From 2fa43ecf0935682418ee72c77136fd3dd87e6dbe Mon Sep 17 00:00:00 2001 From: Martin Wind Date: Tue, 26 Mar 2019 08:10:20 +0100 Subject: [PATCH 001/880] refactor: split argparse and run_pipline --- setup.py | 2 +- src/ocrmypdf/__init__.py | 1 + src/ocrmypdf/__main__.py | 686 +------------------------------------- src/ocrmypdf/run.py | 702 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 708 insertions(+), 683 deletions(-) create mode 100644 src/ocrmypdf/run.py diff --git a/setup.py b/setup.py index 560434f8..74aa1039 100644 --- a/setup.py +++ b/setup.py @@ -108,7 +108,7 @@ setup( ], extras_require={'pdfminer': ['pdfminer.six == 20181108']}, tests_require=tests_require, - entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run_pipeline']}, + entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']}, package_data={'ocrmypdf': ['data/sRGB.icc']}, include_package_data=True, zip_safe=False, diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index a37d2658..bc2e6791 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,3 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo +from .run import run_pipeline as run diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 3f2042df..9a7b1b66 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -17,68 +17,19 @@ # along with OCRmyPDF. If not, see . import argparse -import atexit -import logging import os -import re import sys -import textwrap -from pathlib import Path -from tempfile import mkdtemp - -import PIL -import ruffus.cmdline as cmdline -import ruffus.proxy_logger as proxy_logger -import ruffus.ruffus_exceptions as ruffus_exceptions from . import PROGRAM_NAME, VERSION -from . import exceptions as ocrmypdf_exceptions -from ._jobcontext import JobContext, JobContextManager, cleanup_working_files -from ._pipeline import build_pipeline -from ._unicodefun import verify_python3_env -from .exceptions import ( - BadArgsError, - ExitCode, - ExitCodeException, - InputFileError, - MissingDependencyError, - OutputFileAccessError, -) -from .exec import ( - ghostscript, - jbig2enc, - qpdf, - tesseract, - check_external_program, - unpaper, - pngquant, -) -from .helpers import available_cpu_count, is_file_writable, re_symlink -from .pdfa import file_claims_pdfa - -# ------------- -# External dependencies - -HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) - - -def complain(message): - print(*textwrap.wrap(message), file=sys.stderr) - +from .run import run_pipeline # Hack to help debugger context find /usr/local/bin if 'IDE_PROJECT_ROOTS' in os.environ: os.environ['PATH'] = '/usr/local/bin:' + os.environ['PATH'] -# -------- -# Critical environment tests - -verify_python3_env() - # ------------- # Parser - def numeric(basetype, min_=None, max_=None): """Validator for numeric params""" min_ = basetype(min_) if min_ is not None else None @@ -514,638 +465,9 @@ debugging.add_argument( '--flowchart', type=str, help="Generate the pipeline execution flowchart" ) - -def check_options_languages(options, _log): - if not options.language: - options.language = ['eng'] # Enforce English hegemony - - # Support v2.x "eng+deu" language syntax - if '+' in options.language[0]: - options.language = options.language[0].split('+') - - languages = set(options.language) - if not languages.issubset(tesseract.languages()): - msg = ( - "The installed version of tesseract does not have language " - "data for the following requested languages: \n" - ) - for lang in languages - tesseract.languages(): - msg += lang + '\n' - raise MissingDependencyError(msg) - - -def check_options_output(options, log): - # We have these constraints to check for. - # 1. Ghostscript < 9.20 mangles multibyte Unicode - # 2. hocr doesn't work on non-Latin languages (so don't select it) - - languages = set(options.language) - is_latin = languages.issubset(HOCR_OK_LANGS) - - if options.pdf_renderer == 'hocr' and not is_latin: - msg = ( - "The 'hocr' PDF renderer is known to cause problems with one " - "or more of the languages in your document. Use " - "--pdf-renderer auto (the default) to avoid this issue." - ) - log.warning(msg) - - if ghostscript.version() < '9.20' and options.output_type != 'pdf' and not is_latin: - # https://bugs.ghostscript.com/show_bug.cgi?id=696874 - # Ghostscript < 9.20 fails to encode multibyte characters properly - msg = ( - "The installed version of Ghostscript does not work correctly " - "with the OCR languages you specified. Use --output-type pdf or " - "upgrade to Ghostscript 9.20 or later to avoid this issue." - ) - msg += f"Found Ghostscript {ghostscript.version()}" - log.warning(msg) - - # Decide on what renderer to use - if options.pdf_renderer == 'auto': - options.pdf_renderer = 'sandwich' - - if options.output_type == 'pdfa': - options.output_type = 'pdfa-2' - - if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19': - raise MissingDependencyError( - "--output-type pdfa-3 requires Ghostscript 9.19 or later" - ) - - lossless_reconstruction = False - if not any( - ( - options.deskew, - options.clean_final, - options.force_ocr, - options.remove_background, - ) - ): - lossless_reconstruction = True - options.lossless_reconstruction = lossless_reconstruction - - if not options.lossless_reconstruction and options.redo_ocr: - raise argparse.ArgumentError( - None, - "--redo-ocr is not currently compatible with --deskew, " - "--clean-final, and --remove-background", - ) - - -def check_options_sidecar(options, log): - if options.sidecar == '\0': - if options.output_file == '-': - raise argparse.ArgumentError( - None, - "--sidecar filename must be specified when output file is " "stdout.", - ) - options.sidecar = options.output_file + '.txt' - - -def check_options_preprocessing(options, log): - if options.clean_final: - options.clean = True - if options.unpaper_args and not options.clean: - raise argparse.ArgumentError(None, "--clean is required for --unpaper-args") - if options.clean: - check_external_program( - log=log, - program='unpaper', - package='unpaper', - version_checker=unpaper.version, - need_version='6.1', - required_for=['--clean, --clean-final'], - ) - try: - if options.unpaper_args: - options.unpaper_args = unpaper.validate_custom_args( - options.unpaper_args - ) - except Exception as e: - raise argparse.ArgumentError(None, str(e)) - - -def check_options_ocr_behavior(options, log): - exclusive_options = sum( - [ - (1 if opt else 0) - for opt in (options.force_ocr, options.skip_text, options.redo_ocr) - ] - ) - if exclusive_options >= 2: - raise argparse.ArgumentError( - None, "Error: choose only one of --force-ocr, --skip-text, --redo-ocr." - ) - - -def check_options_optimizing(options, log): - if options.optimize >= 2: - check_external_program( - log=log, - program='pngquant', - package='pngquant', - version_checker=pngquant.version, - need_version='2.0.1', - required_for='--optimize {2,3}', - ) - - if options.optimize >= 2: - # Although we use JBIG2 for optimize=1, don't nag about it unless the - # user is asking for more optimization - check_external_program( - log=log, - program='jbig2', - package='jbig2enc', - version_checker=jbig2enc.version, - need_version='0.28', - required_for='--optimize {2,3} | --jbig2-lossy', - recommended=True if not options.jbig2_lossy else False, - ) - - if options.optimize == 0 and any( - [options.jbig2_lossy, options.png_quality, options.jpeg_quality] - ): - log.warning( - "The arguments --jbig2-lossy, --png-quality, and --jpeg-quality " - "will be ignored because --optimize=0." - ) - - -def check_options_advanced(options, log): - if options.pdfa_image_compression != 'auto' and options.output_type.startswith( - 'pdfa' - ): - log.warning( - "--pdfa-image-compression argument has no effect when " - "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" - ) - if tesseract.v4() and (options.user_words or options.user_patterns): - log.warning('Tesseract 4.x ignores --user-words, so this has no effect') - - -def check_options_metadata(options, log): - import unicodedata - - docinfo = [options.title, options.author, options.keywords, options.subject] - for s in (m for m in docinfo if m): - for c in s: - if unicodedata.category(c) == 'Co' or ord(c) >= 0x10000: - raise ValueError( - "One of the metadata strings contains " - "an unsupported Unicode character: '{}' (U+{})".format( - c, hex(ord(c))[2:].upper() - ) - ) - - -def check_options_pillow(options, log): - PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000) - if PIL.Image.MAX_IMAGE_PIXELS == 0: - PIL.Image.MAX_IMAGE_PIXELS = None - - -def check_options(options, log): - try: - check_options_languages(options, log) - check_options_metadata(options, log) - check_options_output(options, log) - check_options_sidecar(options, log) - check_options_preprocessing(options, log) - check_options_ocr_behavior(options, log) - check_options_optimizing(options, log) - check_options_advanced(options, log) - check_options_pillow(options, log) - except ValueError as e: - log.error(e) - sys.exit(ExitCode.bad_args) - except argparse.ArgumentError as e: - log.error(e) - sys.exit(ExitCode.bad_args) - except MissingDependencyError as e: - log.error(e) - sys.exit(ExitCode.missing_dependency) - - -# ---------- -# Logging - - -def logging_factory(logger_name, logger_args): - verbose = logger_args['verbose'] - quiet = logger_args['quiet'] - - root_logger = logging.getLogger(logger_name) - root_logger.setLevel(logging.DEBUG) - - handler = logging.StreamHandler(sys.stderr) - formatter_ = logging.Formatter("%(levelname)7s - %(message)s") - handler.setFormatter(formatter_) - if verbose: - handler.setLevel(logging.DEBUG) - elif quiet: - handler.setLevel(logging.WARNING) - else: - handler.setLevel(logging.INFO) - root_logger.addHandler(handler) - return root_logger - - -def cleanup_ruffus_error_message(msg): - msg = re.sub(r'\s+', r' ', msg) - msg = re.sub(r"\((.+?)\)", r'\1', msg) - msg = msg.strip() - return msg - - -def do_ruffus_exception(ruffus_five_tuple, options, log): - """Replace the elaborate ruffus stack trace with a user friendly - description of the error message that occurred.""" - exit_code = None - - _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple - - if isinstance(exc_name, type): - # ruffus is full of mystery... sometimes (probably when the process - # group leader is killed) exc_name is the class object of the exception, - # rather than a str. So reach into the object and get its name. - exc_name = exc_name.__name__ - - if exc_name.startswith('ocrmypdf.exceptions.'): - base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') - exc_class = getattr(ocrmypdf_exceptions, base_exc_name) - exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) - try: - if isinstance(exc_value, exc_class): - exc_msg = str(exc_value) - elif isinstance(exc_value, str): - exc_msg = exc_value - else: - exc_msg = str(exc_class()) - except Exception: - exc_msg = "Unknown" - - if exc_name in ('builtins.SystemExit', 'SystemExit'): - match = re.search(r"\.(.+?)\)", exc_value) - exit_code_name = match.groups()[0] - exit_code = getattr(ExitCode, exit_code_name, 'other_error') - elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError': - log.error(cleanup_ruffus_error_message(exc_value)) - exit_code = ExitCode.input_file - elif exc_name in ('builtins.KeyboardInterrupt', 'KeyboardInterrupt'): - # We have to print in this case because the log daemon might be toast - print("Interrupted by user", file=sys.stderr) - exit_code = ExitCode.ctrl_c - elif exc_name == 'subprocess.CalledProcessError': - # It's up to the subprocess handler to report something useful - msg = "Error occurred while running this command:" - log.error(msg + '\n' + exc_value) - exit_code = ExitCode.child_process_error - elif exc_name.startswith('ocrmypdf.exceptions.'): - if exc_msg: - log.error(exc_msg) - elif exc_name == 'PIL.Image.DecompressionBombError': - msg = cleanup_ruffus_error_message(exc_value) - msg += ( - "\nUse the --max-image-mpixels argument to set increase the " - "maximum number of megapixels to accept." - ) - log.error(msg) - exit_code = ExitCode.input_file - - if exit_code is not None: - return exit_code - - if not options.verbose: - log.error(exc_stack) - return ExitCode.other_error - - -def traverse_ruffus_exception(exceptions, options, log): - """Traverse a RethrownJobError and output the exceptions - - Ruffus presents exceptions as 5 element tuples. The RethrownJobException - has a list of exceptions like - e.job_exceptions = [(5-tuple), (5-tuple), ...] - - ruffus < 2.7.0 had a bug with exception marshalling that would give - different output whether the main or child process raised the exception. - We no longer support this. - - Attempting to log the exception itself will re-marshall it to the logger - which is normally running in another process. It's better to avoid re- - marshalling. - - The exit code will be based on this, even if multiple exceptions occurred - at the same time.""" - - exit_codes = [] - for exc in exceptions: - exit_code = do_ruffus_exception(exc, options, log) - exit_codes.append(exit_code) - - return exit_codes[0] # Multiple codes are rare so take the first one - - -def check_closed_streams(options): - """Work around Python issue with multiprocessing forking on closed streams - - https://bugs.python.org/issue28326 - - Attempting to a fork/exec a new Python process when any of std{in,out,err} - are closed or not flushable for some reason may raise an exception. - Fix this by opening devnull if the handle seems to be closed. Do this - globally to avoid tracking places all places that fork. - - Seems to be specific to multiprocessing.Process not all Python process - forkers. - - The error actually occurs when the stream object is not flushable, - but replacing an open stream object that is not flushable with - /dev/null is a bad idea since it will create a silent failure. Replacing - a closed handle with /dev/null seems safe. - - """ - - if sys.version_info[0:3] >= (3, 6, 4): - return True # Issued fixed in Python 3.6.4+ - - if sys.stderr is None: - sys.stderr = open(os.devnull, 'w') - - if sys.stdin is None: - if options.input_file == '-': - print("Trying to read from stdin but stdin seems closed", file=sys.stderr) - return False - sys.stdin = open(os.devnull, 'r') - - if sys.stdout is None: - if options.output_file == '-': - # Can't replace stdout if the user is piping - # If this case can even happen, it must be some kind of weird - # stream. - print( - textwrap.dedent( - """\ - Output was set to stdout '-' but the stream attached to - stdout does not support the flush() system call. This - will fail.""" - ), - file=sys.stderr, - ) - return False - sys.stdout = open(os.devnull, 'w') - - return True - - -def log_page_orientations(pdfinfo, _log): - direction = {0: 'n', 90: 'e', 180: 's', 270: 'w'} - orientations = [] - for n, page in enumerate(pdfinfo): - angle = page.rotation or 0 - if angle != 0: - orientations.append('{0}{1}'.format(n + 1, direction.get(angle, ''))) - if orientations: - _log.info('Page orientations detected: ' + ' '.join(orientations)) - - -def preamble(_log): - _log.debug('ocrmypdf ' + VERSION) - - -def check_environ(options, _log): - old_envvars = ( - 'OCRMYPDF_TESSERACT', - 'OCRMYPDF_QPDF', - 'OCRMYPDF_GS', - 'OCRMYPDF_UNPAPER', - ) - for k in old_envvars: - if k in os.environ: - _log.warning( - textwrap.dedent( - f"""\ - OCRmyPDF no longer uses the environment variable {k}. - Change PATH to select alternate programs.""" - ) - ) - - -def check_input_file(options, _log, start_input_file): - if options.input_file == '-': - # stdin - _log.info('reading file from standard input') - with open(start_input_file, 'wb') as stream_buffer: - from shutil import copyfileobj - - copyfileobj(sys.stdin.buffer, stream_buffer) - else: - try: - re_symlink(options.input_file, start_input_file, _log) - except FileNotFoundError: - _log.error("File not found - " + options.input_file) - raise InputFileError() - - -def check_requested_output_file(options, _log): - if options.output_file == '-': - if sys.stdout.isatty(): - _log.error( - textwrap.dedent( - """\ - Output was set to stdout '-' but it looks like stdout - is connected to a terminal. Please redirect stdout to a - file.""" - ) - ) - raise BadArgsError() - elif not is_file_writable(options.output_file): - _log.error( - "Output file location (" - + options.output_file - + ") " - + "is not a writable file." - ) - raise OutputFileAccessError() - - -def report_output_file_size(options, _log, input_file, output_file): - try: - output_size = Path(output_file).stat().st_size - input_size = Path(input_file).stat().st_size - except FileNotFoundError: - return # Outputting to stream or something - ratio = output_size / input_size - if ratio < 1.35 or input_size < 25000: - return # Seems fine - - reasons = [] - image_preproc = { - 'deskew', - 'clean_final', - 'remove_background', - 'oversample', - 'force_ocr', - } - for arg in image_preproc: - attr = getattr(options, arg, None) - if not attr: - continue - reasons.append( - f"The argument --{arg.replace('_', '-')} was issued, causing transcoding." - ) - - if reasons: - explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n" - else: - explanation = "No reason for this increase is known. Please report this issue." - - _log.warning( - textwrap.dedent( - f"""\ - The output file size is {ratio:.2f}× larger than the input file. - {explanation} - """ - ) - ) - - -def check_dependency_versions(options, log): - check_external_program( - log=log, - program='tesseract', - package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'}, - version_checker=tesseract.version, - need_version='4.0.0', # using backport for Travis CI - ) - check_external_program( - log=log, - program='gs', - package='ghostscript', - version_checker=ghostscript.version, - need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports - ) - if ghostscript.version() == '9.24': - complain( - "Ghostscript 9.24 contains serious regressions and is not " - "supported. Please upgrade to Ghostscript 9.25 or use an older " - "version." - ) - return ExitCode.missing_dependency - check_external_program( - log=log, - program='qpdf', - package='qpdf', - version_checker=qpdf.version, - need_version='8.0.2', - ) - - -def run_pipeline(args=None): +def run(args=None): options = parser.parse_args(args=args) - options.verbose_abbreviated_path = 1 - if os.environ.get('_OCRMYPDF_THREADS'): - options.use_threads = True - - if not check_closed_streams(options): - return ExitCode.bad_args - - logger_args = {'verbose': options.verbose, 'quiet': options.quiet} - - _log, _log_mutex = proxy_logger.make_shared_logger_and_proxy( - logging_factory, __name__, logger_args - ) - preamble(_log) - check_options(options, _log) - check_dependency_versions(options, _log) - - # Any changes to options will not take effect for options that are already - # bound to function parameters in the pipeline. (For example - # options.input_file, options.pdf_renderer are already bound.) - if not options.jobs: - options.jobs = available_cpu_count() - - # Performance is improved by setting Tesseract to single threaded. In tests - # this gives better throughput than letting a smaller number of Tesseract - # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this - # variable, but harmless to set if ignored. - os.environ.setdefault('OMP_THREAD_LIMIT', '1') - - check_environ(options, _log) - if os.environ.get('PYTEST_CURRENT_TEST'): - os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file - - try: - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite') - start_input_file = os.path.join(work_folder, 'origin') - - check_input_file(options, _log, start_input_file) - check_requested_output_file(options, _log) - - manager = JobContextManager() - manager.register('JobContext', JobContext) # pylint: disable=no-member - manager.start() - - context = manager.JobContext() # pylint: disable=no-member - context.set_options(options) - context.set_work_folder(work_folder) - - build_pipeline(options, work_folder, _log, context) - atexit.register(cleanup_working_files, work_folder, options) - if hasattr(os, 'nice'): - os.nice(5) - cmdline.run(options) - except ruffus_exceptions.RethrownJobError as e: - if options.verbose: - _log.debug(str(e)) # stringify exception so logger doesn't have to - exceptions = e.job_exceptions - exitcode = traverse_ruffus_exception(exceptions, options, _log) - if exitcode is None: - _log.error("Unexpected ruffus exception: " + str(e)) - _log.error(repr(e)) - return ExitCode.other_error - return exitcode - except ExitCodeException as e: - return e.exit_code - except Exception as e: - _log.error(str(e)) - return ExitCode.other_error - - if options.flowchart: - _log.info(f"Flowchart saved to {options.flowchart}") - return ExitCode.ok - elif options.output_file == '-': - _log.info("Output sent to stdout") - elif os.path.samefile(options.output_file, os.devnull): - pass # Say nothing when sending to dev null - else: - if options.output_type.startswith('pdfa'): - pdfa_info = file_claims_pdfa(options.output_file) - if pdfa_info['pass']: - msg = f"Output file is a {pdfa_info['conformance']} (as expected)" - _log.info(msg) - else: - msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" - _log.warning(msg) - return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file, _log): - _log.warning('Output file: The generated PDF is INVALID') - return ExitCode.invalid_output_pdf - - report_output_file_size(options, _log, start_input_file, options.output_file) - - pdfinfo = context.get_pdfinfo() - if options.verbose: - from pprint import pformat - - _log.debug(pformat(pdfinfo)) - - log_page_orientations(pdfinfo, _log) - - return ExitCode.ok - + return run_pipeline(options) if __name__ == '__main__': - sys.exit(run_pipeline()) + sys.exit(run()) diff --git a/src/ocrmypdf/run.py b/src/ocrmypdf/run.py new file mode 100644 index 00000000..9d24226f --- /dev/null +++ b/src/ocrmypdf/run.py @@ -0,0 +1,702 @@ +#!/usr/bin/env python3 +# © 2015-17 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import atexit +import logging +import os +import re +import sys +import textwrap +from pathlib import Path +from tempfile import mkdtemp + +import PIL +import ruffus.cmdline as cmdline +import ruffus.proxy_logger as proxy_logger +import ruffus.ruffus_exceptions as ruffus_exceptions + +from . import VERSION +from . import exceptions as ocrmypdf_exceptions +from ._jobcontext import JobContext, JobContextManager, cleanup_working_files +from ._pipeline import build_pipeline +from ._unicodefun import verify_python3_env +from .exceptions import ( + BadArgsError, + ExitCode, + ExitCodeException, + InputFileError, + MissingDependencyError, + OutputFileAccessError, +) +from .exec import ( + ghostscript, + jbig2enc, + qpdf, + tesseract, + check_external_program, + unpaper, + pngquant, +) +from .helpers import available_cpu_count, is_file_writable, re_symlink +from .pdfa import file_claims_pdfa + +# ------------- +# External dependencies + +HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) + + +def complain(message): + print(*textwrap.wrap(message), file=sys.stderr) + +# -------- +# Critical environment tests + +verify_python3_env() + + +def check_options_languages(options, _log): + if not options.language: + options.language = ['eng'] # Enforce English hegemony + + # Support v2.x "eng+deu" language syntax + if '+' in options.language[0]: + options.language = options.language[0].split('+') + + languages = set(options.language) + if not languages.issubset(tesseract.languages()): + msg = ( + "The installed version of tesseract does not have language " + "data for the following requested languages: \n" + ) + for lang in languages - tesseract.languages(): + msg += lang + '\n' + raise MissingDependencyError(msg) + + +def check_options_output(options, log): + # We have these constraints to check for. + # 1. Ghostscript < 9.20 mangles multibyte Unicode + # 2. hocr doesn't work on non-Latin languages (so don't select it) + + languages = set(options.language) + is_latin = languages.issubset(HOCR_OK_LANGS) + + if options.pdf_renderer == 'hocr' and not is_latin: + msg = ( + "The 'hocr' PDF renderer is known to cause problems with one " + "or more of the languages in your document. Use " + "--pdf-renderer auto (the default) to avoid this issue." + ) + log.warning(msg) + + if ghostscript.version() < '9.20' and options.output_type != 'pdf' and not is_latin: + # https://bugs.ghostscript.com/show_bug.cgi?id=696874 + # Ghostscript < 9.20 fails to encode multibyte characters properly + msg = ( + "The installed version of Ghostscript does not work correctly " + "with the OCR languages you specified. Use --output-type pdf or " + "upgrade to Ghostscript 9.20 or later to avoid this issue." + ) + msg += f"Found Ghostscript {ghostscript.version()}" + log.warning(msg) + + # Decide on what renderer to use + if options.pdf_renderer == 'auto': + options.pdf_renderer = 'sandwich' + + if options.output_type == 'pdfa': + options.output_type = 'pdfa-2' + + if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19': + raise MissingDependencyError( + "--output-type pdfa-3 requires Ghostscript 9.19 or later" + ) + + lossless_reconstruction = False + if not any( + ( + options.deskew, + options.clean_final, + options.force_ocr, + options.remove_background, + ) + ): + lossless_reconstruction = True + options.lossless_reconstruction = lossless_reconstruction + + if not options.lossless_reconstruction and options.redo_ocr: + raise BadArgsError( + "--redo-ocr is not currently compatible with --deskew, " + "--clean-final, and --remove-background", + ) + + +def check_options_sidecar(options, log): + if options.sidecar == '\0': + if options.output_file == '-': + raise BadArgsError( + "--sidecar filename must be specified when output file is " "stdout.", + ) + options.sidecar = options.output_file + '.txt' + + +def check_options_preprocessing(options, log): + if options.clean_final: + options.clean = True + if options.unpaper_args and not options.clean: + raise BadArgsError("--clean is required for --unpaper-args") + if options.clean: + check_external_program( + log=log, + program='unpaper', + package='unpaper', + version_checker=unpaper.version, + need_version='6.1', + required_for=['--clean, --clean-final'], + ) + try: + if options.unpaper_args: + options.unpaper_args = unpaper.validate_custom_args( + options.unpaper_args + ) + except Exception as e: + raise BadArgsError(str(e)) + + +def check_options_ocr_behavior(options, log): + exclusive_options = sum( + [ + (1 if opt else 0) + for opt in (options.force_ocr, options.skip_text, options.redo_ocr) + ] + ) + if exclusive_options >= 2: + raise BadArgsError( + "Error: choose only one of --force-ocr, --skip-text, --redo-ocr." + ) + + +def check_options_optimizing(options, log): + if options.optimize >= 2: + check_external_program( + log=log, + program='pngquant', + package='pngquant', + version_checker=pngquant.version, + need_version='2.0.1', + required_for='--optimize {2,3}', + ) + + if options.optimize >= 2: + # Although we use JBIG2 for optimize=1, don't nag about it unless the + # user is asking for more optimization + check_external_program( + log=log, + program='jbig2', + package='jbig2enc', + version_checker=jbig2enc.version, + need_version='0.28', + required_for='--optimize {2,3} | --jbig2-lossy', + recommended=True if not options.jbig2_lossy else False, + ) + + if options.optimize == 0 and any( + [options.jbig2_lossy, options.png_quality, options.jpeg_quality] + ): + log.warning( + "The arguments --jbig2-lossy, --png-quality, and --jpeg-quality " + "will be ignored because --optimize=0." + ) + + +def check_options_advanced(options, log): + if options.pdfa_image_compression != 'auto' and options.output_type.startswith( + 'pdfa' + ): + log.warning( + "--pdfa-image-compression argument has no effect when " + "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" + ) + if tesseract.v4() and (options.user_words or options.user_patterns): + log.warning('Tesseract 4.x ignores --user-words, so this has no effect') + + +def check_options_metadata(options, log): + import unicodedata + + docinfo = [options.title, options.author, options.keywords, options.subject] + for s in (m for m in docinfo if m): + for c in s: + if unicodedata.category(c) == 'Co' or ord(c) >= 0x10000: + raise ValueError( + "One of the metadata strings contains " + "an unsupported Unicode character: '{}' (U+{})".format( + c, hex(ord(c))[2:].upper() + ) + ) + + +def check_options_pillow(options, log): + PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000) + if PIL.Image.MAX_IMAGE_PIXELS == 0: + PIL.Image.MAX_IMAGE_PIXELS = None + + +def check_options(options, log): + try: + check_options_languages(options, log) + check_options_metadata(options, log) + check_options_output(options, log) + check_options_sidecar(options, log) + check_options_preprocessing(options, log) + check_options_ocr_behavior(options, log) + check_options_optimizing(options, log) + check_options_advanced(options, log) + check_options_pillow(options, log) + return ExitCode.ok + except ValueError as e: + log.error(e) + return ExitCode.bad_args + except BadArgsError as e: + log.error(e) + return e.exit_code + except MissingDependencyError as e: + log.error(e) + return ExitCode.missing_dependency + + +# ---------- +# Logging + + +def logging_factory(logger_name, logger_args): + verbose = logger_args['verbose'] + quiet = logger_args['quiet'] + + root_logger = logging.getLogger(logger_name) + root_logger.setLevel(logging.DEBUG) + + handler = logging.StreamHandler(sys.stderr) + formatter_ = logging.Formatter("%(levelname)7s - %(message)s") + handler.setFormatter(formatter_) + if verbose: + handler.setLevel(logging.DEBUG) + elif quiet: + handler.setLevel(logging.WARNING) + else: + handler.setLevel(logging.INFO) + root_logger.addHandler(handler) + return root_logger + + +def cleanup_ruffus_error_message(msg): + msg = re.sub(r'\s+', r' ', msg) + msg = re.sub(r"\((.+?)\)", r'\1', msg) + msg = msg.strip() + return msg + + +def do_ruffus_exception(ruffus_five_tuple, options, log): + """Replace the elaborate ruffus stack trace with a user friendly + description of the error message that occurred.""" + exit_code = None + + _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple + + if isinstance(exc_name, type): + # ruffus is full of mystery... sometimes (probably when the process + # group leader is killed) exc_name is the class object of the exception, + # rather than a str. So reach into the object and get its name. + exc_name = exc_name.__name__ + + if exc_name.startswith('ocrmypdf.exceptions.'): + base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') + exc_class = getattr(ocrmypdf_exceptions, base_exc_name) + exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) + try: + if isinstance(exc_value, exc_class): + exc_msg = str(exc_value) + elif isinstance(exc_value, str): + exc_msg = exc_value + else: + exc_msg = str(exc_class()) + except Exception: + exc_msg = "Unknown" + + if exc_name in ('builtins.SystemExit', 'SystemExit'): + match = re.search(r"\.(.+?)\)", exc_value) + exit_code_name = match.groups()[0] + exit_code = getattr(ExitCode, exit_code_name, 'other_error') + elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError': + log.error(cleanup_ruffus_error_message(exc_value)) + exit_code = ExitCode.input_file + elif exc_name in ('builtins.KeyboardInterrupt', 'KeyboardInterrupt'): + # We have to print in this case because the log daemon might be toast + print("Interrupted by user", file=sys.stderr) + exit_code = ExitCode.ctrl_c + elif exc_name == 'subprocess.CalledProcessError': + # It's up to the subprocess handler to report something useful + msg = "Error occurred while running this command:" + log.error(msg + '\n' + exc_value) + exit_code = ExitCode.child_process_error + elif exc_name.startswith('ocrmypdf.exceptions.'): + if exc_msg: + log.error(exc_msg) + elif exc_name == 'PIL.Image.DecompressionBombError': + msg = cleanup_ruffus_error_message(exc_value) + msg += ( + "\nUse the --max-image-mpixels argument to set increase the " + "maximum number of megapixels to accept." + ) + log.error(msg) + exit_code = ExitCode.input_file + + if exit_code is not None: + return exit_code + + if not options.verbose: + log.error(exc_stack) + return ExitCode.other_error + + +def traverse_ruffus_exception(exceptions, options, log): + """Traverse a RethrownJobError and output the exceptions + + Ruffus presents exceptions as 5 element tuples. The RethrownJobException + has a list of exceptions like + e.job_exceptions = [(5-tuple), (5-tuple), ...] + + ruffus < 2.7.0 had a bug with exception marshalling that would give + different output whether the main or child process raised the exception. + We no longer support this. + + Attempting to log the exception itself will re-marshall it to the logger + which is normally running in another process. It's better to avoid re- + marshalling. + + The exit code will be based on this, even if multiple exceptions occurred + at the same time.""" + + exit_codes = [] + for exc in exceptions: + exit_code = do_ruffus_exception(exc, options, log) + exit_codes.append(exit_code) + + return exit_codes[0] # Multiple codes are rare so take the first one + + +def check_closed_streams(options): + """Work around Python issue with multiprocessing forking on closed streams + + https://bugs.python.org/issue28326 + + Attempting to a fork/exec a new Python process when any of std{in,out,err} + are closed or not flushable for some reason may raise an exception. + Fix this by opening devnull if the handle seems to be closed. Do this + globally to avoid tracking places all places that fork. + + Seems to be specific to multiprocessing.Process not all Python process + forkers. + + The error actually occurs when the stream object is not flushable, + but replacing an open stream object that is not flushable with + /dev/null is a bad idea since it will create a silent failure. Replacing + a closed handle with /dev/null seems safe. + + """ + + if sys.version_info[0:3] >= (3, 6, 4): + return True # Issued fixed in Python 3.6.4+ + + if sys.stderr is None: + sys.stderr = open(os.devnull, 'w') + + if sys.stdin is None: + if options.input_file == '-': + print("Trying to read from stdin but stdin seems closed", file=sys.stderr) + return False + sys.stdin = open(os.devnull, 'r') + + if sys.stdout is None: + if options.output_file == '-': + # Can't replace stdout if the user is piping + # If this case can even happen, it must be some kind of weird + # stream. + print( + textwrap.dedent( + """\ + Output was set to stdout '-' but the stream attached to + stdout does not support the flush() system call. This + will fail.""" + ), + file=sys.stderr, + ) + return False + sys.stdout = open(os.devnull, 'w') + + return True + + +def log_page_orientations(pdfinfo, _log): + direction = {0: 'n', 90: 'e', 180: 's', 270: 'w'} + orientations = [] + for n, page in enumerate(pdfinfo): + angle = page.rotation or 0 + if angle != 0: + orientations.append('{0}{1}'.format(n + 1, direction.get(angle, ''))) + if orientations: + _log.info('Page orientations detected: ' + ' '.join(orientations)) + + +def preamble(_log): + _log.debug('ocrmypdf ' + VERSION) + + +def check_environ(options, _log): + old_envvars = ( + 'OCRMYPDF_TESSERACT', + 'OCRMYPDF_QPDF', + 'OCRMYPDF_GS', + 'OCRMYPDF_UNPAPER', + ) + for k in old_envvars: + if k in os.environ: + _log.warning( + textwrap.dedent( + f"""\ + OCRmyPDF no longer uses the environment variable {k}. + Change PATH to select alternate programs.""" + ) + ) + + +def check_input_file(options, _log, start_input_file): + if options.input_file == '-': + # stdin + _log.info('reading file from standard input') + with open(start_input_file, 'wb') as stream_buffer: + from shutil import copyfileobj + + copyfileobj(sys.stdin.buffer, stream_buffer) + else: + try: + re_symlink(options.input_file, start_input_file, _log) + except FileNotFoundError: + _log.error("File not found - " + options.input_file) + raise InputFileError() + + +def check_requested_output_file(options, _log): + if options.output_file == '-': + if sys.stdout.isatty(): + _log.error( + textwrap.dedent( + """\ + Output was set to stdout '-' but it looks like stdout + is connected to a terminal. Please redirect stdout to a + file.""" + ) + ) + raise BadArgsError() + elif not is_file_writable(options.output_file): + _log.error( + "Output file location (" + + options.output_file + + ") " + + "is not a writable file." + ) + raise OutputFileAccessError() + + +def report_output_file_size(options, _log, input_file, output_file): + try: + output_size = Path(output_file).stat().st_size + input_size = Path(input_file).stat().st_size + except FileNotFoundError: + return # Outputting to stream or something + ratio = output_size / input_size + if ratio < 1.35 or input_size < 25000: + return # Seems fine + + reasons = [] + image_preproc = { + 'deskew', + 'clean_final', + 'remove_background', + 'oversample', + 'force_ocr', + } + for arg in image_preproc: + attr = getattr(options, arg, None) + if not attr: + continue + reasons.append( + f"The argument --{arg.replace('_', '-')} was issued, causing transcoding." + ) + + if reasons: + explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n" + else: + explanation = "No reason for this increase is known. Please report this issue." + + _log.warning( + textwrap.dedent( + f"""\ + The output file size is {ratio:.2f}× larger than the input file. + {explanation} + """ + ) + ) + + +def check_dependency_versions(options, log): + check_external_program( + log=log, + program='tesseract', + package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'}, + version_checker=tesseract.version, + need_version='4.0.0', # using backport for Travis CI + ) + check_external_program( + log=log, + program='gs', + package='ghostscript', + version_checker=ghostscript.version, + need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports + ) + if ghostscript.version() == '9.24': + complain( + "Ghostscript 9.24 contains serious regressions and is not " + "supported. Please upgrade to Ghostscript 9.25 or use an older " + "version." + ) + return ExitCode.missing_dependency + check_external_program( + log=log, + program='qpdf', + package='qpdf', + version_checker=qpdf.version, + need_version='8.0.2', + ) + + +def run_pipeline(options): + options.verbose_abbreviated_path = 1 + if os.environ.get('_OCRMYPDF_THREADS'): + options.use_threads = True + + if not check_closed_streams(options): + return ExitCode.bad_args + + logger_args = {'verbose': options.verbose, 'quiet': options.quiet} + + _log, _log_mutex = proxy_logger.make_shared_logger_and_proxy( + logging_factory, __name__, logger_args + ) + preamble(_log) + check_code = check_options(options, _log) + if check_code != ExitCode.ok: + return check_code + check_dependency_versions(options, _log) + + # Any changes to options will not take effect for options that are already + # bound to function parameters in the pipeline. (For example + # options.input_file, options.pdf_renderer are already bound.) + if not options.jobs: + options.jobs = available_cpu_count() + + # Performance is improved by setting Tesseract to single threaded. In tests + # this gives better throughput than letting a smaller number of Tesseract + # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this + # variable, but harmless to set if ignored. + os.environ.setdefault('OMP_THREAD_LIMIT', '1') + + check_environ(options, _log) + if os.environ.get('PYTEST_CURRENT_TEST'): + os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file + + try: + work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite') + start_input_file = os.path.join(work_folder, 'origin') + + check_input_file(options, _log, start_input_file) + check_requested_output_file(options, _log) + + manager = JobContextManager() + manager.register('JobContext', JobContext) # pylint: disable=no-member + manager.start() + + context = manager.JobContext() # pylint: disable=no-member + context.set_options(options) + context.set_work_folder(work_folder) + + build_pipeline(options, work_folder, _log, context) + atexit.register(cleanup_working_files, work_folder, options) + if hasattr(os, 'nice'): + os.nice(5) + cmdline.run(options) + except ruffus_exceptions.RethrownJobError as e: + if options.verbose: + _log.debug(str(e)) # stringify exception so logger doesn't have to + exceptions = e.job_exceptions + exitcode = traverse_ruffus_exception(exceptions, options, _log) + if exitcode is None: + _log.error("Unexpected ruffus exception: " + str(e)) + _log.error(repr(e)) + return ExitCode.other_error + return exitcode + except ExitCodeException as e: + return e.exit_code + except Exception as e: + _log.error(str(e)) + return ExitCode.other_error + + if options.flowchart: + _log.info(f"Flowchart saved to {options.flowchart}") + return ExitCode.ok + elif options.output_file == '-': + _log.info("Output sent to stdout") + elif os.path.samefile(options.output_file, os.devnull): + pass # Say nothing when sending to dev null + else: + if options.output_type.startswith('pdfa'): + pdfa_info = file_claims_pdfa(options.output_file) + if pdfa_info['pass']: + msg = f"Output file is a {pdfa_info['conformance']} (as expected)" + _log.info(msg) + else: + msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" + _log.warning(msg) + return ExitCode.pdfa_conversion_failed + if not qpdf.check(options.output_file, _log): + _log.warning('Output file: The generated PDF is INVALID') + return ExitCode.invalid_output_pdf + + report_output_file_size(options, _log, start_input_file, options.output_file) + + pdfinfo = context.get_pdfinfo() + if options.verbose: + from pprint import pformat + + _log.debug(pformat(pdfinfo)) + + log_page_orientations(pdfinfo, _log) + + return ExitCode.ok From f65a3d37623a3c6dee0f7559d40a581de208dea9 Mon Sep 17 00:00:00 2001 From: Martin Wind Date: Tue, 26 Mar 2019 10:04:26 +0100 Subject: [PATCH 002/880] fix import in unpaper test --- src/ocrmypdf/__init__.py | 2 +- tests/test_unpaper.py | 7 ++++--- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index bc2e6791..fd06f2be 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,4 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo -from .run import run_pipeline as run +from . import run diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 23f6d698..9d5b60df 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -22,7 +22,8 @@ from unittest.mock import MagicMock, patch import pytest -from ocrmypdf import __main__ as main +from ocrmypdf.__main__ import parser +from ocrmypdf.run import check_options from ocrmypdf.exceptions import ExitCode from ocrmypdf.exec import unpaper @@ -52,12 +53,12 @@ def spoof_unpaper_oldversion(tmpdir_factory): def test_no_unpaper(resources, no_outpdf): input_ = fspath(resources / "c02-22.pdf") output = fspath(no_outpdf) - options = main.parser.parse_args(args=["--clean", input_, output]) + options = parser.parse_args(args=["--clean", input_, output]) with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") with pytest.raises(SystemExit): - main.check_options(options, log=MagicMock()) + check_options(options, log=MagicMock()) def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): From a4667b5656ff7a377cd760e7d0245358fbc8d11b Mon Sep 17 00:00:00 2001 From: Martin Wind Date: Thu, 28 Mar 2019 20:16:10 +0100 Subject: [PATCH 003/880] refactor: move ruffus related code to one file --- src/ocrmypdf/__init__.py | 1 - src/ocrmypdf/__main__.py | 5 +- src/ocrmypdf/_pipeline.py | 251 +----------- src/ocrmypdf/_ruffus.py | 508 ++++++++++++++++++++++++ src/ocrmypdf/{run.py => _validation.py} | 240 +---------- tests/test_unpaper.py | 4 +- 6 files changed, 541 insertions(+), 468 deletions(-) create mode 100644 src/ocrmypdf/_ruffus.py rename src/ocrmypdf/{run.py => _validation.py} (64%) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index fd06f2be..a37d2658 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,4 +44,3 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo -from . import run diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 9a7b1b66..759fffe4 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,7 +21,7 @@ import os import sys from . import PROGRAM_NAME, VERSION -from .run import run_pipeline +from ._ruffus import run_pipeline # Hack to help debugger context find /usr/local/bin if 'IDE_PROJECT_ROOTS' in os.environ: @@ -30,6 +30,7 @@ if 'IDE_PROJECT_ROOTS' in os.environ: # ------------- # Parser + def numeric(basetype, min_=None, max_=None): """Validator for numeric params""" min_ = basetype(min_) if min_ is not None else None @@ -465,9 +466,11 @@ debugging.add_argument( '--flowchart', type=str, help="Generate the pipeline execution flowchart" ) + def run(args=None): options = parser.parse_args(args=args) return run_pipeline(options) + if __name__ == '__main__': sys.exit(run()) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index b283a988..d6e0c491 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -25,13 +25,11 @@ from shutil import copyfile, copyfileobj import img2pdf from PIL import Image -from ruffus import Pipeline, formatter, regex, suffix import pikepdf from pikepdf.models.metadata import encode_pdf_date from . import PROGRAM_NAME, VERSION, leptonica -from ._weave import weave_layers from .exceptions import ( DpiError, EncryptedPdfError, @@ -40,7 +38,12 @@ from .exceptions import ( UnsupportedImageFormatError, ) from .exec import ghostscript, tesseract -from .helpers import flatten_groups, is_iterable_notstr, page_number, re_symlink +from .helpers import ( + flatten_groups, + is_iterable_notstr, + page_number, + re_symlink +) from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps @@ -115,7 +118,10 @@ def triage_image_file(input_file, output_file, log, options): ) with open(output_file, 'wb') as outf: img2pdf.convert( - input_file, layout_fun=layout_fun, with_pdfrw=False, outputstream=outf + input_file, + layout_fun=layout_fun, + with_pdfrw=False, + outputstream=outf ) log.info("Successfully converted to PDF, processing...") except img2pdf.ImageOpenError as e: @@ -170,7 +176,7 @@ def repair_and_parse_pdf(input_file, output_file, log, context): pdfinfo = PdfInfo( output_file, detailed_page_analysis=detailed_page_analysis, log=log ) - except pikepdf.PasswordError as e: + except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError as e: log.error(e) @@ -202,7 +208,8 @@ def repair_and_parse_pdf(input_file, output_file, log, context): raise PriorOcrFoundError() else: log.warning( - "This PDF has a fillable form. Chances are it is a pure digital " + "This PDF has a fillable form. " + "Chances are it is a pure digital " "document that does not need OCR." ) if not options.force_ocr: @@ -281,8 +288,7 @@ def is_ocr_required(pageinfo, log, options): elif options.redo_ocr: if pageinfo.has_corrupt_text: log.warning( - prefix - + ( + prefix + ( "some text on this page cannot be mapped to characters: " "consider using --force-ocr instead", ) @@ -936,232 +942,3 @@ def copy_final(input_files, output_file, log, context): # get the appropriate umask, ownership, etc. with open(output_file, 'wb') as output_stream: copyfileobj(input_stream, output_stream) - - -def build_pipeline(options, work_folder, log, context): - main_pipeline = Pipeline.pipelines['main'] - - # Triage - task_triage = main_pipeline.transform( - task_func=triage, - input=os.path.join(work_folder, 'origin'), - filter=formatter('(?i)'), - output=os.path.join(work_folder, 'origin.pdf'), - extras=[log, context], - ) - - task_repair_and_parse_pdf = main_pipeline.transform( - task_func=repair_and_parse_pdf, - input=task_triage, - filter=suffix('.pdf'), - output='.repaired.pdf', - output_dir=work_folder, - extras=[log, context], - ) - - # Split (kwargs for split seems to be broken, so pass plain args) - task_marker_pages = main_pipeline.split( - marker_pages, - task_repair_and_parse_pdf, - os.path.join(work_folder, '*.marker.pdf'), - extras=[log, context], - ) - - task_ocr_or_skip = main_pipeline.split( - ocr_or_skip, - task_marker_pages, - [ - os.path.join(work_folder, '*.ocr.page.pdf'), - os.path.join(work_folder, '*.skip.page.pdf'), - ], - extras=[log, context], - ) - - # Rasterize preview - task_rasterize_preview = main_pipeline.transform( - task_func=rasterize_preview, - input=task_ocr_or_skip, - filter=suffix('.page.pdf'), - output='.preview.jpg', - output_dir=work_folder, - extras=[log, context], - ) - task_rasterize_preview.active_if(options.rotate_pages) - - # Orient - task_orient_page = main_pipeline.collate( - task_func=orient_page, - input=[task_ocr_or_skip, task_rasterize_preview], - filter=regex(r".*/(\d{6})(\.ocr|\.skip)(?:\.page\.pdf|\.preview\.jpg)"), - output=os.path.join(work_folder, r'\1\2.oriented.pdf'), - extras=[log, context], - ) - - # Rasterize actual - task_rasterize_with_ghostscript = main_pipeline.transform( - task_func=rasterize_with_ghostscript, - input=task_orient_page, - filter=suffix('.ocr.oriented.pdf'), - output='.page.png', - output_dir=work_folder, - extras=[log, context], - ) - - # Preprocessing subpipeline - task_preprocess_remove_background = main_pipeline.transform( - task_func=preprocess_remove_background, - input=task_rasterize_with_ghostscript, - filter=suffix(".page.png"), - output=".pp-background.png", - extras=[log, context], - ) - - task_preprocess_deskew = main_pipeline.transform( - task_func=preprocess_deskew, - input=task_preprocess_remove_background, - filter=suffix(".pp-background.png"), - output=".pp-deskew.png", - extras=[log, context], - ) - - task_preprocess_clean = main_pipeline.transform( - task_func=preprocess_clean, - input=task_preprocess_deskew, - filter=suffix(".pp-deskew.png"), - output=".pp-clean.png", - extras=[log, context], - ) - - task_select_ocr_image = main_pipeline.collate( - task_func=select_ocr_image, - input=[task_preprocess_clean], - filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), - output=os.path.join(work_folder, r"\1.ocr.png"), - extras=[log, context], - ) - - # HOCR OCR - task_ocr_tesseract_hocr = main_pipeline.transform( - task_func=ocr_tesseract_hocr, - input=task_select_ocr_image, - filter=suffix(".ocr.png"), - output=[".hocr", ".txt"], - extras=[log, context], - ) - task_ocr_tesseract_hocr.graphviz(fillcolor='"#00cc66"') - task_ocr_tesseract_hocr.active_if(options.pdf_renderer == 'hocr') - - task_select_visible_page_image = main_pipeline.collate( - task_func=select_visible_page_image, - input=[ - task_rasterize_with_ghostscript, - task_preprocess_remove_background, - task_preprocess_deskew, - task_preprocess_clean, - ], - filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), - output=os.path.join(work_folder, r'\1.image'), - extras=[log, context], - ) - task_select_visible_page_image.graphviz(shape='diamond') - - task_select_image_layer = main_pipeline.collate( - task_func=select_image_layer, - input=[task_select_visible_page_image, task_orient_page], - filter=regex(r".*/(\d{6})(?:\.image|\.ocr\.oriented\.pdf)"), - output=os.path.join(work_folder, r'\1.image-layer.pdf'), - extras=[log, context], - ) - task_select_image_layer.graphviz(fillcolor='"#00cc66"', shape='diamond') - - task_render_hocr_page = main_pipeline.transform( - task_func=render_hocr_page, - input=task_ocr_tesseract_hocr, - filter=regex(r".*/(\d{6})(?:\.hocr)"), - output=os.path.join(work_folder, r'\1.text.pdf'), - extras=[log, context], - ) - task_render_hocr_page.graphviz(fillcolor='"#00cc66"') - task_render_hocr_page.active_if(options.pdf_renderer == 'hocr') - - # Tesseract OCR + text only PDF - task_ocr_tesseract_textonly_pdf = main_pipeline.collate( - task_func=ocr_tesseract_textonly_pdf, - input=[task_select_ocr_image], - filter=regex(r".*/(\d{6})(?:\.ocr.png)"), - output=[ - os.path.join(work_folder, r'\1.text.pdf'), - os.path.join(work_folder, r'\1.text.txt'), - ], - extras=[log, context], - ) - task_ocr_tesseract_textonly_pdf.graphviz(fillcolor='"#ff69b4"') - task_ocr_tesseract_textonly_pdf.active_if(options.pdf_renderer == 'sandwich') - - task_weave_layers = main_pipeline.collate( - task_func=weave_layers, - input=[ - task_repair_and_parse_pdf, - task_render_hocr_page, - task_ocr_tesseract_textonly_pdf, - task_select_image_layer, - ], - filter=regex( - r".*/((?:\d{6}(?:\.text\.pdf|\.image-layer\.pdf))|(?:origin\.repaired\.pdf))" - ), - output=os.path.join(work_folder, r'layers.rendered.pdf'), - extras=[log, context], - ) - task_weave_layers.graphviz(fillcolor='"#00cc66"') - - # PDF/A pdfmark - task_generate_postscript_stub = main_pipeline.transform( - task_func=generate_postscript_stub, - input=task_repair_and_parse_pdf, - filter=formatter(r'\.repaired\.pdf'), - output=os.path.join(work_folder, 'pdfa.ps'), - extras=[log, context], - ) - task_generate_postscript_stub.active_if(options.output_type.startswith('pdfa')) - - # PDF/A conversion - task_convert_to_pdfa = main_pipeline.merge( - task_func=convert_to_pdfa, - input=[task_generate_postscript_stub, task_weave_layers], - output=os.path.join(work_folder, 'pdfa.pdf'), - extras=[log, context], - ) - task_convert_to_pdfa.active_if(options.output_type.startswith('pdfa')) - - task_metadata_fixup = main_pipeline.merge( - task_func=metadata_fixup, - input=[task_repair_and_parse_pdf, task_weave_layers, task_convert_to_pdfa], - output=os.path.join(work_folder, 'metafix.pdf'), - extras=[log, context], - ) - - task_merge_sidecars = main_pipeline.merge( - task_func=merge_sidecars, - input=[task_ocr_tesseract_hocr, task_ocr_tesseract_textonly_pdf], - output=options.sidecar, - extras=[log, context], - ) - task_merge_sidecars.active_if(options.sidecar) - - # Optimize - task_optimize_pdf = main_pipeline.transform( - task_func=optimize_pdf, - input=task_metadata_fixup, - filter=suffix('.pdf'), - output='.optimized.pdf', - output_dir=work_folder, - extras=[log, context], - ) - - # Finalize - main_pipeline.merge( - task_func=copy_final, - input=[task_optimize_pdf], - output=options.output_file, - extras=[log, context], - ) diff --git a/src/ocrmypdf/_ruffus.py b/src/ocrmypdf/_ruffus.py new file mode 100644 index 00000000..4cc69043 --- /dev/null +++ b/src/ocrmypdf/_ruffus.py @@ -0,0 +1,508 @@ +# © 2016 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os +import re +import sys +import atexit +from tempfile import mkdtemp +from ruffus import ( + Pipeline, + formatter, + regex, + suffix, + cmdline, + proxy_logger, + ruffus_exceptions +) +from .exec import qpdf +from ._jobcontext import JobContext, JobContextManager, cleanup_working_files +from ._weave import weave_layers +from ._pipeline import ( + triage, + repair_and_parse_pdf, + marker_pages, + ocr_or_skip, + rasterize_preview, + orient_page, + rasterize_with_ghostscript, + preprocess_remove_background, + preprocess_deskew, + preprocess_clean, + select_ocr_image, + ocr_tesseract_hocr, + select_visible_page_image, + select_image_layer, + render_hocr_page, + ocr_tesseract_textonly_pdf, + generate_postscript_stub, + convert_to_pdfa, + metadata_fixup, + merge_sidecars, + optimize_pdf, + copy_final +) +from . import exceptions as ocrmypdf_exceptions +from .exceptions import ( + ExitCode, + ExitCodeException, +) +from .helpers import available_cpu_count +from .pdfa import file_claims_pdfa +from ._validation import ( + check_closed_streams, + preamble, + check_options, + check_dependency_versions, + check_environ, + check_input_file, + check_requested_output_file, + report_output_file_size, + log_page_orientations, + logging_factory, +) + + +def cleanup_ruffus_error_message(msg): + msg = re.sub(r'\s+', r' ', msg) + msg = re.sub(r"\((.+?)\)", r'\1', msg) + msg = msg.strip() + return msg + + +def do_ruffus_exception(ruffus_five_tuple, options, log): + """Replace the elaborate ruffus stack trace with a user friendly + description of the error message that occurred.""" + exit_code = None + + _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple + + if isinstance(exc_name, type): + # ruffus is full of mystery... sometimes (probably when the process + # group leader is killed) exc_name is the class object of the exception, + # rather than a str. So reach into the object and get its name. + exc_name = exc_name.__name__ + + if exc_name.startswith('ocrmypdf.exceptions.'): + base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') + exc_class = getattr(ocrmypdf_exceptions, base_exc_name) + exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) + try: + if isinstance(exc_value, exc_class): + exc_msg = str(exc_value) + elif isinstance(exc_value, str): + exc_msg = exc_value + else: + exc_msg = str(exc_class()) + except Exception: + exc_msg = "Unknown" + + if exc_name in ('builtins.SystemExit', 'SystemExit'): + match = re.search(r"\.(.+?)\)", exc_value) + exit_code_name = match.groups()[0] + exit_code = getattr(ExitCode, exit_code_name, 'other_error') + elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError': + log.error(cleanup_ruffus_error_message(exc_value)) + exit_code = ExitCode.input_file + elif exc_name in ('builtins.KeyboardInterrupt', 'KeyboardInterrupt'): + # We have to print in this case because the log daemon might be toast + print("Interrupted by user", file=sys.stderr) + exit_code = ExitCode.ctrl_c + elif exc_name == 'subprocess.CalledProcessError': + # It's up to the subprocess handler to report something useful + msg = "Error occurred while running this command:" + log.error(msg + '\n' + exc_value) + exit_code = ExitCode.child_process_error + elif exc_name.startswith('ocrmypdf.exceptions.'): + if exc_msg: + log.error(exc_msg) + elif exc_name == 'PIL.Image.DecompressionBombError': + msg = cleanup_ruffus_error_message(exc_value) + msg += ( + "\nUse the --max-image-mpixels argument to set increase the " + "maximum number of megapixels to accept." + ) + log.error(msg) + exit_code = ExitCode.input_file + + if exit_code is not None: + return exit_code + + if not options.verbose: + log.error(exc_stack) + return ExitCode.other_error + + +def traverse_ruffus_exception(exceptions, options, log): + """Traverse a RethrownJobError and output the exceptions + + Ruffus presents exceptions as 5 element tuples. The RethrownJobException + has a list of exceptions like + e.job_exceptions = [(5-tuple), (5-tuple), ...] + + ruffus < 2.7.0 had a bug with exception marshalling that would give + different output whether the main or child process raised the exception. + We no longer support this. + + Attempting to log the exception itself will re-marshall it to the logger + which is normally running in another process. It's better to avoid re- + marshalling. + + The exit code will be based on this, even if multiple exceptions occurred + at the same time.""" + + exit_codes = [] + for exc in exceptions: + exit_code = do_ruffus_exception(exc, options, log) + exit_codes.append(exit_code) + + return exit_codes[0] # Multiple codes are rare so take the first one + + +def build_pipeline(options, work_folder, log, context): + main_pipeline = Pipeline.pipelines['main'] + + # Triage + task_triage = main_pipeline.transform( + task_func=triage, + input=os.path.join(work_folder, 'origin'), + filter=formatter('(?i)'), + output=os.path.join(work_folder, 'origin.pdf'), + extras=[log, context], + ) + + task_repair_and_parse_pdf = main_pipeline.transform( + task_func=repair_and_parse_pdf, + input=task_triage, + filter=suffix('.pdf'), + output='.repaired.pdf', + output_dir=work_folder, + extras=[log, context], + ) + + # Split (kwargs for split seems to be broken, so pass plain args) + task_marker_pages = main_pipeline.split( + marker_pages, + task_repair_and_parse_pdf, + os.path.join(work_folder, '*.marker.pdf'), + extras=[log, context], + ) + + task_ocr_or_skip = main_pipeline.split( + ocr_or_skip, + task_marker_pages, + [ + os.path.join(work_folder, '*.ocr.page.pdf'), + os.path.join(work_folder, '*.skip.page.pdf'), + ], + extras=[log, context], + ) + + # Rasterize preview + task_rasterize_preview = main_pipeline.transform( + task_func=rasterize_preview, + input=task_ocr_or_skip, + filter=suffix('.page.pdf'), + output='.preview.jpg', + output_dir=work_folder, + extras=[log, context], + ) + task_rasterize_preview.active_if(options.rotate_pages) + + # Orient + task_orient_page = main_pipeline.collate( + task_func=orient_page, + input=[task_ocr_or_skip, task_rasterize_preview], + filter=regex(r".*/(\d{6})(\.ocr|\.skip)(?:\.page\.pdf|\.preview\.jpg)"), + output=os.path.join(work_folder, r'\1\2.oriented.pdf'), + extras=[log, context], + ) + + # Rasterize actual + task_rasterize_with_ghostscript = main_pipeline.transform( + task_func=rasterize_with_ghostscript, + input=task_orient_page, + filter=suffix('.ocr.oriented.pdf'), + output='.page.png', + output_dir=work_folder, + extras=[log, context], + ) + + # Preprocessing subpipeline + task_preprocess_remove_background = main_pipeline.transform( + task_func=preprocess_remove_background, + input=task_rasterize_with_ghostscript, + filter=suffix(".page.png"), + output=".pp-background.png", + extras=[log, context], + ) + + task_preprocess_deskew = main_pipeline.transform( + task_func=preprocess_deskew, + input=task_preprocess_remove_background, + filter=suffix(".pp-background.png"), + output=".pp-deskew.png", + extras=[log, context], + ) + + task_preprocess_clean = main_pipeline.transform( + task_func=preprocess_clean, + input=task_preprocess_deskew, + filter=suffix(".pp-deskew.png"), + output=".pp-clean.png", + extras=[log, context], + ) + + task_select_ocr_image = main_pipeline.collate( + task_func=select_ocr_image, + input=[task_preprocess_clean], + filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), + output=os.path.join(work_folder, r"\1.ocr.png"), + extras=[log, context], + ) + + # HOCR OCR + task_ocr_tesseract_hocr = main_pipeline.transform( + task_func=ocr_tesseract_hocr, + input=task_select_ocr_image, + filter=suffix(".ocr.png"), + output=[".hocr", ".txt"], + extras=[log, context], + ) + task_ocr_tesseract_hocr.graphviz(fillcolor='"#00cc66"') + task_ocr_tesseract_hocr.active_if(options.pdf_renderer == 'hocr') + + task_select_visible_page_image = main_pipeline.collate( + task_func=select_visible_page_image, + input=[ + task_rasterize_with_ghostscript, + task_preprocess_remove_background, + task_preprocess_deskew, + task_preprocess_clean, + ], + filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), + output=os.path.join(work_folder, r'\1.image'), + extras=[log, context], + ) + task_select_visible_page_image.graphviz(shape='diamond') + + task_select_image_layer = main_pipeline.collate( + task_func=select_image_layer, + input=[task_select_visible_page_image, task_orient_page], + filter=regex(r".*/(\d{6})(?:\.image|\.ocr\.oriented\.pdf)"), + output=os.path.join(work_folder, r'\1.image-layer.pdf'), + extras=[log, context], + ) + task_select_image_layer.graphviz(fillcolor='"#00cc66"', shape='diamond') + + task_render_hocr_page = main_pipeline.transform( + task_func=render_hocr_page, + input=task_ocr_tesseract_hocr, + filter=regex(r".*/(\d{6})(?:\.hocr)"), + output=os.path.join(work_folder, r'\1.text.pdf'), + extras=[log, context], + ) + task_render_hocr_page.graphviz(fillcolor='"#00cc66"') + task_render_hocr_page.active_if(options.pdf_renderer == 'hocr') + + # Tesseract OCR + text only PDF + task_ocr_tesseract_textonly_pdf = main_pipeline.collate( + task_func=ocr_tesseract_textonly_pdf, + input=[task_select_ocr_image], + filter=regex(r".*/(\d{6})(?:\.ocr.png)"), + output=[ + os.path.join(work_folder, r'\1.text.pdf'), + os.path.join(work_folder, r'\1.text.txt'), + ], + extras=[log, context], + ) + task_ocr_tesseract_textonly_pdf.graphviz(fillcolor='"#ff69b4"') + task_ocr_tesseract_textonly_pdf.active_if(options.pdf_renderer == 'sandwich') + + task_weave_layers = main_pipeline.collate( + task_func=weave_layers, + input=[ + task_repair_and_parse_pdf, + task_render_hocr_page, + task_ocr_tesseract_textonly_pdf, + task_select_image_layer, + ], + filter=regex( + r".*/((?:\d{6}(?:\.text\.pdf|\.image-layer\.pdf))|(?:origin\.repaired\.pdf))" + ), + output=os.path.join(work_folder, r'layers.rendered.pdf'), + extras=[log, context], + ) + task_weave_layers.graphviz(fillcolor='"#00cc66"') + + # PDF/A pdfmark + task_generate_postscript_stub = main_pipeline.transform( + task_func=generate_postscript_stub, + input=task_repair_and_parse_pdf, + filter=formatter(r'\.repaired\.pdf'), + output=os.path.join(work_folder, 'pdfa.ps'), + extras=[log, context], + ) + task_generate_postscript_stub.active_if(options.output_type.startswith('pdfa')) + + # PDF/A conversion + task_convert_to_pdfa = main_pipeline.merge( + task_func=convert_to_pdfa, + input=[task_generate_postscript_stub, task_weave_layers], + output=os.path.join(work_folder, 'pdfa.pdf'), + extras=[log, context], + ) + task_convert_to_pdfa.active_if(options.output_type.startswith('pdfa')) + + task_metadata_fixup = main_pipeline.merge( + task_func=metadata_fixup, + input=[task_repair_and_parse_pdf, task_weave_layers, task_convert_to_pdfa], + output=os.path.join(work_folder, 'metafix.pdf'), + extras=[log, context], + ) + + task_merge_sidecars = main_pipeline.merge( + task_func=merge_sidecars, + input=[task_ocr_tesseract_hocr, task_ocr_tesseract_textonly_pdf], + output=options.sidecar, + extras=[log, context], + ) + task_merge_sidecars.active_if(options.sidecar) + + # Optimize + task_optimize_pdf = main_pipeline.transform( + task_func=optimize_pdf, + input=task_metadata_fixup, + filter=suffix('.pdf'), + output='.optimized.pdf', + output_dir=work_folder, + extras=[log, context], + ) + + # Finalize + main_pipeline.merge( + task_func=copy_final, + input=[task_optimize_pdf], + output=options.output_file, + extras=[log, context], + ) + + +def run_pipeline(options): + options.verbose_abbreviated_path = 1 + if os.environ.get('_OCRMYPDF_THREADS'): + options.use_threads = True + + if not check_closed_streams(options): + return ExitCode.bad_args + + logger_args = {'verbose': options.verbose, 'quiet': options.quiet} + + _log, _log_mutex = proxy_logger.make_shared_logger_and_proxy( + logging_factory, __name__, logger_args + ) + preamble(_log) + check_code = check_options(options, _log) + if check_code != ExitCode.ok: + return check_code + check_dependency_versions(options, _log) + + # Any changes to options will not take effect for options that are already + # bound to function parameters in the pipeline. (For example + # options.input_file, options.pdf_renderer are already bound.) + if not options.jobs: + options.jobs = available_cpu_count() + + # Performance is improved by setting Tesseract to single threaded. In tests + # this gives better throughput than letting a smaller number of Tesseract + # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this + # variable, but harmless to set if ignored. + os.environ.setdefault('OMP_THREAD_LIMIT', '1') + + check_environ(options, _log) + if os.environ.get('PYTEST_CURRENT_TEST'): + os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file + + try: + work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite') + start_input_file = os.path.join(work_folder, 'origin') + + check_input_file(options, _log, start_input_file) + check_requested_output_file(options, _log) + + manager = JobContextManager() + manager.register('JobContext', JobContext) # pylint: disable=no-member + manager.start() + + context = manager.JobContext() # pylint: disable=no-member + context.set_options(options) + context.set_work_folder(work_folder) + + build_pipeline(options, work_folder, _log, context) + atexit.register(cleanup_working_files, work_folder, options) + if hasattr(os, 'nice'): + os.nice(5) + cmdline.run(options) + except ruffus_exceptions.RethrownJobError as e: + if options.verbose: + _log.debug(str(e)) # stringify exception so logger doesn't have to + exceptions = e.job_exceptions + exitcode = traverse_ruffus_exception(exceptions, options, _log) + if exitcode is None: + _log.error("Unexpected ruffus exception: " + str(e)) + _log.error(repr(e)) + return ExitCode.other_error + return exitcode + except ExitCodeException as e: + return e.exit_code + except Exception as e: + _log.error(str(e)) + return ExitCode.other_error + + if options.flowchart: + _log.info(f"Flowchart saved to {options.flowchart}") + return ExitCode.ok + elif options.output_file == '-': + _log.info("Output sent to stdout") + elif os.path.samefile(options.output_file, os.devnull): + pass # Say nothing when sending to dev null + else: + if options.output_type.startswith('pdfa'): + pdfa_info = file_claims_pdfa(options.output_file) + if pdfa_info['pass']: + msg = f"Output file is a {pdfa_info['conformance']} (as expected)" + _log.info(msg) + else: + msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" + _log.warning(msg) + return ExitCode.pdfa_conversion_failed + if not qpdf.check(options.output_file, _log): + _log.warning('Output file: The generated PDF is INVALID') + return ExitCode.invalid_output_pdf + + report_output_file_size(options, _log, start_input_file, options.output_file) + + pdfinfo = context.get_pdfinfo() + if options.verbose: + from pprint import pformat + + _log.debug(pformat(pdfinfo)) + + log_page_orientations(pdfinfo, _log) + + return ExitCode.ok diff --git a/src/ocrmypdf/run.py b/src/ocrmypdf/_validation.py similarity index 64% rename from src/ocrmypdf/run.py rename to src/ocrmypdf/_validation.py index 9d24226f..871c412d 100644 --- a/src/ocrmypdf/run.py +++ b/src/ocrmypdf/_validation.py @@ -16,33 +16,20 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import atexit + import logging import os -import re + import sys import textwrap from pathlib import Path -from tempfile import mkdtemp import PIL -import ruffus.cmdline as cmdline -import ruffus.proxy_logger as proxy_logger -import ruffus.ruffus_exceptions as ruffus_exceptions from . import VERSION -from . import exceptions as ocrmypdf_exceptions -from ._jobcontext import JobContext, JobContextManager, cleanup_working_files -from ._pipeline import build_pipeline + from ._unicodefun import verify_python3_env -from .exceptions import ( - BadArgsError, - ExitCode, - ExitCodeException, - InputFileError, - MissingDependencyError, - OutputFileAccessError, -) + from .exec import ( ghostscript, jbig2enc, @@ -52,8 +39,14 @@ from .exec import ( unpaper, pngquant, ) -from .helpers import available_cpu_count, is_file_writable, re_symlink -from .pdfa import file_claims_pdfa +from .helpers import is_file_writable, re_symlink +from .exceptions import ( + BadArgsError, + ExitCode, + InputFileError, + MissingDependencyError, + OutputFileAccessError, +) # ------------- # External dependencies @@ -64,9 +57,9 @@ HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) def complain(message): print(*textwrap.wrap(message), file=sys.stderr) + # -------- # Critical environment tests - verify_python3_env() @@ -305,102 +298,6 @@ def logging_factory(logger_name, logger_args): return root_logger -def cleanup_ruffus_error_message(msg): - msg = re.sub(r'\s+', r' ', msg) - msg = re.sub(r"\((.+?)\)", r'\1', msg) - msg = msg.strip() - return msg - - -def do_ruffus_exception(ruffus_five_tuple, options, log): - """Replace the elaborate ruffus stack trace with a user friendly - description of the error message that occurred.""" - exit_code = None - - _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple - - if isinstance(exc_name, type): - # ruffus is full of mystery... sometimes (probably when the process - # group leader is killed) exc_name is the class object of the exception, - # rather than a str. So reach into the object and get its name. - exc_name = exc_name.__name__ - - if exc_name.startswith('ocrmypdf.exceptions.'): - base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') - exc_class = getattr(ocrmypdf_exceptions, base_exc_name) - exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) - try: - if isinstance(exc_value, exc_class): - exc_msg = str(exc_value) - elif isinstance(exc_value, str): - exc_msg = exc_value - else: - exc_msg = str(exc_class()) - except Exception: - exc_msg = "Unknown" - - if exc_name in ('builtins.SystemExit', 'SystemExit'): - match = re.search(r"\.(.+?)\)", exc_value) - exit_code_name = match.groups()[0] - exit_code = getattr(ExitCode, exit_code_name, 'other_error') - elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError': - log.error(cleanup_ruffus_error_message(exc_value)) - exit_code = ExitCode.input_file - elif exc_name in ('builtins.KeyboardInterrupt', 'KeyboardInterrupt'): - # We have to print in this case because the log daemon might be toast - print("Interrupted by user", file=sys.stderr) - exit_code = ExitCode.ctrl_c - elif exc_name == 'subprocess.CalledProcessError': - # It's up to the subprocess handler to report something useful - msg = "Error occurred while running this command:" - log.error(msg + '\n' + exc_value) - exit_code = ExitCode.child_process_error - elif exc_name.startswith('ocrmypdf.exceptions.'): - if exc_msg: - log.error(exc_msg) - elif exc_name == 'PIL.Image.DecompressionBombError': - msg = cleanup_ruffus_error_message(exc_value) - msg += ( - "\nUse the --max-image-mpixels argument to set increase the " - "maximum number of megapixels to accept." - ) - log.error(msg) - exit_code = ExitCode.input_file - - if exit_code is not None: - return exit_code - - if not options.verbose: - log.error(exc_stack) - return ExitCode.other_error - - -def traverse_ruffus_exception(exceptions, options, log): - """Traverse a RethrownJobError and output the exceptions - - Ruffus presents exceptions as 5 element tuples. The RethrownJobException - has a list of exceptions like - e.job_exceptions = [(5-tuple), (5-tuple), ...] - - ruffus < 2.7.0 had a bug with exception marshalling that would give - different output whether the main or child process raised the exception. - We no longer support this. - - Attempting to log the exception itself will re-marshall it to the logger - which is normally running in another process. It's better to avoid re- - marshalling. - - The exit code will be based on this, even if multiple exceptions occurred - at the same time.""" - - exit_codes = [] - for exc in exceptions: - exit_code = do_ruffus_exception(exc, options, log) - exit_codes.append(exit_code) - - return exit_codes[0] # Multiple codes are rare so take the first one - - def check_closed_streams(options): """Work around Python issue with multiprocessing forking on closed streams @@ -516,10 +413,7 @@ def check_requested_output_file(options, _log): raise BadArgsError() elif not is_file_writable(options.output_file): _log.error( - "Output file location (" - + options.output_file - + ") " - + "is not a writable file." + "Output file location (" + options.output_file + ") is not a writable file." ) raise OutputFileAccessError() @@ -594,109 +488,3 @@ def check_dependency_versions(options, log): version_checker=qpdf.version, need_version='8.0.2', ) - - -def run_pipeline(options): - options.verbose_abbreviated_path = 1 - if os.environ.get('_OCRMYPDF_THREADS'): - options.use_threads = True - - if not check_closed_streams(options): - return ExitCode.bad_args - - logger_args = {'verbose': options.verbose, 'quiet': options.quiet} - - _log, _log_mutex = proxy_logger.make_shared_logger_and_proxy( - logging_factory, __name__, logger_args - ) - preamble(_log) - check_code = check_options(options, _log) - if check_code != ExitCode.ok: - return check_code - check_dependency_versions(options, _log) - - # Any changes to options will not take effect for options that are already - # bound to function parameters in the pipeline. (For example - # options.input_file, options.pdf_renderer are already bound.) - if not options.jobs: - options.jobs = available_cpu_count() - - # Performance is improved by setting Tesseract to single threaded. In tests - # this gives better throughput than letting a smaller number of Tesseract - # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this - # variable, but harmless to set if ignored. - os.environ.setdefault('OMP_THREAD_LIMIT', '1') - - check_environ(options, _log) - if os.environ.get('PYTEST_CURRENT_TEST'): - os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file - - try: - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite') - start_input_file = os.path.join(work_folder, 'origin') - - check_input_file(options, _log, start_input_file) - check_requested_output_file(options, _log) - - manager = JobContextManager() - manager.register('JobContext', JobContext) # pylint: disable=no-member - manager.start() - - context = manager.JobContext() # pylint: disable=no-member - context.set_options(options) - context.set_work_folder(work_folder) - - build_pipeline(options, work_folder, _log, context) - atexit.register(cleanup_working_files, work_folder, options) - if hasattr(os, 'nice'): - os.nice(5) - cmdline.run(options) - except ruffus_exceptions.RethrownJobError as e: - if options.verbose: - _log.debug(str(e)) # stringify exception so logger doesn't have to - exceptions = e.job_exceptions - exitcode = traverse_ruffus_exception(exceptions, options, _log) - if exitcode is None: - _log.error("Unexpected ruffus exception: " + str(e)) - _log.error(repr(e)) - return ExitCode.other_error - return exitcode - except ExitCodeException as e: - return e.exit_code - except Exception as e: - _log.error(str(e)) - return ExitCode.other_error - - if options.flowchart: - _log.info(f"Flowchart saved to {options.flowchart}") - return ExitCode.ok - elif options.output_file == '-': - _log.info("Output sent to stdout") - elif os.path.samefile(options.output_file, os.devnull): - pass # Say nothing when sending to dev null - else: - if options.output_type.startswith('pdfa'): - pdfa_info = file_claims_pdfa(options.output_file) - if pdfa_info['pass']: - msg = f"Output file is a {pdfa_info['conformance']} (as expected)" - _log.info(msg) - else: - msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" - _log.warning(msg) - return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file, _log): - _log.warning('Output file: The generated PDF is INVALID') - return ExitCode.invalid_output_pdf - - report_output_file_size(options, _log, start_input_file, options.output_file) - - pdfinfo = context.get_pdfinfo() - if options.verbose: - from pprint import pformat - - _log.debug(pformat(pdfinfo)) - - log_page_orientations(pdfinfo, _log) - - return ExitCode.ok diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 9d5b60df..213ef7d0 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -15,15 +15,13 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import argparse from os import fspath -from pathlib import Path from unittest.mock import MagicMock, patch import pytest from ocrmypdf.__main__ import parser -from ocrmypdf.run import check_options +from ocrmypdf._validation import check_options from ocrmypdf.exceptions import ExitCode from ocrmypdf.exec import unpaper From aa512b61811cb49a42677f2525d8c9e396733e61 Mon Sep 17 00:00:00 2001 From: Martin Wind Date: Tue, 2 Apr 2019 20:03:09 +0200 Subject: [PATCH 004/880] feat: move to sync (none ETL) implementation (WIP) --- src/ocrmypdf/__main__.py | 2 +- src/ocrmypdf/_pipeline_simple.py | 973 +++++++++++++++++++++++++++++++ src/ocrmypdf/_sync.py | 242 ++++++++ src/ocrmypdf/_validation.py | 19 + src/ocrmypdf/exec/tesseract.py | 2 +- 5 files changed, 1236 insertions(+), 2 deletions(-) create mode 100644 src/ocrmypdf/_pipeline_simple.py create mode 100644 src/ocrmypdf/_sync.py diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 759fffe4..f9c6dced 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,7 +21,7 @@ import os import sys from . import PROGRAM_NAME, VERSION -from ._ruffus import run_pipeline +from ._sync import run_pipeline # Hack to help debugger context find /usr/local/bin if 'IDE_PROJECT_ROOTS' in os.environ: diff --git a/src/ocrmypdf/_pipeline_simple.py b/src/ocrmypdf/_pipeline_simple.py new file mode 100644 index 00000000..272805b4 --- /dev/null +++ b/src/ocrmypdf/_pipeline_simple.py @@ -0,0 +1,973 @@ +# © 2016 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os +import re +import sys +from datetime import datetime, timezone +from shutil import copyfileobj + +import img2pdf +from PIL import Image + +import pikepdf +from pikepdf.models.metadata import encode_pdf_date + +from . import PROGRAM_NAME, VERSION, leptonica +from .exceptions import ( + DpiError, + EncryptedPdfError, + InputFileError, + UnsupportedImageFormatError, +) +from .exec import ghostscript, tesseract +from .helpers import ( + flatten_groups, + page_number, + re_symlink +) +from .hocrtransform import HocrTransform +from .optimize import optimize +from .pdfa import generate_pdfa_ps +from .pdfinfo import Colorspace, PdfInfo + +VECTOR_PAGE_DPI = 400 + +# +# The Pipeline +# + + +def triage_image_file(input_file, output_file, log, options): + try: + log.info("Input file is not a PDF, checking if it is an image...") + im = Image.open(input_file) + except EnvironmentError as e: + msg = str(e) + + # Recover the original filename + realpath = '' + if os.path.islink(input_file): + realpath = os.path.realpath(input_file) + elif os.path.isfile(input_file): + realpath = '' + msg = msg.replace(input_file, realpath) + log.error(msg) + raise UnsupportedImageFormatError() from e + else: + log.info("Input file is an image") + + if 'dpi' in im.info: + if im.info['dpi'] <= (96, 96) and not options.image_dpi: + log.info("Image size: (%d, %d)" % im.size) + log.info("Image resolution: (%d, %d)" % im.info['dpi']) + log.error( + "Input file is an image, but the resolution (DPI) is " + "not credible. Estimate the resolution at which the " + "image was scanned and specify it using --image-dpi." + ) + raise DpiError() + elif not options.image_dpi: + log.info("Image size: (%d, %d)" % im.size) + log.error( + "Input file is an image, but has no resolution (DPI) " + "in its metadata. Estimate the resolution at which " + "image was scanned and specify it using --image-dpi." + ) + raise DpiError() + + if im.mode in ('RGBA', 'LA'): + log.error( + "The input image has an alpha channel. Remove the alpha " + "channel first." + ) + raise UnsupportedImageFormatError() + + if 'iccprofile' not in im.info: + if im.mode == 'RGB': + log.info('Input image has no ICC profile, assuming sRGB') + elif im.mode == 'CMYK': + log.info('Input CMYK image has no ICC profile, not usable') + raise UnsupportedImageFormatError() + im.close() + + try: + log.info("Image seems valid. Try converting to PDF...") + layout_fun = img2pdf.default_layout_fun + if options.image_dpi: + layout_fun = img2pdf.get_fixed_dpi_layout_fun( + (options.image_dpi, options.image_dpi) + ) + with open(output_file, 'wb') as outf: + img2pdf.convert( + input_file, + layout_fun=layout_fun, + with_pdfrw=False, + outputstream=outf + ) + log.info("Successfully converted to PDF, processing...") + except img2pdf.ImageOpenError as e: + log.error(e) + raise UnsupportedImageFormatError() from e + + +def _pdf_guess_version(input_file, search_window=1024): + """Try to find version signature at start of file. + + Not robust enough to deal with appended files. + + Returns empty string if not found, indicating file is probably not PDF. + """ + + with open(input_file, 'rb') as f: + signature = f.read(search_window) + m = re.search(br'%PDF-(\d\.\d)', signature) + if m: + return m.group(1) + return '' + + +def triage(input_file, output_file, log, context): + + options = context.get_options() + try: + if _pdf_guess_version(input_file): + if options.image_dpi: + log.warning( + "Argument --image-dpi ignored because the " + "input file is a PDF, not an image." + ) + re_symlink(input_file, output_file, log) + return + except EnvironmentError as e: + log.error(e) + raise InputFileError() from e + + triage_image_file(input_file, output_file, log, options) + + +def get_pdfinfo(input_file, detailed_page_analysis=False): + try: + return PdfInfo( + input_file, detailed_page_analysis=detailed_page_analysis + ) + except pikepdf.PasswordError: + raise EncryptedPdfError() + except pikepdf.PdfError: + raise InputFileError() + + +def validate_pdfinfo_options(context): + log = context.log + pdfinfo = context.pdfinfo + options = context.options + + if pdfinfo.needs_rendering: + log.error( + "This PDF contains dynamic XFA forms created by Adobe LiveCycle " + "Designer and can only be read by Adobe Acrobat or Adobe Reader." + ) + raise InputFileError() + if pdfinfo.has_userunit and options.output_type.startswith('pdfa'): + log.error( + "This input file uses a PDF feature that is not supported " + "by Ghostscript, so you cannot use --output-type=pdfa for this " + "file. (Specifically, it uses the PDF-1.6 /UserUnit feature to " + "support very large or small page sizes, and Ghostscript cannot " + "output these files.) Use --output-type=pdf instead." + ) + raise InputFileError() + if pdfinfo.has_acroform: + if options.redo_ocr: + log.error( + "This PDF has a user fillable form. --redo-ocr is not " + "currently possible on such files." + ) + raise InputFileError() + else: + log.warn( + "This PDF has a fillable form. " + "Chances are it is a pure digital " + "document that does not need OCR." + ) + if not options.force_ocr: + log.info( + "Use the option --force-ocr to produce an image of the " + "form and all filled form fields. The output PDF will be " + "'flattened' and will no longer be fillable." + ) + + +""" +def repair_and_parse_pdf(input_file, output_file, log, context): + options = context.get_options() + copyfile(input_file, output_file) + + detailed_page_analysis = False + if options.redo_ocr: + detailed_page_analysis = True + + try: + pdfinfo = PdfInfo( + output_file, detailed_page_analysis=detailed_page_analysis, log=log + ) + except pikepdf.PasswordError: + raise EncryptedPdfError() + except pikepdf.PdfError as e: + log.error(e) + raise InputFileError() + + if pdfinfo.needs_rendering: + log.error( + "This PDF contains dynamic XFA forms created by Adobe LiveCycle " + "Designer and can only be read by Adobe Acrobat or Adobe Reader." + ) + raise InputFileError() + + if pdfinfo.has_userunit and options.output_type.startswith('pdfa'): + log.error( + "This input file uses a PDF feature that is not supported " + "by Ghostscript, so you cannot use --output-type=pdfa for this " + "file. (Specifically, it uses the PDF-1.6 /UserUnit feature to " + "support very large or small page sizes, and Ghostscript cannot " + "output these files.) Use --output-type=pdf instead." + ) + raise InputFileError() + + if pdfinfo.has_acroform: + if options.redo_ocr: + log.error( + "This PDF has a user fillable form. --redo-ocr is not " + "currently possible on such files." + ) + raise PriorOcrFoundError() + else: + log.warning( + "This PDF has a fillable form. " + "Chances are it is a pure digital " + "document that does not need OCR." + ) + if not options.force_ocr: + log.info( + "Use the option --force-ocr to produce an image of the " + "form and all filled form fields. The output PDF will be " + "'flattened' and will no longer be fillable." + ) + + context.set_pdfinfo(pdfinfo) + log.debug(pdfinfo) +""" + + +def get_pageinfo(input_file, context): + "Get zero-based page info implied by filename, e.g. 000002.pdf -> 1" + pageno = page_number(input_file) - 1 + pageinfo = context.get_pdfinfo()[pageno] + return pageinfo + + +def get_page_dpi(pageinfo, options): + "Get the DPI when nonsquare DPI is tolerable" + xres = max( + pageinfo.xres or VECTOR_PAGE_DPI, + options.oversample or 0, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + ) + yres = max( + pageinfo.yres or VECTOR_PAGE_DPI, + options.oversample or 0, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + ) + return (float(xres), float(yres)) + + +def get_page_square_dpi(pageinfo, options): + "Get the DPI when we require xres == yres, scaled to physical units" + xres = pageinfo.xres or 0 + yres = pageinfo.yres or 0 + userunit = pageinfo.userunit or 1 + return float( + max( + (xres * userunit) or VECTOR_PAGE_DPI, + (yres * userunit) or VECTOR_PAGE_DPI, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + options.oversample or 0, + ) + ) + + +def get_canvas_square_dpi(pageinfo, options): + """Get the DPI when we require xres == yres, in Postscript units""" + return float( + max( + (pageinfo.xres) or VECTOR_PAGE_DPI, + (pageinfo.yres) or VECTOR_PAGE_DPI, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + options.oversample or 0, + ) + ) + + +def is_ocr_required(page_context): + pageinfo = page_context.pageinfo + options = page_context.options + log = page_context.log + + ocr_required = True + + if pageinfo.has_text: + if not options.force_ocr and not (options.skip_text or options.redo_ocr): + log.error("page already has text! - aborting (use --force-ocr to force OCR)") + ocr_required = False + elif options.force_ocr: + log.info("page already has text! - rasterizing text and running OCR anyway") + ocr_required = True + elif options.redo_ocr: + if pageinfo.has_corrupt_text: + log.warn( + "some text on this page cannot be mapped to characters: " + "consider using --force-ocr instead", + ) + else: + log.info("redoing OCR") + ocr_required = True + elif options.skip_text: + log.info("skipping all processing on this page") + ocr_required = False + elif not pageinfo.images and not options.lossless_reconstruction: + # We found a page with no images and no text. That means it may + # have vector art that the user wants to OCR. If we determined + # lossless reconstruction is not possible then we have to rasterize + # the image. So if OCR is being forced, take that to mean YES, go + # ahead and rasterize. If not forced, then pretend there's no text + # on the page at all so we don't lose anything. + # This could be made smarter by explicitly searching for vector art. + if options.force_ocr and options.oversample: + # The user really wants to reprocess this file + log.info( + "page has no images - " + f"rasterizing at {options.oversample} DPI because " + "--force-ocr --oversample was specified" + ) + elif options.force_ocr: + # Warn the user they might not want to do this + log.warn( + "page has no images - " + "all vector content will be " + f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely " + "increasing file size. Use --oversample to adjust the " + "DPI." + ) + else: + log.info( + "page has no images - " + "skipping all processing on this page to avoid losing detail. " + "Use --force-ocr if you wish to perform OCR on pages that " + "have vector content." + ) + ocr_required = False + + if ocr_required and options.skip_big and pageinfo.images: + pixel_count = pageinfo.width_pixels * pageinfo.height_pixels + if pixel_count > (options.skip_big * 1_000_000): + ocr_required = False + log.warn( + "page too big, skipping OCR " + f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)" + ) + return ocr_required + + +""" +def marker_pages(input_files, output_files, log, context): + + options = context.get_options() + work_folder = context.get_work_folder() + + if is_iterable_notstr(input_files): + input_file = input_files[0] + else: + input_file = input_files + + for oo in output_files: + with suppress(FileNotFoundError): + os.unlink(oo) + + # If no files were repaired the input will be empty + if not input_file: + log.error(f"{options.input_file}: file not found or invalid argument") + raise InputFileError() + + pdfinfo = context.get_pdfinfo() + npages = len(pdfinfo) + + # Ruffus needs to see a file for any task it generates, so make very + # file a symlink back to the source. + for n in range(npages): + page = Path(work_folder) / f'{(n + 1):06d}.marker.pdf' + page.symlink_to(input_file) # pylint: disable=E1101 +""" + +""" +def ocr_or_skip(input_files, output_files, log, context): + options = context.get_options() + work_folder = context.get_work_folder() + pdfinfo = context.get_pdfinfo() + + for input_file in input_files: + pageno = page_number(input_file) - 1 + pageinfo = pdfinfo[pageno] + alt_suffix = ( + '.ocr.page.pdf' + if is_ocr_required(pageinfo, log, options) + else '.skip.page.pdf' + ) + + re_symlink( + input_file, + os.path.join(work_folder, os.path.basename(input_file)[0:6] + alt_suffix), + log, + ) +""" + + +def rasterize_preview(input_file, page_context): + output_file = page_context.get_path('rasterize_preview.jpg') + canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) + page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + ghostscript.rasterize_pdf( + input_file, + output_file, + xres=canvas_dpi, + yres=canvas_dpi, + raster_device='jpeggray', + log=page_context.log, + page_dpi=(page_dpi, page_dpi), + pageno=page_context.pageinfo.pageno + 1, + ) + return output_file + + +def get_orientation_correction(preview, page_context): + """ + Work out orientation correct for each page. + + We ask Ghostscript to draw a preview page, which will rasterize with the + current /Rotate applied, and then ask Tesseract which way the page is + oriented. If the value of /Rotate is correct (e.g., a user already + manually fixed rotation), then Tesseract will say the page is pointing + up and the correction is zero. Otherwise, the orientation found by + Tesseract represents the clockwise rotation, or the counterclockwise + correction to rotation. + + When we draw the real page for OCR, we rotate it by the CCW correction, + which points it (hopefully) upright. _weave.py takes care of the orienting + the image and text layers. + + """ + + orient_conf = tesseract.get_orientation( + preview, + engine_mode=page_context.options.tesseract_oem, + timeout=page_context.options.tesseract_timeout, + log=page_context.log, + ) + + direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} + + existing_rotation = page_context.pageinfo.rotation + + correction = orient_conf.angle % 360 + + apply_correction = False + action = '' + if orient_conf.confidence >= page_context.options.rotate_pages_threshold: + if correction != 0: + apply_correction = True + action = ' - will rotate' + else: + action = ' - rotation appears correct' + else: + if correction != 0: + action = ' - confidence too low to rotate' + else: + action = ' - no change' + + facing = '' + if existing_rotation != 0: + facing = 'with existing rotation {}, '.format( + direction.get(existing_rotation, '?') + ) + facing += 'page is facing {}'.format(direction.get(orient_conf.angle, '?')) + + page_context.log.debug( + '{pagenum:4d}: {facing}, confidence {conf:.2f}{action}'.format( + pagenum=page_context.pageinfo.pageno, + facing=facing, + conf=orient_conf.confidence, + action=action, + ) + ) + + if apply_correction: + return correction + return 0 + + +def rasterize(input_file, page_context, correction=0): + colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m'] + device_idx = 0 + output_file = page_context.get_path('rasterize.png') + pageinfo = page_context.pageinfo + + def at_least(cs): + return max(device_idx, colorspaces.index(cs)) + + for image in pageinfo.images: + if image.type_ != 'image': + continue # ignore masks + if image.bpc > 1: + if image.color == Colorspace.index: + device_idx = at_least('png256') + elif image.color == Colorspace.gray: + device_idx = at_least('pnggray') + else: + device_idx = at_least('png16m') + + device = colorspaces[device_idx] + + page_context.log.debug(f"Rasterize with {device}") + + # Produce the page image with square resolution or else deskew and OCR + # will not work properly. + canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options) + page_dpi = get_page_square_dpi(pageinfo, page_context.options) + + ghostscript.rasterize_pdf( + input_file, + output_file, + xres=canvas_dpi, + yres=canvas_dpi, + raster_device=device, + log=page_context.log, + page_dpi=(page_dpi, page_dpi), + pageno=pageinfo.pageno + 1, + rotation=correction, + filter_vector=page_context.options.remove_vectors, + ) + return output_file + + +def preprocess_remove_background(input_file, page_context): + if any(image.bpc > 1 for image in page_context.pageinfo.images): + output_file = page_context.get_path('pp_rm_bg.png') + leptonica.remove_background(input_file, output_file) + return output_file + else: + page_context.log.info("background removal skipped on mono page") + return input_file + + +def preprocess_deskew(input_file, page_context): + output_file = page_context.get_path('pp_deskew.png') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + leptonica.deskew(input_file, output_file, dpi) + return output_file + + +def preprocess_clean(input_file, page_context): + from .exec import unpaper + output_file = page_context.get_path('pp_clean.png') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + unpaper.clean(input_file, output_file, dpi, page_context.log, page_context.options.unpaper_args) + return output_file + + +def create_ocr_image(image, page_context): + """Create the image we send for OCR. May not be the same as the display + image depending on preprocessing. This image will never be shown to the + user.""" + + output_file = page_context.get_path('ocr.png') + options = page_context.options + with Image.open(image) as im: + from PIL import ImageColor + from PIL import ImageDraw + + white = ImageColor.getcolor('#ffffff', im.mode) + # pink = ImageColor.getcolor('#ff0080', im.mode) + draw = ImageDraw.ImageDraw(im) + + xres, yres = im.info['dpi'] + print('resolution %r %r', xres, yres) + + if not options.force_ocr: + # Do not mask text areas when forcing OCR, because we need to OCR + # all text areas + mask = None # Exclude both visible and invisible text from OCR + if options.redo_ocr: + mask = True # Mask visible text, but not invisible text + + for textarea in page_context.pageinfo.get_textareas(visible=mask, corrupt=None): + # Calculate resolution based on the image size and page dimensions + # without regard whatever resolution is in pageinfo (may differ or + # be None) + bbox = [float(v) for v in textarea] + xscale, yscale = float(xres) / 72.0, float(yres) / 72.0 + pixcoords = [ + bbox[0] * xscale, + im.height - bbox[3] * yscale, + bbox[2] * xscale, + im.height - bbox[1] * yscale, + ] + pixcoords = [int(round(c)) for c in pixcoords] + print('blanking %r', pixcoords) + draw.rectangle(pixcoords, fill=white) + # draw.rectangle(pixcoords, outline=pink) + + if options.mask_barcodes or options.threshold: + pix = leptonica.Pix.frompil(im) + if options.threshold: + pix = pix.masked_threshold_on_background_norm() + if options.mask_barcodes: + barcodes = pix.locate_barcodes() + for barcode in barcodes: + decoded, rect = barcode + print('masking barcode %s %r', decoded, rect) + draw.rectangle(rect, fill=white) + im = pix.topil() + + del draw + # Pillow requires integer DPI + dpi = round(xres), round(yres) + im.save(output_file, dpi=dpi) + return output_file + + +def ocr_tesseract_hocr(input_file, page_context): + hocr_out = page_context.get_path('ocr_hocr.hocr') + hocr_text_out = page_context.get_path('ocr_hocr.txt') + options = page_context.options + tesseract.generate_hocr( + input_file=input_file, + output_files=[hocr_out, hocr_text_out], + language=options.language, + engine_mode=options.tesseract_oem, + tessconfig=options.tesseract_config, + timeout=options.tesseract_timeout, + pagesegmode=options.tesseract_pagesegmode, + user_words=options.user_words, + user_patterns=options.user_patterns, + log=page_context.log, + ) + return (hocr_out, hocr_text_out) + + +def should_visible_page_image_use_jpg(pageinfo): + # If all images were JPEGs originally, produce a JPEG as output + return pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images) + + +def create_visible_page_jpg(image, page_context): + output_file = page_context.get_path('visible.jpg') + with Image.open(image) as im: + # At this point the image should be a .png, but deskew, unpaper + # might have removed the DPI information. In this case, fall back to + # square DPI used to rasterize. When the preview image was + # rasterized, it was also converted to square resolution, which is + # what we want to give tesseract, so keep it square. + fallback_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi)) + + # Pillow requires integer DPI + dpi = round(dpi[0]), round(dpi[1]) + im.save(output_file, format='JPEG', dpi=dpi) + return output_file + + +def create_pdf_page_from_image(image, page_context): + # We rasterize a square DPI version of each page because most image + # processing tools don't support rectangular DPI. Use the square DPI as it + # accurately describes the image. It would be possible to resample the image + # at this stage back to non-square DPI to more closely resemble the input, + # except that the hocr renderer does not understand non-square DPI. The + # sandwich renderer would be fine. + output_file = page_context.get_path('visible.pdf') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) + + # This create a single page PDF + with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: + page_context.log.debug('convert') + img2pdf.convert( + imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf + ) + page_context.log.debug('convert done') + return output_file + + +""" +def select_image_layer(infiles, output_file, log, context): + # Selects the image layer for the output page. If possible this is the + # orientation-corrected input page, or an image of the whole page converted + # to PDF. + + options = context.get_options() + page_pdf = next(ii for ii in infiles if ii.endswith('.ocr.oriented.pdf')) + image = next(ii for ii in infiles if ii.endswith('.image')) + + if options.lossless_reconstruction: + log.debug( + f"{page_number(page_pdf):4d}: page eligible for lossless reconstruction" + ) + re_symlink(page_pdf, output_file, log) # Still points to multipage + return + + pageinfo = get_pageinfo(image, context) + + # We rasterize a square DPI version of each page because most image + # processing tools don't support rectangular DPI. Use the square DPI as it + # accurately describes the image. It would be possible to resample the image + # at this stage back to non-square DPI to more closely resemble the input, + # except that the hocr renderer does not understand non-square DPI. The + # sandwich renderer would be fine. + dpi = get_page_square_dpi(pageinfo, options) + layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) + + # This create a single page PDF + with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: + log.debug(f'{page_number(page_pdf):4d}: convert') + img2pdf.convert( + imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf + ) + log.debug(f'{page_number(page_pdf):4d}: convert done') +""" + + +def render_hocr_page(hocr, page_context): + output_file = page_context.get_path('ocr_hocr.pdf') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + hocrtransform = HocrTransform(hocr, dpi) + hocrtransform.to_pdf( + output_file, + imageFileName=None, + showBoundingboxes=False, + invisibleText=True, + interwordSpaces=True, + ) + return output_file + + +def ocr_tesseract_textonly_pdf(input_image, page_context): + output_pdf = page_context.get_path('ocr_tess.pdf') + output_text = page_context.get_path('ocr_tess.txt') + options = page_context.options + tesseract.generate_pdf( + input_image=input_image, + skip_pdf=None, + output_pdf=output_pdf, + output_text=output_text, + language=options.language, + engine_mode=options.tesseract_oem, + text_only=True, + tessconfig=options.tesseract_config, + timeout=options.tesseract_timeout, + pagesegmode=options.tesseract_pagesegmode, + user_words=options.user_words, + user_patterns=options.user_patterns, + log=page_context.log, + ) + return (output_pdf, output_text) + + +def get_docinfo(base_pdf, options): + def from_document_info(key): + try: + s = base_pdf.docinfo[key] + return str(s) + except (KeyError, TypeError): + return '' + + pdfmark = { + k: from_document_info(k) + for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate') + } + if options.title: + pdfmark['/Title'] = options.title + if options.author: + pdfmark['/Author'] = options.author + if options.keywords: + pdfmark['/Keywords'] = options.keywords + if options.subject: + pdfmark['/Subject'] = options.subject + + if options.pdf_renderer == 'sandwich': + renderer_tag = 'OCR-PDF' + else: + renderer_tag = 'OCR' + + pdfmark['/Creator'] = ( + f'{PROGRAM_NAME} {VERSION} / ' f'Tesseract {renderer_tag} {tesseract.version()}' + ) + pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}' + if 'OCRMYPDF_CREATOR' in os.environ: + pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR'] + if 'OCRMYPDF_PRODUCER' in os.environ: + pdfmark['/Producer'] = os.environ['OCRMYPDF_PRODUCER'] + + pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc)) + return pdfmark + + +def generate_postscript_stub(input_file, output_file, log, context): + generate_pdfa_ps(output_file) + + +def convert_to_pdfa(input_files_groups, output_file, log, context): + options = context.get_options() + input_pdfinfo = context.get_pdfinfo() + + input_files = list(f for f in flatten_groups(input_files_groups)) + layers_file = next( + (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None + ) + + # If the DocumentInfo record contains NUL characters, Ghostscript will + # produce XMP metadata which contains invalid XML entities (�). + # NULs in DocumentInfo seem to be common since older Acrobats included them. + # pikepdf can deal with this, but we make the world a better place by + # stamping them out as soon as possible. + pdf_layers_file = pikepdf.open(layers_file) + if pdf_layers_file.docinfo: + modified = False + for k, v in pdf_layers_file.docinfo.items(): + if b'\x00' in bytes(v): + pdf_layers_file.docinfo[k] = bytes(v).replace(b'\x00', b'') + modified = True + if modified: + pdf_layers_file.save(layers_file) + del pdf_layers_file + + ps = next((ii for ii in input_files if ii.endswith('.ps')), None) + ghostscript.generate_pdfa( + pdf_version=input_pdfinfo.min_version, + pdf_pages=[layers_file, ps], + output_file=output_file, + compression=options.pdfa_image_compression, + log=log, + threads=options.jobs or 1, + pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 + ) + + +def metadata_fixup(input_files_groups, output_file, log, context): + options = context.get_options() + + input_files = list(f for f in flatten_groups(input_files_groups)) + original_file = next( + (ii for ii in input_files if ii.endswith('.repaired.pdf')), None + ) + layers_file = next( + (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None + ) + pdfa_file = next((ii for ii in input_files if ii.endswith('pdfa.pdf')), None) + original = pikepdf.open(original_file) + docinfo = get_docinfo(original, options) + + working_file = pdfa_file if pdfa_file else layers_file + + pdf = pikepdf.open(working_file) + with pdf.open_metadata() as meta: + meta.load_from_docinfo(docinfo, delete_missing=False) + # If xmp:CreateDate is missing, set it to the modify date to + # match Ghostscript, for consistency + if 'xmp:CreateDate' not in meta: + meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') + if pdfa_file: + meta_original = original.open_metadata() + not_copied = set(meta_original.keys()) - set(meta.keys()) + if not_copied: + log.warning( + "Some input metadata could not be copied because it is not " + "permitted in PDF/A. You may wish to examine the output " + "PDF's XMP metadata." + ) + log.debug( + "The following metadata fields were not copied: %r", not_copied + ) + + pdf.save( + output_file, + compress_streams=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + ) + + +def optimize_pdf(input_file, output_file, log, context): + optimize(input_file, output_file, log, context) + + +def merge_sidecars(input_files_groups, output_file, log, context): + pdfinfo = context.get_pdfinfo() + + txt_files = [None] * len(pdfinfo) + + for infile in flatten_groups(input_files_groups): + if infile.endswith('.txt'): + idx = page_number(infile) - 1 + txt_files[idx] = infile + + def write_pages(stream): + for page_num, txt_file in enumerate(txt_files): + if page_num != 0: + stream.write('\f') # Form feed between pages + if txt_file: + with open(txt_file, 'r', encoding="utf-8") as in_: + txt = in_.read() + # Tesseract v4 alpha started adding form feeds in + # commit aa6eb6b + # No obvious way to detect what binaries will do this, so + # for consistency just ignore its form feeds and insert our + # own + if txt.endswith('\f'): + stream.write(txt[:-1]) + else: + stream.write(txt) + else: + stream.write(f'[OCR skipped on page {(page_num + 1)}]') + + if output_file == '-': + write_pages(sys.stdout) + sys.stdout.flush() + else: + with open(output_file, 'w', encoding="utf-8") as out: + write_pages(out) + + +def copy_final(input_files, output_file, log, context): + input_file = next((ii for ii in input_files if ii.endswith('.pdf'))) + log.debug('%s -> %s', input_file, output_file) + with open(input_file, 'rb') as input_stream: + if output_file == '-': + copyfileobj(input_stream, sys.stdout.buffer) + sys.stdout.flush() + else: + # At this point we overwrite the output_file specified by the user + # use copyfileobj because then we use open() to create the file and + # get the appropriate umask, ownership, etc. + with open(output_file, 'wb') as output_stream: + copyfileobj(input_stream, output_stream) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py new file mode 100644 index 00000000..32b435aa --- /dev/null +++ b/src/ocrmypdf/_sync.py @@ -0,0 +1,242 @@ +# © 2016 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os +# import re +# import sys +import atexit +from tempfile import mkdtemp +from .helpers import re_symlink +from ._jobcontext import cleanup_working_files +from .exec import qpdf +# from ._weave import weave_layers +from ._pipeline_simple import ( + get_pdfinfo, + validate_pdfinfo_options, + is_ocr_required, + rasterize_preview, + get_orientation_correction, + rasterize, + preprocess_remove_background, + preprocess_deskew, + preprocess_clean, + create_ocr_image, + ocr_tesseract_hocr, + should_visible_page_image_use_jpg, + create_visible_page_jpg, + create_pdf_page_from_image, + render_hocr_page, + ocr_tesseract_textonly_pdf, +) +from .exceptions import ( + ExitCode, +) +from .helpers import available_cpu_count +from .pdfa import file_claims_pdfa +from ._validation import ( + check_closed_streams, + preamble, + check_options, + check_dependency_versions, + check_environ, + check_input_file, + check_requested_output_file, + report_output_file_size, + create_input_file, +) + + +class Logger: + def __init__(self, prefix): + self.prefix = prefix + + def debug(self, *argv): + print(self.prefix, *argv) + + def info(self, *argv): + print(self.prefix, *argv) + + def warn(self, *argv): + print(self.prefix, *argv) + + def error(self, *argv): + print(self.prefix, *argv) + + +class PageContext: + def __init__(self, pdf_context, pageno): + self.pdf_context = pdf_context + self.options = pdf_context.options + self.pageno = pageno + self.pageinfo = pdf_context.pdfinfo[pageno] + self.log = Logger('%s Page %d: ' % (os.path.basename(pdf_context.origin), pageno + 1)) + + def get_path(self, name): + return os.path.join(self.pdf_context.work_folder, "page_%d_%s" % (self.pageno, name)) + + +class PDFContext: + def __init__(self, options, work_folder, origin, pdfinfo): + self.options = options + self.work_folder = work_folder + self.origin = origin + self.pdfinfo = pdfinfo + self.log = Logger('%s: ' % os.path.basename(origin)) + + def get_path(self, name): + return os.path.join(self.work_folder, name) + + def get_page_contexts(self): + npages = len(self.pdfinfo) + for n in range(npages): + yield PageContext(self, n) + + +def build_pipeline(options, work_folder, origin): + # Gather info of pdf + pdfinfo = get_pdfinfo(origin) + context = PDFContext(options, work_folder, origin, pdfinfo) + + # Validate options are okey for this pdf + validate_pdfinfo_options(context) + + # For every page in the pdf + page_res = [] + for page_context in context.get_page_contexts(): + # Check if OCR is required + ocr_required = is_ocr_required(page_context) + if not ocr_required: + continue + + orientation_correction = 0 + if options.rotate_pages: + # Rasterize + rasterize_preview_out = rasterize_preview(origin, page_context) + orientation_correction = get_orientation_correction(rasterize_preview_out, page_context) + + rasterize_out = rasterize(origin, page_context, correction=orientation_correction) + + preprocess_out = rasterize_out + if options.remove_background: + preprocess_out = preprocess_remove_background(preprocess_out, page_context) + + if options.deskew: + preprocess_out = preprocess_deskew(preprocess_out, page_context) + + if options.clean: + preprocess_out = preprocess_clean(preprocess_out, page_context) + + ocr_image_out = create_ocr_image(preprocess_out, page_context) + + pdf_page_from_image_out = None + if not options.lossless_reconstruction: + visible_image_out = preprocess_out + if should_visible_page_image_use_jpg(page_context.pageinfo): + visible_image_out = create_visible_page_jpg(visible_image_out, page_context) + pdf_page_from_image_out = create_pdf_page_from_image(visible_image_out, page_context) + + if options.pdf_renderer == 'hocr': + (hocr_out, text_out) = ocr_tesseract_hocr(ocr_image_out, page_context) + ocr_out = render_hocr_page(hocr_out, page_context) + + if options.pdf_renderer == 'sandwich': + (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) + + page_res.append((pdf_page_from_image_out, ocr_out, orientation_correction)) + + print(page_res) + + +def run_pipeline(options): + if not check_closed_streams(options): + return ExitCode.bad_args + + log = Logger('Pipeline') + preamble(log) + check_code = check_options(options, log) + if check_code != ExitCode.ok: + return check_code + check_dependency_versions(options, log) + + # Any changes to options will not take effect for options that are already + # bound to function parameters in the pipeline. (For example + # options.input_file, options.pdf_renderer are already bound.) + if not options.jobs: + options.jobs = available_cpu_count() + + # Performance is improved by setting Tesseract to single threaded. In tests + # this gives better throughput than letting a smaller number of Tesseract + # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this + # variable, but harmless to set if ignored. + os.environ.setdefault('OMP_THREAD_LIMIT', '1') + + check_environ(options, log) + if os.environ.get('PYTEST_CURRENT_TEST'): + os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file + + work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + + start_input_file = create_input_file(options, log, work_folder) + check_requested_output_file(options, log) + + build_pipeline(options, work_folder, start_input_file) + + return ExitCode.ok + + +""" + try: + # build_pipeline(options, work_folder, log, context) + atexit.register(cleanup_working_files, work_folder, options) + if hasattr(os, 'nice'): + os.nice(5) + except Exception as e: + log.error(str(e)) + return ExitCode.other_error + + if options.flowchart: + log.info(f"Flowchart saved to {options.flowchart}") + return ExitCode.ok + elif options.output_file == '-': + log.info("Output sent to stdout") + elif os.path.samefile(options.output_file, os.devnull): + pass # Say nothing when sending to dev null + else: + if options.output_type.startswith('pdfa'): + pdfa_info = file_claims_pdfa(options.output_file) + if pdfa_info['pass']: + msg = f"Output file is a {pdfa_info['conformance']} (as expected)" + log.info(msg) + else: + msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" + log.warning(msg) + return ExitCode.pdfa_conversion_failed + if not qpdf.check(options.output_file, log): + log.warning('Output file: The generated PDF is INVALID') + return ExitCode.invalid_output_pdf + + report_output_file_size(options, log, start_input_file, options.output_file) + + # pdfinfo = context.get_pdfinfo() + # if options.verbose: + # from pprint import pformat + # log.debug(pformat(pdfinfo)) + + # log_page_orientations(pdfinfo, log) + + return ExitCode.ok +""" diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 871c412d..0d93585c 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -383,6 +383,25 @@ def check_environ(options, _log): ) +def create_input_file(options, log, work_folder): + if options.input_file == '-': + # stdin + log.info('reading file from standard input') + target = os.path.join(work_folder, 'stdin.pdf') + with open(target, 'wb') as stream_buffer: + from shutil import copyfileobj + copyfileobj(sys.stdin.buffer, stream_buffer) + return target + else: + try: + target = os.path.join(work_folder, os.path.basename(options.input_file)) + re_symlink(options.input_file, target, log) + return target + except FileNotFoundError: + log.error("File not found - " + options.input_file) + raise InputFileError() + + def check_input_file(options, _log, start_input_file): if options.input_file == '-': # stdin diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index a81d591d..6e9b29e0 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -159,7 +159,7 @@ def get_orientation(input_file, engine_mode, timeout: float, log): def tesseract_log_output(log, stdout, input_file): - prefix = f"{(page_number(input_file)):4d}: [tesseract] " + prefix = "[tesseract] " try: text = stdout.decode() From b214aa5b38091594e05ce18ce9ef5e5d998ba743 Mon Sep 17 00:00:00 2001 From: Martin Wind Date: Wed, 3 Apr 2019 19:59:43 +0200 Subject: [PATCH 005/880] feat: move to sync (none ETL) implementation --- src/ocrmypdf/_pipeline_simple.py | 277 +++++-------------------------- src/ocrmypdf/_sync.py | 42 +++-- src/ocrmypdf/_weave.py | 61 +++---- src/ocrmypdf/optimize.py | 8 +- 4 files changed, 99 insertions(+), 289 deletions(-) diff --git a/src/ocrmypdf/_pipeline_simple.py b/src/ocrmypdf/_pipeline_simple.py index 272805b4..3b89df3d 100644 --- a/src/ocrmypdf/_pipeline_simple.py +++ b/src/ocrmypdf/_pipeline_simple.py @@ -36,8 +36,6 @@ from .exceptions import ( ) from .exec import ghostscript, tesseract from .helpers import ( - flatten_groups, - page_number, re_symlink ) from .hocrtransform import HocrTransform @@ -47,10 +45,6 @@ from .pdfinfo import Colorspace, PdfInfo VECTOR_PAGE_DPI = 400 -# -# The Pipeline -# - def triage_image_file(input_file, output_file, log, options): try: @@ -212,74 +206,6 @@ def validate_pdfinfo_options(context): ) -""" -def repair_and_parse_pdf(input_file, output_file, log, context): - options = context.get_options() - copyfile(input_file, output_file) - - detailed_page_analysis = False - if options.redo_ocr: - detailed_page_analysis = True - - try: - pdfinfo = PdfInfo( - output_file, detailed_page_analysis=detailed_page_analysis, log=log - ) - except pikepdf.PasswordError: - raise EncryptedPdfError() - except pikepdf.PdfError as e: - log.error(e) - raise InputFileError() - - if pdfinfo.needs_rendering: - log.error( - "This PDF contains dynamic XFA forms created by Adobe LiveCycle " - "Designer and can only be read by Adobe Acrobat or Adobe Reader." - ) - raise InputFileError() - - if pdfinfo.has_userunit and options.output_type.startswith('pdfa'): - log.error( - "This input file uses a PDF feature that is not supported " - "by Ghostscript, so you cannot use --output-type=pdfa for this " - "file. (Specifically, it uses the PDF-1.6 /UserUnit feature to " - "support very large or small page sizes, and Ghostscript cannot " - "output these files.) Use --output-type=pdf instead." - ) - raise InputFileError() - - if pdfinfo.has_acroform: - if options.redo_ocr: - log.error( - "This PDF has a user fillable form. --redo-ocr is not " - "currently possible on such files." - ) - raise PriorOcrFoundError() - else: - log.warning( - "This PDF has a fillable form. " - "Chances are it is a pure digital " - "document that does not need OCR." - ) - if not options.force_ocr: - log.info( - "Use the option --force-ocr to produce an image of the " - "form and all filled form fields. The output PDF will be " - "'flattened' and will no longer be fillable." - ) - - context.set_pdfinfo(pdfinfo) - log.debug(pdfinfo) -""" - - -def get_pageinfo(input_file, context): - "Get zero-based page info implied by filename, e.g. 000002.pdf -> 1" - pageno = page_number(input_file) - 1 - pageinfo = context.get_pdfinfo()[pageno] - return pageinfo - - def get_page_dpi(pageinfo, options): "Get the DPI when nonsquare DPI is tolerable" xres = max( @@ -392,59 +318,6 @@ def is_ocr_required(page_context): return ocr_required -""" -def marker_pages(input_files, output_files, log, context): - - options = context.get_options() - work_folder = context.get_work_folder() - - if is_iterable_notstr(input_files): - input_file = input_files[0] - else: - input_file = input_files - - for oo in output_files: - with suppress(FileNotFoundError): - os.unlink(oo) - - # If no files were repaired the input will be empty - if not input_file: - log.error(f"{options.input_file}: file not found or invalid argument") - raise InputFileError() - - pdfinfo = context.get_pdfinfo() - npages = len(pdfinfo) - - # Ruffus needs to see a file for any task it generates, so make very - # file a symlink back to the source. - for n in range(npages): - page = Path(work_folder) / f'{(n + 1):06d}.marker.pdf' - page.symlink_to(input_file) # pylint: disable=E1101 -""" - -""" -def ocr_or_skip(input_files, output_files, log, context): - options = context.get_options() - work_folder = context.get_work_folder() - pdfinfo = context.get_pdfinfo() - - for input_file in input_files: - pageno = page_number(input_file) - 1 - pageinfo = pdfinfo[pageno] - alt_suffix = ( - '.ocr.page.pdf' - if is_ocr_required(pageinfo, log, options) - else '.skip.page.pdf' - ) - - re_symlink( - input_file, - os.path.join(work_folder, os.path.basename(input_file)[0:6] + alt_suffix), - log, - ) -""" - - def rasterize_preview(input_file, page_context): output_file = page_context.get_path('rasterize_preview.jpg') canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) @@ -613,7 +486,7 @@ def create_ocr_image(image, page_context): draw = ImageDraw.ImageDraw(im) xres, yres = im.info['dpi'] - print('resolution %r %r', xres, yres) + page_context.log.info('resolution %r %r' % (xres, yres)) if not options.force_ocr: # Do not mask text areas when forcing OCR, because we need to OCR @@ -720,44 +593,6 @@ def create_pdf_page_from_image(image, page_context): return output_file -""" -def select_image_layer(infiles, output_file, log, context): - # Selects the image layer for the output page. If possible this is the - # orientation-corrected input page, or an image of the whole page converted - # to PDF. - - options = context.get_options() - page_pdf = next(ii for ii in infiles if ii.endswith('.ocr.oriented.pdf')) - image = next(ii for ii in infiles if ii.endswith('.image')) - - if options.lossless_reconstruction: - log.debug( - f"{page_number(page_pdf):4d}: page eligible for lossless reconstruction" - ) - re_symlink(page_pdf, output_file, log) # Still points to multipage - return - - pageinfo = get_pageinfo(image, context) - - # We rasterize a square DPI version of each page because most image - # processing tools don't support rectangular DPI. Use the square DPI as it - # accurately describes the image. It would be possible to resample the image - # at this stage back to non-square DPI to more closely resemble the input, - # except that the hocr renderer does not understand non-square DPI. The - # sandwich renderer would be fine. - dpi = get_page_square_dpi(pageinfo, options) - layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) - - # This create a single page PDF - with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: - log.debug(f'{page_number(page_pdf):4d}: convert') - img2pdf.convert( - imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf - ) - log.debug(f'{page_number(page_pdf):4d}: convert done') -""" - - def render_hocr_page(hocr, page_context): output_file = page_context.get_path('ocr_hocr.pdf') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) @@ -833,63 +668,51 @@ def get_docinfo(base_pdf, options): return pdfmark -def generate_postscript_stub(input_file, output_file, log, context): +def generate_postscript_stub(context): + output_file = context.get_path('pdfa.ps') generate_pdfa_ps(output_file) + return output_file -def convert_to_pdfa(input_files_groups, output_file, log, context): - options = context.get_options() - input_pdfinfo = context.get_pdfinfo() - - input_files = list(f for f in flatten_groups(input_files_groups)) - layers_file = next( - (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None - ) +def convert_to_pdfa(input_pdf, input_ps_stub, context): + options = context.options + input_pdfinfo = context.pdfinfo + output_file = context.get_path('pdfa.pdf') # If the DocumentInfo record contains NUL characters, Ghostscript will # produce XMP metadata which contains invalid XML entities (�). # NULs in DocumentInfo seem to be common since older Acrobats included them. # pikepdf can deal with this, but we make the world a better place by # stamping them out as soon as possible. - pdf_layers_file = pikepdf.open(layers_file) - if pdf_layers_file.docinfo: + pdf_file = pikepdf.open(input_pdf) + if pdf_file.docinfo: modified = False - for k, v in pdf_layers_file.docinfo.items(): + for k, v in pdf_file.docinfo.items(): if b'\x00' in bytes(v): - pdf_layers_file.docinfo[k] = bytes(v).replace(b'\x00', b'') + pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') modified = True if modified: - pdf_layers_file.save(layers_file) - del pdf_layers_file + pdf_file.save(input_pdf) + del pdf_file - ps = next((ii for ii in input_files if ii.endswith('.ps')), None) ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, - pdf_pages=[layers_file, ps], + pdf_pages=[input_pdf, input_ps_stub], output_file=output_file, compression=options.pdfa_image_compression, - log=log, + log=context.log, threads=options.jobs or 1, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 ) + return output_file -def metadata_fixup(input_files_groups, output_file, log, context): - options = context.get_options() - input_files = list(f for f in flatten_groups(input_files_groups)) - original_file = next( - (ii for ii in input_files if ii.endswith('.repaired.pdf')), None - ) - layers_file = next( - (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None - ) - pdfa_file = next((ii for ii in input_files if ii.endswith('pdfa.pdf')), None) - original = pikepdf.open(original_file) +def metadata_fixup(working_file, context): + output_file = context.get_path('metafix.pdf') + options = context.options + original = pikepdf.open(context.origin) docinfo = get_docinfo(original, options) - - working_file = pdfa_file if pdfa_file else layers_file - pdf = pikepdf.open(working_file) with pdf.open_metadata() as meta: meta.load_from_docinfo(docinfo, delete_missing=False) @@ -897,41 +720,36 @@ def metadata_fixup(input_files_groups, output_file, log, context): # match Ghostscript, for consistency if 'xmp:CreateDate' not in meta: meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') - if pdfa_file: - meta_original = original.open_metadata() - not_copied = set(meta_original.keys()) - set(meta.keys()) - if not_copied: - log.warning( - "Some input metadata could not be copied because it is not " - "permitted in PDF/A. You may wish to examine the output " - "PDF's XMP metadata." - ) - log.debug( - "The following metadata fields were not copied: %r", not_copied - ) + + meta_original = original.open_metadata() + not_copied = set(meta_original.keys()) - set(meta.keys()) + if not_copied: + context.log.warning( + "Some input metadata could not be copied because it is not " + "permitted in PDF/A. You may wish to examine the output " + "PDF's XMP metadata." + ) + context.log.debug( + "The following metadata fields were not copied: %r", not_copied + ) pdf.save( output_file, compress_streams=True, object_stream_mode=pikepdf.ObjectStreamMode.generate, ) + return output_file -def optimize_pdf(input_file, output_file, log, context): - optimize(input_file, output_file, log, context) +def optimize_pdf(input_file, context): + output_file = context.get_path('optimize.pdf') + optimize(input_file, output_file, context) + return output_file -def merge_sidecars(input_files_groups, output_file, log, context): - pdfinfo = context.get_pdfinfo() - - txt_files = [None] * len(pdfinfo) - - for infile in flatten_groups(input_files_groups): - if infile.endswith('.txt'): - idx = page_number(infile) - 1 - txt_files[idx] = infile - - def write_pages(stream): +def merge_sidecars(txt_files, context): + output_file = context.get_path('sidecar.txt') + with open(output_file, 'w', encoding="utf-8") as stream: for page_num, txt_file in enumerate(txt_files): if page_num != 0: stream.write('\f') # Form feed between pages @@ -949,18 +767,11 @@ def merge_sidecars(input_files_groups, output_file, log, context): stream.write(txt) else: stream.write(f'[OCR skipped on page {(page_num + 1)}]') - - if output_file == '-': - write_pages(sys.stdout) - sys.stdout.flush() - else: - with open(output_file, 'w', encoding="utf-8") as out: - write_pages(out) + return output_file -def copy_final(input_files, output_file, log, context): - input_file = next((ii for ii in input_files if ii.endswith('.pdf'))) - log.debug('%s -> %s', input_file, output_file) +def copy_final(input_file, output_file, context): + context.log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': copyfileobj(input_stream, sys.stdout.buffer) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 32b435aa..ec339bf5 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -20,10 +20,8 @@ import os # import sys import atexit from tempfile import mkdtemp -from .helpers import re_symlink from ._jobcontext import cleanup_working_files -from .exec import qpdf -# from ._weave import weave_layers +from ._weave import weave_layers from ._pipeline_simple import ( get_pdfinfo, validate_pdfinfo_options, @@ -41,21 +39,24 @@ from ._pipeline_simple import ( create_pdf_page_from_image, render_hocr_page, ocr_tesseract_textonly_pdf, + generate_postscript_stub, + convert_to_pdfa, + metadata_fixup, + merge_sidecars, + optimize_pdf, + copy_final, ) from .exceptions import ( ExitCode, ) from .helpers import available_cpu_count -from .pdfa import file_claims_pdfa from ._validation import ( check_closed_streams, preamble, check_options, check_dependency_versions, check_environ, - check_input_file, check_requested_output_file, - report_output_file_size, create_input_file, ) @@ -106,7 +107,7 @@ class PDFContext: yield PageContext(self, n) -def build_pipeline(options, work_folder, origin): +def _exec_pipeline(options, work_folder, origin): # Gather info of pdf pdfinfo = get_pdfinfo(origin) context = PDFContext(options, work_folder, origin, pdfinfo) @@ -115,7 +116,7 @@ def build_pipeline(options, work_folder, origin): validate_pdfinfo_options(context) # For every page in the pdf - page_res = [] + layers = [] for page_context in context.get_page_contexts(): # Check if OCR is required ocr_required = is_ocr_required(page_context) @@ -156,9 +157,24 @@ def build_pipeline(options, work_folder, origin): if options.pdf_renderer == 'sandwich': (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) - page_res.append((pdf_page_from_image_out, ocr_out, orientation_correction)) + layers.append((page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction)) - print(page_res) + weave_layers_out = weave_layers(layers, context) + + pdf_out = weave_layers_out + if options.output_type.startswith('pdfa'): + ps_stub_out = generate_postscript_stub(context) + pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context) + + pdf_out = metadata_fixup(pdf_out, context) + + if options.sidecar: + sidecars = [layer[3] for layer in layers] + sidecar_out = merge_sidecars(sidecars, context) + copy_final(sidecar_out, context.options.sidecar, context) + + pdf_out = optimize_pdf(pdf_out, context) + copy_final(pdf_out, context.options.output_file, context) def run_pipeline(options): @@ -193,7 +209,11 @@ def run_pipeline(options): start_input_file = create_input_file(options, log, work_folder) check_requested_output_file(options, log) - build_pipeline(options, work_folder, start_input_file) + atexit.register(cleanup_working_files, work_folder, options) + if hasattr(os, 'nice'): + os.nice(5) + + _exec_pipeline(options, work_folder, start_input_file) return ExitCode.ok diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index 4ea07e4c..c6a00bd2 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -15,15 +15,10 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from itertools import groupby from pathlib import Path import os - import pikepdf - from .exec import tesseract -from .helpers import flatten_groups, page_number - MAX_OPEN_PAGE_PDFS = int(os.environ.get('_OCRMYPDF_MAX_OPEN_PAGE_PDFS', 100)) @@ -112,7 +107,7 @@ def _weave_layers_graft( stream = bytearray(pdf_text_contents) pattern = b'/Im1 Do' idx = stream.find(pattern) - stream[idx : (idx + len(pattern))] = b' ' * len(pattern) + stream[idx:(idx + len(pattern))] = b' ' * len(pattern) pdf_text_contents = bytes(stream) base_page = pdf_base.pages.p(page_num) @@ -142,7 +137,7 @@ def _weave_layers_graft( scale_x = wp / wt scale_y = hp / ht - log.debug('%r', (scale_x, scale_y)) + # log.debug('%r', scale_x, scale_y) scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) # Translate the text so it is centered at (0, 0), rotate it there, adjust @@ -200,7 +195,7 @@ def _traverse_toc(pdf_base, visitor_fn, log): queue = set() link_keys = ('/Parent', '/First', '/Last', '/Prev', '/Next') - if not '/Outlines' in pdf_base.root: + if '/Outlines' not in pdf_base.root: return queue.add(pdf_base.root.Outlines.objgen) @@ -277,7 +272,7 @@ def _fix_toc(pdf_base, pageref_remap, log): _traverse_toc(pdf_base, visit_remap_dest, log) -def weave_layers(infiles, output_file, log, context): +def weave_layers(layers, context): """Apply text layer and/or image layer changes to baseline file This is where the magic happens. infiles will be the main PDF to modify, @@ -303,23 +298,15 @@ def weave_layers(infiles, output_file, log, context): """ - def input_sorter(key): - try: - return page_number(key) - except ValueError: - return -1 + log = context.log - flat_inputs = sorted(flatten_groups(infiles), key=input_sorter) - groups = groupby(flat_inputs, key=input_sorter) - - # Extract first item - _, basegroup = next(groups) - base = list(basegroup)[0] - path_base = Path(base).resolve() + path_base = Path(context.origin).resolve() pdf_base = pikepdf.open(path_base) keep_open = [] font, font_key, procset = None, None, None - pdfinfo = context.get_pdfinfo() + pdfinfo = context.pdfinfo + interim_output_file = context.get_path('weave_layers_interim.pdf') + output_file = context.get_path('weave_layers.pdf') pagerefs = {} # Walk the table of contents first, to trigger pikepdf/qpdf to resolve all @@ -333,40 +320,32 @@ def weave_layers(infiles, output_file, log, context): ) # Iterate rest - for page_num, layers in groups: - layers = list(layers) - log.debug(page_num) - log.debug(layers) - - text = next((ii for ii in layers if ii.endswith('.text.pdf')), None) - image = next((ii for ii in layers if ii.endswith('.image-layer.pdf')), None) - + for (pageno, image, text, sidecar, autorotate_correction) in layers: if text and not font: font, font_key = _find_font(text, pdf_base) replacing = False - content_rotation = pdfinfo[page_num - 1].rotation + content_rotation = pdfinfo[pageno].rotation path_image = Path(image).resolve() if image else None if path_image is not None and path_image != path_base: # We are replacing the old page with a rasterized PDF of the new # page log.debug("Replace") - old_objgen = pdf_base.pages[page_num - 1].objgen + old_objgen = pdf_base.pages[pageno].objgen pdf_image = pikepdf.open(image) keep_open.append(pdf_image) image_page = pdf_image.pages[0] - pdf_base.pages[page_num - 1] = image_page + pdf_base.pages[pageno] = image_page # We're adding a new page, which will get a new objgen number pair, # so we need to update any references to it. qpdf did not like # my attempt to update the old object in place, but that is an # option to consider - pagerefs[old_objgen] = pdf_base.pages[page_num - 1].objgen + pagerefs[old_objgen] = pdf_base.pages[pageno].objgen replacing = True - autorotate_correction = context.get_rotation(page_num - 1) if replacing: content_rotation = autorotate_correction text_rotation = autorotate_correction @@ -378,10 +357,10 @@ def weave_layers(infiles, output_file, log, context): if text and font: # Graft the text layer onto this page, whether new or old - strip_old = context.get_options().redo_ocr + strip_old = context.options.redo_ocr _weave_layers_graft( pdf_base=pdf_base, - page_num=page_num, + page_num=pageno + 1, text=text, font=font, font_key=font_key, @@ -392,7 +371,7 @@ def weave_layers(infiles, output_file, log, context): ) # Correct the rotation if applicable - pdf_base.pages[page_num - 1].Rotate = ( + pdf_base.pages[pageno].Rotate = ( content_rotation - autorotate_correction ) % 360 @@ -406,14 +385,14 @@ def weave_layers(infiles, output_file, log, context): _update_page_resources( page=page0, font=font, font_key=font_key, procset=procset ) - interim = output_file + f'_working{page_num}.pdf' - pdf_base.save(interim) + pdf_base.save(interim_output_file) del pdf_base keep_open = [] - pdf_base = pikepdf.open(interim) + pdf_base = pikepdf.open(interim_output_file) procset = pdf_base.pages[0].Resources.ProcSet font, font_key = None, None # Reacquire this information _fix_toc(pdf_base, pagerefs, log) pdf_base.save(output_file) + return output_file diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 4ab6920c..000f24d8 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -25,7 +25,7 @@ from pathlib import Path from PIL import Image import pikepdf -from pikepdf import Name, Dictionary, Array +from pikepdf import Name, Dictionary from . import leptonica from ._jobcontext import JobContext @@ -427,9 +427,9 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) -def optimize(input_file, output_file, log, context): - - options = context.get_options() +def optimize(input_file, output_file, context): + log = context.log + options = context.options if options.optimize == 0: re_symlink(input_file, output_file, log) return From 783a128bd1bf3195a2c96b78690b8a26c0599689 Mon Sep 17 00:00:00 2001 From: mawi Date: Thu, 4 Apr 2019 21:02:38 +0200 Subject: [PATCH 006/880] feat: move to sync (none ETL) implementation - remove ruffus --- src/ocrmypdf/__main__.py | 6 +- src/ocrmypdf/_jobcontext.py | 130 ++- src/ocrmypdf/_pipeline.py | 489 ++++------- src/ocrmypdf/_pipeline_simple.py | 784 ------------------ src/ocrmypdf/_ruffus.py | 508 ------------ src/ocrmypdf/_sync.py | 63 +- src/ocrmypdf/optimize.py | 19 +- ...processing.py => _test_multiprocessing.py} | 0 tests/test_main.py | 2 +- tests/test_metadata.py | 16 +- 10 files changed, 277 insertions(+), 1740 deletions(-) delete mode 100644 src/ocrmypdf/_pipeline_simple.py delete mode 100644 src/ocrmypdf/_ruffus.py rename tests/{test_multiprocessing.py => _test_multiprocessing.py} (100%) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index f9c6dced..afa3cb20 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -462,9 +462,9 @@ debugging.add_argument( action='store_true', help="Keep temporary files (helpful for debugging)", ) -debugging.add_argument( - '--flowchart', type=str, help="Generate the pipeline execution flowchart" -) +# debugging.add_argument( +# '--flowchart', type=str, help="Generate the pipeline execution flowchart" +# ) def run(args=None): diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index c9367ba6..6b059644 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -17,62 +17,41 @@ import shutil import sys +import os from contextlib import suppress -from multiprocessing.managers import SyncManager - -from .pdfinfo import PdfInfo -class JobContext: - """Holds our context for a particular run of the pipeline +class PDFContext: + """Holds our context for a particular run of the pipeline""" - A multiprocessing manager effectively creates a separate process - that keeps the master job context object. Other threads access - job context via multiprocessing proxy objects. - - While this would naturally lend itself @property's it seems to make - a little more sense to use functions to make it explicitly that the - invocation requires marshalling data across a process boundary. - - """ - - def __init__(self): - self.pdfinfo = None - self.options = None - self.work_folder = None - self.rotations = {} - - def generate_pdfinfo(self, infile): - self.pdfinfo = PdfInfo(infile) - - def get_pdfinfo(self): - "What we know about the input PDF" - return self.pdfinfo - - def set_pdfinfo(self, pdfinfo): - self.pdfinfo = pdfinfo - - def get_options(self): - return self.options - - def set_options(self, options): + def __init__(self, options, work_folder, origin, pdfinfo): self.options = options - - def get_work_folder(self): - return self.work_folder - - def set_work_folder(self, work_folder): self.work_folder = work_folder + self.origin = origin + self.pdfinfo = pdfinfo + self.log = get_logger(options, '%s: ' % os.path.basename(origin)) - def get_rotation(self, pageno): - return self.rotations.get(pageno, 0) + def get_path(self, name): + return os.path.join(self.work_folder, name) - def set_rotation(self, pageno, value): - self.rotations[pageno] = value + def get_page_contexts(self): + npages = len(self.pdfinfo) + for n in range(npages): + yield PageContext(self, n) -class JobContextManager(SyncManager): - pass +class PageContext: + """Holds our context for a page""" + + def __init__(self, pdf_context, pageno): + self.pdf_context = pdf_context + self.options = pdf_context.options + self.pageno = pageno + self.pageinfo = pdf_context.pdfinfo[pageno] + self.log = get_logger(pdf_context.options, '%s Page %d: ' % (os.path.basename(pdf_context.origin), pageno + 1)) + + def get_path(self, name): + return os.path.join(self.pdf_context.work_folder, "page_%d_%s" % (self.pageno, name)) def cleanup_working_files(work_folder, options): @@ -81,3 +60,62 @@ def cleanup_working_files(work_folder, options): else: with suppress(FileNotFoundError): shutil.rmtree(work_folder) + + +def get_logger(options=None, prefix=''): + level = INFO # TODO: add option + if options is not None and options.output_file == '-' or options.sidecar == '-': + return NullLogger() + return Logger(prefix, level) + + +ERROR = 40 +WARN = 30 +INFO = 20 +DEBUG = 10 + + +class Logger: + def __init__(self, prefix, level=INFO): + self.prefix = prefix + self.level = level + + def debug(self, *args, **kwargs): + if self.level <= DEBUG: + print('DEBUG', self.prefix, end='') + print(*args, **kwargs) + + def info(self, *args, **kwargs): + if self.level <= INFO: + print('INFO', self.prefix, end='') + print(*args, **kwargs) + + def warning(self, *args, **kwargs): + self.warn(*args, **kwargs) + + def warn(self, *args, **kwargs): + if self.level <= WARN: + print('WARN', self.prefix, end='') + print(*args, **kwargs) + + def error(self, *args, **kwargs): + if self.level <= ERROR: + print('ERROR', self.prefix, end='') + print(*args, **kwargs) + + +class NullLogger: + def debug(self, *args, **kwargs): + pass + + def info(self, *args, **kwargs): + pass + + def warning(self, *args, **kwargs): + pass + + def warn(self, *args, **kwargs): + pass + + def error(self, *args, **kwargs): + pass diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index d6e0c491..3b553a7e 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -18,10 +18,8 @@ import os import re import sys -from contextlib import suppress from datetime import datetime, timezone -from pathlib import Path -from shutil import copyfile, copyfileobj +from shutil import copyfileobj import img2pdf from PIL import Image @@ -34,14 +32,10 @@ from .exceptions import ( DpiError, EncryptedPdfError, InputFileError, - PriorOcrFoundError, UnsupportedImageFormatError, ) from .exec import ghostscript, tesseract from .helpers import ( - flatten_groups, - is_iterable_notstr, - page_number, re_symlink ) from .hocrtransform import HocrTransform @@ -51,10 +45,6 @@ from .pdfinfo import Colorspace, PdfInfo VECTOR_PAGE_DPI = 400 -# -# The Pipeline -# - def triage_image_file(input_file, output_file, log, options): try: @@ -164,31 +154,28 @@ def triage(input_file, output_file, log, context): triage_image_file(input_file, output_file, log, options) -def repair_and_parse_pdf(input_file, output_file, log, context): - options = context.get_options() - copyfile(input_file, output_file) - - detailed_page_analysis = False - if options.redo_ocr: - detailed_page_analysis = True - +def get_pdfinfo(input_file, detailed_page_analysis=False): try: - pdfinfo = PdfInfo( - output_file, detailed_page_analysis=detailed_page_analysis, log=log + return PdfInfo( + input_file, detailed_page_analysis=detailed_page_analysis ) except pikepdf.PasswordError: raise EncryptedPdfError() - except pikepdf.PdfError as e: - log.error(e) + except pikepdf.PdfError: raise InputFileError() + +def validate_pdfinfo_options(context): + log = context.log + pdfinfo = context.pdfinfo + options = context.options + if pdfinfo.needs_rendering: log.error( "This PDF contains dynamic XFA forms created by Adobe LiveCycle " "Designer and can only be read by Adobe Acrobat or Adobe Reader." ) raise InputFileError() - if pdfinfo.has_userunit and options.output_type.startswith('pdfa'): log.error( "This input file uses a PDF feature that is not supported " @@ -198,16 +185,15 @@ def repair_and_parse_pdf(input_file, output_file, log, context): "output these files.) Use --output-type=pdf instead." ) raise InputFileError() - if pdfinfo.has_acroform: if options.redo_ocr: log.error( "This PDF has a user fillable form. --redo-ocr is not " "currently possible on such files." ) - raise PriorOcrFoundError() + raise InputFileError() else: - log.warning( + log.warn( "This PDF has a fillable form. " "Chances are it is a pure digital " "document that does not need OCR." @@ -219,16 +205,6 @@ def repair_and_parse_pdf(input_file, output_file, log, context): "'flattened' and will no longer be fillable." ) - context.set_pdfinfo(pdfinfo) - log.debug(pdfinfo) - - -def get_pageinfo(input_file, context): - "Get zero-based page info implied by filename, e.g. 000002.pdf -> 1" - pageno = page_number(input_file) - 1 - pageinfo = context.get_pdfinfo()[pageno] - return pageinfo - def get_page_dpi(pageinfo, options): "Get the DPI when nonsquare DPI is tolerable" @@ -272,33 +248,31 @@ def get_canvas_square_dpi(pageinfo, options): ) -def is_ocr_required(pageinfo, log, options): - page = pageinfo.pageno + 1 +def is_ocr_required(page_context): + pageinfo = page_context.pageinfo + options = page_context.options + log = page_context.log + ocr_required = True if pageinfo.has_text: - prefix = f"{page:4d}: page already has text! - " - if not options.force_ocr and not (options.skip_text or options.redo_ocr): - log.error(prefix + "aborting (use --force-ocr to force OCR)") - raise PriorOcrFoundError() + log.error("page already has text! - aborting (use --force-ocr to force OCR)") + ocr_required = False elif options.force_ocr: - log.info(prefix + "rasterizing text and running OCR anyway") + log.info("page already has text! - rasterizing text and running OCR anyway") ocr_required = True elif options.redo_ocr: if pageinfo.has_corrupt_text: - log.warning( - prefix + ( - "some text on this page cannot be mapped to characters: " - "consider using --force-ocr instead", - ) + log.warn( + "some text on this page cannot be mapped to characters: " + "consider using --force-ocr instead", ) - raise PriorOcrFoundError() # Wrong error but will do for now else: - log.info(prefix + "redoing OCR") + log.info("redoing OCR") ocr_required = True elif options.skip_text: - log.info(prefix + "skipping all processing on this page") + log.info("skipping all processing on this page") ocr_required = False elif not pageinfo.images and not options.lossless_reconstruction: # We found a page with no images and no text. That means it may @@ -311,14 +285,14 @@ def is_ocr_required(pageinfo, log, options): if options.force_ocr and options.oversample: # The user really wants to reprocess this file log.info( - f"{page:4d}: page has no images - " + "page has no images - " f"rasterizing at {options.oversample} DPI because " "--force-ocr --oversample was specified" ) elif options.force_ocr: # Warn the user they might not want to do this - log.warning( - f"{page:4d}: page has no images - " + log.warn( + "page has no images - " "all vector content will be " f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely " "increasing file size. Use --oversample to adjust the " @@ -326,7 +300,7 @@ def is_ocr_required(pageinfo, log, options): ) else: log.info( - f"{page:4d}: page has no images - " + "page has no images - " "skipping all processing on this page to avoid losing detail. " "Use --force-ocr if you wish to perform OCR on pages that " "have vector content." @@ -337,82 +311,31 @@ def is_ocr_required(pageinfo, log, options): pixel_count = pageinfo.width_pixels * pageinfo.height_pixels if pixel_count > (options.skip_big * 1_000_000): ocr_required = False - log.warning( - f"{page:4d}: page too big, skipping OCR " + log.warn( + "page too big, skipping OCR " f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)" ) return ocr_required -def marker_pages(input_files, output_files, log, context): - - options = context.get_options() - work_folder = context.get_work_folder() - - if is_iterable_notstr(input_files): - input_file = input_files[0] - else: - input_file = input_files - - for oo in output_files: - with suppress(FileNotFoundError): - os.unlink(oo) - - # If no files were repaired the input will be empty - if not input_file: - log.error(f"{options.input_file}: file not found or invalid argument") - raise InputFileError() - - pdfinfo = context.get_pdfinfo() - npages = len(pdfinfo) - - # Ruffus needs to see a file for any task it generates, so make very - # file a symlink back to the source. - for n in range(npages): - page = Path(work_folder) / f'{(n + 1):06d}.marker.pdf' - page.symlink_to(input_file) # pylint: disable=E1101 - - -def ocr_or_skip(input_files, output_files, log, context): - options = context.get_options() - work_folder = context.get_work_folder() - pdfinfo = context.get_pdfinfo() - - for input_file in input_files: - pageno = page_number(input_file) - 1 - pageinfo = pdfinfo[pageno] - alt_suffix = ( - '.ocr.page.pdf' - if is_ocr_required(pageinfo, log, options) - else '.skip.page.pdf' - ) - - re_symlink( - input_file, - os.path.join(work_folder, os.path.basename(input_file)[0:6] + alt_suffix), - log, - ) - - -def rasterize_preview(input_file, output_file, log, context): - pageinfo = get_pageinfo(input_file, context) - options = context.get_options() - canvas_dpi = get_canvas_square_dpi(pageinfo, options) - page_dpi = get_page_square_dpi(pageinfo, options) - +def rasterize_preview(input_file, page_context): + output_file = page_context.get_path('rasterize_preview.jpg') + canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) + page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) ghostscript.rasterize_pdf( input_file, output_file, xres=canvas_dpi, yres=canvas_dpi, raster_device='jpeggray', - log=log, + log=page_context.log, page_dpi=(page_dpi, page_dpi), - pageno=page_number(input_file), + pageno=page_context.pageinfo.pageno + 1, ) + return output_file -def orient_page(infiles, output_file, log, context): +def get_orientation_correction(preview, page_context): """ Work out orientation correct for each page. @@ -430,32 +353,22 @@ def orient_page(infiles, output_file, log, context): """ - options = context.get_options() - page_pdf = next(ii for ii in infiles if ii.endswith('.page.pdf')) - - if not options.rotate_pages: - re_symlink(page_pdf, output_file, log) - return - preview = next(ii for ii in infiles if ii.endswith('.preview.jpg')) - orient_conf = tesseract.get_orientation( preview, - engine_mode=options.tesseract_oem, - timeout=options.tesseract_timeout, - log=log, + engine_mode=page_context.options.tesseract_oem, + timeout=page_context.options.tesseract_timeout, + log=page_context.log, ) direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} - pageno = page_number(page_pdf) - 1 - pdfinfo = context.get_pdfinfo() - existing_rotation = pdfinfo[pageno].rotation + existing_rotation = page_context.pageinfo.rotation correction = orient_conf.angle % 360 apply_correction = False action = '' - if orient_conf.confidence >= options.rotate_pages_threshold: + if orient_conf.confidence >= page_context.options.rotate_pages_threshold: if correction != 0: apply_correction = True action = ' - will rotate' @@ -474,26 +387,25 @@ def orient_page(infiles, output_file, log, context): ) facing += 'page is facing {}'.format(direction.get(orient_conf.angle, '?')) - log.info( + page_context.log.debug( '{pagenum:4d}: {facing}, confidence {conf:.2f}{action}'.format( - pagenum=page_number(preview), + pagenum=page_context.pageinfo.pageno, facing=facing, conf=orient_conf.confidence, action=action, ) ) - re_symlink(page_pdf, output_file, log) if apply_correction: - context.set_rotation(pageno, correction) + return correction + return 0 -def rasterize_with_ghostscript(input_file, output_file, log, context): - options = context.get_options() - pageinfo = get_pageinfo(input_file, context) - +def rasterize(input_file, page_context, correction=0): colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m'] device_idx = 0 + output_file = page_context.get_path('rasterize.png') + pageinfo = page_context.pageinfo def at_least(cs): return max(device_idx, colorspaces.index(cs)) @@ -511,14 +423,12 @@ def rasterize_with_ghostscript(input_file, output_file, log, context): device = colorspaces[device_idx] - log.debug(f"Rasterize {os.path.basename(input_file)} with {device}") + page_context.log.debug(f"Rasterize with {device}") # Produce the page image with square resolution or else deskew and OCR # will not work properly. - canvas_dpi = get_canvas_square_dpi(pageinfo, options) - page_dpi = get_page_square_dpi(pageinfo, options) - - correction = context.get_rotation(page_number(input_file) - 1) + canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options) + page_dpi = get_page_square_dpi(pageinfo, page_context.options) ghostscript.rasterize_pdf( input_file, @@ -526,64 +436,47 @@ def rasterize_with_ghostscript(input_file, output_file, log, context): xres=canvas_dpi, yres=canvas_dpi, raster_device=device, - log=log, + log=page_context.log, page_dpi=(page_dpi, page_dpi), - pageno=page_number(input_file), + pageno=pageinfo.pageno + 1, rotation=correction, - filter_vector=options.remove_vectors, + filter_vector=page_context.options.remove_vectors, ) + return output_file -def preprocess_remove_background(input_file, output_file, log, context): - options = context.get_options() - if not options.remove_background: - re_symlink(input_file, output_file, log) - return - - pageinfo = get_pageinfo(input_file, context) - - if any(image.bpc > 1 for image in pageinfo.images): +def preprocess_remove_background(input_file, page_context): + if any(image.bpc > 1 for image in page_context.pageinfo.images): + output_file = page_context.get_path('pp_rm_bg.png') leptonica.remove_background(input_file, output_file) + return output_file else: - log.info(f"{pageinfo.pageno:4d}: background removal skipped on mono page") - re_symlink(input_file, output_file, log) + page_context.log.info("background removal skipped on mono page") + return input_file -def preprocess_deskew(input_file, output_file, log, context): - options = context.get_options() - if not options.deskew: - re_symlink(input_file, output_file, log) - return - - pageinfo = get_pageinfo(input_file, context) - dpi = get_page_square_dpi(pageinfo, options) - +def preprocess_deskew(input_file, page_context): + output_file = page_context.get_path('pp_deskew.png') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) leptonica.deskew(input_file, output_file, dpi) + return output_file -def preprocess_clean(input_file, output_file, log, context): - options = context.get_options() - if not options.clean: - re_symlink(input_file, output_file, log) - return - +def preprocess_clean(input_file, page_context): from .exec import unpaper - - pageinfo = get_pageinfo(input_file, context) - dpi = get_page_square_dpi(pageinfo, options) - - unpaper.clean(input_file, output_file, dpi, log, options.unpaper_args) + output_file = page_context.get_path('pp_clean.png') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + unpaper.clean(input_file, output_file, dpi, page_context.log, page_context.options.unpaper_args) + return output_file -def select_ocr_image(infiles, output_file, log, context): - """Select the image we send for OCR. May not be the same as the display +def create_ocr_image(image, page_context): + """Create the image we send for OCR. May not be the same as the display image depending on preprocessing. This image will never be shown to the user.""" - image = infiles[0] - options = context.get_options() - pageinfo = get_pageinfo(image, context) - + output_file = page_context.get_path('ocr.png') + options = page_context.options with Image.open(image) as im: from PIL import ImageColor from PIL import ImageDraw @@ -593,7 +486,7 @@ def select_ocr_image(infiles, output_file, log, context): draw = ImageDraw.ImageDraw(im) xres, yres = im.info['dpi'] - log.debug('resolution %r %r', xres, yres) + page_context.log.info('resolution %r %r' % (xres, yres)) if not options.force_ocr: # Do not mask text areas when forcing OCR, because we need to OCR @@ -602,7 +495,7 @@ def select_ocr_image(infiles, output_file, log, context): if options.redo_ocr: mask = True # Mask visible text, but not invisible text - for textarea in pageinfo.get_textareas(visible=mask, corrupt=None): + for textarea in page_context.pageinfo.get_textareas(visible=mask, corrupt=None): # Calculate resolution based on the image size and page dimensions # without regard whatever resolution is in pageinfo (may differ or # be None) @@ -615,7 +508,7 @@ def select_ocr_image(infiles, output_file, log, context): im.height - bbox[1] * yscale, ] pixcoords = [int(round(c)) for c in pixcoords] - log.debug('blanking %r', pixcoords) + print('blanking %r', pixcoords) draw.rectangle(pixcoords, fill=white) # draw.rectangle(pixcoords, outline=pink) @@ -627,7 +520,7 @@ def select_ocr_image(infiles, output_file, log, context): barcodes = pix.locate_barcodes() for barcode in barcodes: decoded, rect = barcode - log.info('masking barcode %s %r', decoded, rect) + print('masking barcode %s %r', decoded, rect) draw.rectangle(rect, fill=white) im = pix.topil() @@ -635,13 +528,16 @@ def select_ocr_image(infiles, output_file, log, context): # Pillow requires integer DPI dpi = round(xres), round(yres) im.save(output_file, dpi=dpi) + return output_file -def ocr_tesseract_hocr(input_file, output_files, log, context): - options = context.get_options() +def ocr_tesseract_hocr(input_file, page_context): + hocr_out = page_context.get_path('ocr_hocr.hocr') + hocr_text_out = page_context.get_path('ocr_hocr.txt') + options = page_context.options tesseract.generate_hocr( input_file=input_file, - output_files=output_files, + output_files=[hocr_out, hocr_text_out], language=options.language, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, @@ -649,86 +545,57 @@ def ocr_tesseract_hocr(input_file, output_files, log, context): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, - log=log, + log=page_context.log, ) + return (hocr_out, hocr_text_out) -def select_visible_page_image(infiles, output_file, log, context): - """Selects a whole page image that we can show the user (if necessary)""" - - options = context.get_options() - if options.clean_final: - image_suffix = '.pp-clean.png' - elif options.deskew: - image_suffix = '.pp-deskew.png' - elif options.remove_background: - image_suffix = '.pp-background.png' - else: - image_suffix = '.page.png' - image = next(ii for ii in infiles if ii.endswith(image_suffix)) - - pageinfo = get_pageinfo(image, context) - if pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images): - log.debug(f'{page_number(image):4d}: JPEG input -> JPEG output') - # If all images were JPEGs originally, produce a JPEG as output - with Image.open(image) as im: - # At this point the image should be a .png, but deskew, unpaper - # might have removed the DPI information. In this case, fall back to - # square DPI used to rasterize. When the preview image was - # rasterized, it was also converted to square resolution, which is - # what we want to give tesseract, so keep it square. - fallback_dpi = get_page_square_dpi(pageinfo, options) - dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi)) - - # Pillow requires integer DPI - dpi = round(dpi[0]), round(dpi[1]) - im.save(output_file, format='JPEG', dpi=dpi) - else: - re_symlink(image, output_file, log) +def should_visible_page_image_use_jpg(pageinfo): + # If all images were JPEGs originally, produce a JPEG as output + return pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images) -def select_image_layer(infiles, output_file, log, context): - """Selects the image layer for the output page. If possible this is the - orientation-corrected input page, or an image of the whole page converted - to PDF.""" +def create_visible_page_jpg(image, page_context): + output_file = page_context.get_path('visible.jpg') + with Image.open(image) as im: + # At this point the image should be a .png, but deskew, unpaper + # might have removed the DPI information. In this case, fall back to + # square DPI used to rasterize. When the preview image was + # rasterized, it was also converted to square resolution, which is + # what we want to give tesseract, so keep it square. + fallback_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi)) - options = context.get_options() - page_pdf = next(ii for ii in infiles if ii.endswith('.ocr.oriented.pdf')) - image = next(ii for ii in infiles if ii.endswith('.image')) + # Pillow requires integer DPI + dpi = round(dpi[0]), round(dpi[1]) + im.save(output_file, format='JPEG', dpi=dpi) + return output_file - if options.lossless_reconstruction: - log.debug( - f"{page_number(page_pdf):4d}: page eligible for lossless reconstruction" - ) - re_symlink(page_pdf, output_file, log) # Still points to multipage - return - - pageinfo = get_pageinfo(image, context) +def create_pdf_page_from_image(image, page_context): # We rasterize a square DPI version of each page because most image # processing tools don't support rectangular DPI. Use the square DPI as it # accurately describes the image. It would be possible to resample the image # at this stage back to non-square DPI to more closely resemble the input, # except that the hocr renderer does not understand non-square DPI. The # sandwich renderer would be fine. - dpi = get_page_square_dpi(pageinfo, options) + output_file = page_context.get_path('visible.pdf') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) # This create a single page PDF with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: - log.debug(f'{page_number(page_pdf):4d}: convert') + page_context.log.debug('convert') img2pdf.convert( imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf ) - log.debug(f'{page_number(page_pdf):4d}: convert done') + page_context.log.debug('convert done') + return output_file -def render_hocr_page(infiles, output_file, log, context): - options = context.get_options() - hocr = next(ii for ii in infiles if ii.endswith('.hocr')) - pageinfo = get_pageinfo(hocr, context) - dpi = get_page_square_dpi(pageinfo, options) - +def render_hocr_page(hocr, page_context): + output_file = page_context.get_path('ocr_hocr.pdf') + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) hocrtransform = HocrTransform(hocr, dpi) hocrtransform.to_pdf( output_file, @@ -737,17 +604,13 @@ def render_hocr_page(infiles, output_file, log, context): invisibleText=True, interwordSpaces=True, ) + return output_file -def ocr_tesseract_textonly_pdf(infiles, outfiles, log, context): - options = context.get_options() - input_image = next((ii for ii in infiles if ii.endswith('.ocr.png')), '') - if not input_image: - raise ValueError("No image rendered?") - - output_pdf = next((ii for ii in outfiles if ii.endswith('.pdf'))) - output_text = next((ii for ii in outfiles if ii.endswith('.txt'))) - +def ocr_tesseract_textonly_pdf(input_image, page_context): + output_pdf = page_context.get_path('ocr_tess.pdf') + output_text = page_context.get_path('ocr_tess.txt') + options = page_context.options tesseract.generate_pdf( input_image=input_image, skip_pdf=None, @@ -761,8 +624,9 @@ def ocr_tesseract_textonly_pdf(infiles, outfiles, log, context): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, - log=log, + log=page_context.log, ) + return (output_pdf, output_text) def get_docinfo(base_pdf, options): @@ -804,63 +668,51 @@ def get_docinfo(base_pdf, options): return pdfmark -def generate_postscript_stub(input_file, output_file, log, context): +def generate_postscript_stub(context): + output_file = context.get_path('pdfa.ps') generate_pdfa_ps(output_file) + return output_file -def convert_to_pdfa(input_files_groups, output_file, log, context): - options = context.get_options() - input_pdfinfo = context.get_pdfinfo() - - input_files = list(f for f in flatten_groups(input_files_groups)) - layers_file = next( - (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None - ) +def convert_to_pdfa(input_pdf, input_ps_stub, context): + options = context.options + input_pdfinfo = context.pdfinfo + output_file = context.get_path('pdfa.pdf') # If the DocumentInfo record contains NUL characters, Ghostscript will # produce XMP metadata which contains invalid XML entities (�). # NULs in DocumentInfo seem to be common since older Acrobats included them. # pikepdf can deal with this, but we make the world a better place by # stamping them out as soon as possible. - pdf_layers_file = pikepdf.open(layers_file) - if pdf_layers_file.docinfo: + pdf_file = pikepdf.open(input_pdf) + if pdf_file.docinfo: modified = False - for k, v in pdf_layers_file.docinfo.items(): + for k, v in pdf_file.docinfo.items(): if b'\x00' in bytes(v): - pdf_layers_file.docinfo[k] = bytes(v).replace(b'\x00', b'') + pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') modified = True if modified: - pdf_layers_file.save(layers_file) - del pdf_layers_file + pdf_file.save(input_pdf) + del pdf_file - ps = next((ii for ii in input_files if ii.endswith('.ps')), None) ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, - pdf_pages=[layers_file, ps], + pdf_pages=[input_pdf, input_ps_stub], output_file=output_file, compression=options.pdfa_image_compression, - log=log, + log=context.log, threads=options.jobs or 1, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 ) + return output_file -def metadata_fixup(input_files_groups, output_file, log, context): - options = context.get_options() - input_files = list(f for f in flatten_groups(input_files_groups)) - original_file = next( - (ii for ii in input_files if ii.endswith('.repaired.pdf')), None - ) - layers_file = next( - (ii for ii in input_files if ii.endswith('layers.rendered.pdf')), None - ) - pdfa_file = next((ii for ii in input_files if ii.endswith('pdfa.pdf')), None) - original = pikepdf.open(original_file) +def metadata_fixup(working_file, context): + output_file = context.get_path('metafix.pdf') + options = context.options + original = pikepdf.open(context.origin) docinfo = get_docinfo(original, options) - - working_file = pdfa_file if pdfa_file else layers_file - pdf = pikepdf.open(working_file) with pdf.open_metadata() as meta: meta.load_from_docinfo(docinfo, delete_missing=False) @@ -868,16 +720,25 @@ def metadata_fixup(input_files_groups, output_file, log, context): # match Ghostscript, for consistency if 'xmp:CreateDate' not in meta: meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') - if pdfa_file: - meta_original = original.open_metadata() - not_copied = set(meta_original.keys()) - set(meta.keys()) - if not_copied: - log.warning( + + meta_original = original.open_metadata() + not_copied = set(meta_original.keys()) - set(meta.keys()) + if not_copied: + if options.output_type.startswith('pdfa'): + context.log.warn( "Some input metadata could not be copied because it is not " "permitted in PDF/A. You may wish to examine the output " "PDF's XMP metadata." ) - log.debug( + context.log.debug( + "The following metadata fields were not copied: %r", not_copied + ) + else: + context.log.error( + "Some input metadata could not be copied." + "You may wish to examine the output PDF's XMP metadata." + ) + context.log.info( "The following metadata fields were not copied: %r", not_copied ) @@ -886,23 +747,18 @@ def metadata_fixup(input_files_groups, output_file, log, context): compress_streams=True, object_stream_mode=pikepdf.ObjectStreamMode.generate, ) + return output_file -def optimize_pdf(input_file, output_file, log, context): - optimize(input_file, output_file, log, context) +def optimize_pdf(input_file, context): + output_file = context.get_path('optimize.pdf') + optimize(input_file, output_file, context) + return output_file -def merge_sidecars(input_files_groups, output_file, log, context): - pdfinfo = context.get_pdfinfo() - - txt_files = [None] * len(pdfinfo) - - for infile in flatten_groups(input_files_groups): - if infile.endswith('.txt'): - idx = page_number(infile) - 1 - txt_files[idx] = infile - - def write_pages(stream): +def merge_sidecars(txt_files, context): + output_file = context.get_path('sidecar.txt') + with open(output_file, 'w', encoding="utf-8") as stream: for page_num, txt_file in enumerate(txt_files): if page_num != 0: stream.write('\f') # Form feed between pages @@ -920,18 +776,11 @@ def merge_sidecars(input_files_groups, output_file, log, context): stream.write(txt) else: stream.write(f'[OCR skipped on page {(page_num + 1)}]') - - if output_file == '-': - write_pages(sys.stdout) - sys.stdout.flush() - else: - with open(output_file, 'w', encoding="utf-8") as out: - write_pages(out) + return output_file -def copy_final(input_files, output_file, log, context): - input_file = next((ii for ii in input_files if ii.endswith('.pdf'))) - log.debug('%s -> %s', input_file, output_file) +def copy_final(input_file, output_file, context): + context.log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': copyfileobj(input_stream, sys.stdout.buffer) diff --git a/src/ocrmypdf/_pipeline_simple.py b/src/ocrmypdf/_pipeline_simple.py deleted file mode 100644 index 3b89df3d..00000000 --- a/src/ocrmypdf/_pipeline_simple.py +++ /dev/null @@ -1,784 +0,0 @@ -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -import os -import re -import sys -from datetime import datetime, timezone -from shutil import copyfileobj - -import img2pdf -from PIL import Image - -import pikepdf -from pikepdf.models.metadata import encode_pdf_date - -from . import PROGRAM_NAME, VERSION, leptonica -from .exceptions import ( - DpiError, - EncryptedPdfError, - InputFileError, - UnsupportedImageFormatError, -) -from .exec import ghostscript, tesseract -from .helpers import ( - re_symlink -) -from .hocrtransform import HocrTransform -from .optimize import optimize -from .pdfa import generate_pdfa_ps -from .pdfinfo import Colorspace, PdfInfo - -VECTOR_PAGE_DPI = 400 - - -def triage_image_file(input_file, output_file, log, options): - try: - log.info("Input file is not a PDF, checking if it is an image...") - im = Image.open(input_file) - except EnvironmentError as e: - msg = str(e) - - # Recover the original filename - realpath = '' - if os.path.islink(input_file): - realpath = os.path.realpath(input_file) - elif os.path.isfile(input_file): - realpath = '' - msg = msg.replace(input_file, realpath) - log.error(msg) - raise UnsupportedImageFormatError() from e - else: - log.info("Input file is an image") - - if 'dpi' in im.info: - if im.info['dpi'] <= (96, 96) and not options.image_dpi: - log.info("Image size: (%d, %d)" % im.size) - log.info("Image resolution: (%d, %d)" % im.info['dpi']) - log.error( - "Input file is an image, but the resolution (DPI) is " - "not credible. Estimate the resolution at which the " - "image was scanned and specify it using --image-dpi." - ) - raise DpiError() - elif not options.image_dpi: - log.info("Image size: (%d, %d)" % im.size) - log.error( - "Input file is an image, but has no resolution (DPI) " - "in its metadata. Estimate the resolution at which " - "image was scanned and specify it using --image-dpi." - ) - raise DpiError() - - if im.mode in ('RGBA', 'LA'): - log.error( - "The input image has an alpha channel. Remove the alpha " - "channel first." - ) - raise UnsupportedImageFormatError() - - if 'iccprofile' not in im.info: - if im.mode == 'RGB': - log.info('Input image has no ICC profile, assuming sRGB') - elif im.mode == 'CMYK': - log.info('Input CMYK image has no ICC profile, not usable') - raise UnsupportedImageFormatError() - im.close() - - try: - log.info("Image seems valid. Try converting to PDF...") - layout_fun = img2pdf.default_layout_fun - if options.image_dpi: - layout_fun = img2pdf.get_fixed_dpi_layout_fun( - (options.image_dpi, options.image_dpi) - ) - with open(output_file, 'wb') as outf: - img2pdf.convert( - input_file, - layout_fun=layout_fun, - with_pdfrw=False, - outputstream=outf - ) - log.info("Successfully converted to PDF, processing...") - except img2pdf.ImageOpenError as e: - log.error(e) - raise UnsupportedImageFormatError() from e - - -def _pdf_guess_version(input_file, search_window=1024): - """Try to find version signature at start of file. - - Not robust enough to deal with appended files. - - Returns empty string if not found, indicating file is probably not PDF. - """ - - with open(input_file, 'rb') as f: - signature = f.read(search_window) - m = re.search(br'%PDF-(\d\.\d)', signature) - if m: - return m.group(1) - return '' - - -def triage(input_file, output_file, log, context): - - options = context.get_options() - try: - if _pdf_guess_version(input_file): - if options.image_dpi: - log.warning( - "Argument --image-dpi ignored because the " - "input file is a PDF, not an image." - ) - re_symlink(input_file, output_file, log) - return - except EnvironmentError as e: - log.error(e) - raise InputFileError() from e - - triage_image_file(input_file, output_file, log, options) - - -def get_pdfinfo(input_file, detailed_page_analysis=False): - try: - return PdfInfo( - input_file, detailed_page_analysis=detailed_page_analysis - ) - except pikepdf.PasswordError: - raise EncryptedPdfError() - except pikepdf.PdfError: - raise InputFileError() - - -def validate_pdfinfo_options(context): - log = context.log - pdfinfo = context.pdfinfo - options = context.options - - if pdfinfo.needs_rendering: - log.error( - "This PDF contains dynamic XFA forms created by Adobe LiveCycle " - "Designer and can only be read by Adobe Acrobat or Adobe Reader." - ) - raise InputFileError() - if pdfinfo.has_userunit and options.output_type.startswith('pdfa'): - log.error( - "This input file uses a PDF feature that is not supported " - "by Ghostscript, so you cannot use --output-type=pdfa for this " - "file. (Specifically, it uses the PDF-1.6 /UserUnit feature to " - "support very large or small page sizes, and Ghostscript cannot " - "output these files.) Use --output-type=pdf instead." - ) - raise InputFileError() - if pdfinfo.has_acroform: - if options.redo_ocr: - log.error( - "This PDF has a user fillable form. --redo-ocr is not " - "currently possible on such files." - ) - raise InputFileError() - else: - log.warn( - "This PDF has a fillable form. " - "Chances are it is a pure digital " - "document that does not need OCR." - ) - if not options.force_ocr: - log.info( - "Use the option --force-ocr to produce an image of the " - "form and all filled form fields. The output PDF will be " - "'flattened' and will no longer be fillable." - ) - - -def get_page_dpi(pageinfo, options): - "Get the DPI when nonsquare DPI is tolerable" - xres = max( - pageinfo.xres or VECTOR_PAGE_DPI, - options.oversample or 0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - ) - yres = max( - pageinfo.yres or VECTOR_PAGE_DPI, - options.oversample or 0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - ) - return (float(xres), float(yres)) - - -def get_page_square_dpi(pageinfo, options): - "Get the DPI when we require xres == yres, scaled to physical units" - xres = pageinfo.xres or 0 - yres = pageinfo.yres or 0 - userunit = pageinfo.userunit or 1 - return float( - max( - (xres * userunit) or VECTOR_PAGE_DPI, - (yres * userunit) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - options.oversample or 0, - ) - ) - - -def get_canvas_square_dpi(pageinfo, options): - """Get the DPI when we require xres == yres, in Postscript units""" - return float( - max( - (pageinfo.xres) or VECTOR_PAGE_DPI, - (pageinfo.yres) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - options.oversample or 0, - ) - ) - - -def is_ocr_required(page_context): - pageinfo = page_context.pageinfo - options = page_context.options - log = page_context.log - - ocr_required = True - - if pageinfo.has_text: - if not options.force_ocr and not (options.skip_text or options.redo_ocr): - log.error("page already has text! - aborting (use --force-ocr to force OCR)") - ocr_required = False - elif options.force_ocr: - log.info("page already has text! - rasterizing text and running OCR anyway") - ocr_required = True - elif options.redo_ocr: - if pageinfo.has_corrupt_text: - log.warn( - "some text on this page cannot be mapped to characters: " - "consider using --force-ocr instead", - ) - else: - log.info("redoing OCR") - ocr_required = True - elif options.skip_text: - log.info("skipping all processing on this page") - ocr_required = False - elif not pageinfo.images and not options.lossless_reconstruction: - # We found a page with no images and no text. That means it may - # have vector art that the user wants to OCR. If we determined - # lossless reconstruction is not possible then we have to rasterize - # the image. So if OCR is being forced, take that to mean YES, go - # ahead and rasterize. If not forced, then pretend there's no text - # on the page at all so we don't lose anything. - # This could be made smarter by explicitly searching for vector art. - if options.force_ocr and options.oversample: - # The user really wants to reprocess this file - log.info( - "page has no images - " - f"rasterizing at {options.oversample} DPI because " - "--force-ocr --oversample was specified" - ) - elif options.force_ocr: - # Warn the user they might not want to do this - log.warn( - "page has no images - " - "all vector content will be " - f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely " - "increasing file size. Use --oversample to adjust the " - "DPI." - ) - else: - log.info( - "page has no images - " - "skipping all processing on this page to avoid losing detail. " - "Use --force-ocr if you wish to perform OCR on pages that " - "have vector content." - ) - ocr_required = False - - if ocr_required and options.skip_big and pageinfo.images: - pixel_count = pageinfo.width_pixels * pageinfo.height_pixels - if pixel_count > (options.skip_big * 1_000_000): - ocr_required = False - log.warn( - "page too big, skipping OCR " - f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)" - ) - return ocr_required - - -def rasterize_preview(input_file, page_context): - output_file = page_context.get_path('rasterize_preview.jpg') - canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) - page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - ghostscript.rasterize_pdf( - input_file, - output_file, - xres=canvas_dpi, - yres=canvas_dpi, - raster_device='jpeggray', - log=page_context.log, - page_dpi=(page_dpi, page_dpi), - pageno=page_context.pageinfo.pageno + 1, - ) - return output_file - - -def get_orientation_correction(preview, page_context): - """ - Work out orientation correct for each page. - - We ask Ghostscript to draw a preview page, which will rasterize with the - current /Rotate applied, and then ask Tesseract which way the page is - oriented. If the value of /Rotate is correct (e.g., a user already - manually fixed rotation), then Tesseract will say the page is pointing - up and the correction is zero. Otherwise, the orientation found by - Tesseract represents the clockwise rotation, or the counterclockwise - correction to rotation. - - When we draw the real page for OCR, we rotate it by the CCW correction, - which points it (hopefully) upright. _weave.py takes care of the orienting - the image and text layers. - - """ - - orient_conf = tesseract.get_orientation( - preview, - engine_mode=page_context.options.tesseract_oem, - timeout=page_context.options.tesseract_timeout, - log=page_context.log, - ) - - direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} - - existing_rotation = page_context.pageinfo.rotation - - correction = orient_conf.angle % 360 - - apply_correction = False - action = '' - if orient_conf.confidence >= page_context.options.rotate_pages_threshold: - if correction != 0: - apply_correction = True - action = ' - will rotate' - else: - action = ' - rotation appears correct' - else: - if correction != 0: - action = ' - confidence too low to rotate' - else: - action = ' - no change' - - facing = '' - if existing_rotation != 0: - facing = 'with existing rotation {}, '.format( - direction.get(existing_rotation, '?') - ) - facing += 'page is facing {}'.format(direction.get(orient_conf.angle, '?')) - - page_context.log.debug( - '{pagenum:4d}: {facing}, confidence {conf:.2f}{action}'.format( - pagenum=page_context.pageinfo.pageno, - facing=facing, - conf=orient_conf.confidence, - action=action, - ) - ) - - if apply_correction: - return correction - return 0 - - -def rasterize(input_file, page_context, correction=0): - colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m'] - device_idx = 0 - output_file = page_context.get_path('rasterize.png') - pageinfo = page_context.pageinfo - - def at_least(cs): - return max(device_idx, colorspaces.index(cs)) - - for image in pageinfo.images: - if image.type_ != 'image': - continue # ignore masks - if image.bpc > 1: - if image.color == Colorspace.index: - device_idx = at_least('png256') - elif image.color == Colorspace.gray: - device_idx = at_least('pnggray') - else: - device_idx = at_least('png16m') - - device = colorspaces[device_idx] - - page_context.log.debug(f"Rasterize with {device}") - - # Produce the page image with square resolution or else deskew and OCR - # will not work properly. - canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options) - page_dpi = get_page_square_dpi(pageinfo, page_context.options) - - ghostscript.rasterize_pdf( - input_file, - output_file, - xres=canvas_dpi, - yres=canvas_dpi, - raster_device=device, - log=page_context.log, - page_dpi=(page_dpi, page_dpi), - pageno=pageinfo.pageno + 1, - rotation=correction, - filter_vector=page_context.options.remove_vectors, - ) - return output_file - - -def preprocess_remove_background(input_file, page_context): - if any(image.bpc > 1 for image in page_context.pageinfo.images): - output_file = page_context.get_path('pp_rm_bg.png') - leptonica.remove_background(input_file, output_file) - return output_file - else: - page_context.log.info("background removal skipped on mono page") - return input_file - - -def preprocess_deskew(input_file, page_context): - output_file = page_context.get_path('pp_deskew.png') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - leptonica.deskew(input_file, output_file, dpi) - return output_file - - -def preprocess_clean(input_file, page_context): - from .exec import unpaper - output_file = page_context.get_path('pp_clean.png') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - unpaper.clean(input_file, output_file, dpi, page_context.log, page_context.options.unpaper_args) - return output_file - - -def create_ocr_image(image, page_context): - """Create the image we send for OCR. May not be the same as the display - image depending on preprocessing. This image will never be shown to the - user.""" - - output_file = page_context.get_path('ocr.png') - options = page_context.options - with Image.open(image) as im: - from PIL import ImageColor - from PIL import ImageDraw - - white = ImageColor.getcolor('#ffffff', im.mode) - # pink = ImageColor.getcolor('#ff0080', im.mode) - draw = ImageDraw.ImageDraw(im) - - xres, yres = im.info['dpi'] - page_context.log.info('resolution %r %r' % (xres, yres)) - - if not options.force_ocr: - # Do not mask text areas when forcing OCR, because we need to OCR - # all text areas - mask = None # Exclude both visible and invisible text from OCR - if options.redo_ocr: - mask = True # Mask visible text, but not invisible text - - for textarea in page_context.pageinfo.get_textareas(visible=mask, corrupt=None): - # Calculate resolution based on the image size and page dimensions - # without regard whatever resolution is in pageinfo (may differ or - # be None) - bbox = [float(v) for v in textarea] - xscale, yscale = float(xres) / 72.0, float(yres) / 72.0 - pixcoords = [ - bbox[0] * xscale, - im.height - bbox[3] * yscale, - bbox[2] * xscale, - im.height - bbox[1] * yscale, - ] - pixcoords = [int(round(c)) for c in pixcoords] - print('blanking %r', pixcoords) - draw.rectangle(pixcoords, fill=white) - # draw.rectangle(pixcoords, outline=pink) - - if options.mask_barcodes or options.threshold: - pix = leptonica.Pix.frompil(im) - if options.threshold: - pix = pix.masked_threshold_on_background_norm() - if options.mask_barcodes: - barcodes = pix.locate_barcodes() - for barcode in barcodes: - decoded, rect = barcode - print('masking barcode %s %r', decoded, rect) - draw.rectangle(rect, fill=white) - im = pix.topil() - - del draw - # Pillow requires integer DPI - dpi = round(xres), round(yres) - im.save(output_file, dpi=dpi) - return output_file - - -def ocr_tesseract_hocr(input_file, page_context): - hocr_out = page_context.get_path('ocr_hocr.hocr') - hocr_text_out = page_context.get_path('ocr_hocr.txt') - options = page_context.options - tesseract.generate_hocr( - input_file=input_file, - output_files=[hocr_out, hocr_text_out], - language=options.language, - engine_mode=options.tesseract_oem, - tessconfig=options.tesseract_config, - timeout=options.tesseract_timeout, - pagesegmode=options.tesseract_pagesegmode, - user_words=options.user_words, - user_patterns=options.user_patterns, - log=page_context.log, - ) - return (hocr_out, hocr_text_out) - - -def should_visible_page_image_use_jpg(pageinfo): - # If all images were JPEGs originally, produce a JPEG as output - return pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images) - - -def create_visible_page_jpg(image, page_context): - output_file = page_context.get_path('visible.jpg') - with Image.open(image) as im: - # At this point the image should be a .png, but deskew, unpaper - # might have removed the DPI information. In this case, fall back to - # square DPI used to rasterize. When the preview image was - # rasterized, it was also converted to square resolution, which is - # what we want to give tesseract, so keep it square. - fallback_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi)) - - # Pillow requires integer DPI - dpi = round(dpi[0]), round(dpi[1]) - im.save(output_file, format='JPEG', dpi=dpi) - return output_file - - -def create_pdf_page_from_image(image, page_context): - # We rasterize a square DPI version of each page because most image - # processing tools don't support rectangular DPI. Use the square DPI as it - # accurately describes the image. It would be possible to resample the image - # at this stage back to non-square DPI to more closely resemble the input, - # except that the hocr renderer does not understand non-square DPI. The - # sandwich renderer would be fine. - output_file = page_context.get_path('visible.pdf') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) - - # This create a single page PDF - with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: - page_context.log.debug('convert') - img2pdf.convert( - imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf - ) - page_context.log.debug('convert done') - return output_file - - -def render_hocr_page(hocr, page_context): - output_file = page_context.get_path('ocr_hocr.pdf') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - hocrtransform = HocrTransform(hocr, dpi) - hocrtransform.to_pdf( - output_file, - imageFileName=None, - showBoundingboxes=False, - invisibleText=True, - interwordSpaces=True, - ) - return output_file - - -def ocr_tesseract_textonly_pdf(input_image, page_context): - output_pdf = page_context.get_path('ocr_tess.pdf') - output_text = page_context.get_path('ocr_tess.txt') - options = page_context.options - tesseract.generate_pdf( - input_image=input_image, - skip_pdf=None, - output_pdf=output_pdf, - output_text=output_text, - language=options.language, - engine_mode=options.tesseract_oem, - text_only=True, - tessconfig=options.tesseract_config, - timeout=options.tesseract_timeout, - pagesegmode=options.tesseract_pagesegmode, - user_words=options.user_words, - user_patterns=options.user_patterns, - log=page_context.log, - ) - return (output_pdf, output_text) - - -def get_docinfo(base_pdf, options): - def from_document_info(key): - try: - s = base_pdf.docinfo[key] - return str(s) - except (KeyError, TypeError): - return '' - - pdfmark = { - k: from_document_info(k) - for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate') - } - if options.title: - pdfmark['/Title'] = options.title - if options.author: - pdfmark['/Author'] = options.author - if options.keywords: - pdfmark['/Keywords'] = options.keywords - if options.subject: - pdfmark['/Subject'] = options.subject - - if options.pdf_renderer == 'sandwich': - renderer_tag = 'OCR-PDF' - else: - renderer_tag = 'OCR' - - pdfmark['/Creator'] = ( - f'{PROGRAM_NAME} {VERSION} / ' f'Tesseract {renderer_tag} {tesseract.version()}' - ) - pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}' - if 'OCRMYPDF_CREATOR' in os.environ: - pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR'] - if 'OCRMYPDF_PRODUCER' in os.environ: - pdfmark['/Producer'] = os.environ['OCRMYPDF_PRODUCER'] - - pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc)) - return pdfmark - - -def generate_postscript_stub(context): - output_file = context.get_path('pdfa.ps') - generate_pdfa_ps(output_file) - return output_file - - -def convert_to_pdfa(input_pdf, input_ps_stub, context): - options = context.options - input_pdfinfo = context.pdfinfo - output_file = context.get_path('pdfa.pdf') - - # If the DocumentInfo record contains NUL characters, Ghostscript will - # produce XMP metadata which contains invalid XML entities (�). - # NULs in DocumentInfo seem to be common since older Acrobats included them. - # pikepdf can deal with this, but we make the world a better place by - # stamping them out as soon as possible. - pdf_file = pikepdf.open(input_pdf) - if pdf_file.docinfo: - modified = False - for k, v in pdf_file.docinfo.items(): - if b'\x00' in bytes(v): - pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') - modified = True - if modified: - pdf_file.save(input_pdf) - del pdf_file - - ghostscript.generate_pdfa( - pdf_version=input_pdfinfo.min_version, - pdf_pages=[input_pdf, input_ps_stub], - output_file=output_file, - compression=options.pdfa_image_compression, - log=context.log, - threads=options.jobs or 1, - pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 - ) - - return output_file - - -def metadata_fixup(working_file, context): - output_file = context.get_path('metafix.pdf') - options = context.options - original = pikepdf.open(context.origin) - docinfo = get_docinfo(original, options) - pdf = pikepdf.open(working_file) - with pdf.open_metadata() as meta: - meta.load_from_docinfo(docinfo, delete_missing=False) - # If xmp:CreateDate is missing, set it to the modify date to - # match Ghostscript, for consistency - if 'xmp:CreateDate' not in meta: - meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') - - meta_original = original.open_metadata() - not_copied = set(meta_original.keys()) - set(meta.keys()) - if not_copied: - context.log.warning( - "Some input metadata could not be copied because it is not " - "permitted in PDF/A. You may wish to examine the output " - "PDF's XMP metadata." - ) - context.log.debug( - "The following metadata fields were not copied: %r", not_copied - ) - - pdf.save( - output_file, - compress_streams=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, - ) - return output_file - - -def optimize_pdf(input_file, context): - output_file = context.get_path('optimize.pdf') - optimize(input_file, output_file, context) - return output_file - - -def merge_sidecars(txt_files, context): - output_file = context.get_path('sidecar.txt') - with open(output_file, 'w', encoding="utf-8") as stream: - for page_num, txt_file in enumerate(txt_files): - if page_num != 0: - stream.write('\f') # Form feed between pages - if txt_file: - with open(txt_file, 'r', encoding="utf-8") as in_: - txt = in_.read() - # Tesseract v4 alpha started adding form feeds in - # commit aa6eb6b - # No obvious way to detect what binaries will do this, so - # for consistency just ignore its form feeds and insert our - # own - if txt.endswith('\f'): - stream.write(txt[:-1]) - else: - stream.write(txt) - else: - stream.write(f'[OCR skipped on page {(page_num + 1)}]') - return output_file - - -def copy_final(input_file, output_file, context): - context.log.debug('%s -> %s', input_file, output_file) - with open(input_file, 'rb') as input_stream: - if output_file == '-': - copyfileobj(input_stream, sys.stdout.buffer) - sys.stdout.flush() - else: - # At this point we overwrite the output_file specified by the user - # use copyfileobj because then we use open() to create the file and - # get the appropriate umask, ownership, etc. - with open(output_file, 'wb') as output_stream: - copyfileobj(input_stream, output_stream) diff --git a/src/ocrmypdf/_ruffus.py b/src/ocrmypdf/_ruffus.py deleted file mode 100644 index 4cc69043..00000000 --- a/src/ocrmypdf/_ruffus.py +++ /dev/null @@ -1,508 +0,0 @@ -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -import os -import re -import sys -import atexit -from tempfile import mkdtemp -from ruffus import ( - Pipeline, - formatter, - regex, - suffix, - cmdline, - proxy_logger, - ruffus_exceptions -) -from .exec import qpdf -from ._jobcontext import JobContext, JobContextManager, cleanup_working_files -from ._weave import weave_layers -from ._pipeline import ( - triage, - repair_and_parse_pdf, - marker_pages, - ocr_or_skip, - rasterize_preview, - orient_page, - rasterize_with_ghostscript, - preprocess_remove_background, - preprocess_deskew, - preprocess_clean, - select_ocr_image, - ocr_tesseract_hocr, - select_visible_page_image, - select_image_layer, - render_hocr_page, - ocr_tesseract_textonly_pdf, - generate_postscript_stub, - convert_to_pdfa, - metadata_fixup, - merge_sidecars, - optimize_pdf, - copy_final -) -from . import exceptions as ocrmypdf_exceptions -from .exceptions import ( - ExitCode, - ExitCodeException, -) -from .helpers import available_cpu_count -from .pdfa import file_claims_pdfa -from ._validation import ( - check_closed_streams, - preamble, - check_options, - check_dependency_versions, - check_environ, - check_input_file, - check_requested_output_file, - report_output_file_size, - log_page_orientations, - logging_factory, -) - - -def cleanup_ruffus_error_message(msg): - msg = re.sub(r'\s+', r' ', msg) - msg = re.sub(r"\((.+?)\)", r'\1', msg) - msg = msg.strip() - return msg - - -def do_ruffus_exception(ruffus_five_tuple, options, log): - """Replace the elaborate ruffus stack trace with a user friendly - description of the error message that occurred.""" - exit_code = None - - _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple - - if isinstance(exc_name, type): - # ruffus is full of mystery... sometimes (probably when the process - # group leader is killed) exc_name is the class object of the exception, - # rather than a str. So reach into the object and get its name. - exc_name = exc_name.__name__ - - if exc_name.startswith('ocrmypdf.exceptions.'): - base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') - exc_class = getattr(ocrmypdf_exceptions, base_exc_name) - exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) - try: - if isinstance(exc_value, exc_class): - exc_msg = str(exc_value) - elif isinstance(exc_value, str): - exc_msg = exc_value - else: - exc_msg = str(exc_class()) - except Exception: - exc_msg = "Unknown" - - if exc_name in ('builtins.SystemExit', 'SystemExit'): - match = re.search(r"\.(.+?)\)", exc_value) - exit_code_name = match.groups()[0] - exit_code = getattr(ExitCode, exit_code_name, 'other_error') - elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError': - log.error(cleanup_ruffus_error_message(exc_value)) - exit_code = ExitCode.input_file - elif exc_name in ('builtins.KeyboardInterrupt', 'KeyboardInterrupt'): - # We have to print in this case because the log daemon might be toast - print("Interrupted by user", file=sys.stderr) - exit_code = ExitCode.ctrl_c - elif exc_name == 'subprocess.CalledProcessError': - # It's up to the subprocess handler to report something useful - msg = "Error occurred while running this command:" - log.error(msg + '\n' + exc_value) - exit_code = ExitCode.child_process_error - elif exc_name.startswith('ocrmypdf.exceptions.'): - if exc_msg: - log.error(exc_msg) - elif exc_name == 'PIL.Image.DecompressionBombError': - msg = cleanup_ruffus_error_message(exc_value) - msg += ( - "\nUse the --max-image-mpixels argument to set increase the " - "maximum number of megapixels to accept." - ) - log.error(msg) - exit_code = ExitCode.input_file - - if exit_code is not None: - return exit_code - - if not options.verbose: - log.error(exc_stack) - return ExitCode.other_error - - -def traverse_ruffus_exception(exceptions, options, log): - """Traverse a RethrownJobError and output the exceptions - - Ruffus presents exceptions as 5 element tuples. The RethrownJobException - has a list of exceptions like - e.job_exceptions = [(5-tuple), (5-tuple), ...] - - ruffus < 2.7.0 had a bug with exception marshalling that would give - different output whether the main or child process raised the exception. - We no longer support this. - - Attempting to log the exception itself will re-marshall it to the logger - which is normally running in another process. It's better to avoid re- - marshalling. - - The exit code will be based on this, even if multiple exceptions occurred - at the same time.""" - - exit_codes = [] - for exc in exceptions: - exit_code = do_ruffus_exception(exc, options, log) - exit_codes.append(exit_code) - - return exit_codes[0] # Multiple codes are rare so take the first one - - -def build_pipeline(options, work_folder, log, context): - main_pipeline = Pipeline.pipelines['main'] - - # Triage - task_triage = main_pipeline.transform( - task_func=triage, - input=os.path.join(work_folder, 'origin'), - filter=formatter('(?i)'), - output=os.path.join(work_folder, 'origin.pdf'), - extras=[log, context], - ) - - task_repair_and_parse_pdf = main_pipeline.transform( - task_func=repair_and_parse_pdf, - input=task_triage, - filter=suffix('.pdf'), - output='.repaired.pdf', - output_dir=work_folder, - extras=[log, context], - ) - - # Split (kwargs for split seems to be broken, so pass plain args) - task_marker_pages = main_pipeline.split( - marker_pages, - task_repair_and_parse_pdf, - os.path.join(work_folder, '*.marker.pdf'), - extras=[log, context], - ) - - task_ocr_or_skip = main_pipeline.split( - ocr_or_skip, - task_marker_pages, - [ - os.path.join(work_folder, '*.ocr.page.pdf'), - os.path.join(work_folder, '*.skip.page.pdf'), - ], - extras=[log, context], - ) - - # Rasterize preview - task_rasterize_preview = main_pipeline.transform( - task_func=rasterize_preview, - input=task_ocr_or_skip, - filter=suffix('.page.pdf'), - output='.preview.jpg', - output_dir=work_folder, - extras=[log, context], - ) - task_rasterize_preview.active_if(options.rotate_pages) - - # Orient - task_orient_page = main_pipeline.collate( - task_func=orient_page, - input=[task_ocr_or_skip, task_rasterize_preview], - filter=regex(r".*/(\d{6})(\.ocr|\.skip)(?:\.page\.pdf|\.preview\.jpg)"), - output=os.path.join(work_folder, r'\1\2.oriented.pdf'), - extras=[log, context], - ) - - # Rasterize actual - task_rasterize_with_ghostscript = main_pipeline.transform( - task_func=rasterize_with_ghostscript, - input=task_orient_page, - filter=suffix('.ocr.oriented.pdf'), - output='.page.png', - output_dir=work_folder, - extras=[log, context], - ) - - # Preprocessing subpipeline - task_preprocess_remove_background = main_pipeline.transform( - task_func=preprocess_remove_background, - input=task_rasterize_with_ghostscript, - filter=suffix(".page.png"), - output=".pp-background.png", - extras=[log, context], - ) - - task_preprocess_deskew = main_pipeline.transform( - task_func=preprocess_deskew, - input=task_preprocess_remove_background, - filter=suffix(".pp-background.png"), - output=".pp-deskew.png", - extras=[log, context], - ) - - task_preprocess_clean = main_pipeline.transform( - task_func=preprocess_clean, - input=task_preprocess_deskew, - filter=suffix(".pp-deskew.png"), - output=".pp-clean.png", - extras=[log, context], - ) - - task_select_ocr_image = main_pipeline.collate( - task_func=select_ocr_image, - input=[task_preprocess_clean], - filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), - output=os.path.join(work_folder, r"\1.ocr.png"), - extras=[log, context], - ) - - # HOCR OCR - task_ocr_tesseract_hocr = main_pipeline.transform( - task_func=ocr_tesseract_hocr, - input=task_select_ocr_image, - filter=suffix(".ocr.png"), - output=[".hocr", ".txt"], - extras=[log, context], - ) - task_ocr_tesseract_hocr.graphviz(fillcolor='"#00cc66"') - task_ocr_tesseract_hocr.active_if(options.pdf_renderer == 'hocr') - - task_select_visible_page_image = main_pipeline.collate( - task_func=select_visible_page_image, - input=[ - task_rasterize_with_ghostscript, - task_preprocess_remove_background, - task_preprocess_deskew, - task_preprocess_clean, - ], - filter=regex(r".*/(\d{6})(?:\.page|\.pp-.*)\.png"), - output=os.path.join(work_folder, r'\1.image'), - extras=[log, context], - ) - task_select_visible_page_image.graphviz(shape='diamond') - - task_select_image_layer = main_pipeline.collate( - task_func=select_image_layer, - input=[task_select_visible_page_image, task_orient_page], - filter=regex(r".*/(\d{6})(?:\.image|\.ocr\.oriented\.pdf)"), - output=os.path.join(work_folder, r'\1.image-layer.pdf'), - extras=[log, context], - ) - task_select_image_layer.graphviz(fillcolor='"#00cc66"', shape='diamond') - - task_render_hocr_page = main_pipeline.transform( - task_func=render_hocr_page, - input=task_ocr_tesseract_hocr, - filter=regex(r".*/(\d{6})(?:\.hocr)"), - output=os.path.join(work_folder, r'\1.text.pdf'), - extras=[log, context], - ) - task_render_hocr_page.graphviz(fillcolor='"#00cc66"') - task_render_hocr_page.active_if(options.pdf_renderer == 'hocr') - - # Tesseract OCR + text only PDF - task_ocr_tesseract_textonly_pdf = main_pipeline.collate( - task_func=ocr_tesseract_textonly_pdf, - input=[task_select_ocr_image], - filter=regex(r".*/(\d{6})(?:\.ocr.png)"), - output=[ - os.path.join(work_folder, r'\1.text.pdf'), - os.path.join(work_folder, r'\1.text.txt'), - ], - extras=[log, context], - ) - task_ocr_tesseract_textonly_pdf.graphviz(fillcolor='"#ff69b4"') - task_ocr_tesseract_textonly_pdf.active_if(options.pdf_renderer == 'sandwich') - - task_weave_layers = main_pipeline.collate( - task_func=weave_layers, - input=[ - task_repair_and_parse_pdf, - task_render_hocr_page, - task_ocr_tesseract_textonly_pdf, - task_select_image_layer, - ], - filter=regex( - r".*/((?:\d{6}(?:\.text\.pdf|\.image-layer\.pdf))|(?:origin\.repaired\.pdf))" - ), - output=os.path.join(work_folder, r'layers.rendered.pdf'), - extras=[log, context], - ) - task_weave_layers.graphviz(fillcolor='"#00cc66"') - - # PDF/A pdfmark - task_generate_postscript_stub = main_pipeline.transform( - task_func=generate_postscript_stub, - input=task_repair_and_parse_pdf, - filter=formatter(r'\.repaired\.pdf'), - output=os.path.join(work_folder, 'pdfa.ps'), - extras=[log, context], - ) - task_generate_postscript_stub.active_if(options.output_type.startswith('pdfa')) - - # PDF/A conversion - task_convert_to_pdfa = main_pipeline.merge( - task_func=convert_to_pdfa, - input=[task_generate_postscript_stub, task_weave_layers], - output=os.path.join(work_folder, 'pdfa.pdf'), - extras=[log, context], - ) - task_convert_to_pdfa.active_if(options.output_type.startswith('pdfa')) - - task_metadata_fixup = main_pipeline.merge( - task_func=metadata_fixup, - input=[task_repair_and_parse_pdf, task_weave_layers, task_convert_to_pdfa], - output=os.path.join(work_folder, 'metafix.pdf'), - extras=[log, context], - ) - - task_merge_sidecars = main_pipeline.merge( - task_func=merge_sidecars, - input=[task_ocr_tesseract_hocr, task_ocr_tesseract_textonly_pdf], - output=options.sidecar, - extras=[log, context], - ) - task_merge_sidecars.active_if(options.sidecar) - - # Optimize - task_optimize_pdf = main_pipeline.transform( - task_func=optimize_pdf, - input=task_metadata_fixup, - filter=suffix('.pdf'), - output='.optimized.pdf', - output_dir=work_folder, - extras=[log, context], - ) - - # Finalize - main_pipeline.merge( - task_func=copy_final, - input=[task_optimize_pdf], - output=options.output_file, - extras=[log, context], - ) - - -def run_pipeline(options): - options.verbose_abbreviated_path = 1 - if os.environ.get('_OCRMYPDF_THREADS'): - options.use_threads = True - - if not check_closed_streams(options): - return ExitCode.bad_args - - logger_args = {'verbose': options.verbose, 'quiet': options.quiet} - - _log, _log_mutex = proxy_logger.make_shared_logger_and_proxy( - logging_factory, __name__, logger_args - ) - preamble(_log) - check_code = check_options(options, _log) - if check_code != ExitCode.ok: - return check_code - check_dependency_versions(options, _log) - - # Any changes to options will not take effect for options that are already - # bound to function parameters in the pipeline. (For example - # options.input_file, options.pdf_renderer are already bound.) - if not options.jobs: - options.jobs = available_cpu_count() - - # Performance is improved by setting Tesseract to single threaded. In tests - # this gives better throughput than letting a smaller number of Tesseract - # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this - # variable, but harmless to set if ignored. - os.environ.setdefault('OMP_THREAD_LIMIT', '1') - - check_environ(options, _log) - if os.environ.get('PYTEST_CURRENT_TEST'): - os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file - - try: - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite') - start_input_file = os.path.join(work_folder, 'origin') - - check_input_file(options, _log, start_input_file) - check_requested_output_file(options, _log) - - manager = JobContextManager() - manager.register('JobContext', JobContext) # pylint: disable=no-member - manager.start() - - context = manager.JobContext() # pylint: disable=no-member - context.set_options(options) - context.set_work_folder(work_folder) - - build_pipeline(options, work_folder, _log, context) - atexit.register(cleanup_working_files, work_folder, options) - if hasattr(os, 'nice'): - os.nice(5) - cmdline.run(options) - except ruffus_exceptions.RethrownJobError as e: - if options.verbose: - _log.debug(str(e)) # stringify exception so logger doesn't have to - exceptions = e.job_exceptions - exitcode = traverse_ruffus_exception(exceptions, options, _log) - if exitcode is None: - _log.error("Unexpected ruffus exception: " + str(e)) - _log.error(repr(e)) - return ExitCode.other_error - return exitcode - except ExitCodeException as e: - return e.exit_code - except Exception as e: - _log.error(str(e)) - return ExitCode.other_error - - if options.flowchart: - _log.info(f"Flowchart saved to {options.flowchart}") - return ExitCode.ok - elif options.output_file == '-': - _log.info("Output sent to stdout") - elif os.path.samefile(options.output_file, os.devnull): - pass # Say nothing when sending to dev null - else: - if options.output_type.startswith('pdfa'): - pdfa_info = file_claims_pdfa(options.output_file) - if pdfa_info['pass']: - msg = f"Output file is a {pdfa_info['conformance']} (as expected)" - _log.info(msg) - else: - msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" - _log.warning(msg) - return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file, _log): - _log.warning('Output file: The generated PDF is INVALID') - return ExitCode.invalid_output_pdf - - report_output_file_size(options, _log, start_input_file, options.output_file) - - pdfinfo = context.get_pdfinfo() - if options.verbose: - from pprint import pformat - - _log.debug(pformat(pdfinfo)) - - log_page_orientations(pdfinfo, _log) - - return ExitCode.ok diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index ec339bf5..4e246544 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -16,13 +16,11 @@ # along with OCRmyPDF. If not, see . import os -# import re -# import sys import atexit from tempfile import mkdtemp -from ._jobcontext import cleanup_working_files +from ._jobcontext import PDFContext, get_logger, cleanup_working_files from ._weave import weave_layers -from ._pipeline_simple import ( +from ._pipeline import ( get_pdfinfo, validate_pdfinfo_options, is_ocr_required, @@ -48,6 +46,7 @@ from ._pipeline_simple import ( ) from .exceptions import ( ExitCode, + ExitCodeException, ) from .helpers import available_cpu_count from ._validation import ( @@ -61,52 +60,6 @@ from ._validation import ( ) -class Logger: - def __init__(self, prefix): - self.prefix = prefix - - def debug(self, *argv): - print(self.prefix, *argv) - - def info(self, *argv): - print(self.prefix, *argv) - - def warn(self, *argv): - print(self.prefix, *argv) - - def error(self, *argv): - print(self.prefix, *argv) - - -class PageContext: - def __init__(self, pdf_context, pageno): - self.pdf_context = pdf_context - self.options = pdf_context.options - self.pageno = pageno - self.pageinfo = pdf_context.pdfinfo[pageno] - self.log = Logger('%s Page %d: ' % (os.path.basename(pdf_context.origin), pageno + 1)) - - def get_path(self, name): - return os.path.join(self.pdf_context.work_folder, "page_%d_%s" % (self.pageno, name)) - - -class PDFContext: - def __init__(self, options, work_folder, origin, pdfinfo): - self.options = options - self.work_folder = work_folder - self.origin = origin - self.pdfinfo = pdfinfo - self.log = Logger('%s: ' % os.path.basename(origin)) - - def get_path(self, name): - return os.path.join(self.work_folder, name) - - def get_page_contexts(self): - npages = len(self.pdfinfo) - for n in range(npages): - yield PageContext(self, n) - - def _exec_pipeline(options, work_folder, origin): # Gather info of pdf pdfinfo = get_pdfinfo(origin) @@ -181,7 +134,7 @@ def run_pipeline(options): if not check_closed_streams(options): return ExitCode.bad_args - log = Logger('Pipeline') + log = get_logger(options, 'Pipeline') preamble(log) check_code = check_options(options, log) if check_code != ExitCode.ok: @@ -213,7 +166,13 @@ def run_pipeline(options): if hasattr(os, 'nice'): os.nice(5) - _exec_pipeline(options, work_folder, start_input_file) + try: + _exec_pipeline(options, work_folder, start_input_file) + except ExitCodeException as e: + return e.exit_code + except Exception as e: + log.error(str(e)) + return ExitCode.other_error return ExitCode.ok diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 000f24d8..9cf1e69b 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -16,7 +16,6 @@ # along with OCRmyPDF. If not, see . import concurrent.futures -import logging import sys from collections import defaultdict from os import fspath @@ -28,7 +27,7 @@ import pikepdf from pikepdf import Name, Dictionary from . import leptonica -from ._jobcontext import JobContext +from ._jobcontext import PDFContext from .exec import jbig2enc, pngquant from .helpers import re_symlink @@ -82,9 +81,7 @@ def extract_image_jbig2(*, pike, root, log, image, xref, options): pim, filtdp = result if ( - pim.bits_per_component == 1 - and filtdp != Name.JBIG2Decode - and jbig2enc.available() + pim.bits_per_component == 1 and filtdp != Name.JBIG2Decode and jbig2enc.available() ): try: imgname = Path(root / f'{xref:08d}') @@ -129,9 +126,7 @@ def extract_image_generic(*, pike, root, log, image, xref, options): return None return xref, ext elif ( - pim.indexed - and pim.colorspace in pim.SIMPLE_COLORSPACES - and options.optimize >= 3 + pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES and options.optimize >= 3 ): # Try to improve on indexed images - these are far from low hanging # fruit in most cases @@ -492,10 +487,6 @@ def main(infile, outfile, level, jobs=1): self.jbig2_page_group_size = 0 self.jbig2_lossy = jb2lossy - logging.basicConfig(level=logging.DEBUG) - log = logging.getLogger() - - ctx = JobContext() options = OptimizeOptions( jobs=jobs, optimize=int(level), @@ -503,11 +494,11 @@ def main(infile, outfile, level, jobs=1): png_quality=0, jb2lossy=False, ) - ctx.set_options(options) with TemporaryDirectory() as td: + context = PDFContext(options, td, infile, None) tmpout = Path(td) / 'out.pdf' - optimize(infile, tmpout, log, ctx) + optimize(infile, tmpout, context) copy(fspath(tmpout), fspath(outfile)) diff --git a/tests/test_multiprocessing.py b/tests/_test_multiprocessing.py similarity index 100% rename from tests/test_multiprocessing.py rename to tests/_test_multiprocessing.py diff --git a/tests/test_main.py b/tests/test_main.py index 248a9003..49d93025 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -28,7 +28,7 @@ import pytest from PIL import Image from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import ghostscript, qpdf, tesseract, unpaper +from ocrmypdf.exec import ghostscript, qpdf, tesseract from ocrmypdf.leptonica import Pix from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo diff --git a/tests/test_metadata.py b/tests/test_metadata.py index be1d952c..a4d60b55 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -28,7 +28,7 @@ from unittest.mock import MagicMock, patch import pytest import pikepdf -from ocrmypdf._jobcontext import JobContext +from ocrmypdf._jobcontext import PDFContext from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps from pikepdf.models.metadata import decode_pdf_date @@ -330,23 +330,15 @@ def test_prevent_gs_invalid_xml(resources, outdir): from ocrmypdf.pdfinfo import PdfInfo generate_pdfa_ps(outdir / 'pdfa.ps') - input_files = [str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps')] copyfile(resources / 'enron1.pdf', outdir / 'layers.rendered.pdf') - log = logging.getLogger() - context = JobContext() options = parser.parse_args( args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) - context.options = options - context.pdfinfo = PdfInfo(resources / 'enron1.pdf') + pdfinfo = PdfInfo(resources / 'enron1.pdf') + context = PDFContext(options, outdir, resources / 'enron1.pdf', pdfinfo) - convert_to_pdfa( - input_files_groups=input_files, - output_file=outdir / 'pdfa.pdf', - log=log, - context=context, - ) + convert_to_pdfa(str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context) with open(outdir / 'pdfa.pdf', 'rb') as f: with mmap.mmap( From 2647382cf6ac5d8c0dc843db3f67c908811bd221 Mon Sep 17 00:00:00 2001 From: mawi Date: Fri, 5 Apr 2019 14:06:07 +0200 Subject: [PATCH 007/880] fix: most of the tests (37 failed, 133 passed, 28 skipped) --- src/ocrmypdf/_jobcontext.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 6b059644..2b56e03c 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -20,6 +20,11 @@ import sys import os from contextlib import suppress +ERROR = 40 +WARN = 30 +INFO = 20 +DEBUG = 10 + class PDFContext: """Holds our context for a particular run of the pipeline""" @@ -51,7 +56,7 @@ class PageContext: self.log = get_logger(pdf_context.options, '%s Page %d: ' % (os.path.basename(pdf_context.origin), pageno + 1)) def get_path(self, name): - return os.path.join(self.pdf_context.work_folder, "page_%d_%s" % (self.pageno, name)) + return os.path.join(self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name)) def cleanup_working_files(work_folder, options): @@ -63,20 +68,14 @@ def cleanup_working_files(work_folder, options): def get_logger(options=None, prefix=''): - level = INFO # TODO: add option + level = ERROR # TODO: add option if options is not None and options.output_file == '-' or options.sidecar == '-': return NullLogger() return Logger(prefix, level) -ERROR = 40 -WARN = 30 -INFO = 20 -DEBUG = 10 - - class Logger: - def __init__(self, prefix, level=INFO): + def __init__(self, prefix, level=DEBUG): self.prefix = prefix self.level = level From fc1c4f12f55e0e1e2f6175b62b1067c28292358b Mon Sep 17 00:00:00 2001 From: mawi Date: Fri, 5 Apr 2019 18:48:34 +0200 Subject: [PATCH 008/880] feat: add concurrent.futures pipeline --- src/ocrmypdf/_sync.py | 132 +++++++++++++++++++++++++----------------- 1 file changed, 78 insertions(+), 54 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 4e246544..a63228f3 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -17,6 +17,7 @@ import os import atexit +import concurrent.futures from tempfile import mkdtemp from ._jobcontext import PDFContext, get_logger, cleanup_working_files from ._weave import weave_layers @@ -57,32 +58,25 @@ from ._validation import ( check_environ, check_requested_output_file, create_input_file, + report_output_file_size, ) +from .pdfa import file_claims_pdfa +from .exec import qpdf -def _exec_pipeline(options, work_folder, origin): - # Gather info of pdf - pdfinfo = get_pdfinfo(origin) - context = PDFContext(options, work_folder, origin, pdfinfo) - - # Validate options are okey for this pdf - validate_pdfinfo_options(context) - - # For every page in the pdf - layers = [] - for page_context in context.get_page_contexts(): - # Check if OCR is required - ocr_required = is_ocr_required(page_context) - if not ocr_required: - continue - - orientation_correction = 0 +def exec_page_sync(page_context): + options = page_context.options + orientation_correction = 0 + pdf_page_from_image_out = None + ocr_out = None + text_out = None + if is_ocr_required(page_context): if options.rotate_pages: # Rasterize - rasterize_preview_out = rasterize_preview(origin, page_context) + rasterize_preview_out = rasterize_preview(page_context.pdf_context.origin, page_context) orientation_correction = get_orientation_correction(rasterize_preview_out, page_context) - rasterize_out = rasterize(origin, page_context, correction=orientation_correction) + rasterize_out = rasterize(page_context.pdf_context.origin, page_context, correction=orientation_correction) preprocess_out = rasterize_out if options.remove_background: @@ -110,24 +104,70 @@ def _exec_pipeline(options, work_folder, origin): if options.pdf_renderer == 'sandwich': (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) - layers.append((page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction)) + return (page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction) - weave_layers_out = weave_layers(layers, context) - pdf_out = weave_layers_out - if options.output_type.startswith('pdfa'): +def post_process(pdf_file, context): + pdf_out = pdf_file + if context.options.output_type.startswith('pdfa'): ps_stub_out = generate_postscript_stub(context) pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context) pdf_out = metadata_fixup(pdf_out, context) + return optimize_pdf(pdf_out, context) - if options.sidecar: + +def exec_sync(context): + """Execute the pipeline single threaded""" + + # TODO: triage + + # Run exec_page_sync on every page context + layers = map(exec_page_sync, context.get_page_contexts()) + + # Output sidecar text + if context.options.sidecar: sidecars = [layer[3] for layer in layers] - sidecar_out = merge_sidecars(sidecars, context) - copy_final(sidecar_out, context.options.sidecar, context) + text = merge_sidecars(sidecars, context) + # Copy final text file to destination + copy_final(text, context.options.sidecar, context) - pdf_out = optimize_pdf(pdf_out, context) - copy_final(pdf_out, context.options.output_file, context) + # Merge layers to one single pdf + pdf = weave_layers(layers, context) + + # PDF/A and metadata + pdf = post_process(pdf, context) + + # Copy final PDF file to destination + copy_final(pdf, context.options.output_file, context) + + +def exec_concurrent(context): + """Execute the pipeline concurrent""" + + # TODO: triage + + # Run exec_page_sync on every page context + max_workers = min(len(context.pdfinfo), context.options.jobs) + context.log.info("Start processing %d pages concurrent" % max_workers) + with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: + layers = executor.map(exec_page_sync, context.get_page_contexts()) + + # Output sidecar text + if context.options.sidecar: + sidecars = [layer[3] for layer in layers] + text = merge_sidecars(sidecars, context) + # Copy text file to destination + copy_final(text, context.options.sidecar, context) + + # Merge layers to one single pdf + pdf = weave_layers(layers, context) + + # PDF/A and metadata + pdf = post_process(pdf, context) + + # Copy PDF file to destination + copy_final(pdf, context.options.output_file, context) def run_pipeline(options): @@ -167,30 +207,22 @@ def run_pipeline(options): os.nice(5) try: - _exec_pipeline(options, work_folder, start_input_file) + # Gather pdfinfo and create context + pdfinfo = get_pdfinfo(start_input_file) + context = PDFContext(options, work_folder, start_input_file, pdfinfo) + + # Validate options are okey for this pdf + validate_pdfinfo_options(context) + + # Execute the pipeline + exec_concurrent(context) except ExitCodeException as e: return e.exit_code except Exception as e: log.error(str(e)) return ExitCode.other_error - return ExitCode.ok - - -""" - try: - # build_pipeline(options, work_folder, log, context) - atexit.register(cleanup_working_files, work_folder, options) - if hasattr(os, 'nice'): - os.nice(5) - except Exception as e: - log.error(str(e)) - return ExitCode.other_error - - if options.flowchart: - log.info(f"Flowchart saved to {options.flowchart}") - return ExitCode.ok - elif options.output_file == '-': + if options.output_file == '-': log.info("Output sent to stdout") elif os.path.samefile(options.output_file, os.devnull): pass # Say nothing when sending to dev null @@ -210,12 +242,4 @@ def run_pipeline(options): report_output_file_size(options, log, start_input_file, options.output_file) - # pdfinfo = context.get_pdfinfo() - # if options.verbose: - # from pprint import pformat - # log.debug(pformat(pdfinfo)) - - # log_page_orientations(pdfinfo, log) - return ExitCode.ok -""" From 01bbf064e0070583f8bbf5f1a58f229f848a2e08 Mon Sep 17 00:00:00 2001 From: mawi Date: Fri, 5 Apr 2019 19:52:38 +0200 Subject: [PATCH 009/880] feat: add tqdm progress bar This is just a POC. Will be removed. --- src/ocrmypdf/_jobcontext.py | 10 +++++++++- src/ocrmypdf/_sync.py | 23 ++++++++++++++++------- 2 files changed, 25 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 2b56e03c..d3301616 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -29,12 +29,13 @@ DEBUG = 10 class PDFContext: """Holds our context for a particular run of the pipeline""" - def __init__(self, options, work_folder, origin, pdfinfo): + def __init__(self, options, work_folder, origin, pdfinfo, tick=None): self.options = options self.work_folder = work_folder self.origin = origin self.pdfinfo = pdfinfo self.log = get_logger(options, '%s: ' % os.path.basename(origin)) + self.tick_callback = tick def get_path(self, name): return os.path.join(self.work_folder, name) @@ -44,6 +45,10 @@ class PDFContext: for n in range(npages): yield PageContext(self, n) + def tick(self, times=1): + if self.tick_callback: + self.tick_callback(times) + class PageContext: """Holds our context for a page""" @@ -58,6 +63,9 @@ class PageContext: def get_path(self, name): return os.path.join(self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name)) + def tick(self, times=1): + self.pdf_context.tick(times) + def cleanup_working_files(work_folder, options): if options.keep_temporary_files: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a63228f3..82936a7c 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,6 +18,7 @@ import os import atexit import concurrent.futures +from tqdm import tqdm from tempfile import mkdtemp from ._jobcontext import PDFContext, get_logger, cleanup_working_files from ._weave import weave_layers @@ -77,7 +78,7 @@ def exec_page_sync(page_context): orientation_correction = get_orientation_correction(rasterize_preview_out, page_context) rasterize_out = rasterize(page_context.pdf_context.origin, page_context, correction=orientation_correction) - + page_context.tick() preprocess_out = rasterize_out if options.remove_background: preprocess_out = preprocess_remove_background(preprocess_out, page_context) @@ -89,7 +90,7 @@ def exec_page_sync(page_context): preprocess_out = preprocess_clean(preprocess_out, page_context) ocr_image_out = create_ocr_image(preprocess_out, page_context) - + page_context.tick() pdf_page_from_image_out = None if not options.lossless_reconstruction: visible_image_out = preprocess_out @@ -103,7 +104,9 @@ def exec_page_sync(page_context): if options.pdf_renderer == 'sandwich': (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) - + page_context.tick() + else: + page_context.tick(3) return (page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction) @@ -146,13 +149,14 @@ def exec_concurrent(context): """Execute the pipeline concurrent""" # TODO: triage - + context.tick() # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) context.log.info("Start processing %d pages concurrent" % max_workers) with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: layers = executor.map(exec_page_sync, context.get_page_contexts()) + context.tick() # Output sidecar text if context.options.sidecar: sidecars = [layer[3] for layer in layers] @@ -162,10 +166,10 @@ def exec_concurrent(context): # Merge layers to one single pdf pdf = weave_layers(layers, context) - + context.tick() # PDF/A and metadata pdf = post_process(pdf, context) - + context.tick() # Copy PDF file to destination copy_final(pdf, context.options.output_file, context) @@ -209,13 +213,18 @@ def run_pipeline(options): try: # Gather pdfinfo and create context pdfinfo = get_pdfinfo(start_input_file) - context = PDFContext(options, work_folder, start_input_file, pdfinfo) + steps = 5 + len(pdfinfo) * 3 + t = tqdm(total=steps, bar_format='{l_bar}{bar}{n_fmt}/{total_fmt}') + + context = PDFContext(options, work_folder, start_input_file, pdfinfo, tick=lambda n: t.update(n)) # Validate options are okey for this pdf validate_pdfinfo_options(context) # Execute the pipeline exec_concurrent(context) + t.update() + t.close() except ExitCodeException as e: return e.exit_code except Exception as e: From 659087575658a89d3d915f7ef30ec92d7659517e Mon Sep 17 00:00:00 2001 From: mawi Date: Mon, 8 Apr 2019 10:26:56 +0200 Subject: [PATCH 010/880] feat: add triage step remove tqdm demo --- src/ocrmypdf/__main__.py | 6 ++-- src/ocrmypdf/_jobcontext.py | 26 +++++++-------- src/ocrmypdf/_pipeline.py | 23 +++++--------- src/ocrmypdf/_sync.py | 63 ++++++++++--------------------------- src/ocrmypdf/_validation.py | 10 ++---- 5 files changed, 40 insertions(+), 88 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index afa3cb20..0fd3c63b 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -179,10 +179,8 @@ jobcontrol.add_argument( jobcontrol.add_argument( '-v', '--verbose', - const="+", - default=[], - nargs='?', - action="append", + default=0, + action="count", help="Print more verbose messages for each additional verbose level", ) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index d3301616..f18ac1ac 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -29,13 +29,15 @@ DEBUG = 10 class PDFContext: """Holds our context for a particular run of the pipeline""" - def __init__(self, options, work_folder, origin, pdfinfo, tick=None): + def __init__(self, options, work_folder, origin, pdfinfo): self.options = options self.work_folder = work_folder self.origin = origin self.pdfinfo = pdfinfo - self.log = get_logger(options, '%s: ' % os.path.basename(origin)) - self.tick_callback = tick + self.name = os.path.basename(options.input_file) + if self.name == '-': + self.name = 'stdin' + self.log = get_logger(options, '%s: ' % self.name) def get_path(self, name): return os.path.join(self.work_folder, name) @@ -45,10 +47,6 @@ class PDFContext: for n in range(npages): yield PageContext(self, n) - def tick(self, times=1): - if self.tick_callback: - self.tick_callback(times) - class PageContext: """Holds our context for a page""" @@ -58,14 +56,11 @@ class PageContext: self.options = pdf_context.options self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] - self.log = get_logger(pdf_context.options, '%s Page %d: ' % (os.path.basename(pdf_context.origin), pageno + 1)) + self.log = get_logger(pdf_context.options, '%s Page %d: ' % (pdf_context.name, pageno + 1)) def get_path(self, name): return os.path.join(self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name)) - def tick(self, times=1): - self.pdf_context.tick(times) - def cleanup_working_files(work_folder, options): if options.keep_temporary_files: @@ -77,8 +72,13 @@ def cleanup_working_files(work_folder, options): def get_logger(options=None, prefix=''): level = ERROR # TODO: add option - if options is not None and options.output_file == '-' or options.sidecar == '-': - return NullLogger() + if options is not None: + if options.quiet or options.output_file == '-' or options.sidecar == '-': + return NullLogger() + if options.verbose > 0: + level = INFO + if options.verbose > 1: + level = DEBUG return Logger(prefix, level) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 3b553a7e..89e337db 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -46,21 +46,12 @@ from .pdfinfo import Colorspace, PdfInfo VECTOR_PAGE_DPI = 400 -def triage_image_file(input_file, output_file, log, options): +def triage_image_file(input_file, output_file, options, log): try: log.info("Input file is not a PDF, checking if it is an image...") im = Image.open(input_file) except EnvironmentError as e: - msg = str(e) - - # Recover the original filename - realpath = '' - if os.path.islink(input_file): - realpath = os.path.realpath(input_file) - elif os.path.isfile(input_file): - realpath = '' - msg = msg.replace(input_file, realpath) - log.error(msg) + log.error(str(e)) raise UnsupportedImageFormatError() from e else: log.info("Input file is an image") @@ -135,9 +126,7 @@ def _pdf_guess_version(input_file, search_window=1024): return '' -def triage(input_file, output_file, log, context): - - options = context.get_options() +def triage(input_file, output_file, options, log): try: if _pdf_guess_version(input_file): if options.image_dpi: @@ -145,13 +134,15 @@ def triage(input_file, output_file, log, context): "Argument --image-dpi ignored because the " "input file is a PDF, not an image." ) + # Origin file is a pdf create a symlink with pdf extension re_symlink(input_file, output_file, log) - return + return output_file except EnvironmentError as e: log.error(e) raise InputFileError() from e - triage_image_file(input_file, output_file, log, options) + triage_image_file(input_file, output_file, options, log) + return output_file def get_pdfinfo(input_file, detailed_page_analysis=False): diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 82936a7c..6f1672be 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,11 +18,11 @@ import os import atexit import concurrent.futures -from tqdm import tqdm from tempfile import mkdtemp from ._jobcontext import PDFContext, get_logger, cleanup_working_files from ._weave import weave_layers from ._pipeline import ( + triage, get_pdfinfo, validate_pdfinfo_options, is_ocr_required, @@ -50,10 +50,10 @@ from .exceptions import ( ExitCode, ExitCodeException, ) +from . import VERSION from .helpers import available_cpu_count from ._validation import ( check_closed_streams, - preamble, check_options, check_dependency_versions, check_environ, @@ -78,7 +78,7 @@ def exec_page_sync(page_context): orientation_correction = get_orientation_correction(rasterize_preview_out, page_context) rasterize_out = rasterize(page_context.pdf_context.origin, page_context, correction=orientation_correction) - page_context.tick() + preprocess_out = rasterize_out if options.remove_background: preprocess_out = preprocess_remove_background(preprocess_out, page_context) @@ -90,7 +90,7 @@ def exec_page_sync(page_context): preprocess_out = preprocess_clean(preprocess_out, page_context) ocr_image_out = create_ocr_image(preprocess_out, page_context) - page_context.tick() + pdf_page_from_image_out = None if not options.lossless_reconstruction: visible_image_out = preprocess_out @@ -104,9 +104,7 @@ def exec_page_sync(page_context): if options.pdf_renderer == 'sandwich': (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) - page_context.tick() - else: - page_context.tick(3) + return (page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction) @@ -120,43 +118,16 @@ def post_process(pdf_file, context): return optimize_pdf(pdf_out, context) -def exec_sync(context): - """Execute the pipeline single threaded""" - - # TODO: triage - - # Run exec_page_sync on every page context - layers = map(exec_page_sync, context.get_page_contexts()) - - # Output sidecar text - if context.options.sidecar: - sidecars = [layer[3] for layer in layers] - text = merge_sidecars(sidecars, context) - # Copy final text file to destination - copy_final(text, context.options.sidecar, context) - - # Merge layers to one single pdf - pdf = weave_layers(layers, context) - - # PDF/A and metadata - pdf = post_process(pdf, context) - - # Copy final PDF file to destination - copy_final(pdf, context.options.output_file, context) - - def exec_concurrent(context): """Execute the pipeline concurrent""" - # TODO: triage - context.tick() # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) - context.log.info("Start processing %d pages concurrent" % max_workers) + if max_workers > 1: + context.log.info("Start processing %d pages concurrent" % max_workers) with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: layers = executor.map(exec_page_sync, context.get_page_contexts()) - context.tick() # Output sidecar text if context.options.sidecar: sidecars = [layer[3] for layer in layers] @@ -166,10 +137,10 @@ def exec_concurrent(context): # Merge layers to one single pdf pdf = weave_layers(layers, context) - context.tick() + # PDF/A and metadata pdf = post_process(pdf, context) - context.tick() + # Copy PDF file to destination copy_final(pdf, context.options.output_file, context) @@ -178,8 +149,8 @@ def run_pipeline(options): if not check_closed_streams(options): return ExitCode.bad_args - log = get_logger(options, 'Pipeline') - preamble(log) + log = get_logger(options, 'Setup: ') + log.debug('ocrmypdf ' + VERSION) check_code = check_options(options, log) if check_code != ExitCode.ok: return check_code @@ -211,20 +182,18 @@ def run_pipeline(options): os.nice(5) try: - # Gather pdfinfo and create context - pdfinfo = get_pdfinfo(start_input_file) - steps = 5 + len(pdfinfo) * 3 - t = tqdm(total=steps, bar_format='{l_bar}{bar}{n_fmt}/{total_fmt}') + # Triage image or pdf + origin_pdf = triage(start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log) - context = PDFContext(options, work_folder, start_input_file, pdfinfo, tick=lambda n: t.update(n)) + # Gather pdfinfo and create context + pdfinfo = get_pdfinfo(origin_pdf) + context = PDFContext(options, work_folder, origin_pdf, pdfinfo) # Validate options are okey for this pdf validate_pdfinfo_options(context) # Execute the pipeline exec_concurrent(context) - t.update() - t.close() except ExitCodeException as e: return e.exit_code except Exception as e: diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 0d93585c..10671272 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -26,8 +26,6 @@ from pathlib import Path import PIL -from . import VERSION - from ._unicodefun import verify_python3_env from .exec import ( @@ -361,10 +359,6 @@ def log_page_orientations(pdfinfo, _log): _log.info('Page orientations detected: ' + ' '.join(orientations)) -def preamble(_log): - _log.debug('ocrmypdf ' + VERSION) - - def check_environ(options, _log): old_envvars = ( 'OCRMYPDF_TESSERACT', @@ -387,14 +381,14 @@ def create_input_file(options, log, work_folder): if options.input_file == '-': # stdin log.info('reading file from standard input') - target = os.path.join(work_folder, 'stdin.pdf') + target = os.path.join(work_folder, 'stdin') with open(target, 'wb') as stream_buffer: from shutil import copyfileobj copyfileobj(sys.stdin.buffer, stream_buffer) return target else: try: - target = os.path.join(work_folder, os.path.basename(options.input_file)) + target = os.path.join(work_folder, 'origin') re_symlink(options.input_file, target, log) return target except FileNotFoundError: From 39617dd739891665549446c584b302881e2fde28 Mon Sep 17 00:00:00 2001 From: mawi Date: Mon, 8 Apr 2019 11:07:32 +0200 Subject: [PATCH 011/880] fix: remove ruffus --- requirements/main.txt | 1 - setup.py | 1 - 2 files changed, 2 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index b2ebc357..e2429787 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -10,4 +10,3 @@ Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 reportlab == 3.5.13 -ruffus == 2.8.1 diff --git a/setup.py b/setup.py index 74aa1039..fcccc173 100644 --- a/setup.py +++ b/setup.py @@ -104,7 +104,6 @@ setup( # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels 'reportlab >= 3.3.0', # oldest released version with sane image handling - 'ruffus >= 2.7.0', ], extras_require={'pdfminer': ['pdfminer.six == 20181108']}, tests_require=tests_require, From 1137534e97fee09b81f67c8bfcca4094e3e032fc Mon Sep 17 00:00:00 2001 From: mawi Date: Mon, 8 Apr 2019 11:08:29 +0200 Subject: [PATCH 012/880] fix: update pytest version Solves install error: pkg_resources.ContextualVersionConflict: (pytest 4.3.0 (/app/.eggs/pytest-4.3.0-py3.6.egg), Requirement.parse('pytest>=4.4.0'), {'pytest-xdist'}) --- requirements/test.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/test.txt b/requirements/test.txt index 90919da7..7071ff9e 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,4 +1,4 @@ -pytest == 4.3.0 +pytest == 4.4.0 pytest-helpers-namespace >= 2019.1.8 pytest-xdist pytest-cov >= 2.6.1 From c92ccc6134738edc3992ab011ae702a6b9f766ca Mon Sep 17 00:00:00 2001 From: mawi Date: Mon, 8 Apr 2019 14:57:42 +0200 Subject: [PATCH 013/880] fix: tests --- src/ocrmypdf/__main__.py | 15 +++++++++++-- src/ocrmypdf/_jobcontext.py | 18 +++++++++++----- src/ocrmypdf/_pipeline.py | 30 ++++++++++++++------------ src/ocrmypdf/_sync.py | 13 ++++++++---- src/ocrmypdf/exec/tesseract.py | 8 +++---- src/ocrmypdf/optimize.py | 5 ++++- tests/test_main.py | 15 +++++++------ tests/test_metadata.py | 39 +++++++++++++++------------------- tests/test_rotation.py | 4 ++-- 9 files changed, 86 insertions(+), 61 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 0fd3c63b..447ae477 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -179,9 +179,20 @@ jobcontrol.add_argument( jobcontrol.add_argument( '-v', '--verbose', + type=int, default=0, - action="count", - help="Print more verbose messages for each additional verbose level", + nargs='?', + metavar='LEVEL', + choices=range(0, 4), + help=( + "Print more verbose messages for each additional verbose level. Use " + "`-v 1` typically for much more detailed logging. Higher numbers " + "are probably only useful in debugging. " + "0 - Only errors (default); " + "1 - Error and warngings; " + "2 - Info, errors and warngings; " + "3 - All messages including debug messages" + ), ) metadata = parser.add_argument_group( diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index f18ac1ac..4ee576d2 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -34,7 +34,10 @@ class PDFContext: self.work_folder = work_folder self.origin = origin self.pdfinfo = pdfinfo - self.name = os.path.basename(options.input_file) + if options: + self.name = os.path.basename(options.input_file) + else: + self.name = 'origin.pdf' if self.name == '-': self.name = 'stdin' self.log = get_logger(options, '%s: ' % self.name) @@ -75,10 +78,15 @@ def get_logger(options=None, prefix=''): if options is not None: if options.quiet or options.output_file == '-' or options.sidecar == '-': return NullLogger() - if options.verbose > 0: + if options.verbose == 0: + level = ERROR + elif options.verbose == 1: + level = WARN + elif options.verbose == 2: level = INFO - if options.verbose > 1: + elif options.verbose >= 3: level = DEBUG + return Logger(prefix, level) @@ -107,8 +115,8 @@ class Logger: def error(self, *args, **kwargs): if self.level <= ERROR: - print('ERROR', self.prefix, end='') - print(*args, **kwargs) + print('ERROR', self.prefix, end='', file=sys.stderr) + print(*args, file=sys.stderr) class NullLogger: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 89e337db..aea5642c 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -33,6 +33,7 @@ from .exceptions import ( EncryptedPdfError, InputFileError, UnsupportedImageFormatError, + PriorOcrFoundError, ) from .exec import ghostscript, tesseract from .helpers import ( @@ -51,7 +52,8 @@ def triage_image_file(input_file, output_file, options, log): log.info("Input file is not a PDF, checking if it is an image...") im = Image.open(input_file) except EnvironmentError as e: - log.error(str(e)) + # Recover the original filename + log.error(str(e).replace(input_file, options.input_file)) raise UnsupportedImageFormatError() from e else: log.info("Input file is an image") @@ -249,7 +251,7 @@ def is_ocr_required(page_context): if pageinfo.has_text: if not options.force_ocr and not (options.skip_text or options.redo_ocr): log.error("page already has text! - aborting (use --force-ocr to force OCR)") - ocr_required = False + raise PriorOcrFoundError() elif options.force_ocr: log.info("page already has text! - rasterizing text and running OCR anyway") ocr_required = True @@ -632,19 +634,19 @@ def get_docinfo(base_pdf, options): k: from_document_info(k) for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate') } - if options.title: - pdfmark['/Title'] = options.title - if options.author: - pdfmark['/Author'] = options.author - if options.keywords: - pdfmark['/Keywords'] = options.keywords - if options.subject: - pdfmark['/Subject'] = options.subject + renderer_tag = 'OCR' + if options is not None: + if options.title: + pdfmark['/Title'] = options.title + if options.author: + pdfmark['/Author'] = options.author + if options.keywords: + pdfmark['/Keywords'] = options.keywords + if options.subject: + pdfmark['/Subject'] = options.subject - if options.pdf_renderer == 'sandwich': - renderer_tag = 'OCR-PDF' - else: - renderer_tag = 'OCR' + if options.pdf_renderer == 'sandwich': + renderer_tag = 'OCR-PDF' pdfmark['/Creator'] = ( f'{PROGRAM_NAME} {VERSION} / ' f'Tesseract {renderer_tag} {tesseract.version()}' diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 6f1672be..3daf4255 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -149,6 +149,10 @@ def run_pipeline(options): if not check_closed_streams(options): return ExitCode.bad_args + # Default to INFO level + if options.verbose is None: + options.verbose = 2 + log = get_logger(options, 'Setup: ') log.debug('ocrmypdf ' + VERSION) check_code = check_options(options, log) @@ -174,14 +178,14 @@ def run_pipeline(options): work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - start_input_file = create_input_file(options, log, work_folder) - check_requested_output_file(options, log) - atexit.register(cleanup_working_files, work_folder, options) if hasattr(os, 'nice'): os.nice(5) try: + check_requested_output_file(options, log) + start_input_file = create_input_file(options, log, work_folder) + # Triage image or pdf origin_pdf = triage(start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log) @@ -195,9 +199,10 @@ def run_pipeline(options): # Execute the pipeline exec_concurrent(context) except ExitCodeException as e: + log.error("%s: %s" % (type(e).__name__, str(e))) return e.exit_code except Exception as e: - log.error(str(e)) + log.error("%s: %s" % (type(e).__name__, str(e))) return ExitCode.other_error if options.output_file == '-': diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 6e9b29e0..84a728af 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -33,7 +33,7 @@ from subprocess import ( from textwrap import dedent from . import get_version -from ..exceptions import MissingDependencyError, TesseractConfigError +from ..exceptions import MissingDependencyError, TesseractConfigError, SubprocessOutputError from ..helpers import page_number OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) @@ -142,7 +142,7 @@ def get_orientation(input_file, engine_mode, timeout: float, log): or b'Image too large' in e.output ): return OrientationConfidence(0, 0) - raise e from e + raise SubprocessOutputError() from e else: osd = {} for line in stdout.decode().splitlines(): @@ -267,7 +267,7 @@ def generate_hocr( _generate_null_hocr(output_hocr, output_sidecar, input_file) return - raise e from e + raise SubprocessOutputError() from e else: tesseract_log_output(log, stdout, input_file) # The sidecar text file will get the suffix .txt; rename it to @@ -356,6 +356,6 @@ def generate_pdf( if b'Image too large' in e.output: use_skip_page(text_only, skip_pdf, output_pdf, output_text) return - raise e from e + raise SubprocessOutputError() from e else: tesseract_log_output(log, stdout, input_image) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 9cf1e69b..b722228e 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -479,15 +479,18 @@ def main(infile, outfile, level, jobs=1): class OptimizeOptions: """Emulate ocrmypdf's options""" - def __init__(self, jobs, optimize, jpeg_quality, png_quality, jb2lossy): + def __init__(self, input_file, jobs, optimize, jpeg_quality, png_quality, jb2lossy): + self.input_file = input_file self.jobs = jobs self.optimize = optimize self.jpeg_quality = jpeg_quality self.png_quality = png_quality self.jbig2_page_group_size = 0 self.jbig2_lossy = jb2lossy + self.quiet = True options = OptimizeOptions( + input_file=infile, jobs=jobs, optimize=int(level), jpeg_quality=0, # Use default diff --git a/tests/test_main.py b/tests/test_main.py index 49d93025..9461503f 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -165,8 +165,8 @@ def test_exotic_image( resources / pdf, outfile, '-dc' if pytest.helpers.have_unpaper() else '-d', - '-v', - '1', + # '-v', + # '1', '--output-type', output_type, '--sidecar', @@ -409,8 +409,8 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): outpdf, '--tesseract-pagesegmode', '7', - '-v', - '1', + # '-v', + # '1', '--pdf-renderer', renderer, env=spoof_tesseract_cache, @@ -422,8 +422,8 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): p, out, err = run_ocrmypdf( resources / 'ccitt.pdf', no_outpdf, - '-v', - '1', + # '-v', + # '1', '--pdf-renderer', renderer, env=spoof_tesseract_crash, @@ -1014,7 +1014,8 @@ def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): assert out == '', "stdout not clean" assert p.returncode != 0 assert 'not utf-8' in err, "should whine about utf-8" - assert '\\x96' in err, 'should repeat backslash encoded output' + # TODO: find out why this should be tested + # assert '\\x96' in err, 'should repeat backslash encoded output' @pytest.mark.skipif( diff --git a/tests/test_metadata.py b/tests/test_metadata.py index a4d60b55..3ec07fa5 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -18,7 +18,6 @@ import datetime from datetime import timezone -import logging import mmap from os import fspath from pathlib import Path @@ -286,41 +285,37 @@ def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): def test_metadata_fixup_warning(resources, outdir): + from ocrmypdf.__main__ import parser from ocrmypdf._pipeline import metadata_fixup - input_files = [ - str(outdir / 'graph.repaired.pdf'), - str(outdir / 'layers.rendered.pdf'), - str(outdir / 'pdfa.pdf'), # It is okay that this is not a PDF/A - ] - for f in input_files: - copyfile(resources / 'graph.pdf', f) + options = parser.parse_args( + args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf'] + ) - log = MagicMock() - context = MagicMock() + copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') + + context = PDFContext(options, outdir, outdir / 'graph.pdf', None) + context.log = MagicMock() metadata_fixup( - input_files_groups=input_files, - output_file=outdir / 'out.pdf', - log=log, + working_file=outdir / 'graph.pdf', context=context, ) - log.warning.assert_not_called() + context.log.warn.assert_not_called() + context.log.error.assert_not_called() # Now add some metadata that will not be copyable - graph = pikepdf.open(outdir / 'graph.repaired.pdf') + graph = pikepdf.open(outdir / 'graph.pdf') with graph.open_metadata() as meta: meta['prism2:publicationName'] = 'OCRmyPDF Test' - graph.save(outdir / 'graph.repaired.pdf') + graph.save(outdir / 'graph_mod.pdf') - log = MagicMock() - context = MagicMock() + context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None) + context.log = MagicMock() metadata_fixup( - input_files_groups=input_files, - output_file=outdir / 'out.pdf', - log=log, + working_file=outdir / 'graph.pdf', context=context, ) - log.warning.assert_called_once() + context.log.warn.assert_called_once() def test_prevent_gs_invalid_xml(resources, outdir): diff --git a/tests/test_rotation.py b/tests/test_rotation.py index bfcf3dae..3f09d43d 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -140,8 +140,8 @@ def test_autorotate_threshold( '--rotate-pages-threshold', threshold, '-r', - '-v', - '1', + # '-v', + # '1', env=spoof_tesseract_cache, ) From 1c44fd4f3b82a6e98cdf27d056823ee2d14ca3ad Mon Sep 17 00:00:00 2001 From: mawi Date: Mon, 8 Apr 2019 15:01:04 +0200 Subject: [PATCH 014/880] fix: typo --- src/ocrmypdf/__main__.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 447ae477..31b5abd1 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -186,11 +186,11 @@ jobcontrol.add_argument( choices=range(0, 4), help=( "Print more verbose messages for each additional verbose level. Use " - "`-v 1` typically for much more detailed logging. Higher numbers " + "`-v 2` typically for much more detailed logging. Higher numbers " "are probably only useful in debugging. " "0 - Only errors (default); " - "1 - Error and warngings; " - "2 - Info, errors and warngings; " + "1 - Error and warnings; " + "2 - Info, errors and warnings; " "3 - All messages including debug messages" ), ) From f4b87915df5c6e17f76a3441d9cf119aa40e37d1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Apr 2019 13:11:26 -0700 Subject: [PATCH 015/880] Fix --redo-ocr --- src/ocrmypdf/_sync.py | 45 ++++++++++++++++++++++++++++++------------- 1 file changed, 32 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 3daf4255..d7676a78 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -46,10 +46,7 @@ from ._pipeline import ( optimize_pdf, copy_final, ) -from .exceptions import ( - ExitCode, - ExitCodeException, -) +from .exceptions import ExitCode, ExitCodeException from . import VERSION from .helpers import available_cpu_count from ._validation import ( @@ -74,10 +71,18 @@ def exec_page_sync(page_context): if is_ocr_required(page_context): if options.rotate_pages: # Rasterize - rasterize_preview_out = rasterize_preview(page_context.pdf_context.origin, page_context) - orientation_correction = get_orientation_correction(rasterize_preview_out, page_context) + rasterize_preview_out = rasterize_preview( + page_context.pdf_context.origin, page_context + ) + orientation_correction = get_orientation_correction( + rasterize_preview_out, page_context + ) - rasterize_out = rasterize(page_context.pdf_context.origin, page_context, correction=orientation_correction) + rasterize_out = rasterize( + page_context.pdf_context.origin, + page_context, + correction=orientation_correction, + ) preprocess_out = rasterize_out if options.remove_background: @@ -95,17 +100,29 @@ def exec_page_sync(page_context): if not options.lossless_reconstruction: visible_image_out = preprocess_out if should_visible_page_image_use_jpg(page_context.pageinfo): - visible_image_out = create_visible_page_jpg(visible_image_out, page_context) - pdf_page_from_image_out = create_pdf_page_from_image(visible_image_out, page_context) + visible_image_out = create_visible_page_jpg( + visible_image_out, page_context + ) + pdf_page_from_image_out = create_pdf_page_from_image( + visible_image_out, page_context + ) if options.pdf_renderer == 'hocr': (hocr_out, text_out) = ocr_tesseract_hocr(ocr_image_out, page_context) ocr_out = render_hocr_page(hocr_out, page_context) if options.pdf_renderer == 'sandwich': - (ocr_out, text_out) = ocr_tesseract_textonly_pdf(ocr_image_out, page_context) + (ocr_out, text_out) = ocr_tesseract_textonly_pdf( + ocr_image_out, page_context + ) - return (page_context.pageno, pdf_page_from_image_out, ocr_out, text_out, orientation_correction) + return ( + page_context.pageno, + pdf_page_from_image_out, + ocr_out, + text_out, + orientation_correction, + ) def post_process(pdf_file, context): @@ -187,10 +204,12 @@ def run_pipeline(options): start_input_file = create_input_file(options, log, work_folder) # Triage image or pdf - origin_pdf = triage(start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log) + origin_pdf = triage( + start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log + ) # Gather pdfinfo and create context - pdfinfo = get_pdfinfo(origin_pdf) + pdfinfo = get_pdfinfo(origin_pdf, detailed_page_analysis=options.redo_ocr) context = PDFContext(options, work_folder, origin_pdf, pdfinfo) # Validate options are okey for this pdf From 07d4fff3d43f9eaab3f83e5ebb83b8fa683e1d05 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 May 2019 02:13:56 -0700 Subject: [PATCH 016/880] docs: mention FreeBSD works --- docs/installation.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/installation.rst b/docs/installation.rst index 59f9d58e..73380839 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -364,6 +364,16 @@ The command line program should now be available: ocrmypdf --help +Installing on FreeBSD +--------------------- + +FreeBSD 11.2 is known to work. Other versions likely work but have not been tested. + +In general it should work to: + +#. `Install and build pikepdf `_. +#. Install the equivalent list of dependencies for Linux. + Installing the Docker image --------------------------- From b10285d11b2002809b582a2511b6681a31f3ba46 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 May 2019 16:34:42 -0700 Subject: [PATCH 017/880] Fix warnings --- setup.cfg | 2 ++ src/ocrmypdf/_pipeline.py | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/setup.cfg b/setup.cfg index 73ea8553..b1b0a740 100644 --- a/setup.cfg +++ b/setup.cfg @@ -13,6 +13,8 @@ norecursedirs = lib .pc .git output cache resources testpaths = tests filterwarnings = ignore:.*XMLParser.*:DeprecationWarning +markers = + slow [isort] multi_line_output=3 diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 71d5870e..5de49e33 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -721,7 +721,7 @@ def metadata_fixup(working_file, context): not_copied = set(meta_original.keys()) - set(meta.keys()) if not_copied: if options.output_type.startswith('pdfa'): - context.log.warn( + context.log.warning( "Some input metadata could not be copied because it is not " "permitted in PDF/A. You may wish to examine the output " "PDF's XMP metadata." From 486f73d5d6cc2887190424d17944a7e5ba8cc091 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 May 2019 02:28:13 -0700 Subject: [PATCH 018/880] Remove custom logger --- src/ocrmypdf/_jobcontext.py | 67 +--------------------------------- src/ocrmypdf/exec/tesseract.py | 7 +++- tests/test_main.py | 3 +- 3 files changed, 8 insertions(+), 69 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 9a0a1eb2..f5ac6978 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -15,16 +15,12 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import shutil import sys import os from contextlib import suppress -ERROR = 40 -WARN = 30 -INFO = 20 -DEBUG = 10 - class PDFContext: """Holds our context for a particular run of the pipeline""" @@ -78,63 +74,4 @@ def cleanup_working_files(work_folder, options): def get_logger(options=None, prefix=''): - level = ERROR # TODO: add option - if options is not None: - if options.quiet or options.output_file == '-' or options.sidecar == '-': - return NullLogger() - if options.verbose == 0: - level = ERROR - elif options.verbose == 1: - level = WARN - elif options.verbose == 2: - level = INFO - elif options.verbose >= 3: - level = DEBUG - - return Logger(prefix, level) - - -class Logger: - def __init__(self, prefix, level=DEBUG): - self.prefix = prefix - self.level = level - - def debug(self, *args, **kwargs): - if self.level <= DEBUG: - print('DEBUG', self.prefix, end='', file=sys.stderr) - print(*args, file=sys.stderr, **kwargs) - - def info(self, *args, **kwargs): - if self.level <= INFO: - print('INFO', self.prefix, end='', file=sys.stderr) - print(*args, file=sys.stderr, **kwargs) - - def warning(self, *args, **kwargs): - self.warn(*args, **kwargs) - - def warn(self, *args, **kwargs): - if self.level <= WARN: - print('WARN', self.prefix, end='', file=sys.stderr) - print(*args, file=sys.stderr, **kwargs) - - def error(self, *args, **kwargs): - if self.level <= ERROR: - print('ERROR', self.prefix, end='', file=sys.stderr) - print(*args, file=sys.stderr) - - -class NullLogger: - def debug(self, *args, **kwargs): - pass - - def info(self, *args, **kwargs): - pass - - def warning(self, *args, **kwargs): - pass - - def warn(self, *args, **kwargs): - pass - - def error(self, *args, **kwargs): - pass + return logging.getLogger(prefix) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 84a728af..110c49f3 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -33,7 +33,11 @@ from subprocess import ( from textwrap import dedent from . import get_version -from ..exceptions import MissingDependencyError, TesseractConfigError, SubprocessOutputError +from ..exceptions import ( + MissingDependencyError, + TesseractConfigError, + SubprocessOutputError, +) from ..helpers import page_number OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) @@ -342,7 +346,6 @@ def generate_pdf( # to the number of order parameters here args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig) - try: log.debug(args_tesseract) stdout = check_output(args_tesseract, stderr=STDOUT, timeout=timeout) diff --git a/tests/test_main.py b/tests/test_main.py index 9461503f..2165fc78 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -1014,8 +1014,7 @@ def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): assert out == '', "stdout not clean" assert p.returncode != 0 assert 'not utf-8' in err, "should whine about utf-8" - # TODO: find out why this should be tested - # assert '\\x96' in err, 'should repeat backslash encoded output' + assert '\\x96' in err, 'should repeat backslash encoded output' @pytest.mark.skipif( From 4410503349910c3f5d932d7d7877a0d26f74d6c9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 May 2019 03:08:09 -0700 Subject: [PATCH 019/880] More fixes to logging and disabled tests --- src/ocrmypdf/__main__.py | 13 +++++++++---- src/ocrmypdf/_jobcontext.py | 34 ++++++++++++++++++++++++++++++---- 2 files changed, 39 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 624f69dd..544eb1f6 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -17,6 +17,7 @@ # along with OCRmyPDF. If not, see . import argparse +import logging import os import sys @@ -459,14 +460,18 @@ debugging.add_argument( action='store_true', help="Keep temporary files (helpful for debugging)", ) -# debugging.add_argument( -# '--flowchart', type=str, help="Generate the pipeline execution flowchart" -# ) def run(args=None): options = parser.parse_args(args=args) - return run_pipeline(options) + + log = logging.getLogger() + console = logging.StreamHandler(stream=sys.stderr) + console.setLevel(logging.DEBUG) + log.addHandler(console) + + result = run_pipeline(options) + return result if __name__ == '__main__': diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index f5ac6978..78465bb2 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -36,7 +36,7 @@ class PDFContext: self.name = 'origin.pdf' if self.name == '-': self.name = 'stdin' - self.log = get_logger(options, '%s: ' % self.name) + self.log = get_logger(options, filename=self.name) def get_path(self, name): return os.path.join(self.work_folder, name) @@ -56,7 +56,7 @@ class PageContext: self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] self.log = get_logger( - pdf_context.options, '%s Page %d: ' % (pdf_context.name, pageno + 1) + pdf_context.options, filename=self.pdf_context.name, page=(self.pageno + 1) ) def get_path(self, name): @@ -73,5 +73,31 @@ def cleanup_working_files(work_folder, options): shutil.rmtree(work_folder) -def get_logger(options=None, prefix=''): - return logging.getLogger(prefix) +class LogNameAdapter(logging.LoggerAdapter): + def process(self, msg, kwargs): + return '[%s] %s' % (self.extra['filename'], msg), kwargs + + +class LogNamePageAdapter(logging.LoggerAdapter): + def process(self, msg, kwargs): + return ( + '[%s:%05u] %s' % (self.extra['filename'], self.extra['page'], msg), + kwargs, + ) + + +def get_logger(options=None, prefix='ocrmypdf', filename=None, page=None): + log = logging.getLogger(prefix) + if filename and page: + adapter = LogNamePageAdapter(log, dict(filename=filename, page=page)) + elif filename: + adapter = LogNameAdapter(log, dict(filename=filename)) + else: + adapter = log + if options.quiet: + log.setLevel(logging.ERROR) + elif options.verbose >= 2: + log.setLevel(logging.DEBUG) + else: + log.setLevel(logging.INFO) + return adapter From 5e025c3382703ace0f87fb957d7a80178d4608c8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 May 2019 15:46:36 -0700 Subject: [PATCH 020/880] Reinstate log level in messages to be closer to old behavior --- src/ocrmypdf/__main__.py | 2 ++ tests/test_main.py | 12 ++++++------ 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 544eb1f6..7ac5939f 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -466,8 +466,10 @@ def run(args=None): options = parser.parse_args(args=args) log = logging.getLogger() + formatter = logging.Formatter('%(levelname)s - %(message)s') console = logging.StreamHandler(stream=sys.stderr) console.setLevel(logging.DEBUG) + console.setFormatter(formatter) log.addHandler(console) result = run_pipeline(options) diff --git a/tests/test_main.py b/tests/test_main.py index 2165fc78..49d93025 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -165,8 +165,8 @@ def test_exotic_image( resources / pdf, outfile, '-dc' if pytest.helpers.have_unpaper() else '-d', - # '-v', - # '1', + '-v', + '1', '--output-type', output_type, '--sidecar', @@ -409,8 +409,8 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): outpdf, '--tesseract-pagesegmode', '7', - # '-v', - # '1', + '-v', + '1', '--pdf-renderer', renderer, env=spoof_tesseract_cache, @@ -422,8 +422,8 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): p, out, err = run_ocrmypdf( resources / 'ccitt.pdf', no_outpdf, - # '-v', - # '1', + '-v', + '1', '--pdf-renderer', renderer, env=spoof_tesseract_crash, From 9d750828c73700dde68a652ed9c7226a0adbd46e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 May 2019 00:48:40 -0700 Subject: [PATCH 021/880] Make logging format consistent with v8.3.0 --- src/ocrmypdf/__main__.py | 2 +- src/ocrmypdf/_jobcontext.py | 6 ++++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 7ac5939f..3a154a2c 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -466,7 +466,7 @@ def run(args=None): options = parser.parse_args(args=args) log = logging.getLogger() - formatter = logging.Formatter('%(levelname)s - %(message)s') + formatter = logging.Formatter('%(levelname)7s - %(message)s') console = logging.StreamHandler(stream=sys.stderr) console.setLevel(logging.DEBUG) console.setFormatter(formatter) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 78465bb2..290a15e5 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -75,13 +75,15 @@ def cleanup_working_files(work_folder, options): class LogNameAdapter(logging.LoggerAdapter): def process(self, msg, kwargs): - return '[%s] %s' % (self.extra['filename'], msg), kwargs + # return '[%s] %s' % (self.extra['filename'], msg), kwargs + return '%s' % (msg,), kwargs class LogNamePageAdapter(logging.LoggerAdapter): def process(self, msg, kwargs): return ( - '[%s:%05u] %s' % (self.extra['filename'], self.extra['page'], msg), + #'[%s:%05u] %s' % (self.extra['filename'], self.extra['page'], msg), + '%4u: %s' % (self.extra['page'], msg), kwargs, ) From 471cdea23281b8ac106604b17a5023d7c08981e9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 May 2019 01:29:26 -0700 Subject: [PATCH 022/880] Move app specific settings a library may not want to __main__ --- src/ocrmypdf/__main__.py | 31 +++++++++++++++++++++++++++---- src/ocrmypdf/_jobcontext.py | 7 +------ src/ocrmypdf/_sync.py | 26 ++++++++++---------------- 3 files changed, 38 insertions(+), 26 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 3a154a2c..074ba51b 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -22,7 +22,9 @@ import os import sys from . import PROGRAM_NAME, VERSION +from .exceptions import ExitCode from ._sync import run_pipeline +from ._validation import check_closed_streams # ------------- # Parser @@ -35,7 +37,7 @@ def numeric(basetype, min_=None, max_=None): def _numeric(string): value = basetype(string) - if min_ is not None and value < min_ or max_ is not None and value > max_: + if (min_ is not None and value < min_) or (max_ is not None and value > max_): msg = "%r not in valid range %r" % (string, (min_, max_)) raise argparse.ArgumentTypeError(msg) return value @@ -462,16 +464,37 @@ debugging.add_argument( ) -def run(args=None): - options = parser.parse_args(args=args) +def setup_app_logging(options): + """Set up logging""" log = logging.getLogger() formatter = logging.Formatter('%(levelname)7s - %(message)s') console = logging.StreamHandler(stream=sys.stderr) - console.setLevel(logging.DEBUG) console.setFormatter(formatter) log.addHandler(console) + if options.quiet: + log.setLevel(logging.ERROR) + elif options.verbose >= 2: + log.setLevel(logging.DEBUG) + else: + log.setLevel(logging.INFO) + +def configure_app_environment(options): + """Configure the application environment + + Don't do anything here that a library user would not expect. + """ + if not check_closed_streams(options): + return ExitCode.bad_args + if hasattr(os, 'nice'): + os.nice(5) + + +def run(args=None): + options = parser.parse_args(args=args) + setup_app_logging(options) + configure_app_environment(options) result = run_pipeline(options) return result diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 290a15e5..9b4b87b0 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -96,10 +96,5 @@ def get_logger(options=None, prefix='ocrmypdf', filename=None, page=None): adapter = LogNameAdapter(log, dict(filename=filename)) else: adapter = log - if options.quiet: - log.setLevel(logging.ERROR) - elif options.verbose >= 2: - log.setLevel(logging.DEBUG) - else: - log.setLevel(logging.INFO) + adapter.setLevel(logging.DEBUG) return adapter diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index d7676a78..40534b18 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -50,7 +50,6 @@ from .exceptions import ExitCode, ExitCodeException from . import VERSION from .helpers import available_cpu_count from ._validation import ( - check_closed_streams, check_options, check_dependency_versions, check_environ, @@ -163,18 +162,12 @@ def exec_concurrent(context): def run_pipeline(options): - if not check_closed_streams(options): - return ExitCode.bad_args - - # Default to INFO level - if options.verbose is None: - options.verbose = 2 - - log = get_logger(options, 'Setup: ') + log = get_logger(options, __name__) log.debug('ocrmypdf ' + VERSION) - check_code = check_options(options, log) - if check_code != ExitCode.ok: - return check_code + + result = check_options(options, log) + if result != ExitCode.ok: + return result check_dependency_versions(options, log) # Any changes to options will not take effect for options that are already @@ -196,8 +189,6 @@ def run_pipeline(options): work_folder = mkdtemp(prefix="com.github.ocrmypdf.") atexit.register(cleanup_working_files, work_folder, options) - if hasattr(os, 'nice'): - os.nice(5) try: check_requested_output_file(options, log) @@ -212,7 +203,7 @@ def run_pipeline(options): pdfinfo = get_pdfinfo(origin_pdf, detailed_page_analysis=options.redo_ocr) context = PDFContext(options, work_folder, origin_pdf, pdfinfo) - # Validate options are okey for this pdf + # Validate options are okay for this pdf validate_pdfinfo_options(context) # Execute the pipeline @@ -235,7 +226,10 @@ def run_pipeline(options): msg = f"Output file is a {pdfa_info['conformance']} (as expected)" log.info(msg) else: - msg = f"Output file is okay but is not PDF/A (seems to be {pdfa_info['conformance']})" + msg = ( + f"Output file is okay but is not PDF/A " + f"(seems to be {pdfa_info['conformance']})" + ) log.warning(msg) return ExitCode.pdfa_conversion_failed if not qpdf.check(options.output_file, log): From 50bd129d7a4a1e04cb45964018740bce6b0b3b6e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 May 2019 01:58:48 -0700 Subject: [PATCH 023/880] logging: don't pass log object to validation --- src/ocrmypdf/_sync.py | 14 ++--- src/ocrmypdf/_validation.py | 107 +++++++++++++--------------------- src/ocrmypdf/exec/__init__.py | 22 +++---- 3 files changed, 59 insertions(+), 84 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 40534b18..90f87848 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -165,10 +165,10 @@ def run_pipeline(options): log = get_logger(options, __name__) log.debug('ocrmypdf ' + VERSION) - result = check_options(options, log) + result = check_options(options) if result != ExitCode.ok: return result - check_dependency_versions(options, log) + check_dependency_versions(options) # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example @@ -182,7 +182,7 @@ def run_pipeline(options): # variable, but harmless to set if ignored. os.environ.setdefault('OMP_THREAD_LIMIT', '1') - check_environ(options, log) + check_environ(options) if os.environ.get('PYTEST_CURRENT_TEST'): os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file @@ -191,8 +191,8 @@ def run_pipeline(options): atexit.register(cleanup_working_files, work_folder, options) try: - check_requested_output_file(options, log) - start_input_file = create_input_file(options, log, work_folder) + check_requested_output_file(options) + start_input_file = create_input_file(options, work_folder) # Triage image or pdf origin_pdf = triage( @@ -212,7 +212,7 @@ def run_pipeline(options): log.error("%s: %s" % (type(e).__name__, str(e))) return e.exit_code except Exception as e: - log.error("%s: %s" % (type(e).__name__, str(e))) + log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error if options.output_file == '-': @@ -236,6 +236,6 @@ def run_pipeline(options): log.warning('Output file: The generated PDF is INVALID') return ExitCode.invalid_output_pdf - report_output_file_size(options, log, start_input_file, options.output_file) + report_output_file_size(options, start_input_file, options.output_file) return ExitCode.ok diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 10671272..e80cd98f 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -51,6 +51,8 @@ from .exceptions import ( HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) +log = logging.getLogger(__name__) + def complain(message): print(*textwrap.wrap(message), file=sys.stderr) @@ -61,7 +63,7 @@ def complain(message): verify_python3_env() -def check_options_languages(options, _log): +def check_options_languages(options): if not options.language: options.language = ['eng'] # Enforce English hegemony @@ -80,7 +82,7 @@ def check_options_languages(options, _log): raise MissingDependencyError(msg) -def check_options_output(options, log): +def check_options_output(options): # We have these constraints to check for. # 1. Ghostscript < 9.20 mangles multibyte Unicode # 2. hocr doesn't work on non-Latin languages (so don't select it) @@ -134,27 +136,26 @@ def check_options_output(options, log): if not options.lossless_reconstruction and options.redo_ocr: raise BadArgsError( "--redo-ocr is not currently compatible with --deskew, " - "--clean-final, and --remove-background", + "--clean-final, and --remove-background" ) -def check_options_sidecar(options, log): +def check_options_sidecar(options): if options.sidecar == '\0': if options.output_file == '-': raise BadArgsError( - "--sidecar filename must be specified when output file is " "stdout.", + "--sidecar filename must be specified when output file is " "stdout." ) options.sidecar = options.output_file + '.txt' -def check_options_preprocessing(options, log): +def check_options_preprocessing(options): if options.clean_final: options.clean = True if options.unpaper_args and not options.clean: raise BadArgsError("--clean is required for --unpaper-args") if options.clean: check_external_program( - log=log, program='unpaper', package='unpaper', version_checker=unpaper.version, @@ -170,7 +171,7 @@ def check_options_preprocessing(options, log): raise BadArgsError(str(e)) -def check_options_ocr_behavior(options, log): +def check_options_ocr_behavior(options): exclusive_options = sum( [ (1 if opt else 0) @@ -183,10 +184,9 @@ def check_options_ocr_behavior(options, log): ) -def check_options_optimizing(options, log): +def check_options_optimizing(options): if options.optimize >= 2: check_external_program( - log=log, program='pngquant', package='pngquant', version_checker=pngquant.version, @@ -198,7 +198,6 @@ def check_options_optimizing(options, log): # Although we use JBIG2 for optimize=1, don't nag about it unless the # user is asking for more optimization check_external_program( - log=log, program='jbig2', package='jbig2enc', version_checker=jbig2enc.version, @@ -216,7 +215,7 @@ def check_options_optimizing(options, log): ) -def check_options_advanced(options, log): +def check_options_advanced(options): if options.pdfa_image_compression != 'auto' and options.output_type.startswith( 'pdfa' ): @@ -228,7 +227,7 @@ def check_options_advanced(options, log): log.warning('Tesseract 4.x ignores --user-words, so this has no effect') -def check_options_metadata(options, log): +def check_options_metadata(options): import unicodedata docinfo = [options.title, options.author, options.keywords, options.subject] @@ -243,23 +242,23 @@ def check_options_metadata(options, log): ) -def check_options_pillow(options, log): +def check_options_pillow(options): PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000) if PIL.Image.MAX_IMAGE_PIXELS == 0: PIL.Image.MAX_IMAGE_PIXELS = None -def check_options(options, log): +def check_options(options): try: - check_options_languages(options, log) - check_options_metadata(options, log) - check_options_output(options, log) - check_options_sidecar(options, log) - check_options_preprocessing(options, log) - check_options_ocr_behavior(options, log) - check_options_optimizing(options, log) - check_options_advanced(options, log) - check_options_pillow(options, log) + check_options_languages(options) + check_options_metadata(options) + check_options_output(options) + check_options_sidecar(options) + check_options_preprocessing(options) + check_options_ocr_behavior(options) + check_options_optimizing(options) + check_options_advanced(options) + check_options_pillow(options) return ExitCode.ok except ValueError as e: log.error(e) @@ -272,30 +271,6 @@ def check_options(options, log): return ExitCode.missing_dependency -# ---------- -# Logging - - -def logging_factory(logger_name, logger_args): - verbose = logger_args['verbose'] - quiet = logger_args['quiet'] - - root_logger = logging.getLogger(logger_name) - root_logger.setLevel(logging.DEBUG) - - handler = logging.StreamHandler(sys.stderr) - formatter_ = logging.Formatter("%(levelname)7s - %(message)s") - handler.setFormatter(formatter_) - if verbose: - handler.setLevel(logging.DEBUG) - elif quiet: - handler.setLevel(logging.WARNING) - else: - handler.setLevel(logging.INFO) - root_logger.addHandler(handler) - return root_logger - - def check_closed_streams(options): """Work around Python issue with multiprocessing forking on closed streams @@ -348,7 +323,7 @@ def check_closed_streams(options): return True -def log_page_orientations(pdfinfo, _log): +def log_page_orientations(pdfinfo): direction = {0: 'n', 90: 'e', 180: 's', 270: 'w'} orientations = [] for n, page in enumerate(pdfinfo): @@ -356,10 +331,10 @@ def log_page_orientations(pdfinfo, _log): if angle != 0: orientations.append('{0}{1}'.format(n + 1, direction.get(angle, ''))) if orientations: - _log.info('Page orientations detected: ' + ' '.join(orientations)) + log.info('Page orientations detected: ' + ' '.join(orientations)) -def check_environ(options, _log): +def check_environ(options): old_envvars = ( 'OCRMYPDF_TESSERACT', 'OCRMYPDF_QPDF', @@ -368,7 +343,7 @@ def check_environ(options, _log): ) for k in old_envvars: if k in os.environ: - _log.warning( + log.warning( textwrap.dedent( f"""\ OCRmyPDF no longer uses the environment variable {k}. @@ -377,13 +352,14 @@ def check_environ(options, _log): ) -def create_input_file(options, log, work_folder): +def create_input_file(options, work_folder): if options.input_file == '-': # stdin log.info('reading file from standard input') target = os.path.join(work_folder, 'stdin') with open(target, 'wb') as stream_buffer: from shutil import copyfileobj + copyfileobj(sys.stdin.buffer, stream_buffer) return target else: @@ -392,30 +368,30 @@ def create_input_file(options, log, work_folder): re_symlink(options.input_file, target, log) return target except FileNotFoundError: - log.error("File not found - " + options.input_file) + log.error("File not found - %s", options.input_file) raise InputFileError() -def check_input_file(options, _log, start_input_file): +def check_input_file(options, start_input_file): if options.input_file == '-': # stdin - _log.info('reading file from standard input') + log.info('reading file from standard input') with open(start_input_file, 'wb') as stream_buffer: from shutil import copyfileobj copyfileobj(sys.stdin.buffer, stream_buffer) else: try: - re_symlink(options.input_file, start_input_file, _log) + re_symlink(options.input_file, start_input_file, log) except FileNotFoundError: - _log.error("File not found - " + options.input_file) + log.error("File not found - " + options.input_file) raise InputFileError() -def check_requested_output_file(options, _log): +def check_requested_output_file(options): if options.output_file == '-': if sys.stdout.isatty(): - _log.error( + log.error( textwrap.dedent( """\ Output was set to stdout '-' but it looks like stdout @@ -425,13 +401,13 @@ def check_requested_output_file(options, _log): ) raise BadArgsError() elif not is_file_writable(options.output_file): - _log.error( + log.error( "Output file location (" + options.output_file + ") is not a writable file." ) raise OutputFileAccessError() -def report_output_file_size(options, _log, input_file, output_file): +def report_output_file_size(options, input_file, output_file): try: output_size = Path(output_file).stat().st_size input_size = Path(input_file).stat().st_size @@ -462,7 +438,7 @@ def report_output_file_size(options, _log, input_file, output_file): else: explanation = "No reason for this increase is known. Please report this issue." - _log.warning( + log.warning( textwrap.dedent( f"""\ The output file size is {ratio:.2f}× larger than the input file. @@ -472,16 +448,14 @@ def report_output_file_size(options, _log, input_file, output_file): ) -def check_dependency_versions(options, log): +def check_dependency_versions(options): check_external_program( - log=log, program='tesseract', package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'}, version_checker=tesseract.version, need_version='4.0.0', # using backport for Travis CI ) check_external_program( - log=log, program='gs', package='ghostscript', version_checker=ghostscript.version, @@ -495,7 +469,6 @@ def check_dependency_versions(options, log): ) return ExitCode.missing_dependency check_external_program( - log=log, program='qpdf', package='qpdf', version_checker=qpdf.version, diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 31d3603f..670651fb 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -17,6 +17,7 @@ """Wrappers to manage subprocess calls""" +import logging import os import re import sys @@ -24,6 +25,8 @@ from subprocess import run, STDOUT, PIPE, CalledProcessError from ..exceptions import MissingDependencyError, ExitCode from collections.abc import Mapping +log = logging.Logger(__name__) + def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)'): "Get the version of the specified program" @@ -115,7 +118,7 @@ def _get_platform(): return sys.platform -def _error_trailer(log, program, package, **kwargs): +def _error_trailer(program, package, **kwargs): if isinstance(package, Mapping): package = package[_get_platform()] @@ -125,7 +128,7 @@ def _error_trailer(log, program, package, **kwargs): log.info(linux_install_advice.format(**locals())) -def _error_missing_program(log, program, package, required_for, recommended): +def _error_missing_program(program, package, required_for, recommended): if required_for: log.error(missing_optional_program.format(**locals())) elif recommended: @@ -135,9 +138,7 @@ def _error_missing_program(log, program, package, required_for, recommended): _error_trailer(**locals()) -def _error_old_version( - log, program, package, need_version, found_version, required_for -): +def _error_old_version(program, package, need_version, found_version, required_for): if required_for: log.error(old_version_required_for.format(**locals())) else: @@ -147,26 +148,27 @@ def _error_old_version( def check_external_program( *, - log, program, package, version_checker, need_version, required_for=None, recommended=False, + **kwargs, # To consume log parameter ): + if kwargs: + if not 'log' in kwargs: + log.warning('check_external_program(log=...) is deprecated') try: found_version = version_checker() except (CalledProcessError, FileNotFoundError, MissingDependencyError): - _error_missing_program(log, program, package, required_for, recommended) + _error_missing_program(program, package, required_for, recommended) if not recommended: sys.exit(ExitCode.missing_dependency) return if found_version < need_version: - _error_old_version( - log, program, package, need_version, found_version, required_for - ) + _error_old_version(program, package, need_version, found_version, required_for) if not recommended: sys.exit(ExitCode.missing_dependency) From 19263f00c6b1f0a45ba22f3bddbb3068b20a57f1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 May 2019 13:44:44 -0700 Subject: [PATCH 024/880] Additional logging fixes; silence extremely verbose pdfminer logging --- src/ocrmypdf/__main__.py | 15 ++++++++++----- src/ocrmypdf/_jobcontext.py | 1 - 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 074ba51b..14d91cfb 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -468,16 +468,21 @@ def setup_app_logging(options): """Set up logging""" log = logging.getLogger() + log.setLevel(logging.INFO) formatter = logging.Formatter('%(levelname)7s - %(message)s') - console = logging.StreamHandler(stream=sys.stderr) - console.setFormatter(formatter) - log.addHandler(console) + console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) if options.quiet: - log.setLevel(logging.ERROR) + console.setLevel(logging.ERROR) elif options.verbose >= 2: + console.setLevel(logging.DEBUG) log.setLevel(logging.DEBUG) else: - log.setLevel(logging.INFO) + console.setLevel(logging.INFO) + console.setFormatter(formatter) + log.addHandler(console) + + pdfminer_log = logging.getLogger('pdfminer') + pdfminer_log.setLevel(logging.ERROR) def configure_app_environment(options): diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 9b4b87b0..bef45426 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -96,5 +96,4 @@ def get_logger(options=None, prefix='ocrmypdf', filename=None, page=None): adapter = LogNameAdapter(log, dict(filename=filename)) else: adapter = log - adapter.setLevel(logging.DEBUG) return adapter From 13ab23ba547dc4fdaed6aca9e8e337192cd9e3d1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 May 2019 13:49:07 -0700 Subject: [PATCH 025/880] Refactor weave_layers, introduce progress bar Fixes a bug in this branch where --sidecar would fail by trying to iterator the executor futures twice. --- requirements/main.txt | 1 + setup.py | 1 + src/ocrmypdf/__main__.py | 17 +++++ src/ocrmypdf/_sync.py | 49 +++++++++--- src/ocrmypdf/_weave.py | 157 +++++++++++++++++---------------------- tests/test_unpaper.py | 2 +- 6 files changed, 127 insertions(+), 100 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 3d2e9ad1..aeb55d15 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -10,3 +10,4 @@ Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 reportlab == 3.5.13 +tqdm == 4.32.1 diff --git a/setup.py b/setup.py index ca4afb83..9550b975 100644 --- a/setup.py +++ b/setup.py @@ -104,6 +104,7 @@ setup( # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels 'reportlab >= 3.3.0', # oldest released version with sane image handling + 'tqdm >= 4', ], extras_require={'pdfminer': ['pdfminer.six == 20181108']}, tests_require=tests_require, diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 14d91cfb..e48283c2 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,6 +21,8 @@ import logging import os import sys +from tqdm import tqdm + from . import PROGRAM_NAME, VERSION from .exceptions import ExitCode from ._sync import run_pipeline @@ -464,6 +466,21 @@ debugging.add_argument( ) +class TqdmConsole: + """Wrapper to log messages in a way that is compatible with the progress bar""" + + def __init__(self, file): + self.file = file + + def write(self, msg): + # When no progress bar is active, tqdm.write() routes to print() + tqdm.write(msg.rstrip(), file=self.file) + + def flush(self): + if hasattr(self.file, "flush"): + self.file.flush() + + def setup_app_logging(options): """Set up logging""" diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 90f87848..2322bb52 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,9 +18,10 @@ import os import atexit import concurrent.futures +from collections import namedtuple from tempfile import mkdtemp from ._jobcontext import PDFContext, get_logger, cleanup_working_files -from ._weave import weave_layers +from ._weave import OcrGrafter from ._pipeline import ( triage, get_pdfinfo, @@ -60,6 +61,12 @@ from ._validation import ( from .pdfa import file_claims_pdfa from .exec import qpdf +from tqdm import tqdm + +PageResult = namedtuple( + 'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction' +) + def exec_page_sync(page_context): options = page_context.options @@ -115,12 +122,12 @@ def exec_page_sync(page_context): ocr_image_out, page_context ) - return ( - page_context.pageno, - pdf_page_from_image_out, - ocr_out, - text_out, - orientation_correction, + return PageResult( + pageno=page_context.pageno, + pdf_page_from_image=pdf_page_from_image_out, + ocr=ocr_out, + text=text_out, + orientation_correction=orientation_correction, ) @@ -141,18 +148,36 @@ def exec_concurrent(context): max_workers = min(len(context.pdfinfo), context.options.jobs) if max_workers > 1: context.log.info("Start processing %d pages concurrent" % max_workers) - with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: - layers = executor.map(exec_page_sync, context.get_page_contexts()) + + sidecars = {} + layers = [] + ocrgraft = OcrGrafter(context) + with tqdm( + total=(2 * len(context.pdfinfo)), desc='OCR', unit='page', unit_scale=0.5 + ) as pbar, concurrent.futures.ProcessPoolExecutor( + max_workers=max_workers + ) as executor: + # layers = executor.map(exec_page_sync, context.get_page_contexts()) + futures = [ + executor.submit(exec_page_sync, ctx) for ctx in context.get_page_contexts() + ] + for future in concurrent.futures.as_completed(futures): + page_result = future.result() + sidecars[page_result.pageno] = page_result.text + pbar.update() + ocrgraft.graft_page(page_result) + pbar.update() # Output sidecar text if context.options.sidecar: - sidecars = [layer[3] for layer in layers] - text = merge_sidecars(sidecars, context) + ordered_sidecars = [sidecars[pageno] for pageno in sorted(sidecars)] + text = merge_sidecars(ordered_sidecars, context) # Copy text file to destination copy_final(text, context.options.sidecar, context) # Merge layers to one single pdf - pdf = weave_layers(layers, context) + # pdf = weave_layers(layers, context) + pdf = ocrgraft.finalize() # PDF/A and metadata pdf = post_process(pdf, context) diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index 397b89cd..da4ca640 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -185,126 +185,109 @@ def _find_font(text, pdf_base): return None, None -def weave_layers(layers, context): - """Apply text layer and/or image layer changes to baseline file +class OcrGrafter: + def __init__(self, context): + self.context = context + self.log = context.log + self.path_base = Path(context.origin).resolve() - This is where the magic happens. infiles will be the main PDF to modify, - and optional .text.pdf and .image-layer.pdf files, organized however ruffus - organizes them. + self.pdf_base = pikepdf.open(self.path_base) + self.font, self.font_key = None, None - From .text.pdf, we copy the content stream (which contains the Tesseract - OCR results), and rotate it into place. The first time we do this, we also - copy the GlyphlessFont, and then reference that font again. + self.pdfinfo = context.pdfinfo + self.output_file = context.get_path('weave_layers.pdf') - For .image-layer.pdf, we check if this is a "pointer" to the original file, - or a new file. If a new file, we replace the page and remember that we - replaced this page. + self.procset = self.pdf_base.make_indirect( + pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]') + ) - Every 100 open files, we save intermediate results, to avoid any resource - limits, since pikepdf/qpdf need to keep a lot of open file handles in the - background. When objects are copied from one file to another qpdf, qpdf - doesn't actually copy the data until asked to write, so all the resources - it may need to remain available. + self.emplacements = 1 + self.interim_count = 0 - For completeness, we set up a /ProcSet on every page, although it's - unlikely any PDF viewer cares about this anymore. - - """ - - log = context.log - - path_base = Path(context.origin).resolve() - pdf_base = pikepdf.open(path_base) - font, font_key, procset = None, None, None - - pdfinfo = context.pdfinfo - output_file = context.get_path('weave_layers.pdf') - - procset = pdf_base.make_indirect( - pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]') - ) - - emplacements = 1 - interim_count = 0 - - # Iterate rest - for (pageno, image, text, sidecar, autorotate_correction) in layers: - if text and not font: - font, font_key = _find_font(text, pdf_base) + def graft_page(self, page_result): + pageno, image, text, sidecar, autorotate_correction = page_result + if text and not self.font: + self.font, self.font_key = _find_font(text, self.pdf_base) emplaced_page = False - content_rotation = pdfinfo[pageno].rotation + content_rotation = self.pdfinfo[pageno].rotation path_image = Path(image).resolve() if image else None - if path_image is not None and path_image != path_base: + if path_image is not None and path_image != self.path_base: # We are updating the old page with a rasterized PDF of the new # page (without changing objgen, to preserve references) - log.debug("Emplacement update") + self.log.debug("Emplacement update") with pikepdf.open(image) as pdf_image: - emplacements += 1 + self.emplacements += 1 foreign_image_page = pdf_image.pages[0] - pdf_base.pages.append(foreign_image_page) - local_image_page = pdf_base.pages[-1] - pdf_base.pages[pageno].emplace(local_image_page) - del pdf_base.pages[-1] + self.pdf_base.pages.append(foreign_image_page) + local_image_page = self.pdf_base.pages[-1] + self.pdf_base.pages[pageno].emplace(local_image_page) + del self.pdf_base.pages[-1] emplaced_page = True if emplaced_page: content_rotation = autorotate_correction text_rotation = autorotate_correction text_misaligned = (text_rotation - content_rotation) % 360 - log.debug( + self.log.debug( '%r', [text_rotation, autorotate_correction, text_misaligned, content_rotation], ) - if text and font: + if text and self.font: # Graft the text layer onto this page, whether new or old - strip_old = context.options.redo_ocr + strip_old = self.context.options.redo_ocr _weave_layers_graft( - pdf_base=pdf_base, + pdf_base=self.pdf_base, page_num=pageno + 1, text=text, - font=font, - font_key=font_key, + font=self.font, + font_key=self.font_key, rotation=text_misaligned, - procset=procset, + procset=self.procset, strip_old_text=strip_old, - log=log, + log=self.log, ) # Correct the rotation if applicable - pdf_base.pages[pageno].Rotate = (content_rotation - autorotate_correction) % 360 + self.pdf_base.pages[pageno].Rotate = ( + content_rotation - autorotate_correction + ) % 360 - if emplacements % MAX_REPLACE_PAGES == 0: - # Periodically save and reload the Pdf object. This will keep a - # lid on our memory usage for very large files. Attach the font to - # page 1 even if page 1 doesn't use it, so we have a way to get it - # back. - # TODO refactor this to outside the loop - page0 = pdf_base.pages[0] - _update_page_resources( - page=page0, font=font, font_key=font_key, procset=procset - ) + if self.emplacements % MAX_REPLACE_PAGES == 0: + self.save_and_reload() - # We cannot read and write the same file, that will corrupt it - # but we don't to keep more copies than we need to. Delete intermediates. - # {interim_count} is the opened file we were updateing - # {interim_count - 1} can be deleted - # {interim_count + 1} is the new file will produce and open - old_file = output_file + f'_working{interim_count - 1}.pdf' - if not context.options.keep_temporary_files: - with suppress(FileNotFoundError): - os.unlink(old_file) + def save_and_reload(self): + # Periodically save and reload the Pdf object. This will keep a + # lid on our memory usage for very large files. Attach the font to + # page 1 even if page 1 doesn't use it, so we have a way to get it + # back. + # TODO refactor this to outside the loop + page0 = self.pdf_base.pages[0] + _update_page_resources( + page=page0, font=self.font, font_key=self.font_key, procset=self.procset + ) - next_file = output_file + f'_working{interim_count + 1}.pdf' - pdf_base.save(next_file) - pdf_base.close() + # We cannot read and write the same file, that will corrupt it + # but we don't to keep more copies than we need to. Delete intermediates. + # {interim_count} is the opened file we were updateing + # {interim_count - 1} can be deleted + # {interim_count + 1} is the new file will produce and open + old_file = self.output_file + f'_working{self.interim_count - 1}.pdf' + if not self.context.options.keep_temporary_files: + with suppress(FileNotFoundError): + os.unlink(old_file) - pdf_base = pikepdf.open(next_file) - procset = pdf_base.pages[0].Resources.ProcSet - font, font_key = None, None # Ensure we reacquire this information - interim_count += 1 + next_file = self.output_file + f'_working{self.interim_count + 1}.pdf' + self.pdf_base.save(next_file) + self.pdf_base.close() - pdf_base.save(output_file) - pdf_base.close() - return output_file + self.pdf_base = pikepdf.open(next_file) + self.procset = self.pdf_base.pages[0].Resources.ProcSet + self.font, self.font_key = None, None # Ensure we reacquire this information + self.interim_count += 1 + + def finalize(self): + self.pdf_base.save(self.output_file) + self.pdf_base.close() + return self.output_file diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 5b55e819..e3ff1b67 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -59,7 +59,7 @@ def test_no_unpaper(resources, no_outpdf): with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") with pytest.raises(SystemExit): - check_options(options, log=logging.getLogger()) + check_options(options) def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): From 0cb4e854e5db3cde270c2923bbaba4e72d167266 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 00:48:06 -0700 Subject: [PATCH 026/880] Replace ProcessPoolExecutor with multiprocessing.Pool There seems to be no reasonable way to handle Ctrl-C with a ProcessPoolExecutor. Or at least you have to press it several times to actually kill. Pool does the job. --- src/ocrmypdf/_sync.py | 92 ++++++++++++++++++++++++++++++++++--------- 1 file changed, 74 insertions(+), 18 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 2322bb52..a5cec90d 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,6 +18,12 @@ import os import atexit import concurrent.futures +import logging +import logging.handlers +import multiprocessing +import threading +import sys +import signal from collections import namedtuple from tempfile import mkdtemp from ._jobcontext import PDFContext, get_logger, cleanup_working_files @@ -141,37 +147,84 @@ def post_process(pdf_file, context): return optimize_pdf(pdf_out, context) +def worker_init(queue): + """Initialize a process pool worker""" + + # Ignore SIGINT (our parent process will kill us gracefully) + signal.signal(signal.SIGINT, signal.SIG_IGN) + + # Reconfigure the root logger for this process to send all messages to a queue + h = logging.handlers.QueueHandler(queue) + root = logging.getLogger() + root.handlers = [] + root.addHandler(h) + + +def log_listener(queue): + """Listen to the worker processes and forward the messages to logging + + For simplicity this is a thread rather than a process. Only one process + should actually write to sys.stderr or whatever we're using, so if this is + made into a process the main application needs to be directed to it. + + See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes + """ + + while True: + try: + record = queue.get() + if record is None: + break + logger = logging.getLogger(record.name) + logger.handle(record) + except Exception: + import sys, traceback + + print("Logging problem", file=sys.stderr) + traceback.print_exc(file=sys.stderr) + + def exec_concurrent(context): - """Execute the pipeline concurrent""" + """Execute the pipeline concurrently""" # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) if max_workers > 1: context.log.info("Start processing %d pages concurrent" % max_workers) - sidecars = {} - layers = [] + sidecars = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) + + log_queue = multiprocessing.Queue(-1) + listener = threading.Thread(target=log_listener, args=(log_queue,)) + listener.start() with tqdm( total=(2 * len(context.pdfinfo)), desc='OCR', unit='page', unit_scale=0.5 - ) as pbar, concurrent.futures.ProcessPoolExecutor( - max_workers=max_workers - ) as executor: - # layers = executor.map(exec_page_sync, context.get_page_contexts()) - futures = [ - executor.submit(exec_page_sync, ctx) for ctx in context.get_page_contexts() - ] - for future in concurrent.futures.as_completed(futures): - page_result = future.result() - sidecars[page_result.pageno] = page_result.text - pbar.update() - ocrgraft.graft_page(page_result) - pbar.update() + ) as pbar, multiprocessing.Pool( + processes=max_workers, initializer=worker_init, initargs=(log_queue,) + ) as pool: + results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) + while True: + try: + page_result = results.next() + sidecars[page_result.pageno] = page_result.text + pbar.update() + ocrgraft.graft_page(page_result) + pbar.update() + except StopIteration: + break + except (Exception, KeyboardInterrupt): + pool.terminate() + log_queue.put_nowait(None) # Terminate log listener + # Don't try listener.join() here, will deadlock + raise + + log_queue.put_nowait(None) + listener.join() # Output sidecar text if context.options.sidecar: - ordered_sidecars = [sidecars[pageno] for pageno in sorted(sidecars)] - text = merge_sidecars(ordered_sidecars, context) + text = merge_sidecars(sidecars, context) # Copy text file to destination copy_final(text, context.options.sidecar, context) @@ -233,6 +286,9 @@ def run_pipeline(options): # Execute the pipeline exec_concurrent(context) + except KeyboardInterrupt as e: + log.error("KeyboardInterrupt") + return ExitCode.ctrl_c except ExitCodeException as e: log.error("%s: %s" % (type(e).__name__, str(e))) return e.exit_code From c1af0fb18ddfd399d46f78335c9ce9291e0a21e5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 00:50:29 -0700 Subject: [PATCH 027/880] Cleanup ghostscript error output --- src/ocrmypdf/exec/ghostscript.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 70497092..85e82465 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -164,8 +164,7 @@ def rasterize_pdf( log.debug(p.stdout) if p.returncode != 0: - log.error('Ghostscript rasterizing failed') - raise SubprocessOutputError() + raise SubprocessOutputError('Ghostscript rasterizing failed') tmp.seek(0) with Image.open(tmp) as im: @@ -287,5 +286,4 @@ def generate_pdfa( # PDF/A - check PDF/A status elsewhere copy(gs_pdf.name, fspath(output_file)) else: - log.error('Ghostscript PDF/A rendering failed') - raise SubprocessOutputError() + raise SubprocessOutputError('Ghostscript PDF/A rendering failed') From e528adc6030d7d2268825c9406a5b540b113871d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 01:09:06 -0700 Subject: [PATCH 028/880] pylint removal --- setup.py | 3 -- src/ocrmypdf/_sync.py | 76 +++++++++++++++++----------------- src/ocrmypdf/_validation.py | 28 ++++++------- src/ocrmypdf/_weave.py | 14 +++---- src/ocrmypdf/pdfinfo/layout.py | 1 + tests/test_main.py | 2 +- tests/test_unpaper.py | 4 -- 7 files changed, 60 insertions(+), 68 deletions(-) diff --git a/setup.py b/setup.py index 9550b975..794983cf 100644 --- a/setup.py +++ b/setup.py @@ -26,9 +26,6 @@ if sys.version_info < (3, 6): sys.exit(1) from setuptools import setup, find_packages -from subprocess import STDOUT, check_output, CalledProcessError -from collections.abc import Mapping -import re # pylint: disable=w0613 diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a5cec90d..ad79c306 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -15,59 +15,59 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import os import atexit -import concurrent.futures import logging import logging.handlers import multiprocessing -import threading -import sys +import os import signal +import sys +import threading from collections import namedtuple from tempfile import mkdtemp -from ._jobcontext import PDFContext, get_logger, cleanup_working_files -from ._weave import OcrGrafter -from ._pipeline import ( - triage, - get_pdfinfo, - validate_pdfinfo_options, - is_ocr_required, - rasterize_preview, - get_orientation_correction, - rasterize, - preprocess_remove_background, - preprocess_deskew, - preprocess_clean, - create_ocr_image, - ocr_tesseract_hocr, - should_visible_page_image_use_jpg, - create_visible_page_jpg, - create_pdf_page_from_image, - render_hocr_page, - ocr_tesseract_textonly_pdf, - generate_postscript_stub, - convert_to_pdfa, - metadata_fixup, - merge_sidecars, - optimize_pdf, - copy_final, -) -from .exceptions import ExitCode, ExitCodeException + +from tqdm import tqdm + from . import VERSION -from .helpers import available_cpu_count +from ._jobcontext import PDFContext, cleanup_working_files, get_logger +from ._pipeline import ( + convert_to_pdfa, + copy_final, + create_ocr_image, + create_pdf_page_from_image, + create_visible_page_jpg, + generate_postscript_stub, + get_orientation_correction, + get_pdfinfo, + is_ocr_required, + merge_sidecars, + metadata_fixup, + ocr_tesseract_hocr, + ocr_tesseract_textonly_pdf, + optimize_pdf, + preprocess_clean, + preprocess_deskew, + preprocess_remove_background, + rasterize, + rasterize_preview, + render_hocr_page, + should_visible_page_image_use_jpg, + triage, + validate_pdfinfo_options, +) from ._validation import ( - check_options, check_dependency_versions, check_environ, + check_options, check_requested_output_file, create_input_file, report_output_file_size, ) -from .pdfa import file_claims_pdfa +from ._weave import OcrGrafter +from .exceptions import ExitCode, ExitCodeException from .exec import qpdf - -from tqdm import tqdm +from .helpers import available_cpu_count +from .pdfa import file_claims_pdfa PageResult = namedtuple( 'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction' @@ -178,7 +178,7 @@ def log_listener(queue): logger = logging.getLogger(record.name) logger.handle(record) except Exception: - import sys, traceback + import traceback print("Logging problem", file=sys.stderr) traceback.print_exc(file=sys.stderr) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index e80cd98f..6f01fbe8 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -19,7 +19,6 @@ import logging import os - import sys import textwrap from pathlib import Path @@ -27,17 +26,6 @@ from pathlib import Path import PIL from ._unicodefun import verify_python3_env - -from .exec import ( - ghostscript, - jbig2enc, - qpdf, - tesseract, - check_external_program, - unpaper, - pngquant, -) -from .helpers import is_file_writable, re_symlink from .exceptions import ( BadArgsError, ExitCode, @@ -45,6 +33,16 @@ from .exceptions import ( MissingDependencyError, OutputFileAccessError, ) +from .exec import ( + check_external_program, + ghostscript, + jbig2enc, + pngquant, + qpdf, + tesseract, + unpaper, +) +from .helpers import is_file_writable, re_symlink # ------------- # External dependencies @@ -331,7 +329,7 @@ def log_page_orientations(pdfinfo): if angle != 0: orientations.append('{0}{1}'.format(n + 1, direction.get(angle, ''))) if orientations: - log.info('Page orientations detected: ' + ' '.join(orientations)) + log.info('Page orientations detected: %s', ' '.join(orientations)) def check_environ(options): @@ -384,7 +382,7 @@ def check_input_file(options, start_input_file): try: re_symlink(options.input_file, start_input_file, log) except FileNotFoundError: - log.error("File not found - " + options.input_file) + log.error("File not found - %s", options.input_file) raise InputFileError() @@ -402,7 +400,7 @@ def check_requested_output_file(options): raise BadArgsError() elif not is_file_writable(options.output_file): log.error( - "Output file location (" + options.output_file + ") is not a writable file." + "Output file location (%s) is not a writable file.", options.output_file ) raise OutputFileAccessError() diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index da4ca640..ac989a72 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -15,11 +15,12 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from contextlib import suppress -from itertools import groupby -from pathlib import Path import os +from contextlib import suppress +from pathlib import Path + import pikepdf + from .exec import tesseract MAX_REPLACE_PAGES = int(os.environ.get('_OCRMYPDF_MAX_REPLACE_PAGES', 100)) @@ -44,7 +45,7 @@ def _update_page_resources(*, page, font, font_key, procset): resources['/ProcSet'] = procset -def strip_invisible_text(pdf, page, log): +def strip_invisible_text(pdf, page): stream = [] in_text_obj = False render_mode = 0 @@ -151,7 +152,7 @@ def _weave_layers_graft( new_text_layer = pikepdf.Stream(pdf_base, pdf_text_contents) if strip_old_text: - strip_invisible_text(pdf_base, base_page, log) + strip_invisible_text(pdf_base, base_page) base_page.page_contents_add(new_text_layer, prepend=True) @@ -205,7 +206,7 @@ class OcrGrafter: self.interim_count = 0 def graft_page(self, page_result): - pageno, image, text, sidecar, autorotate_correction = page_result + pageno, image, text, _sidecar, autorotate_correction = page_result if text and not self.font: self.font, self.font_key = _find_font(text, self.pdf_base) @@ -262,7 +263,6 @@ class OcrGrafter: # lid on our memory usage for very large files. Attach the font to # page 1 even if page 1 doesn't use it, so we have a way to get it # back. - # TODO refactor this to outside the loop page0 = self.pdf_base.pages[0] _update_page_resources( page=page0, font=self.font, font_key=self.font_key, procset=self.procset diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index e6d04c9c..89caa938 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -172,6 +172,7 @@ class LTStateAwareChar(LTChar): - the Unicode mapping is known, and both have the same render mode - the Unicode mapping is unknown but both are part of the same font """ + # pylint: disable=protected-access both_unicode_mapped = isinstance(self._text, str) and isinstance(obj._text, str) try: if both_unicode_mapped: diff --git a/tests/test_main.py b/tests/test_main.py index 49d93025..3056d527 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -112,7 +112,7 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): ) pix = Pix.open(deskewed_png) - skew_angle, skew_confidence = pix.find_skew() + skew_angle, _skew_confidence = pix.find_skew() print(skew_angle) assert -0.5 < skew_angle < 0.5, "Deskewing failed" diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index e3ff1b67..8df1d5a2 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -15,14 +15,10 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import argparse -import logging from os import fspath -from pathlib import Path from unittest.mock import patch import pytest - from ocrmypdf.__main__ import parser from ocrmypdf._validation import check_options from ocrmypdf.exceptions import ExitCode From 8df1ea2754d67158e4b694de9e9ded840a7d8f88 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 01:42:27 -0700 Subject: [PATCH 029/880] Mark some slow tests --- tests/test_main.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_main.py b/tests/test_main.py index 3056d527..128ca797 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -445,6 +445,7 @@ def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf) @pytest.mark.parametrize('renderer', RENDERERS) +@pytest.mark.slow def test_tesseract_image_too_big( renderer, spoof_tesseract_big_image_error, resources, outpdf ): @@ -1020,6 +1021,7 @@ def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): @pytest.mark.skipif( PIL.__version__ < '5.0.0', reason="Pillow < 5.0.0 doesn't raise the exception" ) +@pytest.mark.slow def test_decompression_bomb(resources, outpdf): p, out, err = run_ocrmypdf(resources / 'hugemono.pdf', outpdf) assert 'decompression bomb' in err From 70def4a0d0187c4e5a16e66c7eff9138a7f1364e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 01:43:26 -0700 Subject: [PATCH 030/880] validation: eliminate print() --- src/ocrmypdf/_validation.py | 20 ++++++-------------- 1 file changed, 6 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 6f01fbe8..af6cf552 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -52,10 +52,6 @@ HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) log = logging.getLogger(__name__) -def complain(message): - print(*textwrap.wrap(message), file=sys.stderr) - - # -------- # Critical environment tests verify_python3_env() @@ -297,7 +293,7 @@ def check_closed_streams(options): if sys.stdin is None: if options.input_file == '-': - print("Trying to read from stdin but stdin seems closed", file=sys.stderr) + log.error("Trying to read from stdin but stdin seems closed") return False sys.stdin = open(os.devnull, 'r') @@ -306,14 +302,10 @@ def check_closed_streams(options): # Can't replace stdout if the user is piping # If this case can even happen, it must be some kind of weird # stream. - print( - textwrap.dedent( - """\ - Output was set to stdout '-' but the stream attached to - stdout does not support the flush() system call. This - will fail.""" - ), - file=sys.stderr, + log.error( + "Output was set to stdout '-' but the stream attached to " + "stdout does not support the flush() system call. This " + "will fail." ) return False sys.stdout = open(os.devnull, 'w') @@ -460,7 +452,7 @@ def check_dependency_versions(options): need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports ) if ghostscript.version() == '9.24': - complain( + log.error( "Ghostscript 9.24 contains serious regressions and is not " "supported. Please upgrade to Ghostscript 9.25 or use an older " "version." From 4340ad9f1275c3d4f9da203ab358a8c9043b6e95 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 01:45:06 -0700 Subject: [PATCH 031/880] Update test cache --- .../pdf.bin | Bin 4010 -> 4010 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 16 +- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 2986 -> 2986 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 10249 -> 0 bytes .../hocr.bin | 1462 ++++++++-------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 10249 -> 10249 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 10249 -> 0 bytes .../hocr.bin | 1462 ++++++++-------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 0 -> 11243 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 6 +- .../hocr.bin | 1462 ++++++++-------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 0 -> 11513 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 128 ++ .../pdf.bin | Bin 10249 -> 0 bytes .../hocr.bin | 1462 ++++++++-------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 0 -> 12456 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 124 ++ .../stderr.bin | 0 .../stdout.bin | 0 .../stderr.bin | 0 .../stdout.bin | 0 .../stderr.bin | 0 .../stdout.bin | 0 .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 123 -- .../txt.bin | 123 -- .../hocr.bin | 1462 ++++++++-------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 10242 -> 10242 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 28 +- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 3073 -> 3073 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../txt.bin | 13 - .../pdf.bin | Bin 3613 -> 3610 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 13 + .../pdf.bin | Bin 3614 -> 0 bytes .../txt.bin | 13 - .../pdf.bin | Bin 3309 -> 3309 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 606 +++---- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 5909 -> 5909 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 4 +- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 2860 -> 2860 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 tests/cache/manifest.jsonl | 112 +- .../hocr.bin | 424 ++--- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 5173 -> 5173 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 2998 -> 2998 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 224 +-- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 4229 -> 4229 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 192 +-- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 3986 -> 3986 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 1070 ++++++------ .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 8535 -> 8535 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 1070 ++++++------ .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 8060 -> 8060 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../hocr.bin | 576 +++---- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 5744 -> 5744 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 10889 -> 0 bytes .../pdf.bin | Bin 0 -> 10842 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 11 +- .../stderr.bin | 0 .../stdout.bin | 4 +- .../hocr.bin | 2 +- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 2798 -> 2798 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../stderr.bin | 1 - .../stdout.bin | 0 .../hocr.bin | 1474 ++++++++--------- .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 .../pdf.bin | Bin 12674 -> 12674 bytes .../stderr.bin | 0 .../stdout.bin | 0 .../txt.bin | 0 178 files changed, 6838 insertions(+), 6829 deletions(-) rename tests/cache/2400dpi/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/2400dpi/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/2400dpi/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/2400dpi/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (73%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/aspect/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) delete mode 100644 tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (58%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) delete mode 100644 tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002__hocr__txt => __-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt}/hocr.bin (58%) rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt}/txt.bin (100%) create mode 100644 tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002__hocr__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002__hocr__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/txt.bin (98%) rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/hocr.bin (58%) rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000002.ocr.png__000002__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/txt.bin (100%) create mode 100644 tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/stdout.bin (100%) create mode 100644 tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/hocr.bin (58%) rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/txt.bin (100%) create mode 100644 tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/stdout.bin (100%) create mode 100644 tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin rename tests/cache/cardinal/{__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout}/stdout.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout}/stderr.bin (100%) rename tests/cache/cardinal/{__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout}/stdout.bin (100%) delete mode 100644 tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin delete mode 100644 tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (58%) rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/{cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt => ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/ccitt/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/{cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt => ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (74%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/cmyk/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) delete mode 100644 tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/txt.bin rename tests/cache/francais/{__-l__deu__000001.ocr.png__000001.text__pdf__txt => __-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (73%) rename tests/cache/francais/{__-l__deu__000001.ocr.png__000001.text__pdf__txt => __-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/francais/{__-l__deu__000001.ocr.png__000001.text__pdf__txt => __-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) create mode 100644 tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/pdf.bin delete mode 100644 tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/txt.bin rename tests/cache/graph_ocred/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/{francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt => graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/{francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt => graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/graph_ocred/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (60%) rename tests/cache/{graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt => jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/{graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt => jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/jbig2/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (79%) rename tests/cache/{jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt => lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/{jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt => lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/lichtenstein/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (60%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/{lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt => multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/{lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt => multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/hocr.bin (66%) rename tests/cache/multipage/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000002.ocr.png__000002.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003.text__pdf__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/hocr.bin (67%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000003.ocr.png__000003__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004.text__pdf__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005__hocr__txt => __-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt}/hocr.bin (57%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000004.ocr.png__000004__hocr__txt => __-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005.text__pdf__txt => __-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005.text__pdf__txt => __-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005.text__pdf__txt => __-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005.text__pdf__txt => __-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005__hocr__txt => __-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006__hocr__txt => __-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt}/hocr.bin (56%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005__hocr__txt => __-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000005.ocr.png__000005__hocr__txt => __-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006.text__pdf__txt => __-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006.text__pdf__txt => __-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006.text__pdf__txt => __-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006.text__pdf__txt => __-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/multipage/{__-l__eng__000006.ocr.png__000006__hocr__txt => __-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt}/txt.bin (100%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (59%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/{multipage/__-l__eng__000006.ocr.png__000006__hocr__txt => palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/palette/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) delete mode 100644 tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin create mode 100644 tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin rename tests/cache/{multipage/__-l__eng__000006.ocr.png__000006__hocr__txt => poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/{palette/__-l__eng__000001.ocr.png__000001__hocr__txt => poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/poster/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (91%) rename tests/cache/poster/{__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout}/stderr.bin (100%) rename tests/cache/poster/{__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout => __-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout}/stdout.bin (54%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (74%) rename tests/cache/{poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt => skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/{poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt => skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (98%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt => __-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) delete mode 100644 tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin delete mode 100644 tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/hocr.bin (56%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stderr.bin (100%) rename tests/cache/skew/{__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/stdout.bin (100%) rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt}/txt.bin (100%) rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/pdf.bin (99%) rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stderr.bin (100%) rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001.text__pdf__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/stdout.bin (100%) rename tests/cache/skew/{__-l__eng__000001.ocr.png__000001__hocr__txt => __-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt}/txt.bin (100%) diff --git a/tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 5bf76e4808033b5cda514189c74fadc502c7197c..fc70a6311190482b159ffcb0654ea99c0175220f 100644 GIT binary patch delta 27 icmZ1_ze;|CIUk>;fvKUHp^2%9iGi+x`DRzXR7L=39|p|; delta 27 icmZ1_ze;|CIUk>ep^>qHfq|)^p|P%k#b#H&R7L=2Q3k&N diff --git a/tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/2400dpi/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 73% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index bc723882..f9726b75 100644 --- a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,23 +9,23 @@ -
+

- +

- This - should - be - a - perfect - circle: + This + should + be + a + perfect + circle:

diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 4420bebc588c779812cffdee22130e6c37836447..8f965ffd680a9ddf70de6a811922f679238f0a67 100644 GIT binary patch delta 27 icmZ1_zDj(9ITxR$fvKUHp^2%9k*ThM`DRzHR7L=1NCvn7 delta 27 icmZ1_zDj(9ITxRWp^>qHfq|)kp}DSs#b#HoR7L=0dIqWh diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/aspect/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin deleted file mode 100644 index e308fb4c54605719eb771b48594c0e24e4f76924..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 10249 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a?! zq-S8DU~UPb!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?N12Mi`_tqkB?)C6D@)mvbrtz{c=FL5>u$iG;EM>}qblVp^pTAB_nHV$W$#KiOju&?9 zxEgwUo6pCMJ3LO+SFgRc*XLjSt^WQ0Kd)b3|NHUz`}IGrzyDu%^Vj+}rz3Z-xvky$ z;n&~m*X{p4J-cV-bmR5+>nkIli+}(1*M5C{{ol=oh7z0ee)|8Je!YHA-T&9xJ5u96 z+t&TRemZKi#Cdz(_dnCC|9!6f)44Au&1ch3f17_h|C#^#zV7a){_C^1F1+*Y?95c& zr}ws~L{Cq7l%#Bb;<aGWPu~)2H9#b9XKjKmYuF;{R=X z+;3;!t*xuPd0;Qs>0_$LzSgk6-PW>go~`b=-TUj=lcQ|jHl2^&aq98uzmt}`w{P0( zD^m6`&Sm4RZ?pF-c`W`@_M@Aoy-91bb@1egzb&3@TX)=Ez3%$A`)?F4i)3hjF55ks zb&GAp=~|!L{P*J&Ob$-@|IbtZt@vU6$*EEwQfDoAvC~X>MzmywPd^f*61DO zJ6UtTi+#;||MrrB`MSHgQSv9{>KZ35uRGefHTC+kaDBzg73{tRw}r0%>VOFMe*h@_ySNf!qF;#-C$qkF?n- zz5U3OWxFE(*QpCm_F1&!-d7iM*2^jHWv(0Sn6I^MTLqU%aE_od#}S>`bqiD{cC=Y@ z9=*LnKYWLR^2P%@m<*#!wM4d>UG!tK{kc)PL7>>>--K}S@(;F=MSWB6>nQAH;^A7S zQ5~Z<-J9wCGx`vII^Q-YPDXxPCH$l@*yH;f6TPd zlP^!-YEH`9{z85iTbZ|6%KZKZ>vVU-MH@VPz9S(j|6s`Tb0;Gk-5*R{ePRDBmz>EO zE8FfLnVsGo!)syhq&0s*N4&46ZN)C(Co@{Nik&&V<6_v9y{Xz81fJaTH9N@9c1_Ol zXXNoo4NYs38jkWCC_J0SVja=(P5I1?i0tjU@v7xtem?wp^QZswYsE`?7k0;nOH7OL z5m_)LymkAFr?K~U{|wn@Q*qF5c9rJTbvGLP3#(^X89goXn4R>QRd-n@SNF0f>gyFs zoh02m^8eQ}bA&No?Ra>3-;_6k<*wGd{myT?I(hzyZ7&X3&tG}+RCU(js*egglailL z{rK6=C8_Ta>z>r(+z));yYX%>+$MdhPBh7P1GD2EefP{%i`o=Vz8aH`HuJ{j#)T%u z=LIjWUw+KEKF^fn$JC17rq7$NORig=bD;3@lJJ+aPfZY;BvTd7T05?AD|dg~f`0F81ZVKi$}xgw(byQF>!>v&{Ew zxPgh;s=3OmPETS{)$+UYv?IqpW4`k5+izTLlD0WK5VY-hA{2dY@}hs0u@O5{T&rVl zuFrEjDz{9DW0ocl-iZt2`_=aDLGfh__tEc|nok!b{H8f2+%X&si;fa>Ez5x{G}ZbIg+_S!yw8tTnqC z&{(Zr!ob0((IitXZ1-}rt!~L|OVd-6qW3?_OMiB8XP#$+T7z5u^h>v_xp^&P!n>u< z7_Z1MKY7_oM)a4#lHhxV&OdxxJT`7Sede^<{%*ew?EPFK=L(NXO!C^d=*qhOpA+1s zF7ej=sdDEwlcv?QElhRQ41Z^3&zR=$W7^IMv(z7!zA>A{vfKE1>y!&Y@{Q|0h+7K# zEIc0){lZ0pwNGtV;(Q5%ExUKUITvu(WA7T48(UuO{&TaW{P>hdU!N6)oRV|QX!3jh zW3I#Vq9p!lQc{g=EiCR@B?lR_HeJ7WNX0qt)v-7Ffs8fWidt*$UyZx^Lk0bi%_*6_*yj=#_Kj>}#ACrhLF7Z^|vEJ+n{E+|Aqb zboz$+TcNw-5C6MrqWdS|{|T*Pv+fxUM%kBnPCF&(OC;zwSpJFlbJSn`?TlAWovP~k zPTN1e*mGd_lRsTDNo@;`^=^@?tbF!tkBf}!#U&FDJ~@6P_Q>6rvCPHpZ|3}%##3x~ z-HtmgPq-|AS?cLlgYp-9H7j{NEv>)gxM*EfvDL9Xv@cPsR48Th@BYaAKPN&`rt~ga z!+%HWh}v}I(Vhn#`NMv-8-#mNgiYagnYOx^S>S^j|0(tCAA zw_**~vhnd&n(m8CoccmmYxOOObxSTyPGLEe^u|^BQqHNaGoK&ywEVe!nWI@DIy&96 z?%u-r3}=1Sxn4`Y?U^Rf(CB3~G3d@pmpISBB?^V`R9g^3UN zjCn#X7H_zh{wl+_(DeTGDNn^P2MoWb;r)W zFRu@8V~^Wb?lfg3%MaUS%~6U5rpw+Y9^0b&(9~A$m2f|c{I}9)i>eO3|1WTmJM4C- zL+Dr4FgZS_H5tK%ei2W0e?Rxae95ANA8tK+amC;?TgcbUX6N_TO;22ZTOFINaP@R$c6%+$pLp6@*J0|h z1$*;u-Hv;)>u}bhZ!#=>OZI5bO8IzPi_P=DXxBo1#^6b4tcD z1^$LTS*nXS&(%s)%zf*1j59PNmW$f~q5^rGhKRB-K(f4IDzt8he@P9UkMa1KjS7CmD;)AX#_JgMEr}RBIU#36bt-Slu*6gPT%eS?C)i@jLAfG)~_UUoa`jQ!*( z+hs1iz5M)pPs{yD;HlSc93c7 zyf3yMIA}xs$&!q`oSf|Td1v%`PMIB( z3hLM3{`T7Cqrdgx1&f2d+?TR13M=GHnEEc~1?vpcutY=E9sjPE{|#}RF!QYC{+C4u zPuFRkUwi+Ch*6RKq=gm?55v!f1gPjvX4B+bS98z1)uB{avFZ4)mD27-%r>rbUhp-= z-c30s#i@9##j$A9wgn$1?tPe`v%TfcQ|_B@873v%IKAO+iRVkb+KX=b*QH;aabj-! z=4!0fzDMcH(&umZmpWd!HJkIgGUI&dM5pI=hMH@m5ApGt|I#U(Vz%IYI+tLTc!;Ho zcuMP)I~yi$J2mmwS?9-3=kE~AyAaFj7`s7r!8fkdmqz;%`6^45N=_~|IK_M9vN867yZ$@y5`jl6-ACqLVx}mRrD{E ze#7?Ips4uPp|Vpg0Xmm#!cMXCo1IwrV7~IS3LE!{4V>-oOnky3ggjXo%EL;~AFtI1Voqu-Mf{cIt ztc6RL+w{LVc_RTCzg2d%!q;Je)0 zzpWTCY;n#A2k)yrSW9hPl9sZAZ1eA2-E?|J-)Gujte36I*yp`b!x! zmL1#wyHreZ&&3)ce>;zs35FLGH>@e)^fQ^ZkJD{3|7q9An}!Qk2wrCS=6Cj>N9kJU zUF~n*nAHDVpl+31kgv7rP1ihCM|OR|ceA$rv#5AAxoXM9y^5QwmZ&{vyXRO^V|3cm zab=g2Cud6c>SHo$|39at@pN8f_2zkVOk?Y=RS|uR4%_)pcOBZYpyJkz_J99Z?LAVo zc|vJdt>x-jVe!+XUTYgUZx zE%*6dm){0VO_}q!I>)TjXm0;ZzJ?j8lXSAbd_Frz;rB|JnOmZ5&fjW$vgSyjR+%z7-a*{#9$- zeej*N=jPIHl75+S5A@_8?Otc&RqSg#?GO7=o$asmX1>jc-;?m^`Mu5O<(HYQ{(LWg ziILh$!#$f7EH9fj=pTvYOW8X?mH+J7@=nq1uG?6$oqAUOnR0xxw~WE2iGL2sZC&|Z zyG%Y|Z_KF&|5_#*zny*5@0N?+&rRwl4|_~sD(d_9jFw9K=8evuCvR5XcJ8G>wFk$f zAKE$VSnF!9U);pSzRKU@#k=HF##`dGHdJ1YdwPWN--^9iS;ts^Sj)Esml+B5>^psR zo|p==neYo01sls}iQ59|7o41MEmdRQ``LO)afkFX`DRFd&zZ7}XHf&wd2PcP`#C}m z9&Y#R3d~X4XD#?(n~3O)Z=5fWa3*bNO*Kbiy^tTzNh+3QjRMx@>T%vg26$H`&q&K4G7iaIfi<^7Yh}eYf_@&v(C{=nA~# zsPT_IRe9KTb=rjdsYa)cRctlfW^L%hE&YjMxqCv{OyR{Pi_hJh@M!yFLnZUi3SXau zd_J~#rU>kzd%DGPPm#A?9 z&ez)}Fuq-p(cjgZmprvy&*7~ptLIs!!rZr;xy&kSFWoXfl`z@pqfW4eV$R_JWzp-~ z-gikX+9sUc_A$riZd!F(z#0jQ$;Wlmbl-bV*4S<-y6Wi+LBYu}&la?mMVhbGT`YA! zxLLDjeyiY>)t*5Mo}JyfEUY^&Iz9bvUG197nSs8K=D!PJ=e;Pva@QfC-Kj8Rk&sIE z?~J?r)$>Do=LhL!$Q{(kR^O@0$7abCk~}YaeS{{1iTL6p2Y#29yz~0oRc(Ce%FMv0 z+KaO8ZMtcnXeS^w%UWn--6?h7y&(lm7auK&>p1RWJL`VaymR>h79UsMUK8xQ?e(Rg zKaB>l&ks6o5c{)4{>s4vkw1StN@~#wik-(Ilwo=2;_{;6hUIB~(Y*?XUkJDAed69P z&))b|#Nv}B>;DHe-=?teJGjO1|B3L~r95knz^i=s>2Eog@|nD;xZ^m>^g55V-adWl z6E(*de%)>S$5CSP=lgj!73~w;%jEc^xMqDna+v3B@PhWM_fAh#XPJ59rpx=)r}6@v zD{fqOO8d8g`JbL%XuaM(-aT*696Otm|6*TOpVC2VCR^q%Ut577uK&l<5`yE+4t#UZ zJM#GD=9xWdw{IvexRMuqHtM0sE-tm%Ht$Zw`CYwe?|$PcPil(6x{ys*r#6@DReD&p zXxio?bD=d|Z=Gt^Mj89bu1R3F%So1Y6LeeB+_R2{=k>c;mYXiHhMVo$t1q+d&<^cC zs=*FMTBUJLjb~1k?l(NtBzVzLMr~6<>ik2?+!`*tzT-Rjp*EZ1=J`Av+v1zN_b%8z z@6q7}%nJTK-}av9m>43={$1c9C&RR#?CUR0b)R#vxAft{RR+iY<(055c&9mM&sCX~ zr*4_s`6&sW(l2bZ_1{W%rhC_FtE97@y2l zdwjcmYlDB_oQ-QYeY+57ro*hW%An5ckWznx`w?3eA*=R$$HS`}U#;=&w|DE*HQxGW zW9pGLf2wy)u$p%Eo2Q+OUYp35TlsPyOt#A?&HG!+uJWOvK#B9pOIs&BkIQSUwzoUU z6#WSK(3!GAU42KniO2Q5`{SQD7gRI^{zg@oP`f4wv@Bg~R zp8eR`^*@}{Wf@b|rU+ZaB)pNA`gyr3`o1S?&=q%=wHYqL;Wy{6?NENO%y?HMzx(t_ zr}WRXsO`1=Bh)4rvbD#Q_3dWghZc-K(_?jt(vGD~=iJP7cap87ZcXr!;#=++1v(p# zPv>78tuR=3^5e`6^?{u`td}ukzhhI)Uy=C}wfH}2o7AHKCObkni+n1Hv_S3Wm0)Y<#u{oN~ui)9Os#EK+)ty*!+ z{qyEs%rm}qF@4ywSf=FL?9H`Wn-zU$FF71#{-ERF-rqVwhcA|#Tt4%7w1vUy^n-fV zF3aDCln0f!F?Yjy+`@P1cYi`1a^3=hQ0?r*8MU(75!#!F{j) zFbXl)?OmvKLFC_J!^_X&R$PufIA^79_>bo~e{TfW&+1BgbV%)JRAx)-#!i_RQ}hJu zeU=pnB^Kp)eC*^>YHJe@Ykyg@yjY0uvSzD?jsJ4ff(M^pemd>D@58iZ-wNu_$F6do zVgGdI*V*h1!Y3zRPPr!X;<&6oYxw=JYKw-mwYs})${MUW_wg-|`mEmwuLj2JFlKtCqHFEZAo*ik=I zELS-0#}bZ+w7u$w|G&BrY@9Oxk^YsHA_1<~+4pzMIAh~Dv+mFN-mM~+xBL74y1PCq zY|h*XzrMa~;QGz)_o!Xq9pAKxldt(+UsxI!ZM5rCTY7fu{NDHTT6X+VO8s?c*GJ~< zb)E+z`!Cn+u+xt;@R@zif5GwBW|?fS`V3$BYyaK#LoP04ZmpNiU4y#v^bU8%aB)$8)S zq(snb-hHcQPo4yCfAoU;vx$9Ld#2(c9|z%c$62bPj1?OGOknu%fJgYhu8}L(ixZcx z$SwKD?JX2IX}{=_+288DZheeVJQXtg8XE`yUb89L3-}veyO_t8@>RWgu0=DaHh+mav9-M0bzx7I1>3siBrPJzGUR5_}>a222Tz~%j z_5IPSn5tX_o}BpN$F*_EC#QLL^WOVSJ7A{c6Roaq#Vpl$&DSKt?aHo~Qy2L$maoDPJbn$enz|*0O#+w**T^&f(dv)tgT~jk+_7DZ}b!n#08fM>?2P zN@h)s&^!3!g`WS_8z))|>}E$7Y_O@TI(qQM;TWmeK9ds~63(o?%_f@Ibd`6>d(L&{ zdFLKA23bb#`IDGpyrV95(wF>R)@q3(O)SdycuJF^xpqBoy2kM!z{FBg_3h&&x#5$4 zws}8FGZ1Lwy0I^geUYP4!=Hxxt7c_ZY_xv*I3eNFRJ}Wm4BLgH`;rttF-CE`VmT?r z{(ILA@yp*f8@Y?;7zuCUi>sOHvxWCs$q$SDA)DRXb#%@hJNx^^`=d6V2~Upx2YPhLi=h zDy$L4#^*TBeB3N5)csS}W9h>}BdJn8{|MK^8bNDo=IrTd`p|qYm(QAsisq0TQWzF-?Uu45lx zuSvBRy=W?K|Mb{Z=^YXa8=n}Sk}WHJEdFK=|J~Tq1s``j+EEvO>2{69h3eLWGMRd{ zoL^_kea$!?$?#p${f_yHdmg&~cHh-p_gZj<_l*fB@*mlYYlennOeY)Rm`(=G|LS{Z zrW7kgD}bh)gTRxz`p)^Kc_j*lNXsD%QP+_L6y>LsCZ`rDXoRE|7pE2_CYLCf=o#o4 zfLFmd7o{eaWaj6&B$lKqXt-Dz85mj^8W|fH7?>KF80Z>Us2dolgH?uPmgJ-=*tofZ zmbBP`SG9nqh%1Ux)3^*242`%zYhEB&!OYau*i<1+0WM}}ssNT!$b*Slnwwaniy0W0 zSzw4685v-RnV6bmh?$#Us53M)F-BKsXl!bTA!ccbq1VXB07J~u!q5bvx1=aBGbgnO zx?C?fvnmx73JMDPLHYS53ZO^;&) -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

- extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

- ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls. + FORWARD, + REWIND, + and + LOCATE + controls.

- e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

- synthesizers! + synthesizers!

- ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

- per - disk! + per + disk!

- ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

- rhythmic - value. + rhythmic + value.

- ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

- ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization.

- © - Optional - remote - control. + © + Optional + remote + control.

- Recording - a - Sequence + Recording + a + Sequence

- To - record - a - sequence, - simply - press - RECORD - and - PLAY, + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

- corrected! - (Timing - correction - may - be - adjusted - or - defeated). + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

- Any - additional - notes - played - will - be - added - into - the - track + Any + additional + notes + played + will + be + added + into + the + track - — - existing - notes - are - not - erased - while - recording! + — + existing + notes + are + not + erased + while + recording!

- FAST - FORWARD, - REWIND, - and - LOCATE - controls + FAST + FORWARD, + REWIND, + and + LOCATE + controls - may - be - used - at - any - time - to - quickly - access - any - location - in + may + be + used + at + any + time + to + quickly + access + any + location + in - your - sequence - for - spot-recording. - To - overdub - a - new - part, + your + sequence + for + spot-recording. + To + overdub + a + new + part, - select - a - different - track - and - start - recording—while - you + select + a + different + track + and + start + recording—while + you - record, - the - first - track - will - play - in - perfect - sync - (unless - you + record, + the + first + track + will + play + in + perfect + sync + (unless + you - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - including - pitch - bend, - modulation, - velocity, - aftertouch, + including + pitch + bend, + modulation, + velocity, + aftertouch, - sustain - pedal, - and - program - changes! + sustain + pedal, + and + program + changes!

- Editing + Editing

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - when - played - back, - it - will - be - gone. - Notes - may - also - be + when + played + back, + it + will + be + gone. + Notes + may + also + be

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

- Additional - Features + Additional + Features

- simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - find - the - desired - bar - number, - then - start - recording. + find + the + desired + bar + number, + then + start + recording.

- The - INSERT/COPY - function - allows - you - to - move - bars + The + INSERT/COPY + function + allows + you + to + move + bars - from - one - location - to - another—in - the - same - sequence - or - a + from + one + location + to + another—in + the + same + sequence + or + a - different - one. - For - example, - you - might - insert - a - copy - of - the + different + one. + For + example, + you + might + insert + a + copy + of + the - first - verse - between - the - second - chorus - and - the - bridge. + first + verse + between + the + second + chorus + and + the + bridge. - DELETE - BARS - operates - the - same - way - to - remove + DELETE + BARS + operates + the + same + way + to + remove - unwanted - sections, + unwanted + sections,

- Creating - a - Song + Creating + a + Song

- One - way - to - create - a - song - is - to - record - each - track - all - the + One + way + to + create + a + song + is + to + record + each + track + all + the - way - through - (up - to - 999 - bars). - Another - way - is - to - record + way + through + (up + to + 999 + bars). + Another + way + is + to + record - each - basic - section - (verse, - chorus, - etc.) - in - individual + each + basic + section + (verse, + chorus, + etc.) + in + individual - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - them - together. - CREATE - SONG - will - then - automatically + them + together. + CREATE + SONG + will + then + automatically - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

- Composition - Without - Compromise + Composition + Without + Compromise

- The - technology - you - use - should - never - be - so - complex - that + The + technology + you + use + should + never + be + so + complex + that - it - interferes - with - the - creative - process. - That’s - precisely - why + it + interferes + with + the + creative + process. + That’s + precisely + why - the - LinnSequencer - is - designed - to - let - you - compose, - record + the + LinnSequencer + is + designed + to + let + you + compose, + record - and - edit - while - devoting - your - undivided - attention - to - your + and + edit + while + devoting + your + undivided + attention + to + your - music. - See - your - Linn - dealer - today - for - a - demonstration! + music. + See + your + Linn + dealer + today + for + a + demonstration!

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

- HELP - button - displays - additional - explanations. + HELP + button + displays + additional + explanations.

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

- ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

- ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

- © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

- © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

- (even - drop - frame!) + (even + drop + frame!)

- ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

- on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

- linn + linn - Linn - Electronics, - Inc. + Linn + Electronics, + Inc.

- 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/pdf.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index e308fb4c54605719eb771b48594c0e24e4f76924..6e958fc5a44e28c784e79c0ee82bd52d8d8efb88 100644 GIT binary patch delta 27 icmeAS=nU8}O^wgez|_#p(8Sc#z*yJ7eDe~uR7L=H;Ro9Q delta 27 icmeAS=nU8}O^wgO(8$=pz`)eV#6Z`;V)GKUR7L=HBL~g^ diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/pdf.bin deleted file mode 100644 index e308fb4c54605719eb771b48594c0e24e4f76924..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 10249 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a?! zq-S8DU~UPb!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?N12Mi`_tqkB?)C6D@)mvbrtz{c=FL5>u$iG;EM>}qblVp^pTAB_nHV$W$#KiOju&?9 zxEgwUo6pCMJ3LO+SFgRc*XLjSt^WQ0Kd)b3|NHUz`}IGrzyDu%^Vj+}rz3Z-xvky$ z;n&~m*X{p4J-cV-bmR5+>nkIli+}(1*M5C{{ol=oh7z0ee)|8Je!YHA-T&9xJ5u96 z+t&TRemZKi#Cdz(_dnCC|9!6f)44Au&1ch3f17_h|C#^#zV7a){_C^1F1+*Y?95c& zr}ws~L{Cq7l%#Bb;<aGWPu~)2H9#b9XKjKmYuF;{R=X z+;3;!t*xuPd0;Qs>0_$LzSgk6-PW>go~`b=-TUj=lcQ|jHl2^&aq98uzmt}`w{P0( zD^m6`&Sm4RZ?pF-c`W`@_M@Aoy-91bb@1egzb&3@TX)=Ez3%$A`)?F4i)3hjF55ks zb&GAp=~|!L{P*J&Ob$-@|IbtZt@vU6$*EEwQfDoAvC~X>MzmywPd^f*61DO zJ6UtTi+#;||MrrB`MSHgQSv9{>KZ35uRGefHTC+kaDBzg73{tRw}r0%>VOFMe*h@_ySNf!qF;#-C$qkF?n- zz5U3OWxFE(*QpCm_F1&!-d7iM*2^jHWv(0Sn6I^MTLqU%aE_od#}S>`bqiD{cC=Y@ z9=*LnKYWLR^2P%@m<*#!wM4d>UG!tK{kc)PL7>>>--K}S@(;F=MSWB6>nQAH;^A7S zQ5~Z<-J9wCGx`vII^Q-YPDXxPCH$l@*yH;f6TPd zlP^!-YEH`9{z85iTbZ|6%KZKZ>vVU-MH@VPz9S(j|6s`Tb0;Gk-5*R{ePRDBmz>EO zE8FfLnVsGo!)syhq&0s*N4&46ZN)C(Co@{Nik&&V<6_v9y{Xz81fJaTH9N@9c1_Ol zXXNoo4NYs38jkWCC_J0SVja=(P5I1?i0tjU@v7xtem?wp^QZswYsE`?7k0;nOH7OL z5m_)LymkAFr?K~U{|wn@Q*qF5c9rJTbvGLP3#(^X89goXn4R>QRd-n@SNF0f>gyFs zoh02m^8eQ}bA&No?Ra>3-;_6k<*wGd{myT?I(hzyZ7&X3&tG}+RCU(js*egglailL z{rK6=C8_Ta>z>r(+z));yYX%>+$MdhPBh7P1GD2EefP{%i`o=Vz8aH`HuJ{j#)T%u z=LIjWUw+KEKF^fn$JC17rq7$NORig=bD;3@lJJ+aPfZY;BvTd7T05?AD|dg~f`0F81ZVKi$}xgw(byQF>!>v&{Ew zxPgh;s=3OmPETS{)$+UYv?IqpW4`k5+izTLlD0WK5VY-hA{2dY@}hs0u@O5{T&rVl zuFrEjDz{9DW0ocl-iZt2`_=aDLGfh__tEc|nok!b{H8f2+%X&si;fa>Ez5x{G}ZbIg+_S!yw8tTnqC z&{(Zr!ob0((IitXZ1-}rt!~L|OVd-6qW3?_OMiB8XP#$+T7z5u^h>v_xp^&P!n>u< z7_Z1MKY7_oM)a4#lHhxV&OdxxJT`7Sede^<{%*ew?EPFK=L(NXO!C^d=*qhOpA+1s zF7ej=sdDEwlcv?QElhRQ41Z^3&zR=$W7^IMv(z7!zA>A{vfKE1>y!&Y@{Q|0h+7K# zEIc0){lZ0pwNGtV;(Q5%ExUKUITvu(WA7T48(UuO{&TaW{P>hdU!N6)oRV|QX!3jh zW3I#Vq9p!lQc{g=EiCR@B?lR_HeJ7WNX0qt)v-7Ffs8fWidt*$UyZx^Lk0bi%_*6_*yj=#_Kj>}#ACrhLF7Z^|vEJ+n{E+|Aqb zboz$+TcNw-5C6MrqWdS|{|T*Pv+fxUM%kBnPCF&(OC;zwSpJFlbJSn`?TlAWovP~k zPTN1e*mGd_lRsTDNo@;`^=^@?tbF!tkBf}!#U&FDJ~@6P_Q>6rvCPHpZ|3}%##3x~ z-HtmgPq-|AS?cLlgYp-9H7j{NEv>)gxM*EfvDL9Xv@cPsR48Th@BYaAKPN&`rt~ga z!+%HWh}v}I(Vhn#`NMv-8-#mNgiYagnYOx^S>S^j|0(tCAA zw_**~vhnd&n(m8CoccmmYxOOObxSTyPGLEe^u|^BQqHNaGoK&ywEVe!nWI@DIy&96 z?%u-r3}=1Sxn4`Y?U^Rf(CB3~G3d@pmpISBB?^V`R9g^3UN zjCn#X7H_zh{wl+_(DeTGDNn^P2MoWb;r)W zFRu@8V~^Wb?lfg3%MaUS%~6U5rpw+Y9^0b&(9~A$m2f|c{I}9)i>eO3|1WTmJM4C- zL+Dr4FgZS_H5tK%ei2W0e?Rxae95ANA8tK+amC;?TgcbUX6N_TO;22ZTOFINaP@R$c6%+$pLp6@*J0|h z1$*;u-Hv;)>u}bhZ!#=>OZI5bO8IzPi_P=DXxBo1#^6b4tcD z1^$LTS*nXS&(%s)%zf*1j59PNmW$f~q5^rGhKRB-K(f4IDzt8he@P9UkMa1KjS7CmD;)AX#_JgMEr}RBIU#36bt-Slu*6gPT%eS?C)i@jLAfG)~_UUoa`jQ!*( z+hs1iz5M)pPs{yD;HlSc93c7 zyf3yMIA}xs$&!q`oSf|Td1v%`PMIB( z3hLM3{`T7Cqrdgx1&f2d+?TR13M=GHnEEc~1?vpcutY=E9sjPE{|#}RF!QYC{+C4u zPuFRkUwi+Ch*6RKq=gm?55v!f1gPjvX4B+bS98z1)uB{avFZ4)mD27-%r>rbUhp-= z-c30s#i@9##j$A9wgn$1?tPe`v%TfcQ|_B@873v%IKAO+iRVkb+KX=b*QH;aabj-! z=4!0fzDMcH(&umZmpWd!HJkIgGUI&dM5pI=hMH@m5ApGt|I#U(Vz%IYI+tLTc!;Ho zcuMP)I~yi$J2mmwS?9-3=kE~AyAaFj7`s7r!8fkdmqz;%`6^45N=_~|IK_M9vN867yZ$@y5`jl6-ACqLVx}mRrD{E ze#7?Ips4uPp|Vpg0Xmm#!cMXCo1IwrV7~IS3LE!{4V>-oOnky3ggjXo%EL;~AFtI1Voqu-Mf{cIt ztc6RL+w{LVc_RTCzg2d%!q;Je)0 zzpWTCY;n#A2k)yrSW9hPl9sZAZ1eA2-E?|J-)Gujte36I*yp`b!x! zmL1#wyHreZ&&3)ce>;zs35FLGH>@e)^fQ^ZkJD{3|7q9An}!Qk2wrCS=6Cj>N9kJU zUF~n*nAHDVpl+31kgv7rP1ihCM|OR|ceA$rv#5AAxoXM9y^5QwmZ&{vyXRO^V|3cm zab=g2Cud6c>SHo$|39at@pN8f_2zkVOk?Y=RS|uR4%_)pcOBZYpyJkz_J99Z?LAVo zc|vJdt>x-jVe!+XUTYgUZx zE%*6dm){0VO_}q!I>)TjXm0;ZzJ?j8lXSAbd_Frz;rB|JnOmZ5&fjW$vgSyjR+%z7-a*{#9$- zeej*N=jPIHl75+S5A@_8?Otc&RqSg#?GO7=o$asmX1>jc-;?m^`Mu5O<(HYQ{(LWg ziILh$!#$f7EH9fj=pTvYOW8X?mH+J7@=nq1uG?6$oqAUOnR0xxw~WE2iGL2sZC&|Z zyG%Y|Z_KF&|5_#*zny*5@0N?+&rRwl4|_~sD(d_9jFw9K=8evuCvR5XcJ8G>wFk$f zAKE$VSnF!9U);pSzRKU@#k=HF##`dGHdJ1YdwPWN--^9iS;ts^Sj)Esml+B5>^psR zo|p==neYo01sls}iQ59|7o41MEmdRQ``LO)afkFX`DRFd&zZ7}XHf&wd2PcP`#C}m z9&Y#R3d~X4XD#?(n~3O)Z=5fWa3*bNO*Kbiy^tTzNh+3QjRMx@>T%vg26$H`&q&K4G7iaIfi<^7Yh}eYf_@&v(C{=nA~# zsPT_IRe9KTb=rjdsYa)cRctlfW^L%hE&YjMxqCv{OyR{Pi_hJh@M!yFLnZUi3SXau zd_J~#rU>kzd%DGPPm#A?9 z&ez)}Fuq-p(cjgZmprvy&*7~ptLIs!!rZr;xy&kSFWoXfl`z@pqfW4eV$R_JWzp-~ z-gikX+9sUc_A$riZd!F(z#0jQ$;Wlmbl-bV*4S<-y6Wi+LBYu}&la?mMVhbGT`YA! zxLLDjeyiY>)t*5Mo}JyfEUY^&Iz9bvUG197nSs8K=D!PJ=e;Pva@QfC-Kj8Rk&sIE z?~J?r)$>Do=LhL!$Q{(kR^O@0$7abCk~}YaeS{{1iTL6p2Y#29yz~0oRc(Ce%FMv0 z+KaO8ZMtcnXeS^w%UWn--6?h7y&(lm7auK&>p1RWJL`VaymR>h79UsMUK8xQ?e(Rg zKaB>l&ks6o5c{)4{>s4vkw1StN@~#wik-(Ilwo=2;_{;6hUIB~(Y*?XUkJDAed69P z&))b|#Nv}B>;DHe-=?teJGjO1|B3L~r95knz^i=s>2Eog@|nD;xZ^m>^g55V-adWl z6E(*de%)>S$5CSP=lgj!73~w;%jEc^xMqDna+v3B@PhWM_fAh#XPJ59rpx=)r}6@v zD{fqOO8d8g`JbL%XuaM(-aT*696Otm|6*TOpVC2VCR^q%Ut577uK&l<5`yE+4t#UZ zJM#GD=9xWdw{IvexRMuqHtM0sE-tm%Ht$Zw`CYwe?|$PcPil(6x{ys*r#6@DReD&p zXxio?bD=d|Z=Gt^Mj89bu1R3F%So1Y6LeeB+_R2{=k>c;mYXiHhMVo$t1q+d&<^cC zs=*FMTBUJLjb~1k?l(NtBzVzLMr~6<>ik2?+!`*tzT-Rjp*EZ1=J`Av+v1zN_b%8z z@6q7}%nJTK-}av9m>43={$1c9C&RR#?CUR0b)R#vxAft{RR+iY<(055c&9mM&sCX~ zr*4_s`6&sW(l2bZ_1{W%rhC_FtE97@y2l zdwjcmYlDB_oQ-QYeY+57ro*hW%An5ckWznx`w?3eA*=R$$HS`}U#;=&w|DE*HQxGW zW9pGLf2wy)u$p%Eo2Q+OUYp35TlsPyOt#A?&HG!+uJWOvK#B9pOIs&BkIQSUwzoUU z6#WSK(3!GAU42KniO2Q5`{SQD7gRI^{zg@oP`f4wv@Bg~R zp8eR`^*@}{Wf@b|rU+ZaB)pNA`gyr3`o1S?&=q%=wHYqL;Wy{6?NENO%y?HMzx(t_ zr}WRXsO`1=Bh)4rvbD#Q_3dWghZc-K(_?jt(vGD~=iJP7cap87ZcXr!;#=++1v(p# zPv>78tuR=3^5e`6^?{u`td}ukzhhI)Uy=C}wfH}2o7AHKCObkni+n1Hv_S3Wm0)Y<#u{oN~ui)9Os#EK+)ty*!+ z{qyEs%rm}qF@4ywSf=FL?9H`Wn-zU$FF71#{-ERF-rqVwhcA|#Tt4%7w1vUy^n-fV zF3aDCln0f!F?Yjy+`@P1cYi`1a^3=hQ0?r*8MU(75!#!F{j) zFbXl)?OmvKLFC_J!^_X&R$PufIA^79_>bo~e{TfW&+1BgbV%)JRAx)-#!i_RQ}hJu zeU=pnB^Kp)eC*^>YHJe@Ykyg@yjY0uvSzD?jsJ4ff(M^pemd>D@58iZ-wNu_$F6do zVgGdI*V*h1!Y3zRPPr!X;<&6oYxw=JYKw-mwYs})${MUW_wg-|`mEmwuLj2JFlKtCqHFEZAo*ik=I zELS-0#}bZ+w7u$w|G&BrY@9Oxk^YsHA_1<~+4pzMIAh~Dv+mFN-mM~+xBL74y1PCq zY|h*XzrMa~;QGz)_o!Xq9pAKxldt(+UsxI!ZM5rCTY7fu{NDHTT6X+VO8s?c*GJ~< zb)E+z`!Cn+u+xt;@R@zif5GwBW|?fS`V3$BYyaK#LoP04ZmpNiU4y#v^bU8%aB)$8)S zq(snb-hHcQPo4yCfAoU;vx$9Ld#2(c9|z%c$62bPj1?OGOknu%fJgYhu8}L(ixZcx z$SwKD?JX2IX}{=_+288DZheeVJQXtg8XE`yUb89L3-}veyO_t8@>RWgu0=DaHh+mav9-M0bzx7I1>3siBrPJzGUR5_}>a222Tz~%j z_5IPSn5tX_o}BpN$F*_EC#QLL^WOVSJ7A{c6Roaq#Vpl$&DSKt?aHo~Qy2L$maoDPJbn$enz|*0O#+w**T^&f(dv)tgT~jk+_7DZ}b!n#08fM>?2P zN@h)s&^!3!g`WS_8z))|>}E$7Y_O@TI(qQM;TWmeK9ds~63(o?%_f@Ibd`6>d(L&{ zdFLKA23bb#`IDGpyrV95(wF>R)@q3(O)SdycuJF^xpqBoy2kM!z{FBg_3h&&x#5$4 zws}8FGZ1Lwy0I^geUYP4!=Hxxt7c_ZY_xv*I3eNFRJ}Wm4BLgH`;rttF-CE`VmT?r z{(ILA@yp*f8@Y?;7zuCUi>sOHvxWCs$q$SDA)DRXb#%@hJNx^^`=d6V2~Upx2YPhLi=h zDy$L4#^*TBeB3N5)csS}W9h>}BdJn8{|MK^8bNDo=IrTd`p|qYm(QAsisq0TQWzF-?Uu45lx zuSvBRy=W?K|Mb{Z=^YXa8=n}Sk}WHJEdFK=|J~Tq1s``j+EEvO>2{69h3eLWGMRd{ zoL^_kea$!?$?#p${f_yHdmg&~cHh-p_gZj<_l*fB@*mlYYlennOeY)Rm`(=G|LS{Z zrW7kgD}bh)gTRxz`p)^Kc_j*lNXsD%QP+_L6y>LsCZ`rDXoRE|7pE2_CYLCf=o#o4 zfLFmd7o{eaWaj6&B$lKqXt-Dz85mj^8W|fH7?>KF80Z>Us2dolgH?uPmgJ-=*tofZ zmbBP`SG9nqh%1Ux)3^*242`%zYhEB&!OYau*i<1+0WM}}ssNT!$b*Slnwwaniy0W0 zSzw4685v-RnV6bmh?$#Us53M)F-BKsXl!bTA!ccbq1VXB07J~u!q5bvx1=aBGbgnO zx?C?fvnmx73JMDPLHYS53ZO^;&) -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

- extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

- ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls. + FORWARD, + REWIND, + and + LOCATE + controls.

- e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

- synthesizers! + synthesizers!

- ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

- per - disk! + per + disk!

- ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

- rhythmic - value. + rhythmic + value.

- ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

- ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization.

- © - Optional - remote - control. + © + Optional + remote + control.

- Recording - a - Sequence + Recording + a + Sequence

- To - record - a - sequence, - simply - press - RECORD - and - PLAY, + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

- corrected! - (Timing - correction - may - be - adjusted - or - defeated). + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

- Any - additional - notes - played - will - be - added - into - the - track + Any + additional + notes + played + will + be + added + into + the + track - — - existing - notes - are - not - erased - while - recording! + — + existing + notes + are + not + erased + while + recording!

- FAST - FORWARD, - REWIND, - and - LOCATE - controls + FAST + FORWARD, + REWIND, + and + LOCATE + controls - may - be - used - at - any - time - to - quickly - access - any - location - in + may + be + used + at + any + time + to + quickly + access + any + location + in - your - sequence - for - spot-recording. - To - overdub - a - new - part, + your + sequence + for + spot-recording. + To + overdub + a + new + part, - select - a - different - track - and - start - recording—while - you + select + a + different + track + and + start + recording—while + you - record, - the - first - track - will - play - in - perfect - sync - (unless - you + record, + the + first + track + will + play + in + perfect + sync + (unless + you - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - including - pitch - bend, - modulation, - velocity, - aftertouch, + including + pitch + bend, + modulation, + velocity, + aftertouch, - sustain - pedal, - and - program - changes! + sustain + pedal, + and + program + changes!

- Editing + Editing

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - when - played - back, - it - will - be - gone. - Notes - may - also - be + when + played + back, + it + will + be + gone. + Notes + may + also + be

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

- Additional - Features + Additional + Features

- simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - find - the - desired - bar - number, - then - start - recording. + find + the + desired + bar + number, + then + start + recording.

- The - INSERT/COPY - function - allows - you - to - move - bars + The + INSERT/COPY + function + allows + you + to + move + bars - from - one - location - to - another—in - the - same - sequence - or - a + from + one + location + to + another—in + the + same + sequence + or + a - different - one. - For - example, - you - might - insert - a - copy - of - the + different + one. + For + example, + you + might + insert + a + copy + of + the - first - verse - between - the - second - chorus - and - the - bridge. + first + verse + between + the + second + chorus + and + the + bridge. - DELETE - BARS - operates - the - same - way - to - remove + DELETE + BARS + operates + the + same + way + to + remove - unwanted - sections, + unwanted + sections,

- Creating - a - Song + Creating + a + Song

- One - way - to - create - a - song - is - to - record - each - track - all - the + One + way + to + create + a + song + is + to + record + each + track + all + the - way - through - (up - to - 999 - bars). - Another - way - is - to - record + way + through + (up + to + 999 + bars). + Another + way + is + to + record - each - basic - section - (verse, - chorus, - etc.) - in - individual + each + basic + section + (verse, + chorus, + etc.) + in + individual - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - them - together. - CREATE - SONG - will - then - automatically + them + together. + CREATE + SONG + will + then + automatically - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

- Composition - Without - Compromise + Composition + Without + Compromise

- The - technology - you - use - should - never - be - so - complex - that + The + technology + you + use + should + never + be + so + complex + that - it - interferes - with - the - creative - process. - That’s - precisely - why + it + interferes + with + the + creative + process. + That’s + precisely + why - the - LinnSequencer - is - designed - to - let - you - compose, - record + the + LinnSequencer + is + designed + to + let + you + compose, + record - and - edit - while - devoting - your - undivided - attention - to - your + and + edit + while + devoting + your + undivided + attention + to + your - music. - See - your - Linn - dealer - today - for - a - demonstration! + music. + See + your + Linn + dealer + today + for + a + demonstration!

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

- HELP - button - displays - additional - explanations. + HELP + button + displays + additional + explanations.

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

- ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

- ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

- © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

- © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

- (even - drop - frame!) + (even + drop + frame!)

- ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

- on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

- linn + linn - Linn - Electronics, - Inc. + Linn + Electronics, + Inc.

- 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002.text__pdf__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..cb807c801d7651276e88678ddee0521150aa1cad GIT binary patch literal 11243 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}bF- zq-S8DU}gxS!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?M_HJdnG>$i z*2JD%Cu3c7cmMho@&PkHGf1#4(Z1VS&bGyP&!&sJ<8FA@@3lPQVYK?wDPHFZx7Hg? z_B>Yb=*Wcse@|Ss_4&O&<@x>pKffQ}|M%_5iK{r~vm@B01+ z^Gvs<|M|DR-Spv2`QH7!@^8zx{{3nC-{t;#W3}7g)X&xT|8M(W&;I?->b2+E(!c)u zcy?xbUVZ=VRxZnG_MFL^ri>C9`DOFl{_@3ISup11If^4gy-|JXlpNq+ZwX3(Rv z#;SES`{Td;c=5r9ap_5|lip%=m#3N32kzMwn8$xMab8i|+amktV#`ea&wUyDaR1Gn zdo3TlG~+A~=WclH@z;*6@7Q5(mK!tv-LHS>{d@Vs>6z71Pfj0jdssYsmTN^`uW#Sa zwLE!GYq*6v=0DO4+QD@HVg<9s^p9opz8Aey+#|cr=$5+9rY)u-`)*tGYOFi9KQUisUYmc;iaEjE z{GlEVU(a)K`#;oKde*MB%sM#8JmkDhk6-Ysq@%30zNvXWZms_~>&;D1W%xI}4Z5UT zcz{L<~mBvW+RCEq=^`(gL?&hgr|ojQT@Yg=D82W@@G$kkO{>e{Wb>O}d%@0GQU z7g#fFEPBK*Xj!gvp88Wj=u%Jcoa;ABqCNq>FTX=JvtW|ejo!>pzg?~fjlRVdB zzx(g{`*!L5xxnD8lYiAVxy*g1Tz&n6caQyCLyqkXHjO>8^{}v^WKO+LqH$ib`@_}? zy|RqHJv)LKW}eynLOuG9)-{3fmNt|A__CK(tu9hbJgyS+`t}_OOMR%w?Y%WP>tK!OQG8Q4ZdREq*zaie^Jj|bZgnelN`a_ zA6`zMaNmDIu71ucmu>9{s-_M18gtF%SaRNSgn3<7)3!0Vv%%ni;iJ?=7vzP`oLRB7 zb4iDyT2$iJvpjig8kDX1q_pRrSh~1y-{O4!{PTJBH)cvWFc|9ieGxt3wpdj1z7Zr-a>Ede!bM zL-+g3S4@vDm~$a~i$mf|ncXuCE(I2J2y9_Z{h$+Lx6*lK(4k|KPn>Z1dwWsKN3J<{ zEcg#F6r9oZSY5yD=1bm7y4f3r38%8VCJ6fU$s`sTi*VIT9hc2S81 zp>ijAI*J}v76pEr)}XyxZf|8E2ivXggYGH|H!YbbD#`!));_Mt*KIdV84~Ijskm%< zc-E+rQdmtPMK>R!@)e)svQeZ@JPmZt?3BzoMY_*y$?v50oh znIKf7v%NgepvB|VisnB@geQJ%Q!H9MbE}fU$ADl5sdvnR?GCa{;+g+!POJZITHgB9 zpr1=<^<1t?6Q3BWMZ$(+{bPqoAi&zNfL+@zJVRQ}_haIVYZ; zI6qNK@Mm{h&Lg!eTOKdrm>2d&!B}ALos2~H4?7R1=9KUrxuM3D=5hS{X~pd?tfp0- z(EQolTYi6g@HY{QHhYWI2T>FLPnhi!)4-U!rD@^^)(JnBL|3Qzb3LB>x$}sd->T)2 zUuSKNm41E9hFMB+&-Cfr{3Vu$Oc0xUB&K=qv9~JfRlG`?w!aN-h)MScp7dI6#?aXw z=y!-Kpzv4fwgp{Mc@ojLCTpkPWKf>8>|MjGEv8R4sO>X!^K9jmIwE~<)50fldwfNX zYbPH#JGbhg-(O|3mml*g%BQNby^p(E*559E;8(`ciCU^GQ|BFxD5wgH{&%FPspoiX zmBy}#*MBJarC-mlO^+{mSJ5&_BB-oNce7rQ$9$Dcg$kFGx|hycb++l7L}XNKt9`bq zV0p)~`ix5T`qbU#LVfX*zOSA;+fwPv0%--|jj2`>CElL+$MQw&?e)u|(fV7}INEey zZ}1a5_2#=}jFA4#KUT4~4ykNe-x_Wh=Jfl@Wz$J!t`4j2d|UQ1eu7{2uBE?{U*=RR zipYMJyA`(hoebaFlU;&IR&&0ef2n6O{j~(kwzoQYiWcEV9;JTaPIoJ3Zg@Nmo8N zzy9cX&X2{sie(GmMPK;%^nU59T$OH?#QMm#O_8E~ce8HQhCWj`nsKu_TI2G{3pSHn z=5oruoA#+S_1dP0)Ri-LZ&JSF=%5(+>T&kQgk5!4U5b`(m-J|M2;q(9Jf*#Ao^a6R z2J_BYB1u=bM+7~SIkI%&^xRrOr@tj%eog*3l_zJnnqOYPg(c6oi0f@UH)&N=-^0~j zvzAAmXSo@v?RoliOq~9wRb@F(w*S&GzI|niPF8TUN00HlfK9AN*}~!$%$d<=uyM~K zXXip!Q9hRf!OsOTEjg5dnA^LMnmo%Jx$T8z0$K$$S@$8UmziVx81o2;NRoHr9&oe!i!mvvw zEix{P_>`5p6zV?ge!4M}p{ev#QmXF>H!JfwqKP?z3EyO1sTCSW`tIMq)p|}{zj<%7 zmrG#8rYSpWxb8T;%5_vRNbxwYablWq-|UP0`7vRudiz=zInI@Orv7xbOdwO@yzS=` zqkYrr_MExiS}^Tx!gHOaYYUbgbqn^n6e3i4Z^<#H1B^f0EdR%RiL2A`o^f&JB@H3g zY_nyPw(dP;?8X=rD4X|v&8~@_8F4=%4_4n@aEf)kL2XOH+ogt&{%x)3;W%c0wrTp( zJ>Kt{O@lw_NIcpc8(F)*`$JycyiS1#pS24tEH$^t-Pyc-!n)XR8`iC!{j}vo^@&|f zC*$@t-d|`LAYR5`vcX03{LvRRUFJ6)r8&A*KV!RnD#Uh!HiOSPvkTEY##Z%p?|*9R zUXI!NTB=&J5R?rEZHTQj!Y z{8~3fbQY`D(?>JbDoi_cjJwk)CUT{-{{BrJ>NoE99WG+1Ql77r^i*t##Am3R( z?}sgzoqm7qt)sVBi?T%CQQ}$dvzu#sgX5#1S%se#vWHjeta-t)=*M#Yt-ISI|G(GR zeM49I=1&F5q`6y;XT_c|k&R3W?^Hb(ZyWRM%c(Sn#L!pc5dvU(1~I?mnYPbDSFVxl`Vlv-II9BbYq=uy6)JpI{jrY&1;gnn+?&EIk3 z)Ln^=`)^_yQXJLiI2f_q&}1<5j)+q{@+G^);aOMK*CnY34jc+~c&8sH^^t#133v5V zxx|78XB#DxmrPW9ruEwMgh26;r*AKu3Ar#Uym3d@%+1McA65tYxA3-HIkj3$M6In` z=f0eE_g*1JA^CD1;iub#jwzM({t8d?Y|vo+R+8DKsAp$s*Q9VSc;!sB8EjGUXX^KB z|5~(YR@t;B&#bJd7t`id89h)>&j{OIe92SE?Ek8LA5L^zu2Y#5Hr32VGVlAr=AWu^ zm(~TOvK0j-PiEOoWoJ;#( zf!nJNZt%Z6t|`ef0*FthpNnjdEt{an!Aab#=4+u5Il zH$Ul1y|ZW9gg$TK6_z{xZL^8|^{+TgA?rb718W8EP1R8G8;Jj30}N`%U!v(muRca{Ws;2*1QuR#OAIF)!pl!7SJ{=M)l3-rq#`2vKm)f*X+A{ z)x%W8b;E)O`}c~KaJ<(ukEl|PJ?9-;|Bd(H>S@|Se5-cUcldu+SFpROCwTgF(UwRB z#q3T0?}xCYi&nQDP)c?8@BVHuYY}JUTHhzlhum5^7umYLi=Fs?%AZxI+5~1VeA9kI zKVKogVBM|o=!2|TrY9>Ngtm)aFDu!8(IIajPf{gW=1pa} z^1;l@QBjIbtxB_C!{-Z~Tm>t7+(I<;Zs=^Q$W7YZux8IUk9!%ndZQ|CC^Mc<_{=qb zLmt=q$+d?!uL`>2cegpnZDb-KSg%@Rs4D9{w5`t+m(fB=@E`O%_Vy; z>p!1vP~D-gRTwbO<*HZ8w!oX`tkpd(PD?jbFFX0drnFtIF6LzcM|+^sPp#0m+A*2p z(-)tM1(*`gJ{fb*eA%mF*ajb&4r zAFTbVk@|Y^gc|;SFY^`bh2`zwoH2!vA0B^^NLaKpIhNVv-D)Scb(EXRXg!OWR8t` z;T^@+)w(UuggB<|>ta}$a%G+GN%nmGrKS4X!JFek%G%V9+*x80^00mS)x83f^xixX zdwzLGAtOMbq(a6|l~JG(XC1i137ab8l8 z9&~0mOYOT0>-N4pnIk%vv7XaNdHQoXpCc=muPtxkylDUYQOW}KRiCGylx-JLEtIO0 zSH?S;wWg;Y`b>Nca-S)7ZJB5?i zUHsDK#k9FpDRf3)bcj}kv$+1*4}nLnZG>g)?# z8n4U3Hq8^#{ub(Av%ZfZi#xa0rsIW>|F?HL3v;d|Cq8%(`s7@qMSnu0^z{vY3M7{; zyftOz#L1hF{dqmxGI>sQNZwn?P~BA`4^*@Hr*>c3%;v&tc*k7l#-)s{_cyqOGwE3N zrFGm9H@>>A;hRH*LRP}ojeb=jw>TAgR)uvR(06=y`}M-f@h9fJ3k^O#<3&>yfAq4s ze|^_>Ejyf~nZ0cxZ_yL6hySj7d5Y|RvvX;bCr8uggD1t7PhjFVd6fI;^qh>?>Y1hT ztXHf&4o=FxXe-6XtG{XSoYgY%SG)POC!|fUIR9L3*`-*uXyLF&Z$uU!`6R^=sJiJx zZ}t<;4Sb6YU%p;u`QXlt-$r)tPL$4G^zxxp*Rf@%5+prOpR;W@@OQ1hv*}|O6VGN9 zsYR0l*2OcWExzR8>w8jt&DpgJmRDZfa`1yo*P8_yI;&o&F+7+f`21G;^os|M?vdK@ zK*GyyZtAlgE_^$8UoC&RQn)%y{EqQ{4(mHnKZ-LO5{-AwoN`%DLtM+kReiGCbv}+- z>u(REv3&M$##N3vVYQdnrK@raF5Ont_T;vXo&4X*qB-hUqPYKWS~yv!jX$=V zAv$Qbo|1&LtMATD;y1 z-&4ER`D@SY($E_hKkQ3?GHv?ol>$$;FKM*X+P;bq2ZG8DJw2PSL%LVjJ>R6xWFQ2FXb9T59Pb;T_WNPpB^(vNcoZM%3mrQ(DXM2I~y4%c&98ptawk?|{ zo9r1kC!PI8)OzE#i#IDSUEY<}yKu>&Qob1)2JYAXSnW>jx79uJQevJFbAa08bEkz` z4crc@^RYToMKFb@Q)GluP!K}@a`{K2V*2#0vSnmW*yya;1 zwy|lRme(fbLrcB~9V|_(T*&v<%fsN)Nonu-*O;G9x$k=@B;?3Fb)~1zr6=4DFg~Kv zkp4DKoq6%%DV+jm7Pww{@p!ofQ;+dw>y2MFKVX-86zbu2`K*F+qDXUtakA!4f98qr z_}gBV7VrCSdf~I3W7#QQ zejZUpkHdD`4k$=&zy7JS;mH)EnasR0eUIWMeO|Oe*wO0Z>qv=rGoQa$IqCaexyuQ2 z>{b;u)>j-l8MH{CnWoveX6lLTTzZICByVltD$EIsNZ#!z`D$)Y)?GE~UX1S|Gi2(oS zT)+9pHWX!D{Ou(6D*MhZ7k`@x=b}Pi{N$0|tajD2RL^UDRzmloszZTq{~3K`Ws>pI zn9D8^cz5dyFJ;|NaxMl*Y{B`Nkqp;7^7@_qBTTES<9|O7P|5lGn18{-TPZ&ZJg*k1 zC*{c%E7l%ta_uYrGrhO+`fIDZyFL`JNPHruyCt>Bz5Z&&k!EkEv|_*9)2ojMXf9@1 zB)I*@C)v(`Ul)!Q9$C7hbZs@8!}`#&cE&xy+Y&Y&nN(1F+2H(Bk6q_v`GpT2*myo^ z;-}fNJMR40QnP?p^|Qd+Y{f&378l|ormz3tsM9e0uTlDReq~0@#rb(B9^EdNalb3A zy3DNk>COGqRqWZGyPUk{6W7`r;GjCqY{9C$AK#S^zY2c4*GDmH_CfJ8UnEKv=4^62 zIHzde4Z}WG=G)KbE)aL?|5tWd^l-<#?`N;7KPr`PU=Mf^a>_Z^a=(Fxfz?x$)sp#q zCFOIK?%HQsx_|G;JN15x->cJEyTiU_>a(WK+VeP{J;e7w;2CAE`r@jyw*}U(Kh7On zx1rDJb)MtX?dJNI%ePFwkh!O7%L&(A;<5FMF8b|oJE?tAh4Hc6F^iA@g-t7!7r%5h zl-hq=>fi$3PtWe0(%!dy^O19}?cze&yH0O`#G}f)L-%+-s|n(A^Gjv z-cxG?j;^(r_!!0^Hbu;)P=41j?*jQpx4l6WjoF!u|&LX`X&uSc=|D6(}c(^ZXqou|i`?+6D7I;@qIk9j5UfbSxzq}T%mAHHL z@}8n9@sOJ*SRXvq@Bibw;{6#9)4vlHLxVrofBbxWc{YoaYjywKzT-cX!;I#I)~Eb> z7r?6Jvg*$bUiZ|G6UDC@e@aRa?tM^ezB==>&~roI?4<9puSLK9x_5L|(UouF+X{b9 zb9^8EqO@nqY4Jx6hgYPFm&Pft?=W?q|3-Q)$C4cRC%dQ2<`w%dY^U>RXG^6D^RDL( zt7jIJ_#X2Y*N8TI70P?wWI>Tcc*BmF7mm9n*s;dMNLOx3333tLu4b;&a&B7Gfko3q z{_BNkZ|2b4b9PNp(pLUL3!YyA#!JqMX{jIlAYr@C?TXQh$e=y7ljk0k(>Qub&OZG0 z1kKx45v8SbYSxR-uUVLJCus9C%N0{PivR905-<=`{Jgs}wBzZaGXhcT@*`YZMCEeCxcux!l@F zLF=nQGaHR2Za#ngQPLsHSx+OT^nR&UX#H_}vY_)q-sY=cZ?-s^S$J_SUw6arWsY&W zwe14Ea&3`%nW&$Q5$0XhH@2C@y|h~IcVSOdz{3yM`U-5`yg%}CPU9bDo>OvKs_%JL z&I+=4UQ(`*b0)~D=PxI#lbim}zTZ|%j*o;xjNU)Ju_%OLuBxo!+Zq2>ym-&{r~RXY z#a(%pwf_s;KhBQ#`gi01g7dK}(v9Dm?4RW1#sA)k_1rP7*q*e&G&X}3Z+udIM1L_$ zob_uWbMxg(trA;UzH+p)OkCn6v$?G1km?=X_Z#9*DBQk&d+L-W$v>Z-$orIbK=9M= z^>xNwFVzZ+J6Ig`JviH+Z06iN%kS6Q4+WW*jiRosJ(*cES3@>$rQuH5!pYaCt>HY| z{nW8wp62V$JG;+^R=o=mJh;N})tsY?ZN2aR^L@#;!_w(PFx#piws&gJLp)4&yw}K0 zyUl$zkKuAh=7gwkiT2&ELuLo$Ded2}?9z|4oS@H_#WwJT?h|?BlX&UMwZ;1!G}8j- zb4kyeqBB9hPf(?JP1Szc@JCi_6}RlL^P72FN@>BjzpsvLwsQBo^l$&vFF7yGR^C$+ z73Sq!Z8Ph`#FF(X8c+YfsK1(deCo5;YO?A z*43beQDd82KN8fAu{Ay#t#@D+`;RjDj`Wi!{xU7T`9<_&8voSq&F)vt|L$N` zlZ%Qkb5?WSvY_E;&J8i!pL3obp80Q0Pu?-Zk327BX}V*i%fPF z`Zp}UY5m;B@#x_SXCqf8iKV-CRzA|+vtZ4ChLY(Ym%d_;$BpApIXpW`jCUTCu6qO7^W zAH^drj{4}ihQ{vL*5KIA=lAjH@^5Vsrf+)o$JEWbdv?!*>71!*7Q6~CnG({fE$gSI zId#09Iq&WDbE}^`V~qNv*Ol>C@JHj^&4%J94tzMI=5^ELKf^SZtL+%;MT~H)7XdBs z(D%+v0WHl0ttkltuiDUe&M(a?Q7}YW?_`L&P&A+@KczG|wMaoDB(=CWwJ0&UM8QPQ zK+gcY%+0wdHL)Z!KhGtxBvnDf#mdOQ(9*!v(9F=p)YQOO*T7uez(5_WG9sF@6uu%*;tG zg09C6&a6rWg@S^Deo%gXi2^7RzzcLhOH>uWaROeKTU?S@R00k(Ljy}gb1qd?SARDy E00VLr0{{R3 literal 0 HcmV?d00001 diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin similarity index 98% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin index d3c2e860..686fd1ac 100644 --- a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin @@ -103,7 +103,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE. © Will sync to standard LinnDrum or Linn 9000 sync tone. -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +® Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. * TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, (even drop frame!) @@ -115,9 +115,9 @@ on the TAP TEMPO button. ¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. ¢ Any TIME SIGNATURE may be used, and may be changed within a song. -linn -Linn Electronics, Inc. +nn +Linn Electronics, Inc. 18720 Oxnard Street, Tarzana, CA 91356 (818) 708-8131 TELEX #298949 LINN UR \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin similarity index 58% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin index a4a254f1..0971ea05 100644 --- a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin @@ -9,1054 +9,1054 @@ -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

- extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

- ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls. + FORWARD, + REWIND, + and + LOCATE + controls.

- e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

- synthesizers! + synthesizers!

- ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

- per - disk! + per + disk!

- ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

- rhythmic - value. + rhythmic + value.

- ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

- ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization.

- © - Optional - remote - control. + © + Optional + remote + control.

- Recording - a - Sequence + Recording + a + Sequence

- To - record - a - sequence, - simply - press - RECORD - and - PLAY, + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

- corrected! - (Timing - correction - may - be - adjusted - or - defeated). + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

- Any - additional - notes - played - will - be - added - into - the - track + Any + additional + notes + played + will + be + added + into + the + track - — - existing - notes - are - not - erased - while - recording! + — + existing + notes + are + not + erased + while + recording!

- FAST - FORWARD, - REWIND, - and - LOCATE - controls + FAST + FORWARD, + REWIND, + and + LOCATE + controls - may - be - used - at - any - time - to - quickly - access - any - location - in + may + be + used + at + any + time + to + quickly + access + any + location + in - your - sequence - for - spot-recording. - To - overdub - a - new - part, + your + sequence + for + spot-recording. + To + overdub + a + new + part, - select - a - different - track - and - start - recording—while - you + select + a + different + track + and + start + recording—while + you - record, - the - first - track - will - play - in - perfect - sync - (unless - you + record, + the + first + track + will + play + in + perfect + sync + (unless + you - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - including - pitch - bend, - modulation, - velocity, - aftertouch, + including + pitch + bend, + modulation, + velocity, + aftertouch, - sustain - pedal, - and - program - changes! + sustain + pedal, + and + program + changes!

- Editing + Editing

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - when - played - back, - it - will - be - gone. - Notes - may - also - be + when + played + back, + it + will + be + gone. + Notes + may + also + be

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

- Additional - Features + Additional + Features

- simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - find - the - desired - bar - number, - then - start - recording. + find + the + desired + bar + number, + then + start + recording.

- The - INSERT/COPY - function - allows - you - to - move - bars + The + INSERT/COPY + function + allows + you + to + move + bars - from - one - location - to - another—in - the - same - sequence - or - a + from + one + location + to + another—in + the + same + sequence + or + a - different - one. - For - example, - you - might - insert - a - copy - of - the + different + one. + For + example, + you + might + insert + a + copy + of + the - first - verse - between - the - second - chorus - and - the - bridge. + first + verse + between + the + second + chorus + and + the + bridge. - DELETE - BARS - operates - the - same - way - to - remove + DELETE + BARS + operates + the + same + way + to + remove - unwanted - sections, + unwanted + sections,

- Creating - a - Song + Creating + a + Song

- One - way - to - create - a - song - is - to - record - each - track - all - the + One + way + to + create + a + song + is + to + record + each + track + all + the - way - through - (up - to - 999 - bars). - Another - way - is - to - record + way + through + (up + to + 999 + bars). + Another + way + is + to + record - each - basic - section - (verse, - chorus, - etc.) - in - individual + each + basic + section + (verse, + chorus, + etc.) + in + individual - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - them - together. - CREATE - SONG - will - then - automatically + them + together. + CREATE + SONG + will + then + automatically - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

- Composition - Without - Compromise + Composition + Without + Compromise

- The - technology - you - use - should - never - be - so - complex - that + The + technology + you + use + should + never + be + so + complex + that - it - interferes - with - the - creative - process. - That’s - precisely - why + it + interferes + with + the + creative + process. + That’s + precisely + why - the - LinnSequencer - is - designed - to - let - you - compose, - record + the + LinnSequencer + is + designed + to + let + you + compose, + record - and - edit - while - devoting - your - undivided - attention - to - your + and + edit + while + devoting + your + undivided + attention + to + your - music. - See - your - Linn - dealer - today - for - a - demonstration! + music. + See + your + Linn + dealer + today + for + a + demonstration!

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

- HELP - button - displays - additional - explanations. + HELP + button + displays + additional + explanations.

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

- ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

- ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

- © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

- © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

- (even - drop - frame!) + (even + drop + frame!)

- ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

- on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

- linn + linn - Linn - Electronics, - Inc. + Linn + Electronics, + Inc.

- 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000002.ocr.png__000002__hocr__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..75cefdccafdac074ddc0199e567fd57fe6e3fe0f GIT binary patch literal 11513 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a?! zq-S8DU~UPb!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?M_HH~n-i|k zmc;JtQ`ueg^?vvl{*q4X$qfn%sx+PFHGE`=6k7M4=j4^^-O(;{6e6$ca`OMJdgd3Z zHo52NkxteBpLHX@r@i`LB)R|p-}m|dKW*=~um7B1|L^tLZ~s&7uPfQTZu^`|XLoP6 zudn~zzW?6Y-IcrV{rmGZExGdd=ll8pe|_seR*~+r;s2L!`T6z#cK$oOHvaed`aO}^ zd5!j>`=0w39*cYX{ZAYBcAxtOAI0vdhWmXj&iwQ3`k%SIg>`k;chC0yy!+he?@7D& zt=)h4hpgRg{@UaBPVI@`xLt!==d1cP=^KT{-%azg&c*M0qx*6CT~m9X_+`s~-~K*7 zH@W+@w6yS>tjyg}}K+4 z^bh9w5KuhFJ~uU9U+?9j%vG~q*)0i8G=6uw;*na+mnB6N%T7M4Fp6|4_&oph!3i&J zY8`#GaW|81AXhwlqKA3;!DQc?nbP;=^uI37YA*=Ld;FvN?Svl+t2!5cwTM`!%DB8g zb=ulpd0SfwvnLApZ#%e=$B)tQ^F7~yeJ?+*nXy|uyXWPe6Ekj~Pn=lzGv;>CuH}1= zBvj?@-}SG)`@*M(ni_R27S!ayhi{X@T*EEH}zoOfDDq-8)?hKv4O`p$7tYz6UtIY1~nGdHMMOa$2{NcWR8&)rrOW#|U zm1mOUaY7_l?a_^MUu^3=_TE*!Ey$g~9`}A%nEV52w`)=k|JyDJN=(SUWNP$S^w0A$ zKc2uf>-}c+EUwrjleoIy;FyV`|4OF`x#0(@y0$I1_Pmv6EAE@~>GsuyoUxLJZj@hs z%eMGRFY98#lk=x+41KG<_+Ry5qu}p;x6K6Ow69NJ{OZHGH=lTasb?&@yXCeQf7?sN z{Zs4MEL&Df@o&g~skrElzWek|!ZT`ob;?W=C+{z>f91$9S(&>_>$S(EOW(HKooW9h z^Y6je7n;A_fAi&~r-@$DLY<;FaklARZ#ka)Xfn3Ua@+I&*UvtHq-I!MhwJZ|csH(u_!5ePic_M^k3AB=||nygx3c&0N}Y{;a}v z*BCtXs)vVL~oEvbTMIbtuRlGQ{J}B& zbG7W*>6>@|-1XsGPZ6`Lk&1K2>l$Xk-D@{I6S*4lMdP}^d~iweLWJ!PZGr>(cr%{Pl=W-F~w{uyr-*>FWPFX4*ZyH8X3jF~Jt**V9 z{pPF_Wj@ig=z3v2r^32BWfrB=N*V>Gr|DZT=uY8g(8tJzZ@spu4AgX3FG~ zHhD>}`?sxM>2xpbqQw8xaaH!qx|N-j0u81ouCG&4`lG*a`*oHWyP7@qqakYfikpcg9)G4-^D|7@#3yJa@Vh2eJ=yz=%Hxs;acs}} z&4s7VRS6O8?5kR1 zR3(#iM)a}gwJ*|Ztn*@fBg-}~-ti`@x_QRt*3W+#CQWU)_*}7Gpuv1)#eLliY{3%K z9vFi&RZyu3U@n-l|-2 zu+K>+NJDB<%%aE5>eGK@M@*^g)wtNOb~%GyJ8$%z{O>CB3N_96|KHobq}t}pi<+#| zC0CXhZFBWuS@PS#i0gsS)UTTg&b^uty{Sg#)gRe!&u?B~Ok|0$>{z>N&yv`U1zYue z_U~{Ko_geyd5grFgtlU>zvcE98pR&{J*g+=;3XN86npQ~^`cF6@Amx`JAHe4R@w!> zsNa7t?fn{|nYLef_c2lVT+QHbf(^%?w^-=bJi3xE=vI4nk;_go!RxYi%Eg>!JC!c1 znR&(J@T(Y=KT_d>8YgzztAA*IKkMwI5{7Rl<0fp#O*nNvHvgjIj<}!Zms5P2{GTQsw=0k9ybAw%@P!Z?pZS`)1^27Js@i>z&??FK^~EKU=`{wLoWr#=GcBKMl@Q@hOX|j=eGM+11d~ zq4^X|}f7jaqBX&DPgmVz!;RqOWbwJ>$21PJyxy zy0t4>Y_43(Xs4RP0(r0!d`D3$Yf^iJbW9`c&bZ<$mv%%8^?pZnz<2dv)=`EhXB&4Rtgwd&y}?b#OW)mRoTM-`lW& zx!^0)CsV%ODdEP&&W|jG@`hef9g%{ZRo+;yVV-z;tC_t&=yZ-;1Z3CLOR zQhds#>|NP9!*dD_s`YzVUvnASKQ{lSe1MQ;)m{%@zn`u2=34hPgrr^kgomEJYK=xhm-#g*0?+fCfpPykC=+zFY?F%m`@W6c z$6h|%ck}3`7#Z&@jj3nW{cS1Ht8Iu#SGx6alUVK?r5URY6}6T9iT z`)SU6#}%tNmOIrt1_i%bs>adOq!a(N-z@D`Ny{bS3q@+1t9;*lUm`zIAm_jDtzw4V ze^2;q`NS|);2?{TN`HhoWB;v;>%TsQSI=92J8rV+w-_I0@4~8uB45|sbZ|)URlLBs zYK4mptNg4fEhf|N_xAm5eVExLTJ>76y{v5F42PW-Nk`Io=A`q?QT$u+mZ$&Jl@p3h z#|y4|Bn3pBoqqYVqqVHfH>P=Dg2!^@smUoY%G_(SGNyVnmt$BawsmIdEiKl7gl zyTRAu2Gwkz=vdo)`p-u8qrW=qf41sq8GXLNm~mdzw0%Q@W^T@u zTJha4HEt?d)}MN+)1OdZ{OHEf6o39n2W{n|Rch_^;XYS=}HO)C@)Ks*B>)|s)^8;HR zF%GxzUJ7g>bW1UxOfXT?>7*1T`jb{Wd75s zdrYo3@IH5belK@nS+w7yTN^dMt^I!O6_2!hwe<3U=*ZvYTFVQB%g+>X7tT$;SDZT` zh&NH;pv!&ho3-C!d8bPMeGpP(m2iwveb(Wv@FJn7vsiPzjDr4 zzKi{!Ei5EF$+>f?k9@=3vy-1^oR>TLq4*txM1ydC+spSwrz#8LW+koS_MK&7o^nU{ z)2g-eR2X+nY;szhXW|lYB>EWZ8qph5#2%Y%(r;XI@xaaJ9qIK;Zr*8Z(cU3oQ`G(R zaQ2p>cPef7@?X3Ns{Njz!7p=7+pb=5`L7E86Gu89r5AqOF})>gXV4d`KtumezG4w8 zotHG8`1yLt(*p|AzJ5_|J2&fa#e}B^0~QxY3S>1%d^@%J%oO#OXF^1uAz zN%JdgSEd3Op zWOH8egzvsYwbJg}Z|0^iX>qu%+idVpytIM4{PbaQgAKKl&UHO1n#eg>t7Abn^Rn=M zu_VWwnAVvlt9)Y$@5*%QUbI}UCc9w=|Kd}1d~cT4vPza%)*ow{|Blx>`m>$Yn)-G| z#cm;i$O$bU7#SD75!Y8ZHbpVn;B-%aoQk*mNyU3d?-zurJGBJ0uX4~iy+rGB&k4t_ zz?{V&rn#NfE4=t8BCo&ET=nMurxrVocsrl5`Ppal`Ki-%haZV~OS#@ums-DNdG>$% ze-S5Tz2rBh*4bZ{%&y$IB9ZOa760z7tZT)LDg=M?Ecor*?Bq9V=FiVsp4X%fuF7%E z=Vz9_)+n<%_n*fvO>?cr_nM!Y>tmNOvMSHa3ssR(tXyW+$91K1_EDvW8z#3l@0OB! zX_=f>a$i#{XQDt2-_3lGF!ElwOo>K_GIRAZwn6$Y4~BBJne}2x3>%Y84TKZ zoAV~@SUyw6)jhPzr*JlACtFulTfftJN;q z_*W-hM?C+w#9PXSW7-3`<5sPv<{Q6FSg~cp{!BLZYKIv@jCH5oawFa9UwR&?lbba; zl;>CJ28E8m#)iaq0;`rYUDP>ZeMFnkNVD7LES&J?CxjvYhqvK0cVF zwOfR5!pC1H<-EUtJS|ir(Xk--(MyBgBO1nM{(gH^Y3bC}5WVSE>C1f%CBA99t|#1` zUhF>0T<_~LpJ~o4)eV-bPU-l?$yRtX&anT+_jl2SNM5xWCOzg1-j4lUfpZgjrhR#oav=uw_ldMP1$jOO0cSc+O$P(3j#w|ukthdE&C8sf#}^s4=@tLQhU7$vUvD3;n0bj? z#aVCGnY`-X;l;X|uZ@r2HF^-%H(~vDowkx!&)7uw=$~iVd~Qqlm*tcGZB?H6x?uj5 z!hcoQ7Njk7v*UXEPR`Qgl{gb`jO4THA!a2jyLGk;OkX8k#5rZAa^P|!`5e{ub2FMFUe6@7?!Pa?C5gTUhhmJU4M2o93Ro(v2ESU)`QM+G<=3Dt7Bj z-~Quh?%6#yrsvP_wrdv{>*xJ@!WjBo+{9^OvqJpr1hLl#eLn`cYV3U&l(6_mzS2UC z!yY>8pKJO(IBEKU)7a?Q&0waQJBlxU&L}b1t?j`aEwF2$#l(5{`sSPT&!|qxlrC)w9=lHbdl9YqbmaV)wTkDza z6w7p-k_T}W&9b~)!G*RvPF-}G#G7n?mf7+C#P`d5Tv?58NCz+3B6Yk(@9wYM>ibLr z68CJ5?dzS#q_eiM>g}Ed>-;#9*72;W4=D@SHtnLg`NTIowkg>nA#G7hmmXGJSmUsJ z-lgtAWNB`R8 zmOVdHJ8QHJ9ai5tA%54%ZuhGBg_l^@%~p6D%FE{vowa3o(bVROh&4@t2VyVn($`9G zu~}*>RO2A9Xm&Bv6tP1hk|N4`);XR}H5TjC3NzGsv&{0WW18^G1xsJ13b{;KIQ#gn z=}Tutls{Phc*AjLuTtF;{l?Dj&u6ADWVx-9aAcF;MAkLGQ*=WUo(D55lSPlM&yJT<-EK24dq)=qrQIc~G}nn|ZObJ)A+ip+4mHRWvg!Ub$1 zrXN^erazy$`c3fi<(#(56OAUsS%03e(21?>8Iv~;uaVaDtNzLnu%x^8Vch6buTd+xF^`HN< zrYur>bzq6e_K<&pujkL3lC)s+fmP?9OD;@%W$;)~bJi_6SEg*$30%^)OwWQH+vc-> z4E{BB;+KV$OACG5O+I*?a6b0ZeAVin=fCtgOy*hdo-=>LX`jQn%175$S_aje(f95C z%QTH=hpO!&)+5*7o_mn$yy(5s)5*4Hn1a`+wR-*e!Ra!$YyFwL{J1AuCAS{E+~MfT zUAw0LP5oCn?wN-cE?zp#*>dl(MXfK=$`AjT#rEsvnje~Hc^yT4lpo!nTs7@cp$1=O z=@S9_`wtIx<}6V78XWd8et!PKnt$T@3Oi~7nWHQ#R+-jq(% z-4N_{**wcg%P%1I8DG{iPQJ_2)#ujV)4tnwra0=j=#}iXZ2rw40Mo{J2g)s;knG`g(st)y*m*sd#Szm#wE5M zwL1aHll4r+B=%2|ec7%z??LtU-ScD5W{dMP&G;jI^ou$(r>(80&#i@v+Yc$8d+RT= zaLwJ1hCcN?<)zGfRb3|Q@$Q*cF>OaL3C&oYTXKF_L+~S8=T1w+`W*Y z%-_G^t?TApdmkUGy86re%)vADr($BBoLwWNTv5u{uA1eta018vyAM8QsZAG~VDGG1P>*F;nv14(w%ihFv_`K2ai@(Y6 zlqGl9rm+26GcrUcE%6U+Pd&xgmymLRqoV0$!EQfE{zoM@9&UNN^kTHl(aHJu6Xf^t z2Ax-Qytnh1r)#_9@{i(z7k2Flf6vyewsKG7?XZAjO?#dwYrbmUI>l(Ji9_?VcU5}Q zCA!ngw|vQ6Rw};Y`^4RU7ITXw-1xhsJatBU$j>y{+ z3!i2&bytt=;-4IavzeWrM*Y|-S-mnaVS=AermX)#aY*Td(+C1A;?Il}sO60coe#QNnoJtOo{|*YXCGBRrdL_z5 zitTUP(w&Ri)(BKwH~pPPRRZq~7=%=?=G>hexwO`-rVYv3kR#{FF54*QZkIUIT zy#CZ`+BJCzhxd)2AGE&P`KE5?-sDF=gRDY-As-f+R=d}`YV2|0z5ThbcaZa&Xou@-L9&{=o&Tp;TY_QyO- zHKFs|XEOyWICICJbvyRTa8r?R)YHPbGt6(q`O4W($dYJW{ch(%u}F*WtMjf;63ml# zpX&eim60m@jA{4Y%DuNglVY+;Bb^X>8Shgmb*~+l|;Fk+~_oUFR(s}wHHV(l_ zH~!yv%aVDhDzE3jOBS8Cl@81IMBi*ZU2juspmy(C^X%gbL$&2UGii&j)_d@+Lcd~P zbK`Eelaqn?2Ih%&`<_U%&td*1vS73Eg}%WIb6;;R>pjWRW^-aKJ%qtS zggW}S3u%j#eq}h)y-Q2Ii>r-Oa&myK9nZPIz0B9u7N0SC_k8i|(5Mv>%N&ds7rrT5 z?9J59?5$kRTYWoECgq!!(IlSh5BZY1#7wp;Fm9^%QeXIqb7I*J+XXW2e+2Uyv4Tu zP1pSyb+=!i_$TJ7Y})SY2aZ2@q>yPY)Uf1`z`-3k%l$OJFF3P(TlE2j=N=O}GC$0{ zxrA|cNRsH5>v>@-TEd#F%FN}%7xntaPX7I3hPme1fUS4UPvkSnOJB6w7sh|z`H)2T z@-T<;1t}b2|2z^@OpKnL3jK4s|7^xdud*2P|Bl&jqiS8|7TgOCTt0aRXHBT$PvLKl zj?*@lzAi~xSG(ce%h-Q1cT0^=Kf5Fx({xt*q7Tq z@+*pH^jdOm7Z=OJ8?vt*ZqHoORjlDLZ>q}2`D=atN1s^!@usk#;f2TV0%p=h@h&?E9iJDb^&+v0uj|?0si{b1g&e1aFhpAYWr(_LHlQd! zr8GIUNI@ecwYWI7C^5N2!9>qM&j7sW&$%cyu_QA;&n2-WRYAkW%E-Xb(!kWv%+SQt z)WB5Nz+BzHKpm_yB(o$ZRl&y16|{m6u{5TlC^e1CK*7+63$(-!f)&h6O^r(8UZ54a|}3Eh$RO%t%@me>W}wt$j|0 literal 0 HcmV?d00001 diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..d80b111f --- /dev/null +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,128 @@ +2A NNI‘I 6F6867# XATALL IE18-80L (818) + +9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI +“Uy ‘soTUOMOI,q UUrT + +uut] + +“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV +“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e + +‘uonng OdNAL dV L 9) uO + +sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e + +(jouer doup u3a9) + +“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « +‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e + +"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © + +“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML + +"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV + +SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « +“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON + +‘suoneurldxa peuoyippe sdeydsip uowng g1TqH + +oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « + +jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL +INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue +p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) +Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT +yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy + +ISTUMOIAUIO?) NOAA UOHISOdWIO) + +"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd +uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo +ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} +,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes +JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes +Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM + +dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, + +SUOS & SUTVAID + +*suoT}oes poJUBMUN + +SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG + +“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI + +ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP + +B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] +$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL + +‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy + +0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns + +sainjeay [PUOHIPPY + +‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} +-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe +aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM +—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy +ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL + +sunipa + +jsesdueyo ureisoid pue ‘fepod ureysns +‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour +pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen +Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN +NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar +NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas +*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k +UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE +SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd +{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— +yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy + +*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 + +2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA + +‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor + +§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 +AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, + +g0uaNbas & SUIP10I0y] + +‘JONWOD s}JouNaI TeuONdGO e + +"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e + +‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e + +‘onqea ory AY + +pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e +‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e +‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e + +i ASIP Jed + +S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN + +jSIOZISOUJUAS + +stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq +ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e + +‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA +LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ +LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO +St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay + +JOps1odady soUINbIS [GTI YVAL ZE +Jgouanbaguury oy + \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/pdf.bin deleted file mode 100644 index e308fb4c54605719eb771b48594c0e24e4f76924..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 10249 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a?! zq-S8DU~UPb!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?N12Mi`_tqkB?)C6D@)mvbrtz{c=FL5>u$iG;EM>}qblVp^pTAB_nHV$W$#KiOju&?9 zxEgwUo6pCMJ3LO+SFgRc*XLjSt^WQ0Kd)b3|NHUz`}IGrzyDu%^Vj+}rz3Z-xvky$ z;n&~m*X{p4J-cV-bmR5+>nkIli+}(1*M5C{{ol=oh7z0ee)|8Je!YHA-T&9xJ5u96 z+t&TRemZKi#Cdz(_dnCC|9!6f)44Au&1ch3f17_h|C#^#zV7a){_C^1F1+*Y?95c& zr}ws~L{Cq7l%#Bb;<aGWPu~)2H9#b9XKjKmYuF;{R=X z+;3;!t*xuPd0;Qs>0_$LzSgk6-PW>go~`b=-TUj=lcQ|jHl2^&aq98uzmt}`w{P0( zD^m6`&Sm4RZ?pF-c`W`@_M@Aoy-91bb@1egzb&3@TX)=Ez3%$A`)?F4i)3hjF55ks zb&GAp=~|!L{P*J&Ob$-@|IbtZt@vU6$*EEwQfDoAvC~X>MzmywPd^f*61DO zJ6UtTi+#;||MrrB`MSHgQSv9{>KZ35uRGefHTC+kaDBzg73{tRw}r0%>VOFMe*h@_ySNf!qF;#-C$qkF?n- zz5U3OWxFE(*QpCm_F1&!-d7iM*2^jHWv(0Sn6I^MTLqU%aE_od#}S>`bqiD{cC=Y@ z9=*LnKYWLR^2P%@m<*#!wM4d>UG!tK{kc)PL7>>>--K}S@(;F=MSWB6>nQAH;^A7S zQ5~Z<-J9wCGx`vII^Q-YPDXxPCH$l@*yH;f6TPd zlP^!-YEH`9{z85iTbZ|6%KZKZ>vVU-MH@VPz9S(j|6s`Tb0;Gk-5*R{ePRDBmz>EO zE8FfLnVsGo!)syhq&0s*N4&46ZN)C(Co@{Nik&&V<6_v9y{Xz81fJaTH9N@9c1_Ol zXXNoo4NYs38jkWCC_J0SVja=(P5I1?i0tjU@v7xtem?wp^QZswYsE`?7k0;nOH7OL z5m_)LymkAFr?K~U{|wn@Q*qF5c9rJTbvGLP3#(^X89goXn4R>QRd-n@SNF0f>gyFs zoh02m^8eQ}bA&No?Ra>3-;_6k<*wGd{myT?I(hzyZ7&X3&tG}+RCU(js*egglailL z{rK6=C8_Ta>z>r(+z));yYX%>+$MdhPBh7P1GD2EefP{%i`o=Vz8aH`HuJ{j#)T%u z=LIjWUw+KEKF^fn$JC17rq7$NORig=bD;3@lJJ+aPfZY;BvTd7T05?AD|dg~f`0F81ZVKi$}xgw(byQF>!>v&{Ew zxPgh;s=3OmPETS{)$+UYv?IqpW4`k5+izTLlD0WK5VY-hA{2dY@}hs0u@O5{T&rVl zuFrEjDz{9DW0ocl-iZt2`_=aDLGfh__tEc|nok!b{H8f2+%X&si;fa>Ez5x{G}ZbIg+_S!yw8tTnqC z&{(Zr!ob0((IitXZ1-}rt!~L|OVd-6qW3?_OMiB8XP#$+T7z5u^h>v_xp^&P!n>u< z7_Z1MKY7_oM)a4#lHhxV&OdxxJT`7Sede^<{%*ew?EPFK=L(NXO!C^d=*qhOpA+1s zF7ej=sdDEwlcv?QElhRQ41Z^3&zR=$W7^IMv(z7!zA>A{vfKE1>y!&Y@{Q|0h+7K# zEIc0){lZ0pwNGtV;(Q5%ExUKUITvu(WA7T48(UuO{&TaW{P>hdU!N6)oRV|QX!3jh zW3I#Vq9p!lQc{g=EiCR@B?lR_HeJ7WNX0qt)v-7Ffs8fWidt*$UyZx^Lk0bi%_*6_*yj=#_Kj>}#ACrhLF7Z^|vEJ+n{E+|Aqb zboz$+TcNw-5C6MrqWdS|{|T*Pv+fxUM%kBnPCF&(OC;zwSpJFlbJSn`?TlAWovP~k zPTN1e*mGd_lRsTDNo@;`^=^@?tbF!tkBf}!#U&FDJ~@6P_Q>6rvCPHpZ|3}%##3x~ z-HtmgPq-|AS?cLlgYp-9H7j{NEv>)gxM*EfvDL9Xv@cPsR48Th@BYaAKPN&`rt~ga z!+%HWh}v}I(Vhn#`NMv-8-#mNgiYagnYOx^S>S^j|0(tCAA zw_**~vhnd&n(m8CoccmmYxOOObxSTyPGLEe^u|^BQqHNaGoK&ywEVe!nWI@DIy&96 z?%u-r3}=1Sxn4`Y?U^Rf(CB3~G3d@pmpISBB?^V`R9g^3UN zjCn#X7H_zh{wl+_(DeTGDNn^P2MoWb;r)W zFRu@8V~^Wb?lfg3%MaUS%~6U5rpw+Y9^0b&(9~A$m2f|c{I}9)i>eO3|1WTmJM4C- zL+Dr4FgZS_H5tK%ei2W0e?Rxae95ANA8tK+amC;?TgcbUX6N_TO;22ZTOFINaP@R$c6%+$pLp6@*J0|h z1$*;u-Hv;)>u}bhZ!#=>OZI5bO8IzPi_P=DXxBo1#^6b4tcD z1^$LTS*nXS&(%s)%zf*1j59PNmW$f~q5^rGhKRB-K(f4IDzt8he@P9UkMa1KjS7CmD;)AX#_JgMEr}RBIU#36bt-Slu*6gPT%eS?C)i@jLAfG)~_UUoa`jQ!*( z+hs1iz5M)pPs{yD;HlSc93c7 zyf3yMIA}xs$&!q`oSf|Td1v%`PMIB( z3hLM3{`T7Cqrdgx1&f2d+?TR13M=GHnEEc~1?vpcutY=E9sjPE{|#}RF!QYC{+C4u zPuFRkUwi+Ch*6RKq=gm?55v!f1gPjvX4B+bS98z1)uB{avFZ4)mD27-%r>rbUhp-= z-c30s#i@9##j$A9wgn$1?tPe`v%TfcQ|_B@873v%IKAO+iRVkb+KX=b*QH;aabj-! z=4!0fzDMcH(&umZmpWd!HJkIgGUI&dM5pI=hMH@m5ApGt|I#U(Vz%IYI+tLTc!;Ho zcuMP)I~yi$J2mmwS?9-3=kE~AyAaFj7`s7r!8fkdmqz;%`6^45N=_~|IK_M9vN867yZ$@y5`jl6-ACqLVx}mRrD{E ze#7?Ips4uPp|Vpg0Xmm#!cMXCo1IwrV7~IS3LE!{4V>-oOnky3ggjXo%EL;~AFtI1Voqu-Mf{cIt ztc6RL+w{LVc_RTCzg2d%!q;Je)0 zzpWTCY;n#A2k)yrSW9hPl9sZAZ1eA2-E?|J-)Gujte36I*yp`b!x! zmL1#wyHreZ&&3)ce>;zs35FLGH>@e)^fQ^ZkJD{3|7q9An}!Qk2wrCS=6Cj>N9kJU zUF~n*nAHDVpl+31kgv7rP1ihCM|OR|ceA$rv#5AAxoXM9y^5QwmZ&{vyXRO^V|3cm zab=g2Cud6c>SHo$|39at@pN8f_2zkVOk?Y=RS|uR4%_)pcOBZYpyJkz_J99Z?LAVo zc|vJdt>x-jVe!+XUTYgUZx zE%*6dm){0VO_}q!I>)TjXm0;ZzJ?j8lXSAbd_Frz;rB|JnOmZ5&fjW$vgSyjR+%z7-a*{#9$- zeej*N=jPIHl75+S5A@_8?Otc&RqSg#?GO7=o$asmX1>jc-;?m^`Mu5O<(HYQ{(LWg ziILh$!#$f7EH9fj=pTvYOW8X?mH+J7@=nq1uG?6$oqAUOnR0xxw~WE2iGL2sZC&|Z zyG%Y|Z_KF&|5_#*zny*5@0N?+&rRwl4|_~sD(d_9jFw9K=8evuCvR5XcJ8G>wFk$f zAKE$VSnF!9U);pSzRKU@#k=HF##`dGHdJ1YdwPWN--^9iS;ts^Sj)Esml+B5>^psR zo|p==neYo01sls}iQ59|7o41MEmdRQ``LO)afkFX`DRFd&zZ7}XHf&wd2PcP`#C}m z9&Y#R3d~X4XD#?(n~3O)Z=5fWa3*bNO*Kbiy^tTzNh+3QjRMx@>T%vg26$H`&q&K4G7iaIfi<^7Yh}eYf_@&v(C{=nA~# zsPT_IRe9KTb=rjdsYa)cRctlfW^L%hE&YjMxqCv{OyR{Pi_hJh@M!yFLnZUi3SXau zd_J~#rU>kzd%DGPPm#A?9 z&ez)}Fuq-p(cjgZmprvy&*7~ptLIs!!rZr;xy&kSFWoXfl`z@pqfW4eV$R_JWzp-~ z-gikX+9sUc_A$riZd!F(z#0jQ$;Wlmbl-bV*4S<-y6Wi+LBYu}&la?mMVhbGT`YA! zxLLDjeyiY>)t*5Mo}JyfEUY^&Iz9bvUG197nSs8K=D!PJ=e;Pva@QfC-Kj8Rk&sIE z?~J?r)$>Do=LhL!$Q{(kR^O@0$7abCk~}YaeS{{1iTL6p2Y#29yz~0oRc(Ce%FMv0 z+KaO8ZMtcnXeS^w%UWn--6?h7y&(lm7auK&>p1RWJL`VaymR>h79UsMUK8xQ?e(Rg zKaB>l&ks6o5c{)4{>s4vkw1StN@~#wik-(Ilwo=2;_{;6hUIB~(Y*?XUkJDAed69P z&))b|#Nv}B>;DHe-=?teJGjO1|B3L~r95knz^i=s>2Eog@|nD;xZ^m>^g55V-adWl z6E(*de%)>S$5CSP=lgj!73~w;%jEc^xMqDna+v3B@PhWM_fAh#XPJ59rpx=)r}6@v zD{fqOO8d8g`JbL%XuaM(-aT*696Otm|6*TOpVC2VCR^q%Ut577uK&l<5`yE+4t#UZ zJM#GD=9xWdw{IvexRMuqHtM0sE-tm%Ht$Zw`CYwe?|$PcPil(6x{ys*r#6@DReD&p zXxio?bD=d|Z=Gt^Mj89bu1R3F%So1Y6LeeB+_R2{=k>c;mYXiHhMVo$t1q+d&<^cC zs=*FMTBUJLjb~1k?l(NtBzVzLMr~6<>ik2?+!`*tzT-Rjp*EZ1=J`Av+v1zN_b%8z z@6q7}%nJTK-}av9m>43={$1c9C&RR#?CUR0b)R#vxAft{RR+iY<(055c&9mM&sCX~ zr*4_s`6&sW(l2bZ_1{W%rhC_FtE97@y2l zdwjcmYlDB_oQ-QYeY+57ro*hW%An5ckWznx`w?3eA*=R$$HS`}U#;=&w|DE*HQxGW zW9pGLf2wy)u$p%Eo2Q+OUYp35TlsPyOt#A?&HG!+uJWOvK#B9pOIs&BkIQSUwzoUU z6#WSK(3!GAU42KniO2Q5`{SQD7gRI^{zg@oP`f4wv@Bg~R zp8eR`^*@}{Wf@b|rU+ZaB)pNA`gyr3`o1S?&=q%=wHYqL;Wy{6?NENO%y?HMzx(t_ zr}WRXsO`1=Bh)4rvbD#Q_3dWghZc-K(_?jt(vGD~=iJP7cap87ZcXr!;#=++1v(p# zPv>78tuR=3^5e`6^?{u`td}ukzhhI)Uy=C}wfH}2o7AHKCObkni+n1Hv_S3Wm0)Y<#u{oN~ui)9Os#EK+)ty*!+ z{qyEs%rm}qF@4ywSf=FL?9H`Wn-zU$FF71#{-ERF-rqVwhcA|#Tt4%7w1vUy^n-fV zF3aDCln0f!F?Yjy+`@P1cYi`1a^3=hQ0?r*8MU(75!#!F{j) zFbXl)?OmvKLFC_J!^_X&R$PufIA^79_>bo~e{TfW&+1BgbV%)JRAx)-#!i_RQ}hJu zeU=pnB^Kp)eC*^>YHJe@Ykyg@yjY0uvSzD?jsJ4ff(M^pemd>D@58iZ-wNu_$F6do zVgGdI*V*h1!Y3zRPPr!X;<&6oYxw=JYKw-mwYs})${MUW_wg-|`mEmwuLj2JFlKtCqHFEZAo*ik=I zELS-0#}bZ+w7u$w|G&BrY@9Oxk^YsHA_1<~+4pzMIAh~Dv+mFN-mM~+xBL74y1PCq zY|h*XzrMa~;QGz)_o!Xq9pAKxldt(+UsxI!ZM5rCTY7fu{NDHTT6X+VO8s?c*GJ~< zb)E+z`!Cn+u+xt;@R@zif5GwBW|?fS`V3$BYyaK#LoP04ZmpNiU4y#v^bU8%aB)$8)S zq(snb-hHcQPo4yCfAoU;vx$9Ld#2(c9|z%c$62bPj1?OGOknu%fJgYhu8}L(ixZcx z$SwKD?JX2IX}{=_+288DZheeVJQXtg8XE`yUb89L3-}veyO_t8@>RWgu0=DaHh+mav9-M0bzx7I1>3siBrPJzGUR5_}>a222Tz~%j z_5IPSn5tX_o}BpN$F*_EC#QLL^WOVSJ7A{c6Roaq#Vpl$&DSKt?aHo~Qy2L$maoDPJbn$enz|*0O#+w**T^&f(dv)tgT~jk+_7DZ}b!n#08fM>?2P zN@h)s&^!3!g`WS_8z))|>}E$7Y_O@TI(qQM;TWmeK9ds~63(o?%_f@Ibd`6>d(L&{ zdFLKA23bb#`IDGpyrV95(wF>R)@q3(O)SdycuJF^xpqBoy2kM!z{FBg_3h&&x#5$4 zws}8FGZ1Lwy0I^geUYP4!=Hxxt7c_ZY_xv*I3eNFRJ}Wm4BLgH`;rttF-CE`VmT?r z{(ILA@yp*f8@Y?;7zuCUi>sOHvxWCs$q$SDA)DRXb#%@hJNx^^`=d6V2~Upx2YPhLi=h zDy$L4#^*TBeB3N5)csS}W9h>}BdJn8{|MK^8bNDo=IrTd`p|qYm(QAsisq0TQWzF-?Uu45lx zuSvBRy=W?K|Mb{Z=^YXa8=n}Sk}WHJEdFK=|J~Tq1s``j+EEvO>2{69h3eLWGMRd{ zoL^_kea$!?$?#p${f_yHdmg&~cHh-p_gZj<_l*fB@*mlYYlennOeY)Rm`(=G|LS{Z zrW7kgD}bh)gTRxz`p)^Kc_j*lNXsD%QP+_L6y>LsCZ`rDXoRE|7pE2_CYLCf=o#o4 zfLFmd7o{eaWaj6&B$lKqXt-Dz85mj^8W|fH7?>KF80Z>Us2dolgH?uPmgJ-=*tofZ zmbBP`SG9nqh%1Ux)3^*242`%zYhEB&!OYau*i<1+0WM}}ssNT!$b*Slnwwaniy0W0 zSzw4685v-RnV6bmh?$#Us53M)F-BKsXl!bTA!ccbq1VXB07J~u!q5bvx1=aBGbgnO zx?C?fvnmx73JMDPLHYS53ZO^;&) -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

- extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

- ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls. + FORWARD, + REWIND, + and + LOCATE + controls.

- e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

- synthesizers! + synthesizers!

- ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

- per - disk! + per + disk!

- ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

- rhythmic - value. + rhythmic + value.

- ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

- ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization.

- © - Optional - remote - control. + © + Optional + remote + control.

- Recording - a - Sequence + Recording + a + Sequence

- To - record - a - sequence, - simply - press - RECORD - and - PLAY, + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

- corrected! - (Timing - correction - may - be - adjusted - or - defeated). + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

- Any - additional - notes - played - will - be - added - into - the - track + Any + additional + notes + played + will + be + added + into + the + track - — - existing - notes - are - not - erased - while - recording! + — + existing + notes + are + not + erased + while + recording!

- FAST - FORWARD, - REWIND, - and - LOCATE - controls + FAST + FORWARD, + REWIND, + and + LOCATE + controls - may - be - used - at - any - time - to - quickly - access - any - location - in + may + be + used + at + any + time + to + quickly + access + any + location + in - your - sequence - for - spot-recording. - To - overdub - a - new - part, + your + sequence + for + spot-recording. + To + overdub + a + new + part, - select - a - different - track - and - start - recording—while - you + select + a + different + track + and + start + recording—while + you - record, - the - first - track - will - play - in - perfect - sync - (unless - you + record, + the + first + track + will + play + in + perfect + sync + (unless + you - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - including - pitch - bend, - modulation, - velocity, - aftertouch, + including + pitch + bend, + modulation, + velocity, + aftertouch, - sustain - pedal, - and - program - changes! + sustain + pedal, + and + program + changes!

- Editing + Editing

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - when - played - back, - it - will - be - gone. - Notes - may - also - be + when + played + back, + it + will + be + gone. + Notes + may + also + be

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

- Additional - Features + Additional + Features

- simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - find - the - desired - bar - number, - then - start - recording. + find + the + desired + bar + number, + then + start + recording.

- The - INSERT/COPY - function - allows - you - to - move - bars + The + INSERT/COPY + function + allows + you + to + move + bars - from - one - location - to - another—in - the - same - sequence - or - a + from + one + location + to + another—in + the + same + sequence + or + a - different - one. - For - example, - you - might - insert - a - copy - of - the + different + one. + For + example, + you + might + insert + a + copy + of + the - first - verse - between - the - second - chorus - and - the - bridge. + first + verse + between + the + second + chorus + and + the + bridge. - DELETE - BARS - operates - the - same - way - to - remove + DELETE + BARS + operates + the + same + way + to + remove - unwanted - sections, + unwanted + sections,

- Creating - a - Song + Creating + a + Song

- One - way - to - create - a - song - is - to - record - each - track - all - the + One + way + to + create + a + song + is + to + record + each + track + all + the - way - through - (up - to - 999 - bars). - Another - way - is - to - record + way + through + (up + to + 999 + bars). + Another + way + is + to + record - each - basic - section - (verse, - chorus, - etc.) - in - individual + each + basic + section + (verse, + chorus, + etc.) + in + individual - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - them - together. - CREATE - SONG - will - then - automatically + them + together. + CREATE + SONG + will + then + automatically - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

- Composition - Without - Compromise + Composition + Without + Compromise

- The - technology - you - use - should - never - be - so - complex - that + The + technology + you + use + should + never + be + so + complex + that - it - interferes - with - the - creative - process. - That’s - precisely - why + it + interferes + with + the + creative + process. + That’s + precisely + why - the - LinnSequencer - is - designed - to - let - you - compose, - record + the + LinnSequencer + is + designed + to + let + you + compose, + record - and - edit - while - devoting - your - undivided - attention - to - your + and + edit + while + devoting + your + undivided + attention + to + your - music. - See - your - Linn - dealer - today - for - a - demonstration! + music. + See + your + Linn + dealer + today + for + a + demonstration!

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

- HELP - button - displays - additional - explanations. + HELP + button + displays + additional + explanations.

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

- ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

- ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

- © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

- © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

- (even - drop - frame!) + (even + drop + frame!)

- ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

- on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

- linn + linn - Linn - Electronics, - Inc. + Linn + Electronics, + Inc.

- 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003.text__pdf__txt/txt.bin rename to tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..415d519d70f90431e8b334ff0521a9751a1cdd58 GIT binary patch literal 12456 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}bF- zq-S8DU}gxS!1b_meqKpxUP-ZnAxJ(5WJD0OBrHa%20=9;th9_)&<}8NQ_v4dtte5@ z_smU9Pj!OQo>1BuoFf$=6_|pJje>rfu7R$B0;uf}1S*OZKm?*q0k#|?M_HO#5K*Dc ziQUFM?)~%g_tEG7K0U9G|M&Ol{JGr+cRu<5^r?N^zdx^kv>)H~YX1MK zkN0dR)o9gT@aI3b{paj`f|nM5GLNZSa^cIV)S7i~?r*&K{#8uJm%9>j`!19_PkH~e zvev$C`Cse3XXXF()tBsw|K7ib`ORbXe(4{zn?LxOmwu1W`xp4+^umknKTrQDsjvOB zbLOp>xJ~cN;(pJ$9+q(Gkwjo$bx3i^#t&0k+q1vi*l6o@vGIvz>ik_2Cf26*QG8!k zZe%ESI=1Pl&;I!DvY-BZ{vrFMQaOI%QTKC4;~tit_xo6#X{>+RKr-@TBD1HwruD9g zx6R|Vj~?dzJF&8;Q0|g??}y&!P3qsDhObL**#9bK>dpm!jZRcfKNbJs>BT37QMtcP z8unSuTUwL1I#Ss9-<*3dzt5WTQF7boDMvn4)g0mIpYd{8vEJqBX%Aj6UOHXmB9Bw- zZ|$3vt3NmWe3rET&SA^_>pklAlEqGW>m0Vby)$3i?_AxR%)~!Cb~U}bn6Ye^R5;g{ z^U0+*{?$~Ne14NHKhbvScj<18{tI7bP36fr_pj#Erfb>jj+~V%cid+l?iH3?AtmSE z@UUuI(Z2{$=FfIB6yHRzp6OC1=*Tm1&5?_Hmv624@VN48ReX=(^xuxptm^aFX7-;y ztNFD0*Rnd}DD!P|4zx$D)5$)=KmCQZ=7P_Q110=9yTdLYKG?Ii-``q)m+<2s74xUP zkYD$G>+!bq^J&JXlKMk6^LIU66)bM>=l_Ps6Q+JCsp^a#K%unp6_?;TeQ2R{*>sfWjQgg=3MM~ zH|0!SkKWpgS3idC_O3`N{+rblSoY~cN!!v23CU29Pq{3S92B{*IIr1sUfQxH8}`;16g+U%JI&ea+V{D7;lnBQPamJ| z+4J&fzrw}!8mCHMM}M04=#!)U8)2v5`xC7n@JW8Ld@$pIv>*HIgv;mV&SG(UVfs@l zbOYbN&oSnW=5~DsN9C@zY0oS&)|XnV5ixb=^v?wgj7@I7WW2w6-J(64B;#t|Z!BHj zxW3j{cT>K1>}9Dn+au#;jd|{=6`Q=OKKIPub-AGI?%&&brPqY6z4iKQ=~*$ilItrN zq(c}_?s@rXi{PG{%m0~4TfEHL@-Xv&<}`oPH>H;Qrl{w9_A6NDyZZ9}^%*h;g-*vz z_Fll96ZEu8GEQ)U73Ug>%|9Hk=Qr~FQ4x#kEG>)umN9Shl$|WQ>a2tEd-lzV2s?h> zpsHn6NYl((KZR%hCtv-pxW)3irXk;p&2oWj;frXor`u)q!=>`S{rJ!9JjY@Ar{~AS zOlLp3cJ^*{->=^3(GA9oKYa9489y<`s5#l(o%>xa@zPqKHn!hE3KOI&+s*%ZH%)l6 zaAWV>=~MJ1%tap0%YVG>UF_G0Ju_#Vy-;AWXrft)huA#^MF z)Ojm;c290L*8jrxEPF;P17{HP%h3JX8$6l}K5?=cue;10!!Gl2Q#8Ne&FZ?K{m*v1 z+!w`iY4;3?S}oxn+XG^5XtP_F3HxS?na7=*aA(h@G?DF3Uh)+tK3NjBEs!-$HCw@D zUQ5;d#MN`(zg6W>Y|xjPkRHEQBuzZwbI_wLAGf%5Gfui@aozmXk2dY_^S`-@w=-Y* z<9EuQuZLY*SZ%53v3J}rQ~tX31swEoTCntVwdIw7-H&!BW$ZsP?TMoC(wyRE!B`%4 zd)M_cM{FhR42|D3y_A?gZ9+l%L-UODk+X_^NtZ3XYVAGCGwIz?e%Y3bkNAy@%!{YJ zi%r_fpQ5(KKI!7mHD>F7W!be~ta$RLE;uCN(9PA_sVc=ikH2Q?Y+7+!WLmrJ1&_^h zx|ZykyMuTA?CG&}t-c#Sy(_PCIk7TR>fDRZJ1-vp^;XjHaftOy(f2xPTbIlVW8+KL zk#BtZqRd(ETe3}l((y_a&U=h*HKFgd=7m{2Q#zt%b!w;ht5lKAau;hhe|Ee$DRaW1 z$fVYtRxkRwzY5lOyj=X0_4V23W-oT^6ifNb*3*+QJro?^h{BbxuB8wrdvpJ{$!as zQ%VkBHOv&?yJC@}`g+aG-fz|XCA0kN+BZ$9+3GdNTCh>wZTb84>z=PY@hisp=|jQ% zLr)sAL&a5;Rb@`vDd`GWf0fAFd{zCc^n;=!%u;<)tCSZ;bNiY{TWmYbZS-ye!=iaw zrQNBs-hDs&D7$O>UG7+A_pfJ?MHNIk51+n%+tzcYljOwxenCfVSKZ$z;2n@(W?kfX zb;rYXe20Z!^lzz3P>?O?UU^mb_|uNl~b*=<${aZg2|^Y{Msv4v$*HauarJ(g*CgJUheVt zIJ0%G&ZZXv8I6b9cfB&raDT#>|F&QHyY3Ey>z0g)-)CnXTCySEcHI;aqXW~-nLLb5 zCod`sKlXpkpOn2~pI>sz{%TaM2qtqGm z6;A#dVwtznizeRUS5AryniAQw;~@k8_12Z9EG5D24K>pyi?VG_TD73kLZjhIHFH`0F`0J9HfOwg2IFPLE@= z_S~x1Cc?EgrC(j86+c|x#8s;!wvJihoYuj$l1?6910v7P`?h+Ar+T+sjH$PJ%La=_ zI)Y4zS6@#v-rMFaG3SCY)AwzYwsE>@a85E+7cY%`A5*Hm&3U0>X7~2~&&47IODaFA zY%>k~raZ?oy?5J>&0iLl?zp|8;x*^vN6U_Fig0R;RlH`yW%)UE`=1Z_(&7*KqAFvP zy!V`YQhU;SkC>EtYj<<9NI><1Nj*Ei@oIg!wr9t(nf(jfR6gffoYCs5*J&t>@p`hx zsay7c>g!dHmNajFYO?ed-*lJzv3n8^F1Xz{>)rwTeY$rqtxwHOZ+*NwaA*3o>e>Do zC$37pJ|c8kCr)*l>dncIPpfGs<=xB8eiWsZeOqOcf9Ji}z0Y6E#_ctwb-0F(`{GYax8f+zVb@UM!BbQZez)4aA|r2+rbNVa~T(YcUh3bdNMU!a#Jerx%#)l+%emtrv?3;bJ=N& z{q$_Zi`i^kK6he5>rS0&`6_mNM%I&g_kVumh{;I2DscR0I-i30y45-V{;)l5>GR(E z>UQzUY?XPBPyN{YTE)uKProoar|Ix@^}m)c-)xwA&8|pM;l$;z^;d81VapsqE}e+A&jZg}hj@sY0mL^>X`v+s&JLoYNy!!uuWG z=BVgRZu-T-d{S1jfA-3Tmkd+S9+A+MubStd6tzq0+p6xcz2Dd7JyvVq=Tq-j zxy8TI`{ny*yQxdYg$AQ01FL|YtGZ@=VZ5)|w)fS#jJLmb`dmLO^E0a3=SqIV8!hHb zS>N{v9_Wflx?y}`l5uE{@80CrwI4W=r!M0@qvB$*1j~Wx6}qt43kr*3wgn8zrm_ zKE^25m1-TCGGS%qmrlD)TY2vJus3dORSsujR^~3=FP)LFZK9mz)umDArH@FqXFB(n zFA7RjWSY(_tP#2`$U;!H{LgEQl}q9oOIhaJj8H2H ze|*D_^Tiqn@BZr)Z?c8_z>#4$JgZ!3Jd zW-HgnMK0<$m)ctfd1>e^Nciz;ldt;R(AjtLuN5Q&&i^JgtuA2S%#6u%t=_oz7AZet zi`l_xwemri;E&CmTQoVY9rZtPe1dAi%Pm&t4^1*(e*16KNg-8>w?BBEUTIR^zSR1% zx>SS7?Hu`gVkhGB8g-8yeBJJ^+P%u_AQc?zXOZ{FjqGpUl+SoVW4Dj_nbDl)rgz{p_*kxmVzO@lE%REzq^+xW7t) z>2_>rdwcMvY+c9Y^8ZuT*Bw;vWdC((&erf#0^7IWoilIuhC@raU!O={viw|^NBY#H znYyk;g$;^DpQCMUUVn^W@z}SO|E**4{QA#1qDen;g=DJdJ?_dCoSQC|BYpY(@A$1x zS{fFe;L4c0#pYMwr>8l748p(u{Qh`tYVOfndAHPNz18Pizb#|qsyQq#b^o|OblrdP zq6c%r4bMi!wc#9bCbNWg-}|H$I5{GPDtwY;RJTyk4s#Ss4c0VzgT+Zs`R_dVj0cl5;s)6 z(O%;lw{}l*?w_>0=&d)JU9}oN2zEx8OyT~#o4xQ%FXLSU^Qn`hPOnQ`wSv*am9b$QO8N;Z2yKto6?J)Sud3q zZ=8{GQ1^hZN;&7#q(_S!HmZs)x_N$0M|Zu{uETRzTeaTsIym*Xd8XltGxJ@z9y0OS z=$Y`Q$|w|E{;ZX-NPDCG=a-w^Lw_Qt()7j0kHWx1JgP9Br*mG7w&;=LMWV(wXKv%Q-8N?VKh!Cl7ws2ay7 zi(cgR_#d9>Qxkgn^MMz4Q-VAG-`DR`oRrQev~B*fNvdzQPR=?L`hJ?A)%K$te(UG2 zFk%vV;Zm8j!9=fgoj}~)D?vUi$D_J-t;>^_KyYtNd} zH@ETafsVpedXthe*cVO=x@V==xT_?rZBm9!0q>K!&b^ygT@<=8|GJM)`aX#TFIUAX zR+nn)t&iic`(n}cH2u&y>sc{<{}1^5_mniuUg5qduvF{hDszsY6#kd`X7-04dOSPB z7#udi=zwFmho=QY?7x>idav8mzHJY9dZ==9-ojZ5yIoi8_WmFJFW+A5ha`(pj>aa3 z19y%qZ&^`1?cIteeZgtd-p!mIH0i3^eU5wA=4>fr6!N=rWmC=$C9S}n6Xu7!xOi{3 zj^1wlTbk=kgTt(AUKD*=_cJ>KaZ@-V(%c-5rfosEI@=^9b9ET6EksO<{;F?Zdky=R}leJi;%=w;4pqf_yJ z-<3VxE9BV1y4U(yz=R1i(jLxp<4q_~5EiPKnRe+%c<{v!4;=Y4?Mtq6d${T;vZPON zGTjnhv&GN#L{aUO@`&8Ub*vkw9p1&G?qq#V;uF)#J)95bsM-Yf*A?4|1hie?5H|Z> zTmHp+`^&ku6^~m)&KO)<{k#3n!*>z}#+#n69-Q)(N8`25<3@KCjmr<^9qQ&u z**+^=;+FAufz0dFtKvd-bCzyXUvm1kkk-V=_D{`wAC&jSO5HZ^{gnFKam!cUD>-MQ z@64I;xF>yju)-M@i_H&>8Wv|=bMbp<*fHwz5e_d>7w>4X@lYcbqE`<;pQFxMG`?d%LVdzxD>t zNd?L7A6r~sdGFNoQ4Z|C=hjv^{n3)ck*|Vh-MaMb^)m;DYYx)2x4f)v?z|Ovrh087 zGtZsE$A{kREHcV%nRxEK5Xqj&b~8$CD8*>XX9^90G$ zrhVK7OM~2&)kS<&Xf7mo<#ZwU*k=cmpS++WW%v$3SO z^2V_wzKM%pgw@y{+)*+ueJ_jm%d$gHgRgk$l-fO(JFu&X_0&!6Es}F~Ol4_fT_LM3 z`d#|)2irqUvA-_Lu?9V0xLT$Xf2ydWs^RXBn9C=h3G6#`uA_Iy$_w>&1ux4sREGx| z7ki}J&u6*gb4qVtR&#Mk`dy0+!Ank>Jb$ptx2UWt^ltc+JttQ3_w5n?IWMuggJr4f zL6IBRZys{`!|v0n>nrv{c~8-7Kk;`#UfKuy)ekDp z_0OBrH0=o6uGiXTwZGrynJ%9qzHRmE^KZ7SHV$&MI${y)Ws&arLZ|0#bXd-9qx{P{X1B^Ts~pe9 z!t^2yR_$49G#$_Kiw3K?PPE!<|FwR3j2G|h;zP0z0zGDKS)h{W7OC%B zwmTWtycM?(iRkLp{M^N%{A%~snwe94UL2cN*7nomh~lCNNt|6%Pv73G-L=qc--?Oe zLPcNZrD$#YbSXUOQpaUwzDbc{84Qn@Jzf4kIcRCteCeKu<3_D%zgcE>&%L~}!#yv` zJ+Exu!(WRY&An`IefiFvk6+%cTx2b=;r(n;p|h{l7M{6n8n(*KqKfyw`MQewAA29I zJbGDc?Vjt03tyD4;!pg}EU{7d9nu*!Uo_WTWBa@7 zWxs>3&nNdu5_U0c;U&v6%EKegza4zdWZshg@-#d9QhA}kj*K^N|9+jYyqj`zT*%`aoO77h?qvZxC%TyUe7Iof#?NIgU`uC$IYG$67 z%ML#XnGo)JAxNrBUh1?@;-wvNvT6I|j%B^>dy+0$r&7wZHqqqR*U4VWWg%xX`;wd6 zdXH^;Y;L;i{`uxbFL)OI%5Ct;o7O$4p<3X;6}!Kz#Tl;{*WB2<+Wd}8+LCWQB~Du` zAD21&4EmecnYd3T^~sA!&VNp>U306K1^Y}Zo}k>5Q@M!MtM~koYfmlNa_QW>bw|?9>r3rqQ+P7@u+2Ip0g*0oyP20AEAKH?H?-Eg zd$i!3@IgP%=H@-OxJsY6omlnPZhoM|jEcxt+R^%I;_G;2pZ_jVD4TJ8)}6gtZS`Sq z9|hQOD6_p(o};d|f2G!fSfPV^9G9G3#<@T^F41Vlgoi4BecgpOCIqy%du>^N+cMVh z&fA{1KlQKfczpb{z)K}hi4yW8=pEs;#mARYjj?VeyGIveS4Bj=D4*$O!sBlr;`GME* z3oMpXMITv|ngq{^+Pn6=;LX$8nx>aG-}H~j(>_~uXPd~(;3N6AYc6~$u$ytqRr!LG za9mzRqV=?MeP(n2&0NJh=Ro{zDN8SjkWSOCZNMA$jL>Q zId@-WTYltj@ZUatbIBtIr{8{ZL?x=NWUW{mx1EWXhzD=O?_X*^t(_)MS-dRvBj+AB z#jqW$zuCTWoowd0?s$FnkY}{@w>Y-$= zIA{6fp4}}~bHaR%zqtFg&PG$8dr$KQ!-Z`r>on{M!2bwQiD5mVE4+KVNx^C|i^DeB}$f zgL+f8%0^mE_hvlsQL^V-MAs?(-HTe|CnpH`*|sj76?>jT`E{;1V-{b_1v}v@bN+v6 z^-BE4y6Z2~)M6&LoOXB1qwRN;^!3Z11`0{mR_=GaY&Lgu_TfUUylhwYAcyxgmHpHE zm(|a=6x7Oj9e8K5_m?&0Uydz!AgVJ-bEdPOy7?hn+xtFSr6MB_Jp9D9>t)RUxEMc+ zK7%WV!>9beIK8g4q2uA4x5`@s=lyf!N%HxGmMK5SC-u77#;iq!lH=L*|pIciJ~3y%EITk$UEaQmEX zx6eGbyZ`OmRd&U-a*tJY_eQ0D7T?R6-*IXwZ{pt{zO6_1opGxES=qVtVA73yhvlpO zKG!O*W()7&)?X5~R=z-cp)g((B=S|6}6ioYm{D zDo>2QKj~)vN*P<}9qiSq>@iX$$6rT0Ui$Pzo`G%s_8fy2!LIFwFJ1`td6f6t-*1>; z;^o;g>++~V z`D$8iJ0{*eeKg6=wcY=BpOP={=8`IDfoGM!f8VeB`BRR?vh&LUQHQM8JM9*Esj>J zfalMtnuM519uq%ZT&eQE{!+ogOT4E}-YF``>11)#>CHPWy{^|cb!cio**ALmTTHGIUqM_|44L!Pf? z7u~eC{L13!3GaG#@SOP3;OoCPpDOe&cm5gp$?1`5%gnnIrKTJ;7Tg=7#^CaK`;U+q zp6?Hz{dCFwSol3x!Qx`>!xJ6b%cS)mFo_rZye#~8-k+t)46VWY8~f)ibhb#+>??ko zHs#qCT{j-~fXUZy9h!aMO2>h&+Y_okIF&BF>L{>jiD|jBwyDw^^_?AYpKhGoE_8Ui z%>Dem>^o;0nzvf+of~`2{3zQMuZ$bLKU{z9d?)P@Jj>+$mZ{k$3U)V515TWtuKe_< zLDX(9^Xi|c&fT}p6}WqZ{rn8gdlNqhFP?B|LTwJW=9;J10+d!{hhJt?u3*ngb7a*{ zwz%r%#(pZ-`sdNBcD0ui7F`h1HI2=f_Qj}U?VMl_r$^tTLN8XUD_vTn{;`K?%|gY8 z@>652KCFq2-m4kE&a@yeo$2E?{rcA%JN185?)^6PE_?8w_@?jcZr_npD*xK>Rq@vE zv};dSJDgf~&(!(Bn?UP1*O+bJ90}NUR%$CpDwkmHlg!t`_Qw{)C~r{zU2p&1W!0(f z6~8BVUF>>b9>Uqmm-{46=vhR@y`00{l$v0J3g#VyuN8V=XMSK7SrqMA$5sPha&Ry z_K9!Qzmv4xXx5FS*iw3nN-egXSm|y|IgX4 zTh4#ydi(j97L(TD-6uS`-WZAA$eXcy%0K)5B?nHK-uq&j^v=tqTwATS@#pHD)3s77 ze|p{1n6~TUPZrt!#j^wyqPcb%?ODI++m*hn@d=L;YA*UeeH(mxThkd6n^ z?EJ1S-a6yKGB5TMNBZhJ|KFXkVQ$B9t2b`S_&?U99EvkN;q~{nT3I zPUU5_7j>3?zvgSWm_pUp~Hrp|@V&Rj9Pp@X(5AQ6Tkazes zd#v)T1v3^(&!1I%|M-r$?(Fpsi!@%mJKY%gUDxg0#Y&@sJ3$LN&aJ-VqCI2QoLkF{ zyEeMtYP$UU-VDtH4%;_hyt0&;>%A4{9I0;s6XV`aY&@AGbA16L>)xEGCf8Sw8v4?n z@AoS4;SpbU_kajr;codQ)~h2qO9IyBv~4h*U3ER{#ABvwDKe1(9#=SC>@yRdY;j+w z)_2eC8S8#qcn6+#3X;6u`S{x8dnxIQ#Fb}CxW~_Zl#?L1(qPHLtqn?hI4Uyq&Aj#I zBzk^o6T81BB|ZIJ$gGr)!E?&FHD?B~=LRf4oa4Gsd{MxmaL1eG*}M{!+|Ob(PZz7Z zhD|%5Yr4ps`DCt}dtKC#wrxu?9tXdEcI3%2%L_)Q&CBg;YBX528Mkz*N&MOnpvot7 zjfJy0R+;JZt6AyCU#W3FaGJpO=+I9y-}_rm^kua2e&BVyF*n(Au19oAKbz8X<|daU zqn#;}w{HK=BjDD%F*oHQ>+Y0^3taei-m!T$X@9p(YroH-Qz!STO6py}%h*~H-rCTuTc>DFj*eS(c_N}Ly{=HjT-XwQILrXT$__OItMphe@ zgVJ^}SF2>%4lQc^bEB)kc=Fnp9;RlKGgmks_IY)agWpO0VHij%_So&`MQ??B;Tnb{RZ0gO=O6xg_ZJpof^Y>lOJO<^t84teO zG>e{EeZ(^F;%d3u+TJhICYnv}d!V?>cR7!P?EHN}fAe)W_nx;iTr zY&prlsPFpB2-!73XY_<(CI6iXRIn{N`ccSeulgR-zOP@`RThh0vhzOk<-g*>#I{zw zJM8M3!LlE2JiNj0x%@73$}DLa51|(`x2{++^Y{IEKgH+lwljNr{hrgGZ_LwYzFQR@ z{HDRa-DIBSZ@z>*Lc95H-K<}@?dnvF#coD87Q2B~r|ElVrhwL!gBHXEftQ@=JLi|? zl_(e@E!H$dUF9B7l%GQUZd+=l$uzQnV;v9Sdyxs z;bLWEU}$MzYG`I?Vrpt&s%v1bZeXAeRvD66l9Q@nnVK4#Dx@jE#SBdqz)}i%P%%S8OH*_)0|Nsy3p8~G21Z5(7-A--<``n;CK&1r z4NZ*E)fpO_8e)i9T4LxmGBQ9HGc+#AOB6tn l0A5`PT3@dKjuY^vg5r|Iq7rbR85$WH8*r(ry863u0RZ~aki!4~ literal 0 HcmV?d00001 diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/stderr.bin rename to tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/stdout.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000004.ocr.png__000004__hocr__txt/stdout.bin rename to tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..137fef56 --- /dev/null +++ b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,124 @@ +2A NNI‘I 6F6867# XATALL IE18-80L (818) + +9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI +“Uy ‘soTUOMOI,q UUrT + +uu + +“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV +“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e + +‘uonng OdNAL dV L 9) uO + +sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e + +(jouer doup u3a9) + +“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « +‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e + +"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © + +“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML + +"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV + +SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « +“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON + +‘suoneurldxa peuoyippe sdeydsip uowng g1TqH + +oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « + +jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL +INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue +p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) +Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT +yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy + +ISTUMOIAUIO?) NOAA UOHISOdWIO) + +"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd +uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo +ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} +,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes +JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes +Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM + +dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, + +SUOS & SUTVAID + +*suoT}oes poJUBMUN + +SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG + +“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI + +ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP + +B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] +$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL + +‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy + +0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns + +sainjeay [PUOHIPPY + +‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} +-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe +aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM +—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy +ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL + +sunipa + +jsesdueyo ureisoid pue ‘fepod ureysns +‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour +pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen +Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN +NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar +NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas +*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k +UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE +SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd +{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— +yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy +*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 +2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA +‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor +§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 +AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, + +g0uaNbas & SUIP10I0y] + +‘JONWOD s}JouNaI TeuONdGO e + +"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e + +‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e + +‘onqea ory AY + +pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e +‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e +‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e + +i ASIP Jed + +S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN + +jSIOZISOUJUAS + +stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq +ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e + +‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA +LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ +LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO +St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay + +JOps1odady soUINbIS [GTI YVAL ZE +Jgouanbaguury oy + \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stderr.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stderr.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stderr.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stdout.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stdout.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stdout.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout/stderr.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout/stderr.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout/stderr.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout/stdout.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout/stdout.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout/stdout.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout/stderr.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout/stderr.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout/stderr.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout/stdout.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout/stdout.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout/stdout.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout/stderr.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout/stderr.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout/stderr.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout/stderr.bin diff --git a/tests/cache/cardinal/__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout/stdout.bin b/tests/cache/cardinal/__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout/stdout.bin similarity index 100% rename from tests/cache/cardinal/__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout/stdout.bin rename to tests/cache/cardinal/__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout/stdout.bin diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin deleted file mode 100644 index d3c2e860..00000000 --- a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin +++ /dev/null @@ -1,123 +0,0 @@ -The LinnSequencer -32 Track MIDI Sequence Recorder - -The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is - -extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: - -¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST -FORWARD, REWIND, and LOCATE controls. - -e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may -be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic - -synthesizers! - -¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes - -per disk! - -¢ One or all tracks may be TRANSPOSED at the touch of a key. -e Exclusive real-time ERASE function makes editing FAST. -* Exclusive REPEAT function automatically repeats any held notes at a pre-selected - -rhythmic value. - -¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. - -¢ Optional SMPTE time code synchronization. - -© Optional remote control. - -Recording a Sequence - -To record a sequence, simply press RECORD and PLAY, -then play your MIDI keyboard in time to the Sequencer’s -click track. When the sequence loops back around to bar 1, -you’ ll hear what you played—only all timing errors will be - -corrected! (Timing correction may be adjusted or defeated). - -Any additional notes played will be added into the track -— existing notes are not erased while recording! - -FAST FORWARD, REWIND, and LOCATE controls -may be used at any time to quickly access any location in -your sequence for spot-recording. To overdub a new part, -select a different track and start recording—while you -record, the first track will play in perfect sync (unless you -MUTE it, or SOLO another track). In this way, up to 32 -tracks may be overdubbed! All MIDI effects are recorded -including pitch bend, modulation, velocity, aftertouch, -sustain pedal, and program changes! - -Editing - -To erase a wrong note, simply hold ERASE and press -the note to be erased just before it plays in the sequence— -when played back, it will be gone. Notes may also be - -added, erased, or changed using the SINGLE STEP func- -tion. To overdub notes at specific points within a sequence, - -Additional Features - -simply use LOCATE, FAST FORWARD, or REWIND to -find the desired bar number, then start recording. - -The INSERT/COPY function allows you to move bars -from one location to another—in the same sequence or a -different one. For example, you might insert a copy of the -first verse between the second chorus and the bridge. -DELETE BARS operates the same way to remove -unwanted sections, - -Creating a Song - -One way to create a song is to record each track all the -way through (up to 999 bars). Another way is to record -each basic section (verse, chorus, etc.) in individual -sequences, then use the CREATE SONG function to “chain” -them together. CREATE SONG will then automatically -copy all the parts into a new sequence. If desired, you can -even set the last few bars to repeat infinitely, for a fadeout. - -Composition Without Compromise - -The technology you use should never be so complex that -it interferes with the creative process. That’s precisely why -the LinnSequencer is designed to let you compose, record -and edit while devoting your undivided attention to your -music. See your Linn dealer today for a demonstration! - -* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the - -HELP button displays additional explanations. - -* Non-destructive recording—existing notes are not erased while recording. -¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including - -ERASE, REPEAT, PLAY/STOP, or LOCATE. - -¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. - -© Will sync to standard LinnDrum or Linn 9000 sync tone. - -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. -* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, - -(even drop frame!) - -¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes - -on the TAP TEMPO button. - -¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. -¢ Any TIME SIGNATURE may be used, and may be changed within a song. - -linn -Linn Electronics, Inc. - -18720 Oxnard Street, Tarzana, CA 91356 -(818) 708-8131 TELEX #298949 LINN UR - \ No newline at end of file diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin deleted file mode 100644 index d3c2e860..00000000 --- a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin +++ /dev/null @@ -1,123 +0,0 @@ -The LinnSequencer -32 Track MIDI Sequence Recorder - -The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is - -extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: - -¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST -FORWARD, REWIND, and LOCATE controls. - -e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may -be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic - -synthesizers! - -¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes - -per disk! - -¢ One or all tracks may be TRANSPOSED at the touch of a key. -e Exclusive real-time ERASE function makes editing FAST. -* Exclusive REPEAT function automatically repeats any held notes at a pre-selected - -rhythmic value. - -¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. - -¢ Optional SMPTE time code synchronization. - -© Optional remote control. - -Recording a Sequence - -To record a sequence, simply press RECORD and PLAY, -then play your MIDI keyboard in time to the Sequencer’s -click track. When the sequence loops back around to bar 1, -you’ ll hear what you played—only all timing errors will be - -corrected! (Timing correction may be adjusted or defeated). - -Any additional notes played will be added into the track -— existing notes are not erased while recording! - -FAST FORWARD, REWIND, and LOCATE controls -may be used at any time to quickly access any location in -your sequence for spot-recording. To overdub a new part, -select a different track and start recording—while you -record, the first track will play in perfect sync (unless you -MUTE it, or SOLO another track). In this way, up to 32 -tracks may be overdubbed! All MIDI effects are recorded -including pitch bend, modulation, velocity, aftertouch, -sustain pedal, and program changes! - -Editing - -To erase a wrong note, simply hold ERASE and press -the note to be erased just before it plays in the sequence— -when played back, it will be gone. Notes may also be - -added, erased, or changed using the SINGLE STEP func- -tion. To overdub notes at specific points within a sequence, - -Additional Features - -simply use LOCATE, FAST FORWARD, or REWIND to -find the desired bar number, then start recording. - -The INSERT/COPY function allows you to move bars -from one location to another—in the same sequence or a -different one. For example, you might insert a copy of the -first verse between the second chorus and the bridge. -DELETE BARS operates the same way to remove -unwanted sections, - -Creating a Song - -One way to create a song is to record each track all the -way through (up to 999 bars). Another way is to record -each basic section (verse, chorus, etc.) in individual -sequences, then use the CREATE SONG function to “chain” -them together. CREATE SONG will then automatically -copy all the parts into a new sequence. If desired, you can -even set the last few bars to repeat infinitely, for a fadeout. - -Composition Without Compromise - -The technology you use should never be so complex that -it interferes with the creative process. That’s precisely why -the LinnSequencer is designed to let you compose, record -and edit while devoting your undivided attention to your -music. See your Linn dealer today for a demonstration! - -* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the - -HELP button displays additional explanations. - -* Non-destructive recording—existing notes are not erased while recording. -¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including - -ERASE, REPEAT, PLAY/STOP, or LOCATE. - -¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. - -© Will sync to standard LinnDrum or Linn 9000 sync tone. - -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. -* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, - -(even drop frame!) - -¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes - -on the TAP TEMPO button. - -¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. -¢ Any TIME SIGNATURE may be used, and may be changed within a song. - -linn -Linn Electronics, Inc. - -18720 Oxnard Street, Tarzana, CA 91356 -(818) 708-8131 TELEX #298949 LINN UR - \ No newline at end of file diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 58% rename from tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index b76beb1d..48da5a5e 100644 --- a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,1054 +9,1054 @@ -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

- extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

- ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls. + FORWARD, + REWIND, + and + LOCATE + controls.

- e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

- synthesizers! + synthesizers!

- ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

- per - disk! + per + disk!

- ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

- rhythmic - value. + rhythmic + value.

- ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

- ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization.

- © - Optional - remote - control. + © + Optional + remote + control.

- Recording - a - Sequence + Recording + a + Sequence

- To - record - a - sequence, - simply - press - RECORD - and - PLAY, + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

- corrected! - (Timing - correction - may - be - adjusted - or - defeated). + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

- Any - additional - notes - played - will - be - added - into - the - track + Any + additional + notes + played + will + be + added + into + the + track - — - existing - notes - are - not - erased - while - recording! + — + existing + notes + are + not + erased + while + recording!

- FAST - FORWARD, - REWIND, - and - LOCATE - controls + FAST + FORWARD, + REWIND, + and + LOCATE + controls - may - be - used - at - any - time - to - quickly - access - any - location - in + may + be + used + at + any + time + to + quickly + access + any + location + in - your - sequence - for - spot-recording. - To - overdub - a - new - part, + your + sequence + for + spot-recording. + To + overdub + a + new + part, - select - a - different - track - and - start - recording—while - you + select + a + different + track + and + start + recording—while + you - record, - the - first - track - will - play - in - perfect - sync - (unless - you + record, + the + first + track + will + play + in + perfect + sync + (unless + you - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - including - pitch - bend, - modulation, - velocity, - aftertouch, + including + pitch + bend, + modulation, + velocity, + aftertouch, - sustain - pedal, - and - program - changes! + sustain + pedal, + and + program + changes!

- Editing + Editing

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - when - played - back, - it - will - be - gone. - Notes - may - also - be + when + played + back, + it + will + be + gone. + Notes + may + also + be

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

- Additional - Features + Additional + Features

- simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - find - the - desired - bar - number, - then - start - recording. + find + the + desired + bar + number, + then + start + recording.

- The - INSERT/COPY - function - allows - you - to - move - bars + The + INSERT/COPY + function + allows + you + to + move + bars - from - one - location - to - another—in - the - same - sequence - or - a + from + one + location + to + another—in + the + same + sequence + or + a - different - one. - For - example, - you - might - insert - a - copy - of - the + different + one. + For + example, + you + might + insert + a + copy + of + the - first - verse - between - the - second - chorus - and - the - bridge. + first + verse + between + the + second + chorus + and + the + bridge. - DELETE - BARS - operates - the - same - way - to - remove + DELETE + BARS + operates + the + same + way + to + remove - unwanted - sections, + unwanted + sections,

- Creating - a - Song + Creating + a + Song

- One - way - to - create - a - song - is - to - record - each - track - all - the + One + way + to + create + a + song + is + to + record + each + track + all + the - way - through - (up - to - 999 - bars). - Another - way - is - to - record + way + through + (up + to + 999 + bars). + Another + way + is + to + record - each - basic - section - (verse, - chorus, - etc.) - in - individual + each + basic + section + (verse, + chorus, + etc.) + in + individual - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - them - together. - CREATE - SONG - will - then - automatically + them + together. + CREATE + SONG + will + then + automatically - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

- Composition - Without - Compromise + Composition + Without + Compromise

- The - technology - you - use - should - never - be - so - complex - that + The + technology + you + use + should + never + be + so + complex + that - it - interferes - with - the - creative - process. - That’s - precisely - why + it + interferes + with + the + creative + process. + That’s + precisely + why - the - LinnSequencer - is - designed - to - let - you - compose, - record + the + LinnSequencer + is + designed + to + let + you + compose, + record - and - edit - while - devoting - your - undivided - attention - to - your + and + edit + while + devoting + your + undivided + attention + to + your - music. - See - your - Linn - dealer - today - for - a - demonstration! + music. + See + your + Linn + dealer + today + for + a + demonstration!

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

- HELP - button - displays - additional - explanations. + HELP + button + displays + additional + explanations.

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

- ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

- ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

- © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

- © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

- (even - drop - frame!) + (even + drop + frame!)

- ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

- on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

- linn + linn - Linn - Electronics, - Inc. + Linn + Electronics, + Inc.

- 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/txt.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cardinal/__-l__eng__000003.ocr.png__000003__hocr__txt/txt.bin rename to tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/ccitt/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 089cb041fe3784e9fc9fbac72470b91cd2afb0a9..2a214e8cafb950d4b622494c24a8885436197f1d 100644 GIT binary patch delta 27 icmZn)XbRZSuf}I-U}|V)Xkuz&WTI -
+

- Multicolor - Black + Multicolor + Black - Pure - Black - (K - = - 100} + Pure + Black + (K + = + 100}

- Pure - Magenta + Pure + Magenta

- Pure - Cyan + Pure + Cyan

- Pure - Yellow + Pure + Yellow

diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 7cedf01c7ff152a1036d5f0177d7bd86421a2a8b..df52c271ea1ed1e337e04f2215fc00608a6c7305 100644 GIT binary patch delta 27 icmZpaXq4E{$Hix9U}|V)Xkuz&Xr^mmzIhH;DkA`9`v(>P delta 27 icmZpaXq4E{$Hix1Xk=_)U|?clV5)0iv3U+xDkA`9UIz>S diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/cmyk/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/txt.bin deleted file mode 100644 index 080288f6..00000000 --- a/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/txt.bin +++ /dev/null @@ -1,13 +0,0 @@ -Portez ce vieux whisky au juge -blond qui fume sur son ile -interieure, a cöte de l'alcöve -ovolde, oU les büches se -consument dans l'ätre, ce qui -lui permet de penser ä la -cgnogenese de l'&tre dont il -est question dans la cause -ambigu6 entendue ä Moy, dans -un capharnaüÜm qui, pense-t-il, -diminue ca et 13 la qualite de son -ceuvre. - \ No newline at end of file diff --git a/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 73% rename from tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 05592c58b4a792b1427731a23152b02e6607d76b..188f998a04be08ebf18edd8adfe6536fa3624162 100644 GIT binary patch delta 896 zcmbO$GfQT}1`bBE$qyNo>ZgY7&XY0ZdGlBF7o+wfxhX6xj5+srdWa|NRtRk>(VJux zFP{~@MC0wv!UUPJ_dbD73euLuA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z( zp?uxF-PP%D7G7seZ`;Rz zo%du?|C-{WHxtEq)3)tXZ+r6m%tGduJL@|Viv4>FxvXD3opR2~@8w^U;}O;MToa_& zB$L!e7 zGqRTq>YY@RXF79!e17lbI+?vX6^fZhXUbRITWQ$G>X37_%SG|>gZl@=)13UL%qa_%6PQ`jpS`iO zxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cxhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk z5DsNS=0{dlPjAn?UZ3;(r-zaX_tmCL$4mqbd5%s$@xef(`S)vwookiMHm}l%+jTH3 zRg5KU(#-o0PH?@m`+Y-Cph$U!`=;yFvQepLw}z_gH?6yQ>3;2d>0O!eH*6+U+a>=gAoIy!k8oi&48OUA^x{8i6ah|)Qho_;Xr z{jS}oAvhWm7HR|wBOnLW}iQ&`dlV%?&(H5hxcEc>wYHH&3L=IQ}_0_`qNjE z1UIffc;k`9v39w~%3o&mhy+{4{}$=vKlzF4{U+bcC1+swO!JAUUtXcxRX-34SIRjLXX1VrO3?vGQE9I&;N`3u`nQg;lTb z{>YbZ)F`-r#VwDiebtAOg89rsN>6-oN`<4uC-pMs3am6z_gfv;M1nu~`_3AQ*&TZfO&D!``cKv>^rq%GD zG5e*bC#?V6)h+2f^GoQS*KT7C_4dhNin9RVk%t_^c2c9JN4=J{g9 zvrC&-9R2cnqRJwUJyGlH*^-l%+-|+Z&0)iIG~k6(hInD1nz1(5xtz11)~;Jt7T38e zYwxiBe&HSeUnhg$P1ma>cd<$+uybrzu-=w>{?T5U_h&bq?_Iz7%&8goV&1LxFn4p- zoIZ!i#q`g0$^31 zB3nE)jQxt5ok|~kWtsl=rTp7F-ENabd8Y6f8yFiH8yOjz>Ka%~-pUirXgpbvcQd2K RTSTmT#KwMGB{ diff --git a/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/francais/__-l__deu__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..25fdded2 --- /dev/null +++ b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,13 @@ +Portez ce vieux whisky au juge +blond qui fume sur son Ile +interieure, a cöte de l'alcöve +ovoide, oU les büches se +consument dans l'ätre, ce qui +lui permet de penser & la +caenogenese de |'etre dont il +est question dans la cause +ambigu& entendue a MoY, dans +un capharnaüm qui, pense-t-il, +diminue ca et la la qualite de son +ceuvre. + \ No newline at end of file diff --git a/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/pdf.bin deleted file mode 100644 index 40830d1ace3edebe8ec5dc6a0b56ae9088d5eb61..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 3614 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%WewwabOocK0L8b z^8E#dzbAkH-u<)4>izPBZ}*SI))%RG~ez40jkNEmyx}{&zy0{mlHd|2a-)xBWLzjdppHZE(e+uW;Y-9Rk%pPFJs11L&ee4UWuhwiEAFcdFi^P^p&s|uf+Dsyk30dU`J(cYC_P``DeO| z&nfNwbN5IK=RVV#f~D$S{29f0v!D3PXFm6&LVo3KU9p`q>rXlI?OsvDW~3VR{;!9W zgtPsmXZKPzoIAU5ZCt171WVm>73;OPFaM~(p}?h3IrF0iZ~LmrTTRaT$o&5H_P#}; zOzE?Wf4Kg<4{`nFkq|wlmqkv=eL;(nxY$+&=Ws8+VhPjN!NO;@7T#n_7bpsp5J^(y zXrdePQi(lKAYLkG{|d|ws?r2KegST)#t5el_K+2 zRROI{XPo!VHJvJU_iV_5;3kc&7xX%E9A3LhpSO$ef7u(lPR~d^xhQzw`UQ(-Eoz*??bqj0{AaoS4{zta`(~dDUFB7I z>H7SmVipM@GfFLgnI<#lCfc2PxP8Q{M*gkFL@toe|N=L^tXjaru^==?bhq` z+v_dw=)Y)_l$X83zF(@a&@QrX`ubHGMQIgFG`63(%6HZCdieU^FZHcDx*lSbF-ADb z7*L_0@1290#i>P!$t4OV zdIow1;HHstQEFmIW`3SaVo9okhKrSvfuV(=k+Fe+fvJI^xvqhQx`BZ@SY=3NNlvPQ zjhic|or5SqDvDCmxC|5wjkrJ!AqZA5Gc`3fRY+5Siy4|KfTa}jU}DCGh9>A@1_ovp z7-B|71{h){rsf!8<|Y{G3=K_;(bX9mn;K$>Sz2P~H8L{55HmJ3w?ybIDN4-DNiBl* z;es=(QbD1hpr9X=pI@Q?iUe?>4(iq@fa3(*mn$wwEGhv9nz50EC6}tItG^o;0HK-} A4*&oF diff --git a/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/txt.bin deleted file mode 100644 index 4ded2d60..00000000 --- a/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/txt.bin +++ /dev/null @@ -1,13 +0,0 @@ -Portez ce vieux whisky au juge -blond qui fume sur son île -intérieure, à côté de l'alcôve -ovoide, où les bûches se -consument dans l'âtre, ce qui -lui permet de penser à la -cænogénèse de l'être dont il -est question dans la cause -ambiguë entendue à Moÿ, dans -un capharnaüm qui, pense-t-il, -diminue cà et là la qualité de son -œuvre. - \ No newline at end of file diff --git a/tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index e0cb3317afbc856c23b9d9f9377733e313f2036d..c741ec2f74ba81b1503fa37e08e5bf8abc94d3d3 100644 GIT binary patch delta 27 icmaDW`Brj66*r%yfvKUHp^2%9p{1^Y`Q{GpR7L=MNC(XT delta 27 icmaDW`Brj66*r%Sp^>qHfq{vIiJ`86#pVv~R7L=Lln1~7 diff --git a/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/francais/__-l__fra__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 60% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index c9ae9039..e7472ec6 100644 --- a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,467 +9,467 @@ -
+

- 4ist - ConGREss, - } - SENATE. - { - Ex. - Doc, + 4ist + ConGREss, + } + SENATE. + { + Ex. + Doc, - 3d - Session. - No. - 25. + 3d + Session. + No. + 25.

- MESSAGE + MESSAGE

- OF - THE + OF + THE

- PRESIDENT - OF - THE - UNITED - STATES, + PRESIDENT + OF + THE + UNITED + STATES,

- COMMUNICATING + COMMUNICATING

- A - copy - of - regulations - for - the - consular - courts - of - the - United - States - in - Japan, + A + copy + of + regulations + for + the + consular + courts + of + the + United + States + in + Japan, - decreed - and - issued - by - the - minister - of - the - United - States - in - that - country. + decreed + and + issued + by + the + minister + of + the + United + States + in + that + country.

- JANUARY - 27, - 1871,—Read, - referred - to - the - Committee - on - Commerce, - and - ordered - to - be + JANUARY + 27, + 1871,—Read, + referred + to + the + Committee + on + Commerce, + and + ordered + to + be - printed. + printed.

- To - the - Senate - and - House - of - Representatives - : + To + the + Senate + and + House + of + Representatives + :

- I - transmit - herewith, - for - the - consideration - of - Congress, - a - report - from + I + transmit + herewith, + for + the + consideration + of + Congress, + a + report + from - the - Secretary - of - State, - and - the - papers - which - accompanied - it, - concern- + the + Secretary + of + State, + and + the + papers + which + accompanied + it, + concern- - ing - regulations - for - the - consular - courts - of - the - United - States - in - Japan. + ing + regulations + for + the + consular + courts + of + the + United + States + in + Japan.

- U. - 8. - GRANT. + U. + 8. + GRANT.

- ‘WASHINGTON, - January - 27, - 1871. + ‘WASHINGTON, + January + 27, + 1871.

- DEPARTMENT - OF - STATE, + DEPARTMENT + OF + STATE, - Washington, - January - 26, - 1870, + Washington, + January + 26, + 1870,

- The - Secretary - of - State - has - the - honor - to - submit - herewith, - for - revision + The + Secretary + of + State + has + the + honor + to + submit + herewith, + for + revision - by - Congress, - in - conformity - with - the - provisions - of - section - 6 - of - the - act + by + Congress, + in + conformity + with + the + provisions + of + section + 6 + of + the + act - approved - 22d - of - June, - 1860, - a - copy - of - “regulations - for - the - consular + approved + 22d + of + June, + 1860, + a + copy + of + “regulations + for + the + consular - courts - of - the - United - States - in - Japan,” - decreed - and - issued - by - C. - BE. + courts + of + the + United + States + in + Japan,” + decreed + and + issued + by + C. + BE. - De - Long, - the - minister - of - the - United - States - in - that - country, - in - Septem- + De + Long, + the + minister + of + the + United + States + in + that + country, + in + Septem- - ber, - 1870; - and - also - the - papers - mentioned - in - the - subjoined - list, - which, + ber, + 1870; + and + also + the + papers + mentioned + in + the + subjoined + list, + which, - contain - suggestions - on - the - subject - thereof. + contain + suggestions + on + the + subject + thereof.

- A - copy - of - Article - XXVI - of - the - consular - regulations - is - also - submitted, + A + copy + of + Article + XXVI + of + the + consular + regulations + is + also + submitted, - and - the - Secretary - of - State - respectfully - suggests, - for - the - consideration + and + the + Secretary + of + State + respectfully + suggests, + for + the + consideration - of - Congress, - the - propriety - of - limiting - the - power - of - ministers - to - make + of + Congress, + the + propriety + of + limiting + the + power + of + ministers + to + make - decrees - and - regulation, - in - the - sense - in - which - it - is - limited - by - paragraph + decrees + and + regulation, + in + the + sense + in + which + it + is + limited + by + paragraph - 431 - of - the - article - before - named—that - is, - “to - acts - necessary - to - organize + 431 + of + the + article + before + named—that + is, + “to + acts + necessary + to + organize - and - give - efficiency - to - the - courts - created - by - the - act.” + and + give + efficiency + to + the + courts + created + by + the + act.”

- Respectfully - submitted. + Respectfully + submitted.

- HAMILTON - FISH. + HAMILTON + FISH.

- The - PRESIDENT, + The + PRESIDENT,

- List - of - accompanying - papers. + List + of + accompanying + papers.

- 1, - Regulations - for - the - consular - courts - of - the - United - States - in - Japan. + 1, + Regulations + for + the + consular + courts + of + the + United + States + in + Japan. - 2, - Mr. - Fish - to - Mr. - De - Long, - September - 10, - 1870, + 2, + Mr. + Fish + to + Mr. + De + Long, + September + 10, + 1870,

- +

- +

- +

diff --git a/tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/graph_ocred/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index d89942dab3e48af5a7bb5451992fdf4da74cfacc..30f64b9bebd2a4df7a1b8f32e4df4fac39646850 100644 GIT binary patch delta 27 icmbQLH&t)LB2hj|15-mYLlaXILvvjN^UWJXQyBqra0jUX delta 27 icmbQLH&t)LB2hjILnC7Y0|OItQ%hX~i_IHEQyBqr6$hpO diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 79% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 1ffbc424..99418cee 100644 --- a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,11 +9,11 @@ -
+

- +

diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/jbig2/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index db319e971594ad24d875e4fbdde1a6bc83fc6839..db5de01cc9913b1d766118de8c1e1927393b3c85 100644 GIT binary patch delta 27 icmZ1@wnl8jPEI~c15-mYLlaXILo;0i^UcROQyBqsnFqW8 delta 27 icmZ1@wnl8jPEI}xLnC7Y0|OHa15;fCi_OP5QyBqr{0FiC diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 6ec2a17e..325350eb 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -1,47 +1,65 @@ -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000004.ocr.png__000004__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000004.ocr.png", "$TMPDIR/000004", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000003.ocr.png__000003__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000003.ocr.png", "$TMPDIR/000003", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000005.ocr.png__000005__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000005.ocr.png", "$TMPDIR/000005", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000004.ocr.png__000004.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004.ocr.png", "$TMPDIR/000004.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000003.ocr.png__000003.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003.ocr.png", "$TMPDIR/000003.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000006.ocr.png__000006__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006.ocr.png", "$TMPDIR/000006", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000005.ocr.png__000005.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005.ocr.png", "$TMPDIR/000005.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000006.ocr.png__000006__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006.ocr.png", "$TMPDIR/000006", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000006.ocr.png__000006.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006.ocr.png", "$TMPDIR/000006.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__fra__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "fra", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000002.ocr.png__000002.text__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002.ocr.png", "$TMPDIR/000002.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/2400dpi.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001.ocr.preview.jpg", "stdout"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__osd__--psm__0__000004.ocr.preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004.ocr.preview.jpg", "stdout"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__osd__--psm__0__000003.ocr.preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003.ocr.preview.jpg", "stdout"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__osd__--psm__0__000002.ocr.preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002.ocr.preview.jpg", "stdout"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000002.ocr.png__000002.text__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002.ocr.png", "$TMPDIR/000002.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000003.ocr.png__000003.text__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003.ocr.png", "$TMPDIR/000003.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000004.ocr.png__000004.text__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004.ocr.png", "$TMPDIR/000004.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout", "sourcefile": "resources/poster.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001.ocr.preview.jpg", "stdout"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000001.ocr.png__000001__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000004.ocr.png__000004__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000004.ocr.png", "$TMPDIR/000004", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000003.ocr.png__000003__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000003.ocr.png", "$TMPDIR/000003", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.0.0-x86_64-i386-64bit", "python": "3.7.1", "argv_slug": "__-l__eng__000002.ocr.png__000002__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000002.ocr.png", "$TMPDIR/000002", "hocr", "txt"]} -{"tesseract_version": "tesseract 4.0.0 leptonica-1.77.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.36 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.0 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.2.0-x86_64-i386-64bit", "python": "3.7.2", "argv_slug": "__-l__deu__000001.ocr.png__000001.text__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001.ocr.png", "$TMPDIR/000001.text", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/2400dpi.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/poster.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 60% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 873adcaa..01a67514 100644 --- a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,315 +9,315 @@ -
+

- +

- THEY - TIP-TOED - ALONG. + THEY + TIP-TOED + ALONG.

- ee - . + ee + . - Se - We - went - tip-toeing - along - a - path - amongst + Se + We + went + tip-toeing + along + a + path + amongst

- the - trees - back - towards - the - end - of - the + the + trees + back + towards + the + end + of + the - widow’s - garden, - stooping - down - so - as + widow’s + garden, + stooping + down + so + as - the - branches - wouldn’t - scrape - our - heads. + the + branches + wouldn’t + scrape + our + heads. - When - we - was - passing - by - the - kitchen + When + we + was + passing + by + the + kitchen - I - fell - over - a - root - and - made - a - noise. + I + fell + over + a + root + and + made + a + noise. - We - scrouched - down - and - laid - still. + We + scrouched + down + and + laid + still. - Miss - Watson’s - big - nigger, - named + Miss + Watson’s + big + nigger, + named - Jim, - was - setting - in - the - kitchen - door - ; + Jim, + was + setting + in + the + kitchen + door + ; - we - could - see - him - pretty - clear, - because + we + could + see + him + pretty + clear, + because - there - was - a - light - behind - him. - He + there + was + a + light + behind + him. + He - got - up - and - stretched - his - neck - out + got + up + and + stretched + his + neck + out - about - a - minute, - listening. - Then - he + about + a + minute, + listening. + Then + he - says, + says,

- “Who - dah?” + “Who + dah?”

- He - listened - some - more; - then - he + He + listened + some + more; + then + he - come - tip-toeing - down - and_ - stood + come + tip-toeing + down + and_ + stood - right - between - us; - we - could - a - touched + right + between + us; + we + could + a + touched - him, - nearly. - Well, - likely - it - was - min- + him, + nearly. + Well, + likely + it + was + min- - utes - and - minutes - that - there - warn’t - a + utes + and + minutes + that + there + warn’t + a - sound, - and - we - all - there - so - close + sound, + and + we + all + there + so + close - together. - There - was - a - place - on - my + together. + There + was + a + place + on + my - ankle - that - got - to - itching; - but - I + ankle + that + got + to + itching; + but + I

- dasn’t - scratch - it; - and - then - my - ear - begun - to - itch; - and - next - my - back, - right - be- + dasn’t + scratch + it; + and + then + my + ear + begun + to + itch; + and + next + my + back, + right + be- - tween - my - shoulders. - Seemed - like - I’d - die - if - I - couldn’t - scratch. - Well, - I’ve + tween + my + shoulders. + Seemed + like + I’d + die + if + I + couldn’t + scratch. + Well, + I’ve

- noticed - that - thing - plenty - of - times - since. + noticed + that + thing + plenty + of + times + since.

- Tf - you - are - with - the - quality, - or - at - a + Tf + you + are + with + the + quality, + or + at + a

- funeral, - or - trying - to - go - to - sleep - when - you - ain’t - sleepy—if - you - are - anywheres + funeral, + or + trying + to + go + to + sleep + when + you + ain’t + sleepy—if + you + are + anywheres

diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index b34a5d2d6fc9be5f74d7e462cd9c2cab936e43b0..ce465f7c8cd1d76dd861f5572245b43432f3f007 100644 GIT binary patch delta 27 icmdn0u~lQkAt63X15-mYLlaXI6JuQi^UW88QW*hywg?sg delta 27 icmdn0u~lQkAt62sLnC7Y0|OIFBXeB?i_I5=QW*hyM+gxB diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin index 13be989ca61388e6c1869ba538919e01d07e3c59..60fa7ef5fe5bf7055bb05ebfcb497b07ef0255f5 100644 GIT binary patch delta 27 icmdlczD<0CGZ&wwfvKUHp^2%fk%6v(`Q|{bR7L=4&IaTF delta 27 icmdlczD<0CGZ&wQp^>qHfq|)ksim%g#pXb+R7L=4Tn69( diff --git a/tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/lichtenstein/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin similarity index 66% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/hocr.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin index 4d626250..07cd49a4 100644 --- a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin @@ -9,211 +9,211 @@ -
+

- Replacement - of - "creationism" - with - "intelligent - design" + Replacement + of + "creationism" + with + "intelligent + design"

- +

- + - +

- +

- +

- +

- +

- +

- 120 + 120 - 100 - - + 100 + - - Cc - 80 + Cc + 80 - > + > - 5 + 5 - 5 - 607 - —@— - "Creation" - and - "creationist" + 5 + 607 + —@— + "Creation" + and + "creationist" - 5 - —@— - "Intelligent - design" + 5 + —@— + "Intelligent + design" - = - and - "design - proponent" + = + and + "design + proponent" - 40 - - + 40 + - - 20 - - + 20 + - - —@— - —@® + —@— + —@® - 0 - e— - T - T - T - ' - w - ° + 0 + e— + T + T + T + ' + w + ° - gp) - ee) - 0 - oN - g\ - gD) - op) + gp) + ee) + 0 + oN + g\ + gD) + op) - oO - NC) - NC) - LN - eo - N - N + oO + NC) + NC) + LN + eo + N + N - S - os - o* - vs - ws - os - os + S + os + o* + vs + ws + os + os - cs) - Re - ss - & - ow - x - & - s? + cs) + Re + ss + & + ow + x + & + s? - ge - ee - Oo - ss - Ss - Ne - qs + ge + ee + Oo + ss + Ss + Ne + qs - G - % - S - S - © - S + G + % + S + S + © + S - Ros - % - se - se - oe - AN + Ros + % + se + se + oe + AN - Ss - 3s - S - Ss - Ss - Ss - Ss + Ss + 3s + S + Ss + Ss + Ss + Ss - ow? - \O - Xo) - R - g - g - Q + ow? + \O + Xo) + R + g + g + Q

diff --git a/tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000002.ocr.png__000002.text__pdf__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin index 842a84b672df8dc44f9b82c771bc5edf83fa189c..a59aaf6198443e101ffc76be5decccf72bbc6fcc 100644 GIT binary patch delta 27 icmZowY*pMK#Ls7GU}|V)Xkuz&Y^G~qzFC1kl@S14fd)hX delta 27 icmZowY*pMK#Ls78Xk=_)U|?ctWUOmou~~sXl@S13-3B=T diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003.text__pdf__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin similarity index 67% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin index 755b0b97..2daad740 100644 --- a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin @@ -9,189 +9,189 @@ -
+

- Replacement - of - "creationism" - with - "intelligent - design" + Replacement + of + "creationism" + with + "intelligent + design"

- +

- + - +

- +

- +

- +

- +

- +

- 120 + 120 - 100 - 4 + 100 + 4 - = - 80 + = + 80 - — + — - S + S - _ - 6047 - —@— - "Creation" - and - "creationist" + _ + 6047 + —@— + "Creation" + and + "creationist" - 5 - —@— - "Intelligent - design" + 5 + —@— + "Intelligent + design" - and - "design - proponent" + and + "design + proponent" - S - «4 + S + «4 - 20 - - + 20 + - - 0 - oe - I - T - T - T - T - © + 0 + oe + I + T + T + T + T + © - 3) - ©) - Ay - Ay - Ay - 9 - o>) + 3) + ©) + Ay + Ay + Ay + 9 + o>) - ee - ow - oe - oe - Cs - Cs - eS + ee + ow + oe + oe + Cs + Cs + eS - RQ - Q - R - R - XR - Q - R + RQ + Q + R + R + XR + Q + R - & - es - o - a - al - & - eo + & + es + o + a + al + & + eo - 3 - e - oe - we - a? - i) - as? + 3 + e + oe + we + a? + i) + as? - 3 - 4? - 3 - & - oe - & - & + 3 + 4? + 3 + & + oe + & + & - oe - ww - 6 - e - Qe - Qe - Qe + oe + ww + 6 + e + Qe + Qe + Qe

diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000003.ocr.png__000003__hocr__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin index 5262d6da30d6534f58feefe4588bad785534fae7..603059cf45eae5ab7b28a9617bf1e03c6f9723db 100644 GIT binary patch delta 27 icmbOvKS_RrJRhH>fvKUHp^2%9v6-%c`DR_dR7L<``v!6V delta 27 icmbOvKS_RrJRhHhp^>qHfq{vok+H6U#b#Z;R7L<`R|aPQ diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004.text__pdf__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin similarity index 57% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/hocr.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin index 3510c868..edb5eee7 100644 --- a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin @@ -9,689 +9,689 @@ -
+

- with - a - plain - face, - on - the - throne - of - England; + with + a + plain + face, + on + the + throne + of + England; - there - were - a - king - with - a - large - jaw - and - a - queen + there + were + a + king + with + a + large + jaw + and + a + queen - with - a - fair - face, - on - the - throne - of - France. - In - both + with + a + fair + face, + on + the + throne + of + France. + In + both - countries - it - was - clearer - than - crystal - to - the - lords + countries + it + was + clearer + than + crystal + to + the + lords - of - the - State - preserves - of - loaves - and - fishes, - that + of + the + State + preserves + of + loaves + and + fishes, + that - things - in - general - were - settled - for - ever. + things + in + general + were + settled + for + ever.

- It - was - the - year - of - Our - Lord - one - thousand + It + was + the + year + of + Our + Lord + one + thousand - seven - hundred - and - seventy-five. - Spiritual - reve- + seven + hundred + and + seventy-five. + Spiritual + reve- - lations - were - conceded - to - England - at - that + lations + were + conceded + to + England + at + that - favoured - period, - as - at - this, - Mrs. - Southcott - had + favoured + period, + as + at + this, + Mrs. + Southcott + had - recently - attained - her - five-and-twentieth - blessed + recently + attained + her + five-and-twentieth + blessed - birthday, - of - whom - a - prophetic - private - in - the - Life + birthday, + of + whom + a + prophetic + private + in + the + Life - Guards - had - heralded - the - sublime - appearance - by + Guards + had + heralded + the + sublime + appearance + by - announcing - that - arrangements - were - made - for - the + announcing + that + arrangements + were + made + for + the - swallowing - up - of - London - and - Westminster. + swallowing + up + of + London + and + Westminster. - Even - the - Cock-lane - ghost - had - been - laid - only - a + Even + the + Cock-lane + ghost + had + been + laid + only + a - round - dozen - of - years, - after - rapping - out - its - mes- + round + dozen + of + years, + after + rapping + out + its + mes- - sages, - as - the - spirits - of - this - very - year - last - past + sages, + as + the + spirits + of + this + very + year + last + past - (supematurally - deficient - in - originality) - rapped + (supematurally + deficient + in + originality) + rapped - out - theirs. - Mere - messages - in - the - earthly - order - of + out + theirs. + Mere + messages + in + the + earthly + order + of - events - had - lately - come - to - the - English - Crown - and + events + had + lately + come + to + the + English + Crown + and - People, - from - a - congress - of - British - subjects - in + People, + from + a + congress + of + British + subjects + in - America: - which, - strange - to - relate, - have - proved + America: + which, + strange + to + relate, + have + proved - more - important - to - the - human - race - than - any - com- + more + important + to + the + human + race + than + any + com- - munications - yet - received - through - any - of - the + munications + yet + received + through + any + of + the - chickens - of - the - Cock-lane - brood. + chickens + of + the + Cock-lane + brood.

- France, - less - favoured - on - the - whole - as - to - mat- + France, + less + favoured + on + the + whole + as + to + mat- - ters - spiritual - than - her - sister - of - the - shield - and - tri- + ters + spiritual + than + her + sister + of + the + shield + and + tri- - dent, - rolled - with - exceeding - smoothness - down + dent, + rolled + with + exceeding + smoothness + down - hill, - making - paper - money - and - spending - it. - Under + hill, + making + paper + money + and + spending + it. + Under - the - guidance - of - her - Christian - pastors, - she - enter- + the + guidance + of + her + Christian + pastors, + she + enter- - tained - herself, - besides, - with - such - humane + tained + herself, + besides, + with + such + humane - achievements - as - sentencing - a - youth - to - have - his + achievements + as + sentencing + a + youth + to + have + his

- hands - cut - off, - his - tongue - torn - out - with - pincers, + hands + cut + off, + his + tongue + torn + out + with + pincers, - and - his - body - burned - alive, - because - he - had - not + and + his + body + burned + alive, + because + he + had + not - kneeled - down - in - the - rain - to - do - honour - to - a - dirty + kneeled + down + in + the + rain + to + do + honour + to + a + dirty - procession - of - monks - which - passed - within - his + procession + of + monks + which + passed + within + his - view, - at - a - distance - of - some - fifty - or - sixty - yards. - It + view, + at + a + distance + of + some + fifty + or + sixty + yards. + It - is - likely - enough - that, - rooted - in - the - woods - of + is + likely + enough + that, + rooted + in + the + woods + of - France - and - Norway, - there - were - growing - trees, + France + and + Norway, + there + were + growing + trees, - when - that - sufferer - was - put - to - death, - already + when + that + sufferer + was + put + to + death, + already - marked - by - the - Woodman, - Fate, - to - come - down + marked + by + the + Woodman, + Fate, + to + come + down - and - be - sawn - into - boards, - to - make - a - certain - mov- + and + be + sawn + into + boards, + to + make + a + certain + mov- - able - framework - with - a - sack - and - a - knife - in - it, - ter- + able + framework + with + a + sack + and + a + knife + in + it, + ter- - rible - in - history. - It - is - likely - enough - that - in - the + rible + in + history. + It + is + likely + enough + that + in + the - rough - outhouses - of - some - tillers - of - the - heavy + rough + outhouses + of + some + tillers + of + the + heavy - lands - adjacent - to - Paris, - there - were - sheltered + lands + adjacent + to + Paris, + there + were + sheltered - from - the - weather - that - very - day, - rude - carts, + from + the + weather + that + very + day, + rude + carts, - bespattered - with - rustic - mire, - snuffed - about - by + bespattered + with + rustic + mire, + snuffed + about + by - pigs, - and - roosted - in - by - poultry, - which - the + pigs, + and + roosted + in + by + poultry, + which + the - Farmer, - Death, - had - already - set - apart - to - be - his + Farmer, + Death, + had + already + set + apart + to + be + his - tumbrils - of - the - Revolution. - But - that - Woodman + tumbrils + of + the + Revolution. + But + that + Woodman - and - that - Farmer, - though - they - work - unceasingly, + and + that + Farmer, + though + they + work + unceasingly, - work - silently, - and - no - one - heard - them - as - they + work + silently, + and + no + one + heard + them + as + they - went - about - with - muffled - tread: - the - rather, - foras- + went + about + with + muffled + tread: + the + rather, + foras- - much - as - to - entertain - any - suspicion - that - they + much + as + to + entertain + any + suspicion + that + they - were - awake, - was - to - be - atheistical - and - traitorous. + were + awake, + was + to + be + atheistical + and + traitorous.

- In - England, - there - was - scarcely - an - amount - of + In + England, + there + was + scarcely + an + amount + of - order - and - protection - to - justify - much - national + order + and + protection + to + justify + much + national - boasting. - Daring - burglaries - by - armed - men, - and + boasting. + Daring + burglaries + by + armed + men, + and - highway - robberies, - took - place - in - the - capital + highway + robberies, + took + place + in + the + capital - itself - every - night; - families - were - publicly - cau- + itself + every + night; + families + were + publicly + cau- - tioned - not - to - go - out - of - town - without - removing + tioned + not + to + go + out + of + town + without + removing - their - furniture - to - upholsterers' - warehouses - for + their + furniture + to + upholsterers' + warehouses + for - security; - the - highwayman - in - the - dark - was - a - City + security; + the + highwayman + in + the + dark + was + a + City - tradesman - in - the - light, - and, - being - recognised - and + tradesman + in + the + light, + and, + being + recognised + and

diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000004.ocr.png__000004__hocr__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin index 397f32093208dedbb7e0c499e235decaeb49cfbb..1c10f1167bb81026a7e62cc79ed87ff9ee3b69ff 100644 GIT binary patch delta 27 icmccablqvg6L~&M15-mYLlaXI6AN7f^Ua^+QyBq^iwN-m delta 27 icmccablqvg6L~%hLnC7Y0|OIFV?$j7i_M?pQyBq@!wBL4 diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005.text__pdf__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin similarity index 56% rename from tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/hocr.bin rename to tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin index 4f7b0583..a4d97447 100644 --- a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin @@ -9,690 +9,690 @@ -
+

- with - a - plain - face, - on - the - throne - of - England; + with + a + plain + face, + on + the + throne + of + England; - there - were - a - king - with - a - large - jaw - and - a - queen + there + were + a + king + with + a + large + jaw + and + a + queen - with - a - fair - face, - on - the - throne - of - France. - In - both + with + a + fair + face, + on + the + throne + of + France. + In + both - countries - it - was - clearer - than - crystal - to - the - lords + countries + it + was + clearer + than + crystal + to + the + lords - of - the - State - preserves - of - loaves - and - fishes, - that + of + the + State + preserves + of + loaves + and + fishes, + that - things - in - general - were - settled - for - ever. + things + in + general + were + settled + for + ever.

- It - was - the - year - of - Our - Lord - one - thousand + It + was + the + year + of + Our + Lord + one + thousand - seven - hundred - and - seventy-five. - Spiritual - reve- + seven + hundred + and + seventy-five. + Spiritual + reve- - lations - were - conceded - to - England - at - that + lations + were + conceded + to + England + at + that - favoured - period, - as - at - this. - Mrs. - Southcott - had + favoured + period, + as + at + this. + Mrs. + Southcott + had - recently - attained - her - five-and-twentieth - blessed + recently + attained + her + five-and-twentieth + blessed - birthday, - of - whom - a - prophetic - private - in - the - Life + birthday, + of + whom + a + prophetic + private + in + the + Life - Guards - had - heralded - the - sublime - appearance - by + Guards + had + heralded + the + sublime + appearance + by - announcing - that - arrangements - were - made - for - the + announcing + that + arrangements + were + made + for + the - swallowing - up - of - London - and - Westminster. + swallowing + up + of + London + and + Westminster. - Even - the - Cock-lane - ghost - had - been - laid - only - a + Even + the + Cock-lane + ghost + had + been + laid + only + a - round - dozen - of - years, - after - rapping - out - its - mes- + round + dozen + of + years, + after + rapping + out + its + mes- - sages, - as - the - spirits - of - this - very - year - last - past + sages, + as + the + spirits + of + this + very + year + last + past - (supernaturally - deficient - in - originality) - rapped + (supernaturally + deficient + in + originality) + rapped - out - theirs. - Mere - messages - in - the - earthly - order - of + out + theirs. + Mere + messages + in + the + earthly + order + of - events - had - lately - come - to - the - English - Crown - and + events + had + lately + come + to + the + English + Crown + and - People, - from - a - congress - of - British - subjects - in + People, + from + a + congress + of + British + subjects + in - America: - which, - strange - to - relate, - have - proved + America: + which, + strange + to + relate, + have + proved - more - important - to - the - human - race - than - any - com- + more + important + to + the + human + race + than + any + com- - munications - yet - received - through - any - of - the + munications + yet + received + through + any + of + the - chickens - of - the - Cock-lane - brood. + chickens + of + the + Cock-lane + brood.

- France, - less - favoured - on - the - whole - as - to - mat- + France, + less + favoured + on + the + whole + as + to + mat- - ters - spiritual - than - her - sister - of - the - shield - and - tri- + ters + spiritual + than + her + sister + of + the + shield + and + tri- - dent, - rolled - with - exceeding - smoothness - down + dent, + rolled + with + exceeding + smoothness + down - hill, - making - paper - money - and - spending - it. - Under + hill, + making + paper + money + and + spending + it. + Under - the - guidance - of - her - Christian - pastors, - she - enter- + the + guidance + of + her + Christian + pastors, + she + enter- - tained - herself, - besides, - with - such - humane + tained + herself, + besides, + with + such + humane - achievements - as - sentencing - a - youth - to - have - his + achievements + as + sentencing + a + youth + to + have + his

- hands - cut - off, - his - tongue - torn - out - with - pincers, + hands + cut + off, + his + tongue + torn + out + with + pincers, - and - his - body - burned - alive, - because - he - had - not + and + his + body + burned + alive, + because + he + had + not - kneeled - down - in - the - rain - to - do - honour - to - a - dirty + kneeled + down + in + the + rain + to + do + honour + to + a + dirty - procession - of - monks - which - passed - within - his + procession + of + monks + which + passed + within + his - view, - at - a - distance - of - some - fifty - or - sixty - yards. - It + view, + at + a + distance + of + some + fifty + or + sixty + yards. + It - is - likely - enough - that, - rooted - in - the - woods - of + is + likely + enough + that, + rooted + in + the + woods + of - France - and - Norway, - there - were - growing - trees, + France + and + Norway, + there + were + growing + trees, - when - that - sufferer - was - put - to - death, - already + when + that + sufferer + was + put + to + death, + already - marked - by - the - Woodman, - Fate, - to - come - down + marked + by + the + Woodman, + Fate, + to + come + down - and - be - sawn - into - boards, - to - make - a - certain - mov- + and + be + sawn + into + boards, + to + make + a + certain + mov- - able - framework - with - a - sack - and - a - knife - in - it, - ter- + able + framework + with + a + sack + and + a + knife + in + it, + ter- - rible - in - history. - It - is - likely - enough - that - in - the + rible + in + history. + It + is + likely + enough + that + in + the - rough - outhouses - of - some - tillers - of - the - heavy + rough + outhouses + of + some + tillers + of + the + heavy - lands - adjacent - to - Paris, - there - were - sheltered + lands + adjacent + to + Paris, + there + were + sheltered - from - the - weather - that - very - day, - rude - carts, + from + the + weather + that + very + day, + rude + carts, - bespattered - with - rustic - mire, - snuffed - about - by + bespattered + with + rustic + mire, + snuffed + about + by - pigs, - and - roosted - in - by - poultry, - which - the + pigs, + and + roosted + in + by + poultry, + which + the - Farmer, - Death, - had - already - set - apart - to - be - his + Farmer, + Death, + had + already + set + apart + to + be + his - tumbrils - of - the - Revolution. - But - that - Woodman + tumbrils + of + the + Revolution. + But + that + Woodman - and - that - Farmer, - though - they - work - unceasingly, + and + that + Farmer, + though + they + work + unceasingly, - work - silently, - and - no - one - heard - them - as - they + work + silently, + and + no + one + heard + them + as + they - went - about - with - muffled - tread: - the - rather, - foras- + went + about + with + muffled + tread: + the + rather, + foras- - much - as - to - entertain - any - suspicion - that - they + much + as + to + entertain + any + suspicion + that + they - were - awake, - was - to - be - atheistical - and - traitorous. + were + awake, + was + to + be + atheistical + and + traitorous.

- In - England, - there - was - scarcely - an - amount - of + In + England, + there + was + scarcely + an + amount + of - order - and - protection - to - justify - much - national + order + and + protection + to + justify + much + national - boasting. - Daring - burglaries - by - armed - men, - and + boasting. + Daring + burglaries + by + armed + men, + and - highway - robberies, - took - place - in - the - capital + highway + robberies, + took + place + in + the + capital - itself - every - night; - families - were - publicly - cau- + itself + every + night; + families + were + publicly + cau- - tioned - not - to - go - out - of - town - without - removing + tioned + not + to + go + out + of + town + without + removing - their - furniture - to - upholsterers' - warehouses - for + their + furniture + to + upholsterers' + warehouses + for - security; - the - highwayman - in - the - dark - was - a - City + security; + the + highwayman + in + the + dark + was + a + City - tradesman - in - the - light, - and, - being - recognised - and + tradesman + in + the + light, + and, + being + recognised + and

diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/stderr.bin rename to tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000005.ocr.png__000005__hocr__txt/stdout.bin rename to tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006.text__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000006.ocr.png__000006.text__pdf__txt/txt.bin rename to tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006.text__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/multipage/__-l__eng__000006.ocr.png__000006.text__pdf__txt/pdf.bin rename to tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin index 2fa051d2dc1b4f62c542fe227f0e36ef9a164263..3dc65020979a474a3659ffd6bff360b9b33b2b0b 100644 GIT binary patch delta 27 icmexk_s4DnryQT9fvKUHp^2%9sfDhA`DSssR7L=M`v-gg delta 27 icmexk_s4DnryQS!p^>qHfq{vov5BsM#b$B2R7L=ML -
+

- 41st - CONGRESS, - } - SENATE. + 41st + CONGRESS, + } + SENATE. - 3d - Session. + 3d + Session.

- MESSAGE + MESSAGE

- OF - THE + OF + THE

- PRESIDENT - OF - THE - UNITED - STATES + PRESIDENT + OF + THE + UNITED + STATES

- A - copy - of - regulations - for - the - consular - courts - of - the - United - States - in - Japan, + A + copy + of + regulations + for + the + consular + courts + of + the + United + States + in + Japan, - decreed - and - issued - by - the - minister - of - the - United - States - in - that - country. + decreed + and + issued + by + the + minister + of + the + United + States + in + that + country.

- Janvary - 27, - 1871—Read, - referred - to - the - Committee - on - Commerce, - and - ordered - to - be + Janvary + 27, + 1871—Read, + referred + to + the + Committee + on + Commerce, + and + ordered + to + be - printed, + printed,

- To - the - Senate - and - House - of - Representatives - : + To + the + Senate + and + House + of + Representatives + :

- I - transmit - herewith, - for - the - consideration - of - Congress, - a - report - from + I + transmit + herewith, + for + the + consideration + of + Congress, + a + report + from - the - Secretary - of - State, - and - the - papers - which - accompanied - it, - concern- + the + Secretary + of + State, + and + the + papers + which + accompanied + it, + concern- - ing - regulations - for - the - consular - courts - of - the - United - States - in - Ja + ing + regulations + for + the + consular + courts + of + the + United + States + in + Ja

- U. - 8. - GRAN + U. + 8. + GRAN

- WASHINGTON, - January - 27, - 1871. + WASHINGTON, + January + 27, + 1871.

- DEPARTMENT - OF - STATE, + DEPARTMENT + OF + STATE, - Washington, - January - 26, - 1870. + Washington, + January + 26, + 1870. - The - Secretary - of - State - has - the - honor - to - submit - herewith, - for - revision + The + Secretary + of + State + has + the + honor + to + submit + herewith, + for + revision - by - Congress, - in - conformity - with - the - provisions - of - section - 6 - of - the - act + by + Congress, + in + conformity + with + the + provisions + of + section + 6 + of + the + act - approved - 22d - of - June, - 1860, - a - copy - of - “regulations - for - the - consular + approved + 22d + of + June, + 1860, + a + copy + of + “regulations + for + the + consular - courts - of - the - United - States - in - Japan,” - decreed - and - issued - by - C. - E. + courts + of + the + United + States + in + Japan,” + decreed + and + issued + by + C. + E. - De - Long, - the - minister - of - the - United - States - in - that - country, - in - Septem- + De + Long, + the + minister + of + the + United + States + in + that + country, + in + Septem- - ber, - 1870; - and - also - the - papers - mentioned - in - the - subjoined - list, - which + ber, + 1870; + and + also + the + papers + mentioned + in + the + subjoined + list, + which - contain - suggestions - on - the - subject - thereof. - : + contain + suggestions + on + the + subject + thereof. + : - A - copy - of - Art - XVI - of - the - consist - regulations - so - Submitted, + A + copy + of + Art + XVI + of + the + consist + regulations + so + Submitted, - and - the - Secretary - of - r - i - ly - , - for - the - consideration + and + the + Secretary + of + r + i + ly + , + for + the + consideration - of - ministers - to - make + of + ministers + to + make - ion, - in - the - sense - in - which - it - is - limited - by - paragraph + ion, + in + the + sense + in + which + it + is + limited + by + paragraph - 431 - of - the - e - be: - fore - named—that - is, - “ - to - acts - necessary - to - organize + 431 + of + the + e + be: + fore + named—that + is, + “ + to + acts + necessary + to + organize - and - give - efficiency - to - the - courts - created - by - the - act.” + and + give + efficiency + to + the + courts + created + by + the + act.” - Respectful - submitted. + Respectful + submitted. - HAMILTON - FISH. + HAMILTON + FISH. - The - PRESIDENT. + The + PRESIDENT.

- List - of - accompanying - papers. + List + of + accompanying + papers.

- 1. - Regulations - for - the - consular - courts - of - the - United - States - in - Japan. + 1. + Regulations + for + the + consular + courts + of + the + United + States + in + Japan. - 2. - Mr. - Fish - to - Mr. - De - Long, - September - 10, - 1870, + 2. + Mr. + Fish + to + Mr. + De + Long, + September + 10, + 1870,

- +

diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/stdout.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/stdout.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 68839285905c4b953f0382505d8249a9f99ee1b6..8ed0fa16996f97f73d94aab6524cf2921c4fbe99 100644 GIT binary patch delta 27 icmeyM^Fe3BUlBe_15-mYLlaXIBO_e{^Ud6%sf+-LjtC3@ delta 27 icmeyM^Fe3BUlBeFLnC7Y0|OItQwv=Ki_P4ksf+-LO$ZDC diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin deleted file mode 100644 index c09b9eb338524f518efd0ecc042e1b667055504a..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 10889 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a|v zV4!DUpkQnUrolC_bADb)YF;6uzK5{+o#2n|S$C!C?_WQkpT4z8R5;H5|Ld>o*Vmh~a;S7FPRDuiUrf##)i%uQ_J-CSAOFQ8}mniu|4*KIUh(#XmV5wq|zs ze>eLrKLl>guW_~fcJf12&b;tD`{VXEFHYa~-SBho1HN_fSJ(Zx7%uzeS2}RX!qoo>5qR$x9qcjdw>7uyGr}Fn?@V_u&UkuW9LP`##Jg(!8zaAkLs~-vOJQ$ z!{&PW*RH1`KM$6DRyg;yzu-m|3(c*yWk>&%>2+4yCi(x1Wld;~~?Y z6ng2*j3e1^*I)EcuXnZo{`to7e_ip%>u&wbDsj$sn6gi{$s+GS{Wsg9#{oQCOO#9X z6IN?+I2-IXdn@mwlD%;a>-0BMO7DKEQT%-|Co@`MrlbA(jBM@WYN=r#^8fk=t+VOr zHvY>xTT(Yj;Ff2Krohho2YO?bitpt-Td16=Etwbe`{+ZbmqrnrYF|fq*RlUSzg+ct zW?A;VP1Cb>#-H%|EXuN^<*Qb?=G*xn%hpXecgHf~!S#LhvAyk&mfAd3nLOc|^wq^% z{d>b78L((IF-**r-8Jiy*jh39^fjBZgXKN4Cdpwxr8nhsz1gw6C}k`2$A_cl?w#p3p={lb ztp@Mw>z*b{>=rv{Ke1qk_R4dgdsZthkMWz9bN}W)wKsdNF{mw^tIGata?OgTx=CL( zK2+X1S2bV?zGe!b-@5S7UF7PDaz$~~ZU;dt}=@LkL|&aN(0mlpZLIK}#F=)3c5xAX6AyrjQx z@w#tMHb_pnWHMi(c>npAFI6~#&v9z~tcm(nS9o~w{jWzAo33&^dCc%)Zcm(BkHZYD z2H#u$kKOO9-u_;BL*%uf!o<0;4;H^ySYy9Sp?brD@5>&Fw8$(|oP4K&vrGH7L$~&W zrOqt5iHSOFE%99M8QmhBPk&joM8#_#+PP_PJ zCwaXSY&LP%V1LujICJIS_pBct@VyP&$a*w|ul#e0?o@$o>gQ)%4H9^MpM$SOuH?+t zU9;QQJ_>Y_Qu>i7A9r!ulf#A!bQfBfDW?@ER5Qjcs$!J*Uah?3T!De|gM%mjt~{K@ zxZ2w*f8XL}L+|vGmGQ3*?!MM@zW2t(TsFQ6$(fhM9!M_Kl1u0ObTWG3+Q}RXJlvDH z4!y5Y`og4U&7LqjXYbjdHRb_-Mumujam9u?MnSq zN;2PkW_%T^P<`W$sYLbm#|IoHZr@@$bFIXY56Zv()iNFUtMgXaUwBQZ&dHAj9s8yW zOc8q?pyt4LgX7H;Cni~$Uut(u9M3l0ONzI;c|5;6pqcOLLOs^DwF!Aa^O(XO33eQM z{{8CKY@rufdI=0C?yT1kb!&Rjnqe!PA^4{B;@rIxIDSPj?D^!db+%QBj*h3~e2>6I z+^46?>G?(W%;GTA_`rAL8-vRWvD||ei{JR~ZVBLAl+zT6^KYqGu2ZGo`kZ(Wr&)1GfWbN1 z&(@D?G!Nf%H7?y|Ym{Lm)L8PYA>v-a427_Jn+zOxEAMRh)cf_>_p&U(=19KGQ=E+k z=?`4~zhH_r?5M2O{OG_HZx!qjUMIClV8yWY!N++(m@6v}k7J1D+R?_^hyCZI1ZeRP$*3EzIt;YF_diHUt+$-qIX>p? z(e7`pCSCluabvQB)?0pE*~6O+)3S=^w6njynIJCl!Q}jW$!>F%^Mbj%#GY-NJx^wb z$+^p&)d>Q%PMbGg4iOg7XkVVo!>Q=X%86Ny)M8+n$`iz@ zXe0Zxr=!LHW5O>DfuzTg&%C}}?cvSdIqe`Dqj-?Um38G6-#0G#W%{gEY2_ACv8TQF z9d0CsDO&90T$`iqEAYly`13PP7VT)}imMI>s{%D%Kl#P$wRI`TI#T`fI#h(pZ z@Z?fw$*#wb6`!|=va00i-G9D6eVKTyc5Z;CcR<009X(YpTixzzxp=*1x!&ZxVV19z zYKP38pxY`ovsX`%om=d)m+z3QWtXy&t4qtDyABJLTAlXKE_wJqEyB`onrU84z`@fD ztcJ#N^O%Z0w|SnESg~b|-X>oGezj|ft6Th+HF;(RY@V>L$RXQPGj`Uyb2E&#n7PfY z3X@`2tjfLe)Axa_K%T%U){_c&IVfB+S*9atE3zga-NMg$uEF^;9EBE1ENA^1W*&>> zI(@PB&vd;Vx13GlyIvk&__Y0x=P}-jg4DWWHy;~a{PSi_%c1p0HNv=a4I=+%-!&c2+Nt%Saf6POaEikM)%?&) zy48KHttS>PVB+>XaHe`^`>!5@#HDkWCFs6SO*Yq9>Uqubufen`u0@R*VWPZj7kjm3 zJ7aQ1e}y<5&FIM&FHB0*xu#(4&92SewB-Bdh`O9>x1Xi1U9zHh!*RJK_tn))l_VyV zu67sMyX4Tux^*Xf_`>;KHf=6s>+6 z?ti;Af9LgSb!+`)9ks*vd{`Lave*5ChsLqi8!M-*<%kpD+^J)oxM$n`j5QmDszgQq z^|x51_-#Dj@bO27ZjVcdD%W45UH|S%YaD3)EM&SiKVF8rB(=i2V81zcQ%~(i@mT$< z@hy{lJ5GIEv-$@2-{YZ6Y!uI~KVlK*(RX~4z+In5scGw3PVVcO)_y@)*}vy{Lx>2= z0Zz|8hfCgauLW5drnpL`ZTOZclCgvP`W~ejYnDc=(@?+i?u?vvg6-lGq2F;mWvZf9 zN0$@^pAJ2C`Q*u-OG+Ou$K~u{7319MY#UziroB5_^yZ_FVX@5ZsuC}B>y?YUH8*d5 z*mn4wR>jd{207n^0_O+b{u_3MPod~#bM8us`!Rtb1s!`09vb~yt8`{h#C-woz-K!C zU#I;$sd6DME9a>L|3e;^t?8GtYUeQ?^3u*?zqcrE(<@Ky!=1t`=|86A2hUzr{dUW$ z4&P8d9~DP2A-8Y3jy@NiP9Hp2P&uc-K+>{vgX`U6s!yJ)WhMl_y7Ybeu{<}ifA!W~ zlB^p9qdX$Gb{kl%kaAMLB{9`lv`juKCyKS1ds4xj#A|%p#eDYJY)<5?Taa*gbBVbq zTeP;2U~*8PpNu(!ucL=Z%Yn6<7wH{Rx+qwteDh|{gu*=S2MgT7kF2;LJ^jzuMIWbf zc^ps)nEfQqMx6_6eYUMC;jB?BcHaEP^V`Cea&EF6aVopsN$pD0_3XC2q?(fRGLC2N`d$Ce zh4Rec%H1k{s>xWr*|he8mBJ1g2a!YP6|b+q$X#B)nPuM z`2JOaf8UBWnm>(EUBA?KUyq*0vPpZD>%@MtUV7fJ*oBG({|PGledI&8P-R`vf$%KP^g@r*lR_TG*BsSbw|RlnvsI?KhibY{m;CNK z!~bsG%LIMa+y~3Fo>$G(m?9VV?laeJyA-Q?J5P9S)!1C&ZO*jT@fbJz+7mekJ0jex zwngj}Q&tb_nJ&8X#|}ve@d*VC%>~(RmzcwOgFC|I)9>1M^v_zA`2M!7=A|YVgJ)M` zts--0hG%9Ou34zs`RQXvYPbn70vxTX`id~xnLvqShESfT0T)) ze?{XJvX`#gRLd)&^7~%A{?UMgA_({zU$}%cr8#`zl~~a8EyN?yFUs}VT zKDB=V6Q+4T^SpAkQ)$MY3JA5S} zUra80{Tk!+kj?jOE=^lCE7ACR@#X?+3uT@4^&w}u3VWV0&sR}4x%OA(?bV`hOTSg5 zR?Jkp;c3nJ(E4h?7KY_#gpTGe*llAjnHKuWx#H1^pl*gc(ei5@e|%e`%{$wP<#QKr zx($DB=G{`ypR0`W(?aL7#jfjBurBslG}pgDZ0ftxf^QPZ%PLO&UH$dF^l?Yq=B^b{ z*QVA_zk4?D%YM<2C;gi*_<#4_e?&F*K6mwn=ci_^X$bnD<`yht`CT)qkWsjIc~QrC z?aK8Jjnu6CuCf;V`kL@#;?rzN`Ai;p)kwjl`NjszXT0ny=C_K~?_D7rdwj9d=~I-OhAu zHu85ex^-E0%f|i1Gpe21PPe3Oo8L1juvF~y8r^H+ArA!&geJr+*t$*VUsz3tB$1k z``zJ|zc%fZv6Gil_w{d3Sb8l$rmuI0#`2?Uj<|B@xR-kQ887aeGr7&&m(R+t zmRR&@i`2^s{l!KpGtLJ!^Gf9|nv-QbYf7AuGjGYp|C!9;kInq+a-PPB86M8Of5c|n z1oM1RTWbkPjt+V8K&~UT_meq0qHkZ=C#kgm^~TVdC7MrnJ27obzp&(2;PunY`Lnv7 z#g|U~5Siq1D7bRfKc0)}Qxln2iC%unAX4CYVb05M-0sV>18bkoXZ`D^usQ1C%(R?E zUWo-uKG}uJNZ-DH^+AY)+VbB!zC2Tss?xjhYFYSEz4;~l`&V~m*rw&T#F`ka>Wfx+ zJn@DI=Ng@JpKhu9*h-e^Z(lokQ)yGh466fY3MO2hI9^q_i#HrU zv#R9e|G4nib2kUX@Ef_kDhZkT(_+K4L&Y=Suh9?KRE?D!fdwt`KU$KD6BFUty#N`J~ekg==3;+3l^ zlgKk?)k9KR>!e<%Z2Q6JvZe1%OK7NS?qhdt#!VmO`L@kap7c;{)r=(l23a09lZ#t8 z+xb0ed~B0VR~?J{zGT9y-H9&`)y=43IQP0yaa^Gr%pV`*s&*Y@ie76`-^Wg*n;#7d;gnlyi*Yny0M{JZ1f|s3l{q@kG(4LRtM?Ds0xxYPiJMyf=o{787yA#4Q{@!c3v0w0U zQk?UYw*iahoMD_DedboKk?=|*MFZzc-5)RAIJ5sxo6r%TU9+C>C7N4AJc(OZVdALL zWq2*?i)q%p7gnwsn(2NIq83~5o6cZgY|`~%Pppl&R^Ep6f*0Yg4X$P;l3O>rChdy! z->@$4cZcROgOd-wT#NtbwX5T=@+NgwbDu5Wz0DUToc**~v|#!eSAsKlKP!W`YjSC&vcc@$4otXy~bSpz(Jd*r}AE%p1yuZ zxx9RvtKXXg4ar~c)_i__Wy22d3`uR1C}suUc`^)-RHts7`m9<1#Ppx13Jql3A2mPO zV$tusz+K}^sC_fz>7)d2&GHVh-6?tkxf4&STr@e9{k(6*dzmzenU@YHT`re3t8P1X zHMrgPMf{UCm!mVzbgs_%x-@Ijl2r^M*E~XM=0!`ny)fDI-fx4q-}me_Df={nzJKCo z$i4`u%(_&d>OsivWyPkZaztHyD*Z0;J zk1Z^mwJ2cQnW(nFWc^tOqa0#Gd0DT|-B25Ay<25}@#~-r$BrkzPZ!t5gyb%C+27j~ zP!l}y{PWlL=f3zm-zmOVwfj&tct7=Z^}EHD^FR6qx}JF-wCL2-lWBhIBusr- zy#t$y-fZ8@{zoa)+J))0c$xLGefo#@b8<&(nA|<+9J9!3@pa)7Q~CeqgO?A?jdh%K zM}cS0bJ3p%mTlPBTkDfLN&LzGd`<(7U9B5t-YHvMR+f|FvRbt?Uj5}93-5<}gH<+P z%L~own3it1)&H9_Pucp%hi^LZHfknhn%h5AXS|xFompvc_eEA8PPl*G=dt2?G4c5u-dZ?s_vljjc$xR#n<&{o z=jX=0KWBG>MX0^Y@b#z1F4x~)j5vSg;LOzrzzO~xiwNpEf zDERYih+E9RN4b6Po=HL<*SwgX6wH0#VM;i^?7xV!=Xjm&H*M7@4dE@7i`l$0f13Q| zsLMAttl`VnIQ=BweO`}yqjUI%wfE*uO__RteX7sfSM4g>!!86aKM?R&f9qH2%a6B* z*{hd7>z$gpw&w8v4<6rY-rZ2XeI)kO#^+}rtcr7GXmob1Ul~$3spV3j@ujHGw%H{_Teoc5b$4y#q}#^Y$6Sr=Jr4=K&{aIE{HadmkN&%L*VL|VKD=q8_U6<( ze~-337fll1p(_`&b=g|sl%i6v0M4>4`3FS(*338)7p0@Mw86f&V(FCrKauku2i1vb zw5Bb}5hC)-89d#}qc5uOdoIjn&H| zgWox)TcurStUg-6?ct^z%5^TI!Swu@xomr`Eu6V^+k?oJY&*W3xfXtS0jJi|`xl?B z71oa7581e*adL>jdF8n_+hXSi#6P|59_}2j`@hCDyVn>ExwNNUIC`-l-~U0hqHH$5WS|83PBX2bUWZ#KUc&-;}aS9AT$%rz#;SMP7@oxJv<*J()= zHIeDpoMPuBs^2tP_iarLXpKO$+8_ zDXpJZmD70!I>lH_+)uH|>HtT=h3B>2g!)sHsuMsA&O@l#&z+uUTSo6Ng^N-x;^;nG*G zPS$tT?HS#7&VDP(YTjEUIBET|oGFJUFHv0(a{c!*(>LsgB*>m&Ax^EXIo%wu{f1&lgi?grsZSu~3!}o5j*ZvKqn_cT_fB#EM*w=OG(0xDC zkIrnm^L|KgJ1jj%hx3)#o|6j4n&0X_IK+G5!r@yn?@jqyP2LDi@9tRb#o0QE{gh^J z828PRBZW^DJmmK-vn~`-e`LMp&ifraszvN~tR%QXl;sUpC$Z?p=Zh~82bAG*k+amWXe>Z2Zv;}hDz*0 z{@2GZPdzbpu|$_vXZQ!jEV<=hXI9ueyz%wb>E!*V#N+aBE$FM7th+;O`R}V0Q?iQU zRriNB|LZGV`Q_O^=GfNRk1NK=4|8Ja49r4;g@Vul9B78Yn?1_lOZm|{jo1{ms0OwBRG%uO)V85)`x zqpLGCHZ{Z$v$Vv}Yh+|#gr?5Wz{C{Eypp2C%$(FB=xWj6%&JsSC@3iC2j%CND1agX kym$n(fKmY*C*W11#U+VFCE!3aG%z(Y=2BI4^>^a}0H={L)&Kwi diff --git a/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..c786f4e6750037403708462c03825a2c4a7eb5f9 GIT binary patch literal 10842 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a|v zV4!DUpkQnUrolC_bADb)YF;k^ITYgL}l+c>3N z`<(sa)&24HwSO+pzkkO3v&XlOug|w{eH!Ug@#ok4Y#^LPBO{hj~k|L%WR?B(w7 z(f6^WZjYu5AiBENNxezZDt*=^&CH!noD?SI9tZ+Dq_^V|4O z4F=%U2o1Z@ya`n^L!|q^{raO!k_mk+1L9cbbXbSoz(5*P4%4z8!h< zTb=pub_>3%_PQ}en`#QC@$I%>dt~`s`IBFs$$a~L_V1QE&+D2r-`1Z`KKyt8&A%nZ z66VtLHJ?m&{qo!6t>?1)z3+E({H?iF|Nqf)*L&Md&nFaE|K0Lq??u1HRVq@!IorjL z>alRLJQ9{`UX)r}`%L8L(Qhvu=WI{?_om+N$J0{bX@9wwoLuQAICK3I{^ym9S8Z+I zXk`~F70PsldKxDS-M+Q$Vt)GVMf0EUoS%IE(QylZ>+Np*Ti9G)7Kz+5?B!70=Hr7d!Up`Cc=BC;&Im7ku z=6}yqQ`-v*m*sDmc~OaV;a!($OcQ@I-?30NdVV(Toz2JE^lvU~a*kW@%+v^8@ zZ(;3=uWt)pp3N+uUw3Rp$*g|eFR@nF4lY!C_Od3vY~}fgr8niuo-giBFyHy_ z_tNH5w^MgTlq;LcPPsq9P4S}a`k&|6oi?OMX*{0C*Z*hvrinTK`j1KUu-~2iF>&p= zdp_zO>fv{yW^Avo`(WA=-2Gs_hsB+lA-DA$o*C&U8-J?A zXZ`Gmu;r?`XIpH0>bmHcj3^@><(1X0?2`}5i-~1$e|Hy9X|TK0Y+fLkF>&J}N#hmf zW*7E*Ot^Z%d-vaOEI-5~a=pcDn1pf;&703)zn;z;bkI z?go`J4R@6_yd&SloP5{k8sT`lWYH33uf;4GZNiG|si$fdE#H3m^h^HwYrc!U7_6Tv zv*P`IRLHhT$q}AlL^u5b`&@{c#QkLTTec}t#fqVMwXYI2% zS2gz5uRY0C;wR+H@_D!2%QY|vc%Yl>o2JdC_=jWPe*Zt)?=Ic@IHPH6m&(m}zJuS( zw|_87$?RgD$zYkFyrGbrv!vjKbnB4=Z8Of=EO=Qe^6%z(*CmeAUj55B(V`6#3HfuSj?HQ-bQfj?j?|fo(ROQ&j8O(A-a1FQoD_+im z&7~6`rDVz&3oGiBi6-4&cJ;;bw=LrDniJU^cn=m|5&L~0Z{IKO=3Q5}ZhOZ#@$+5F z%+{5=(`}5xxWhJ@wAu?l-aYH}N$XWFKF)hKDV2RC=dJvc9<_z`+A&woY`WlcJAWRd zKhvfeDHiH4llnDMAD+oDeSW~PrFd$Lam*LVh>Q#t`x_4b#FB5=ZoAPWbGcdU{^4bg zGiMkjuIY96J*geHS+kb6tJ_7PH!*1{zie8E=nkDT7oPd}J*ZLMSNZ;&W^L+1?o25L zF7r8!(SI+t?n;vR{3Gm9pV+?AO%wM0%)8?BeU1*pY(EjDoklXUQSZd=_Ej%0Hdqj@ z-1q*F#ff}1eYI#Cr(Urgxy!#+W$BMYPyuRIkTTlPpzw11kNY*sVg>SuluDf!p zv^a>C1f7W5Q?J@vu6Jo(ZuYY)jI-tICb-XTueiQrrQg{v)0!(bt(Ne4k|N(Ozie&m zVtKjz&Iemc_gqYT^sfD8MA*drE1tE5>+BHZYJPm{MsL7TfgS7JKR;?%cv0f+8RwfN zR}WuOz9Yky?=3&G%s0f>3a#YY^k>kGpzT2+-o;_pQ zniu#qg_Vf8`cIh1WpzOibaXulkcYi|$i*m`n6^{SQyb`)E@!y=e`+5F}9p>>hx1(Az zdp;Y+etE>i^h)-o?DmGR`5Imm{!8{vNm+la`pb`QV~*KE){jkNFI9Z9|2}8N^n}Ej zM=d30B)C4hmi!_iylnct5E1!rtFA5mIIAvQa^FQZsUtHrZvNb>*Lub>p-)3YoBg5h zq$3xK7VE4l=oIm@sJ=66?ST_ZnV*-q9(O**dDi;n#M9SrzMuKrb-fpR$SEQIH!6CE zWdm0S&KI4zBDE^pQ8mI>P4<#A-?l4!obR9MB)Wf>30~h+P~DZp?;*HfvDW;-@)HX# z-1oKmSoEi|`IRJV=QQmf(+)bgaJ;cSs{f32;vC)}$HMBnS|UDc4u^5h_hFfQX+fG| z^wWl`lY(|_xi!}y)hD-S)2B5id`e$#&G_8C!`fll{4ETQXI~rXhrV%e61Y{C>8-%B z=1t<~8D07>3NLo3O6=yib)rM5Pc+}iC*x3*%lZ?y_6Y2^G2D51WwDKV{K5Z%yS|zC z9@Tmr(mZ7)zwxV0W_Q_76+GH5c3dfcqf?Cpzv#&|e2S?`u8I2<+s#Z*mrUa1@|hsv zX<)2T61-vgBrC=wseH{dZr|VD-*qMH`+_g&t1K4S>0HaQj9ZiFY{oZzY0hNd~dmJkD?H1YZ7DS;1GsTvSlTjHW;WMujn-zs6 zOe|Y26z=v-<=y@@5Htrz#YSoq$*UdqS*HttAdq{O+c+c?TD@7eb|Z1214 z&l^tfc=XQQm$Qm($+fs&mzYlOd7M7|s>p1IH_t>F)09dMZ@>6fIe49@%1?pc5!#1t zcE~+GH+7ndmS0~lH#`gCvd&viWY z?lBL}Hd!a{f8*-!^HqIe+UEJG_2M`DbtTgozwB(e!(F#$g|A@u6#;ua z>$oskwS3LB17RY!9S*(_eg9;mU!ap?C+`%%bLdEv}StsQ|ra}~B^2tH59+R@?nV`h`<8KHeT ztQxAkN-drlk~YmNo)zmCh-A2F%$?eGWa5a8g zNv3)x|JAvS_V^q~yL3?7d-0;6(4?Npe_5k6u2WS9JgJk;*lEl?Ro6 z8tU-8YkhL2yVcqF1=FgnZ?sj`xp+GWFWNKz;Gy+7VI57HkJt5=u!Tf$f7#EJZu`xA z?(f{Xw!jtvF@6(i&VxN_S1-6|ttyzAa$z59wlKT5k`9kd%+%PdRfhxF7pZq{SY5V0 zGR?_6?`ZC2%NMtzi*wj)rX@1o^*k-U^-bC=zQFtDQJhQudoakgcnd$VXAFI_;bG6) z9VKOaR*JTf@(dYk|2t3J`@vXZ!FOH`*U)cDQK`#>Ht_V7UUFEt_(+MOVS_wd%ue-u zm1c9Rh8y1!tD8^p6k3Ya^WX0Jtgx#2iPyS>YuD=MJ&|^6n0-*M;N!Sayy%Uqu50?Df^L3ejXl)R1C-c7AM(Ur>=I+Q*v9Mk1>$|J*)2dsV zAsLyQx;U&)@6T7LQGY#)>(r$C@BVqLT=huL?{bIGrc=T3j@M?}9RF+6!1YY;q?>~2 ziKh<)#Acd5U`{J>+jdVc$J}poN3e=wu>Yr*rrTmHO{*H+UaZd3Vp+Gn_*4E7SKa*P z-ooo=Jh=HL-YZ_-xT~)%+%w1UM5BsH*6zA#xhoh1wr}{WbFhZ#75`kLN7EXd6}bZz z-#U7K?&%A?x{_yP!WO9+ZcMkcm}B(!gqQV9!<|mcy60GLmvG{ejqIKr8@GIa+tQ0Y zOHMpdFc4Zg)AO8PWY%65L*bn#gH~-m6Q7(cB&I#5;{8$?ttqEOwtPF%G_`8+iKR!U zi!Hiyw!LT5L%DRlnckx3`wdynD)xr?9I;>fPfYmZDyjRmk8 z+F|UTw)Kfu#X8$7lKcDPnpNLd{{3R&dd%ec0U@uTEt}MMfjjCSuDj-Wfwecm^kd7hMU&gB6YtNe%}ZoI zwAEy54(G1;nK_J8GBaO)Udb4lxpDfMaLDgT3j2oieekJWm4iLUPp);}bk=jMyd9!&oYklRk zExfYwXj5w=%SXZecVCA!bFqbXu=Hlw=;UVJEq#_BI@|p5m)(KWvp=7FqO$y@NB9EQ z)N0u~wr+DYUv24+`(IWd^VoPr$Et;~Kg)0amT`@badfS-=B)j`=Zl(O<*$b- zE}UnMS^gz%`y`Xt)!mg6(yL9H++>zi3QW4VD=af?@eO6 zFHc$ZfkT4RBP}RG;?taVue*G_mbaY!H5Q#R-*8}IQmb>=iY2?)SvD-aU>Fx8F!zaN zkdC{x>8m+>Ut`0My>L1sJe~dex@(aupG;%25Mh+*x@c8)m-*)Eq%A7TV_3{5X2_^d z+{j|rWx4MAFQ;6Kw>1@>D`r+*4!LdS?|UrU;}VbAC$rY5BN2O?@5@_#Ez=Erb|R(G zv{@?dn#Md)5ss!p;X(tZQ#<%>#F=d@+LKy*bFYm_;H-HJj$&&cn$1xt46@hWEOdIo zb<0kNlCb|1CPnD&|6X=OX6~zlZyVda_%m{vEyaDcdHFR`f>3>S@+5*Wn)pOMVpUs$^9 z$CdDZn-lJEhJ|U@Tv4)HtA1pid!rh zZ0sG}^7XKm6Hj{68MZxzx@z3Mm9_d=kDeH@ckg|_>rk@h?oHMAHLc!lPjff@%5yzu z^Bd*Zwi|H)9$T>`$@?#?E=rr|d&SmFC*sEMN#$!guXMb6 z(sNn+T;U=Kn-y#OH#Z+!aXUk}b%pP>Bl8UOt_l3#b+D zG|k)`rI=DAV>EZF^QFiI1^Sn~o&8E@+Jqb|axvT_D5uk34 z)8Ut0pIiggRz91wDfimtx}zL?i}#fTGL&xc&|a{F|LN17G|?qHLp_zVeq4*)FF{3~mRd+s;JXu*$l&@N<4fi0c~{H6zK=4X%f_M#>Zf zt^UUM)?0AGLCLlI|2{4jjCQP%7TvDK@3MYhi7?l8oq9J${@3fiUwI?(`o6~8BYc;> zZxC3!tfkoZe`epue-b->v0cBOXDgFto0~mVILLqYi&DY7IFfxoXtg!V#PwNR zh-Imn-y;%sXPdyk`!D;~MeIFtY3Bpg7Lk)CoeVD*AL-ivIn}SA@u%*j568rA=a&DI zO@Hgbmitw^=5zby4ME--lG-Lw%nzL3#V{OFoLV>aS@Zr0{*(J3Ct0{05kGmdKtx)n zE@q;41>4T4M?XvrTF0-Q=3oH?KP^u_9DGj)GHdbO}p z?aRZM6OKKMbHbDqLQPkn{@G$E^mO)dwpKyO-jH3wM=mn1T9dL*W8b$g%UO8un3mbh zQMEgM@Y}Qew=Z%I1^iib`O^IA`X@UKoQ)JO8?Za_g86}jZu251 zZ?7xdw*2YrNgTHY>ZTkkN}8ikxuM81gSWO%d}@onuu0Mscj&wQ1?m{nEjx6F6H`K;j?^dagdmWPD*zx4w)4liBO*odAo4cU-RGG-nx_|#} zd;ZDnuvT7rKB{bH{r{+&iISbgpHDwyvFkms>y5zO0%D9$Yy6xLWqtwCfq~(*J0M2p--sWlhoCjWdNc-CBkIIW+U0+AP0oYe7lV zs#DWVxW66GY4iE3)mASkwk;szSM%%xOH8jHS6;1kymz+qhQ7$a&7M#0gO?5*jrE>% zSAplwbCI0~mKfA+OPqP?^qv16jM9$V1#TG6eluy+^NqcZ^QT=|yJ>%4+?gor^$dmH z^CZ+yehHK;)4OLOp7s8;Saw55V#I2d72;j9C(bE1KBx4IyNe}1ZcWLqOQ$>RkLvr+ z6wJ9^RXaV^a$PRRv%1|YqGWFUu2YlxyC*8{&#`Trc=jrV34Fe)%XG^`^2c+Z+2zR> zle>i5V+>v%X1h`?*Zb1u)7 z`@VcRH&tYI`px}0b7%3|F%|F5Vm|StBroBQ*cvmHezckN%Lt24H961vC@bmhSqn8D1uxC6%PLP=st!%~I_Kra$~g_Cj!Djf zI(|tG4u!WBPo346c$UxUf<90G{nFL_?2}LI6LbtT>sl;rx*_X?aCzD1%#ZFR`I>I= zi{`H{n7VG$tm|tpPI&pvGwq{iTaxSSnc7U8l1nRHwgvbJxxBq%H<*p*7UmEgStPqO!??`mT&6kBlp`}gFmgv)|g+V`F~nL z_l(aKq8>NJdnM~9?B$W$B6Os!Pw9QS^I5l5M`mw+pdR)@tt@u+nefTkviEi-?7Jyq zcXi3t?Z&%yE&RZk<{UNQo_}cGl&@Qwd9E_+KJ3k~Q~WDi@#l_!?a7J)FZV+E7d`B6 z&fF`V&By-U^0nm)i=ND|3v1U!Zrf0+b@l%=`1W)W-xqR)?PuwLg=WTLyL?_Mu@y0$>n~4b9oyR0jZ?}Mx#Cq?XH3i&ULInXt##tibpDF0 zX@a`vByuW^7S7sMlu+HhRwwCR*5jg$Td!t@=kwo?t-G<|`@DHoY4`q|<5~5?Vs`J= zLZ<9Vm##4uY9xmL6tnU?Ic>t$slI+(w+)v3XZclovTvo+vuBmpXRY|k!_(Eu88E+E zc3bdNmc%o=^<)k`4B7lTQ$uXcZ2@_&ZCVez!&Ki!J(sL~+ilCgxi7-+MtRXS9m%fG zuMRIiKb!s7#7m*83jgaA=&lR&%w@T(V0w94c4zUNU3Xs{Tjr!$vc0u+zN6_1vuoSE zeMRHvdFr%$ z(E`VPPbyE_Jdxe_U3u>@TvmZ`derum9!g;tyIc z*8D!r5MA^`Kd0SXAUf%??m{(p1MLikmuQ06WQuNN=>*ic2rp{U(aqE78WYc6y~Nz2%EHB9K_$_G2wq*i8^wKdyWIw6w&~Yh+|#gr?Wfz{J1|Ay!h9n3JcjE#8666t_ literal 0 HcmV?d00001 diff --git a/tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/stderr.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/multipage/__-l__eng__000006.ocr.png__000006__hocr__txt/stderr.bin rename to tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/palette/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 91% rename from tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin index 34a37290..b3381922 100644 --- a/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin +++ b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -12,7 +12,7 @@ be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic synthesizers! -© Ultra-fast 3!” disk drive stores complex songs in seconds and holds over 110,000 notes +¢ Ultra-fast 3!” disk drive stores complex songs in seconds and holds over 110,000 notes per disk! @@ -33,7 +33,7 @@ Recording a Sequence To record a sequence, simply press RECORD and PLAY, then play your MIDI keyboard in time to the Sequencer’s click track. When the sequence loops back around to bar 1, -you’ll hear what you played—only all timing errors will be +you’ ll hear what you played—only all timing errors will be corrected! (Timing correction may be adjusted or defeated). Any additional notes played will be added into the track —existing notes are not erased while recording! @@ -43,7 +43,6 @@ may be used at any time to quickly access any location in your sequence for spot-recording. To overdub a new part, select a different track and start recording—while you record, the first track will play in perfect sync (unless you - MUTE it, or SOLO another track). In this way, up to 32 tracks may be overdubbed! All MIDI effects are recorded including pitch bend, modulation, velocity, aftertouch, @@ -91,7 +90,7 @@ music. See your Linn dealer today for a demonstration! HELP button displays additional explanations. -© Non-destructive recording—existing notes are not erased while recording. +® Non-destructive recording—existing notes are not erased while recording. © Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including ERASE, REPEAT, PLAY/STOP, or LOCATE. @@ -100,7 +99,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE. e Will sync to standard LinnDrum or Linn 9000 sync tone. -Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. ¢ TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, (even drop frame!) @@ -109,7 +108,7 @@ Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST opera on the TAP TEMPO button. -¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. +e TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. e Any TIME SIGNATURE may be used, and may be changed within a song. linn diff --git a/tests/cache/poster/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stderr.bin b/tests/cache/poster/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stderr.bin similarity index 100% rename from tests/cache/poster/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stderr.bin rename to tests/cache/poster/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stderr.bin diff --git a/tests/cache/poster/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stdout.bin b/tests/cache/poster/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stdout.bin similarity index 54% rename from tests/cache/poster/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stdout.bin rename to tests/cache/poster/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stdout.bin index d02007b8..a690fe5b 100644 --- a/tests/cache/poster/__-l__osd__--psm__0__000001.ocr.preview.jpg__stdout/stdout.bin +++ b/tests/cache/poster/__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout/stdout.bin @@ -1,6 +1,6 @@ Page number: 0 Orientation in degrees: 0 Rotate: 0 -Orientation confidence: 29.83 +Orientation confidence: 33.87 Script: Latin -Script confidence: 1.88 +Script confidence: 4.54 diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 74% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index ada2aee1..5f5be8ad 100644 --- a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -
+
diff --git a/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/poster/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 98% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index fd936de35c709017b229e9b083250613ff672db7..7db2edeeaefde567820b92a060331ae9caf5351c 100644 GIT binary patch delta 27 icmaDS`c8C1H7B2?fvKUHp^2%9k+H6U`Q}c}R7L=LatFWw delta 27 icmaDS`c8C1H7B2ip^>qHfq|)kiLtJM#pX`VR7L=Kum`gM diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin similarity index 56% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 87f8ad9b..714b7115 100644 --- a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/hocr.bin +++ b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,962 +9,962 @@ -
+

- The - LinnSequencer + The + LinnSequencer - 32 - Track - MIDI - Sequence - Recorder + 32 + Track + MIDI + Sequence + Recorder

- The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is - extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It - ’s - many - remarkable - features - include: + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It + ’s + many + remarkable + features + include:

- * - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST + * + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - FORWARD, - REWIND, - and - LOCATE - controls, + FORWARD, + REWIND, + and + LOCATE + controls,

- © - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may + © + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic - synthesizers! + synthesizers!

- * - Ultra-fast - 314" - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes + * + Ultra-fast + 314" + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes - per - disk! + per + disk!

- * - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. + * + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - ¢ - Exclusive - real-time - ERASE - function - makes - editing - FAST, + ¢ + Exclusive + real-time + ERASE + function + makes + editing + FAST,

- * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected - rhythmic - value. + rhythmic + value.

- * - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes, + * + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes, - ¢ - Optional - SMPTE - time - code - synchronization. + ¢ + Optional + SMPTE + time + code + synchronization. - * - Optional - remote - control. + * + Optional + remote + control.

- Recording - a - Sequence - simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to + Recording + a + Sequence + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - To - record - a - sequence, - simply - press - RECORD - and - PL - AY, - _ - find - the - desired - bar - number, - then - Start - recording. + To + record + a + sequence, + simply + press + RECORD + and + PL + AY, + _ + find + the + desired + bar + number, + then + Start + recording. - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s - The - INSERT/COPY - function - allows - you - to - move - bars + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s + The + INSERT/COPY + function + allows + you + to + move + bars

- click - track, - When - the - sequence - loops - back - around - to - bar - 1, - from - one - location - to - another—in - the - same - sequence - or - a + click + track, + When + the + sequence + loops + back + around + to + bar + 1, + from + one + location + to + another—in + the + same + sequence + or + a - you'll - hear - what - you - played—only - all - timing - errors - will - be - _different - one. - For - example, - you - might - insert - a - copy - of - the + you'll + hear + what + you + played—only + all + timing + errors + will + be + _different + one. + For + example, + you + might + insert + a + copy + of + the - corrected! - (Timing - correction - may - be - adjusted - or - defeated), - _ - first - verse - between - the - second - chorus - and - the - bridge. + corrected! + (Timing + correction + may + be + adjusted + or + defeated), + _ + first + verse + between + the + second + chorus + and + the + bridge. - Any - additional - notes - played - will - be - added - into - the - track - DELETE - BARS - operates - the - same - way - to - remove + Any + additional + notes + played + will + be + added + into + the + track + DELETE + BARS + operates + the + same + way + to + remove - —existing - notes - are - not - erased - while - recording! - unwanted - sections, + —existing + notes + are + not + erased + while + recording! + unwanted + sections,

- FAST - FORWARD, - REWIND, - and - LOCATE - controls - . + FAST + FORWARD, + REWIND, + and + LOCATE + controls + . - may - be - used - at - any - time - to - quickly - access - any - location - in - Creating - a - Song + may + be + used + at + any + time + to + quickly + access + any + location + in + Creating + a + Song

- your - sequence - for - spot-recording. - To - overdub - a - new - part, - One - way - to - create - a - song - is - to - record - each - track - all - the + your + sequence + for + spot-recording. + To + overdub + a + new + part, + One + way + to + create + a + song + is + to + record + each + track + all + the - select - a - different - track - and - start - recording—while - you - way - through - (up - to - 999 - bars), - Another - way - is - to - record + select + a + different + track + and + start + recording—while + you + way + through + (up + to + 999 + bars), + Another + way + is + to + record - record, - the - first - track - will - play - in - perfect - sync - (unless - you - each - basic - section - (verse, - chorus, - etc.) - in - individual + record, + the + first + track + will + play + in + perfect + sync + (unless + you + each + basic + section + (verse, + chorus, + etc.) + in + individual

- MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded - them - together. - CREATE - SONG - will - then - automatically + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded + them + together. + CREATE + SONG + will + then + automatically

- including - pitch - bend, - modulation, - velocity, - aftertouch, - Copy - all - the - parts - into - a - new - sequence. - If - desir - ed, - you - can + including + pitch + bend, + modulation, + velocity, + aftertouch, + Copy + all + the + parts + into + a + new + sequence. + If + desir + ed, + you + can - sustain - pedal, - and - program - changes! - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + sustain + pedal, + and + program + changes! + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout. - Editing - Composition - Without - Compromise + Editing + Composition + Without + Compromise

- To - erase - a - wrong - note, - simply - hold - ERASE - and - press - The - technology - you - use - should - never - be - so - complex - that + To + erase + a + wrong + note, + simply + hold + ERASE + and + press + The + technology + you + use + should + never + be + so + complex + that - the - note - to - be - erased - just - before - it - plays - in - the - sequence— - it - interferes - with - the - creative - process. - That’s - precisely - why + the + note + to + be + erased + just + before + it + plays + in + the + sequence— + it + interferes + with + the + creative + process. + That’s + precisely + why - when - played - back, - it - will - be - gone. - Notes - may - also - be - the - LinnSequencer - is - designed - to - let - you - compose, - record + when + played + back, + it + will + be + gone. + Notes + may + also + be + the + LinnSequencer + is + designed + to + let + you + compose, + record

- added, - erased, - or - changed - using - the - SINGLE - STEP - func- - and - edit - while - devoting - your - undivided - attention - to - your + added, + erased, + or + changed + using + the + SINGLE + STEP + func- + and + edit + while + devoting + your + undivided + attention + to + your - tion. - To - overdub - notes - at - specific - points - within - a - Sequence, - — - music. - See - your - Linn - dealer - today - for - a - demonstration! + tion. + To + overdub + notes + at + specific + points + within + a + Sequence, + — + music. + See + your + Linn + dealer + today + for + a + demonstration!

- Additional - Features + Additional + Features

- * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations, - If - needed, - the + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations, + If + needed, + the - HELP - button - displays - additional - explanations, + HELP + button + displays + additional + explanations,

- * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording.

- * - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + * + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including - ERASE, - REPEAT, - PLAY/ - STOP, - or - LOCATE, + ERASE, + REPEAT, + PLAY/ + STOP, + or + LOCATE,

- * - Two - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. + * + Two + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value. - © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone, + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone, - * - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. + * + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation.

- * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second, - (even - drop - frame!) + (even + drop + frame!)

- * - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes + * + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes - on - the - TAP - TEMPO - button. + on + the + TAP + TEMPO + button.

- * - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. + * + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired.

- « - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + « + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song. - linn + linn

- Linn - Electronics, - Inc. + Linn + Electronics, + Inc. - 18720 - Oxnard - Street, - Tarzana, - CA - 91356 + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - (818) - 708-8131 - TELEX - #298949 - LINN - UR + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/stderr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/stderr.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/stdout.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin similarity index 100% rename from tests/cache/skew/__-l__eng__--psm__7__000001.ocr.png__000001__hocr__txt/stdout.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin similarity index 100% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/txt.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin similarity index 99% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/pdf.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 634dc61be74aae399b5f60b868e5e804cc705620..ed122d0b931947adc2d4ce1249b1160119808143 100644 GIT binary patch delta 27 icmZolZc5(3XTWD^U}|V)Xkuz&Y_4lyzFEc~jS&EBga-Zq delta 27 icmZolZc5(3XTWD+Xk=_)U|?clYM^Uiv026-jS&EA&j#@T diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin similarity index 100% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stderr.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin similarity index 100% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001.text__pdf__txt/stdout.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin diff --git a/tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin similarity index 100% rename from tests/cache/skew/__-l__eng__000001.ocr.png__000001__hocr__txt/txt.bin rename to tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin From 56067b590b3ee4dfa8346d79b736af6d907bd9fa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 01:59:36 -0700 Subject: [PATCH 032/880] Make re_symlink() not require a log object --- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_validation.py | 4 ++-- src/ocrmypdf/helpers.py | 31 +++++++++++++++++-------------- src/ocrmypdf/optimize.py | 18 ++++++++++++------ 4 files changed, 32 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 5de49e33..d2881407 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -132,7 +132,7 @@ def triage(input_file, output_file, options, log): "input file is a PDF, not an image." ) # Origin file is a pdf create a symlink with pdf extension - re_symlink(input_file, output_file, log) + re_symlink(input_file, output_file) return output_file except EnvironmentError as e: log.error(e) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index af6cf552..fb1be0d1 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -355,7 +355,7 @@ def create_input_file(options, work_folder): else: try: target = os.path.join(work_folder, 'origin') - re_symlink(options.input_file, target, log) + re_symlink(options.input_file, target) return target except FileNotFoundError: log.error("File not found - %s", options.input_file) @@ -372,7 +372,7 @@ def check_input_file(options, start_input_file): copyfileobj(sys.stdin.buffer, stream_buffer) else: try: - re_symlink(options.input_file, start_input_file, log) + re_symlink(options.input_file, start_input_file) except FileNotFoundError: log.error("File not found - %s", options.input_file) raise InputFileError() diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 3d56a1f5..1e9a2cb7 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -15,32 +15,35 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import multiprocessing import os -import sys import warnings from collections.abc import Iterable from contextlib import suppress -from functools import partial, wraps +from functools import wraps from pathlib import Path +log = logging.getLogger(__name__) -def re_symlink(input_file, soft_link_name, log=None): + +def re_symlink(input_file, soft_link_name, *args, **kwargs): """ Helper function: relinks soft symbolic link if necessary """ + if len(args) == 1 and isinstance(args[0], logging.Logger): + log.warning("Deprecated: re_symlink(,log)") + if 'log' in kwargs: + log.warning('Deprecated: re_symlink(...log=)') + input_file = os.fspath(input_file) soft_link_name = os.fspath(soft_link_name) - if log is None: - prdebug = partial(print, file=sys.stderr) - else: - prdebug = log.debug # Guard against soft linking to oneself if input_file == soft_link_name: - prdebug( - "Warning: No symbolic link made. You are using " - + "the original data directory as the working directory." + log.warning( + "No symbolic link made. You are using " + "the original data directory as the working directory." ) return @@ -48,16 +51,16 @@ def re_symlink(input_file, soft_link_name, log=None): if os.path.lexists(soft_link_name): # do not delete or overwrite real (non-soft link) file if not os.path.islink(soft_link_name): - raise FileExistsError("%s exists and is not a link" % soft_link_name) + raise FileExistsError(f"{soft_link_name} exists and is not a link") try: os.unlink(soft_link_name) except OSError: - prdebug("Can't unlink %s" % (soft_link_name)) + log.debug("Can't unlink %s", soft_link_name) if not os.path.exists(input_file): - raise FileNotFoundError("trying to create a broken symlink to %s" % input_file) + raise FileNotFoundError(f"trying to create a broken symlink to {input_file}") - prdebug("os.symlink(%s, %s)" % (input_file, soft_link_name)) + log.debug("os.symlink(%s, %s)", input_file, soft_link_name) # Create symbolic link using absolute path os.symlink(os.path.abspath(input_file), soft_link_name) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index b722228e..87ce0212 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -81,7 +81,9 @@ def extract_image_jbig2(*, pike, root, log, image, xref, options): pim, filtdp = result if ( - pim.bits_per_component == 1 and filtdp != Name.JBIG2Decode and jbig2enc.available() + pim.bits_per_component == 1 + and filtdp != Name.JBIG2Decode + and jbig2enc.available() ): try: imgname = Path(root / f'{xref:08d}') @@ -126,7 +128,9 @@ def extract_image_generic(*, pike, root, log, image, xref, options): return None return xref, ext elif ( - pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES and options.optimize >= 3 + pim.indexed + and pim.colorspace in pim.SIMPLE_COLORSPACES + and options.optimize >= 3 ): # Try to improve on indexed images - these are far from low hanging # fruit in most cases @@ -426,7 +430,7 @@ def optimize(input_file, output_file, context): log = context.log options = context.options if options.optimize == 0: - re_symlink(input_file, output_file, log) + re_symlink(input_file, output_file) return if options.jpeg_quality == 0: @@ -467,9 +471,9 @@ def optimize(input_file, output_file, context): if savings < 0: log.info("Optimize did not improve the file - discarded") - re_symlink(input_file, output_file, log) + re_symlink(input_file, output_file) else: - re_symlink(target_file, output_file, log) + re_symlink(target_file, output_file) def main(infile, outfile, level, jobs=1): @@ -479,7 +483,9 @@ def main(infile, outfile, level, jobs=1): class OptimizeOptions: """Emulate ocrmypdf's options""" - def __init__(self, input_file, jobs, optimize, jpeg_quality, png_quality, jb2lossy): + def __init__( + self, input_file, jobs, optimize, jpeg_quality, png_quality, jb2lossy + ): self.input_file = input_file self.jobs = jobs self.optimize = optimize From cfd67ab6aab8199497d623ba069f786936c10439 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 14:24:47 -0700 Subject: [PATCH 033/880] Fixing threading._RLock exception on Python 3.6 Issue was the usual business: objects that cross process boundaries need to be picklable and Python 3.6 is more strict about this. The logger object in particular interfered, so now we suppress it and rebuild it in process. --- src/ocrmypdf/_jobcontext.py | 50 ++++++++++++++++++++++++++++--------- src/ocrmypdf/_sync.py | 12 +++------ 2 files changed, 42 insertions(+), 20 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index bef45426..cc23de85 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -22,10 +22,29 @@ import os from contextlib import suppress -class PDFContext: +class PicklableLoggerMixin: + def __init__(self): + self._log = None + + @property + def log(self): + if not self._log: + self._log = self.get_logger() + return self._log + + def __getstate__(self): + log = self._log + self._log = None + state = self.__dict__ + self._log = log + return state + + +class PDFContext(PicklableLoggerMixin): """Holds our context for a particular run of the pipeline""" def __init__(self, options, work_folder, origin, pdfinfo): + PicklableLoggerMixin.__init__(self) self.options = options self.work_folder = work_folder self.origin = origin @@ -36,7 +55,9 @@ class PDFContext: self.name = 'origin.pdf' if self.name == '-': self.name = 'stdin' - self.log = get_logger(options, filename=self.name) + + def get_logger(self): + return make_logger(self.options, filename=self.name) def get_path(self, name): return os.path.join(self.work_folder, name) @@ -47,22 +68,27 @@ class PDFContext: yield PageContext(self, n) -class PageContext: - """Holds our context for a page""" +class PageContext(PicklableLoggerMixin): + """Holds our context for a page + + Must be pickable, so only store intrinsic/simple data elements + """ def __init__(self, pdf_context, pageno): - self.pdf_context = pdf_context + PicklableLoggerMixin.__init__(self) + self.work_folder = pdf_context.work_folder + self.origin = pdf_context.origin self.options = pdf_context.options + self.name = pdf_context.name self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] - self.log = get_logger( - pdf_context.options, filename=self.pdf_context.name, page=(self.pageno + 1) - ) + self._log = None + + def get_logger(self): + return make_logger(self.options, filename=self.name, page=self.pageno + 1) def get_path(self, name): - return os.path.join( - self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name) - ) + return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name)) def cleanup_working_files(work_folder, options): @@ -88,7 +114,7 @@ class LogNamePageAdapter(logging.LoggerAdapter): ) -def get_logger(options=None, prefix='ocrmypdf', filename=None, page=None): +def make_logger(options=None, prefix='ocrmypdf', filename=None, page=None): log = logging.getLogger(prefix) if filename and page: adapter = LogNamePageAdapter(log, dict(filename=filename, page=page)) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index ad79c306..06e600b7 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -29,7 +29,7 @@ from tempfile import mkdtemp from tqdm import tqdm from . import VERSION -from ._jobcontext import PDFContext, cleanup_working_files, get_logger +from ._jobcontext import PDFContext, cleanup_working_files, make_logger from ._pipeline import ( convert_to_pdfa, copy_final, @@ -83,17 +83,13 @@ def exec_page_sync(page_context): if is_ocr_required(page_context): if options.rotate_pages: # Rasterize - rasterize_preview_out = rasterize_preview( - page_context.pdf_context.origin, page_context - ) + rasterize_preview_out = rasterize_preview(page_context.origin, page_context) orientation_correction = get_orientation_correction( rasterize_preview_out, page_context ) rasterize_out = rasterize( - page_context.pdf_context.origin, - page_context, - correction=orientation_correction, + page_context.origin, page_context, correction=orientation_correction ) preprocess_out = rasterize_out @@ -240,7 +236,7 @@ def exec_concurrent(context): def run_pipeline(options): - log = get_logger(options, __name__) + log = make_logger(options, __name__) log.debug('ocrmypdf ' + VERSION) result = check_options(options) From 61afef549e8775acc0b7a059bf186d447a32d1f5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 14:25:17 -0700 Subject: [PATCH 034/880] Remove some now-unused code; etc --- .gitignore | 1 + docs/installation.rst | 4 ++-- src/ocrmypdf/helpers.py | 12 ------------ 3 files changed, 3 insertions(+), 14 deletions(-) diff --git a/.gitignore b/.gitignore index b2881fb4..83ec1729 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ .pytest_cache/ .ruffus_history.sqlite .venv/ +.venv*/ *.pyc *.sublime-* diff --git a/docs/installation.rst b/docs/installation.rst index 73380839..b669320c 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -118,13 +118,13 @@ install the system version to get most of the dependencies: ocrmypdf \ python3-pip -There are a few dependency changes between ocrmypdf 6.1.2 and 7.x. Let's get +There are a few system dependency changes since ocrmypdf 6.1.2. Let's get these, too. .. code-block:: bash sudo apt-get install \ - libexempi3 \ + libxml2 \ pngquant Then install the most recent ocrmypdf for the local user and set the user's ``PATH`` to check for the user's Python packages. diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 1e9a2cb7..683cc815 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -66,10 +66,6 @@ def re_symlink(input_file, soft_link_name, *args, **kwargs): os.symlink(os.path.abspath(input_file), soft_link_name) -def is_iterable_notstr(thing): - return isinstance(thing, Iterable) and not isinstance(thing, str) - - def page_number(input_file): """Get one-based page number implied by filename (000002.pdf -> 2)""" return int(os.path.basename(os.fspath(input_file))[0:6]) @@ -125,14 +121,6 @@ def is_file_writable(test_file): return True -def flatten_groups(groups): - for obj in groups: - if is_iterable_notstr(obj): - yield from obj - else: - yield obj - - def deprecated(func): """Warn that function is deprecated""" From 24da92d39ef4b648317b7fc9a091209aa865dc22 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 14:56:33 -0700 Subject: [PATCH 035/880] Fix extra blank lines in output messages in Python 3.6 --- src/ocrmypdf/__main__.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index e48283c2..eaf894ef 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -471,10 +471,15 @@ class TqdmConsole: def __init__(self, file): self.file = file + self.py36 = sys.version_info >= (3, 6) def write(self, msg): # When no progress bar is active, tqdm.write() routes to print() - tqdm.write(msg.rstrip(), file=self.file) + if self.py36: + if msg.strip() != '': + tqdm.write(msg.rstrip(), end='\n', file=self.file) + else: + tqdm.write(msg.rstrip(), end='\n', file=self.file) def flush(self): if hasattr(self.file, "flush"): From ef1ef1cdf02b5f85bc57b69e96caea035f04c876 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 15:20:07 -0700 Subject: [PATCH 036/880] Fix test invalidated by Python 3.6 logging fixes --- tests/test_metadata.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 5b7e235e..fc392c77 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -296,7 +296,6 @@ def test_metadata_fixup_warning(resources, outdir, caplog): copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') context = PDFContext(options, outdir, outdir / 'graph.pdf', None) - context.log = logging.getLogger() metadata_fixup(working_file=outdir / 'graph.pdf', context=context) for record in caplog.records: assert record.levelname != 'WARNING' @@ -308,7 +307,6 @@ def test_metadata_fixup_warning(resources, outdir, caplog): graph.save(outdir / 'graph_mod.pdf') context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None) - context.log = logging.getLogger() metadata_fixup(working_file=outdir / 'graph.pdf', context=context) assert any(record.levelname == 'WARNING' for record in caplog.records) From 188e08e98baccdb6317cc8e83d956a46e02ba5ae Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 May 2019 22:28:28 -0700 Subject: [PATCH 037/880] docs: Remove discussion of ruffus --- docs/batch.rst | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index 6883e992..a98fd142 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -28,7 +28,7 @@ This will walk through a directory tree and run OCR on all files in place, print .. code-block:: bash find . -printf '%p' -name '*.pdf' -exec ocrmypdf '{}' '{}' \; - + Alternatively, with a docker container (mounts a volume to the container where the PDFs are stored): .. code-block:: bash @@ -100,8 +100,6 @@ API OCRmyPDF is currently supported as a command line interface. This means that even if you are using OCRmyPDF in a Python script, you should run it in a subprocess rather importing the ocrmypdf package. -The reason for this limitation is that the `ruffus `_ library that OCRmyPDF depends on is unfortunately not reentrant. OCRmyPDF works by defining each operation it does as a ruffus task that takes one or more files as input and generates one or more files as output. As such ruffus is fairly fundamental. - (If you find individual functions implemented in OCRmyPDF useful (such as ``ocrmypdf.pdfinfo``), you can use these if you wish to.) From ac2fc9c2a0e0d3eb32fd1c7f644eebe416e27597 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 May 2019 15:06:47 -0700 Subject: [PATCH 038/880] Explain picklable logger --- src/ocrmypdf/_jobcontext.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index cc23de85..5daab103 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -33,10 +33,11 @@ class PicklableLoggerMixin: return self._log def __getstate__(self): - log = self._log - self._log = None - state = self.__dict__ - self._log = log + # Python 3.6 is incapable of pickling a logger and marshalling it to another + # process (threading._RLock error), so we disconnect it before pickling, + # and create a new logger in the worker process. + state = self.__dict__.copy() + state['_log'] = None return state From 7ee0c52a5718f7aa1ad4d0e0fdbe01150784cac2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 May 2019 22:34:45 -0700 Subject: [PATCH 039/880] Refactor cli into basic high level api --- src/ocrmypdf/__init__.py | 1 + src/ocrmypdf/__main__.py | 514 ++------------------------------------- src/ocrmypdf/_sync.py | 2 - src/ocrmypdf/api.py | 177 ++++++++++++++ src/ocrmypdf/cli.py | 458 ++++++++++++++++++++++++++++++++++ 5 files changed, 651 insertions(+), 501 deletions(-) create mode 100644 src/ocrmypdf/api.py create mode 100644 src/ocrmypdf/cli.py diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index a37d2658..a7211a25 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,3 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo +from .api import ocrmypdf diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index eaf894ef..23d29b4c 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -# © 2015-17 James R. Barlow: github.com/jbarlow83 +# © 2015-19 James R. Barlow: github.com/jbarlow83 # # This file is part of OCRmyPDF. # @@ -16,513 +16,29 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import argparse import logging import os import sys -from tqdm import tqdm - -from . import PROGRAM_NAME, VERSION -from .exceptions import ExitCode -from ._sync import run_pipeline +from .cli import parser +from .api import configure_logging, run as api_run from ._validation import check_closed_streams - -# ------------- -# Parser - - -def numeric(basetype, min_=None, max_=None): - """Validator for numeric params""" - min_ = basetype(min_) if min_ is not None else None - max_ = basetype(max_) if max_ is not None else None - - def _numeric(string): - value = basetype(string) - if (min_ is not None and value < min_) or (max_ is not None and value > max_): - msg = "%r not in valid range %r" % (string, (min_, max_)) - raise argparse.ArgumentTypeError(msg) - return value - - _numeric.__name__ = basetype.__name__ - return _numeric - - -parser = argparse.ArgumentParser( - prog=PROGRAM_NAME, - fromfile_prefix_chars='@', - formatter_class=argparse.RawDescriptionHelpFormatter, - description="""\ -Generates a searchable PDF or PDF/A from a regular PDF. - -OCRmyPDF rasterizes each page of the input PDF, optionally corrects page -rotation and performs image processing, runs the Tesseract OCR engine on the -image, and then creates a PDF from the OCR information. -""", - epilog="""\ -OCRmyPDF attempts to keep the output file at about the same size. If a file -contains losslessly compressed images, and output file will be losslessly -compressed as well. - -PDF is a page description file that attempts to preserve a layout exactly. -A PDF can contain vector objects (such as text or lines) and raster objects -(images). A page might have multiple images. OCRmyPDF is prepared to deal -with the wide variety of PDFs that exist in the wild. - -When a PDF page contains text, OCRmyPDF assumes that the page has already -been OCRed or is a "born digital" page that should not be OCRed. The default -behavior is to exit in this case without producing a file. You can use the -option --skip-text to ignore pages with text, or --force-ocr to rasterize -all objects on the page and produce an image-only PDF as output. - - ocrmypdf --skip-text file_with_some_text_pages.pdf output.pdf - - ocrmypdf --force-ocr word_document.pdf output.pdf - -If you are concerned about long-term archiving of PDFs, use the default option ---output-type pdfa which converts the PDF to a standardized PDF/A-2b. This -converts images to sRGB colorspace, removes some features from the PDF such -as Javascript or forms. If you want to minimize the number of changes made to -your PDF, use --output-type pdf. - -If OCRmyPDF is given an image file as input, it will attempt to convert the -image to a PDF before processing. For more control over the conversion of -images to PDF, use the Python package img2pdf or other image to PDF software. - -For example, this command uses img2pdf to convert all .png files beginning -with the 'page' prefix to a PDF, fitting each image on A4-sized paper, and -sending the result to OCRmyPDF through a pipe. img2pdf is a dependency of -ocrmypdf so it is already installed. - - img2pdf --pagesize A4 page*.png | ocrmypdf - myfile.pdf - -Online documentation is located at: - https://ocrmypdf.readthedocs.io/en/latest/introduction.html - -""", -) - -parser.add_argument( - 'input_file', - metavar="input_pdf_or_image", - help="PDF file containing the images to be OCRed (or '-' to read from " - "standard input)", -) -parser.add_argument( - 'output_file', - metavar="output_pdf", - help="Output searchable PDF file (or '-' to write to standard output). " - "Existing files will be ovewritten. If same as input file, the " - "input file will be updated only if processing is successful.", -) -parser.add_argument( - '-l', - '--language', - action='append', - help="Language(s) of the file to be OCRed (see tesseract --list-langs for " - "all language packs installed in your system). Use -l eng+deu for " - "multiple languages.", -) -parser.add_argument( - '--image-dpi', - metavar='DPI', - type=int, - help="For input image instead of PDF, use this DPI instead of file's.", -) -parser.add_argument( - '--output-type', - choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'], - default='pdfa', - help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for " - "long term archiving (default, recommended) but may not suitable " - "for users who want their file altered as little as possible. 'pdfa' " - "also has problems with full Unicode text. 'pdf' attempts to " - "preserve file contents as much as possible. 'pdf-a1' creates a " - "PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a " - "PDF/A3-b file.", -) - -# Use null string '\0' as sentinel to indicate the user supplied no argument, -# since that is the only invalid character for filepaths on all platforms -# bool('\0') is True in Python -parser.add_argument( - '--sidecar', - nargs='?', - const='\0', - default=None, - metavar='FILE', - help="Generate sidecar text files that contain the same text recognized " - "by Tesseract. This may be useful for building a OCR text database. " - "If FILE is omitted, the sidecar file be named {output_file}.txt " - "If FILE is set to '-', the sidecar is written to stdout (a " - "convenient way to preview OCR quality). The output file and sidecar " - "may not both use stdout at the same time.", -) - -parser.add_argument( - '--version', - action='version', - version=VERSION, - help="Print program version and exit", -) - -jobcontrol = parser.add_argument_group("Job control options") -jobcontrol.add_argument( - '-j', - '--jobs', - metavar='N', - type=numeric(int, 0, 256), - help="Use up to N CPU cores simultaneously (default: use all).", -) -jobcontrol.add_argument( - '-q', '--quiet', action='store_true', help="Suppress INFO messages" -) -jobcontrol.add_argument( - '-v', - '--verbose', - type=int, - default=0, - nargs='?', - help="Print more verbose messages for each additional verbose level. Use " - "`-v 1` typically for much more detailed logging. Higher numbers " - "are probably only useful in debugging.", -) - -metadata = parser.add_argument_group( - "Metadata options", - "Set output PDF/A metadata (default: copy input document's metadata)", -) -metadata.add_argument( - '--title', type=str, help="Set document title (place multiple words in quotes)" -) -metadata.add_argument('--author', type=str, help="Set document author") -metadata.add_argument('--subject', type=str, help="Set document subject description") -metadata.add_argument('--keywords', type=str, help="Set document keywords") - -preprocessing = parser.add_argument_group( - "Image preprocessing options", - "Options to improve the quality of the final PDF and OCR", -) -preprocessing.add_argument( - '-r', - '--rotate-pages', - action='store_true', - help="Automatically rotate pages based on detected text orientation", -) -preprocessing.add_argument( - '--remove-background', - action='store_true', - help="Attempt to remove background from gray or color pages, setting it " - "to white ", -) -preprocessing.add_argument( - '-d', '--deskew', action='store_true', help="Deskew each page before performing OCR" -) -preprocessing.add_argument( - '-c', - '--clean', - action='store_true', - help="Clean pages from scanning artifacts before performing OCR, and send " - "the cleaned page to OCR, but do not include the cleaned page in " - "the output", -) -preprocessing.add_argument( - '-i', - '--clean-final', - action='store_true', - help="Clean page as above, and incorporate the cleaned image in the final " - "PDF. Might remove desired content.", -) -preprocessing.add_argument( - '--unpaper-args', - type=str, - default=None, - help="A quoted string of arguments to pass to unpaper. Requires --clean. " - "Example: --unpaper-args '--layout double'.", -) -preprocessing.add_argument( - '--oversample', - metavar='DPI', - type=numeric(int, 0, 5000), - default=0, - help="Oversample images to at least the specified DPI, to improve OCR " - "results slightly", -) -preprocessing.add_argument( - '--remove-vectors', - action='store_true', - help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they " - "will not be included in OCR. This can eliminate false characters.", -) -preprocessing.add_argument( - '--mask-barcodes', - action='store_true', - help="EXPERIMENTAL. Mask out any barcodes that appear in the PDF so they are not " - "considered during OCR. Barcodes can introduce false characters into " - "OCR.", -) -preprocessing.add_argument( - '--threshold', - action='store_true', - help="EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract for OCR. Can " - "improve OCR quality compared to Tesseract's thresholder.", -) - -ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied") -ocrsettings.add_argument( - '-f', - '--force-ocr', - action='store_true', - help="Rasterize any text or vector objects on each page, apply OCR, and " - "save the rastered output (this rewrites the PDF)", -) -ocrsettings.add_argument( - '-s', - '--skip-text', - action='store_true', - help="Skip OCR on any pages that already contain text, but include the " - "page in final output; useful for PDFs that contain a mix of " - "images, text pages, and/or previously OCRed pages", -) -ocrsettings.add_argument( - '--redo-ocr', - action='store_true', - help="Attempt to detect and remove the hidden OCR layer from files that " - "were previously OCRed with OCRmyPDF or another program. Apply OCR " - "to text found in raster images. Existing visible text objects will " - "not be changed. If there is no existing OCR, OCR will be added.", -) -ocrsettings.add_argument( - '--skip-big', - type=numeric(float, 0, 5000), - metavar='MPixels', - help="Skip OCR on pages larger than the specified amount of megapixels, " - "but include skipped pages in final output", -) - -optimizing = parser.add_argument_group( - "Optimization options", "Control how the PDF is optimized after OCR" -) -optimizing.add_argument( - '-O', - '--optimize', - type=int, - choices=range(0, 4), - default=1, - help=( - "Control how PDF is optimized after processing:" - "0 - do not optimize; " - "1 - do safe, lossless optimizations (default); " - "2 - do some lossy optimizations; " - "3 - do aggressive lossy optimizations (including lossy JBIG2)" - ), -) -optimizing.add_argument( - '--jpeg-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - help=( - "Adjust JPEG quality level for JPEG optimization. " - "100 is best quality and largest output size; " - "1 is lowest quality and smallest output; " - "0 uses the default." - ), -) -optimizing.add_argument( - '--jpg-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - dest='jpeg_quality', - help=argparse.SUPPRESS, # Alias for --jpeg-quality -) -optimizing.add_argument( - '--png-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - help=( - "Adjust PNG quality level to use when quantizing PNGs. " - "Values have same meaning as with --jpeg-quality" - ), -) -optimizing.add_argument( - '--jbig2-lossy', - action='store_true', - help=( - "Enable JBIG2 lossy mode (better compression, not suitable for some " - "use cases - see documentation)." - ), -) -optimizing.add_argument( - '--jbig2-page-group-size', - type=numeric(int, 1, 10000), - default=0, - metavar='N', - # Adjust number of pages to consider at once for JBIG2 compression - help=argparse.SUPPRESS, -) - -advanced = parser.add_argument_group( - "Advanced", "Advanced options to control Tesseract's OCR behavior" -) -advanced.add_argument( - '--max-image-mpixels', - action='store', - type=numeric(float, 0), - metavar='MPixels', - help="Set maximum number of pixels to unpack before treating an image as a " - "decompression bomb", - default=128.0, -) -advanced.add_argument( - '--tesseract-config', - action='append', - metavar='CFG', - default=[], - help="Additional Tesseract configuration files -- see documentation", -) -advanced.add_argument( - '--tesseract-pagesegmode', - action='store', - type=int, - metavar='PSM', - choices=range(0, 14), - help="Set Tesseract page segmentation mode (see tesseract --help)", -) -advanced.add_argument( - '--tesseract-oem', - action='store', - type=int, - metavar='MODE', - choices=range(0, 4), - help=( - "Set Tesseract 4.0 OCR engine mode: " - "0 - original Tesseract only; " - "1 - neural nets LSTM only; " - "2 - Tesseract + LSTM; " - "3 - default." - ), -) -advanced.add_argument( - '--pdf-renderer', - choices=['auto', 'hocr', 'sandwich'], - default='auto', - help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " - "choose. See documentation for discussion.", -) -advanced.add_argument( - '--tesseract-timeout', - default=180.0, - type=numeric(float, 0), - metavar='SECONDS', - help='Give up on OCR after the timeout, but copy the preprocessed page ' - 'into the final output', -) -advanced.add_argument( - '--rotate-pages-threshold', - default=14.0, - type=numeric(float, 0, 1000), - metavar='CONFIDENCE', - help="Only rotate pages when confidence is above this value (arbitrary " - "units reported by tesseract)", -) -advanced.add_argument( - '--pdfa-image-compression', - choices=['auto', 'jpeg', 'lossless'], - default='auto', - help="Specify how to compress images in the output PDF/A. 'auto' lets " - "OCRmyPDF decide. 'jpeg' changes all grayscale and color images to " - "JPEG compression. 'lossless' uses PNG-style lossless compression " - "for all images. Monochrome images are always compressed using a " - "lossless codec. Compression settings " - "are applied to all pages, including those for which OCR was " - "skipped. Not supported for --output-type=pdf ; that setting " - "preserves the original compression of all images.", -) -advanced.add_argument( - '--user-words', - metavar='FILE', - help="Specify the location of the Tesseract user words file. This is a " - "list of words Tesseract should consider while performing OCR in " - "addition to its standard language dictionaries. This can improve " - "OCR quality especially for specialized and technical documents.", -) -advanced.add_argument( - '--user-patterns', - metavar='FILE', - help="Specify the location of the Tesseract user patterns file.", -) - -debugging = parser.add_argument_group( - "Debugging", "Arguments to help with troubleshooting and debugging" -) -debugging.add_argument( - '-k', - '--keep-temporary-files', - action='store_true', - help="Keep temporary files (helpful for debugging)", -) - - -class TqdmConsole: - """Wrapper to log messages in a way that is compatible with the progress bar""" - - def __init__(self, file): - self.file = file - self.py36 = sys.version_info >= (3, 6) - - def write(self, msg): - # When no progress bar is active, tqdm.write() routes to print() - if self.py36: - if msg.strip() != '': - tqdm.write(msg.rstrip(), end='\n', file=self.file) - else: - tqdm.write(msg.rstrip(), end='\n', file=self.file) - - def flush(self): - if hasattr(self.file, "flush"): - self.file.flush() - - -def setup_app_logging(options): - """Set up logging""" - - log = logging.getLogger() - log.setLevel(logging.INFO) - formatter = logging.Formatter('%(levelname)7s - %(message)s') - console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) - if options.quiet: - console.setLevel(logging.ERROR) - elif options.verbose >= 2: - console.setLevel(logging.DEBUG) - log.setLevel(logging.DEBUG) - else: - console.setLevel(logging.INFO) - console.setFormatter(formatter) - log.addHandler(console) - - pdfminer_log = logging.getLogger('pdfminer') - pdfminer_log.setLevel(logging.ERROR) - - -def configure_app_environment(options): - """Configure the application environment - - Don't do anything here that a library user would not expect. - """ - if not check_closed_streams(options): - return ExitCode.bad_args - if hasattr(os, 'nice'): - os.nice(5) +from .exceptions import ExitCode def run(args=None): options = parser.parse_args(args=args) - setup_app_logging(options) - configure_app_environment(options) - result = run_pipeline(options) + + if not check_closed_streams(options): + return ExitCode.bad_args + + if os.environ.get('PYTEST_CURRENT_TEST'): + os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file + if hasattr(os, 'nice'): + os.nice(5) + + configure_logging(options, manage_root_logger=True) + result = api_run(options=options) return result diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 06e600b7..fd98e63b 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -257,8 +257,6 @@ def run_pipeline(options): os.environ.setdefault('OMP_THREAD_LIMIT', '1') check_environ(options) - if os.environ.get('PYTEST_CURRENT_TEST'): - os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file work_folder = mkdtemp(prefix="com.github.ocrmypdf.") diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py new file mode 100644 index 00000000..2e08118a --- /dev/null +++ b/src/ocrmypdf/api.py @@ -0,0 +1,177 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import argparse +import logging +import sys +from pathlib import Path + +from tqdm import tqdm + +from .cli import parser +from ._sync import run_pipeline + + +class TqdmConsole: + """Wrapper to log messages in a way that is compatible with tqdm progress bar""" + + def __init__(self, file): + self.file = file + self.py36 = sys.version_info >= (3, 6) + + def write(self, msg): + # When no progress bar is active, tqdm.write() routes to print() + if self.py36: + if msg.strip() != '': + tqdm.write(msg.rstrip(), end='\n', file=self.file) + else: + tqdm.write(msg.rstrip(), end='\n', file=self.file) + + def flush(self): + if hasattr(self.file, "flush"): + self.file.flush() + + +def configure_logging(options, progress_bar_friendly=True, manage_root_logger=False): + """Set up logging + + Library users may wish to use this function if they want their log output to be + similar to ocrmypdf's when run as a command. If not use, the external application + should configure logging on its own. + + ocrmypdf will perform all of its logging under the `"ocrmypdf"` logging namespace. + In addition, ocrmypdf imports pdfminer, which logs under `"pdfminer"`. A library + user may wish to configure both; note that pdfminer is extremely chatty at the log + level logging.INFO. + + Library users may perform additional configuration afterwards. + + Args: + options: OCRmyPDF options + progress_bar_friendly (bool): install the TqdmConsole log handler, which is + compatible with the tqdm progress bar; without this log messages will + overwrite the progress bar + manage_root_logger (bool): configure the process's root logger, to ensure + all log output is sent through + """ + + prefix = '' if manage_root_logger else 'ocrmypdf' + log = logging.getLogger(prefix) + log.setLevel(logging.INFO) + + if progress_bar_friendly: + console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) + if options.quiet: + console.setLevel(logging.ERROR) + elif options.verbose >= 2: + console.setLevel(logging.DEBUG) + else: + console.setLevel(logging.INFO) + + formatter = logging.Formatter('%(levelname)7s - %(message)s') + + if options.verbose >= 2: + log.setLevel(logging.DEBUG) + + console.setFormatter(formatter) + log.addHandler(console) + + pdfminer_log = logging.getLogger('pdfminer') + pdfminer_log.setLevel(logging.ERROR) + + +def create_options(*, input_file, output_file, **kwargs): + cmdline = [] + + for arg, val in kwargs.items(): + if val is None: + continue + cmd_style_arg = arg.replace('_', '-') + cmdline.append(f"--{cmd_style_arg}") + if isinstance(val, bool): + continue + if isinstance(val, (int, float)): + cmdline.append(str(val)) + elif isinstance(val, str): + cmdline.append(val) + elif isinstance(val, Path): + cmdline.append(str(val)) + else: + raise TypeError(f"{val} ({type(val)})") + + cmdline.append(str(input_file)) + cmdline.append(str(output_file)) + + try: + options = parser.parse_args(cmdline) + except argparse.ArgumentError as e: + raise ValueError(str(e)) + return options + + +def ocrmypdf( # pylint: disable=unused-argument + input_file, + output_file, + *, + language=None, + image_dpi=None, + output_type=None, + sidecar=None, + jobs=None, + quiet=None, + verbose=None, + title=None, + author=None, + subject=None, + keywords=None, + rotate_pages=None, + remove_background=None, + deskew=None, + clean=None, + clean_final=None, + unpaper_args=None, + oversample=None, + remove_vectors=None, + mask_barcodes=None, + threshold=None, + force_ocr=None, + skip_text=None, + redo_ocr=None, + skip_big=None, + optimize=None, + jpg_quality=None, + png_quality=None, + jbig2_lossy=None, + jbig2_page_group_size=None, + max_image_mpixels=None, + tesseract_config=None, + tesseract_pagesegmode=None, + tesseract_oem=None, + pdf_renderer=None, + tesseract_timeout=None, + rotate_pages_threshold=None, + pdfa_image_compression=None, + user_words=None, + user_patterns=None, + keep_temporary_files=None, +): + options = create_options(**locals()) + return run(options) + + +def run(options): + return run_pipeline(options) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py new file mode 100644 index 00000000..73a3c452 --- /dev/null +++ b/src/ocrmypdf/cli.py @@ -0,0 +1,458 @@ +# © 2015-19 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import argparse + +from . import PROGRAM_NAME, VERSION + + +# ------------- +# Parser + + +def numeric(basetype, min_=None, max_=None): + """Validator for numeric params""" + min_ = basetype(min_) if min_ is not None else None + max_ = basetype(max_) if max_ is not None else None + + def _numeric(string): + value = basetype(string) + if (min_ is not None and value < min_) or (max_ is not None and value > max_): + msg = "%r not in valid range %r" % (string, (min_, max_)) + raise argparse.ArgumentTypeError(msg) + return value + + _numeric.__name__ = basetype.__name__ + return _numeric + + +parser = argparse.ArgumentParser( + prog=PROGRAM_NAME, + fromfile_prefix_chars='@', + formatter_class=argparse.RawDescriptionHelpFormatter, + description="""\ +Generates a searchable PDF or PDF/A from a regular PDF. + +OCRmyPDF rasterizes each page of the input PDF, optionally corrects page +rotation and performs image processing, runs the Tesseract OCR engine on the +image, and then creates a PDF from the OCR information. +""", + epilog="""\ +OCRmyPDF attempts to keep the output file at about the same size. If a file +contains losslessly compressed images, and output file will be losslessly +compressed as well. + +PDF is a page description file that attempts to preserve a layout exactly. +A PDF can contain vector objects (such as text or lines) and raster objects +(images). A page might have multiple images. OCRmyPDF is prepared to deal +with the wide variety of PDFs that exist in the wild. + +When a PDF page contains text, OCRmyPDF assumes that the page has already +been OCRed or is a "born digital" page that should not be OCRed. The default +behavior is to exit in this case without producing a file. You can use the +option --skip-text to ignore pages with text, or --force-ocr to rasterize +all objects on the page and produce an image-only PDF as output. + + ocrmypdf --skip-text file_with_some_text_pages.pdf output.pdf + + ocrmypdf --force-ocr word_document.pdf output.pdf + +If you are concerned about long-term archiving of PDFs, use the default option +--output-type pdfa which converts the PDF to a standardized PDF/A-2b. This +converts images to sRGB colorspace, removes some features from the PDF such +as Javascript or forms. If you want to minimize the number of changes made to +your PDF, use --output-type pdf. + +If OCRmyPDF is given an image file as input, it will attempt to convert the +image to a PDF before processing. For more control over the conversion of +images to PDF, use the Python package img2pdf or other image to PDF software. + +For example, this command uses img2pdf to convert all .png files beginning +with the 'page' prefix to a PDF, fitting each image on A4-sized paper, and +sending the result to OCRmyPDF through a pipe. img2pdf is a dependency of +ocrmypdf so it is already installed. + + img2pdf --pagesize A4 page*.png | ocrmypdf - myfile.pdf + +Online documentation is located at: + https://ocrmypdf.readthedocs.io/en/latest/introduction.html + +""", +) + +parser.add_argument( + 'input_file', + metavar="input_pdf_or_image", + help="PDF file containing the images to be OCRed (or '-' to read from " + "standard input)", +) +parser.add_argument( + 'output_file', + metavar="output_pdf", + help="Output searchable PDF file (or '-' to write to standard output). " + "Existing files will be ovewritten. If same as input file, the " + "input file will be updated only if processing is successful.", +) +parser.add_argument( + '-l', + '--language', + action='append', + help="Language(s) of the file to be OCRed (see tesseract --list-langs for " + "all language packs installed in your system). Use -l eng+deu for " + "multiple languages.", +) +parser.add_argument( + '--image-dpi', + metavar='DPI', + type=int, + help="For input image instead of PDF, use this DPI instead of file's.", +) +parser.add_argument( + '--output-type', + choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'], + default='pdfa', + help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for " + "long term archiving (default, recommended) but may not suitable " + "for users who want their file altered as little as possible. 'pdfa' " + "also has problems with full Unicode text. 'pdf' attempts to " + "preserve file contents as much as possible. 'pdf-a1' creates a " + "PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a " + "PDF/A3-b file.", +) + +# Use null string '\0' as sentinel to indicate the user supplied no argument, +# since that is the only invalid character for filepaths on all platforms +# bool('\0') is True in Python +parser.add_argument( + '--sidecar', + nargs='?', + const='\0', + default=None, + metavar='FILE', + help="Generate sidecar text files that contain the same text recognized " + "by Tesseract. This may be useful for building a OCR text database. " + "If FILE is omitted, the sidecar file be named {output_file}.txt " + "If FILE is set to '-', the sidecar is written to stdout (a " + "convenient way to preview OCR quality). The output file and sidecar " + "may not both use stdout at the same time.", +) + +parser.add_argument( + '--version', + action='version', + version=VERSION, + help="Print program version and exit", +) + +jobcontrol = parser.add_argument_group("Job control options") +jobcontrol.add_argument( + '-j', + '--jobs', + metavar='N', + type=numeric(int, 0, 256), + help="Use up to N CPU cores simultaneously (default: use all).", +) +jobcontrol.add_argument( + '-q', '--quiet', action='store_true', help="Suppress INFO messages" +) +jobcontrol.add_argument( + '-v', + '--verbose', + type=int, + default=0, + nargs='?', + help="Print more verbose messages for each additional verbose level. Use " + "`-v 1` typically for much more detailed logging. Higher numbers " + "are probably only useful in debugging.", +) + +metadata = parser.add_argument_group( + "Metadata options", + "Set output PDF/A metadata (default: copy input document's metadata)", +) +metadata.add_argument( + '--title', type=str, help="Set document title (place multiple words in quotes)" +) +metadata.add_argument('--author', type=str, help="Set document author") +metadata.add_argument('--subject', type=str, help="Set document subject description") +metadata.add_argument('--keywords', type=str, help="Set document keywords") + +preprocessing = parser.add_argument_group( + "Image preprocessing options", + "Options to improve the quality of the final PDF and OCR", +) +preprocessing.add_argument( + '-r', + '--rotate-pages', + action='store_true', + help="Automatically rotate pages based on detected text orientation", +) +preprocessing.add_argument( + '--remove-background', + action='store_true', + help="Attempt to remove background from gray or color pages, setting it " + "to white ", +) +preprocessing.add_argument( + '-d', '--deskew', action='store_true', help="Deskew each page before performing OCR" +) +preprocessing.add_argument( + '-c', + '--clean', + action='store_true', + help="Clean pages from scanning artifacts before performing OCR, and send " + "the cleaned page to OCR, but do not include the cleaned page in " + "the output", +) +preprocessing.add_argument( + '-i', + '--clean-final', + action='store_true', + help="Clean page as above, and incorporate the cleaned image in the final " + "PDF. Might remove desired content.", +) +preprocessing.add_argument( + '--unpaper-args', + type=str, + default=None, + help="A quoted string of arguments to pass to unpaper. Requires --clean. " + "Example: --unpaper-args '--layout double'.", +) +preprocessing.add_argument( + '--oversample', + metavar='DPI', + type=numeric(int, 0, 5000), + default=0, + help="Oversample images to at least the specified DPI, to improve OCR " + "results slightly", +) +preprocessing.add_argument( + '--remove-vectors', + action='store_true', + help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they " + "will not be included in OCR. This can eliminate false characters.", +) +preprocessing.add_argument( + '--mask-barcodes', + action='store_true', + help="EXPERIMENTAL. Mask out any barcodes that appear in the PDF so they are not " + "considered during OCR. Barcodes can introduce false characters into " + "OCR.", +) +preprocessing.add_argument( + '--threshold', + action='store_true', + help="EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract for OCR. Can " + "improve OCR quality compared to Tesseract's thresholder.", +) + +ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied") +ocrsettings.add_argument( + '-f', + '--force-ocr', + action='store_true', + help="Rasterize any text or vector objects on each page, apply OCR, and " + "save the rastered output (this rewrites the PDF)", +) +ocrsettings.add_argument( + '-s', + '--skip-text', + action='store_true', + help="Skip OCR on any pages that already contain text, but include the " + "page in final output; useful for PDFs that contain a mix of " + "images, text pages, and/or previously OCRed pages", +) +ocrsettings.add_argument( + '--redo-ocr', + action='store_true', + help="Attempt to detect and remove the hidden OCR layer from files that " + "were previously OCRed with OCRmyPDF or another program. Apply OCR " + "to text found in raster images. Existing visible text objects will " + "not be changed. If there is no existing OCR, OCR will be added.", +) +ocrsettings.add_argument( + '--skip-big', + type=numeric(float, 0, 5000), + metavar='MPixels', + help="Skip OCR on pages larger than the specified amount of megapixels, " + "but include skipped pages in final output", +) + +optimizing = parser.add_argument_group( + "Optimization options", "Control how the PDF is optimized after OCR" +) +optimizing.add_argument( + '-O', + '--optimize', + type=int, + choices=range(0, 4), + default=1, + help=( + "Control how PDF is optimized after processing:" + "0 - do not optimize; " + "1 - do safe, lossless optimizations (default); " + "2 - do some lossy optimizations; " + "3 - do aggressive lossy optimizations (including lossy JBIG2)" + ), +) +optimizing.add_argument( + '--jpeg-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + help=( + "Adjust JPEG quality level for JPEG optimization. " + "100 is best quality and largest output size; " + "1 is lowest quality and smallest output; " + "0 uses the default." + ), +) +optimizing.add_argument( + '--jpg-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + dest='jpeg_quality', + help=argparse.SUPPRESS, # Alias for --jpeg-quality +) +optimizing.add_argument( + '--png-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + help=( + "Adjust PNG quality level to use when quantizing PNGs. " + "Values have same meaning as with --jpeg-quality" + ), +) +optimizing.add_argument( + '--jbig2-lossy', + action='store_true', + help=( + "Enable JBIG2 lossy mode (better compression, not suitable for some " + "use cases - see documentation)." + ), +) +optimizing.add_argument( + '--jbig2-page-group-size', + type=numeric(int, 1, 10000), + default=0, + metavar='N', + # Adjust number of pages to consider at once for JBIG2 compression + help=argparse.SUPPRESS, +) + +advanced = parser.add_argument_group( + "Advanced", "Advanced options to control Tesseract's OCR behavior" +) +advanced.add_argument( + '--max-image-mpixels', + action='store', + type=numeric(float, 0), + metavar='MPixels', + help="Set maximum number of pixels to unpack before treating an image as a " + "decompression bomb", + default=128.0, +) +advanced.add_argument( + '--tesseract-config', + action='append', + metavar='CFG', + default=[], + help="Additional Tesseract configuration files -- see documentation", +) +advanced.add_argument( + '--tesseract-pagesegmode', + action='store', + type=int, + metavar='PSM', + choices=range(0, 14), + help="Set Tesseract page segmentation mode (see tesseract --help)", +) +advanced.add_argument( + '--tesseract-oem', + action='store', + type=int, + metavar='MODE', + choices=range(0, 4), + help=( + "Set Tesseract 4.0 OCR engine mode: " + "0 - original Tesseract only; " + "1 - neural nets LSTM only; " + "2 - Tesseract + LSTM; " + "3 - default." + ), +) +advanced.add_argument( + '--pdf-renderer', + choices=['auto', 'hocr', 'sandwich'], + default='auto', + help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " + "choose. See documentation for discussion.", +) +advanced.add_argument( + '--tesseract-timeout', + default=180.0, + type=numeric(float, 0), + metavar='SECONDS', + help='Give up on OCR after the timeout, but copy the preprocessed page ' + 'into the final output', +) +advanced.add_argument( + '--rotate-pages-threshold', + default=14.0, + type=numeric(float, 0, 1000), + metavar='CONFIDENCE', + help="Only rotate pages when confidence is above this value (arbitrary " + "units reported by tesseract)", +) +advanced.add_argument( + '--pdfa-image-compression', + choices=['auto', 'jpeg', 'lossless'], + default='auto', + help="Specify how to compress images in the output PDF/A. 'auto' lets " + "OCRmyPDF decide. 'jpeg' changes all grayscale and color images to " + "JPEG compression. 'lossless' uses PNG-style lossless compression " + "for all images. Monochrome images are always compressed using a " + "lossless codec. Compression settings " + "are applied to all pages, including those for which OCR was " + "skipped. Not supported for --output-type=pdf ; that setting " + "preserves the original compression of all images.", +) +advanced.add_argument( + '--user-words', + metavar='FILE', + help="Specify the location of the Tesseract user words file. This is a " + "list of words Tesseract should consider while performing OCR in " + "addition to its standard language dictionaries. This can improve " + "OCR quality especially for specialized and technical documents.", +) +advanced.add_argument( + '--user-patterns', + metavar='FILE', + help="Specify the location of the Tesseract user patterns file.", +) + +debugging = parser.add_argument_group( + "Debugging", "Arguments to help with troubleshooting and debugging" +) +debugging.add_argument( + '-k', + '--keep-temporary-files', + action='store_true', + help="Keep temporary files (helpful for debugging)", +) From 2fdaa76a0dfe9e5d0e87d671ee814d7404a4fce1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 20 May 2019 14:54:22 -0700 Subject: [PATCH 040/880] Refactor configure_logging --- docs/conf.py | 4 +--- src/ocrmypdf/__main__.py | 5 ++++- src/ocrmypdf/api.py | 30 +++++++++++++++++++----------- src/ocrmypdf/cli.py | 4 ---- 4 files changed, 24 insertions(+), 19 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index f1f32a01..b78cc050 100755 --- a/docs/conf.py +++ b/docs/conf.py @@ -30,9 +30,7 @@ # Add any Sphinx extension module names here, as strings. They can be # extensions coming with Sphinx (named 'sphinx.ext.*') or your custom # ones. -extensions = [ - # 'sphinx.ext.mathjax', -] +extensions = ['sphinx.ext.napoleon'] # Add any paths that contain templates here, relative to this directory. templates_path = ['_templates'] diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 23d29b4c..7b08f05e 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -37,7 +37,10 @@ def run(args=None): if hasattr(os, 'nice'): os.nice(5) - configure_logging(options, manage_root_logger=True) + verbosity = options.verbose + if options.quiet: + verbosity = -1 + configure_logging(verbosity, manage_root_logger=True) result = api_run(options=options) return result diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 2e08118a..acf60760 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -46,11 +46,11 @@ class TqdmConsole: self.file.flush() -def configure_logging(options, progress_bar_friendly=True, manage_root_logger=False): +def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=False): """Set up logging Library users may wish to use this function if they want their log output to be - similar to ocrmypdf's when run as a command. If not use, the external application + similar to ocrmypdf command line interface. If not used, the external application should configure logging on its own. ocrmypdf will perform all of its logging under the `"ocrmypdf"` logging namespace. @@ -61,11 +61,15 @@ def configure_logging(options, progress_bar_friendly=True, manage_root_logger=Fa Library users may perform additional configuration afterwards. Args: - options: OCRmyPDF options - progress_bar_friendly (bool): install the TqdmConsole log handler, which is + verbosity: Verbosity level. + * `-1`: Quiet + * `0`: Default + * `1`: Output ocrmypdf debug messages + * `2`: More detailed debugging from ocrmypdf and dependent modules + progress_bar_friendly (bool): Install the TqdmConsole log handler, which is compatible with the tqdm progress bar; without this log messages will overwrite the progress bar - manage_root_logger (bool): configure the process's root logger, to ensure + manage_root_logger (bool): Configure the process's root logger, to ensure all log output is sent through """ @@ -75,23 +79,27 @@ def configure_logging(options, progress_bar_friendly=True, manage_root_logger=Fa if progress_bar_friendly: console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) - if options.quiet: + if verbosity < 0: console.setLevel(logging.ERROR) - elif options.verbose >= 2: + elif verbosity >= 1: console.setLevel(logging.DEBUG) else: console.setLevel(logging.INFO) formatter = logging.Formatter('%(levelname)7s - %(message)s') - - if options.verbose >= 2: + if verbosity >= 1: log.setLevel(logging.DEBUG) + if verbosity >= 2: + formatter = logging.Formatter('%(name)s - %(levelname)7s - %(message)s') console.setFormatter(formatter) log.addHandler(console) - pdfminer_log = logging.getLogger('pdfminer') - pdfminer_log.setLevel(logging.ERROR) + if verbosity <= 1: + pdfminer_log = logging.getLogger('pdfminer') + pdfminer_log.setLevel(logging.ERROR) + pil_log = logging.getLogger('PIL') + pil_log.setLevel(logging.INFO) def create_options(*, input_file, output_file, **kwargs): diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 73a3c452..c83cfda9 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -20,10 +20,6 @@ import argparse from . import PROGRAM_NAME, VERSION -# ------------- -# Parser - - def numeric(basetype, min_=None, max_=None): """Validator for numeric params""" min_ = basetype(min_) if min_ is not None else None From e4baa8c0dd3cf56f18b7e51c09041ac3f03b6ca2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 20 May 2019 15:08:20 -0700 Subject: [PATCH 041/880] Remove sys.exit() calls so we don't terminate caller application --- src/ocrmypdf/exec/__init__.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 670651fb..b09b516b 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -164,12 +164,12 @@ def check_external_program( except (CalledProcessError, FileNotFoundError, MissingDependencyError): _error_missing_program(program, package, required_for, recommended) if not recommended: - sys.exit(ExitCode.missing_dependency) + raise MissingDependencyError() return if found_version < need_version: _error_old_version(program, package, need_version, found_version, required_for) if not recommended: - sys.exit(ExitCode.missing_dependency) + raise MissingDependencyError() - log.debug(f'Found {program} {found_version}') + log.debug('Found %s %s', program, found_version) From 32a076c039264e940c7575b05f1271a22c327832 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 20 May 2019 18:01:17 -0700 Subject: [PATCH 042/880] Refactor validation and exceptions CLI now tracks check_options exceptions. API now works more like an API, without an exception handler, because the caller should provide one. --- src/ocrmypdf/__main__.py | 29 ++++++++++++--- src/ocrmypdf/_sync.py | 10 +---- src/ocrmypdf/_validation.py | 74 ++++++++++++------------------------- src/ocrmypdf/api.py | 6 +-- tests/test_unpaper.py | 7 ++-- 5 files changed, 54 insertions(+), 72 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 7b08f05e..b821c858 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -20,13 +20,16 @@ import logging import os import sys +from . import __version__ from .cli import parser -from .api import configure_logging, run as api_run -from ._validation import check_closed_streams -from .exceptions import ExitCode +from .api import configure_logging +from ._jobcontext import make_logger +from ._sync import run_pipeline +from ._validation import check_closed_streams, check_options +from .exceptions import ExitCode, BadArgsError, MissingDependencyError -def run(args=None): +def main(args=None): options = parser.parse_args(args=args) if not check_closed_streams(options): @@ -41,9 +44,23 @@ def run(args=None): if options.quiet: verbosity = -1 configure_logging(verbosity, manage_root_logger=True) - result = api_run(options=options) + log = make_logger('ocrmypdf') + log.debug('ocrmypdf ' + __version__) + try: + check_options(options) + except ValueError as e: + log.error(e) + return ExitCode.bad_args + except BadArgsError as e: + log.error(e) + return e.exit_code + except MissingDependencyError as e: + log.error(e) + return ExitCode.missing_dependency + + result = run_pipeline(options=options) return result if __name__ == '__main__': - sys.exit(run()) + sys.exit(main()) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index fd98e63b..6a6aa2c5 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -28,7 +28,7 @@ from tempfile import mkdtemp from tqdm import tqdm -from . import VERSION +from . import __version__ from ._jobcontext import PDFContext, cleanup_working_files, make_logger from ._pipeline import ( convert_to_pdfa, @@ -237,12 +237,6 @@ def exec_concurrent(context): def run_pipeline(options): log = make_logger(options, __name__) - log.debug('ocrmypdf ' + VERSION) - - result = check_options(options) - if result != ExitCode.ok: - return result - check_dependency_versions(options) # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example @@ -256,8 +250,6 @@ def run_pipeline(options): # variable, but harmless to set if ignored. os.environ.setdefault('OMP_THREAD_LIMIT', '1') - check_environ(options) - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") atexit.register(cleanup_working_files, work_folder, options) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index fb1be0d1..07a8c481 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -20,7 +20,6 @@ import logging import os import sys -import textwrap from pathlib import Path import PIL @@ -28,7 +27,6 @@ import PIL from ._unicodefun import verify_python3_env from .exceptions import ( BadArgsError, - ExitCode, InputFileError, MissingDependencyError, OutputFileAccessError, @@ -243,26 +241,17 @@ def check_options_pillow(options): def check_options(options): - try: - check_options_languages(options) - check_options_metadata(options) - check_options_output(options) - check_options_sidecar(options) - check_options_preprocessing(options) - check_options_ocr_behavior(options) - check_options_optimizing(options) - check_options_advanced(options) - check_options_pillow(options) - return ExitCode.ok - except ValueError as e: - log.error(e) - return ExitCode.bad_args - except BadArgsError as e: - log.error(e) - return e.exit_code - except MissingDependencyError as e: - log.error(e) - return ExitCode.missing_dependency + check_options_languages(options) + check_options_metadata(options) + check_options_output(options) + check_options_sidecar(options) + check_options_preprocessing(options) + check_options_ocr_behavior(options) + check_options_optimizing(options) + check_options_advanced(options) + check_options_pillow(options) + check_dependency_versions(options) + check_environ(options) def check_closed_streams(options): @@ -334,11 +323,8 @@ def check_environ(options): for k in old_envvars: if k in os.environ: log.warning( - textwrap.dedent( - f"""\ - OCRmyPDF no longer uses the environment variable {k}. - Change PATH to select alternate programs.""" - ) + "OCRmyPDF no longer uses the environment variable {k}." + "Change PATH to select alternate programs." ) @@ -358,8 +344,7 @@ def create_input_file(options, work_folder): re_symlink(options.input_file, target) return target except FileNotFoundError: - log.error("File not found - %s", options.input_file) - raise InputFileError() + raise InputFileError(f"File not found - {options.input_file}") def check_input_file(options, start_input_file): @@ -374,27 +359,21 @@ def check_input_file(options, start_input_file): try: re_symlink(options.input_file, start_input_file) except FileNotFoundError: - log.error("File not found - %s", options.input_file) - raise InputFileError() + raise InputFileError(f"File not found - {options.input_file}") def check_requested_output_file(options): if options.output_file == '-': if sys.stdout.isatty(): - log.error( - textwrap.dedent( - """\ - Output was set to stdout '-' but it looks like stdout - is connected to a terminal. Please redirect stdout to a - file.""" - ) + raise BadArgsError( + "Output was set to stdout '-' but it looks like stdout " + "is connected to a terminal. Please redirect stdout to a " + "file." ) - raise BadArgsError() elif not is_file_writable(options.output_file): - log.error( - "Output file location (%s) is not a writable file.", options.output_file + raise OutputFileAccessError( + f"Output file location ({options.output_file}) is not a writable file." ) - raise OutputFileAccessError() def report_output_file_size(options, input_file, output_file): @@ -429,12 +408,8 @@ def report_output_file_size(options, input_file, output_file): explanation = "No reason for this increase is known. Please report this issue." log.warning( - textwrap.dedent( - f"""\ - The output file size is {ratio:.2f}× larger than the input file. - {explanation} - """ - ) + f"The output file size is {ratio:.2f}× larger than the input file.\n" + f"{explanation}" ) @@ -452,12 +427,11 @@ def check_dependency_versions(options): need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports ) if ghostscript.version() == '9.24': - log.error( + raise MissingDependencyError( "Ghostscript 9.24 contains serious regressions and is not " "supported. Please upgrade to Ghostscript 9.25 or use an older " "version." ) - return ExitCode.missing_dependency check_external_program( program='qpdf', package='qpdf', diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index acf60760..dc5da476 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -24,6 +24,7 @@ from tqdm import tqdm from .cli import parser from ._sync import run_pipeline +from ._validation import check_options class TqdmConsole: @@ -178,8 +179,5 @@ def ocrmypdf( # pylint: disable=unused-argument keep_temporary_files=None, ): options = create_options(**locals()) - return run(options) - - -def run(options): + check_options(options) return run_pipeline(options) diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 8df1d5a2..887184cb 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -19,9 +19,10 @@ from os import fspath from unittest.mock import patch import pytest -from ocrmypdf.__main__ import parser + +from ocrmypdf.cli import parser from ocrmypdf._validation import check_options -from ocrmypdf.exceptions import ExitCode +from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import unpaper # pytest.helpers is dynamic @@ -54,7 +55,7 @@ def test_no_unpaper(resources, no_outpdf): with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") - with pytest.raises(SystemExit): + with pytest.raises(MissingDependencyError): check_options(options) From 23dd77ce0ff0649be7773a3f58ff18f38b061665 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 15:31:03 -0700 Subject: [PATCH 043/880] api: fix progress_bar_friendly=False --- src/ocrmypdf/api.py | 17 +++++++++-------- 1 file changed, 9 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index dc5da476..7d9dcddf 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -80,12 +80,15 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= if progress_bar_friendly: console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) - if verbosity < 0: - console.setLevel(logging.ERROR) - elif verbosity >= 1: - console.setLevel(logging.DEBUG) - else: - console.setLevel(logging.INFO) + else: + console = logging.StreamHandler(stream=sys.stderr) + + if verbosity < 0: + console.setLevel(logging.ERROR) + elif verbosity >= 1: + console.setLevel(logging.DEBUG) + else: + console.setLevel(logging.INFO) formatter = logging.Formatter('%(levelname)7s - %(message)s') if verbosity >= 1: @@ -141,8 +144,6 @@ def ocrmypdf( # pylint: disable=unused-argument output_type=None, sidecar=None, jobs=None, - quiet=None, - verbose=None, title=None, author=None, subject=None, From 09ca1bee975cd830b7a259e6796b197983344886 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 15:31:48 -0700 Subject: [PATCH 044/880] Add progress bar to optimize and add option to disable it --- src/ocrmypdf/__main__.py | 5 ++- src/ocrmypdf/_sync.py | 6 +++- src/ocrmypdf/cli.py | 6 ++++ src/ocrmypdf/optimize.py | 72 +++++++++++++++++++++++----------------- 4 files changed, 57 insertions(+), 32 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index b821c858..78b3cd13 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -43,7 +43,10 @@ def main(args=None): verbosity = options.verbose if options.quiet: verbosity = -1 - configure_logging(verbosity, manage_root_logger=True) + options.progress_bar = False + configure_logging( + verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True + ) log = make_logger('ocrmypdf') log.debug('ocrmypdf ' + __version__) try: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 6a6aa2c5..5c509d22 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -195,7 +195,11 @@ def exec_concurrent(context): listener = threading.Thread(target=log_listener, args=(log_queue,)) listener.start() with tqdm( - total=(2 * len(context.pdfinfo)), desc='OCR', unit='page', unit_scale=0.5 + total=(2 * len(context.pdfinfo)), + desc='OCR', + unit='page', + unit_scale=0.5, + disable=not context.options.progress_bar, ) as pbar, multiprocessing.Pool( processes=max_workers, initializer=worker_init, initargs=(log_queue,) ) as pool: diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index c83cfda9..8491ae1e 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -175,6 +175,12 @@ jobcontrol.add_argument( "`-v 1` typically for much more detailed logging. Higher numbers " "are probably only useful in debugging.", ) +jobcontrol.add_argument( + '--no-progress-bar', + action='store_false', + dest='progress_bar', + help=argparse.SUPPRESS, +) metadata = parser.add_argument_group( "Metadata options", diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 87ce0212..fc1484f7 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -22,7 +22,7 @@ from os import fspath from pathlib import Path from PIL import Image - +from tqdm import tqdm import pikepdf from pikepdf import Name, Dictionary @@ -266,9 +266,13 @@ def _produce_jbig2_images(jbig2_groups, root, log, options): with concurrent.futures.ThreadPoolExecutor(max_workers=options.jobs) as executor: futures = jbig2_futures(executor, root, jbig2_groups) - for future in concurrent.futures.as_completed(futures): - proc = future.result() - log.debug(proc.stderr.decode()) + with tqdm( + total=len(jbig2_groups), desc="JBIG2", disable=not options.progress_bar + ) as pbar: + for future in concurrent.futures.as_completed(futures): + proc = future.result() + log.debug(proc.stderr.decode()) + pbar.update() def convert_to_jbig2(pike, jbig2_groups, root, log, options): @@ -310,7 +314,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options): def transcode_jpegs(pike, jpegs, root, log, options): - for xref in jpegs: + for xref in tqdm(jpegs, desc="JPEGs", disable=not options.progress_bar): in_jpg = Path(jpg_name(root, xref)) opt_jpg = in_jpg.with_suffix('.opt.jpg') @@ -339,15 +343,23 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): with concurrent.futures.ThreadPoolExecutor( max_workers=options.jobs ) as executor: + futures = [] for xref in images: log.debug(image_name_fn(root, xref)) - executor.submit( - pngquant.quantize, - image_name_fn(root, xref), - png_name(root, xref), - png_quality[0], - png_quality[1], + futures.append( + executor.submit( + pngquant.quantize, + image_name_fn(root, xref), + png_name(root, xref), + png_quality[0], + png_quality[1], + ) ) + with tqdm( + desc="PNGs", total=len(futures), disable=not options.progress_bar + ) as pbar: + for _future in concurrent.futures.as_completed(futures): + pbar.update() for xref in images: im_obj = pike.get_object(xref, 0) @@ -440,28 +452,27 @@ def optimize(input_file, output_file, context): if options.jbig2_page_group_size == 0: options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1 - pike = pikepdf.Pdf.open(input_file) + with pikepdf.Pdf.open(input_file) as pike: + root = Path(output_file).parent / 'images' + root.mkdir(exist_ok=True) - root = Path(output_file).parent / 'images' - root.mkdir(exist_ok=True) + jpegs, pngs = extract_images_generic(pike, root, log, options) + transcode_jpegs(pike, jpegs, root, log, options) + # if options.optimize >= 2: + # Try pngifying the jpegs + # transcode_pngs(pike, jpegs, jpg_name, root, log, options) + transcode_pngs(pike, pngs, png_name, root, log, options) - jpegs, pngs = extract_images_generic(pike, root, log, options) - transcode_jpegs(pike, jpegs, root, log, options) - # if options.optimize >= 2: - # Try pngifying the jpegs - # transcode_pngs(pike, jpegs, jpg_name, root, log, options) - transcode_pngs(pike, pngs, png_name, root, log, options) + jbig2_groups = extract_images_jbig2(pike, root, log, options) + convert_to_jbig2(pike, jbig2_groups, root, log, options) - jbig2_groups = extract_images_jbig2(pike, root, log, options) - convert_to_jbig2(pike, jbig2_groups, root, log, options) - - target_file = Path(output_file).with_suffix('.opt.pdf') - pike.remove_unreferenced_resources() - pike.save( - target_file, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, - ) + target_file = Path(output_file).with_suffix('.opt.pdf') + pike.remove_unreferenced_resources() + pike.save( + target_file, + preserve_pdfa=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + ) input_size = Path(input_file).stat().st_size output_size = Path(target_file).stat().st_size @@ -494,6 +505,7 @@ def main(infile, outfile, level, jobs=1): self.jbig2_page_group_size = 0 self.jbig2_lossy = jb2lossy self.quiet = True + self.progress_bar = False options = OptimizeOptions( input_file=infile, From 8bcb85720cef494e4c35255f8bf89c2c8122dfa2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 15:34:23 -0700 Subject: [PATCH 045/880] release notes: clarify --- docs/release_notes.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 6af0f263..38c94e29 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,9 +16,9 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and ar v8.3.0 ------ -- Improved the strategy for updating pages when a new image of the page was produced. We know attempt to preserve more content from the original file, for annotations in particular. +- Improved the strategy for updating pages when a new image of the page was produced. We now attempt to preserve more content from the original file, for annotations in particular. -- For PDFs with more than 100 pages and a sequence where one PDF page was replaced and one or more subsequent ones were skipped, an intermediate file would be corrupted while grafting OCR text, causing processing to fail. +- For PDFs with more than 100 pages and a sequence where one PDF page was replaced and one or more subsequent ones were skipped, an intermediate file would be corrupted while grafting OCR text, causing processing to fail. This is a regression, likely introduced in v8.2.4. - Previously, we resized the images produced by Ghostscript by a small number of pixels to ensure the output image size was an exactly what we wanted. Having discovered a way to get Ghostscript to produce the exact image sizes we require, we eliminated the resizing step. From db69b4d11a9b2116f4e9c76b00c4a880609e0e73 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 15:55:48 -0700 Subject: [PATCH 046/880] Improve argparse behavior for its role in making the API work --- src/ocrmypdf/api.py | 6 ++---- src/ocrmypdf/cli.py | 19 ++++++++++++++++++- 2 files changed, 20 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 7d9dcddf..fc22b20b 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -128,10 +128,8 @@ def create_options(*, input_file, output_file, **kwargs): cmdline.append(str(input_file)) cmdline.append(str(output_file)) - try: - options = parser.parse_args(cmdline) - except argparse.ArgumentError as e: - raise ValueError(str(e)) + parser.api_mode = True + options = parser.parse_args(cmdline) return options diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 8491ae1e..503f591b 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -36,7 +36,24 @@ def numeric(basetype, min_=None, max_=None): return _numeric -parser = argparse.ArgumentParser( +class ArgumentParser(argparse.ArgumentParser): + """Override parser's default behavior of calling sys.exit() + + https://stackoverflow.com/questions/5943249/python-argparse-and-controlling-overriding-the-exit-status-code + """ + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + self.api_mode = False + + def error(self, message): + if not self.api_mode: + super().error(message) + return + raise ValueError(message) + + +parser = ArgumentParser( prog=PROGRAM_NAME, fromfile_prefix_chars='@', formatter_class=argparse.RawDescriptionHelpFormatter, From a139e64c679977c479d16a4dfe4ce05023b7243c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 18:30:30 -0700 Subject: [PATCH 047/880] api: short-circuit exception handler, as caller should provide their own --- src/ocrmypdf/_sync.py | 13 +++++++++++-- src/ocrmypdf/api.py | 2 +- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 5c509d22..a1a3f083 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -239,7 +239,7 @@ def exec_concurrent(context): copy_final(pdf, context.options.output_file, context) -def run_pipeline(options): +def run_pipeline(options, api=False): log = make_logger(options, __name__) # Any changes to options will not take effect for options that are already @@ -277,12 +277,21 @@ def run_pipeline(options): # Execute the pipeline exec_concurrent(context) except KeyboardInterrupt as e: + if api: + raise log.error("KeyboardInterrupt") return ExitCode.ctrl_c except ExitCodeException as e: - log.error("%s: %s" % (type(e).__name__, str(e))) + if api: + raise + if str(e): + log.error("%s: %s", type(e).__name__, str(e)) + else: + log.error(type(e).__name__) return e.exit_code except Exception as e: + if api: + raise log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index fc22b20b..2d6ccecf 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -179,4 +179,4 @@ def ocrmypdf( # pylint: disable=unused-argument ): options = create_options(**locals()) check_options(options) - return run_pipeline(options) + return run_pipeline(options, api=True) From 5cecb3ecb4865d846e7e8c98ce57a818d0c3830d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 May 2019 23:53:48 -0700 Subject: [PATCH 048/880] Convert one test to use API --- tests/conftest.py | 17 +++++++++++++++++ tests/test_weave.py | 25 ++++++++----------------- 2 files changed, 25 insertions(+), 17 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 19679fdf..8ce8f5fb 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -18,6 +18,7 @@ import os import platform import sys +from contextlib import contextmanager from pathlib import Path from subprocess import PIPE, run @@ -115,6 +116,22 @@ def spoof(tmpdir_factory, **kwargs): return env +@pytest.helpers.register +@contextmanager +def os_environ(new_env): + old_env = os.environ.copy() + + for k, v in new_env.items(): + os.environ[k] = v + yield + new_keys = set(os.environ.copy()) - set(old_env) + for k in new_keys: + del os.environ[k] + for k in old_env: + os.environ[k] = old_env[k] + assert os.environ.copy() == old_env + + @pytest.fixture(scope='session') def spoof_tesseract_noop(tmpdir_factory): return spoof(tmpdir_factory, tesseract='tesseract_noop.py') diff --git a/tests/test_weave.py b/tests/test_weave.py index 06fa2a0a..5c72a8fc 100644 --- a/tests/test_weave.py +++ b/tests/test_weave.py @@ -19,9 +19,10 @@ import os import pytest +from ocrmypdf import ocrmypdf import pikepdf -check_ocrmypdf = pytest.helpers.check_ocrmypdf +os_environ = pytest.helpers.os_environ def test_no_glyphless_weave(resources, outdir): @@ -34,26 +35,16 @@ def test_no_glyphless_weave(resources, outdir): env = os.environ.copy() env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2' - check_ocrmypdf( - outdir / 'test.pdf', - outdir / 'out.pdf', - '--deskew', - '--tesseract-timeout', - '0', - env=env, - ) + with os_environ(env): + ocrmypdf( + outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0 + ) @pytest.helpers.needs_pdfminer def test_links(resources, outpdf): - check_ocrmypdf( - resources / 'link.pdf', - outpdf, - '--redo-ocr', - '--oversample', - '200', - '--output-type', - 'pdf', + ocrmypdf( + resources / 'link.pdf', outpdf, redo_ocr=True, oversample=200, output_type='pdf' ) pdf = pikepdf.open(outpdf) p1 = pdf.pages[0] From 22298b31becda46fd444420a4c8d9937d2e4644e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 23 May 2019 01:19:58 -0700 Subject: [PATCH 049/880] Fix distinction between clean and clean_final lost in API refactor --- src/ocrmypdf/_sync.py | 19 ++++++++++++++----- src/ocrmypdf/_validation.py | 2 +- 2 files changed, 15 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a1a3f083..45ccdca9 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -92,17 +92,26 @@ def exec_page_sync(page_context): page_context.origin, page_context, correction=orientation_correction ) - preprocess_out = rasterize_out + preprocess = rasterize_out if options.remove_background: - preprocess_out = preprocess_remove_background(preprocess_out, page_context) + preprocess = preprocess_remove_background(preprocess, page_context) if options.deskew: - preprocess_out = preprocess_deskew(preprocess_out, page_context) + preprocess = preprocess_deskew(preprocess, page_context) if options.clean: - preprocess_out = preprocess_clean(preprocess_out, page_context) + cleaned = preprocess_clean(preprocess, page_context) + if options.clean_final: + preprocess_out = cleaned + ocr_image = cleaned + else: + preprocess_out = preprocess + ocr_image = cleaned + else: + preprocess_out = preprocess + ocr_image = preprocess - ocr_image_out = create_ocr_image(preprocess_out, page_context) + ocr_image_out = create_ocr_image(ocr_image, page_context) pdf_page_from_image_out = None if not options.lossless_reconstruction: diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 07a8c481..3e0e3526 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -136,7 +136,7 @@ def check_options_sidecar(options): if options.sidecar == '\0': if options.output_file == '-': raise BadArgsError( - "--sidecar filename must be specified when output file is " "stdout." + "--sidecar filename must be specified when output file is stdout." ) options.sidecar = options.output_file + '.txt' From d0efdf643cb1ac35af60abff6a6a223609aa0800 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 23 May 2019 01:25:08 -0700 Subject: [PATCH 050/880] Cleanup working files when done with a particular file, rather than end of process --- src/ocrmypdf/_sync.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 45ccdca9..3141fed3 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import atexit import logging import logging.handlers import multiprocessing @@ -264,9 +263,6 @@ def run_pipeline(options, api=False): os.environ.setdefault('OMP_THREAD_LIMIT', '1') work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - - atexit.register(cleanup_working_files, work_folder, options) - try: check_requested_output_file(options) start_input_file = create_input_file(options, work_folder) @@ -303,6 +299,8 @@ def run_pipeline(options, api=False): raise log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error + finally: + cleanup_working_files(work_folder, options) if options.output_file == '-': log.info("Output sent to stdout") From 805aa776ad31e7ff701c92d14eead3e33b90cb39 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 23 May 2019 02:00:35 -0700 Subject: [PATCH 051/880] Re-disable progress bar when not connected to tty --- src/ocrmypdf/__main__.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 78b3cd13..b5997d79 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -41,6 +41,8 @@ def main(args=None): os.nice(5) verbosity = options.verbose + if not os.isatty(sys.stderr.fileno()): + options.progress_bar = False if options.quiet: verbosity = -1 options.progress_bar = False From db6aa22eaef9b564e7809f4ccc711115e1cd15da Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 23 May 2019 02:00:47 -0700 Subject: [PATCH 052/880] Progress bar: unit types --- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/optimize.py | 14 +++++++++++--- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 2d6ccecf..2344970e 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import argparse import logging import sys from pathlib import Path @@ -176,6 +175,7 @@ def ocrmypdf( # pylint: disable=unused-argument user_words=None, user_patterns=None, keep_temporary_files=None, + progress_bar=None, ): options = create_options(**locals()) check_options(options) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index fc1484f7..21351291 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -267,7 +267,10 @@ def _produce_jbig2_images(jbig2_groups, root, log, options): with concurrent.futures.ThreadPoolExecutor(max_workers=options.jobs) as executor: futures = jbig2_futures(executor, root, jbig2_groups) with tqdm( - total=len(jbig2_groups), desc="JBIG2", disable=not options.progress_bar + total=len(jbig2_groups), + desc="JBIG2", + unit='item', + disable=not options.progress_bar, ) as pbar: for future in concurrent.futures.as_completed(futures): proc = future.result() @@ -314,7 +317,9 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options): def transcode_jpegs(pike, jpegs, root, log, options): - for xref in tqdm(jpegs, desc="JPEGs", disable=not options.progress_bar): + for xref in tqdm( + jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar + ): in_jpg = Path(jpg_name(root, xref)) opt_jpg = in_jpg.with_suffix('.opt.jpg') @@ -356,7 +361,10 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): ) ) with tqdm( - desc="PNGs", total=len(futures), disable=not options.progress_bar + desc="PNGs", + total=len(futures), + unit='image', + disable=not options.progress_bar, ) as pbar: for _future in concurrent.futures.as_completed(futures): pbar.update() From ed236e0c27d0f57a68ecae673ce65956412d1ade Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 24 May 2019 01:05:32 -0700 Subject: [PATCH 053/880] Begin API documentation --- docs/api.rst | 72 ++++++++++++++++++++++++++++++++++++++++ docs/batch.rst | 8 ----- docs/index.rst | 1 + docs/introduction.rst | 1 - docs/release_notes.rst | 2 -- src/ocrmypdf/__init__.py | 2 +- src/ocrmypdf/__main__.py | 8 ++--- src/ocrmypdf/api.py | 45 +++++++++++++++++++++---- 8 files changed, 117 insertions(+), 22 deletions(-) create mode 100644 docs/api.rst diff --git a/docs/api.rst b/docs/api.rst new file mode 100644 index 00000000..d4aeeca5 --- /dev/null +++ b/docs/api.rst @@ -0,0 +1,72 @@ +Using the OCRmyPDF API +====================== + +OCRmyPDF originated as a command line program and continues to have this legacy, but parts of it can be imported and used in other Python applications. + +Some applications may want to consider running ocrmypdf from a subprocess call anyway, as this provides isolation of its activities. + +Example +------- + +OCRmyPDF one high-level function to run its main engine from an application. The parameters are symmetric to the command line arguments and largely have the same functions. + +.. code-block:: python + + from ocrmypdf import ocrmypdf + + ocrmypdf('input.pdf', 'output.pdf', deskew=True) + +With a few exceptions, all of the command line arguments are available and may be passed as equivalent keywords. + +A few differences are that ``verbose`` and ``quiet`` are not available. Instead, output should be managed by configuring logging. + +Parent process requirements +^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +The :func:`ocrmypdf.ocrmypdf` function runs OCRmyPDF similar to command line execution. To do this, it will: +- create a monitoring thread +- create worker processes (forking itself) +- manage the signal flags of worker processes +0 execute other subprocesses (forking and executing other programs) + +The Python process that calls ``ocrmypdf()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will fail. + +There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. + +Forking a child process to call ``ocrmypdf()`` is suggested. That way your application will survive even if OCRmyPDF does not. + +Logging +^^^^^^^ + +OCRmyPDF will log under loggers named ``ocrmypdf``. In addition, it imports ``pdfminer`` and ``PIL``, both of which post log messages under those logging namespaces. + +You can configure the logging as desired for your application or call :func:`ocrmypdf.configure_logging` to configure logging the same way OCRmyPDF itself does. The command line parameters such as ``--quiet`` and ``--verbose`` have no equivalents in the API; you must configure logging. + +Progress monitoring +^^^^^^^^^^^^^^^^^^^ + +OCRmyPDF uses the ``tqdm`` package to implement its progress bars. :func:`ocrmypdf.configure_logging` will set up logging output to ``sys.stderr`` in a way that is compatible with the display of the progress bar. + +Exceptions +^^^^^^^^^^ + +OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*`` exceptions, some exceptions related to multiprocessing, and ``KeyboardInterrupt``. The parent process should provide an exception handler. OCRmyPDF will clean up its temporary files and worker processes automatically when an exception occurs. + +Programs that call OCRmyPDF should consider trapping KeyboardInterrupt so that they allow OCR to terminate with the whole program terminating. + +When OCRmyPDF succeeds conditionally, it may return an integer exit code. + +Reference +--------- + +.. autofunction:: ocrmypdf.ocrmypdf + +.. autoclass:: ocrmypdf.Verbosity + :members: + :undoc-members: + +.. autoclass:: ocrmypdf.ExitCode + :members: + :undoc-members: + +.. autofunction:: ocrmypdf.configure_logging diff --git a/docs/batch.rst b/docs/batch.rst index a98fd142..7ae20779 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -95,14 +95,6 @@ This user contributed script also provides an example of batch processing. print("OCR complete") logging.info(result) -API -""" - -OCRmyPDF is currently supported as a command line interface. This means that even if you are using OCRmyPDF in a Python script, you should run it in a subprocess rather importing the ocrmypdf package. - -(If you find individual functions implemented in OCRmyPDF useful (such as ``ocrmypdf.pdfinfo``), you can use these if you wish to.) - - Synology DiskStations """"""""""""""""""""" diff --git a/docs/index.rst b/docs/index.rst index 3329eea8..5936e7da 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -27,6 +27,7 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat cookbook docker advanced + api batch security errors diff --git a/docs/introduction.rst b/docs/introduction.rst index f7f80568..03780225 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -95,7 +95,6 @@ Ghostscript also imposes some limitations: Regarding OCRmyPDF itself: * PDFs that use transparency are not currently represented in the test suite -* The Python API exported by ``import ocrmypdf`` is design to help scripts that use OCRmyPDF but is not currently capable of running OCRmyPDF jobs due to limitations in an underlying library. Similar programs ---------------- diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 38c94e29..13c880d1 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -5,8 +5,6 @@ OCRmyPDF uses `semantic versioning `_ for its command line i The ``ocrmypdf`` package may now be imported. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. -Unfortunately, the public API does **not** expose the ability to actually OCR a PDF. This is due to a limitation in an underlying library (ruffus) that makes OCRmyPDF non-reentrant. - Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. .. Issue regex diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index a7211a25..29491ca9 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,4 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo -from .api import ocrmypdf +from .api import ocrmypdf, configure_logging, Verbosity diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index b5997d79..0ec9050f 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -22,14 +22,14 @@ import sys from . import __version__ from .cli import parser -from .api import configure_logging +from .api import configure_logging, Verbosity from ._jobcontext import make_logger from ._sync import run_pipeline from ._validation import check_closed_streams, check_options from .exceptions import ExitCode, BadArgsError, MissingDependencyError -def main(args=None): +def run(args=None): options = parser.parse_args(args=args) if not check_closed_streams(options): @@ -44,7 +44,7 @@ def main(args=None): if not os.isatty(sys.stderr.fileno()): options.progress_bar = False if options.quiet: - verbosity = -1 + verbosity = Verbosity.quiet options.progress_bar = False configure_logging( verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True @@ -68,4 +68,4 @@ def main(args=None): if __name__ == '__main__': - sys.exit(main()) + sys.exit(run()) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 2344970e..e371324d 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -17,6 +17,7 @@ import logging import sys +from enum import IntEnum from pathlib import Path from tqdm import tqdm @@ -46,8 +47,17 @@ class TqdmConsole: self.file.flush() +class Verbosity(IntEnum): + """Verbosity level for configure_logging.""" + + quiet = -1 #: Suppress most messages + default = 0 #: Default level of logging + debug = 1 #: Output ocrmypdf debug messages + debug_all = 2 #: More detailed debugging from ocrmypdf and dependent modules + + def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=False): - """Set up logging + """Set up logging. Library users may wish to use this function if they want their log output to be similar to ocrmypdf command line interface. If not used, the external application @@ -61,11 +71,7 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= Library users may perform additional configuration afterwards. Args: - verbosity: Verbosity level. - * `-1`: Quiet - * `0`: Default - * `1`: Output ocrmypdf debug messages - * `2`: More detailed debugging from ocrmypdf and dependent modules + verbosity (Verbosity): Verbosity level. progress_bar_friendly (bool): Install the TqdmConsole log handler, which is compatible with the tqdm progress bar; without this log messages will overwrite the progress bar @@ -176,7 +182,34 @@ def ocrmypdf( # pylint: disable=unused-argument user_patterns=None, keep_temporary_files=None, progress_bar=None, + process_ocr_image=None, ): + """Run OCRmyPDF on one PDF or image. + + Raises: + ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging + with the OCR layer. + ocrmypdf.MissingDependencyError: If a required dependency program is missing or + was not found on PATH. + ocrmypdf.UnsupportedImageFormatError: If the input file type was an image that + could not be read, or some other file type that is not a PDF. + ocrmypdf.DpiError: If the input file is an image, but the resolution of the + image is not credible (allowing it to proceed would cause poor OCR). + ocrmypdf.OutputFileAccessError: If an attempt to write to the intended output + file failed. + ocrmypdf.PriorOcrFoundError: If the input PDF seems to have OCR or digital + text already, and settings did not tell us to proceed. + ocrmypdf.InputFileError: Any other problem with the input file. + ocrmypdf.SubprocessOutputError: Any error related to executing a subprocess. + ocrmypdf.EncryptedPdfERror: If the input PDF is encrypted (password protected). + OCRmyPDF does not remove passwords. + ocrmypdf.TesseractConfigError: If Tesseract reported its configuration was not + valid. + + Returns: + :class:`ocrmypdf.ExitCode` + """ + options = create_options(**locals()) check_options(options) return run_pipeline(options, api=True) From 24855045e1dfc8a26d9c2c1f0a4b42be16849626 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 25 May 2019 16:23:39 -0700 Subject: [PATCH 054/880] Provisionally add filters --- src/ocrmypdf/_pipeline.py | 3 +++ src/ocrmypdf/api.py | 6 ++++++ src/ocrmypdf/cli.py | 4 ++++ 3 files changed, 13 insertions(+) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index d2881407..70657a3a 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -521,6 +521,9 @@ def create_ocr_image(image, page_context): draw.rectangle(rect, fill=white) im = pix.topil() + if options.filter_ocr_image: + im = options.filter_ocr_image(im) + del draw # Pillow requires integer DPI dpi = round(xres), round(yres) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index e371324d..701e3b73 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -113,10 +113,14 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= def create_options(*, input_file, output_file, **kwargs): cmdline = [] + filters = [] for arg, val in kwargs.items(): if val is None: continue + if arg.startswith('filter') and callable(val): + filters.append((arg, val)) + continue cmd_style_arg = arg.replace('_', '-') cmdline.append(f"--{cmd_style_arg}") if isinstance(val, bool): @@ -135,6 +139,8 @@ def create_options(*, input_file, output_file, **kwargs): parser.api_mode = True options = parser.parse_args(cmdline) + for keyword, function in filters: + setattr(options, keyword, function) return options diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 503f591b..1480d677 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -466,6 +466,10 @@ advanced.add_argument( help="Specify the location of the Tesseract user patterns file.", ) + +filters = parser.add_argument_group("Filters", argparse.SUPPRESS) +filters.add_argument('--filter-ocr-image', help=argparse.SUPPRESS) + debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" ) From c14f62752b12db78e2115f3431ad0b59054fecdb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 25 May 2019 16:24:09 -0700 Subject: [PATCH 055/880] Tests: add an API test --- tests/test_main.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index 128ca797..36348479 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -27,6 +27,7 @@ import PIL import pytest from PIL import Image +from ocrmypdf import ocrmypdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import ghostscript, qpdf, tesseract from ocrmypdf.leptonica import Pix @@ -39,6 +40,7 @@ from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof +os_environ = pytest.helpers.os_environ RENDERERS = ['hocr', 'sandwich'] @@ -609,11 +611,8 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): def test_masks(spoof_tesseract_noop, resources, outpdf): - p, out, err = run_ocrmypdf( - resources / 'masks.pdf', outpdf, env=spoof_tesseract_noop - ) - - assert p.returncode == ExitCode.ok + with os_environ(spoof_tesseract_noop): + assert ocrmypdf(resources / 'masks.pdf', outpdf) == ExitCode.ok def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop, resources, outpdf): From 0628a890419aff1119dbc32996b7d0158f98289f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 May 2019 00:15:14 -0700 Subject: [PATCH 056/880] docs: mention how to use Docker image shell --- .gitignore | 2 +- docs/docker.rst | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/.gitignore b/.gitignore index 83ec1729..1504fb47 100644 --- a/.gitignore +++ b/.gitignore @@ -42,4 +42,4 @@ tests/output/ tests/resources/private/ tmp/ /debug_tests.py -*.traineddata +private/ diff --git a/docs/docker.rst b/docs/docker.rst index 73f0a3e8..2bbcb2c4 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -76,6 +76,8 @@ By default the Docker image includes English, German and Simplified Chinese, the # Add French RUN apk add tesseract-ocr-data-fra +You can also copy training data to ``/usr/share/tessdata``. + Executing the test suite ------------------------ @@ -85,6 +87,15 @@ The OCRmyPDF test suite is installed with image. To run it: docker run --entrypoint python3 jbarlow83/ocrmypdf-alpine setup.py test +Accessing the shell +------------------- + +``bash`` is not installed in the image. To use the busybox shell in the Docker image: + +.. code-block:: bash + + docker run -it --entrypoint busybox jbarlow83/ocrmypdf-alpine sh + Using the OCRmyPDF web service wrapper -------------------------------------- From e9731b6bac5047a5a4427bb01ce5eda6056085ea Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 May 2019 03:46:53 -0700 Subject: [PATCH 057/880] Docker: upgrade pip, temporarily enable community repository for qpdf --- .docker/alpine.dockerfile | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index d54951f6..5a16c843 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -4,25 +4,30 @@ FROM base as builder ENV LANG=C.UTF-8 +# Normally: +# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories + RUN \ - echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories \ + echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\ + >> /etc/apk/repositories \ # Add runtime dependencies && apk add --update \ python3-dev \ py3-setuptools \ jbig2enc@testing \ ghostscript \ - qpdf \ + qpdf@community \ tesseract-ocr \ unpaper \ pngquant \ libxml2-dev \ libxslt-dev \ zlib-dev \ - qpdf-dev \ + qpdf-dev@community \ libffi-dev \ leptonica-dev \ binutils \ + && pip3 install --upgrade pip \ # Install pybind11 for pikepdf && pip3 install pybind11 \ # Install flask for the webservice @@ -42,14 +47,18 @@ FROM base ENV LANG=C.UTF-8 +# Normally: +# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories + RUN \ - echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories \ + echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\ + >> /etc/apk/repositories \ # Add runtime dependencies && apk add --update \ python3 \ jbig2enc@testing \ ghostscript \ - qpdf \ + qpdf@community \ tesseract-ocr \ tesseract-ocr-data-deu \ tesseract-ocr-data-chi_sim \ @@ -58,7 +67,6 @@ RUN \ libxml2 \ libxslt \ zlib \ - qpdf \ libffi \ leptonica-dev \ binutils \ From 8d0958d7eaec90fae010f2dffc5bc91ca0e43251 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 May 2019 04:30:34 -0700 Subject: [PATCH 058/880] Dockerfile: qpdf-dev needs to be requested explicitly --- .docker/alpine.dockerfile | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index 5a16c843..264968fc 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -17,13 +17,13 @@ RUN \ jbig2enc@testing \ ghostscript \ qpdf@community \ + qpdf-dev@community \ tesseract-ocr \ unpaper \ pngquant \ libxml2-dev \ libxslt-dev \ zlib-dev \ - qpdf-dev@community \ libffi-dev \ leptonica-dev \ binutils \ @@ -59,6 +59,7 @@ RUN \ jbig2enc@testing \ ghostscript \ qpdf@community \ + qpdf-dev@community \ tesseract-ocr \ tesseract-ocr-data-deu \ tesseract-ocr-data-chi_sim \ From 692f7b31516fe73265f5c3adcc4926140efdc5eb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 May 2019 04:31:53 -0700 Subject: [PATCH 059/880] Dockerfile: with newer pip Newer pip seems to install ocrmypdf-*.dist-info and has no problem reporting installed version unlike -egg-info, so skip copying. Also move WORKDIR --- .docker/alpine.dockerfile | 62 +++++++++++++++++++-------------------- 1 file changed, 31 insertions(+), 31 deletions(-) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index 264968fc..7145199f 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -12,21 +12,21 @@ RUN \ >> /etc/apk/repositories \ # Add runtime dependencies && apk add --update \ - python3-dev \ - py3-setuptools \ - jbig2enc@testing \ - ghostscript \ + python3-dev \ + py3-setuptools \ + jbig2enc@testing \ + ghostscript \ qpdf@community \ qpdf-dev@community \ - tesseract-ocr \ - unpaper \ - pngquant \ - libxml2-dev \ - libxslt-dev \ - zlib-dev \ - libffi-dev \ - leptonica-dev \ - binutils \ + tesseract-ocr \ + unpaper \ + pngquant \ + libxml2-dev \ + libxslt-dev \ + zlib-dev \ + libffi-dev \ + leptonica-dev \ + binutils \ && pip3 install --upgrade pip \ # Install pybind11 for pikepdf && pip3 install pybind11 \ @@ -34,8 +34,8 @@ RUN \ && pip3 install flask \ # Add build dependencies && apk add --virtual build-dependencies \ - build-base \ - git + build-base \ + git COPY . /app @@ -55,22 +55,22 @@ RUN \ >> /etc/apk/repositories \ # Add runtime dependencies && apk add --update \ - python3 \ - jbig2enc@testing \ - ghostscript \ + python3 \ + jbig2enc@testing \ + ghostscript \ qpdf@community \ qpdf-dev@community \ - tesseract-ocr \ - tesseract-ocr-data-deu \ - tesseract-ocr-data-chi_sim \ - unpaper \ - pngquant \ - libxml2 \ - libxslt \ - zlib \ - libffi \ - leptonica-dev \ - binutils \ + tesseract-ocr \ + tesseract-ocr-data-deu \ + tesseract-ocr-data-chi_sim \ + unpaper \ + pngquant \ + libxml2 \ + libxslt \ + zlib \ + libffi \ + leptonica-dev \ + binutils \ && mkdir /app WORKDIR /app @@ -87,7 +87,7 @@ COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ COPY --from=builder /app/requirements /app/requirements COPY --from=builder /app/tests /app/tests COPY --from=builder /app/src /app/src -# Copy PKG-INFO from build artifact in app dir to make setuptools-scm happy -RUN cp /usr/lib/python3.6/site-packages/ocrmypdf-*.egg-info/PKG-INFO /app + +WORKDIR /data ENTRYPOINT ["/usr/bin/ocrmypdf"] From 5c4c32ab3c833bdf93c3d17ac6831e00ae376b8a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 27 May 2019 12:07:20 -0700 Subject: [PATCH 060/880] Remove multiprocessing tests - no longer valid --- tests/_test_multiprocessing.py | 63 ---------------------------------- 1 file changed, 63 deletions(-) delete mode 100644 tests/_test_multiprocessing.py diff --git a/tests/_test_multiprocessing.py b/tests/_test_multiprocessing.py deleted file mode 100644 index a25457ba..00000000 --- a/tests/_test_multiprocessing.py +++ /dev/null @@ -1,63 +0,0 @@ -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -from multiprocessing import Process -from multiprocessing.managers import BaseProxy - -from ocrmypdf._jobcontext import JobContext, JobContextManager -from ocrmypdf.pdfinfo import PageInfo, PdfInfo - - -def test_jobcontext_proxy(resources): - # Prove that managers are set up correctly to share state among processes - manager = JobContextManager() - manager.register('JobContext', JobContext) - - # Start the manager in a child process (or maybe thread) - manager.start() - - # Tell the manager process to retrieve pdf info - context = manager.JobContext() - context.generate_pdfinfo(resources / 'graph.pdf') - - # Get a copy of that information for this process - pdfinfo = context.get_pdfinfo() - assert len(pdfinfo) == 1 - assert pdfinfo[0].rotation == 0 - - # Update information and send back to manager - pdfinfo[0].rotation = 90 - context.set_pdfinfo(pdfinfo) - - # Retrieve again, ensure it stayed changed - pdfinfo2 = context.get_pdfinfo() - assert pdfinfo2[0].rotation == 90 - - # Start a new process which gets its own proxy object - def client(context): - assert isinstance(context, BaseProxy) - pdfinfo = context.get_pdfinfo() - page = pdfinfo[0] - assert page.rotation == 90 - page.rotation += 90 - context.set_pdfinfo(pdfinfo) - - p = Process(target=client, args=(context,)) - p.start() - p.join() - assert p.exitcode == 0, "Child process failed" - - assert context.get_pdfinfo()[0].rotation == 180 From 7566d4b76834428ddeaecc6ce0c1601ac4fcccda Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 27 May 2019 16:55:04 -0700 Subject: [PATCH 061/880] Introduce plugins/filters --- src/ocrmypdf/_filters.py | 82 +++++++++++++++++++++++++++++++++++ src/ocrmypdf/_pipeline.py | 5 ++- src/ocrmypdf/api.py | 4 +- src/ocrmypdf/cli.py | 6 ++- src/ocrmypdf/filters.py | 10 +++++ tests/test_filters.py | 91 +++++++++++++++++++++++++++++++++++++++ 6 files changed, 193 insertions(+), 5 deletions(-) create mode 100644 src/ocrmypdf/_filters.py create mode 100644 src/ocrmypdf/filters.py create mode 100644 tests/test_filters.py diff --git a/src/ocrmypdf/_filters.py b/src/ocrmypdf/_filters.py new file mode 100644 index 00000000..44ea1867 --- /dev/null +++ b/src/ocrmypdf/_filters.py @@ -0,0 +1,82 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +import importlib +import os +import sys + + +log = logging.getLogger(__name__) + + +def _load_function_from_module(location): + """Load a function given a module location + + For location=a.b.c, will effectively run "from a.b import c" + + Example: + _load_function_from_module("a.b.c") + + """ + module_parts = location.split('.') + module_name = '.'.join(module_parts[:-1]) + object_name = module_parts[-1] + module = importlib.import_module(module_name) + fn = getattr(module, object_name) + log.debug(f"Loaded function: from {module_name} import {object_name}") + return fn + + +def _load_function_from_pyfile(location): + """Load a function from a file + + Example: + _load_function_from_pyfile("test.py::blur_filter") + """ + filename, object_name = location.split('::', maxsplit=1) + log.debug(f"Loading function {object_name} from {filename}") + + module_name = os.path.basename(filename) + if module_name.endswith('.py'): + module_name = module_name[:-3] + + spec = importlib.util.spec_from_file_location(module_name, filename) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + fn = getattr(module, object_name) + return fn + + +def load_filter(filt): + if callable(filt): + return filt + + if not isinstance(filt, str): + raise TypeError() + + if '::' not in filt: + filt = _load_function_from_module(filt) + else: + filt = _load_function_from_pyfile(filt) + + return filt + + +def check_filter_loadable(filt): + load_filter(filt) + return filt diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 70657a3a..8ab62309 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -28,6 +28,8 @@ import pikepdf from pikepdf.models.metadata import encode_pdf_date from . import PROGRAM_NAME, VERSION, leptonica + +from ._filters import load_filter from .exceptions import ( DpiError, EncryptedPdfError, @@ -522,7 +524,8 @@ def create_ocr_image(image, page_context): im = pix.topil() if options.filter_ocr_image: - im = options.filter_ocr_image(im) + filt = load_filter(options.filter_ocr_image) + im = filt(im) del draw # Pillow requires integer DPI diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 701e3b73..6270a3de 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -118,7 +118,7 @@ def create_options(*, input_file, output_file, **kwargs): for arg, val in kwargs.items(): if val is None: continue - if arg.startswith('filter') and callable(val): + if arg.startswith('filter') and (callable(val) or isinstance(val, str)): filters.append((arg, val)) continue cmd_style_arg = arg.replace('_', '-') @@ -188,7 +188,7 @@ def ocrmypdf( # pylint: disable=unused-argument user_patterns=None, keep_temporary_files=None, progress_bar=None, - process_ocr_image=None, + filter_ocr_image=None, ): """Run OCRmyPDF on one PDF or image. diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 1480d677..d7a9cf69 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -18,6 +18,7 @@ import argparse from . import PROGRAM_NAME, VERSION +from ._filters import check_filter_loadable def numeric(basetype, min_=None, max_=None): @@ -466,9 +467,10 @@ advanced.add_argument( help="Specify the location of the Tesseract user patterns file.", ) - filters = parser.add_argument_group("Filters", argparse.SUPPRESS) -filters.add_argument('--filter-ocr-image', help=argparse.SUPPRESS) +filters.add_argument( + '--filter-ocr-image', help=argparse.SUPPRESS, type=check_filter_loadable +) debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" diff --git a/src/ocrmypdf/filters.py b/src/ocrmypdf/filters.py new file mode 100644 index 00000000..e51381bf --- /dev/null +++ b/src/ocrmypdf/filters.py @@ -0,0 +1,10 @@ +from PIL import Image +import PIL.ImageOps + + +def invert(im): + return PIL.ImageOps.invert(im.convert('L')) + + +def whiteout(im): + return Image.new(im.mode, im.size) diff --git a/tests/test_filters.py b/tests/test_filters.py new file mode 100644 index 00000000..80128dd3 --- /dev/null +++ b/tests/test_filters.py @@ -0,0 +1,91 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os + +from PIL import Image +import pytest + + +from ocrmypdf import ocrmypdf +from ocrmypdf.filters import invert, whiteout +from ocrmypdf._filters import load_filter + + +os_environ = pytest.helpers.os_environ +check_ocrmypdf = pytest.helpers.check_ocrmypdf + + +def filter_42(): + return 42 + + +def test_pyfile(): + obj = load_filter(f'{__file__}::filter_42') + assert obj() == 42 + + +def test_pyfile_notexist(): + with pytest.raises(FileNotFoundError): + load_filter('thisfile.doesnot.exist.py::filter_42') + + +def test_pyfile_noobject(): + with pytest.raises(AttributeError): + load_filter(f'{__file__}::no_function_with_this_name') + + +def test_module(): + obj = load_filter(f'os.getuid') + assert obj() == os.getuid() + + +def test_module_notexist(): + with pytest.raises(ModuleNotFoundError): + load_filter('thismodule.doesnot.exist') + + +def test_filter_from_cmdline(resources, outdir): + (outdir / 'temp.py').write_text( + "from PIL import Image\n" + "def whiteout(im):\n" + " return Image.new(im.mode, im.size)\n" + ) + + check_ocrmypdf( + resources / 'crom.png', + outdir / 'out.pdf', + '--image-dpi', + '100', + '--sidecar', + outdir / 'sidecar.txt', + '--filter-ocr-image', + f"{outdir / 'temp.py'}::whiteout", + ) + + assert (outdir / 'sidecar.txt').read_text().strip() == '' + + +def test_filter_from_api(resources, outdir): + ocrmypdf( + resources / 'crom.png', + outdir / 'out.pdf', + image_dpi=100, + sidecar=outdir / 'sidecar.txt', + filter_ocr_image=whiteout, + ) + assert (outdir / 'sidecar.txt').read_text().strip() == '' From 26a6232e1c011f4793becc150d8f89f257b7a859 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 May 2019 02:33:35 -0700 Subject: [PATCH 062/880] Ignore DSStore --- .gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitignore b/.gitignore index 1504fb47..3343d3b0 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ .venv*/ *.pyc *.sublime-* +*.DS_Store # Package building .eggs/ From 9d5f23e961f4a271125e284f3f511dc5933a10e2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 May 2019 02:39:25 -0700 Subject: [PATCH 063/880] Rename filters to plugins --- src/ocrmypdf/_pipeline.py | 4 ++-- src/ocrmypdf/{_filters.py => _plugins.py} | 22 +++++++++++----------- src/ocrmypdf/cli.py | 4 ++-- tests/test_filters.py | 12 ++++++------ 4 files changed, 21 insertions(+), 21 deletions(-) rename src/ocrmypdf/{_filters.py => _plugins.py} (85%) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 8ab62309..ce2f44c7 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -29,7 +29,7 @@ from pikepdf.models.metadata import encode_pdf_date from . import PROGRAM_NAME, VERSION, leptonica -from ._filters import load_filter +from ._plugins import load_plugin from .exceptions import ( DpiError, EncryptedPdfError, @@ -524,7 +524,7 @@ def create_ocr_image(image, page_context): im = pix.topil() if options.filter_ocr_image: - filt = load_filter(options.filter_ocr_image) + filt = load_plugin(options.filter_ocr_image) im = filt(im) del draw diff --git a/src/ocrmypdf/_filters.py b/src/ocrmypdf/_plugins.py similarity index 85% rename from src/ocrmypdf/_filters.py rename to src/ocrmypdf/_plugins.py index 44ea1867..89655b9f 100644 --- a/src/ocrmypdf/_filters.py +++ b/src/ocrmypdf/_plugins.py @@ -62,21 +62,21 @@ def _load_function_from_pyfile(location): return fn -def load_filter(filt): - if callable(filt): - return filt +def load_plugin(plugin): + if callable(plugin): + return plugin - if not isinstance(filt, str): + if not isinstance(plugin, str): raise TypeError() - if '::' not in filt: - filt = _load_function_from_module(filt) + if '::' not in plugin: + plugin = _load_function_from_module(plugin) else: - filt = _load_function_from_pyfile(filt) + plugin = _load_function_from_pyfile(plugin) - return filt + return plugin -def check_filter_loadable(filt): - load_filter(filt) - return filt +def check_plugin_loadable(plugin): + load_plugin(plugin) + return plugin diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index d7a9cf69..49a2c87d 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -18,7 +18,7 @@ import argparse from . import PROGRAM_NAME, VERSION -from ._filters import check_filter_loadable +from ._plugins import check_plugin_loadable def numeric(basetype, min_=None, max_=None): @@ -469,7 +469,7 @@ advanced.add_argument( filters = parser.add_argument_group("Filters", argparse.SUPPRESS) filters.add_argument( - '--filter-ocr-image', help=argparse.SUPPRESS, type=check_filter_loadable + '--filter-ocr-image', help=argparse.SUPPRESS, type=check_plugin_loadable ) debugging = parser.add_argument_group( diff --git a/tests/test_filters.py b/tests/test_filters.py index 80128dd3..12628e80 100644 --- a/tests/test_filters.py +++ b/tests/test_filters.py @@ -23,7 +23,7 @@ import pytest from ocrmypdf import ocrmypdf from ocrmypdf.filters import invert, whiteout -from ocrmypdf._filters import load_filter +from ocrmypdf._plugins import load_plugin os_environ = pytest.helpers.os_environ @@ -35,28 +35,28 @@ def filter_42(): def test_pyfile(): - obj = load_filter(f'{__file__}::filter_42') + obj = load_plugin(f'{__file__}::filter_42') assert obj() == 42 def test_pyfile_notexist(): with pytest.raises(FileNotFoundError): - load_filter('thisfile.doesnot.exist.py::filter_42') + load_plugin('thisfile.doesnot.exist.py::filter_42') def test_pyfile_noobject(): with pytest.raises(AttributeError): - load_filter(f'{__file__}::no_function_with_this_name') + load_plugin(f'{__file__}::no_function_with_this_name') def test_module(): - obj = load_filter(f'os.getuid') + obj = load_plugin(f'os.getuid') assert obj() == os.getuid() def test_module_notexist(): with pytest.raises(ModuleNotFoundError): - load_filter('thismodule.doesnot.exist') + load_plugin('thismodule.doesnot.exist') def test_filter_from_cmdline(resources, outdir): From 396c39978a2f5e2fe9f58c8c70ce4c239a6a9c95 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 May 2019 14:17:17 -0700 Subject: [PATCH 064/880] Reorganize .docker folder so we don't have to rebuild as much --- .docker/alpine.dockerfile | 4 ++-- .docker/docker-wrapper.sh | 5 ----- .docker/polyglot.dockerfile | 17 ----------------- .docker/webservice.dockerfile | 24 ------------------------ .dockerignore | 2 +- {.docker => misc}/webservice.py | 0 6 files changed, 3 insertions(+), 49 deletions(-) delete mode 100755 .docker/docker-wrapper.sh delete mode 100644 .docker/polyglot.dockerfile delete mode 100644 .docker/webservice.dockerfile rename {.docker => misc}/webservice.py (100%) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index 7145199f..e6e743b5 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -75,12 +75,12 @@ RUN \ WORKDIR /app -# Copy build artifacts (python site-packages9 +# Copy build artifacts (python site-packages) COPY --from=builder /usr/lib/python3.6/site-packages /usr/lib/python3.6/site-packages COPY --from=builder /usr/bin/ocrmypdf /usr/bin/dumppdf.py /usr/bin/latin2ascii.py /usr/bin/pdf2txt.py /usr/bin/img2pdf /usr/bin/chardetect /usr/bin/ # Copy -COPY --from=builder /app/.docker/webservice.py /app/ +COPY --from=builder /app/misc/webservice.py /app/ # Copy minimal project files to get the test suite. COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ diff --git a/.docker/docker-wrapper.sh b/.docker/docker-wrapper.sh deleted file mode 100755 index ecc1af3e..00000000 --- a/.docker/docker-wrapper.sh +++ /dev/null @@ -1,5 +0,0 @@ -#!/bin/bash - -. /appenv/bin/activate -cd /home/docker -exec ocrmypdf "$@" \ No newline at end of file diff --git a/.docker/polyglot.dockerfile b/.docker/polyglot.dockerfile deleted file mode 100644 index c837a197..00000000 --- a/.docker/polyglot.dockerfile +++ /dev/null @@ -1,17 +0,0 @@ -# OCRmyPDF polyglot -# -FROM jbarlow83/ocrmypdf:latest - -USER root - -# Update system and install our dependencies -RUN apt-get update && apt-get install -y --no-install-recommends \ - tesseract-ocr-all - -RUN apt-get autoremove -y && apt-get clean -y - -USER docker - -# Must use array form of ENTRYPOINT -# Non-array form does not append other arguments, because that is "intuitive" -ENTRYPOINT ["/application/.docker/docker-wrapper.sh"] \ No newline at end of file diff --git a/.docker/webservice.dockerfile b/.docker/webservice.dockerfile deleted file mode 100644 index cd0f71be..00000000 --- a/.docker/webservice.dockerfile +++ /dev/null @@ -1,24 +0,0 @@ -# OCRmyPDF webservice -# -FROM jbarlow83/ocrmypdf-polyglot:latest - -USER root - -# Update system and install our dependencies -RUN apt-get update && apt-get install -y --no-install-recommends \ - python3-flask - -RUN apt-get autoremove -y && apt-get clean -y - -EXPOSE 5000 - -COPY .docker/webservice.py /application - -USER docker - -VOLUME ["/config"] - -# This config file is optional -ENV OCRMYPDF_WEBSERVICE_SETTINGS "/config/config.py" - -ENTRYPOINT ["python3", "/application/webservice.py"] diff --git a/.dockerignore b/.dockerignore index ee63ddca..66a0920b 100644 --- a/.dockerignore +++ b/.dockerignore @@ -6,7 +6,6 @@ **/*.pyc .*/ !.git/ -!.docker/ .ruffus_history.sqlite bin/ build/ @@ -17,6 +16,7 @@ include/ lib/ MANIFEST.in ocrmypdf.egg-info/ +private/ staging/ tests/cache/ tests/output/ diff --git a/.docker/webservice.py b/misc/webservice.py similarity index 100% rename from .docker/webservice.py rename to misc/webservice.py From d5b6cbb95ebe44cb41904d7e031b83cbe234dde2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 May 2019 15:36:50 -0700 Subject: [PATCH 065/880] Update Ubuntu dockerfile --- .docker/Dockerfile | 103 ++++++++++++++++++++++++--------------------- 1 file changed, 56 insertions(+), 47 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index ebb63ddd..62fac8d3 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -1,17 +1,60 @@ # OCRmyPDF # -FROM ubuntu:18.04 +FROM ubuntu:19.04 as base + +FROM base as builder + +ENV LANG=C.UTF-8 RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential autoconf automake libtool \ libleptonica-dev \ zlib1g-dev \ - libexempi3 \ ocrmypdf \ pngquant \ python3-pip \ python3-venv \ tesseract-ocr \ + unpaper \ + wget \ + git + + +# Compile and install jbig2 +# Needs libleptonica-dev, zlib1g-dev +RUN \ + mkdir jbig2 \ + && wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \ + tar xz -C jbig2 --strip-components=1 \ + && cd jbig2 \ + && ./autogen.sh && ./configure && make && make install \ + && cd .. \ + && rm -rf jbig2 + +RUN python3 -m venv /appenv + +COPY . /app + +WORKDIR /app + +RUN . /appenv/bin/activate; \ + pip install --upgrade pip \ + && pip install . + +FROM base + +ENV LANG=C.UTF-8 + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ghostscript \ + img2pdf \ + liblept5 \ + zlib1g \ + pngquant \ + python3 \ + python3-venv \ + qpdf \ + tesseract-ocr \ tesseract-ocr-chi-sim \ tesseract-ocr-deu \ tesseract-ocr-eng \ @@ -21,52 +64,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ unpaper \ wget +# Copy +COPY --from=builder /app/misc/webservice.py /app/ -ENV LANG=C.UTF-8 +# Copy minimal project files to get the test suite. +COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ +COPY --from=builder /app/requirements /app/requirements +COPY --from=builder /app/tests /app/tests +COPY --from=builder /app/src /app/src -# Compile and install jbig2 -# Needs libleptonica-dev, zlib1g-dev -RUN \ - mkdir jbig2 \ - && wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \ - tar xz -C jbig2 --strip-components=1 \ - && cd jbig2 \ - && ./autogen.sh && ./configure && make && make install \ - && cd .. \ - && rm -rf jbig2 +COPY --from=builder /appenv /appenv +COPY --from=builder /usr/local /usr/local -RUN apt-get remove -y autoconf automake libtool +WORKDIR /data -RUN python3 -m venv --system-site-packages /appenv - -# This installs the latest binary wheel instead of the code in the current -# folder. Installing from source will fail, apparently because cffi needs -# build-essentials (gcc) to do a source installation -# (i.e. "pip install ."). It's unclear to me why this is the case. -RUN . /appenv/bin/activate; \ - pip install --upgrade pip \ - && pip install --upgrade ocrmypdf - -# Now copy the application in, mainly to get the test suite. -# Do this now to make the best use of Docker cache. -COPY . /application -RUN . /appenv/bin/activate; \ - pip install -r /application/requirements/test.txt - -# Remove the junk, including the source version of application since it was -# already installed -RUN rm -rf /tmp/* /var/tmp/* /root/* /application/ocrmypdf \ - && apt-get remove -y build-essential \ - && apt-get autoremove -y \ - && apt-get autoclean -y - -RUN useradd docker \ - && mkdir /home/docker \ - && chown docker:docker /home/docker - -USER docker -WORKDIR /home/docker - -# Must use array form of ENTRYPOINT -# Non-array form does not append other arguments, because that is "intuitive" -ENTRYPOINT ["/application/.docker/docker-wrapper.sh"] +ENTRYPOINT ["/appenv/bin/ocrmypdf"] From db29cae177ac03864d7ef9aa19111b0f4a0d9f11 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 May 2019 15:43:15 -0700 Subject: [PATCH 066/880] Docker docs: Remove legacy images, revive Ubuntu --- .dockerignore | 1 - docs/docker.rst | 51 ++++++------------------------------------------- 2 files changed, 6 insertions(+), 46 deletions(-) diff --git a/.dockerignore b/.dockerignore index 66a0920b..2ebfdc48 100644 --- a/.dockerignore +++ b/.dockerignore @@ -16,7 +16,6 @@ include/ lib/ MANIFEST.in ocrmypdf.egg-info/ -private/ staging/ tests/cache/ tests/output/ diff --git a/docs/docker.rst b/docs/docker.rst index 2bbcb2c4..df45ef23 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -5,7 +5,7 @@ OCRmyPDF is also available in a Docker image that packages recent versions of al For users who already have Docker installed this may be an easy and convenient option. However, it is less performant than a system installation and may require Docker engine configuration. -OCRmyPDF needs a generous amount of RAM, CPU cores, and temporary storage space. +OCRmyPDF needs a generous amount of RAM, CPU cores, temporary storage space, whether running in a Docker container or on its own. It may be necessary to ensure the container is provisioned with additional resources. .. _docker-install: @@ -109,7 +109,7 @@ Unlike command line usage this program will open a socket and wait for connectio .. warning:: - The OCRmyPDF web service wrapper is intended for demonstration or development. It provides no security, no authentication, no protection against denial of service attacks, and no load balancing. The default Flask WSGI server is used, which is intended for development only. The server is single-threaded and so can respond to only one client at a time. It cannot respond to clients while busy with OCR. + The OCRmyPDF web service wrapper is intended for demonstration or development. It provides no security, no authentication, no protection against denial of service attacks, and no load balancing. The default Flask WSGI server is used, which is intended for development only. The server is single-threaded and so can respond to only one client at a time. While running OCR, it cannot respond to any other clients. Clients must keep their open connection while waiting for OCR to complete. This may entail setting a long timeout; this interface is more useful for internal HTTP API calls. @@ -117,50 +117,11 @@ Unlike the rest of OCRmyPDF, this web service is licensed under the Affero GPLv3 In addition to the above, please read our :ref:`general remarks on using OCRmyPDF as a service `. -Legacy Ubuntu Docker images ---------------------------- +Ubuntu-based Docker image +------------------------- -Previously OCRmyPDF was delivered in several Docker images for different purposes, based on Ubuntu. - -The Ubuntu-based images will be maintained for some time but should not be used for new deployments. They are as follows: - -.. list-table:: - :widths: auto - :header-rows: 1 - - * - Image name - - Download command - - Notes - * - ocrmypdf - - ``docker pull jbarlow83/ocrmypdf`` - - Latest ocrmypdf with Tesseract 4.0.0-beta1 on Ubuntu 18.04. Includes English, French, German, Spanish, Portugeuse and Simplified Chinese. - * - ocrmypdf-polyglot - - ``docker pull jbarlow83/ocrmypdf-polyglot`` - - As above, with all available language packs. - * - ocrmypdf-webservice - - ``docker pull jbarlow83/ocrmypdf-webservice`` - - All language packs, and a simple HTTP wrapper allowing OCRmyPDF to be used as a web service. Note that this component is licensed under AGPLv3. - -To execute the Ubuntu-based OCRmyPDF on a local file, you must `provide a writable volume to the Docker image `_, and both the input and output file must be inside the writable volume. This limitation applies only to the legacy images. - -This example command uses the current working directory as the writable volume: +A Ubuntu-based OCRmyPDF image is also available. The main advantage this image offers is that it supports manylinux Python wheels (which are not supported on Alpine Linux). This may be useful for plugins. .. code-block:: bash - docker run --rm -v "$(pwd):/home/docker" ocrmypdf - -In this worked example, the current working directory contains an input file called ``test.pdf`` and the output will go to ``output.pdf``: - -.. code-block:: bash - - docker run --rm -v "$(pwd):/home/docker" ocrmypdf --skip-text test.pdf output.pdf - -.. note:: The working directory should be a writable local volume or Docker may not have permission to access it. - -Note that ``ocrmypdf`` has its own separate ``-v VERBOSITYLEVEL`` argument to control debug verbosity. All Docker arguments should before the ``ocrmypdf`` image name and all arguments to ``ocrmypdf`` should be listed after. - -In some environments the permissions associated with Docker can be complex to configure. The process that executes Docker may end up not having the permissions to write the specified file system. In that case one can stream the file into and out of the Docker process and avoid all permission hassles, using ``-`` as the input and output filename: - -.. code-block:: bash - - docker run --rm -i ocrmypdf - - output.pdf + docker pull jbarlow83/ocrmypdf From 8ed4e229f39fb203e045358f3df9a93b8baf79cb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 May 2019 13:57:38 -0700 Subject: [PATCH 067/880] ghostscript: avoid log=None construct --- src/ocrmypdf/exec/ghostscript.py | 11 ++++++++++- src/ocrmypdf/pdfinfo/__init__.py | 9 ++++----- src/ocrmypdf/pdfinfo/ghosttext.py | 5 ++++- tests/test_rotation.py | 2 +- 4 files changed, 19 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 85e82465..b744cdbf 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import re from functools import lru_cache from os import fspath @@ -24,8 +25,11 @@ from tempfile import NamedTemporaryFile from PIL import Image -from . import get_version from ..exceptions import SubprocessOutputError +from . import get_version + + +gslog = logging.getLogger() @lru_cache(maxsize=1) @@ -132,6 +136,8 @@ def rasterize_pdf( res = round(xres, 6), round(yres, 6) if not page_dpi: page_dpi = res + if not log: + log = gslog with NamedTemporaryFile(delete=True) as tmp: args_gs = ( @@ -209,6 +215,9 @@ def generate_pdfa( images entirely. (The feature was added in 9.23 but broken, and the 9.24 release of Ghostscript had regressions, so we don't support it until 9.25.) """ + if not log: + log = gslog + compression_args = [] if compression == 'jpeg': compression_args = [ diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index aaad8ebe..c6c4f7e9 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -19,10 +19,10 @@ from collections import namedtuple from decimal import Decimal from enum import Enum +import logging from math import hypot, isclose from os import fspath from pathlib import Path -from unittest.mock import Mock from warnings import warn import re @@ -34,6 +34,8 @@ from . import ghosttext from ..exceptions import EncryptedPdfError, MissingDependencyError +logger = logging.getLogger() + Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000') Encoding = Enum( @@ -615,9 +617,6 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None): - if not log: - log = Mock() - pdf = pikepdf.open(infile) # Do not close in this function if pdf.is_encrypted: pdf.close() @@ -750,7 +749,7 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, detailed_page_analysis=False, log=None): + def __init__(self, infile, detailed_page_analysis=False, log=logger): self._infile = infile self._pages, pdf = _pdf_get_all_pageinfo( infile, detailed_page_analysis, log=log diff --git a/src/ocrmypdf/pdfinfo/ghosttext.py b/src/ocrmypdf/pdfinfo/ghosttext.py index c1a612a5..43156154 100644 --- a/src/ocrmypdf/pdfinfo/ghosttext.py +++ b/src/ocrmypdf/pdfinfo/ghosttext.py @@ -15,11 +15,14 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import re import xml.etree.ElementTree as ET from ..exec import ghostscript +gslog = logging.getLogger() + # Forgive me for I have sinned # I am using regular expressions to parse XML. However the XML in this case, # generated by Ghostscript, is self-consistent enough to be parseable. @@ -74,7 +77,7 @@ def page_get_textblocks(infile, pageno, xmltext, height): return [block for block in joined_blocks()] -def extract_text_xml(infile, pdf, pageno=None, log=None): +def extract_text_xml(infile, pdf, pageno=None, log=gslog): existing_text = ghostscript.extract_text(infile, pageno=None) existing_text = regex_remove_char_tags.sub(b' ', existing_text) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 3f09d43d..a33e66cd 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -268,7 +268,7 @@ def test_tesseract_orientation(resources, tmpdir): pix_rotated = pix.rotate_orth(2) # 180 degrees clockwise pix_rotated.write_implied_format(tmpdir / '000001.png') - log = Mock() + log = logging.getLogger() tesseract.get_orientation( # Test results of this are unreliable tmpdir / '000001.png', engine_mode='3', timeout=10, log=log ) From 522e1e948bf18835f0919e735d82961dd105a39b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 31 May 2019 01:55:29 -0700 Subject: [PATCH 068/880] ghostscript: don't use threads= for generate_pdfa Not supported for pdfwrite --- src/ocrmypdf/_pipeline.py | 1 - 1 file changed, 1 deletion(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index ce2f44c7..592f2411 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -703,7 +703,6 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): output_file=output_file, compression=options.pdfa_image_compression, log=context.log, - threads=options.jobs or 1, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 ) From 45a361d112185ffc5034eac3170ba7c4099bbe71 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 31 May 2019 01:56:16 -0700 Subject: [PATCH 069/880] Add option to use threads instead of processes Mainly since they are more convenient for debugging --- src/ocrmypdf/_sync.py | 16 ++++++++++++++-- src/ocrmypdf/api.py | 1 + src/ocrmypdf/cli.py | 1 + 3 files changed, 16 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 3141fed3..de408919 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -164,6 +164,10 @@ def worker_init(queue): root.addHandler(h) +def worker_thread_init(queue): + pass + + def log_listener(queue): """Listen to the worker processes and forward the messages to logging @@ -196,6 +200,14 @@ def exec_concurrent(context): if max_workers > 1: context.log.info("Start processing %d pages concurrent" % max_workers) + if context.options.use_threads: + from multiprocessing.dummy import Pool + + initializer = worker_thread_init + else: + Pool = multiprocessing.Pool + initializer = worker_init + sidecars = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) @@ -208,8 +220,8 @@ def exec_concurrent(context): unit='page', unit_scale=0.5, disable=not context.options.progress_bar, - ) as pbar, multiprocessing.Pool( - processes=max_workers, initializer=worker_init, initargs=(log_queue,) + ) as pbar, Pool( + processes=max_workers, initializer=initializer, initargs=(log_queue,) ) as pool: results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) while True: diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 6270a3de..ab648170 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -153,6 +153,7 @@ def ocrmypdf( # pylint: disable=unused-argument output_type=None, sidecar=None, jobs=None, + use_threads=None, title=None, author=None, subject=None, diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 49a2c87d..75833362 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -199,6 +199,7 @@ jobcontrol.add_argument( dest='progress_bar', help=argparse.SUPPRESS, ) +jobcontrol.add_argument('--use-threads', action='store_true', help=argparse.SUPPRESS) metadata = parser.add_argument_group( "Metadata options", From 8347c0d6625b043209bee4041a0319a21ff48d7d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 31 May 2019 01:57:08 -0700 Subject: [PATCH 070/880] validation: remove dead code check_input_file --- src/ocrmypdf/_validation.py | 18 +----------------- 1 file changed, 1 insertion(+), 17 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 3e0e3526..2770874a 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -21,6 +21,7 @@ import logging import os import sys from pathlib import Path +from shutil import copyfileobj import PIL @@ -334,8 +335,6 @@ def create_input_file(options, work_folder): log.info('reading file from standard input') target = os.path.join(work_folder, 'stdin') with open(target, 'wb') as stream_buffer: - from shutil import copyfileobj - copyfileobj(sys.stdin.buffer, stream_buffer) return target else: @@ -347,21 +346,6 @@ def create_input_file(options, work_folder): raise InputFileError(f"File not found - {options.input_file}") -def check_input_file(options, start_input_file): - if options.input_file == '-': - # stdin - log.info('reading file from standard input') - with open(start_input_file, 'wb') as stream_buffer: - from shutil import copyfileobj - - copyfileobj(sys.stdin.buffer, stream_buffer) - else: - try: - re_symlink(options.input_file, start_input_file) - except FileNotFoundError: - raise InputFileError(f"File not found - {options.input_file}") - - def check_requested_output_file(options): if options.output_file == '-': if sys.stdout.isatty(): From b9d6e46572fc3979afd7debd105688ff5f91086b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 31 May 2019 15:12:46 -0700 Subject: [PATCH 071/880] shutil.rmtree: use builtin error suppression --- src/ocrmypdf/_jobcontext.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 5daab103..2a141e7f 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -19,7 +19,6 @@ import logging import shutil import sys import os -from contextlib import suppress class PicklableLoggerMixin: @@ -96,8 +95,7 @@ def cleanup_working_files(work_folder, options): if options.keep_temporary_files: print(f"Temporary working files saved at:\n{work_folder}", file=sys.stderr) else: - with suppress(FileNotFoundError): - shutil.rmtree(work_folder) + shutil.rmtree(work_folder, ignore_errors=True) class LogNameAdapter(logging.LoggerAdapter): From df9e286e9cddbbbf2ec2a86fc0863c4458562058 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 1 Jun 2019 01:35:15 -0700 Subject: [PATCH 072/880] Make bypassed exception clearer --- src/ocrmypdf/_sync.py | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index de408919..62fc5a39 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -259,6 +259,12 @@ def exec_concurrent(context): copy_final(pdf, context.options.output_file, context) +class NeverRaise(Exception): + """An exception that is never raised""" + + pass + + def run_pipeline(options, api=False): log = make_logger(options, __name__) @@ -293,22 +299,16 @@ def run_pipeline(options, api=False): # Execute the pipeline exec_concurrent(context) - except KeyboardInterrupt as e: - if api: - raise + except (KeyboardInterrupt if not api else NeverRaise) as e: log.error("KeyboardInterrupt") return ExitCode.ctrl_c - except ExitCodeException as e: - if api: - raise + except (ExitCodeException if not api else NeverRaise) as e: if str(e): log.error("%s: %s", type(e).__name__, str(e)) else: log.error(type(e).__name__) return e.exit_code - except Exception as e: - if api: - raise + except (Exception if not api else NeverRaise) as e: log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error finally: From ba41ccae1bbfb57d8ef0c9b08574577add440820 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 1 Jun 2019 01:41:39 -0700 Subject: [PATCH 073/880] conftest: don't modify PYTEST_CURRENT_TEST when manipulating os.environ It confuses pytest. --- tests/conftest.py | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 8ce8f5fb..e0454e5c 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -120,16 +120,24 @@ def spoof(tmpdir_factory, **kwargs): @contextmanager def os_environ(new_env): old_env = os.environ.copy() + if new_env is None: + new_env = {} for k, v in new_env.items(): - os.environ[k] = v + if k != 'PYTEST_CURRENT_TEST': + os.environ[k] = v yield new_keys = set(os.environ.copy()) - set(old_env) for k in new_keys: - del os.environ[k] + if k != 'PYTEST_CURRENT_TEST': + del os.environ[k] for k in old_env: - os.environ[k] = old_env[k] - assert os.environ.copy() == old_env + if k != 'PYTEST_CURRENT_TEST': + os.environ[k] = old_env[k] + + for k, v in os.environ.copy().items(): + if k != 'PYTEST_CURRENT_TEST': + assert v == old_env[k] @pytest.fixture(scope='session') From fb933edc0f6f2c829eb620579583f97dd9bce72a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 1 Jun 2019 01:55:51 -0700 Subject: [PATCH 074/880] Use newer pytest tmp_path API --- tests/conftest.py | 26 +++++++++++----------- tests/test_hocrtransform.py | 4 ++-- tests/test_lept.py | 4 ++-- tests/test_main.py | 43 ++++++++++++++++++++----------------- tests/test_metadata.py | 4 ++-- tests/test_rotation.py | 6 +++--- tests/test_tess4.py | 6 +++--- tests/test_unpaper.py | 4 ++-- 8 files changed, 50 insertions(+), 47 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index e0454e5c..6ff87b4b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -91,7 +91,7 @@ OCRMYPDF = [sys.executable, '-m', 'ocrmypdf'] @pytest.helpers.register -def spoof(tmpdir_factory, **kwargs): +def spoof(tmp_path_factory, **kwargs): """Modify PATH to override subprocess executables spoof(program1='replacement', ...) @@ -101,8 +101,8 @@ def spoof(tmpdir_factory, **kwargs): """ env = os.environ.copy() slug = '-'.join(v.replace('.py', '') for v in sorted(kwargs.values())) - spoofer_base = Path(str(tmpdir_factory.mktemp('spoofers'))) - tmpdir = spoofer_base / slug + spoofer_base = tmp_path_factory.mktemp('spoofers') + tmpdir = Path(spoofer_base / slug) tmpdir.mkdir(parents=True) for replace_program, with_spoof in kwargs.items(): @@ -141,15 +141,15 @@ def os_environ(new_env): @pytest.fixture(scope='session') -def spoof_tesseract_noop(tmpdir_factory): - return spoof(tmpdir_factory, tesseract='tesseract_noop.py') +def spoof_tesseract_noop(tmp_path_factory): + return spoof(tmp_path_factory, tesseract='tesseract_noop.py') @pytest.fixture(scope='session') -def spoof_tesseract_cache(tmpdir_factory): +def spoof_tesseract_cache(tmp_path_factory): if running_in_docker(): return os.environ.copy() - return spoof(tmpdir_factory, tesseract="tesseract_cache.py") + return spoof(tmp_path_factory, tesseract="tesseract_cache.py") @pytest.fixture @@ -163,22 +163,22 @@ def ocrmypdf_exec(): @pytest.fixture(scope="function") -def outdir(tmpdir): - return Path(str(tmpdir)) +def outdir(tmp_path): + return tmp_path @pytest.fixture(scope="function") -def outpdf(tmpdir): - return str(Path(str(tmpdir)) / 'out.pdf') +def outpdf(tmp_path): + return tmp_path / 'out.pdf' @pytest.fixture(scope="function") -def no_outpdf(tmpdir): +def no_outpdf(tmp_path): """This just documents the fact that a test is not expected to produce output. Unfortunately an assertion failure inside a test fixture produces an error rather than a test failure, so no testing is done. It's up to the test to confirm that no output file was created.""" - return str(Path(str(tmpdir)) / 'no_output.pdf') + return tmp_path / 'no_output.pdf' @pytest.helpers.register diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index e36fe7ee..19e4684d 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -28,8 +28,8 @@ from ocrmypdf.exec.tesseract import HOCR_TEMPLATE @pytest.fixture -def blank_hocr(tmpdir): - filename = Path(str(tmpdir)) / "blank.hocr" +def blank_hocr(tmp_path): + filename = tmp_path / "blank.hocr" filename.write_text(HOCR_TEMPLATE) # pylint: disable=E1101 return filename diff --git a/tests/test_lept.py b/tests/test_lept.py index 5fe68109..804ca215 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -80,13 +80,13 @@ def test_pickle(crom_pix): assert pix.mode == pix2.mode -def test_leptonica_compile(tmpdir): +def test_leptonica_compile(tmp_path): from ocrmypdf.lib.compile_leptonica import ffibuilder # Compile the library but build it somewhere that won't interfere with # existing compiled library. Also compile in API mode so that we test # the interfaces, even though we use it ABI mode. - ffibuilder.compile(tmpdir=fspath(tmpdir), target=fspath(tmpdir / 'lepttest.*')) + ffibuilder.compile(tmpdir=fspath(tmp_path), target=fspath(tmp_path / 'lepttest.*')) def test_with_stderr(capsys): diff --git a/tests/test_main.py b/tests/test_main.py index 36348479..10b6eb5b 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -47,44 +47,46 @@ RENDERERS = ['hocr', 'sandwich'] @pytest.fixture(scope='session') -def spoof_tesseract_crash(tmpdir_factory): - return spoof(tmpdir_factory, tesseract='tesseract_crash.py') +def spoof_tesseract_crash(tmp_path_factory): + return spoof(tmp_path_factory, tesseract='tesseract_crash.py') @pytest.fixture(scope='session') -def spoof_tesseract_big_image_error(tmpdir_factory): - return spoof(tmpdir_factory, tesseract='tesseract_big_image_error.py') +def spoof_tesseract_big_image_error(tmp_path_factory): + return spoof(tmp_path_factory, tesseract='tesseract_big_image_error.py') @pytest.fixture(scope='session') -def spoof_no_tess_no_pdfa(tmpdir_factory): - return spoof(tmpdir_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py') - - -@pytest.fixture(scope='session') -def spoof_no_tess_pdfa_warning(tmpdir_factory): +def spoof_no_tess_no_pdfa(tmp_path_factory): return spoof( - tmpdir_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py' ) @pytest.fixture(scope='session') -def spoof_no_tess_gs_render_fail(tmpdir_factory): +def spoof_no_tess_pdfa_warning(tmp_path_factory): return spoof( - tmpdir_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' ) @pytest.fixture(scope='session') -def spoof_no_tess_gs_raster_fail(tmpdir_factory): +def spoof_no_tess_gs_render_fail(tmp_path_factory): return spoof( - tmpdir_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' ) @pytest.fixture(scope='session') -def spoof_tess_bad_utf8(tmpdir_factory): - return spoof(tmpdir_factory, tesseract='tesseract_badutf8.py') +def spoof_no_tess_gs_raster_fail(tmp_path_factory): + return spoof( + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' + ) + + +@pytest.fixture(scope='session') +def spoof_tess_bad_utf8(tmp_path_factory): + return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') def test_quick(spoof_tesseract_cache, resources, outpdf): @@ -225,7 +227,8 @@ def test_skip_ocr(spoof_tesseract_cache, resources, outpdf): def test_redo_ocr(spoof_tesseract_cache, resources, outpdf): in_ = resources / 'graph_ocred.pdf' before = PdfInfo(in_, detailed_page_analysis=True) - out = check_ocrmypdf(in_, outpdf, '--redo-ocr', env=spoof_tesseract_cache) + out = outpdf + out = check_ocrmypdf(in_, out, '--redo-ocr') after = PdfInfo(out, detailed_page_analysis=True) assert before[0].has_text and after[0].has_text assert ( @@ -936,7 +939,7 @@ def test_compression_changed( def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): - sidecar = outpdf + '.txt' + sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( resources / 'multipage.pdf', outpdf, @@ -960,7 +963,7 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): def test_sidecar_nonempty(spoof_tesseract_cache, resources, outpdf): - sidecar = outpdf + '.txt' + sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( resources / 'ccitt.pdf', outpdf, '--sidecar', sidecar, env=spoof_tesseract_cache ) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index fc392c77..73318ee4 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -261,10 +261,10 @@ def test_xml_metadata_preserved(spoof_tesseract_noop, output_type, resources, ou ) -def test_srgb_in_unicode_path(tmpdir): +def test_srgb_in_unicode_path(tmp_path): """Test that we can produce pdfmark when install path is not ASCII""" - dstdir = Path(fspath(tmpdir)) / b'\xe4\x80\x80'.decode('utf-8') + dstdir = tmp_path / b'\xe4\x80\x80'.decode('utf-8') dstdir.mkdir() dst = dstdir / 'sRGB.icc' diff --git a/tests/test_rotation.py b/tests/test_rotation.py index a33e66cd..b8bc07ca 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -263,12 +263,12 @@ def test_rotate_page_level(image_angle, page_angle, resources, outdir): assert check_monochrome_correlation(outdir, reference, 1, out, 1) > 0.2 -def test_tesseract_orientation(resources, tmpdir): +def test_tesseract_orientation(resources, tmp_path): pix = leptonica.Pix.open(resources / 'crom.png') pix_rotated = pix.rotate_orth(2) # 180 degrees clockwise - pix_rotated.write_implied_format(tmpdir / '000001.png') + pix_rotated.write_implied_format(tmp_path / '000001.png') log = logging.getLogger() tesseract.get_orientation( # Test results of this are unreliable - tmpdir / '000001.png', engine_mode='3', timeout=10, log=log + tmp_path / '000001.png', engine_mode='3', timeout=10, log=log ) diff --git a/tests/test_tess4.py b/tests/test_tess4.py index d4330b83..92ee8c94 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -174,10 +174,10 @@ def test_content_preservation(ensure_tess4, resources, outpdf): assert len(page.images) > 1, "masks were rasterized" -def test_no_languages(ensure_tess4, tmpdir): +def test_no_languages(ensure_tess4, tmp_path): env = ensure_tess4 - (tmpdir / 'tessdata').mkdir() - env['TESSDATA_PREFIX'] = fspath(tmpdir) + (tmp_path / 'tessdata').mkdir() + env['TESSDATA_PREFIX'] = fspath(tmp_path) with modified_os_environ(env): with pytest.raises(MissingDependencyError): diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 887184cb..4242743c 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -44,8 +44,8 @@ def have_unpaper(): @pytest.fixture(scope="session") -def spoof_unpaper_oldversion(tmpdir_factory): - return spoof(tmpdir_factory, unpaper="unpaper_oldversion.py") +def spoof_unpaper_oldversion(tmp_path_factory): + return spoof(tmp_path_factory, unpaper="unpaper_oldversion.py") def test_no_unpaper(resources, no_outpdf): From e73740ae9d6ea9a45a6a60afa5dea1daae51d064 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Jun 2019 01:33:24 -0700 Subject: [PATCH 075/880] test: remove test code that support tess3 or tess4 testing --- tests/test_tess4.py | 98 +++++---------------------------------------- 1 file changed, 10 insertions(+), 88 deletions(-) diff --git a/tests/test_tess4.py b/tests/test_tess4.py index 92ee8c94..bb9cc49a 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -27,85 +27,17 @@ from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.exec import tesseract # pylint: disable=no-member,w0621 -spoof = pytest.helpers.spoof - - -def _ensure_tess4(): - if tesseract.v4(): - # "tesseract" on $PATH is already v4 - return os.environ.copy() - - if os.environ.get('OCRMYPDF_TESS4'): - # OCRMYPDF_TESS4 is a hint environment variable that tells us to look - # somewhere special for tess4 if and only if we need it. This allows - # setting OCRMYPDF_TESS4 to test tess4 and PATH to point to tess3 - # on a system with both installed. - env = os.environ.copy() - tess4 = Path(os.environ['OCRMYPDF_TESS4']) - assert tess4.is_file() - env['PATH'] = tess4.parent + ':' + env['PATH'] - env['OCRMYPDF_TESS4'] = os.environ['OCRMYPDF_TESS4'] - return env - - raise EnvironmentError("Can't find Tesseract 4") - - -@pytest.fixture -def ensure_tess4(): - return _ensure_tess4() - - -@contextmanager -def modified_os_environ(env): - old_env = os.environ.copy() - os.environ.update(env) - yield - for key in env: - del os.environ[key] - if key in old_env: - os.environ[key] = old_env[key] - - -def tess4_available(): - """Check if a tesseract 4 binary is available, even if it's not the - official "tesseract" on PATH - - """ - try: - # _ensure_tess4 locates the tess4 binary we are going to check - env = _ensure_tess4() - with modified_os_environ(env): - # Now jump into this environment and make sure it really is Tess4 - return tesseract.v4() and tesseract.has_textonly_pdf() - except EnvironmentError: - pass - - return False - - -# Skip all tests in this file if not tesseract 4 -pytestmark = pytest.mark.skipif( - not tess4_available(), reason="tesseract 4.0 with textonly_pdf feature required" -) check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -def test_textonly_pdf(ensure_tess4, resources, outdir): - check_ocrmypdf( - resources / 'linn.pdf', - outdir / 'linn_textonly.pdf', - '--pdf-renderer', - 'sandwich', - '--sidecar', - outdir / 'foo.txt', - env=ensure_tess4, - ) +def test_tesseract_v4(): + assert tesseract.v4() -def test_pagesize_consistency_tess4(ensure_tess4, resources, outpdf): +def test_pagesize_consistency_tess4(resources, outpdf): from math import isclose infile = resources / 'linn.pdf' @@ -121,7 +53,6 @@ def test_pagesize_consistency_tess4(ensure_tess4, resources, outpdf): '--deskew', '--remove-background', '--clean-final' if pytest.helpers.have_unpaper() else None, - env=ensure_tess4, ) after_dims = pytest.helpers.first_page_dimensions(outpdf) @@ -131,7 +62,7 @@ def test_pagesize_consistency_tess4(ensure_tess4, resources, outpdf): @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) -def test_skip_pages_does_not_replicate(ensure_tess4, resources, basename, outdir): +def test_skip_pages_does_not_replicate(resources, basename, outdir): infile = resources / basename outpdf = outdir / basename @@ -143,7 +74,6 @@ def test_skip_pages_does_not_replicate(ensure_tess4, resources, basename, outdir '--force-ocr', '--tesseract-timeout', '0', - env=ensure_tess4, ) info_in = pdfinfo.PdfInfo(infile) @@ -156,17 +86,11 @@ def test_skip_pages_does_not_replicate(ensure_tess4, resources, basename, outdir assert info[n].width_inches == info_in[n].width_inches -def test_content_preservation(ensure_tess4, resources, outpdf): +def test_content_preservation(resources, outpdf): infile = resources / 'masks.pdf' check_ocrmypdf( - infile, - outpdf, - '--pdf-renderer', - 'sandwich', - '--tesseract-timeout', - '0', - env=ensure_tess4, + infile, outpdf, '--pdf-renderer', 'sandwich', '--tesseract-timeout', '0' ) info = pdfinfo.PdfInfo(outpdf) @@ -174,12 +98,10 @@ def test_content_preservation(ensure_tess4, resources, outpdf): assert len(page.images) > 1, "masks were rasterized" -def test_no_languages(ensure_tess4, tmp_path): - env = ensure_tess4 +def test_no_languages(tmp_path): + env = os.environ.copy() (tmp_path / 'tessdata').mkdir() env['TESSDATA_PREFIX'] = fspath(tmp_path) - with modified_os_environ(env): - with pytest.raises(MissingDependencyError): - tesseract.languages.cache_clear() - tesseract.languages() + with pytest.raises(MissingDependencyError): + tesseract.languages(tesseract_env=env) From 98a3fda1f51db1b4140af25291bf88f25d7e37d9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Jun 2019 01:39:41 -0700 Subject: [PATCH 076/880] Drop support for Tesseract 4 alpha releases without textonly_pdf (mostly) hocr renderer can still be used --- src/ocrmypdf/_validation.py | 12 +++++++++++- src/ocrmypdf/_weave.py | 10 ---------- 2 files changed, 11 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 2770874a..eacff184 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -106,6 +106,14 @@ def check_options_output(options): if options.pdf_renderer == 'auto': options.pdf_renderer = 'sandwich' + if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( + options.tesseract_env + ): + raise MissingDependencyError( + "You are using an alpha version of Tesseract 4.0 that does not support " + "the textonly_pdf parameter. We don't support versions this old." + ) + if options.output_type == 'pdfa': options.output_type = 'pdfa-2' @@ -216,7 +224,9 @@ def check_options_advanced(options): "--pdfa-image-compression argument has no effect when " "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" ) - if tesseract.v4() and (options.user_words or options.user_patterns): + if tesseract.v4(options.tesseract_env) and ( + options.user_words or options.user_patterns + ): log.warning('Tesseract 4.x ignores --user-words, so this has no effect') diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index ac989a72..c9f64963 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -103,16 +103,6 @@ def _weave_layers_graft( pdf_text = pikepdf.open(text) pdf_text_contents = pdf_text.pages[0].Contents.read_bytes() - if not tesseract.has_textonly_pdf(): - # If we don't have textonly_pdf, edit the stream to delete the - # instruction to draw the image Tesseract generated, which we do not - # use. - stream = bytearray(pdf_text_contents) - pattern = b'/Im1 Do' - idx = stream.find(pattern) - stream[idx : (idx + len(pattern))] = b' ' * len(pattern) - pdf_text_contents = bytes(stream) - base_page = pdf_base.pages.p(page_num) # The text page always will be oriented up by this stage but the original From eb5200d26aca2115ae5cc832d99954d00f83a48a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Jun 2019 01:45:27 -0700 Subject: [PATCH 077/880] Change most tests to use ocrmypdf API instead of subprocess The main benefit of this is code coverage gains can actually follow it. Also removes most ugly os.environ hacks. --- src/ocrmypdf/__main__.py | 2 - src/ocrmypdf/_pipeline.py | 3 + src/ocrmypdf/api.py | 28 +++++-- src/ocrmypdf/cli.py | 1 + src/ocrmypdf/exec/__init__.py | 3 +- src/ocrmypdf/exec/tesseract.py | 72 +++++++++++++----- .../pdf.bin | Bin 0 -> 3610 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 13 ++++ .../pdf.bin | Bin 0 -> 3610 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 13 ++++ tests/cache/manifest.jsonl | 2 + tests/conftest.py | 24 +++--- tests/test_filters.py | 1 - tests/test_main.py | 7 +- 18 files changed, 131 insertions(+), 40 deletions(-) create mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin create mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 0ec9050f..c24796af 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -35,8 +35,6 @@ def run(args=None): if not check_closed_streams(options): return ExitCode.bad_args - if os.environ.get('PYTEST_CURRENT_TEST'): - os.environ['_OCRMYPDF_TEST_INFILE'] = options.input_file if hasattr(os, 'nice'): os.nice(5) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 592f2411..189e6fbc 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -348,6 +348,7 @@ def get_orientation_correction(preview, page_context): engine_mode=page_context.options.tesseract_oem, timeout=page_context.options.tesseract_timeout, log=page_context.log, + tesseract_env=page_context.options.tesseract_env, ) direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} @@ -548,6 +549,7 @@ def ocr_tesseract_hocr(input_file, page_context): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, + tesseract_env=options.tesseract_env, log=page_context.log, ) return (hocr_out, hocr_text_out) @@ -627,6 +629,7 @@ def ocr_tesseract_textonly_pdf(input_image, page_context): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, + tesseract_env=options.tesseract_env, log=page_context.log, ) return (output_pdf, output_text) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index ab648170..8e5c1e1b 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -16,6 +16,7 @@ # along with OCRmyPDF. If not, see . import logging +import os import sys from enum import IntEnum from pathlib import Path @@ -113,13 +114,16 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= def create_options(*, input_file, output_file, **kwargs): cmdline = [] - filters = [] + deferred = [] for arg, val in kwargs.items(): if val is None: continue if arg.startswith('filter') and (callable(val) or isinstance(val, str)): - filters.append((arg, val)) + deferred.append((arg, val)) + continue + elif arg == 'tesseract_env': + deferred.append((arg, val)) continue cmd_style_arg = arg.replace('_', '-') cmdline.append(f"--{cmd_style_arg}") @@ -132,15 +136,20 @@ def create_options(*, input_file, output_file, **kwargs): elif isinstance(val, Path): cmdline.append(str(val)) else: - raise TypeError(f"{val} ({type(val)})") + raise TypeError(f"{arg}: {val} ({type(val)})") cmdline.append(str(input_file)) cmdline.append(str(output_file)) parser.api_mode = True options = parser.parse_args(cmdline) - for keyword, function in filters: - setattr(options, keyword, function) + for keyword, val in deferred: + setattr(options, keyword, val) + + # If we are running a Tesseract spoof, ensure it knows what the input file is + if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env: + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file + return options @@ -190,9 +199,18 @@ def ocrmypdf( # pylint: disable=unused-argument keep_temporary_files=None, progress_bar=None, filter_ocr_image=None, + tesseract_env=None, ): """Run OCRmyPDF on one PDF or image. + For most arguments, see documentation for the equivalent command line parameter. + A few specific arguments are discussed here: + + Args: + use_threads (bool): Use worker threads instead of processes. This reduces + performance but may make debugging easier since it is easier to set + breakpoints. + tesseract_env (dict): Override environment variables for Tesseract Raises: ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging with the OCR layer. diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 75833362..ac2f2713 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -482,3 +482,4 @@ debugging.add_argument( action='store_true', help="Keep temporary files (helpful for debugging)", ) +debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index b09b516b..28193a01 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -28,7 +28,7 @@ from collections.abc import Mapping log = logging.Logger(__name__) -def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)'): +def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): "Get the version of the specified program" args_prog = [program, version_arg] try: @@ -39,6 +39,7 @@ def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)'): stdout=PIPE, stderr=STDOUT, check=True, + env=env, ) output = proc.stdout except FileNotFoundError as e: diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 110c49f3..467a9b7c 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -60,18 +60,16 @@ HOCR_TEMPLATE = """ """ -@lru_cache(maxsize=1) -def version(): - return get_version('tesseract', regex=r'tesseract\s(.+)') +def version(tesseract_env=None): + return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env) -def v4(): +def v4(tesseract_env=None): "Is this Tesseract v4.0?" - return version() >= '4' + return version(tesseract_env) >= '4' -@lru_cache(maxsize=1) -def has_textonly_pdf(): +def has_textonly_pdf(tesseract_env=None): """Does Tesseract have textonly_pdf capability? Available in v4.00.00alpha since January 2017. Best to @@ -80,7 +78,15 @@ def has_textonly_pdf(): args_tess = ['tesseract', '--print-parameters', 'pdf'] params = '' try: - params = check_output(args_tess, universal_newlines=True, stderr=STDOUT) + proc = run( + args_tess, + check=True, + universal_newlines=True, + stdout=PIPE, + stderr=STDOUT, + env=tesseract_env, + ) + params = proc.stdout except CalledProcessError as e: print("Could not --print-parameters from tesseract", file=sys.stderr) raise MissingDependencyError from e @@ -89,8 +95,7 @@ def has_textonly_pdf(): return False -@lru_cache(maxsize=1) -def languages(): +def languages(tesseract_env=None): def lang_error(output): msg = dedent( """Tesseract failed to report available languages. @@ -104,7 +109,12 @@ def languages(): args_tess = ['tesseract', '--list-langs'] try: proc = run( - args_tess, universal_newlines=True, stdout=PIPE, stderr=STDOUT, check=True + args_tess, + universal_newlines=True, + stdout=PIPE, + stderr=STDOUT, + check=True, + env=tesseract_env, ) output = proc.stdout except CalledProcessError as e: @@ -127,7 +137,7 @@ def tess_base_args(langs, engine_mode): return args -def get_orientation(input_file, engine_mode, timeout: float, log): +def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=None): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', @@ -136,7 +146,15 @@ def get_orientation(input_file, engine_mode, timeout: float, log): ] try: - stdout = check_output(args_tesseract, stderr=STDOUT, timeout=timeout) + p = run( + args_tesseract, + stdout=PIPE, + stderr=STDOUT, + timeout=timeout, + check=True, + env=tesseract_env, + ) + stdout = p.stdout except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) except CalledProcessError as e: @@ -235,6 +253,7 @@ def generate_hocr( pagesegmode: int, user_words, user_patterns, + tesseract_env, log, ): @@ -258,7 +277,15 @@ def generate_hocr( args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) try: log.debug(args_tesseract) - stdout = check_output(args_tesseract, stderr=STDOUT, timeout=timeout) + p = run( + args_tesseract, + stdout=PIPE, + stderr=STDOUT, + timeout=timeout, + check=True, + env=tesseract_env, + ) + stdout = p.stdout except TimeoutExpired: # Generate a HOCR file with no recognized text if tesseract times out # Temporary workaround to hocrTransform not being able to function if @@ -310,9 +337,10 @@ def generate_pdf( pagesegmode: int, user_words, user_patterns, + tesseract_env, log, ): - '''Use Tesseract to render a PDF. + """Use Tesseract to render a PDF. input_image -- image to analyze skip_pdf -- if we time out, use this file as output @@ -324,14 +352,14 @@ def generate_pdf( tessconfig -- tesseract configuration timeout -- timeout (seconds) log -- logger object - ''' + """ args_tesseract = tess_base_args(language, engine_mode) if pagesegmode is not None: args_tesseract.extend(['--psm', str(pagesegmode)]) - if text_only and has_textonly_pdf(): + if text_only and has_textonly_pdf(tesseract_env): args_tesseract.extend(['-c', 'textonly_pdf=1']) if user_words: @@ -348,7 +376,15 @@ def generate_pdf( args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig) try: log.debug(args_tesseract) - stdout = check_output(args_tesseract, stderr=STDOUT, timeout=timeout) + p = run( + args_tesseract, + stdout=PIPE, + stderr=STDOUT, + timeout=timeout, + check=True, + env=tesseract_env, + ) + stdout = p.stdout if os.path.exists(prefix + '.txt'): shutil.move(prefix + '.txt', output_text) except TimeoutExpired: diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..bd00d06d58aa38342d2146f3d8aabb668419456a GIT binary patch literal 3610 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fti7^fuWIsp^>hExw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0B;o< AlmGw# literal 0 HcmV?d00001 diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..61f78d82 --- /dev/null +++ b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..25fdded2 --- /dev/null +++ b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,13 @@ +Portez ce vieux whisky au juge +blond qui fume sur son Ile +interieure, a cöte de l'alcöve +ovoide, oU les büches se +consument dans l'ätre, ce qui +lui permet de penser & la +caenogenese de |'etre dont il +est question dans la cause +ambigu& entendue a MoY, dans +un capharnaüm qui, pense-t-il, +diminue ca et la la qualite de son +ceuvre. + \ No newline at end of file diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..7d8ac39f02fe6b90b1dcdac74080f96a3944ada3 GIT binary patch literal 3610 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fti7^fuV_!k+H6Uxw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0C$!e Ang9R* literal 0 HcmV?d00001 diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..61f78d82 --- /dev/null +++ b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..25fdded2 --- /dev/null +++ b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,13 @@ +Portez ce vieux whisky au juge +blond qui fume sur son Ile +interieure, a cöte de l'alcöve +ovoide, oU les büches se +consument dans l'ätre, ce qui +lui permet de penser & la +caenogenese de |'etre dont il +est question dans la cause +ambigu& entendue a MoY, dans +un capharnaüm qui, pense-t-il, +diminue ca et la la qualite de son +ceuvre. + \ No newline at end of file diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 325350eb..c9a1fc25 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -63,3 +63,5 @@ {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} diff --git a/tests/conftest.py b/tests/conftest.py index 6ff87b4b..e386bdac 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -21,6 +21,7 @@ import sys from contextlib import contextmanager from pathlib import Path from subprocess import PIPE, run +from ocrmypdf import api, cli import pytest @@ -185,18 +186,21 @@ def no_outpdf(tmp_path): def check_ocrmypdf(input_file, output_file, *args, env=None): """Run ocrmypdf and confirmed that a valid file was created""" - p, out, err = run_ocrmypdf(input_file, output_file, *args, env=env) - # ensure py.test collects the output, use -s to view - print(err, file=sys.stderr) - assert p.returncode == 0 + # p, out, err = run_ocrmypdf(input_file, output_file, *args, env=env) + + options = cli.parser.parse_args( + [str(input_file), str(output_file)] + [str(arg) for arg in args] + ) + api.check_options(options) + if env: + options.tesseract_env = env + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file + result = api.run_pipeline(options, api=True) + + assert result == 0 assert os.path.exists(str(output_file)), "Output file not created" assert os.stat(str(output_file)).st_size > 100, "PDF too small or empty" - assert out == "", ( - "The following was written to stdout and should not have been: \n" - + "\n" - + out - + "\n" - ) + return output_file diff --git a/tests/test_filters.py b/tests/test_filters.py index 12628e80..89c34a25 100644 --- a/tests/test_filters.py +++ b/tests/test_filters.py @@ -26,7 +26,6 @@ from ocrmypdf.filters import invert, whiteout from ocrmypdf._plugins import load_plugin -os_environ = pytest.helpers.os_environ check_ocrmypdf = pytest.helpers.check_ocrmypdf diff --git a/tests/test_main.py b/tests/test_main.py index 10b6eb5b..86f52b21 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -40,7 +40,6 @@ from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -os_environ = pytest.helpers.os_environ RENDERERS = ['hocr', 'sandwich'] @@ -614,8 +613,10 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): def test_masks(spoof_tesseract_noop, resources, outpdf): - with os_environ(spoof_tesseract_noop): - assert ocrmypdf(resources / 'masks.pdf', outpdf) == ExitCode.ok + assert ( + ocrmypdf(resources / 'masks.pdf', outpdf, tesseract_env=spoof_tesseract_noop) + == ExitCode.ok + ) def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop, resources, outpdf): From 5ab69153eef22bfad200c623f76064f030fb389c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Jun 2019 02:26:49 -0700 Subject: [PATCH 078/880] Fix .coveragerc --- .coveragerc | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/.coveragerc b/.coveragerc index ddcb72b0..b4e940b7 100644 --- a/.coveragerc +++ b/.coveragerc @@ -1,9 +1,18 @@ # Coverage isn't really compatible with subprocesses so results are unreliable +[paths] +source = + src + */site-packages + [run] -branch = True -#concurrency = multiprocessing -source = ocrmypdf/ +branch = true +parallel = true +source = + src/ocrmypdf + tests +omit = + tests/spoof/* [report] exclude_lines = From 9444cf357b917fa72c8d7885b479947804fe976a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 4 Jun 2019 02:01:53 -0700 Subject: [PATCH 079/880] optimize: add divide by zero check --- src/ocrmypdf/optimize.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 21351291..0ee1ef98 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -17,6 +17,7 @@ import concurrent.futures import sys +import tempfile from collections import defaultdict from os import fspath from pathlib import Path @@ -29,6 +30,7 @@ from pikepdf import Name, Dictionary from . import leptonica from ._jobcontext import PDFContext from .exec import jbig2enc, pngquant +from .exceptions import OutputFileAccessError from .helpers import re_symlink DEFAULT_JPEG_QUALITY = 75 @@ -484,6 +486,11 @@ def optimize(input_file, output_file, context): input_size = Path(input_file).stat().st_size output_size = Path(target_file).stat().st_size + if output_size == 0: + raise OutputFileAccessError( + f"Output file not created after optimizing. We probably ran " + f"out of disk space in the temporary folder: {tempfile.gettempdir()}." + ) ratio = input_size / output_size savings = 1 - output_size / input_size log.info(f"Optimize ratio: {ratio:.2f} savings: {(100 * savings):.1f}%") From fd427a8ec14d5dd2a16f593182c542a818165100 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Jun 2019 01:46:56 -0700 Subject: [PATCH 080/880] plugins: replace path manipulation --- src/ocrmypdf/_plugins.py | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_plugins.py b/src/ocrmypdf/_plugins.py index 89655b9f..c4bf5d0a 100644 --- a/src/ocrmypdf/_plugins.py +++ b/src/ocrmypdf/_plugins.py @@ -19,7 +19,7 @@ import logging import importlib import os import sys - +from pathlib import Path log = logging.getLogger(__name__) @@ -51,10 +51,7 @@ def _load_function_from_pyfile(location): filename, object_name = location.split('::', maxsplit=1) log.debug(f"Loading function {object_name} from {filename}") - module_name = os.path.basename(filename) - if module_name.endswith('.py'): - module_name = module_name[:-3] - + module_name = Path(filename).stem spec = importlib.util.spec_from_file_location(module_name, filename) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) From 93f1b73579827479d20fc618ad2540e2ff5374e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Jun 2019 02:04:45 -0700 Subject: [PATCH 081/880] Fix --remove-vectors which was broken in API migration It got dropped during the change. This feature has also been altered so that the final visual appearance of the file is not affected, only the OCR image. --- src/ocrmypdf/_pipeline.py | 14 ++++++--- src/ocrmypdf/_sync.py | 65 ++++++++++++++++++++++++++++----------- 2 files changed, 57 insertions(+), 22 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 189e6fbc..c09c31f1 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -392,10 +392,16 @@ def get_orientation_correction(preview, page_context): return 0 -def rasterize(input_file, page_context, correction=0): +def rasterize( + input_file, page_context, correction=0, output_tag='', remove_vectors=None +): colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m'] device_idx = 0 - output_file = page_context.get_path('rasterize.png') + + if remove_vectors is None: + remove_vectors = page_context.options.remove_vectors + + output_file = page_context.get_path(f'rasterize{output_tag}.png') pageinfo = page_context.pageinfo def at_least(cs): @@ -431,7 +437,7 @@ def rasterize(input_file, page_context, correction=0): page_dpi=(page_dpi, page_dpi), pageno=pageinfo.pageno + 1, rotation=correction, - filter_vector=page_context.options.remove_vectors, + filter_vector=remove_vectors, ) return output_file @@ -520,7 +526,7 @@ def create_ocr_image(image, page_context): barcodes = pix.locate_barcodes() for barcode in barcodes: decoded, rect = barcode - print('masking barcode %s %r', decoded, rect) + page_context.log.debug('masking barcode %s %r', decoded, rect) draw.rectangle(rect, fill=white) im = pix.topil() diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 62fc5a39..1ca758d1 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -73,6 +73,16 @@ PageResult = namedtuple( ) +def preprocess(page_context, image, remove_background, deskew, clean): + if remove_background: + image = preprocess_remove_background(image, page_context) + if deskew: + image = preprocess_deskew(image, page_context) + if clean: + image = preprocess_clean(image, page_context) + return image + + def exec_page_sync(page_context): options = page_context.options orientation_correction = 0 @@ -88,27 +98,46 @@ def exec_page_sync(page_context): ) rasterize_out = rasterize( - page_context.origin, page_context, correction=orientation_correction + page_context.origin, + page_context, + correction=orientation_correction, + remove_vectors=False, ) - preprocess = rasterize_out - if options.remove_background: - preprocess = preprocess_remove_background(preprocess, page_context) - - if options.deskew: - preprocess = preprocess_deskew(preprocess, page_context) - - if options.clean: - cleaned = preprocess_clean(preprocess, page_context) - if options.clean_final: - preprocess_out = cleaned - ocr_image = cleaned - else: - preprocess_out = preprocess - ocr_image = cleaned + if not any([options.clean, options.clean_final, options.remove_vectors]): + ocr_image = preprocess_out = preprocess( + page_context, + rasterize_out, + options.remove_background, + options.deskew, + clean=False, + ) else: - preprocess_out = preprocess - ocr_image = preprocess + if not options.lossless_reconstruction: + preprocess_out = preprocess( + page_context, + rasterize_out, + options.remove_background, + options.deskew, + clean=options.clean_final, + ) + if options.remove_vectors: + rasterize_ocr_out = rasterize( + page_context.origin, + page_context, + correction=orientation_correction, + remove_vectors=True, + output_tag='_ocr', + ) + else: + rasterize_ocr_out = rasterize_out + ocr_image = preprocess( + page_context, + rasterize_ocr_out, + options.remove_background, + options.deskew, + clean=options.clean, + ) ocr_image_out = create_ocr_image(ocr_image, page_context) From 20ad032977ce37b301eb1b3637e36ef9b113e7f7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Jun 2019 03:07:48 -0700 Subject: [PATCH 082/880] Fix some error messages that printed directly to sys.stderr instead of logging --- src/ocrmypdf/exec/tesseract.py | 32 +++++++++++--------------------- src/ocrmypdf/exec/unpaper.py | 16 +++++++--------- tests/conftest.py | 4 ++-- 3 files changed, 20 insertions(+), 32 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 467a9b7c..c16a9202 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -22,15 +22,7 @@ from collections import namedtuple from contextlib import suppress from functools import lru_cache from os import fspath -from subprocess import ( - PIPE, - STDOUT, - CalledProcessError, - TimeoutExpired, - check_output, - run, -) -from textwrap import dedent +from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run from . import get_version from ..exceptions import ( @@ -88,8 +80,9 @@ def has_textonly_pdf(tesseract_env=None): ) params = proc.stdout except CalledProcessError as e: - print("Could not --print-parameters from tesseract", file=sys.stderr) - raise MissingDependencyError from e + raise MissingDependencyError( + "Could not --print-parameters from tesseract" + ) from e if 'textonly_pdf' in params: return True return False @@ -97,14 +90,13 @@ def has_textonly_pdf(tesseract_env=None): def languages(tesseract_env=None): def lang_error(output): - msg = dedent( - """Tesseract failed to report available languages. - Output from Tesseract: - ----------- - """ + msg = ( + "Tesseract failed to report available languages.\n" + "Output from Tesseract:\n" + "-----------\n" ) msg += output - print(msg, file=sys.stderr) + return msg args_tess = ['tesseract', '--list-langs'] try: @@ -118,13 +110,11 @@ def languages(tesseract_env=None): ) output = proc.stdout except CalledProcessError as e: - lang_error(e.output) - raise MissingDependencyError from e + raise MissingDependencyError(lang_error(e.output)) from e header, *rest = output.splitlines() if not header.startswith('List of available languages'): - lang_error(output) - raise MissingDependencyError + raise MissingDependencyError(lang_error(output)) return set(lang.strip() for lang in rest) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 40c79594..9c9db5d5 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -29,11 +29,7 @@ from tempfile import TemporaryDirectory from . import get_version from ..exceptions import MissingDependencyError, SubprocessOutputError -try: - from PIL import Image -except ImportError: - print("Could not find Python3 imaging library", file=sys.stderr) - raise +from PIL import Image @lru_cache(maxsize=1) @@ -55,16 +51,18 @@ def run(input_file, output_file, dpi, log, mode_args): else: im = im.convert(mode='RGB') except IOError as e: - log.error("Could not convert image with type " + im.mode) im.close() - raise MissingDependencyError() from e + raise MissingDependencyError( + "Could not convert image with type " + im.mode + ) from e try: suffix = SUFFIXES[im.mode] except KeyError: - log.error("Failed to convert image to a supported format.") im.close() - raise MissingDependencyError() from e + raise MissingDependencyError( + "Failed to convert image to a supported format." + ) from e with TemporaryDirectory() as tmpdir: input_pnm = os.path.join(tmpdir, f'input{suffix}') diff --git a/tests/conftest.py b/tests/conftest.py index e386bdac..4a4f0347 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -37,8 +37,8 @@ else: # pylint: disable=E1101 # pytest.helpers is dynamic so it confuses pylint -if sys.version_info.major < 3: - print("Requires Python 3.4+") +if sys.version_info < (3, 5): + print("Requires Python 3.5+") sys.exit(1) From 81fc95556c4a6c940f584b0c5f41de65a198857e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Jun 2019 03:08:04 -0700 Subject: [PATCH 083/880] Add progress bar for PdfInfo step --- src/ocrmypdf/_pipeline.py | 6 ++++-- src/ocrmypdf/_sync.py | 9 +++++---- src/ocrmypdf/pdfinfo/__init__.py | 15 +++++++++++---- 3 files changed, 20 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index c09c31f1..bf3e979f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -144,9 +144,11 @@ def triage(input_file, output_file, options, log): return output_file -def get_pdfinfo(input_file, detailed_page_analysis=False): +def get_pdfinfo(input_file, detailed_page_analysis=False, progbar=False): try: - return PdfInfo(input_file, detailed_page_analysis=detailed_page_analysis) + return PdfInfo( + input_file, detailed_page_analysis=detailed_page_analysis, progbar=progbar + ) except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 1ca758d1..013be469 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -55,9 +55,6 @@ from ._pipeline import ( validate_pdfinfo_options, ) from ._validation import ( - check_dependency_versions, - check_environ, - check_options, check_requested_output_file, create_input_file, report_output_file_size, @@ -320,7 +317,11 @@ def run_pipeline(options, api=False): ) # Gather pdfinfo and create context - pdfinfo = get_pdfinfo(origin_pdf, detailed_page_analysis=options.redo_ocr) + pdfinfo = get_pdfinfo( + origin_pdf, + detailed_page_analysis=options.redo_ocr, + progbar=options.progress_bar, + ) context = PDFContext(options, work_folder, origin_pdf, pdfinfo) # Validate options are okay for this pdf diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index c6c4f7e9..ea09ba99 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -28,6 +28,7 @@ import re from pikepdf import PdfMatrix import pikepdf +from tqdm import tqdm from . import ghosttext @@ -616,7 +617,7 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): return pageinfo -def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None): +def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False): pdf = pikepdf.open(infile) # Do not close in this function if pdf.is_encrypted: pdf.close() @@ -627,7 +628,13 @@ def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None): pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) pages = [] - for n in range(len(pdf.pages)): + for n, _ in tqdm( + enumerate(pdf.pages), + total=len(pdf.pages), + desc="Scan", + unit='page', + disable=not progbar, + ): page_xml = pages_xml[n] if pages_xml else None page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) pages.append(page) @@ -749,10 +756,10 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, detailed_page_analysis=False, log=logger): + def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False): self._infile = infile self._pages, pdf = _pdf_get_all_pageinfo( - infile, detailed_page_analysis, log=log + infile, detailed_page_analysis, log=log, progbar=progbar ) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False From 5dd10c961c629d5596b1b8c4d3f9e6df029ced4c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Jun 2019 03:14:36 -0700 Subject: [PATCH 084/880] Docker: prefer streaming --- .docker/Dockerfile | 2 -- .docker/alpine.dockerfile | 2 -- docs/docker.rst | 13 ++++++++++--- 3 files changed, 10 insertions(+), 7 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 62fac8d3..c6bd51a6 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -76,6 +76,4 @@ COPY --from=builder /app/src /app/src COPY --from=builder /appenv /appenv COPY --from=builder /usr/local /usr/local -WORKDIR /data - ENTRYPOINT ["/appenv/bin/ocrmypdf"] diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index e6e743b5..d8472c11 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -88,6 +88,4 @@ COPY --from=builder /app/requirements /app/requirements COPY --from=builder /app/tests /app/tests COPY --from=builder /app/src /app/src -WORKDIR /data - ENTRYPOINT ["/usr/bin/ocrmypdf"] diff --git a/docs/docker.rst b/docs/docker.rst index df45ef23..b8f3657b 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -46,14 +46,15 @@ To start a Docker container (instance of the image): .. code-block:: bash docker tag jbarlow83/ocrmypdf-alpine ocrmypdf - docker run --rm ocrmypdf (... all other arguments here...) + docker run --rm -i ocrmypdf (... all other arguments here...) -For convenience, create a shell alias to hide the Docker command: +For convenience, create a shell alias to hide the Docker command. It is easier to send the input file to file stdin and read the output from stdout – this avoids the occasionally messy permission issues with Docker entirely. .. code-block:: bash - alias ocrmypdf='docker run --rm -v "$(pwd):/home/docker" ocrmypdf' + alias ocrmypdf='docker run --rm -i ocrmypdf' ocrmypdf --version # runs docker version + ocrmypdf output.pdf Or in the wonderful `fish shell `_: @@ -62,6 +63,12 @@ Or in the wonderful `fish shell `_: alias ocrmypdf 'docker run --rm ocrmypdf' funcsave ocrmypdf +Alternately, you could mount the local current working directory as a Docker volume: + +.. code-block:: bash + + docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf + .. _docker-lang-packs: Adding languages to the Docker image From 0bbd6885e2005424450bff540e34569db10f1a7c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 6 Jun 2019 23:07:46 -0700 Subject: [PATCH 085/880] Make the go/no-go decision pluggable --- src/ocrmypdf/_pipeline.py | 6 ++++++ src/ocrmypdf/api.py | 5 ++++- src/ocrmypdf/cli.py | 7 +++++-- src/ocrmypdf/pdfinfo/layout.py | 2 +- 4 files changed, 16 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index bf3e979f..cba5de41 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -160,6 +160,12 @@ def validate_pdfinfo_options(context): pdfinfo = context.pdfinfo options = context.options + if options.plugin_validation: + validate = load_plugin(options.plugin_validation) + result = validate(context) + if result is not None: + return result + if pdfinfo.needs_rendering: log.error( "This PDF contains dynamic XFA forms created by Adobe LiveCycle " diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 8e5c1e1b..3bddf47c 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -119,7 +119,9 @@ def create_options(*, input_file, output_file, **kwargs): for arg, val in kwargs.items(): if val is None: continue - if arg.startswith('filter') and (callable(val) or isinstance(val, str)): + if (arg.startswith('plugin') or arg.startswith('filter')) and ( + callable(val) or isinstance(val, str) + ): deferred.append((arg, val)) continue elif arg == 'tesseract_env': @@ -199,6 +201,7 @@ def ocrmypdf( # pylint: disable=unused-argument keep_temporary_files=None, progress_bar=None, filter_ocr_image=None, + plugin_validation=None, tesseract_env=None, ): """Run OCRmyPDF on one PDF or image. diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index ac2f2713..cca69960 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -468,10 +468,13 @@ advanced.add_argument( help="Specify the location of the Tesseract user patterns file.", ) -filters = parser.add_argument_group("Filters", argparse.SUPPRESS) -filters.add_argument( +plugins = parser.add_argument_group("Filters and Plugins", argparse.SUPPRESS) +plugins.add_argument( '--filter-ocr-image', help=argparse.SUPPRESS, type=check_plugin_loadable ) +plugins.add_argument( + '--plugin-validation', help=argparse.SUPPRESS, type=check_plugin_loadable +) debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 89caa938..9bb7f3a5 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -185,7 +185,7 @@ class LTStateAwareChar(LTChar): def get_text(self): if isinstance(self._text, tuple): - return '�' + return '\ufffd' # standard 'Unknown symbol' return self._text def __repr__(self): From 066a293462b0a9042073889b376a3fb93e5d5a20 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Jun 2019 13:55:43 -0700 Subject: [PATCH 086/880] If verbose, print stacktrace on KeyboardInterrupt --- src/ocrmypdf/_sync.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 013be469..a9f22294 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -330,7 +330,10 @@ def run_pipeline(options, api=False): # Execute the pipeline exec_concurrent(context) except (KeyboardInterrupt if not api else NeverRaise) as e: - log.error("KeyboardInterrupt") + if options.verbose >= 1: + log.exception("KeyboardInterrupt") + else: + log.error("KeyboardInterrupt") return ExitCode.ctrl_c except (ExitCodeException if not api else NeverRaise) as e: if str(e): From aba293fd802e4e7e3c0c89fbbe90ddcd9932ec6f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Jun 2019 13:56:02 -0700 Subject: [PATCH 087/880] Change "Temporary working files" output message --- docs/advanced.rst | 2 +- src/ocrmypdf/_jobcontext.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/advanced.rst b/docs/advanced.rst index da1f1675..518b5bf1 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -226,7 +226,7 @@ If the ``-k`` argument is issued on the command line, OCRmyPDF will keep the tem .. code-block:: none - Temporary working files saved at: + Temporary working files retained at: /tmp/com.github.ocrmypdf.u20wpz07 The organization of this folder is an implementation detail and subject to change between releases. However the general organization is that working files on a per page basis have the page number as a prefix (starting with page 1), an infix indicates the processing stage, and a suffix indicates the file type. Some important files include: diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 2a141e7f..96ad6c7e 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -93,7 +93,7 @@ class PageContext(PicklableLoggerMixin): def cleanup_working_files(work_folder, options): if options.keep_temporary_files: - print(f"Temporary working files saved at:\n{work_folder}", file=sys.stderr) + print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr) else: shutil.rmtree(work_folder, ignore_errors=True) From 8b8de7cc1d783023d2408923c8fbdfea805e33ea Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Jun 2019 17:27:47 -0700 Subject: [PATCH 088/880] Add new --pages feature to limit OCR to only specific pages --- src/ocrmypdf/_pipeline.py | 8 ++++-- src/ocrmypdf/_validation.py | 37 +++++++++++++++++++++--- src/ocrmypdf/api.py | 1 + src/ocrmypdf/cli.py | 5 ++++ src/ocrmypdf/helpers.py | 9 ++++++ tests/test_page_numbers.py | 56 +++++++++++++++++++++++++++++++++++++ 6 files changed, 109 insertions(+), 7 deletions(-) create mode 100644 tests/test_page_numbers.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index cba5de41..e7454a78 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -251,12 +251,14 @@ def is_ocr_required(page_context): ocr_required = True - if pageinfo.has_text: + if options.pages and pageinfo.pageno not in options.pages: + log.debug(f"skipped {pageinfo.pageno} as requested by --pages {options.pages}") + ocr_required = False + elif pageinfo.has_text: if not options.force_ocr and not (options.skip_text or options.redo_ocr): - log.error( + raise PriorOcrFoundError( "page already has text! - aborting (use --force-ocr to force OCR)" ) - raise PriorOcrFoundError() elif options.force_ocr: log.info("page already has text! - rasterizing text and running OCR anyway") ocr_required = True diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index eacff184..8069ae1b 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -41,7 +41,7 @@ from .exec import ( tesseract, unpaper, ) -from .helpers import is_file_writable, re_symlink +from .helpers import is_file_writable, re_symlink, is_iterable_notstr, monotonic # ------------- # External dependencies @@ -172,6 +172,33 @@ def check_options_preprocessing(options): raise BadArgsError(str(e)) +def _pages_from_ranges(ranges): + if is_iterable_notstr(ranges): + return set(ranges) + pages = [] + page_groups = ranges.replace(' ', '').split(',') + for g in page_groups: + if not g: + continue + try: + start, end = g.split('-') + except ValueError: + pages.append(int(g) - 1) + else: + pages.extend(range(int(start) - 1, int(end))) + + if not monotonic(pages): + log.warning( + "List of pages to process contains duplicate pages, or pages that are " + "out of order" + ) + if any(page < 0 for page in pages): + raise BadArgsError("pages refers to a page number less than 1") + + log.debug("OCRing only these pages: %s", pages) + return set(pages) + + def check_options_ocr_behavior(options): exclusive_options = sum( [ @@ -180,9 +207,11 @@ def check_options_ocr_behavior(options): ] ) if exclusive_options >= 2: - raise BadArgsError( - "Error: choose only one of --force-ocr, --skip-text, --redo-ocr." - ) + raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.") + if options.pages and options.sidecar: + raise BadArgsError("--pages and --sidecar are mutually exclusive") + if options.pages: + options.pages = _pages_from_ranges(options.pages) def check_options_optimizing(options): diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 3bddf47c..e108d587 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -188,6 +188,7 @@ def ocrmypdf( # pylint: disable=unused-argument png_quality=None, jbig2_lossy=None, jbig2_page_group_size=None, + pages=None, max_image_mpixels=None, tesseract_config=None, tesseract_pagesegmode=None, diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index cca69960..6fe45275 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -380,6 +380,11 @@ optimizing.add_argument( advanced = parser.add_argument_group( "Advanced", "Advanced options to control Tesseract's OCR behavior" ) +advanced.add_argument( + '--pages', + type=str, + help="Limit OCR to the specified pages (ranges or comma separated), skipping others", +) advanced.add_argument( '--max-image-mpixels', action='store', diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 683cc815..8d52a822 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -66,6 +66,15 @@ def re_symlink(input_file, soft_link_name, *args, **kwargs): os.symlink(os.path.abspath(input_file), soft_link_name) +def is_iterable_notstr(thing): + return isinstance(thing, Iterable) and not isinstance(thing, str) + + +def monotonic(L): + """Does list increase monotonically?""" + return all(b > a for a, b in zip(L, L[1:])) + + def page_number(input_file): """Get one-based page number implied by filename (000002.pdf -> 2)""" return int(os.path.basename(os.fspath(input_file))[0:6]) diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py new file mode 100644 index 00000000..a870a987 --- /dev/null +++ b/tests/test_page_numbers.py @@ -0,0 +1,56 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import pytest + +from ocrmypdf import ocrmypdf as run +from ocrmypdf._validation import _pages_from_ranges +from ocrmypdf.pdfinfo import PdfInfo + + +def test_str_ranges(): + assert _pages_from_ranges('43') == {42} + assert _pages_from_ranges('1, 2, 3') == {0, 1, 2} + assert _pages_from_ranges('1-3') == {0, 1, 2} + assert _pages_from_ranges('1-3,5,7,42') == {0, 1, 2, 4, 6, 41} + assert _pages_from_ranges('3, 3, 3, 3,') == {2} + + +def test_nonmonotonic_warning(caplog): + pages = _pages_from_ranges('1, 3, 2') + assert pages == {0, 1, 2} + assert 'out of order' in caplog.text + + +def test_list_range(): + assert _pages_from_ranges([0, 1, 2]) == {0, 1, 2} + + +def test_limited_pages(resources, outpdf, spoof_tesseract_cache): + multi = resources / 'multipage.pdf' + run( + multi, + outpdf, + pages='5-6', + optimize=0, + output_type='pdf', + tesseract_env=spoof_tesseract_cache, + ) + pi = PdfInfo(outpdf) + assert not pi.pages[0].has_text + assert pi.pages[4].has_text + assert pi.pages[5].has_text From cfb11559d539630f4af24e1dc98ea830bc6784eb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Jun 2019 17:28:02 -0700 Subject: [PATCH 089/880] logging: capture warnings too --- src/ocrmypdf/api.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index e108d587..eaad44f9 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -111,6 +111,9 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= pil_log = logging.getLogger('PIL') pil_log.setLevel(logging.INFO) + if manage_root_logger: + logging.captureWarnings(True) + def create_options(*, input_file, output_file, **kwargs): cmdline = [] From 16990890d8c285ac250d73a4cce552706efb9864 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Jun 2019 17:52:25 -0700 Subject: [PATCH 090/880] Remove "from ocrmypdf import ocrmypdf" Messes up future imports from ocrmypdf, so don't do it. --- docs/api.rst | 10 +++++----- src/ocrmypdf/__init__.py | 2 +- src/ocrmypdf/api.py | 2 +- tests/test_filters.py | 4 ++-- tests/test_main.py | 6 ++++-- tests/test_page_numbers.py | 4 ++-- tests/test_weave.py | 6 +++--- 7 files changed, 18 insertions(+), 16 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index d4aeeca5..32efd3be 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -12,9 +12,9 @@ OCRmyPDF one high-level function to run its main engine from an application. The .. code-block:: python - from ocrmypdf import ocrmypdf + import ocrmypdf - ocrmypdf('input.pdf', 'output.pdf', deskew=True) + ocrmypdf.run('input.pdf', 'output.pdf', deskew=True) With a few exceptions, all of the command line arguments are available and may be passed as equivalent keywords. @@ -29,11 +29,11 @@ The :func:`ocrmypdf.ocrmypdf` function runs OCRmyPDF similar to command line exe - manage the signal flags of worker processes 0 execute other subprocesses (forking and executing other programs) -The Python process that calls ``ocrmypdf()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will fail. +The Python process that calls ``ocrmypdf.run()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will fail. There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. -Forking a child process to call ``ocrmypdf()`` is suggested. That way your application will survive even if OCRmyPDF does not. +Forking a child process to call ``ocrmypdf.run()`` is suggested. That way your application will survive even if OCRmyPDF does not. Logging ^^^^^^^ @@ -59,7 +59,7 @@ When OCRmyPDF succeeds conditionally, it may return an integer exit code. Reference --------- -.. autofunction:: ocrmypdf.ocrmypdf +.. autofunction:: ocrmypdf.run .. autoclass:: ocrmypdf.Verbosity :members: diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 29491ca9..00d757cf 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,4 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo -from .api import ocrmypdf, configure_logging, Verbosity +from .api import run, configure_logging, Verbosity diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index eaad44f9..94c3c6d3 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -158,7 +158,7 @@ def create_options(*, input_file, output_file, **kwargs): return options -def ocrmypdf( # pylint: disable=unused-argument +def run( # pylint: disable=unused-argument input_file, output_file, *, diff --git a/tests/test_filters.py b/tests/test_filters.py index 89c34a25..5d9482ea 100644 --- a/tests/test_filters.py +++ b/tests/test_filters.py @@ -21,7 +21,7 @@ from PIL import Image import pytest -from ocrmypdf import ocrmypdf +import ocrmypdf from ocrmypdf.filters import invert, whiteout from ocrmypdf._plugins import load_plugin @@ -80,7 +80,7 @@ def test_filter_from_cmdline(resources, outdir): def test_filter_from_api(resources, outdir): - ocrmypdf( + ocrmypdf.run( resources / 'crom.png', outdir / 'out.pdf', image_dpi=100, diff --git a/tests/test_main.py b/tests/test_main.py index 86f52b21..7e2693e4 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -27,7 +27,7 @@ import PIL import pytest from PIL import Image -from ocrmypdf import ocrmypdf +import ocrmypdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import ghostscript, qpdf, tesseract from ocrmypdf.leptonica import Pix @@ -614,7 +614,9 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): def test_masks(spoof_tesseract_noop, resources, outpdf): assert ( - ocrmypdf(resources / 'masks.pdf', outpdf, tesseract_env=spoof_tesseract_noop) + ocrmypdf.run( + resources / 'masks.pdf', outpdf, tesseract_env=spoof_tesseract_noop + ) == ExitCode.ok ) diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index a870a987..a2bc9f4d 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -17,7 +17,7 @@ import pytest -from ocrmypdf import ocrmypdf as run +import ocrmypdf from ocrmypdf._validation import _pages_from_ranges from ocrmypdf.pdfinfo import PdfInfo @@ -42,7 +42,7 @@ def test_list_range(): def test_limited_pages(resources, outpdf, spoof_tesseract_cache): multi = resources / 'multipage.pdf' - run( + ocrmypdf.run( multi, outpdf, pages='5-6', diff --git a/tests/test_weave.py b/tests/test_weave.py index 5c72a8fc..cd181877 100644 --- a/tests/test_weave.py +++ b/tests/test_weave.py @@ -19,7 +19,7 @@ import os import pytest -from ocrmypdf import ocrmypdf +import ocrmypdf import pikepdf os_environ = pytest.helpers.os_environ @@ -36,14 +36,14 @@ def test_no_glyphless_weave(resources, outdir): env = os.environ.copy() env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2' with os_environ(env): - ocrmypdf( + ocrmypdf.run( outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0 ) @pytest.helpers.needs_pdfminer def test_links(resources, outpdf): - ocrmypdf( + ocrmypdf.run( resources / 'link.pdf', outpdf, redo_ocr=True, oversample=200, output_type='pdf' ) pdf = pikepdf.open(outpdf) From 5ee45411c92536f8aab544d9bd0fa53b740bbd0e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 13 Jun 2019 01:02:07 -0700 Subject: [PATCH 091/880] Decide on OMP_THREAD_LIMIT more intelligently --- src/ocrmypdf/_sync.py | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a9f22294..c61391e7 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -224,7 +224,21 @@ def exec_concurrent(context): # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) if max_workers > 1: - context.log.info("Start processing %d pages concurrent" % max_workers) + context.log.info("Start processing %d pages concurrent", max_workers) + + # Tesseract 4.0 is multithreaded, and we also run multiple workers. We want to + # avoid the situation where we end up trying to run NxN jobs on N CPU cores, + # as that gives poor performance. Performance testing shows we're better off + # parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we + # get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the + # input file is small, then we allow Tesseract to use threads, subject to the + # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers + tess_threads = min(1, context.options.jobs // max_workers) + if context.options.tesseract_env is None: + context.options.tesseract_env = os.environ.copy() + context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads)) + if tess_threads > 1: + context.log.info("Using Tesseract OpenMP thread limit %d", tess_threads) if context.options.use_threads: from multiprocessing.dummy import Pool @@ -300,12 +314,6 @@ def run_pipeline(options, api=False): if not options.jobs: options.jobs = available_cpu_count() - # Performance is improved by setting Tesseract to single threaded. In tests - # this gives better throughput than letting a smaller number of Tesseract - # jobs run multithreaded. Same story for pngquant. Tess <4 ignores this - # variable, but harmless to set if ignored. - os.environ.setdefault('OMP_THREAD_LIMIT', '1') - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") try: check_requested_output_file(options) From 51ed381bfcfb1e34bfa24d2c37a9493aae4d88d6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 13 Jun 2019 01:16:56 -0700 Subject: [PATCH 092/880] Rename weave -> graft --- src/ocrmypdf/{_weave.py => _graft.py} | 6 +++--- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_sync.py | 3 +-- tests/{test_weave.py => test_graft.py} | 2 +- 4 files changed, 6 insertions(+), 7 deletions(-) rename src/ocrmypdf/{_weave.py => _graft.py} (98%) rename tests/{test_weave.py => test_graft.py} (97%) diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_graft.py similarity index 98% rename from src/ocrmypdf/_weave.py rename to src/ocrmypdf/_graft.py index c9f64963..3bdb79ae 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_graft.py @@ -90,7 +90,7 @@ def strip_invisible_text(pdf, page): page.Contents = pikepdf.Stream(pdf, content_stream) -def _weave_layers_graft( +def _graft_text_layer( *, pdf_base, page_num, text, font, font_key, procset, rotation, strip_old_text, log ): """Insert the text layer from text page 0 on to pdf_base at page_num""" @@ -186,7 +186,7 @@ class OcrGrafter: self.font, self.font_key = None, None self.pdfinfo = context.pdfinfo - self.output_file = context.get_path('weave_layers.pdf') + self.output_file = context.get_path('graft_layers.pdf') self.procset = self.pdf_base.make_indirect( pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]') @@ -228,7 +228,7 @@ class OcrGrafter: if text and self.font: # Graft the text layer onto this page, whether new or old strip_old = self.context.options.redo_ocr - _weave_layers_graft( + _graft_text_layer( pdf_base=self.pdf_base, page_num=pageno + 1, text=text, diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e7454a78..d0d5ad89 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -348,7 +348,7 @@ def get_orientation_correction(preview, page_context): correction to rotation. When we draw the real page for OCR, we rotate it by the CCW correction, - which points it (hopefully) upright. _weave.py takes care of the orienting + which points it (hopefully) upright. _graft.py takes care of the orienting the image and text layers. """ diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c61391e7..ffcb403d 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -59,7 +59,7 @@ from ._validation import ( create_input_file, report_output_file_size, ) -from ._weave import OcrGrafter +from ._graft import OcrGrafter from .exceptions import ExitCode, ExitCodeException from .exec import qpdf from .helpers import available_cpu_count @@ -289,7 +289,6 @@ def exec_concurrent(context): copy_final(text, context.options.sidecar, context) # Merge layers to one single pdf - # pdf = weave_layers(layers, context) pdf = ocrgraft.finalize() # PDF/A and metadata diff --git a/tests/test_weave.py b/tests/test_graft.py similarity index 97% rename from tests/test_weave.py rename to tests/test_graft.py index cd181877..c384f1f2 100644 --- a/tests/test_weave.py +++ b/tests/test_graft.py @@ -25,7 +25,7 @@ import pikepdf os_environ = pytest.helpers.os_environ -def test_no_glyphless_weave(resources, outdir): +def test_no_glyphless_graft(resources, outdir): pdf = pikepdf.open(resources / 'francais.pdf') pdf_aspect = pikepdf.open(resources / 'aspect.pdf') pdf_cmyk = pikepdf.open(resources / 'cmyk.pdf') From 9c4b1aeb8d7802af80fac180dbe8a876cf897c5c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 20 Jun 2019 02:44:29 -0700 Subject: [PATCH 093/880] docs: plugin; renaming --- docs/index.rst | 1 + docs/plugins.rst | 22 ++++++++++++++++++++++ src/ocrmypdf/_plugins.py | 28 ++++++++++++++-------------- 3 files changed, 37 insertions(+), 14 deletions(-) create mode 100644 docs/plugins.rst diff --git a/docs/index.rst b/docs/index.rst index 5936e7da..f3a28781 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -28,6 +28,7 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat docker advanced api + plugins batch security errors diff --git a/docs/plugins.rst b/docs/plugins.rst new file mode 100644 index 00000000..b962081b --- /dev/null +++ b/docs/plugins.rst @@ -0,0 +1,22 @@ +Plugins +======= + +You can use plugins to customize the behavior of OCRmyPDF at certain points of interest. + +Currently, it is possible to: +- override the decision for whether or not to perform OCR on a particular file +- modify the image is about to be sent for OCR + +How plugins are imported +------------------------ + +Plugins are imported on demand, by the OCRmyPDF worker process that needs to use them. +As such, plugins cannot share state with each other, and will be imported many times, +once for each worker process. + +Plugins currently cannot override the same hook. + +How plugins are invoked +----------------------- + +Plugins may be called from the command line: diff --git a/src/ocrmypdf/_plugins.py b/src/ocrmypdf/_plugins.py index c4bf5d0a..2a52b105 100644 --- a/src/ocrmypdf/_plugins.py +++ b/src/ocrmypdf/_plugins.py @@ -24,39 +24,39 @@ from pathlib import Path log = logging.getLogger(__name__) -def _load_function_from_module(location): - """Load a function given a module location +def _load_object_from_module(location): + """Load a object given a module location For location=a.b.c, will effectively run "from a.b import c" Example: - _load_function_from_module("a.b.c") + _load_object_from_module("a.b.c") """ module_parts = location.split('.') module_name = '.'.join(module_parts[:-1]) object_name = module_parts[-1] module = importlib.import_module(module_name) - fn = getattr(module, object_name) - log.debug(f"Loaded function: from {module_name} import {object_name}") - return fn + obj = getattr(module, object_name) + log.debug(f"Loaded object: from {module_name} import {object_name}") + return obj -def _load_function_from_pyfile(location): - """Load a function from a file +def _load_object_from_pyfile(location): + """Load a object from a file Example: - _load_function_from_pyfile("test.py::blur_filter") + _load_object_from_pyfile("test.py::blur_filter") """ filename, object_name = location.split('::', maxsplit=1) - log.debug(f"Loading function {object_name} from {filename}") + log.debug(f"Loading object {object_name} from {filename}") module_name = Path(filename).stem spec = importlib.util.spec_from_file_location(module_name, filename) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) - fn = getattr(module, object_name) - return fn + obj = getattr(module, object_name) + return obj def load_plugin(plugin): @@ -67,9 +67,9 @@ def load_plugin(plugin): raise TypeError() if '::' not in plugin: - plugin = _load_function_from_module(plugin) + plugin = _load_object_from_module(plugin) else: - plugin = _load_function_from_pyfile(plugin) + plugin = _load_object_from_pyfile(plugin) return plugin From f47cb2fade111e99817df025755790fc133f1258 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 20 Jun 2019 02:45:14 -0700 Subject: [PATCH 094/880] docs: update ocrmypdf.ocrmypdf to .run --- docs/api.rst | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index 32efd3be..ffc27145 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -23,7 +23,7 @@ A few differences are that ``verbose`` and ``quiet`` are not available. Instead, Parent process requirements ^^^^^^^^^^^^^^^^^^^^^^^^^^^ -The :func:`ocrmypdf.ocrmypdf` function runs OCRmyPDF similar to command line execution. To do this, it will: +The :func:`ocrmypdf.run` function runs OCRmyPDF similar to command line execution. To do this, it will: - create a monitoring thread - create worker processes (forking itself) - manage the signal flags of worker processes @@ -33,14 +33,15 @@ The Python process that calls ``ocrmypdf.run()`` must be sufficiently privileged There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. -Forking a child process to call ``ocrmypdf.run()`` is suggested. That way your application will survive even if OCRmyPDF does not. +Forking a child process to call ``ocrmypdf.run()`` is suggested. That way your application will survive and remain interactive even if OCRmyPDF does not. Logging ^^^^^^^ OCRmyPDF will log under loggers named ``ocrmypdf``. In addition, it imports ``pdfminer`` and ``PIL``, both of which post log messages under those logging namespaces. -You can configure the logging as desired for your application or call :func:`ocrmypdf.configure_logging` to configure logging the same way OCRmyPDF itself does. The command line parameters such as ``--quiet`` and ``--verbose`` have no equivalents in the API; you must configure logging. +You can configure the logging as desired for your application or call :func:`ocrmypdf.configure_logging` to configure logging the same way OCRmyPDF itself does. The command line parameters such as ``--quiet`` and ``--verbose`` have no equivalents in the API; you must use the provided configuration function or do configuration in a +way that suits your use case. Progress monitoring ^^^^^^^^^^^^^^^^^^^ @@ -54,7 +55,7 @@ OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*`` excepti Programs that call OCRmyPDF should consider trapping KeyboardInterrupt so that they allow OCR to terminate with the whole program terminating. -When OCRmyPDF succeeds conditionally, it may return an integer exit code. +When OCRmyPDF succeeds conditionally, it returns an integer exit code. Reference --------- From c357d4146e74d3b5c8174e7042228521e0d00e69 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 20 Jun 2019 03:10:41 -0700 Subject: [PATCH 095/880] Restructure ocrmypdf.pdfinfo --- src/ocrmypdf/pdfinfo/__init__.py | 806 +----------------------------- src/ocrmypdf/pdfinfo/info.py | 821 +++++++++++++++++++++++++++++++ tests/test_pdfinfo.py | 6 +- 3 files changed, 825 insertions(+), 808 deletions(-) create mode 100644 src/ocrmypdf/pdfinfo/info.py diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index ea09ba99..83cf9a48 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -16,808 +16,4 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from collections import namedtuple -from decimal import Decimal -from enum import Enum -import logging -from math import hypot, isclose -from os import fspath -from pathlib import Path -from warnings import warn -import re - -from pikepdf import PdfMatrix -import pikepdf -from tqdm import tqdm - -from . import ghosttext - -from ..exceptions import EncryptedPdfError, MissingDependencyError - - -logger = logging.getLogger() - -Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000') - -Encoding = Enum( - 'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate ' + 'runlength' -) - -FRIENDLY_COLORSPACE = { - '/DeviceGray': Colorspace.gray, - '/CalGray': Colorspace.gray, - '/DeviceRGB': Colorspace.rgb, - '/CalRGB': Colorspace.rgb, - '/DeviceCMYK': Colorspace.cmyk, - '/Lab': Colorspace.lab, - '/ICCBased': Colorspace.icc, - '/Indexed': Colorspace.index, - '/Separation': Colorspace.sep, - '/DeviceN': Colorspace.devn, - '/Pattern': Colorspace.pattern, - '/G': Colorspace.gray, # Abbreviations permitted in inline images - '/RGB': Colorspace.rgb, - '/CMYK': Colorspace.cmyk, - '/I': Colorspace.index, -} - -FRIENDLY_ENCODING = { - '/CCITTFaxDecode': Encoding.ccitt, - '/DCTDecode': Encoding.jpeg, - '/JPXDecode': Encoding.jpeg2000, - '/JBIG2Decode': Encoding.jbig2, - '/CCF': Encoding.ccitt, # Abbreviations permitted in inline images - '/DCT': Encoding.jpeg, - '/AHx': Encoding.asciihex, - '/A85': Encoding.ascii85, - '/LZW': Encoding.lzw, - '/Fl': Encoding.flate, - '/RL': Encoding.runlength, -} - -FRIENDLY_COMP = { - Colorspace.gray: 1, - Colorspace.rgb: 3, - Colorspace.cmyk: 4, - Colorspace.lab: 3, - Colorspace.index: 1, -} - - -UNIT_SQUARE = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0) - - -def _is_unit_square(shorthand): - values = map(float, shorthand) - pairwise = zip(values, UNIT_SQUARE) - return all([isclose(a, b, rel_tol=1e-3) for a, b in pairwise]) - - -XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_depth']) - -InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth']) - -ContentsInfo = namedtuple( - 'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector'] -) - -TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt']) - - -class VectorInfo: - def __init__(self): - pass - - -def _normalize_stack(graphobjs): - """Convert runs of qQ's in the stack into single graphobjs""" - for operands, operator in graphobjs: - operator = str(operator) - if re.match(r'Q*q+$', operator): # Zero or more Q, one or more q - for char in operator: # Split into individual - yield ([], char) # Yield individual - else: - yield (operands, operator) - - -def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): - """Interpret the PDF content stream. - - The stack represents the state of the PDF graphics stack. We are only - interested in the current transformation matrix (CTM) so we only track - this object; a full implementation would need to track many other items. - - The CTM is initialized to the mapping from user space to device space. - PDF units are 1/72". In a PDF viewer or printer this matrix is initialized - to the transformation to device space. For example if set to - (1/72, 0, 0, 1/72, 0, 0) then all units would be calculated in inches. - - Images are always considered to be (0, 0) -> (1, 1). Before drawing an - image there should be a 'cm' that sets up an image coordinate system - where drawing from (0, 0) -> (1, 1) will draw on the desired area of the - page. - - PDF units suit our needs so we initialize ctm to the identity matrix. - - According to the PDF specification, the maximum stack depth is 32. Other - viewers tolerate some amount beyond this. We issue a warning if the - stack depth exceeds the spec limit and set a hard limit beyond this to - bound our memory requirements. If the stack underflows behavior is - undefined in the spec, but we just pretend nothing happened and leave the - CTM unchanged. - """ - - stack = [] - ctm = PdfMatrix(initial_shorthand) - xobject_settings = [] - inline_images = [] - found_vector = False - vector_ops = set('S s f F f* B B* b b*'.split()) - image_ops = set('BI ID EI q Q Do cm'.split()) - operator_whitelist = ' '.join(vector_ops | image_ops) - - for n, graphobj in enumerate( - _normalize_stack( - pikepdf.parse_content_stream(contentstream, operator_whitelist) - ) - ): - operands, operator = graphobj - if operator == 'q': - stack.append(ctm) - if len(stack) > 32: # See docstring - if len(stack) > 128: - raise RuntimeError( - "PDF graphics stack overflowed hard limit, operator %i" % n - ) - warn("PDF graphics stack overflowed spec limit") - elif operator == 'Q': - try: - ctm = stack.pop() - except IndexError: - # Keeping the ctm the same seems to be the only sensible thing - # to do. Just pretend nothing happened, keep calm and carry on. - warn("PDF graphics stack underflowed - PDF may be malformed") - elif operator == 'cm': - ctm = PdfMatrix(operands) @ ctm - elif operator == 'Do': - image_name = operands[0] - settings = XobjectSettings( - name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack) - ) - xobject_settings.append(settings) - elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this - iimage = operands[0] - inline = InlineSettings( - iimage=iimage, shorthand=ctm.shorthand, stack_depth=len(stack) - ) - inline_images.append(inline) - elif operator in vector_ops: - found_vector = True - - return ContentsInfo( - xobject_settings=xobject_settings, - inline_images=inline_images, - found_vector=found_vector, - ) - - -def _get_dpi(ctm_shorthand, image_size): - """Given the transformation matrix and image size, find the image DPI. - - PDFs do not include image resolution information within image data. - Instead, the PDF page content stream describes the location where the - image will be rasterized, and the effective resolution is the ratio of the - pixel size to raster target size. - - Normally a scanned PDF has the paper size set appropriately but this is - not guaranteed. The most common case is a cropped image will change the - page size (/CropBox) without altering the page content stream. That means - it is not sufficient to assume that the image fills the page, even though - that is the most common case. - - A PDF image may be scaled (always), cropped, translated, rotated in place - to an arbitrary angle (rarely) and skewed. Only equal area mappings can - be expressed, that is, it is not necessary to consider distortions where - the effective DPI varies with position. - - To determine the image scale, transform an offset axis vector v0 (0, 0), - width-axis vector v0 (1, 0), height-axis vector vh (0, 1) with the matrix, - which gives the dimensions of the image in PDF units. From there we can - compare to actual image dimensions. PDF uses - row vector * matrix_tranposed unlike the traditional - matrix * column vector. - - The offset, width and height vectors can be combined in a matrix and - multiplied by the transform matrix. Then we want to calculated - magnitude(width_vector - offset_vector) - and - magnitude(height_vector - offset_vector) - - When the above is worked out algebraically, the effect of translation - cancels out, and the vector magnitudes become functions of the nonzero - transformation matrix indices. The results of the derivation are used - in this code. - - pdfimages -list does calculate the DPI in some way that is not completely - naive, but it does not get the DPI of rotated images right, so cannot be - used anymore to validate this. Photoshop works, or using Acrobat to - rotate the image back to normal. - - It does not matter if the image is partially cropped, or even out of the - /MediaBox. - - """ - - a, b, c, d, _, _ = ctm_shorthand - - # Calculate the width and height of the image in PDF units - image_drawn_width = hypot(a, b) - image_drawn_height = hypot(c, d) - - # The scale of the image is pixels per unit of default user space (1/72") - scale_w = image_size[0] / image_drawn_width - scale_h = image_size[1] / image_drawn_height - - # DPI = scale * 72 - dpi_w = scale_w * 72.0 - dpi_h = scale_h * 72.0 - - return dpi_w, dpi_h - - -class ImageInfo: - DPI_PREC = Decimal('1.000') - - def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None): - - self._name = str(name) - self._shorthand = shorthand - - if inline is not None: - self._origin = 'inline' - pim = inline.iimage - elif pdfimage is not None: - self._origin = 'xobject' - pim = pikepdf.PdfImage(pdfimage) - self._width = pim.width - self._height = pim.height - - # If /ImageMask is true, then this image is a stencil mask - # (Images that draw with this stencil mask will have a reference to - # it in their /Mask, but we don't actually need that information) - if pim.image_mask: - self._type = 'stencil' - else: - self._type = 'image' - - self._bpc = int(pim.bits_per_component) - try: - self._enc = FRIENDLY_ENCODING.get(pim.filters[0], 'image') - except IndexError: - self._enc = '?' - - try: - self._color = FRIENDLY_COLORSPACE.get(pim.colorspace, '?') - except NotImplementedError: - self._color = '?' - if self._enc == Encoding.jpeg2000: - self._color = Colorspace.jpeg2000 - - self._comp = FRIENDLY_COMP.get(self._color, '?') - - # Bit of a hack... infer grayscale if component count is uncertain - # but encoding must be monochrome. This happens if a monochrome image - # has an ICC profile attached. Better solution would be to examine - # the ICC profile. - if self._comp == '?' and self._enc in (Encoding.ccitt, 'jbig2'): - self._comp = FRIENDLY_COMP[Colorspace.gray] - - @property - def name(self): - return self._name - - @property - def type_(self): - return self._type - - @property - def width(self): - return self._width - - @property - def height(self): - return self._height - - @property - def bpc(self): - return self._bpc - - @property - def color(self): - return self._color - - @property - def comp(self): - return self._comp - - @property - def enc(self): - return self._enc - - @property - def xres(self): - return _get_dpi(self._shorthand, (self._width, self._height))[0] - - @property - def yres(self): - return _get_dpi(self._shorthand, (self._width, self._height))[1] - - def __repr__(self): - class_locals = { - attr: getattr(self, attr, None) - for attr in dir(self) - if not attr.startswith('_') - } - return ( - "" - ).format(**class_locals) - - -def _find_inline_images(contentsinfo): - "Find inline images in the contentstream" - - for n, inline in enumerate(contentsinfo.inline_images): - yield ImageInfo( - name='inline-%02d' % n, shorthand=inline.shorthand, inline=inline - ) - - -def _image_xobjects(container): - """Search for all XObject-based images in the container - - Usually the container is a page, but it could also be a Form XObject - that contains images. Filter out the Form XObjects which are dealt with - elsewhere. - - Generate a sequence of tuples (image, xobj container), where container, - where xobj is the name of the object and image is the object itself, - since the object does not know its own name. - - """ - - if '/Resources' not in container: - return - resources = container['/Resources'] - if '/XObject' not in resources: - return - xobjs = resources['/XObject'].as_dict() - for xobj in xobjs: - candidate = xobjs[xobj] - if not '/Subtype' in candidate: - continue - if candidate['/Subtype'] == '/Image': - pdfimage = candidate - yield (pdfimage, xobj) - - -def _find_regular_images(container, contentsinfo): - """Find images stored in the container's /Resources /XObject - - Usually the container is a page, but it could also be a Form XObject - that contains images. - - Generates images with their DPI at time of drawing. - """ - - for pdfimage, xobj in _image_xobjects(container): - - # For each image that is drawn on this, check if we drawing the - # current image - yes this is O(n^2), but n == 1 almost always - for draw in contentsinfo.xobject_settings: - if draw.name != xobj: - continue - - if draw.stack_depth == 0 and _is_unit_square(draw.shorthand): - # At least one PDF in the wild (and test suite) draws an image - # when the graphics stack depth is 0, meaning that the image - # gets drawn into a square of 1x1 PDF units (or 1/72", - # or 0.35 mm). The equivalent DPI will be >100,000. Exclude - # these from our DPI calculation for the page. - continue - - yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand) - - -def _find_form_xobject_images(pdf, container, contentsinfo): - """Find any images that are in Form XObjects in the container - - The container may be a page, or a parent Form XObject. - - """ - if '/Resources' not in container: - return - resources = container['/Resources'] - if '/XObject' not in resources: - return - xobjs = resources['/XObject'].as_dict() - for xobj in xobjs: - candidate = xobjs[xobj] - if candidate['/Subtype'] != '/Form': - continue - - form_xobject = candidate - for settings in contentsinfo.xobject_settings: - if settings.name != xobj: - continue - - # Find images once for each time this Form XObject is drawn. - # This could be optimized to cache the multiple drawing events - # but in practice both Form XObjects and multiple drawing of the - # same object are both very rare. - ctm_shorthand = settings.shorthand - yield from _process_content_streams( - pdf=pdf, container=form_xobject, shorthand=ctm_shorthand - ) - - -def _process_content_streams(*, pdf, container, shorthand=None): - """Find all individual instances of images drawn in the container - - Usually the container is a page, but it may also be a Form XObject. - - On a typical page images are stored inline or as regular images - in an XObject. - - Form XObjects may include inline images, XObject images, - and recursively, other Form XObjects; and also vector graphic objects. - - Every instance of an image being drawn somewhere is flattened and - treated as a unique image, since if the same image is drawn multiple times - on one page it may be drawn at differing resolutions, and our objective - is to find the resolution at which the page can be rastered without - downsampling. - - """ - - if container.get('/Type') == '/Page' and '/Contents' in container: - initial_shorthand = shorthand or UNIT_SQUARE - elif container.get('/Type') == '/XObject' and container['/Subtype'] == '/Form': - # Set the CTM to the state it was when the "Do" operator was - # encountered that is drawing this instance of the Form XObject - ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity() - - # A Form XObject may provide its own matrix to map form space into - # user space. Get this if one exists - form_shorthand = container.get('/Matrix', PdfMatrix.identity()) - form_matrix = PdfMatrix(form_shorthand) - - # Concatenate form matrix with CTM to ensure CTM is correct for - # drawing this instance of the XObject - ctm = form_matrix @ ctm - initial_shorthand = ctm.shorthand - else: - return - - contentsinfo = _interpret_contents(container, initial_shorthand) - - if contentsinfo.found_vector: - yield VectorInfo() - yield from _find_inline_images(contentsinfo) - yield from _find_regular_images(container, contentsinfo) - yield from _find_form_xobject_images(pdf, container, contentsinfo) - - -def _page_has_text(text_blocks, page_width, page_height): - """Smarter text detection that ignores text in margins""" - - pw, ph = float(page_width), float(page_height) - - margin_ratio = 0.125 - interior_bbox = ( - margin_ratio * pw, # left - (1 - margin_ratio) * ph, # top - (1 - margin_ratio) * pw, # right - margin_ratio * ph, # bottom (first quadrant: bottom < top) - ) - - def rects_intersect(a, b): - """ - Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3) - https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other - Formula assumes all boxes are in first quadrant - """ - return a[0] < b[2] and a[2] > b[0] and a[1] > b[3] and a[3] < b[1] - - has_text = False - for bbox in text_blocks: - if rects_intersect(bbox, interior_bbox): - has_text = True - break - return has_text - - -def simplify_textboxes(miner, textbox_getter): - """Extract only limited content from text boxes - - We do this to save memory and ensure that our objects are pickleable. - """ - for box in textbox_getter(miner): - first_line = box._objs[0] - first_char = first_line._objs[0] - - visible = first_char.rendermode != 3 - corrupt = first_char.get_text() == '\ufffd' - yield TextboxInfo(box.bbox, visible, corrupt) - - -def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): - pageinfo = {} - pageinfo['pageno'] = pageno - pageinfo['images'] = [] - - page = pdf.pages[pageno] - mediabox = [Decimal(d) for d in page.MediaBox.as_list()] - width_pt = mediabox[2] - mediabox[0] - height_pt = mediabox[3] - mediabox[1] - - if xmltext is not None: - bboxes = ghosttext.page_get_textblocks( - fspath(infile), pageno, xmltext=xmltext, height=height_pt - ) - pageinfo['bboxes'] = bboxes - else: - # pdfminer required for this section - try: - from .layout import get_page_analysis, get_text_boxes - except ImportError: - raise MissingDependencyError( - "pdfminer is required for this feature. Your distribution " - "may not have installed it." - ) - pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') - miner = get_page_analysis(infile, pageno, pscript5_mode) - pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) - bboxes = (box.bbox for box in pageinfo['textboxes']) - - pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) - - userunit = page.get('/UserUnit', Decimal(1.0)) - if not isinstance(userunit, Decimal): - userunit = Decimal(userunit) - pageinfo['userunit'] = userunit - pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0) - pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0) - - try: - pageinfo['rotate'] = int(page['/Rotate']) - except KeyError: - pageinfo['rotate'] = 0 - - userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) - contentsinfo = [ - ci - for ci in _process_content_streams( - pdf=pdf, container=page, shorthand=userunit_shorthand - ) - ] - - pageinfo['has_vector'] = False - if any(isinstance(ci, VectorInfo) for ci in contentsinfo): - pageinfo['has_vector'] = True - - pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] - if pageinfo['images']: - xres = Decimal(max(image.xres for image in pageinfo['images'])) - yres = Decimal(max(image.yres for image in pageinfo['images'])) - pageinfo['xres'], pageinfo['yres'] = xres, yres - pageinfo['width_pixels'] = int(round(xres * pageinfo['width_inches'])) - pageinfo['height_pixels'] = int(round(yres * pageinfo['height_inches'])) - - return pageinfo - - -def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False): - pdf = pikepdf.open(infile) # Do not close in this function - if pdf.is_encrypted: - pdf.close() - raise EncryptedPdfError() # Triggered by encryption with empty passwd - if detailed_analysis: - pages_xml = None - else: - pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) - - pages = [] - for n, _ in tqdm( - enumerate(pdf.pages), - total=len(pdf.pages), - desc="Scan", - unit='page', - disable=not progbar, - ): - page_xml = pages_xml[n] if pages_xml else None - page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) - pages.append(page) - - return pages, pdf - - -class PageInfo: - def __init__(self, pdf, pageno, infile, xmltext, detailed_analysis=False): - self._pageno = pageno - self._infile = infile - self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, xmltext) - self._detailed_analysis = detailed_analysis - - @property - def pageno(self): - return self._pageno - - @property - def has_text(self): - return self._pageinfo['has_text'] - - @property - def has_corrupt_text(self): - if not self._detailed_analysis: - raise NotImplementedError('Did not do detailed analysis') - return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) - - @property - def has_vector(self): - return self._pageinfo['has_vector'] - - @property - def width_inches(self): - return self._pageinfo['width_inches'] - - @property - def height_inches(self): - return self._pageinfo['height_inches'] - - @property - def width_pixels(self): - return int(round(self.width_inches * self.xres)) - - @property - def height_pixels(self): - return int(round(self.height_inches * self.yres)) - - @property - def rotation(self): - return self._pageinfo.get('rotate', None) - - @rotation.setter - def rotation(self, value): - if value in (0, 90, 180, 270, 360, -90, -180, -270): - self._pageinfo['rotate'] = value - else: - raise ValueError("rotation must be a cardinal angle") - - @property - def images(self): - return self._pageinfo['images'] - - def get_textareas(self, visible=None, corrupt=None): - def predicate(obj, want_visible, want_corrupt): - result = True - if want_visible is not None: - if obj.is_visible != want_visible: - result = False - if want_corrupt is not None: - if obj.is_corrupt != want_corrupt: - result = False - return result - - if 'textboxes' not in self._pageinfo: - if visible is not None and corrupt is not None: - raise NotImplementedError('Ghostscript textboxes cannot be classified') - return self._pageinfo['bboxes'] - - return ( - obj.bbox - for obj in self._pageinfo['textboxes'] - if predicate(obj, visible, corrupt) - ) - - @property - def xres(self): - return self._pageinfo.get('xres', None) - - @property - def yres(self): - return self._pageinfo.get('yres', None) - - @property - def userunit(self): - return self._pageinfo.get('userunit', None) - - @property - def min_version(self): - if self.userunit is not None: - return '1.6' - else: - return '1.5' - - def __repr__(self): - return ( - '' - ).format( - self.pageno, - self.width_inches, - self.height_inches, - self.rotation, - self.xres, - self.yres, - self.has_text, - ) - - -class PdfInfo: - """Get summary information about a PDF""" - - def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False): - self._infile = infile - self._pages, pdf = _pdf_get_all_pageinfo( - infile, detailed_page_analysis, log=log, progbar=progbar - ) - self._needs_rendering = pdf.root.get('/NeedsRendering', False) - self._has_acroform = False - if '/AcroForm' in pdf.root: - if len(pdf.root.AcroForm.get('/Fields', [])) > 0: - self._has_acroform = True - elif '/XFA' in pdf.root.AcroForm: - self._has_acroform = True - pdf.close() - - @property - def pages(self): - return self._pages - - @property - def min_version(self): - # The minimum PDF is the maximum version that any particular page needs - return max(page.min_version for page in self.pages) - - @property - def has_userunit(self): - return any(page.userunit != 1.0 for page in self.pages) - - @property - def has_acroform(self): - return self._has_acroform - - @property - def filename(self): - if not isinstance(self._infile, (str, Path)): - raise NotImplementedError("can't get filename from stream") - return self._infile - - @property - def needs_rendering(self): - return self._needs_rendering - - def __getitem__(self, item): - return self._pages[item] - - def __len__(self): - return len(self._pages) - - def __repr__(self): - return f"" - - -def main(): - import argparse - - parser = argparse.ArgumentParser() - parser.add_argument('infile') - args = parser.parse_args() - info = _pdf_get_all_pageinfo(args.infile) - from pprint import pprint - - pprint(info) - - -if __name__ == '__main__': - main() +from .info import PdfInfo, Colorspace, Encoding diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py new file mode 100644 index 00000000..8e29c44d --- /dev/null +++ b/src/ocrmypdf/pdfinfo/info.py @@ -0,0 +1,821 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from collections import namedtuple +from decimal import Decimal +from enum import Enum +import logging +from math import hypot, isclose +from os import fspath +from pathlib import Path +from warnings import warn +import re + +from pikepdf import PdfMatrix +import pikepdf +from tqdm import tqdm + +from . import ghosttext +from ocrmypdf.exceptions import EncryptedPdfError, MissingDependencyError + +logger = logging.getLogger() + +Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000') + +Encoding = Enum( + 'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate ' + 'runlength' +) + +FRIENDLY_COLORSPACE = { + '/DeviceGray': Colorspace.gray, + '/CalGray': Colorspace.gray, + '/DeviceRGB': Colorspace.rgb, + '/CalRGB': Colorspace.rgb, + '/DeviceCMYK': Colorspace.cmyk, + '/Lab': Colorspace.lab, + '/ICCBased': Colorspace.icc, + '/Indexed': Colorspace.index, + '/Separation': Colorspace.sep, + '/DeviceN': Colorspace.devn, + '/Pattern': Colorspace.pattern, + '/G': Colorspace.gray, # Abbreviations permitted in inline images + '/RGB': Colorspace.rgb, + '/CMYK': Colorspace.cmyk, + '/I': Colorspace.index, +} + +FRIENDLY_ENCODING = { + '/CCITTFaxDecode': Encoding.ccitt, + '/DCTDecode': Encoding.jpeg, + '/JPXDecode': Encoding.jpeg2000, + '/JBIG2Decode': Encoding.jbig2, + '/CCF': Encoding.ccitt, # Abbreviations permitted in inline images + '/DCT': Encoding.jpeg, + '/AHx': Encoding.asciihex, + '/A85': Encoding.ascii85, + '/LZW': Encoding.lzw, + '/Fl': Encoding.flate, + '/RL': Encoding.runlength, +} + +FRIENDLY_COMP = { + Colorspace.gray: 1, + Colorspace.rgb: 3, + Colorspace.cmyk: 4, + Colorspace.lab: 3, + Colorspace.index: 1, +} + + +UNIT_SQUARE = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0) + + +def _is_unit_square(shorthand): + values = map(float, shorthand) + pairwise = zip(values, UNIT_SQUARE) + return all([isclose(a, b, rel_tol=1e-3) for a, b in pairwise]) + + +XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_depth']) + +InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth']) + +ContentsInfo = namedtuple( + 'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector'] +) + +TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt']) + + +class VectorInfo: + def __init__(self): + pass + + +def _normalize_stack(graphobjs): + """Convert runs of qQ's in the stack into single graphobjs""" + for operands, operator in graphobjs: + operator = str(operator) + if re.match(r'Q*q+$', operator): # Zero or more Q, one or more q + for char in operator: # Split into individual + yield ([], char) # Yield individual + else: + yield (operands, operator) + + +def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): + """Interpret the PDF content stream. + + The stack represents the state of the PDF graphics stack. We are only + interested in the current transformation matrix (CTM) so we only track + this object; a full implementation would need to track many other items. + + The CTM is initialized to the mapping from user space to device space. + PDF units are 1/72". In a PDF viewer or printer this matrix is initialized + to the transformation to device space. For example if set to + (1/72, 0, 0, 1/72, 0, 0) then all units would be calculated in inches. + + Images are always considered to be (0, 0) -> (1, 1). Before drawing an + image there should be a 'cm' that sets up an image coordinate system + where drawing from (0, 0) -> (1, 1) will draw on the desired area of the + page. + + PDF units suit our needs so we initialize ctm to the identity matrix. + + According to the PDF specification, the maximum stack depth is 32. Other + viewers tolerate some amount beyond this. We issue a warning if the + stack depth exceeds the spec limit and set a hard limit beyond this to + bound our memory requirements. If the stack underflows behavior is + undefined in the spec, but we just pretend nothing happened and leave the + CTM unchanged. + """ + + stack = [] + ctm = PdfMatrix(initial_shorthand) + xobject_settings = [] + inline_images = [] + found_vector = False + vector_ops = set('S s f F f* B B* b b*'.split()) + image_ops = set('BI ID EI q Q Do cm'.split()) + operator_whitelist = ' '.join(vector_ops | image_ops) + + for n, graphobj in enumerate( + _normalize_stack( + pikepdf.parse_content_stream(contentstream, operator_whitelist) + ) + ): + operands, operator = graphobj + if operator == 'q': + stack.append(ctm) + if len(stack) > 32: # See docstring + if len(stack) > 128: + raise RuntimeError( + "PDF graphics stack overflowed hard limit, operator %i" % n + ) + warn("PDF graphics stack overflowed spec limit") + elif operator == 'Q': + try: + ctm = stack.pop() + except IndexError: + # Keeping the ctm the same seems to be the only sensible thing + # to do. Just pretend nothing happened, keep calm and carry on. + warn("PDF graphics stack underflowed - PDF may be malformed") + elif operator == 'cm': + ctm = PdfMatrix(operands) @ ctm + elif operator == 'Do': + image_name = operands[0] + settings = XobjectSettings( + name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack) + ) + xobject_settings.append(settings) + elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this + iimage = operands[0] + inline = InlineSettings( + iimage=iimage, shorthand=ctm.shorthand, stack_depth=len(stack) + ) + inline_images.append(inline) + elif operator in vector_ops: + found_vector = True + + return ContentsInfo( + xobject_settings=xobject_settings, + inline_images=inline_images, + found_vector=found_vector, + ) + + +def _get_dpi(ctm_shorthand, image_size): + """Given the transformation matrix and image size, find the image DPI. + + PDFs do not include image resolution information within image data. + Instead, the PDF page content stream describes the location where the + image will be rasterized, and the effective resolution is the ratio of the + pixel size to raster target size. + + Normally a scanned PDF has the paper size set appropriately but this is + not guaranteed. The most common case is a cropped image will change the + page size (/CropBox) without altering the page content stream. That means + it is not sufficient to assume that the image fills the page, even though + that is the most common case. + + A PDF image may be scaled (always), cropped, translated, rotated in place + to an arbitrary angle (rarely) and skewed. Only equal area mappings can + be expressed, that is, it is not necessary to consider distortions where + the effective DPI varies with position. + + To determine the image scale, transform an offset axis vector v0 (0, 0), + width-axis vector v0 (1, 0), height-axis vector vh (0, 1) with the matrix, + which gives the dimensions of the image in PDF units. From there we can + compare to actual image dimensions. PDF uses + row vector * matrix_tranposed unlike the traditional + matrix * column vector. + + The offset, width and height vectors can be combined in a matrix and + multiplied by the transform matrix. Then we want to calculated + magnitude(width_vector - offset_vector) + and + magnitude(height_vector - offset_vector) + + When the above is worked out algebraically, the effect of translation + cancels out, and the vector magnitudes become functions of the nonzero + transformation matrix indices. The results of the derivation are used + in this code. + + pdfimages -list does calculate the DPI in some way that is not completely + naive, but it does not get the DPI of rotated images right, so cannot be + used anymore to validate this. Photoshop works, or using Acrobat to + rotate the image back to normal. + + It does not matter if the image is partially cropped, or even out of the + /MediaBox. + + """ + + a, b, c, d, _, _ = ctm_shorthand + + # Calculate the width and height of the image in PDF units + image_drawn_width = hypot(a, b) + image_drawn_height = hypot(c, d) + + # The scale of the image is pixels per unit of default user space (1/72") + scale_w = image_size[0] / image_drawn_width + scale_h = image_size[1] / image_drawn_height + + # DPI = scale * 72 + dpi_w = scale_w * 72.0 + dpi_h = scale_h * 72.0 + + return dpi_w, dpi_h + + +class ImageInfo: + DPI_PREC = Decimal('1.000') + + def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None): + + self._name = str(name) + self._shorthand = shorthand + + if inline is not None: + self._origin = 'inline' + pim = inline.iimage + elif pdfimage is not None: + self._origin = 'xobject' + pim = pikepdf.PdfImage(pdfimage) + self._width = pim.width + self._height = pim.height + + # If /ImageMask is true, then this image is a stencil mask + # (Images that draw with this stencil mask will have a reference to + # it in their /Mask, but we don't actually need that information) + if pim.image_mask: + self._type = 'stencil' + else: + self._type = 'image' + + self._bpc = int(pim.bits_per_component) + try: + self._enc = FRIENDLY_ENCODING.get(pim.filters[0], 'image') + except IndexError: + self._enc = '?' + + try: + self._color = FRIENDLY_COLORSPACE.get(pim.colorspace, '?') + except NotImplementedError: + self._color = '?' + if self._enc == Encoding.jpeg2000: + self._color = Colorspace.jpeg2000 + + self._comp = FRIENDLY_COMP.get(self._color, '?') + + # Bit of a hack... infer grayscale if component count is uncertain + # but encoding must be monochrome. This happens if a monochrome image + # has an ICC profile attached. Better solution would be to examine + # the ICC profile. + if self._comp == '?' and self._enc in (Encoding.ccitt, 'jbig2'): + self._comp = FRIENDLY_COMP[Colorspace.gray] + + @property + def name(self): + return self._name + + @property + def type_(self): + return self._type + + @property + def width(self): + return self._width + + @property + def height(self): + return self._height + + @property + def bpc(self): + return self._bpc + + @property + def color(self): + return self._color + + @property + def comp(self): + return self._comp + + @property + def enc(self): + return self._enc + + @property + def xres(self): + return _get_dpi(self._shorthand, (self._width, self._height))[0] + + @property + def yres(self): + return _get_dpi(self._shorthand, (self._width, self._height))[1] + + def __repr__(self): + class_locals = { + attr: getattr(self, attr, None) + for attr in dir(self) + if not attr.startswith('_') + } + return ( + "" + ).format(**class_locals) + + +def _find_inline_images(contentsinfo): + "Find inline images in the contentstream" + + for n, inline in enumerate(contentsinfo.inline_images): + yield ImageInfo( + name='inline-%02d' % n, shorthand=inline.shorthand, inline=inline + ) + + +def _image_xobjects(container): + """Search for all XObject-based images in the container + + Usually the container is a page, but it could also be a Form XObject + that contains images. Filter out the Form XObjects which are dealt with + elsewhere. + + Generate a sequence of tuples (image, xobj container), where container, + where xobj is the name of the object and image is the object itself, + since the object does not know its own name. + + """ + + if '/Resources' not in container: + return + resources = container['/Resources'] + if '/XObject' not in resources: + return + xobjs = resources['/XObject'].as_dict() + for xobj in xobjs: + candidate = xobjs[xobj] + if not '/Subtype' in candidate: + continue + if candidate['/Subtype'] == '/Image': + pdfimage = candidate + yield (pdfimage, xobj) + + +def _find_regular_images(container, contentsinfo): + """Find images stored in the container's /Resources /XObject + + Usually the container is a page, but it could also be a Form XObject + that contains images. + + Generates images with their DPI at time of drawing. + """ + + for pdfimage, xobj in _image_xobjects(container): + + # For each image that is drawn on this, check if we drawing the + # current image - yes this is O(n^2), but n == 1 almost always + for draw in contentsinfo.xobject_settings: + if draw.name != xobj: + continue + + if draw.stack_depth == 0 and _is_unit_square(draw.shorthand): + # At least one PDF in the wild (and test suite) draws an image + # when the graphics stack depth is 0, meaning that the image + # gets drawn into a square of 1x1 PDF units (or 1/72", + # or 0.35 mm). The equivalent DPI will be >100,000. Exclude + # these from our DPI calculation for the page. + continue + + yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand) + + +def _find_form_xobject_images(pdf, container, contentsinfo): + """Find any images that are in Form XObjects in the container + + The container may be a page, or a parent Form XObject. + + """ + if '/Resources' not in container: + return + resources = container['/Resources'] + if '/XObject' not in resources: + return + xobjs = resources['/XObject'].as_dict() + for xobj in xobjs: + candidate = xobjs[xobj] + if candidate['/Subtype'] != '/Form': + continue + + form_xobject = candidate + for settings in contentsinfo.xobject_settings: + if settings.name != xobj: + continue + + # Find images once for each time this Form XObject is drawn. + # This could be optimized to cache the multiple drawing events + # but in practice both Form XObjects and multiple drawing of the + # same object are both very rare. + ctm_shorthand = settings.shorthand + yield from _process_content_streams( + pdf=pdf, container=form_xobject, shorthand=ctm_shorthand + ) + + +def _process_content_streams(*, pdf, container, shorthand=None): + """Find all individual instances of images drawn in the container + + Usually the container is a page, but it may also be a Form XObject. + + On a typical page images are stored inline or as regular images + in an XObject. + + Form XObjects may include inline images, XObject images, + and recursively, other Form XObjects; and also vector graphic objects. + + Every instance of an image being drawn somewhere is flattened and + treated as a unique image, since if the same image is drawn multiple times + on one page it may be drawn at differing resolutions, and our objective + is to find the resolution at which the page can be rastered without + downsampling. + + """ + + if container.get('/Type') == '/Page' and '/Contents' in container: + initial_shorthand = shorthand or UNIT_SQUARE + elif container.get('/Type') == '/XObject' and container['/Subtype'] == '/Form': + # Set the CTM to the state it was when the "Do" operator was + # encountered that is drawing this instance of the Form XObject + ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity() + + # A Form XObject may provide its own matrix to map form space into + # user space. Get this if one exists + form_shorthand = container.get('/Matrix', PdfMatrix.identity()) + form_matrix = PdfMatrix(form_shorthand) + + # Concatenate form matrix with CTM to ensure CTM is correct for + # drawing this instance of the XObject + ctm = form_matrix @ ctm + initial_shorthand = ctm.shorthand + else: + return + + contentsinfo = _interpret_contents(container, initial_shorthand) + + if contentsinfo.found_vector: + yield VectorInfo() + yield from _find_inline_images(contentsinfo) + yield from _find_regular_images(container, contentsinfo) + yield from _find_form_xobject_images(pdf, container, contentsinfo) + + +def _page_has_text(text_blocks, page_width, page_height): + """Smarter text detection that ignores text in margins""" + + pw, ph = float(page_width), float(page_height) + + margin_ratio = 0.125 + interior_bbox = ( + margin_ratio * pw, # left + (1 - margin_ratio) * ph, # top + (1 - margin_ratio) * pw, # right + margin_ratio * ph, # bottom (first quadrant: bottom < top) + ) + + def rects_intersect(a, b): + """ + Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3) + https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other + Formula assumes all boxes are in first quadrant + """ + return a[0] < b[2] and a[2] > b[0] and a[1] > b[3] and a[3] < b[1] + + has_text = False + for bbox in text_blocks: + if rects_intersect(bbox, interior_bbox): + has_text = True + break + return has_text + + +def simplify_textboxes(miner, textbox_getter): + """Extract only limited content from text boxes + + We do this to save memory and ensure that our objects are pickleable. + """ + for box in textbox_getter(miner): + first_line = box._objs[0] + first_char = first_line._objs[0] + + visible = first_char.rendermode != 3 + corrupt = first_char.get_text() == '\ufffd' + yield TextboxInfo(box.bbox, visible, corrupt) + + +def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): + pageinfo = {} + pageinfo['pageno'] = pageno + pageinfo['images'] = [] + + page = pdf.pages[pageno] + mediabox = [Decimal(d) for d in page.MediaBox.as_list()] + width_pt = mediabox[2] - mediabox[0] + height_pt = mediabox[3] - mediabox[1] + + if xmltext is not None: + bboxes = ghosttext.page_get_textblocks( + fspath(infile), pageno, xmltext=xmltext, height=height_pt + ) + pageinfo['bboxes'] = bboxes + else: + # pdfminer required for this section + try: + from .layout import get_page_analysis, get_text_boxes + except ImportError: + raise MissingDependencyError( + "pdfminer is required for this feature. Your distribution " + "may not have installed it." + ) + pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') + miner = get_page_analysis(infile, pageno, pscript5_mode) + pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) + bboxes = (box.bbox for box in pageinfo['textboxes']) + + pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) + + userunit = page.get('/UserUnit', Decimal(1.0)) + if not isinstance(userunit, Decimal): + userunit = Decimal(userunit) + pageinfo['userunit'] = userunit + pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0) + pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0) + + try: + pageinfo['rotate'] = int(page['/Rotate']) + except KeyError: + pageinfo['rotate'] = 0 + + userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) + contentsinfo = [ + ci + for ci in _process_content_streams( + pdf=pdf, container=page, shorthand=userunit_shorthand + ) + ] + + pageinfo['has_vector'] = False + if any(isinstance(ci, VectorInfo) for ci in contentsinfo): + pageinfo['has_vector'] = True + + pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] + if pageinfo['images']: + xres = Decimal(max(image.xres for image in pageinfo['images'])) + yres = Decimal(max(image.yres for image in pageinfo['images'])) + pageinfo['xres'], pageinfo['yres'] = xres, yres + pageinfo['width_pixels'] = int(round(xres * pageinfo['width_inches'])) + pageinfo['height_pixels'] = int(round(yres * pageinfo['height_inches'])) + + return pageinfo + + +def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False): + pdf = pikepdf.open(infile) # Do not close in this function + if pdf.is_encrypted: + pdf.close() + raise EncryptedPdfError() # Triggered by encryption with empty passwd + if detailed_analysis: + pages_xml = None + else: + pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) + + pages = [] + for n, _ in tqdm( + enumerate(pdf.pages), + total=len(pdf.pages), + desc="Scan", + unit='page', + disable=not progbar, + ): + page_xml = pages_xml[n] if pages_xml else None + page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) + pages.append(page) + + return pages, pdf + + +class PageInfo: + def __init__(self, pdf, pageno, infile, xmltext, detailed_analysis=False): + self._pageno = pageno + self._infile = infile + self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, xmltext) + self._detailed_analysis = detailed_analysis + + @property + def pageno(self): + return self._pageno + + @property + def has_text(self): + return self._pageinfo['has_text'] + + @property + def has_corrupt_text(self): + if not self._detailed_analysis: + raise NotImplementedError('Did not do detailed analysis') + return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) + + @property + def has_vector(self): + return self._pageinfo['has_vector'] + + @property + def width_inches(self): + return self._pageinfo['width_inches'] + + @property + def height_inches(self): + return self._pageinfo['height_inches'] + + @property + def width_pixels(self): + return int(round(self.width_inches * self.xres)) + + @property + def height_pixels(self): + return int(round(self.height_inches * self.yres)) + + @property + def rotation(self): + return self._pageinfo.get('rotate', None) + + @rotation.setter + def rotation(self, value): + if value in (0, 90, 180, 270, 360, -90, -180, -270): + self._pageinfo['rotate'] = value + else: + raise ValueError("rotation must be a cardinal angle") + + @property + def images(self): + return self._pageinfo['images'] + + def get_textareas(self, visible=None, corrupt=None): + def predicate(obj, want_visible, want_corrupt): + result = True + if want_visible is not None: + if obj.is_visible != want_visible: + result = False + if want_corrupt is not None: + if obj.is_corrupt != want_corrupt: + result = False + return result + + if 'textboxes' not in self._pageinfo: + if visible is not None and corrupt is not None: + raise NotImplementedError('Ghostscript textboxes cannot be classified') + return self._pageinfo['bboxes'] + + return ( + obj.bbox + for obj in self._pageinfo['textboxes'] + if predicate(obj, visible, corrupt) + ) + + @property + def xres(self): + return self._pageinfo.get('xres', None) + + @property + def yres(self): + return self._pageinfo.get('yres', None) + + @property + def userunit(self): + return self._pageinfo.get('userunit', None) + + @property + def min_version(self): + if self.userunit is not None: + return '1.6' + else: + return '1.5' + + def __repr__(self): + return ( + '' + ).format( + self.pageno, + self.width_inches, + self.height_inches, + self.rotation, + self.xres, + self.yres, + self.has_text, + ) + + +class PdfInfo: + """Get summary information about a PDF""" + + def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False): + self._infile = infile + self._pages, pdf = _pdf_get_all_pageinfo( + infile, detailed_page_analysis, log=log, progbar=progbar + ) + self._needs_rendering = pdf.root.get('/NeedsRendering', False) + self._has_acroform = False + if '/AcroForm' in pdf.root: + if len(pdf.root.AcroForm.get('/Fields', [])) > 0: + self._has_acroform = True + elif '/XFA' in pdf.root.AcroForm: + self._has_acroform = True + pdf.close() + + @property + def pages(self): + return self._pages + + @property + def min_version(self): + # The minimum PDF is the maximum version that any particular page needs + return max(page.min_version for page in self.pages) + + @property + def has_userunit(self): + return any(page.userunit != 1.0 for page in self.pages) + + @property + def has_acroform(self): + return self._has_acroform + + @property + def filename(self): + if not isinstance(self._infile, (str, Path)): + raise NotImplementedError("can't get filename from stream") + return self._infile + + @property + def needs_rendering(self): + return self._needs_rendering + + def __getitem__(self, item): + return self._pages[item] + + def __len__(self): + return len(self._pages) + + def __repr__(self): + return f"" + + +def main(): + import argparse + + parser = argparse.ArgumentParser() + parser.add_argument('infile') + args = parser.parse_args() + info = _pdf_get_all_pageinfo(args.infile) + from pprint import pprint + + pprint(info) + + +if __name__ == '__main__': + main() diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index a4ab14f6..6482d69c 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -203,15 +203,15 @@ def test_stack_abuse(): stream = pikepdf.Stream(p, b'q ' * 35) with pytest.warns(None) as record: - pdfinfo._interpret_contents(stream) + pdfinfo.info._interpret_contents(stream) assert 'overflowed' in str(record[0].message) stream = pikepdf.Stream(p, b'q Q Q Q Q') with pytest.warns(None) as record: - pdfinfo._interpret_contents(stream) + pdfinfo.info._interpret_contents(stream) assert 'underflowed' in str(record[0].message) stream = pikepdf.Stream(p, b'q ' * 135) with pytest.warns(None): with pytest.raises(RuntimeError): - pdfinfo._interpret_contents(stream) + pdfinfo.info._interpret_contents(stream) From c32ea3b3748e5a35d30f68378ad4d8a6fb9061d0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 22 Jun 2019 00:59:04 -0700 Subject: [PATCH 096/880] If a page have vector content, promote to full color --- src/ocrmypdf/_pipeline.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index d0d5ad89..db7e7464 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -428,6 +428,9 @@ def rasterize( else: device_idx = at_least('png16m') + if pageinfo.has_vector: + device_idx = at_least('png16m') + device = colorspaces[device_idx] page_context.log.debug(f"Rasterize with {device}") From 3331a686fa290b3d29ef6794fc2093b94dda2f44 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 22 Jun 2019 00:59:33 -0700 Subject: [PATCH 097/880] Fix tess_threads clamped to 1 --- src/ocrmypdf/_sync.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index ffcb403d..daf332b1 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -232,8 +232,9 @@ def exec_concurrent(context): # parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we # get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the # input file is small, then we allow Tesseract to use threads, subject to the - # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers - tess_threads = min(1, context.options.jobs // max_workers) + # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers and limiting + # Tesseract to 4 threads. + tess_threads = min(4, context.options.jobs // max_workers) if context.options.tesseract_env is None: context.options.tesseract_env = os.environ.copy() context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads)) From 9b60d3e2851db04b0b2d62faf4281d621716232c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 22 Jun 2019 02:33:04 -0700 Subject: [PATCH 098/880] Improve testing of _validation.py --- src/ocrmypdf/_validation.py | 26 ++------- tests/test_validation.py | 110 ++++++++++++++++++++++++++++++++++++ 2 files changed, 114 insertions(+), 22 deletions(-) create mode 100644 tests/test_validation.py diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 8069ae1b..4163a58d 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -291,7 +291,6 @@ def check_options(options): check_options_advanced(options) check_options_pillow(options) check_dependency_versions(options) - check_environ(options) def check_closed_streams(options): @@ -353,21 +352,6 @@ def log_page_orientations(pdfinfo): log.info('Page orientations detected: %s', ' '.join(orientations)) -def check_environ(options): - old_envvars = ( - 'OCRMYPDF_TESSERACT', - 'OCRMYPDF_QPDF', - 'OCRMYPDF_GS', - 'OCRMYPDF_UNPAPER', - ) - for k in old_envvars: - if k in os.environ: - log.warning( - "OCRmyPDF no longer uses the environment variable {k}." - "Change PATH to select alternate programs." - ) - - def create_input_file(options, work_folder): if options.input_file == '-': # stdin @@ -418,12 +402,10 @@ def report_output_file_size(options, input_file, output_file): 'force_ocr', } for arg in image_preproc: - attr = getattr(options, arg, None) - if not attr: - continue - reasons.append( - f"The argument --{arg.replace('_', '-')} was issued, causing transcoding." - ) + if getattr(options, arg, False): + reasons.append( + f"The argument --{arg.replace('_', '-')} was issued, causing transcoding." + ) if reasons: explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n" diff --git a/tests/test_validation.py b/tests/test_validation.py new file mode 100644 index 00000000..c151ec16 --- /dev/null +++ b/tests/test_validation.py @@ -0,0 +1,110 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os +from unittest.mock import MagicMock, patch + +import pytest + +import ocrmypdf._validation as vd +from ocrmypdf.api import create_options +from ocrmypdf.exceptions import MissingDependencyError, BadArgsError + + +def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): + return create_options( + input_file=input_file, output_file=output_file, language=language, **kwargs + ) + + +def test_hocr_notlatin_warning(caplog): + vd.check_options_output(make_opts(language='chi_sim', pdf_renderer='hocr')) + assert 'PDF renderer is known to cause' in caplog.text + + +def test_old_ghostscript(caplog): + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.19'): + vd.check_options_output(make_opts(language='chi_sim', output_type='pdfa')) + assert 'Ghostscript does not work correctly' in caplog.text + + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.18'): + with pytest.raises(MissingDependencyError): + vd.check_options_output(make_opts(output_type='pdfa-3')) + + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.24'): + with pytest.raises(MissingDependencyError): + vd.check_dependency_versions(make_opts()) + + +def test_old_tesseract_error(): + with patch('ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=False): + with pytest.raises(MissingDependencyError): + opts = make_opts(pdf_renderer='sandwich', language='eng') + vd.check_options_output(opts) + + +def test_lossless_redo(): + with pytest.raises(BadArgsError): + vd.check_options_output(make_opts(redo_ocr=True, deskew=True)) + + +def test_mutex_options(): + with pytest.raises(BadArgsError): + vd.check_options_ocr_behavior(make_opts(force_ocr=True, skip_text=True)) + with pytest.raises(BadArgsError): + vd.check_options_ocr_behavior(make_opts(redo_ocr=True, skip_text=True)) + with pytest.raises(BadArgsError): + vd.check_options_ocr_behavior(make_opts(redo_ocr=True, force_ocr=True)) + with pytest.raises(BadArgsError): + vd.check_options_ocr_behavior(make_opts(pages='1-3', sidecar='file.txt')) + + +def test_optimizing(caplog): + vd.check_options_optimizing( + make_opts(optimize=0, jbig2_lossy=True, png_quality=18, jpeg_quality=10) + ) + assert 'will be ignored because' in caplog.text + + +def test_user_words(caplog): + vd.check_options_advanced(make_opts(user_words='foo')) + assert 'ignores --user-words' in caplog.text + + +def test_pillow_options(): + vd.check_options_pillow(make_opts(max_image_mpixels=0)) + + +def test_output_tty(): + with patch('sys.stdout.isatty', return_value=True): + with pytest.raises(BadArgsError): + vd.check_requested_output_file(make_opts(output_file='-')) + + +def test_report_file_size(tmp_path, caplog): + in_ = tmp_path / 'a.pdf' + out = tmp_path / 'b.pdf' + in_.write_bytes(b'123') + out.write_bytes(b'') + opts = make_opts() + vd.report_output_file_size(opts, in_, out) + assert caplog.text == '' + + os.truncate(in_, 25001) + os.truncate(out, 50000) + vd.report_output_file_size(opts, in_, out) + assert 'No reason' in caplog.text From 1beb7dfd37b95a56275869cc01e273cf6d0708e1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 22 Jun 2019 02:36:06 -0700 Subject: [PATCH 099/880] helpers: don't expect psutil will be installed It's not in stdlib --- src/ocrmypdf/helpers.py | 8 -------- 1 file changed, 8 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 8d52a822..80eff55f 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -85,14 +85,6 @@ def available_cpu_count(): return multiprocessing.cpu_count() except NotImplementedError: pass - - try: - import psutil - - return psutil.cpu_count() - except (ImportError, AttributeError): - pass - warnings.warn( "Could not get CPU count. Assuming one (1) CPU." "Use -j N to set manually." ) From 8aa678859d70c32f804de90256cd6c77466d8bb5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 22 Jun 2019 17:29:26 -0700 Subject: [PATCH 100/880] Use pandoc to rewrite .rst files Fixes all of the long lines, mainly. --- docs/advanced.rst | 225 +++-- docs/api.rst | 76 +- docs/batch.rst | 362 ++++---- docs/cookbook.rst | 204 +++-- docs/docker.rst | 141 +-- docs/errors.rst | 44 +- docs/installation.rst | 306 ++++--- docs/introduction.rst | 222 +++-- docs/jbig2.rst | 48 +- docs/languages.rst | 38 +- docs/plugins.rst | 20 +- docs/release_notes.rst | 1837 ++++++++++++++++++++++++---------------- docs/security.rst | 149 +++- 13 files changed, 2332 insertions(+), 1340 deletions(-) diff --git a/docs/advanced.rst b/docs/advanced.rst index 518b5bf1..fbc14d1a 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -1,16 +1,34 @@ +================= Advanced features ================= Control of unpaper ------------------- +================== -OCRmyPDF uses ``unpaper`` to provide the implementation of the ``--clean`` and ``--clean-final`` arguments. `unpaper `_ provides a variety of image processing filters to improve images. +OCRmyPDF uses ``unpaper`` to provide the implementation of the +``--clean`` and ``--clean-final`` arguments. +`unpaper `__ +provides a variety of image processing filters to improve images. -By default, OCRmyPDF uses only ``unpaper`` arguments that were found to be safe to use on almost all files without having to inspect every page of the file afterwards. This is particularly true when only ``--clean`` is used, since that instructs OCRmyPDF to only clean the image before OCR and not the final image. +By default, OCRmyPDF uses only ``unpaper`` arguments that were found to +be safe to use on almost all files without having to inspect every page +of the file afterwards. This is particularly true when only ``--clean`` +is used, since that instructs OCRmyPDF to only clean the image before +OCR and not the final image. -However, if you wish to use the more aggressive options in ``unpaper``, you may use ``--unpaper-args '...'`` to override the OCRmyPDF's defaults and forward other arguments to unpaper. This option will forward arguments to ``unpaper`` without any knowledge of what that program considers to be valid arguments. The string of arguments must be quoted as shown in the examples below. No filename arguments may be included. OCRmyPDF will assume it can append input and output filename of intermediate images to the ``--unpaper-args`` string. +However, if you wish to use the more aggressive options in ``unpaper``, +you may use ``--unpaper-args '...'`` to override the OCRmyPDF's defaults +and forward other arguments to unpaper. This option will forward +arguments to ``unpaper`` without any knowledge of what that program +considers to be valid arguments. The string of arguments must be quoted +as shown in the examples below. No filename arguments may be included. +OCRmyPDF will assume it can append input and output filename of +intermediate images to the ``--unpaper-args`` string. -In this example, we tell ``unpaper`` to expect two pages of text on a sheet (image), such as occurs when two facing pages of a book are scanned. ``unpaper`` uses this information to deskew each independently and clean up the margins of both. +In this example, we tell ``unpaper`` to expect two pages of text on a +sheet (image), such as occurs when two facing pages of a book are +scanned. ``unpaper`` uses this information to deskew each independently +and clean up the margins of both. .. code-block:: bash @@ -19,40 +37,71 @@ In this example, we tell ``unpaper`` to expect two pages of text on a sheet (ima .. warning:: - Some ``unpaper`` features will reposition text within the image. ``--clean-final`` is recommended to avoid this issue. + Some ``unpaper`` features will reposition text within the image. + ``--clean-final`` is recommended to avoid this issue. .. warning:: - Some ``unpaper`` features cause multiple input or output files to be consumed or produced. OCRmyPDF requires ``unpaper`` to consume one file and produce one file. An deviation from that condition will result in errors. + Some ``unpaper`` features cause multiple input or output files to be + consumed or produced. OCRmyPDF requires ``unpaper`` to consume one + file and produce one file. An deviation from that condition will + result in errors. .. note:: - ``unpaper`` uses uncompressed PBM/PGM/PPM files for its intermediate files. For large images or documents, it can take a lot of temporary disk space. + ``unpaper`` uses uncompressed PBM/PGM/PPM files for its intermediate + files. For large images or documents, it can take a lot of temporary + disk space. Control of OCR options ----------------------- +====================== -OCRmyPDF provides many features to control the behavior of the OCR engine, Tesseract. +OCRmyPDF provides many features to control the behavior of the OCR +engine, Tesseract. When OCR is skipped -""""""""""""""""""" +------------------- -If a page in a PDF seems to have text, by default OCRmyPDF will exit without modifying the PDF. This is to ensure that PDFs that were previously OCRed or were "born digital" rather than scanned are not processed. +If a page in a PDF seems to have text, by default OCRmyPDF will exit +without modifying the PDF. This is to ensure that PDFs that were +previously OCRed or were "born digital" rather than scanned are not +processed. -If ``--skip-text`` is issued, then no OCR will be performed on pages that already have text. The page will be copied to the output. This may be useful for documents that contain both "born digital" and scanned content, or to use OCRmyPDF to normalize and convert to PDF/A regardless of their contents. +If ``--skip-text`` is issued, then no OCR will be performed on pages +that already have text. The page will be copied to the output. This may +be useful for documents that contain both "born digital" and scanned +content, or to use OCRmyPDF to normalize and convert to PDF/A regardless +of their contents. -If ``--redo-ocr`` is issued, then a detailed text analysis is performed. Text is categorized as either visible or invisible. Invisible text (OCR) is stripped out. Then an image of each page is created with visible text masked out. The page image is sent for OCR, and any additional text is inserted as OCR. If a file contains a mix of text and bitmap images that contain text, OCRmyPDF will locate the additional text in images without disrupting the existing text. +If ``--redo-ocr`` is issued, then a detailed text analysis is performed. +Text is categorized as either visible or invisible. Invisible text (OCR) +is stripped out. Then an image of each page is created with visible text +masked out. The page image is sent for OCR, and any additional text is +inserted as OCR. If a file contains a mix of text and bitmap images that +contain text, OCRmyPDF will locate the additional text in images without +disrupting the existing text. -If ``--force-ocr`` is issued, then all pages will be rasterized to images, discarding any hidden OCR text, and rasterizing any printable text. This is useful for redoing OCR, for fixing OCR text with a damaged character map (text is selectable but not searchable), and destroying redacted information. Any forms and vector graphics will be rasterized as well. +If ``--force-ocr`` is issued, then all pages will be rasterized to +images, discarding any hidden OCR text, and rasterizing any printable +text. This is useful for redoing OCR, for fixing OCR text with a damaged +character map (text is selectable but not searchable), and destroying +redacted information. Any forms and vector graphics will be rasterized +as well. Time and image size limits -"""""""""""""""""""""""""" +-------------------------- -By default, OCRmyPDF permits tesseract to run for three minutes (180 seconds) per page. This is usually more than enough time to find all text on a reasonably sized page with modern hardware. +By default, OCRmyPDF permits tesseract to run for three minutes (180 +seconds) per page. This is usually more than enough time to find all +text on a reasonably sized page with modern hardware. -If a page is skipped, it will be inserted without OCR. If preprocessing was requested, the preprocessed image layer will be inserted. +If a page is skipped, it will be inserted without OCR. If preprocessing +was requested, the preprocessed image layer will be inserted. -If you want to adjust the amount of time spent on OCR, change ``--tesseract-timeout``. You can also automatically skip images that exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI, 8.5×11" page is 8.4 megapixels.) +If you want to adjust the amount of time spent on OCR, change +``--tesseract-timeout``. You can also automatically skip images that +exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI, +8.5×11" page is 8.4 megapixels.) .. code-block:: bash @@ -60,21 +109,27 @@ If you want to adjust the amount of time spent on OCR, change ``--tesseract-time ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf Overriding default tesseract -"""""""""""""""""""""""""""" +---------------------------- OCRmyPDF checks the system ``PATH`` for the ``tesseract`` binary. -Some relevant environment variables that influence Tesseract's behavior include: +Some relevant environment variables that influence Tesseract's behavior +include: .. envvar:: TESSDATA_PREFIX - Overrides the path to Tesseract's data files. This can allow simultaneous installation of the "best" and "fast" training data sets. OCRmyPDF does not manage this environment variable. + Overrides the path to Tesseract's data files. This can allow + simultaneous installation of the "best" and "fast" training data + sets. OCRmyPDF does not manage this environment variable. .. envvar:: OMP_THREAD_LIMIT - Controls the number of threads Tesseract will use. OCRmyPDF will manage this environment if it is not already set. (Currently, it will set it to 1 because this gives the best results in testing.) + Controls the number of threads Tesseract will use. OCRmyPDF will + manage this environment if it is not already set. (Currently, it will + set it to 1 because this gives the best results in testing.) -For example, if you have a development build of Tesseract don't wish to use the system installation, you can launch OCRmyPDF as follows: +For example, if you have a development build of Tesseract don't wish to +use the system installation, you can launch OCRmyPDF as follows: .. code-block:: bash @@ -83,26 +138,33 @@ For example, if you have a development build of Tesseract don't wish to use the TESSDATA_PREFIX=/home/user/src/tesseract \ ocrmypdf input.pdf output.pdf -In this example ``TESSDATA_PREFIX`` is required to redirect Tesseract to an alternate folder for its "tessdata" files. +In this example ``TESSDATA_PREFIX`` is required to redirect Tesseract to +an alternate folder for its "tessdata" files. Overriding other support programs -""""""""""""""""""""""""""""""""" +--------------------------------- In addition to tesseract, OCRmyPDF uses the following external binaries: -* ``gs`` (Ghostscript) -* ``unpaper`` -* ``qpdf`` - -In each case OCRmyPDF will search the ``PATH`` environment variable to locate the binaries. +- ``gs`` (Ghostscript) +- ``unpaper`` +- ``qpdf`` +In each case OCRmyPDF will search the ``PATH`` environment variable to +locate the binaries. Changing tesseract configuration variables -"""""""""""""""""""""""""""""""""""""""""" +------------------------------------------ -You can override tesseract's default `control parameters `_ with a configuration file. +You can override tesseract's default `control +parameters `__ +with a configuration file. -As an example, this configuration will disable Tesseract's dictionary for current language. Normally the dictionary is helpful for interpolating words that are unclear, but it may interfere with OCR if the document does not contain many words (for example, a list of part numbers). +As an example, this configuration will disable Tesseract's dictionary +for current language. Normally the dictionary is helpful for +interpolating words that are unclear, but it may interfere with OCR if +the document does not contain many words (for example, a list of part +numbers). Create a file named "no-dict.cfg" with these contents: @@ -120,11 +182,11 @@ then run ocrmypdf as follows (along with any other desired arguments): .. warning:: - Some combinations of control parameters will break Tesseract or break assumptions that OCRmyPDF makes about Tesseract's output. - + Some combinations of control parameters will break Tesseract or break + assumptions that OCRmyPDF makes about Tesseract's output. Changing the PDF renderer -------------------------- +========================= rasterizing Converting a PDF to an image for display. @@ -132,42 +194,63 @@ rasterizing rendering Creating a new PDF from other data (such as an existing PDF). - -OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The renderer may be selected using ``--pdf-renderer``. The default is ``auto`` which lets OCRmyPDF select the renderer to use. Currently, ``auto`` always selects ``sandwich``. +OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The +renderer may be selected using ``--pdf-renderer``. The default is +``auto`` which lets OCRmyPDF select the renderer to use. Currently, +``auto`` always selects ``sandwich``. The ``sandwich`` renderer -""""""""""""""""""""""""" +------------------------- -The ``sandwich`` renderer uses Tesseract's new text-only PDF feature, which produces a PDF page that lays out the OCR in invisible text. This page is then "sandwiched" onto the original PDF page, allowing lossless application of OCR even to PDF pages that contain other vector objects. +The ``sandwich`` renderer uses Tesseract's new text-only PDF feature, +which produces a PDF page that lays out the OCR in invisible text. This +page is then "sandwiched" onto the original PDF page, allowing lossless +application of OCR even to PDF pages that contain other vector objects. -Currently this is the best renderer for most uses, however it is implemented in Tesseract so OCRmyPDF cannot influence it. Currently some problematic PDF viewers like Mozilla PDF.js and macOS Preview have problems with segmenting its text output, and mightrunseveralwordstogether. +Currently this is the best renderer for most uses, however it is +implemented in Tesseract so OCRmyPDF cannot influence it. Currently some +problematic PDF viewers like Mozilla PDF.js and macOS Preview have +problems with segmenting its text output, and +mightrunseveralwordstogether. -When image preprocessing features like ``--deskew`` are used, the original PDF will be rendered as a full page and the OCR layer will be placed on top. +When image preprocessing features like ``--deskew`` are used, the +original PDF will be rendered as a full page and the OCR layer will be +placed on top. The ``hocr`` renderer -""""""""""""""""""""" +--------------------- -The ``hocr`` renderer works with older versions of Tesseract. The image layer is copied from the original PDF page if possible, avoiding potentially lossy transcoding or loss of other PDF information. If preprocessing is specified, then the image layer is a new PDF. +The ``hocr`` renderer works with older versions of Tesseract. The image +layer is copied from the original PDF page if possible, avoiding +potentially lossy transcoding or loss of other PDF information. If +preprocessing is specified, then the image layer is a new PDF. -Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone looking to customize how OCR is presented should look here. A major disadvantage of this renderer is it not capable of correctly handling text outside the Latin alphabet. Pull requests to improve the situation are welcome. +Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone +looking to customize how OCR is presented should look here. A major +disadvantage of this renderer is it not capable of correctly handling +text outside the Latin alphabet. Pull requests to improve the situation +are welcome. -Currently, this renderer has the best compatibility with Mozilla's PDF.js viewer. +Currently, this renderer has the best compatibility with Mozilla's +PDF.js viewer. This works in all versions of Tesseract. The ``tesseract`` renderer -"""""""""""""""""""""""""" +-------------------------- -The ``tesseract`` renderer was removed. OCRmyPDF's new approach to text layer grafting makes it functionally equivalent to ``sandwich``. +The ``tesseract`` renderer was removed. OCRmyPDF's new approach to text +layer grafting makes it functionally equivalent to ``sandwich``. Return code policy ------------------- +================== -OCRmyPDF writes all messages to ``stderr``. ``stdout`` is reserved for piping -output files. ``stdin`` is reserved for piping input files. +OCRmyPDF writes all messages to ``stderr``. ``stdout`` is reserved for +piping output files. ``stdin`` is reserved for piping input files. -The return codes generated by the OCRmyPDF are considered part of the stable -user interface. They may be imported from ``ocrmypdf.exceptions``. +The return codes generated by the OCRmyPDF are considered part of the +stable user interface. They may be imported from +``ocrmypdf.exceptions``. .. list-table:: Return codes :widths: 5 35 60 @@ -218,22 +301,36 @@ user interface. They may be imported from ``ocrmypdf.exceptions``. Debugging the intermediate files --------------------------------- +================================ -OCRmyPDF normally saves its intermediate results to a temporary folder and deletes this folder when it exits, whether it succeeded or failed. +OCRmyPDF normally saves its intermediate results to a temporary folder +and deletes this folder when it exits, whether it succeeded or failed. -If the ``-k`` argument is issued on the command line, OCRmyPDF will keep the temporary folder and print the location, whether it succeeded or failed (provided the Python interpreter did not crash). An example message is: +If the ``-k`` argument is issued on the command line, OCRmyPDF will keep +the temporary folder and print the location, whether it succeeded or +failed (provided the Python interpreter did not crash). An example +message is: .. code-block:: none Temporary working files retained at: /tmp/com.github.ocrmypdf.u20wpz07 -The organization of this folder is an implementation detail and subject to change between releases. However the general organization is that working files on a per page basis have the page number as a prefix (starting with page 1), an infix indicates the processing stage, and a suffix indicates the file type. Some important files include: +The organization of this folder is an implementation detail and subject +to change between releases. However the general organization is that +working files on a per page basis have the page number as a prefix +(starting with page 1), an infix indicates the processing stage, and a +suffix indicates the file type. Some important files include: -* ``.page.png`` - what the input page looks like -* ``.image`` - the image we will show the user if we are in a mode that changes the final appearance; may be in one of several image formats -* ``.text.pdf`` - the OCR file; this will load as a blank page but should have visible text if checked with a tool like pdftotext or pdfminder.six -* ``.ocr.png`` - the file that is sent to Tesseract for OCR; depending on arguments this may differ from the presentation image -* ``layers.rendered.pdf`` - the composite PDF, before metadata repair and optimization -* ``images/*`` - images extracted during the optimization process; here the prefix indicates a PDF object ID not a page number +- ``.page.png`` - what the input page looks like +- ``.image`` - the image we will show the user if we are in a mode that + changes the final appearance; may be in one of several image formats +- ``.text.pdf`` - the OCR file; this will load as a blank page but + should have visible text if checked with a tool like pdftotext or + pdfminder.six +- ``.ocr.png`` - the file that is sent to Tesseract for OCR; depending + on arguments this may differ from the presentation image +- ``layers.rendered.pdf`` - the composite PDF, before metadata repair + and optimization +- ``images/*`` - images extracted during the optimization process; here + the prefix indicates a PDF object ID not a page number diff --git a/docs/api.rst b/docs/api.rst index ffc27145..cc83b55f 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -1,14 +1,20 @@ +====================== Using the OCRmyPDF API ====================== -OCRmyPDF originated as a command line program and continues to have this legacy, but parts of it can be imported and used in other Python applications. +OCRmyPDF originated as a command line program and continues to have this +legacy, but parts of it can be imported and used in other Python +applications. -Some applications may want to consider running ocrmypdf from a subprocess call anyway, as this provides isolation of its activities. +Some applications may want to consider running ocrmypdf from a +subprocess call anyway, as this provides isolation of its activities. Example -------- +======= -OCRmyPDF one high-level function to run its main engine from an application. The parameters are symmetric to the command line arguments and largely have the same functions. +OCRmyPDF one high-level function to run its main engine from an +application. The parameters are symmetric to the command line arguments +and largely have the same functions. .. code-block:: python @@ -16,44 +22,66 @@ OCRmyPDF one high-level function to run its main engine from an application. The ocrmypdf.run('input.pdf', 'output.pdf', deskew=True) -With a few exceptions, all of the command line arguments are available and may be passed as equivalent keywords. +With a few exceptions, all of the command line arguments are available +and may be passed as equivalent keywords. -A few differences are that ``verbose`` and ``quiet`` are not available. Instead, output should be managed by configuring logging. +A few differences are that ``verbose`` and ``quiet`` are not available. +Instead, output should be managed by configuring logging. Parent process requirements -^^^^^^^^^^^^^^^^^^^^^^^^^^^ +--------------------------- -The :func:`ocrmypdf.run` function runs OCRmyPDF similar to command line execution. To do this, it will: -- create a monitoring thread -- create worker processes (forking itself) -- manage the signal flags of worker processes -0 execute other subprocesses (forking and executing other programs) +The :func:`ocrmypdf.run` function runs OCRmyPDF similar to command line +execution. To do this, it will: - create a monitoring thread - create +worker processes (forking itself) - manage the signal flags of worker +processes 0 execute other subprocesses (forking and executing other +programs) -The Python process that calls ``ocrmypdf.run()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will fail. +The Python process that calls ``ocrmypdf.run()`` must be sufficiently +privileged to perform these actions. If it is not, ``ocrmypdf()`` will +fail. -There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. +There is no currently no option to manage how jobs are scheduled other +than the argument ``jobs=`` which will limit the number of worker +processes. -Forking a child process to call ``ocrmypdf.run()`` is suggested. That way your application will survive and remain interactive even if OCRmyPDF does not. +Forking a child process to call ``ocrmypdf.run()`` is suggested. That +way your application will survive and remain interactive even if +OCRmyPDF does not. Logging -^^^^^^^ +------- -OCRmyPDF will log under loggers named ``ocrmypdf``. In addition, it imports ``pdfminer`` and ``PIL``, both of which post log messages under those logging namespaces. +OCRmyPDF will log under loggers named ``ocrmypdf``. In addition, it +imports ``pdfminer`` and ``PIL``, both of which post log messages under +those logging namespaces. -You can configure the logging as desired for your application or call :func:`ocrmypdf.configure_logging` to configure logging the same way OCRmyPDF itself does. The command line parameters such as ``--quiet`` and ``--verbose`` have no equivalents in the API; you must use the provided configuration function or do configuration in a -way that suits your use case. +You can configure the logging as desired for your application or call +:func:`ocrmypdf.configure_logging` to configure logging the same way +OCRmyPDF itself does. The command line parameters such as ``--quiet`` +and ``--verbose`` have no equivalents in the API; you must use the +provided configuration function or do configuration in a way that suits +your use case. Progress monitoring -^^^^^^^^^^^^^^^^^^^ +------------------- -OCRmyPDF uses the ``tqdm`` package to implement its progress bars. :func:`ocrmypdf.configure_logging` will set up logging output to ``sys.stderr`` in a way that is compatible with the display of the progress bar. +OCRmyPDF uses the ``tqdm`` package to implement its progress bars. +:func:`ocrmypdf.configure_logging` will set up logging output to +``sys.stderr`` in a way that is compatible with the display of the +progress bar. Exceptions -^^^^^^^^^^ +---------- -OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*`` exceptions, some exceptions related to multiprocessing, and ``KeyboardInterrupt``. The parent process should provide an exception handler. OCRmyPDF will clean up its temporary files and worker processes automatically when an exception occurs. +OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*`` +exceptions, some exceptions related to multiprocessing, and +``KeyboardInterrupt``. The parent process should provide an exception +handler. OCRmyPDF will clean up its temporary files and worker processes +automatically when an exception occurs. -Programs that call OCRmyPDF should consider trapping KeyboardInterrupt so that they allow OCR to terminate with the whole program terminating. +Programs that call OCRmyPDF should consider trapping KeyboardInterrupt +so that they allow OCR to terminate with the whole program terminating. When OCRmyPDF succeeds conditionally, it returns an integer exit code. diff --git a/docs/batch.rst b/docs/batch.rst index 7ae20779..555879b2 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -1,215 +1,263 @@ +================ Batch processing ================ -This article provides information about running OCRmyPDF on multiple files or configuring it as a service triggered by file system events. +This article provides information about running OCRmyPDF on multiple +files or configuring it as a service triggered by file system events. Batch jobs ----------- +========== -Consider using the excellent `GNU Parallel `_ to apply OCRmyPDF to multiple files at once. +Consider using the excellent `GNU +Parallel `__ to apply OCRmyPDF +to multiple files at once. -Both ``parallel`` and ``ocrmypdf`` will try to use all available processors. To maximize parallelism without overloading your system with processes, consider using ``parallel -j 2`` to limit parallel to running two jobs at once. +Both ``parallel`` and ``ocrmypdf`` will try to use all available +processors. To maximize parallelism without overloading your system with +processes, consider using ``parallel -j 2`` to limit parallel to running +two jobs at once. -This command will run all ocrmypdf all files named ``*.pdf`` in the current directory and write them to the previous created ``output/`` folder. It will not search subdirectories. +This command will run all ocrmypdf all files named ``*.pdf`` in the +current directory and write them to the previous created ``output/`` +folder. It will not search subdirectories. -The ``--tag`` argument tells parallel to print the filename as a prefix whenever a message is printed, so that one can trace any errors to the file that produced them. +The ``--tag`` argument tells parallel to print the filename as a prefix +whenever a message is printed, so that one can trace any errors to the +file that produced them. .. code-block:: bash - parallel --tag -j 2 ocrmypdf '{}' 'output/{}' ::: *.pdf + parallel --tag -j 2 ocrmypdf '{}' 'output/{}' ::: *.pdf -OCRmyPDF automatically repairs PDFs before parsing and gathering information from them. +OCRmyPDF automatically repairs PDFs before parsing and gathering +information from them. Directory trees ---------------- +=============== -This will walk through a directory tree and run OCR on all files in place, printing the output in a way that makes +This will walk through a directory tree and run OCR on all files in +place, printing the output in a way that makes .. code-block:: bash - find . -printf '%p' -name '*.pdf' -exec ocrmypdf '{}' '{}' \; + find . -printf '%p' -name '*.pdf' -exec ocrmypdf '{}' '{}' \; -Alternatively, with a docker container (mounts a volume to the container where the PDFs are stored): +Alternatively, with a docker container (mounts a volume to the container +where the PDFs are stored): .. code-block:: bash - find . -printf '%p' -name '*.pdf' -exec docker run --rm -v : jbarlow83/ocrmypdf-alpine '/{}' '/{}' \; + find . -printf '%p' -name '*.pdf' -exec docker run --rm -v : jbarlow83/ocrmypdf-alpine '/{}' '/{}' \; -This only runs one ``ocrmypdf`` process at a time. This variation uses ``find`` to create a directory list and ``parallel`` to parallelize runs of ``ocrmypdf``, again updating files in place. +This only runs one ``ocrmypdf`` process at a time. This variation uses +``find`` to create a directory list and ``parallel`` to parallelize runs +of ``ocrmypdf``, again updating files in place. .. code-block:: bash - find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}' - + find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}' Sample script -""""""""""""" +------------- -This user contributed script also provides an example of batch processing. +This user contributed script also provides an example of batch +processing. .. code-block:: python - #!/usr/bin/env python3 - # Walk through directory tree, replacing all files with OCR'd version - # Contributed by DeliciousPickle@github + #!/usr/bin/env python3 + # Walk through directory tree, replacing all files with OCR'd version + # Contributed by DeliciousPickle@github - import logging - import os - import subprocess - import sys + import logging + import os + import subprocess + import sys - script_dir = os.path.dirname(os.path.realpath(__file__)) - print(script_dir + '/ocr-tree.py: Start') + script_dir = os.path.dirname(os.path.realpath(__file__)) + print(script_dir + '/ocr-tree.py: Start') - if len(sys.argv) > 1: - start_dir = sys.argv[1] - else: - start_dir = '.' + if len(sys.argv) > 1: + start_dir = sys.argv[1] + else: + start_dir = '.' - if len(sys.argv) > 2: - log_file = sys.argv[2] - else: - log_file = script_dir + '/ocr-tree.log' + if len(sys.argv) > 2: + log_file = sys.argv[2] + else: + log_file = script_dir + '/ocr-tree.log' - logging.basicConfig( - level=logging.INFO, format='%(asctime)s %(message)s', - filename=log_file, filemode='w') + logging.basicConfig( + level=logging.INFO, format='%(asctime)s %(message)s', + filename=log_file, filemode='w') - for dir_name, subdirs, file_list in os.walk(start_dir): - logging.info('\n') - logging.info(dir_name + '\n') - os.chdir(dir_name) - for filename in file_list: - file_ext = os.path.splitext(filename)[1] - if file_ext == '.pdf': - full_path = dir_name + '/' + filename - print(full_path) - cmd = ["ocrmypdf", "--deskew", filename, filename] - logging.info(cmd) - proc = subprocess.run( - cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) - result = proc.stdout - if proc.returncode == 6: - print("Skipped document because it already contained text") - elif proc.returncode == 0: - print("OCR complete") - logging.info(result) + for dir_name, subdirs, file_list in os.walk(start_dir): + logging.info('\n') + logging.info(dir_name + '\n') + os.chdir(dir_name) + for filename in file_list: + file_ext = os.path.splitext(filename)[1] + if file_ext == '.pdf': + full_path = dir_name + '/' + filename + print(full_path) + cmd = ["ocrmypdf", "--deskew", filename, filename] + logging.info(cmd) + proc = subprocess.run( + cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) + result = proc.stdout + if proc.returncode == 6: + print("Skipped document because it already contained text") + elif proc.returncode == 0: + print("OCR complete") + logging.info(result) Synology DiskStations -""""""""""""""""""""" +--------------------- -Synology DiskStations (Network Attached Storage devices) can run the Docker image of OCRmyPDF if the Synology `Docker package `_ is installed. Attached is a script to address particular quirks of using OCRmyPDF on one of these devices. +Synology DiskStations (Network Attached Storage devices) can run the +Docker image of OCRmyPDF if the Synology `Docker +package `__ is +installed. Attached is a script to address particular quirks of using +OCRmyPDF on one of these devices. -This is only possible for x86-based Synology products. Some Synology products use ARM or Power processors and do not support Docker. Further adjustments might be needed to deal with the Synology's relatively limited CPU and RAM. +This is only possible for x86-based Synology products. Some Synology +products use ARM or Power processors and do not support Docker. Further +adjustments might be needed to deal with the Synology's relatively +limited CPU and RAM. .. code-block:: python - #!/bin/env python3 - # Contributed by github.com/Enantiomerie + #!/bin/env python3 + # Contributed by github.com/Enantiomerie - # script needs 2 arguments - # 1. source dir with *.pdf - default is location of script - # 2. move dir where *.pdf and *_OCR.pdf are moved to + # script needs 2 arguments + # 1. source dir with *.pdf - default is location of script + # 2. move dir where *.pdf and *_OCR.pdf are moved to - import logging - import os - import subprocess - import sys - import time - import shutil + import logging + import os + import subprocess + import sys + import time + import shutil - script_dir = os.path.dirname(os.path.realpath(__file__)) - timestamp = time.strftime("%Y-%m-%d-%H%M_") - log_file = script_dir + '/' + timestamp + 'ocrmypdf.log' - logging.basicConfig(level=logging.INFO, format='%(asctime)s %(message)s', filename=log_file, filemode='w') + script_dir = os.path.dirname(os.path.realpath(__file__)) + timestamp = time.strftime("%Y-%m-%d-%H%M_") + log_file = script_dir + '/' + timestamp + 'ocrmypdf.log' + logging.basicConfig(level=logging.INFO, format='%(asctime)s %(message)s', filename=log_file, filemode='w') - if len(sys.argv) > 1: - start_dir = sys.argv[1] - else: - start_dir = '.' + if len(sys.argv) > 1: + start_dir = sys.argv[1] + else: + start_dir = '.' - for dir_name, subdirs, file_list in os.walk(start_dir): - logging.info('\n') - logging.info(dir_name + '\n') - os.chdir(dir_name) - for filename in file_list: - file_ext = os.path.splitext(filename)[1] - if file_ext == '.pdf': - full_path = dir_name + '/' + filename - file_noext = os.path.splitext(filename)[0] - timestamp_OCR = time.strftime("%Y-%m-%d-%H%M_OCR_") - filename_OCR = timestamp_OCR + file_noext + '.pdf' - docker_mount = dir_name + ':/home/docker' - # create string for pdf processing - # diskstation needs a user:group docker:docker. find uid:gid of your diskstation docker:docker with id docker. - # use this uid:gid in -u flag - # rw rights for docker:docker at source dir are also necessary - # the script is processed as root user via chron - cmd = ['docker', 'run', '--rm', '-v', docker_mount, '-u=1030:65538', 'jbarlow83/ocrmypdf', , '--deskew' , filename, filename_OCR] - logging.info(cmd) - proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) - result = proc.stdout.read() - logging.info(result) - full_path_OCR = dir_name + '/' + filename_OCR - os.chmod(full_path_OCR, 0o666) - os.chmod(full_path, 0o666) - full_path_OCR_archive = sys.argv[2] - full_path_archive = sys.argv[2] + '/no_ocr' - shutil.move(full_path_OCR,full_path_OCR_archive) - shutil.move(full_path, full_path_archive) - logging.info('Finished.\n') + for dir_name, subdirs, file_list in os.walk(start_dir): + logging.info('\n') + logging.info(dir_name + '\n') + os.chdir(dir_name) + for filename in file_list: + file_ext = os.path.splitext(filename)[1] + if file_ext == '.pdf': + full_path = dir_name + '/' + filename + file_noext = os.path.splitext(filename)[0] + timestamp_OCR = time.strftime("%Y-%m-%d-%H%M_OCR_") + filename_OCR = timestamp_OCR + file_noext + '.pdf' + docker_mount = dir_name + ':/home/docker' + # create string for pdf processing + # diskstation needs a user:group docker:docker. find uid:gid of your diskstation docker:docker with id docker. + # use this uid:gid in -u flag + # rw rights for docker:docker at source dir are also necessary + # the script is processed as root user via chron + cmd = ['docker', 'run', '--rm', '-v', docker_mount, '-u=1030:65538', 'jbarlow83/ocrmypdf', , '--deskew' , filename, filename_OCR] + logging.info(cmd) + proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) + result = proc.stdout.read() + logging.info(result) + full_path_OCR = dir_name + '/' + filename_OCR + os.chmod(full_path_OCR, 0o666) + os.chmod(full_path, 0o666) + full_path_OCR_archive = sys.argv[2] + full_path_archive = sys.argv[2] + '/no_ocr' + shutil.move(full_path_OCR,full_path_OCR_archive) + shutil.move(full_path, full_path_archive) + logging.info('Finished.\n') Huge batch jobs -""""""""""""""" - -If you have thousands of files to work with, contact the author. Consulting work related to OCRmyPDF helps fund this open source project and all inquiries are appreciated. - -Hot (watched) folders ---------------------- - -To set up a "hot folder" that will trigger OCR for every file inserted, use a program like Python `watchdog `_ (supports all major OS). - -One could then configure a scanner to automatically place scanned files in a hot folder, so that they will be queued for OCR and copied to the destination. - -.. code-block:: bash - - pip install watchdog - -watchdog installs the command line program ``watchmedo``, which can be told to run ``ocrmypdf`` on any .pdf added to the current directory (``.``) and place the result in the previously created ``out/`` folder. - -.. code-block:: bash - - cd hot-folder - mkdir out - watchmedo shell-command \ - --patterns="*.pdf" \ - --ignore-directories \ - --command='ocrmypdf "${watch_src_path}" "out/${watch_src_path}" ' \ - . # don't forget the final dot - -For more complex behavior you can write a Python script around to use the watchdog API. - -On file servers, you could configure watchmedo as a system service so it will run all the time. - -Caveats -""""""" - -* ``watchmedo`` may not work properly on a networked file system, depending on the capabilities of the file system client and server. -* This simple recipe does not filter for the type of file system event, so file copies, deletes and moves, and directory operations, will all be sent to ocrmypdf, producing errors in several cases. Disable your watched folder if you are doing anything other than copying files to it. -* If the source and destination directory are the same, watchmedo may create an infinite loop. -* On BSD, FreeBSD and older versions of macOS, you may need to increase the number of file descriptors to monitor more files, using ``ulimit -n 1024`` to watch a folder of up to 1024 files. - -Alternatives -"""""""""""" - -* `Watchman `_ is a more powerful alternative to ``watchmedo``. - -macOS Automator --------------- -You can use the Automator app with macOS, to create a Workflow or Quick Action. Use a *Run Shell Script* action in your workflow. In the context of Automator, the ``PATH`` may be set differently your Terminal's ``PATH``; you may need to explicitly set the PATH to include ``ocrmypdf``. The following example may serve as a starting point: +If you have thousands of files to work with, contact the author. +Consulting work related to OCRmyPDF helps fund this open source project +and all inquiries are appreciated. -.. image:: images/macos-workflow.png - :alt: Example macOS Automator script +Hot (watched) folders +===================== + +To set up a "hot folder" that will trigger OCR for every file inserted, +use a program like Python +`watchdog `__ (supports all major +OS). + +One could then configure a scanner to automatically place scanned files +in a hot folder, so that they will be queued for OCR and copied to the +destination. + +.. code-block:: bash + + pip install watchdog + +watchdog installs the command line program ``watchmedo``, which can be +told to run ``ocrmypdf`` on any .pdf added to the current directory +(``.``) and place the result in the previously created ``out/`` folder. + +.. code-block:: bash + + cd hot-folder + mkdir out + watchmedo shell-command \ + --patterns="*.pdf" \ + --ignore-directories \ + --command='ocrmypdf "${watch_src_path}" "out/${watch_src_path}" ' \ + . # don't forget the final dot + +For more complex behavior you can write a Python script around to use +the watchdog API. + +On file servers, you could configure watchmedo as a system service so it +will run all the time. + +Caveats +------- + +- ``watchmedo`` may not work properly on a networked file system, + depending on the capabilities of the file system client and server. +- This simple recipe does not filter for the type of file system event, + so file copies, deletes and moves, and directory operations, will all + be sent to ocrmypdf, producing errors in several cases. Disable your + watched folder if you are doing anything other than copying files to + it. +- If the source and destination directory are the same, watchmedo may + create an infinite loop. +- On BSD, FreeBSD and older versions of macOS, you may need to increase + the number of file descriptors to monitor more files, using + ``ulimit -n 1024`` to watch a folder of up to 1024 files. + +Alternatives +------------ + +- `Watchman `__ is a more + powerful alternative to ``watchmedo``. + +macOS Automator +=============== + +You can use the Automator app with macOS, to create a Workflow or Quick +Action. Use a *Run Shell Script* action in your workflow. In the context +of Automator, the ``PATH`` may be set differently your Terminal's +``PATH``; you may need to explicitly set the PATH to include +``ocrmypdf``. The following example may serve as a starting point: + +|Example macOS Automator script| You may customize the command sent to ocrmypdf. + +.. |Example macOS Automator script| image:: images/macos-workflow.png diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 3ea153aa..bc6ca951 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -1,11 +1,12 @@ +======== Cookbook ======== Basic examples --------------- +============== Help! -^^^^^ +----- ocrmypdf has built-in help. @@ -13,30 +14,29 @@ ocrmypdf has built-in help. ocrmypdf --help - Add an OCR layer and convert to PDF/A -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +------------------------------------- .. code-block:: bash ocrmypdf input.pdf output.pdf Add an OCR layer and output a standard PDF -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +------------------------------------------ .. code-block:: bash ocrmypdf --output-type pdf input.pdf output.pdf Create a PDF/A with all color and grayscale images converted to JPEG -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +-------------------------------------------------------------------- .. code-block:: bash ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf Modify a file in place -^^^^^^^^^^^^^^^^^^^^^^ +---------------------- The file will only be overwritten if OCRmyPDF is successful. @@ -45,48 +45,58 @@ The file will only be overwritten if OCRmyPDF is successful. ocrmypdf myfile.pdf myfile.pdf Correct page rotation -^^^^^^^^^^^^^^^^^^^^^ +--------------------- -OCR will attempt to automatic correct the rotation of each page. This can help fix a scanning job that contains a mix of landscape and portrait pages. +OCR will attempt to automatic correct the rotation of each page. This +can help fix a scanning job that contains a mix of landscape and +portrait pages. .. code-block:: bash ocrmypdf --rotate-pages myfile.pdf myfile.pdf -You can increase (decrease) the parameter ``--rotate-pages-threshold`` to make page rotation more (less) aggressive. +You can increase (decrease) the parameter ``--rotate-pages-threshold`` +to make page rotation more (less) aggressive. -If the page is "just a little off horizontal", like a crooked picture, then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal angle is wrong. +If the page is "just a little off horizontal", like a crooked picture, +then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal +angle is wrong. OCR languages other than English -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +-------------------------------- -OCRmyPDF assumes the document is in English unless told otherwise. OCR quality may be poor if the wrong language is used. +OCRmyPDF assumes the document is in English unless told otherwise. OCR +quality may be poor if the wrong language is used. .. code-block:: bash ocrmypdf -l fra LeParisien.pdf LeParisien.pdf ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf -Language packs must be installed for all languages specified. See :ref:`Installing additional language packs `. +Language packs must be installed for all languages specified. See +:ref:`Installing additional language packs `. -Unfortunately, the Tesseract OCR engine has no ability to detect the language when it is unknown. +Unfortunately, the Tesseract OCR engine has no ability to detect the +language when it is unknown. Produce PDF and text file containing OCR text -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +--------------------------------------------- -This produces a file named "output.pdf" and a companion text file named "output.txt". +This produces a file named "output.pdf" and a companion text file named +"output.txt". .. code-block:: bash ocrmypdf --sidecar output.txt input.pdf output.pdf OCR images, not PDFs -^^^^^^^^^^^^^^^^^^^^ +-------------------- Option: use Tesseract -""""""""""""""""""""" +~~~~~~~~~~~~~~~~~~~~~ -If you are starting with images, you can just use Tesseract directly to convert images to PDFs: +If you are starting with images, you can just use Tesseract directly to +convert images to PDFs: .. code-block:: bash @@ -97,62 +107,92 @@ If you are starting with images, you can just use Tesseract directly to convert # When there are multiple images tesseract text-file-containing-list-of-image-filenames.txt output-prefix pdf -Tesseract's PDF output is quite good – OCRmyPDF uses it internally, in some cases. However, OCRmyPDF has many features not available in Tesseract like image processing, metadata control, and PDF/A generation. +Tesseract's PDF output is quite good – OCRmyPDF uses it internally, in +some cases. However, OCRmyPDF has many features not available in +Tesseract like image processing, metadata control, and PDF/A generation. Option: use img2pdf -""""""""""""""""""" +~~~~~~~~~~~~~~~~~~~ -You can also use a program like `img2pdf `_ to convert your images to PDFs, and then pipe the results to run ocrmypdf. The ``-`` tells ocrmypdf to read standard input. +You can also use a program like +`img2pdf `__ to convert +your images to PDFs, and then pipe the results to run ocrmypdf. The +``-`` tells ocrmypdf to read standard input. .. code-block:: bash img2pdf my-images*.jpg | ocrmypdf - myfile.pdf -``img2pdf`` is recommended because it does an excellent job at generating PDFs without transcoding images. +``img2pdf`` is recommended because it does an excellent job at +generating PDFs without transcoding images. Option: use OCRmyPDF (single images only) -""""""""""""""""""""""""""""""""""""""""" +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -For convenience, OCRmyPDF can also convert single images to PDFs on its own. If the resolution (dots per inch, DPI) of an image is not set or is incorrect, it can be overridden with ``--image-dpi``. (As 1 inch is 2.54 cm, 1 dpi = 0.39 dpcm). +For convenience, OCRmyPDF can also convert single images to PDFs on its +own. If the resolution (dots per inch, DPI) of an image is not set or is +incorrect, it can be overridden with ``--image-dpi``. (As 1 inch is 2.54 +cm, 1 dpi = 0.39 dpcm). .. code-block:: bash ocrmypdf --image-dpi 300 image.png myfile.pdf -If you have multiple images, you must use ``img2pdf`` to convert the images to PDF. +If you have multiple images, you must use ``img2pdf`` to convert the +images to PDF. Not recommended -""""""""""""""" +~~~~~~~~~~~~~~~ -We caution against using ImageMagick or Ghostscript to convert images to PDF, since they may transcode images or produce downsampled images, sometimes without warning. +We caution against using ImageMagick or Ghostscript to convert images to +PDF, since they may transcode images or produce downsampled images, +sometimes without warning. Image processing ----------------- +================ -OCRmyPDF perform some image processing on each page of a PDF, if desired. The same processing is applied to each page. It is suggested that the user review files after image processing as these commands might remove desirable content, especially from poor quality scans. +OCRmyPDF perform some image processing on each page of a PDF, if +desired. The same processing is applied to each page. It is suggested +that the user review files after image processing as these commands +might remove desirable content, especially from poor quality scans. -* ``--rotate-pages`` attempts to determine the correct orientation for each page and rotates the page if necessary. - -* ``--remove-background`` attempts to detect and remove a noisy background from grayscale or color images. Monochrome images are ignored. This should not be used on documents that contain color photos as it may remove them. - -* ``--deskew`` will correct pages were scanned at a skewed angle by rotating them back into place. Skew determination and correction is performed using `Postl's variance of line sums `_ algorithm as implemented in `Leptonica `_. - -* ``--clean`` uses `unpaper `_ to clean up pages before OCR, but does not alter the final output. This makes it less likely that OCR will try to find text in background noise. - -* ``--clean-final`` uses unpaper to clean up pages before OCR and inserts the page into the final output. You will want to review each page to ensure that unpaper did not remove something important. - -* ``--mask-barcodes`` will suppress any barcodes detected in a page image. Barcodes are known to confuse Tesseract OCR and interfere with the recognition of text on the same baseline as a barcode. The output file will contain the unaltered image of the barcode. +- ``--rotate-pages`` attempts to determine the correct orientation for + each page and rotates the page if necessary. +- ``--remove-background`` attempts to detect and remove a noisy + background from grayscale or color images. Monochrome images are + ignored. This should not be used on documents that contain color + photos as it may remove them. +- ``--deskew`` will correct pages were scanned at a skewed angle by + rotating them back into place. Skew determination and correction is + performed using `Postl's variance of line + sums `__ algorithm as + implemented in `Leptonica `__. +- ``--clean`` uses + `unpaper `__ to clean up + pages before OCR, but does not alter the final output. This makes it + less likely that OCR will try to find text in background noise. +- ``--clean-final`` uses unpaper to clean up pages before OCR and + inserts the page into the final output. You will want to review each + page to ensure that unpaper did not remove something important. +- ``--mask-barcodes`` will suppress any barcodes detected in a page + image. Barcodes are known to confuse Tesseract OCR and interfere with + the recognition of text on the same baseline as a barcode. The output + file will contain the unaltered image of the barcode. .. note:: - In many cases image processing will rasterize PDF pages as images, potentially losing quality. + In many cases image processing will rasterize PDF pages as images, + potentially losing quality. .. warning:: - ``--clean-final`` and ``-remove-background`` may leave undesirable visual artifacts in some images where their algorithms have shortcomings. Files should be visually reviewed after using these options. + ``--clean-final`` and ``-remove-background`` may leave undesirable + visual artifacts in some images where their algorithms have + shortcomings. Files should be visually reviewed after using these + options. Example: OCR and correct document skew (crooked scan) -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +----------------------------------------------------- Deskew: @@ -160,55 +200,83 @@ Deskew: ocrmypdf --deskew input.pdf output.pdf -Image processing commands can be combined. The order in which options are given does not matter. OCRmyPDF always applies the steps of the image processing pipeline in the same order (rotate, remove background, deskew, clean). +Image processing commands can be combined. The order in which options +are given does not matter. OCRmyPDF always applies the steps of the +image processing pipeline in the same order (rotate, remove background, +deskew, clean). .. code-block:: bash ocrmypdf --deskew --clean --rotate-pages input.pdf output.pdf - Don't actually OCR my PDF -------------------------- +========================= -If you set ``--tesseract-timeout 0`` OCRmyPDF will apply its image processing without performing OCR, if all you want to is to apply image processing or PDF/A conversion. +If you set ``--tesseract-timeout 0`` OCRmyPDF will apply its image +processing without performing OCR, if all you want to is to apply image +processing or PDF/A conversion. .. code-block:: bash ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf - Redo existing OCR ------------------ +================= -To redo OCR on a file OCRed with other OCR software or a previous version of OCRmyPDF and/or Tesseract, you may use the ``--redo-ocr`` argument. (Normally, OCRmyPDF will exit with an error if asked to modify a file with OCR.) +To redo OCR on a file OCRed with other OCR software or a previous +version of OCRmyPDF and/or Tesseract, you may use the ``--redo-ocr`` +argument. (Normally, OCRmyPDF will exit with an error if asked to modify +a file with OCR.) -This may be helpful for users who want to take advantage of accuracy improvements in Tesseract 4.0 for files they previously OCRed with an earlier version of Tesseract and OCRmyPDF. +This may be helpful for users who want to take advantage of accuracy +improvements in Tesseract 4.0 for files they previously OCRed with an +earlier version of Tesseract and OCRmyPDF. .. code-block:: bash ocrmypdf --redo-ocr input.pdf output.pdf -This method will replace OCR without rasterizing, reducing quality or removing vector content. If a file contains a mix of pure digital text and OCR, digital text will be ignored and OCR will be replaced. As such this mode is incompatible with image processing options, since they alter the appearance of the file. +This method will replace OCR without rasterizing, reducing quality or +removing vector content. If a file contains a mix of pure digital text +and OCR, digital text will be ignored and OCR will be replaced. As such +this mode is incompatible with image processing options, since they +alter the appearance of the file. -In some cases, existing OCR cannot be detected or replaced. Files produced by OCRmyPDF v2.2 or earlier, for example, are internally represented as having visible text with an opaque image drawn on top. This situation cannot be detected. +In some cases, existing OCR cannot be detected or replaced. Files +produced by OCRmyPDF v2.2 or earlier, for example, are internally +represented as having visible text with an opaque image drawn on top. +This situation cannot be detected. -If ``--redo-ocr`` does not work, you can use ``--force-ocr``, which will force rasterization of all pages, potentially reducing quality or losing vector content. +If ``--redo-ocr`` does not work, you can use ``--force-ocr``, which will +force rasterization of all pages, potentially reducing quality or losing +vector content. Improving OCR quality ---------------------- +===================== -The `Image processing`_ features can improve OCR quality. +The `Image processing <#image-processing>`__ features can improve OCR +quality. -Rotating pages and deskewing helps to ensure that the page orientation is correct before OCR begins. Removing the background and/or cleaning the page can also improve results. The ``--oversample DPI`` argument can be specified to resample images to higher resolution before attempting OCR; this can improve results as well. +Rotating pages and deskewing helps to ensure that the page orientation +is correct before OCR begins. Removing the background and/or cleaning +the page can also improve results. The ``--oversample DPI`` argument can +be specified to resample images to higher resolution before attempting +OCR; this can improve results as well. -OCR quality will suffer if the resolution of input images is not correct (since the range of pixel sizes that will be checked for possible fonts will also be incorrect). +OCR quality will suffer if the resolution of input images is not correct +(since the range of pixel sizes that will be checked for possible fonts +will also be incorrect). PDF optimization ----------------- +================ -By default OCRmyPDF will attempt to perform lossless optimizations on the images inside PDFs after OCR is complete. Optimization is performed even if no OCR text is found. +By default OCRmyPDF will attempt to perform lossless optimizations on +the images inside PDFs after OCR is complete. Optimization is performed +even if no OCR text is found. -The ``--optimize N`` (short form ``-O``) argument controls optimization, where ``N`` ranges from 0 to 3 inclusive, analogous to the optimization levels in the GCC compiler. +The ``--optimize N`` (short form ``-O``) argument controls optimization, +where ``N`` ranges from 0 to 3 inclusive, analogous to the optimization +levels in the GCC compiler. .. list-table:: :widths: auto @@ -227,9 +295,15 @@ The ``--optimize N`` (short form ``-O``) argument controls optimization, where ` * - ``--optimize 3`` - All of the above, and enables more aggressive optimizations and targets lower image quality. -Optimization is improved when a JBIG2 encoder is available and when ``pngquant`` is installed. If either of these components are missing, then some types of images cannot be optimized. +Optimization is improved when a JBIG2 encoder is available and when +``pngquant`` is installed. If either of these components are missing, +then some types of images cannot be optimized. -The types of optimization available may expand over time. By default, OCRmyPDF compresses data streams inside PDFs, and will change inefficient compression modes to more modern versions. A program like ``qpdf`` can be used to change encodings, e.g. to inspect the internals fo a PDF. +The types of optimization available may expand over time. By default, +OCRmyPDF compresses data streams inside PDFs, and will change +inefficient compression modes to more modern versions. A program like +``qpdf`` can be used to change encodings, e.g. to inspect the internals +fo a PDF. .. code-block:: bash diff --git a/docs/docker.rst b/docs/docker.rst index b8f3657b..3b34ebba 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -1,134 +1,175 @@ +===================== OCRmyPDF Docker image ===================== -OCRmyPDF is also available in a Docker image that packages recent versions of all dependencies. +OCRmyPDF is also available in a Docker image that packages recent +versions of all dependencies. -For users who already have Docker installed this may be an easy and convenient option. However, it is less performant than a system installation and may require Docker engine configuration. +For users who already have Docker installed this may be an easy and +convenient option. However, it is less performant than a system +installation and may require Docker engine configuration. -OCRmyPDF needs a generous amount of RAM, CPU cores, temporary storage space, whether running in a Docker container or on its own. It may be necessary to ensure the container is provisioned with additional resources. +OCRmyPDF needs a generous amount of RAM, CPU cores, temporary storage +space, whether running in a Docker container or on its own. It may be +necessary to ensure the container is provisioned with additional +resources. .. _docker-install: Installing the Docker image ---------------------------- +=========================== -If you have `Docker `_ installed on your system, you can install a Docker image of the latest release. +If you have `Docker `__ installed on your +system, you can install a Docker image of the latest release. -The recommended OCRmyPDF Docker image is currently named ``ocrmypdf-alpine``: +The recommended OCRmyPDF Docker image is currently named +``ocrmypdf-alpine``: .. code-block:: bash - docker pull jbarlow83/ocrmypdf-alpine + docker pull jbarlow83/ocrmypdf-alpine -Follow the Docker installation instructions for your platform. If you can run this command successfully, your system is ready to download and execute the image: +Follow the Docker installation instructions for your platform. If you +can run this command successfully, your system is ready to download and +execute the image: .. code-block:: bash - docker run hello-world + docker run hello-world -OCRmyPDF will use all available CPU cores. By default, the VirtualBox machine instance on Windows and macOS has only a single CPU core enabled. Use the VirtualBox Manager to determine the name of your Docker engine host, and then follow these optional steps to enable multiple CPUs: +OCRmyPDF will use all available CPU cores. By default, the VirtualBox +machine instance on Windows and macOS has only a single CPU core +enabled. Use the VirtualBox Manager to determine the name of your Docker +engine host, and then follow these optional steps to enable multiple +CPUs: .. code-block:: bash - # Optional step for Mac OS X users - docker-machine stop "yourVM" - VBoxManage modifyvm "yourVM" --cpus 2 # or whatever number of core is desired - docker-machine start "yourVM" - eval $(docker-machine env "yourVM") + # Optional step for Mac OS X users + docker-machine stop "yourVM" + VBoxManage modifyvm "yourVM" --cpus 2 # or whatever number of core is desired + docker-machine start "yourVM" + eval $(docker-machine env "yourVM") Using the Docker image on the command line ------------------------------------------- +========================================== -**Unlike typical Docker containers**, in this mode we are using the OCRmyPDF Docker container is intended to be emphemeral – it runs for one OCR job and then terminates, just like a command line program. We are using Docker as a way of delivering an application, not a server. +**Unlike typical Docker containers**, in this mode we are using the +OCRmyPDF Docker container is intended to be emphemeral – it runs for one +OCR job and then terminates, just like a command line program. We are +using Docker as a way of delivering an application, not a server. To start a Docker container (instance of the image): .. code-block:: bash - docker tag jbarlow83/ocrmypdf-alpine ocrmypdf - docker run --rm -i ocrmypdf (... all other arguments here...) + docker tag jbarlow83/ocrmypdf-alpine ocrmypdf + docker run --rm -i ocrmypdf (... all other arguments here...) -For convenience, create a shell alias to hide the Docker command. It is easier to send the input file to file stdin and read the output from stdout – this avoids the occasionally messy permission issues with Docker entirely. +For convenience, create a shell alias to hide the Docker command. It is +easier to send the input file to file stdin and read the output from +stdout – this avoids the occasionally messy permission issues with +Docker entirely. .. code-block:: bash - alias ocrmypdf='docker run --rm -i ocrmypdf' - ocrmypdf --version # runs docker version - ocrmypdf output.pdf + alias ocrmypdf='docker run --rm -i ocrmypdf' + ocrmypdf --version # runs docker version + ocrmypdf output.pdf -Or in the wonderful `fish shell `_: +Or in the wonderful `fish shell `__: .. code-block:: fish - alias ocrmypdf 'docker run --rm ocrmypdf' - funcsave ocrmypdf + alias ocrmypdf 'docker run --rm ocrmypdf' + funcsave ocrmypdf -Alternately, you could mount the local current working directory as a Docker volume: +Alternately, you could mount the local current working directory as a +Docker volume: .. code-block:: bash - docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf + docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf .. _docker-lang-packs: Adding languages to the Docker image ------------------------------------- +==================================== -By default the Docker image includes English, German and Simplified Chinese, the most popular languages for OCRmyPDF users based on feedback. You may add other languages by creating a new Dockerfile based on the public one: +By default the Docker image includes English, German and Simplified +Chinese, the most popular languages for OCRmyPDF users based on +feedback. You may add other languages by creating a new Dockerfile based +on the public one: .. code-block:: dockerfile - FROM jbarlow83/ocrmypdf-alpine + FROM jbarlow83/ocrmypdf-alpine - # Add French - RUN apk add tesseract-ocr-data-fra + # Add French + RUN apk add tesseract-ocr-data-fra You can also copy training data to ``/usr/share/tessdata``. Executing the test suite ------------------------- +======================== -The OCRmyPDF test suite is installed with image. To run it: +The OCRmyPDF test suite is installed with image. To run it: .. code-block:: bash - docker run --entrypoint python3 jbarlow83/ocrmypdf-alpine setup.py test + docker run --entrypoint python3 jbarlow83/ocrmypdf-alpine setup.py test Accessing the shell -------------------- +=================== -``bash`` is not installed in the image. To use the busybox shell in the Docker image: +``bash`` is not installed in the image. To use the busybox shell in the +Docker image: .. code-block:: bash - docker run -it --entrypoint busybox jbarlow83/ocrmypdf-alpine sh + docker run -it --entrypoint busybox jbarlow83/ocrmypdf-alpine sh Using the OCRmyPDF web service wrapper --------------------------------------- +====================================== -The OCRmyPDF Docker image includes an example, barebones HTTP web service. The webservice may be launched as follows: +The OCRmyPDF Docker image includes an example, barebones HTTP web +service. The webservice may be launched as follows: .. code-block:: bash - docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf-alpine webservice.py + docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf-alpine webservice.py -Unlike command line usage this program will open a socket and wait for connections. +Unlike command line usage this program will open a socket and wait for +connections. .. warning:: - The OCRmyPDF web service wrapper is intended for demonstration or development. It provides no security, no authentication, no protection against denial of service attacks, and no load balancing. The default Flask WSGI server is used, which is intended for development only. The server is single-threaded and so can respond to only one client at a time. While running OCR, it cannot respond to any other clients. + The OCRmyPDF web service wrapper is intended for demonstration or + development. It provides no security, no authentication, no + protection against denial of service attacks, and no load balancing. + The default Flask WSGI server is used, which is intended for + development only. The server is single-threaded and so can respond to + only one client at a time. While running OCR, it cannot respond to + any other clients. -Clients must keep their open connection while waiting for OCR to complete. This may entail setting a long timeout; this interface is more useful for internal HTTP API calls. +Clients must keep their open connection while waiting for OCR to +complete. This may entail setting a long timeout; this interface is more +useful for internal HTTP API calls. -Unlike the rest of OCRmyPDF, this web service is licensed under the Affero GPLv3 (AGPLv3) since Ghostscript, a dependency of OCRmyPDF, is also licensed in this way. +Unlike the rest of OCRmyPDF, this web service is licensed under the +Affero GPLv3 (AGPLv3) since Ghostscript, a dependency of OCRmyPDF, is +also licensed in this way. -In addition to the above, please read our :ref:`general remarks on using OCRmyPDF as a service `. +In addition to the above, please read our +:ref:`general remarks on using OCRmyPDF as a service `. Ubuntu-based Docker image -------------------------- +========================= -A Ubuntu-based OCRmyPDF image is also available. The main advantage this image offers is that it supports manylinux Python wheels (which are not supported on Alpine Linux). This may be useful for plugins. +A Ubuntu-based OCRmyPDF image is also available. The main advantage this +image offers is that it supports manylinux Python wheels (which are not +supported on Alpine Linux). This may be useful for plugins. .. code-block:: bash - docker pull jbarlow83/ocrmypdf + docker pull jbarlow83/ocrmypdf diff --git a/docs/errors.rst b/docs/errors.rst index 44bc5732..825cd656 100644 --- a/docs/errors.rst +++ b/docs/errors.rst @@ -1,33 +1,47 @@ +===================== Common error messages ===================== Page already has text ---------------------- +===================== -.. code:: +.. code-block:: - ERROR - 1: page already has text! – aborting (use --force-ocr to force OCR) + ERROR - 1: page already has text! – aborting (use --force-ocr to force OCR) -You ran ocrmypdf on a file that already contains printable text or a hidden OCR text layer (it can't quite tell the difference). You probably don't want to do this, because the file is already searchable. +You ran ocrmypdf on a file that already contains printable text or a +hidden OCR text layer (it can't quite tell the difference). You probably +don't want to do this, because the file is already searchable. As the error message suggests, your options are: -- ``ocrmypdf --force-ocr`` to :ref:`rasterize ` all vector content and run OCR on the images. This is useful if a previous OCR program failed, or if the document contains a text watermark. - -- ``ocrmypdf --skip-text`` to skip OCR and other processing on any pages that contain text. Text pages will be copied into the output PDF without modification. - +- ``ocrmypdf --force-ocr`` to :ref:`rasterize ` all + vector content and run OCR on the images. This is useful if a + previous OCR program failed, or if the document contains a text + watermark. +- ``ocrmypdf --skip-text`` to skip OCR and other processing on any + pages that contain text. Text pages will be copied into the output + PDF without modification. Input file 'filename' is not a valid PDF ----------------------------------------- +======================================== -OCRmyPDF passes files through qpdf, a program that fixes errors in PDFs, before it tries to work on them. In most cases this happens because the PDF is corrupt and -truncated (incomplete file copying) and not much can be done. +OCRmyPDF passes files through qpdf, a program that fixes errors in PDFs, +before it tries to work on them. In most cases this happens because the +PDF is corrupt and truncated (incomplete file copying) and not much can +be done. -You can try rewriting the file with Ghostscript or pdftk: +You can try rewriting the file with Ghostscript: -- ``gs -o output.pdf -dSAFER -sDEVICE=pdfwrite input.pdf`` +.. code-block:: bash -- ``pdftk input.pdf cat output output.pdf`` + gs -o output.pdf -dSAFER -sDEVICE=pdfwrite input.pdf -Sometimes Acrobat can repair PDFs with its `Preflight tool `_. +``pdftk`` can also rewrite PDFs: +.. code-block:: bash + + pdftk input.pdf cat output output.pdf + +Sometimes Acrobat can repair PDFs with its `Preflight +tool `__. diff --git a/docs/installation.rst b/docs/installation.rst index b669320c..7262fdd3 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -1,3 +1,4 @@ +=================== Installing OCRmyPDF =================== @@ -18,10 +19,10 @@ installing the Python binary wheels. :local: Installing on Linux -------------------- +=================== Debian and Ubuntu 16.10 or newer -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +-------------------------------- .. |deb-stable| image:: https://repology.org/badge/version-for-repo/debian_stable/ocrmypdf.svg :alt: Debian 9 stable ("stretch") @@ -52,22 +53,33 @@ Debian and Ubuntu 16.10 or newer | |ubu-1710| |ubu-1804| |ubu-1810| | +-------------------------------------------+ -Users of Debian 9 ("stretch") or later or Ubuntu 16.10 or later may simply +Users of Debian 9 ("stretch") or later or Ubuntu 16.10 or later may +simply .. code-block:: bash apt-get install ocrmypdf -As indicated in the table above, Debian and Ubuntu releases may lag behind the latest version. If the version available for your platform is out of date, you could opt to install the latest version from source. See `Installing HEAD revision from sources`_. +As indicated in the table above, Debian and Ubuntu releases may lag +behind the latest version. If the version available for your platform is +out of date, you could opt to install the latest version from source. +See `Installing HEAD revision from +sources <#installing-head-revision-from-sources>`__. -For full details on version availability for your platform, check the `Debian Package Tracker `_ or `Ubuntu launchpad.net `_. +For full details on version availability for your platform, check the +`Debian Package Tracker `__ or +`Ubuntu launchpad.net `__. .. note:: - OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder. OCRmyPDF works fine without it but will produce larger output files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later will automatically detect it (specifically the ``jbig2`` binary) on the ``PATH``. To add JBIG2 encoding, see :ref:`jbig2`. + OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder. + OCRmyPDF works fine without it but will produce larger output files. + If you build jbig2enc from source, ocrmypdf 7.0.0 and later will + automatically detect it (specifically the ``jbig2`` binary) on the + ``PATH``. To add JBIG2 encoding, see :ref:`jbig2`. Fedora 29 or newer -^^^^^^^^^^^^^^^^^^ +------------------ .. |fedora-29| image:: https://repology.org/badge/version-for-repo/fedora29/ocrmypdf.svg :alt: Fedora 29 @@ -90,26 +102,26 @@ Users of Fedora 29 later may simply dnf install ocrmypdf -For full details on version availability, check the `Fedora Package Tracker -`_. +For full details on version availability, check the `Fedora Package +Tracker `__. -If the version available for your platform is out of date, you could opt to -install the latest version from source. See `Installing HEAD revision from -sources`_. +If the version available for your platform is out of date, you could opt +to install the latest version from source. See `Installing HEAD revision +from sources <#installing-head-revision-from-sources>`__. .. note:: - OCRmyPDF for Fedora currently omits the JBIG2 encoder due to patent issues. - OCRmyPDF works fine without it but will produce larger output files. If you - build jbig2enc from source, ocrmypdf 7.0.0 and later will automatically - detect it on the ``PATH``. To add JBIG2 encoding, see `Installing the JBIG2 - encoder `_. + OCRmyPDF for Fedora currently omits the JBIG2 encoder due to patent + issues. OCRmyPDF works fine without it but will produce larger output + files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later + will automatically detect it on the ``PATH``. To add JBIG2 encoding, + see `Installing the JBIG2 encoder `__. Installing the latest version on Ubuntu 18.04 LTS -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +------------------------------------------------- -Ubuntu 18.04 includes ocrmypdf 6.1.2. To install a more recent version, first -install the system version to get most of the dependencies: +Ubuntu 18.04 includes ocrmypdf 6.1.2. To install a more recent version, +first install the system version to get most of the dependencies: .. code-block:: bash @@ -118,8 +130,8 @@ install the system version to get most of the dependencies: ocrmypdf \ python3-pip -There are a few system dependency changes since ocrmypdf 6.1.2. Let's get -these, too. +There are a few system dependency changes since ocrmypdf 6.1.2. Let's +get these, too. .. code-block:: bash @@ -127,7 +139,8 @@ these, too. libxml2 \ pngquant -Then install the most recent ocrmypdf for the local user and set the user's ``PATH`` to check for the user's Python packages. +Then install the most recent ocrmypdf for the local user and set the +user's ``PATH`` to check for the user's Python packages. .. code-block:: bash @@ -137,13 +150,13 @@ Then install the most recent ocrmypdf for the local user and set the user's ``PA To add JBIG2 encoding, see :ref:`jbig2`. Ubuntu 16.04 LTS -^^^^^^^^^^^^^^^^ +---------------- -No package is available for Ubuntu 16.04. OCRmyPDF 8.0 and newer require Python -3.6. Ubuntu 16.04 ships Python 3.5, but you can install Python 3.6 on it. Or, -you can skip Python 3.6 and install OCRmyPDF 7.x or older - for that procedure, -please see the installation documentation for the version of OCRmyPDF you plan -to use. +No package is available for Ubuntu 16.04. OCRmyPDF 8.0 and newer require +Python 3.6. Ubuntu 16.04 ships Python 3.5, but you can install Python +3.6 on it. Or, you can skip Python 3.6 and install OCRmyPDF 7.x or older +- for that procedure, please see the installation documentation for the +version of OCRmyPDF you plan to use. **Install system packages for OCRmyPDF** @@ -165,13 +178,13 @@ to use. tesseract-ocr \ unpaper -This will install a Python 3.6 binary at ``/usr/bin/python3.6`` alongside the -system's Python 3.5. Do not remove the system Python. This will also install -Tesseract 4.0 from a PPA, since the version available in Ubuntu 16.04 is too old -for OCRmyPDF. +This will install a Python 3.6 binary at ``/usr/bin/python3.6`` +alongside the system's Python 3.5. Do not remove the system Python. This +will also install Tesseract 4.0 from a PPA, since the version available +in Ubuntu 16.04 is too old for OCRmyPDF. -Now install pip for Python 3.6. This will install the Python 3.6 version of -``pip`` at ``/usr/local/bin/pip``. +Now install pip for Python 3.6. This will install the Python 3.6 version +of ``pip`` at ``/usr/local/bin/pip``. .. code-block:: bash @@ -179,8 +192,8 @@ Now install pip for Python 3.6. This will install the Python 3.6 version of **Install OCRmyPDF** -OCRmyPDF requires the locale to be set for UTF-8. **On some minimal Ubuntu -installations systems**, it may be necessary to set the locale. +OCRmyPDF requires the locale to be set for UTF-8. **On some minimal +Ubuntu installations systems**, it may be necessary to set the locale. .. code-block:: bash @@ -199,11 +212,12 @@ environment variable contains ``$HOME/.local/bin``. To add JBIG2 encoding, see :ref:`jbig2`. Ubuntu 14.04 LTS -^^^^^^^^^^^^^^^^ +---------------- -Installing on Ubuntu 14.04 LTS (trusty) is more difficult than some other -options, because of its age. Several backports are required. For explanations of -some steps of this procedure, see the similar steps for Ubuntu 16.04. +Installing on Ubuntu 14.04 LTS (trusty) is more difficult than some +other options, because of its age. Several backports are required. For +explanations of some steps of this procedure, see the similar steps for +Ubuntu 16.04. Install system dependencies: @@ -220,12 +234,12 @@ Install system dependencies: qpdf We will need backports of Ghostscript 9.16, libav-11 (for unpaper 6.1), -Tesseract 4.00 (alpha), and Python 3.6. This will replace Ghostscript and -Tesseract 3.x on your system. Python 3.6 will be installed alongside the system -Python 3.4. +Tesseract 4.00 (alpha), and Python 3.6. This will replace Ghostscript +and Tesseract 3.x on your system. Python 3.6 will be installed alongside +the system Python 3.4. -If you prefer to not modify your system in this matter, consider using a Docker -container. +If you prefer to not modify your system in this matter, consider using a +Docker container. .. code-block:: bash @@ -251,7 +265,10 @@ Now we need to install ``pip`` and let it install ocrmypdf: curl https://bootstrap.pypa.io/ez_setup.py -o - | python3.6 && python3.6 -m easy_install pip pip3.6 install ocrmypdf -These installation instructions omit the optional dependency ``unpaper``, which is only available at version 0.4.2 in Ubuntu 14.04. The author could not find a backport of ``unpaper``, and created a .deb package to do the job of installing unpaper 6.1 (for x86 64-bit only): +These installation instructions omit the optional dependency +``unpaper``, which is only available at version 0.4.2 in Ubuntu 14.04. +The author could not find a backport of ``unpaper``, and created a .deb +package to do the job of installing unpaper 6.1 (for x86 64-bit only): .. code-block:: bash @@ -261,44 +278,52 @@ These installation instructions omit the optional dependency ``unpaper``, which To add JBIG2 encoding, see :ref:`jbig2`. ArchLinux (AUR) -^^^^^^^^^^^^^^^ +--------------- .. image:: https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg :alt: ArchLinux :target: https://repology.org/metapackage/ocrmypdf -There is an `ArchLinux User Repository package for ocrmypdf `_. You can use the following command. +There is an `ArchLinux User Repository package for +ocrmypdf `__. You can use +the following command. .. code-block:: bash yaourt -S ocrmypdf -If you have any difficulties with installation, check the repository package page. +If you have any difficulties with installation, check the repository +package page. Other Linux packages -^^^^^^^^^^^^^^^^^^^^ +-------------------- -See the `Repology `_ page. +See the +`Repology `__ page. -In general, first install the OCRmyPDF package for your system, then optionally use the procedure `Installing with Python pip`_ to install a more recent version. +In general, first install the OCRmyPDF package for your system, then +optionally use the procedure `Installing with Python +pip <#installing-with-python-pip>`__ to install a more recent version. Installing on macOS -------------------- +=================== Homebrew -^^^^^^^^ +-------- .. image:: https://img.shields.io/homebrew/v/ocrmypdf.svg :alt: homebrew :target: http://brewformulas.org/Ocrmypdf -OCRmyPDF is now a standard `Homebrew `_ formula. To install on macOS: +OCRmyPDF is now a standard `Homebrew `__ formula. To +install on macOS: .. code-block:: bash brew install ocrmypdf -This will include only the English language pack. If you need other languages you can optionally install them all: +This will include only the English language pack. If you need other +languages you can optionally install them all: .. code-block:: bash @@ -306,18 +331,23 @@ This will include only the English language pack. If you need other languages yo .. note:: - Users who previously installed OCRmyPDF on macOS using ``pip install ocrmypdf`` should remove the pip version (``pip3 uninstall ocrmypdf``) before switching to the Homebrew version. + Users who previously installed OCRmyPDF on macOS using + ``pip install ocrmypdf`` should remove the pip version + (``pip3 uninstall ocrmypdf``) before switching to the Homebrew + version. .. note:: - Users who previously installed OCRmyPDF from the private tap should switch to the mainline version (``brew untap jbarlow83/ocrmypdf``) and install from there. + Users who previously installed OCRmyPDF from the private tap should + switch to the mainline version (``brew untap jbarlow83/ocrmypdf``) + and install from there. Manual installation on macOS -^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +---------------------------- These instructions probably work on all macOS supported by Homebrew. -If it's not already present, `install Homebrew `_. +If it's not already present, `install Homebrew `__. Update Homebrew: @@ -325,18 +355,22 @@ Update Homebrew: brew update -Install or upgrade the required Homebrew packages, if any are missing. To do this, download the ``Brewfile`` that lists all of the dependencies to the current directory, and run ``brew bundle`` to process them (installing or upgrading as needed). ``Brewfile`` is a plain text file. +Install or upgrade the required Homebrew packages, if any are missing. +To do this, download the ``Brewfile`` that lists all of the dependencies +to the current directory, and run ``brew bundle`` to process them +(installing or upgrading as needed). ``Brewfile`` is a plain text file. .. code-block:: bash wget https://github.com/jbarlow83/OCRmyPDF/raw/master/.travis/Brewfile brew bundle -This will include the English, French, German and Spanish language packs. If you need other languages you can optionally install them all: +This will include the English, French, German and Spanish language +packs. If you need other languages you can optionally install them all: .. _macos-all-languages: -.. code-block:: bash + .. code-block:: bash brew install tesseract --with-all-languages # Option 2: for all language packs @@ -365,103 +399,147 @@ The command line program should now be available: ocrmypdf --help Installing on FreeBSD ---------------------- +===================== -FreeBSD 11.2 is known to work. Other versions likely work but have not been tested. +FreeBSD 11.2 is known to work. Other versions likely work but have not +been tested. In general it should work to: -#. `Install and build pikepdf `_. +#. `Install and build + pikepdf `__. #. Install the equivalent list of dependencies for Linux. Installing the Docker image ---------------------------- +=========================== -For some users, installing the Docker image will be easier than installing all of OCRmyPDF's dependencies. For Windows, it is the only option. +For some users, installing the Docker image will be easier than +installing all of OCRmyPDF's dependencies. For Windows, it is the only +option. -See `OCRmyPDF Docker Image `_ for more information. +See `OCRmyPDF Docker Image `__ for more information. Installing on Windows ---------------------- +===================== -Direct installation on Windows is not possible. `Install the Docker `_ container as described above. Ensure that your command prompt can run the docker "hello world" container. +Direct installation on Windows is not possible. `Install the +Docker `__ container as described above. Ensure that +your command prompt can run the docker "hello world" container. -It would probably not be too difficult to port on Windows. The main reason this has been avoided is the difficulty of packaging and installing the various non-Python dependencies: Tesseract, QPDF, Ghostscript, Leptonica. Pull requests to add or improve Windows support would be quite welcome. +It would probably not be too difficult to port on Windows. The main +reason this has been avoided is the difficulty of packaging and +installing the various non-Python dependencies: Tesseract, QPDF, +Ghostscript, Leptonica. Pull requests to add or improve Windows support +would be quite welcome. Installing with Python pip --------------------------- +========================== -OCRmyPDF is delivered by PyPI because it is a convenient way to install the latest version. However, PyPI and ``pip`` cannot address the fact that ``ocrmypdf`` depends on certain non-Python system libraries and programs being instsalled. +OCRmyPDF is delivered by PyPI because it is a convenient way to install +the latest version. However, PyPI and ``pip`` cannot address the fact +that ``ocrmypdf`` depends on certain non-Python system libraries and +programs being instsalled. -For best results, first install `your platform's version `_ of ``ocrmypdf``, using the instructions elsewhere in this document. Then you can use ``pip`` to get the latest version if your platform version is out of date. Chances are that this will satisfy most dependencies. +For best results, first install `your platform's +version `__ of +``ocrmypdf``, using the instructions elsewhere in this document. Then +you can use ``pip`` to get the latest version if your platform version +is out of date. Chances are that this will satisfy most dependencies. Use ``ocrmypdf --version`` to confirm what version was installed. -Then you can install the latest OCRmyPDF from the Python wheels. First try: +Then you can install the latest OCRmyPDF from the Python wheels. First +try: .. code-block:: bash pip3 install --user ocrmypdf -You should then be able to run ``ocrmypdf --version`` and see that the latest version was located. +You should then be able to run ``ocrmypdf --version`` and see that the +latest version was located. -Since ``pip3 install --user`` does not work correctly on some platforms, notably Ubuntu 16.04 and older, and the Homebrew version of Python, instead use this for a system wide installation: +Since ``pip3 install --user`` does not work correctly on some platforms, +notably Ubuntu 16.04 and older, and the Homebrew version of Python, +instead use this for a system wide installation: .. code-block:: bash pip3 install ocrmypdf Requirements for pip and HEAD install -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +------------------------------------- -OCRmyPDF currently requires these external programs and libraries to be installed, and must be satisfied using the operating system package manager. ``pip`` cannot provide them. +OCRmyPDF currently requires these external programs and libraries to be +installed, and must be satisfied using the operating system package +manager. ``pip`` cannot provide them. -- Python 3.6 or newer -- Ghostscript 9.15 or newer -- qpdf 8.1.0 or newer -- Tesseract 4.0.0-alpha or newer +- Python 3.6 or newer +- Ghostscript 9.15 or newer +- qpdf 8.1.0 or newer +- Tesseract 4.0.0-alpha or newer As of ocrmypdf 7.2.1, the following versions are recommended: -- Python 3.7 -- Ghostscript 9.23 or newer -- qpdf 8.2.1 -- Tesseract 4.0.0 or newer -- jbig2enc 0.29 or newer -- pngquant 2.5 or newer -- unpaper 6.1 +- Python 3.7 +- Ghostscript 9.23 or newer +- qpdf 8.2.1 +- Tesseract 4.0.0 or newer +- jbig2enc 0.29 or newer +- pngquant 2.5 or newer +- unpaper 6.1 -jbig2enc, pngquant, and unpaper are optional. If missing certain features are disabled. OCRmyPDF will discover them as soon as they are available. +jbig2enc, pngquant, and unpaper are optional. If missing certain +features are disabled. OCRmyPDF will discover them as soon as they are +available. -**jbig2enc**, if present, will be used to optimize the encoding of monochrome images. This can significantly reduce the file size of the output file. It is not required. `jbig2enc `_ is not generally available for Ubuntu or Debian due to lingering concerns about patent issues, but can easily be built from source. To add JBIG2 encoding, see :ref:`jbig2`. +**jbig2enc**, if present, will be used to optimize the encoding of +monochrome images. This can significantly reduce the file size of the +output file. It is not required. +`jbig2enc `__ is not generally +available for Ubuntu or Debian due to lingering concerns about patent +issues, but can easily be built from source. To add JBIG2 encoding, see +:ref:`jbig2`. -**pngquant**, if present, is optionally used to optimize the encoding of PNG-style images in PDFs (actually, any that are that losslessly encoded) by lossily quantizing to a smaller color palette. It is only activated then the ``--optimize`` argument is ``2`` or ``3``. +**pngquant**, if present, is optionally used to optimize the encoding of +PNG-style images in PDFs (actually, any that are that losslessly +encoded) by lossily quantizing to a smaller color palette. It is only +activated then the ``--optimize`` argument is ``2`` or ``3``. -**unpaper**, if present, enables the ``--clean`` and ``--clean-final`` command line options. - -These are in addition to the Python packaging dependencies, meaning that unfortunately, the ``pip install`` command cannot satisfy all of them. +**unpaper**, if present, enables the ``--clean`` and ``--clean-final`` +command line options. +These are in addition to the Python packaging dependencies, meaning that +unfortunately, the ``pip install`` command cannot satisfy all of them. Installing HEAD revision from sources -------------------------------------- +===================================== -If you have ``git`` and Python 3.6 or newer installed, you can install from source. When the ``pip`` installer runs, it will alert you if dependencies are missing. +If you have ``git`` and Python 3.6 or newer installed, you can install +from source. When the ``pip`` installer runs, it will alert you if +dependencies are missing. -If you prefer to build every from source, you will need to `build pikepdf from source `_. First ensure you can build and install pikepdf. +If you prefer to build every from source, you will need to `build +pikepdf from +source `__. +First ensure you can build and install pikepdf. -To install the HEAD revision from sources in the current Python 3 environment: +To install the HEAD revision from sources in the current Python 3 +environment: .. code-block:: bash pip3 install git+https://github.com/jbarlow83/OCRmyPDF.git -Or, to install in `development mode `_, allowing customization of OCRmyPDF, use the ``-e`` flag: +Or, to install in `development +mode `__, +allowing customization of OCRmyPDF, use the ``-e`` flag: .. code-block:: bash pip3 install -e git+https://github.com/jbarlow83/OCRmyPDF.git -You may find it easiest to install in a virtual environment, rather than system-wide: +You may find it easiest to install in a virtual environment, rather than +system-wide: .. code-block:: bash @@ -471,8 +549,8 @@ You may find it easiest to install in a virtual environment, rather than system- cd OCRmyPDF pip3 install . -However, ``ocrmypdf`` will only be accessible on the system PATH -when you activate the virtual environment. +However, ``ocrmypdf`` will only be accessible on the system PATH when +you activate the virtual environment. To run the program: @@ -486,7 +564,7 @@ dependencies. Older version than the ones mentioned in the release notes are likely not to be compatible to OCRmyPDF. For development -^^^^^^^^^^^^^^^ +--------------- To install all of the development and test requirements: @@ -502,15 +580,17 @@ To install all of the development and test requirements: To add JBIG2 encoding, see :ref:`jbig2`. Shell completions ------------------ +================= Completions for ``bash`` and ``fish`` are available in the project's ``misc/completion`` folder. The ``bash`` completions are likely ``zsh`` -compatible but this has not been confirmed. Package maintainers, please install -these at the appropriate locations for your system. +compatible but this has not been confirmed. Package maintainers, please +install these at the appropriate locations for your system. -To manually install the ``bash`` completion, copy ``misc/completion/ocrmypdf.bash`` to -``/etc/bash_completion.d/ocrmypdf`` (rename the file). +To manually install the ``bash`` completion, copy +``misc/completion/ocrmypdf.bash`` to ``/etc/bash_completion.d/ocrmypdf`` +(rename the file). -To manually install the ``fish`` completion, copy ``misc/completion/ocrmypdf.fish`` to +To manually install the ``fish`` completion, copy +``misc/completion/ocrmypdf.fish`` to ``~/.config/fish/completions/ocrmypdf.fish``. diff --git a/docs/introduction.rst b/docs/introduction.rst index 03780225..0643c7e7 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -1,118 +1,228 @@ +============ Introduction ============ + OCRmyPDF is a Python 3 package that adds OCR layers to PDFs. About OCR ---------- +========= -`Optical character recognition `_ is technology that converts images of typed or handwritten text, such as in a scanned document, to computer text that can be searched and copied. +`Optical character +recognition `__ +is technology that converts images of typed or handwritten text, such as +in a scanned document, to computer text that can be searched and copied. -OCRmyPDF uses `Tesseract `_, the best available open source OCR engine, to perform OCR. +OCRmyPDF uses +`Tesseract `__, the best +available open source OCR engine, to perform OCR. .. _raster-vector: About PDFs ----------- +========== -PDFs are page description files that attempts to preserve a layout exactly. They contain `vector graphics `_ that can contain raster objects such as scanned images. Because PDFs can contain multiple pages (unlike many image formats) and can contain fonts and text, it is a good formats for exchanging scanned documents. +PDFs are page description files that attempts to preserve a layout +exactly. They contain `vector +graphics `__ +that can contain raster objects such as scanned images. Because PDFs can +contain multiple pages (unlike many image formats) and can contain fonts +and text, it is a good formats for exchanging scanned documents. -.. image:: images/bitmap_vs_svg.svg +|image| -A PDF page might contain multiple images, even if it only appears to have one image. Some scanners or scanning software will segment pages into monochromatic text and color regions for example, to improve the compression ratio and appearance of the page. - -Rasterizing a PDF is the process of generating an image suitable for display or analyzing with an OCR engine. OCR engines like Tesseract work with images, not vector objects. +A PDF page might contain multiple images, even if it only appears to +have one image. Some scanners or scanning software will segment pages +into monochromatic text and color regions for example, to improve the +compression ratio and appearance of the page. +Rasterizing a PDF is the process of generating an image suitable for +display or analyzing with an OCR engine. OCR engines like Tesseract work +with images, not vector objects. About PDF/A ------------ +=========== -`PDF/A `_ is an ISO-standardized subset of the full PDF specification that is designed for archiving (the 'A' stands for Archive). PDF/A differs from PDF primarily by omitting features that would make it difficult to read the file in the future, such as embedded Javascript, video, audio and references to external fonts. All fonts and resources needed to interpret the PDF must be contained within it. Because PDF/A disables Javascript and other types of embedded content, it is probably more secure. +`PDF/A `__ is an ISO-standardized +subset of the full PDF specification that is designed for archiving (the +'A' stands for Archive). PDF/A differs from PDF primarily by omitting +features that would make it difficult to read the file in the future, +such as embedded Javascript, video, audio and references to external +fonts. All fonts and resources needed to interpret the PDF must be +contained within it. Because PDF/A disables Javascript and other types +of embedded content, it is probably more secure. There are various conformance levels and versions, such as "PDF/A-2b". -Generally speaking, the best format for scanned documents is PDF/A. Some governments and jurisdictions, US Courts in particular, `mandate the use of PDF/A `_ for scanned documents. +Generally speaking, the best format for scanned documents is PDF/A. Some +governments and jurisdictions, US Courts in particular, `mandate the use +of PDF/A `__ for scanned +documents. -Since most people who scan documents are interested in reading them indefinitely into the future, OCRmyPDF generates PDF/A-2b by default. - -PDF/A has a few drawbacks. Some PDF viewers include an alert that the file is a PDF/A, which may confuse some users. It also tends to produce larger files than PDF, because it embeds certain resources even if they are commonly available. PDF/A files can be digitally signed, but may not be encrypted, to ensure they can be read in the future. Fortunately, converting from PDF/A to a regular PDF is trivial, and any PDF viewer can view PDF/A. +Since most people who scan documents are interested in reading them +indefinitely into the future, OCRmyPDF generates PDF/A-2b by default. +PDF/A has a few drawbacks. Some PDF viewers include an alert that the +file is a PDF/A, which may confuse some users. It also tends to produce +larger files than PDF, because it embeds certain resources even if they +are commonly available. PDF/A files can be digitally signed, but may not +be encrypted, to ensure they can be read in the future. Fortunately, +converting from PDF/A to a regular PDF is trivial, and any PDF viewer +can view PDF/A. What OCRmyPDF does ------------------- +================== -OCRmyPDF analyzes each page of a PDF to determine the colorspace and resolution (DPI) needed to capture all of the information on that page without losing content. It uses `Ghostscript `_ to rasterize the page, and then performs on OCR on the rasterized image to create an OCR "layer". The layer is then grafted back onto the original PDF. +OCRmyPDF analyzes each page of a PDF to determine the colorspace and +resolution (DPI) needed to capture all of the information on that page +without losing content. It uses +`Ghostscript `__ to rasterize the page, and +then performs on OCR on the rasterized image to create an OCR "layer". +The layer is then grafted back onto the original PDF. -While one can use a program like Ghostscript or ImageMagick to get an image and put the image through Tesseract, that actually creates a new PDF and many details may be lost. OCRmyPDF can produce a minimally changed PDF as output. +While one can use a program like Ghostscript or ImageMagick to get an +image and put the image through Tesseract, that actually creates a new +PDF and many details may be lost. OCRmyPDF can produce a minimally +changed PDF as output. -OCRmyPDF also some image processing options like deskew which improve the appearance of files and quality of OCR. When these are used, the OCR layer is grafted onto the processed image instead. - -By default, OCRmyPDF produces archival PDFs – PDF/A, which are a stricter subset of PDF features designed for long term archives. If regular PDFs are desired, this can be disabled with ``--output-type pdf``. +OCRmyPDF also some image processing options like deskew which improve +the appearance of files and quality of OCR. When these are used, the OCR +layer is grafted onto the processed image instead. +By default, OCRmyPDF produces archival PDFs – PDF/A, which are a +stricter subset of PDF features designed for long term archives. If +regular PDFs are desired, this can be disabled with +``--output-type pdf``. Why you shouldn't do this manually ----------------------------------- +================================== -A PDF is similar to an HTML file, in that it contains document structure along with images. Sometimes a PDF does nothing more than present a full page image, but often there is additional content that would be lost. +A PDF is similar to an HTML file, in that it contains document structure +along with images. Sometimes a PDF does nothing more than present a full +page image, but often there is additional content that would be lost. A manual process could work like either of these: -1. Rasterize each page as an image, OCR the images, and combine the output into a PDF. This preserves the layout of each page, but resamples all images (possibly losing quality, increasing file size, introducing compression artifacts, etc.). +1. Rasterize each page as an image, OCR the images, and combine the + output into a PDF. This preserves the layout of each page, but + resamples all images (possibly losing quality, increasing file size, + introducing compression artifacts, etc.). +2. Extract each image, OCR, and combine the output into a PDF. This + loses the context in which images are used in the PDF, meaning that + cropping, rotation and scaling of pages may be lost. Some scanned + PDFs use multiple images segmented into black and white, grayscale + and color regions, with stencil masks to prevent overlap, as this can + enhance the appearance of a file while reducing file size. Clearly, + reassembling these images will be easy. This also loses and text or + vector art on any pages in a PDF with both scanned and pure digital + content. -2. Extract each image, OCR, and combine the output into a PDF. This loses the context in which images are used in the PDF, meaning that cropping, rotation and scaling of pages may be lost. Some scanned PDFs use multiple images segmented into black and white, grayscale and color regions, with stencil masks to prevent overlap, as this can enhance the appearance of a file while reducing file size. Clearly, reassembling these images will be easy. This also loses and text or vector art on any pages in a PDF with both scanned and pure digital content. +In the case of a PDF that is nothing other than a container of images +(no rotation, scaling, cropping, one image per page), the second +approach can be lossless. -In the case of a PDF that is nothing other than a container of images (no rotation, scaling, cropping, one image per page), the second approach can be lossless. - -OCRmyPDF uses several strategies depending on input options and the input PDF itself, but generally speaking it rasterizes a page for OCR and then grafts the OCR back onto the original. As such it can handle complex PDFs and still preserve their contents as much as possible. - -OCRmyPDF also supports a many, many edge cases that have cropped over several years of development. We support PDF features like images inside of Form XObjects, and pages with UserUnit scaling. We support rare image formats like non-monochrome 1-bit images. We warn about files you may not to OCR. Thanks to pikepdf and QPDF, we auto-repair PDFs that are damaged. (Not that you need to know what any of these are! You should be able to throw any PDF at it.) +OCRmyPDF uses several strategies depending on input options and the +input PDF itself, but generally speaking it rasterizes a page for OCR +and then grafts the OCR back onto the original. As such it can handle +complex PDFs and still preserve their contents as much as possible. +OCRmyPDF also supports a many, many edge cases that have cropped over +several years of development. We support PDF features like images inside +of Form XObjects, and pages with UserUnit scaling. We support rare image +formats like non-monochrome 1-bit images. We warn about files you may +not to OCR. Thanks to pikepdf and QPDF, we auto-repair PDFs that are +damaged. (Not that you need to know what any of these are! You should be +able to throw any PDF at it.) Limitations ------------ +=========== -OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences these limitations, as do any other programs that rely on Tesseract: +OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences +these limitations, as do any other programs that rely on Tesseract: -* The OCR is not as accurate as commercial solutions such as Abbyy. -* It is not capable of recognizing handwriting. -* It may find gibberish and report this as OCR output. -* If a document contains languages outside of those given in the ``-l LANG`` arguments, results may be poor. -* It is not always good at analyzing the natural reading order of documents. For example, it may fail to recognize that a document contains two columns, and may try to join text across columns. -* Poor quality scans may produce poor quality OCR. Garbage in, garbage out. -* It does not expose information about what font family text belongs to. +- The OCR is not as accurate as commercial solutions such as Abbyy. +- It is not capable of recognizing handwriting. +- It may find gibberish and report this as OCR output. +- If a document contains languages outside of those given in the + ``-l LANG`` arguments, results may be poor. +- It is not always good at analyzing the natural reading order of + documents. For example, it may fail to recognize that a document + contains two columns, and may try to join text across columns. +- Poor quality scans may produce poor quality OCR. Garbage in, garbage + out. +- It does not expose information about what font family text belongs + to. OCRmyPDF is also limited by the PDF specification: -* PDF encodes the position of text glyphs but does not encode document structure. There is no markup that divides a document in sections, paragraphs, sentences, or even words (since blank spaces are not represented). As such all elements of document structure including the spaces between words must be derived heuristically. Some PDF viewers do a better job of this than others. -* Because some popular open source PDF viewers have a particularly hard time with spaces betweem words, OCRmyPDF appends a space to each text element as a workaround (when using ``--pdf-renderer hocr``). While this mixes document structure with graphical information that ideally should be left to the PDF viewer to interpret, it improves compatibility with some viewers and does not cause problems for better ones. +- PDF encodes the position of text glyphs but does not encode document + structure. There is no markup that divides a document in sections, + paragraphs, sentences, or even words (since blank spaces are not + represented). As such all elements of document structure including + the spaces between words must be derived heuristically. Some PDF + viewers do a better job of this than others. +- Because some popular open source PDF viewers have a particularly hard + time with spaces betweem words, OCRmyPDF appends a space to each text + element as a workaround (when using ``--pdf-renderer hocr``). While + this mixes document structure with graphical information that ideally + should be left to the PDF viewer to interpret, it improves + compatibility with some viewers and does not cause problems for + better ones. Ghostscript also imposes some limitations: -* PDFs containing JBIG2-encoded content will be converted to CCITT Group4 encoding, which has lower compression ratios, if Ghostscript PDF/A is enabled. -* PDFs containing JPEG 2000-encoded content will be converted to JPEG encoding, which may introduce compression artifacts, if Ghostscript PDF/A is enabled. -* Ghostscript may transcode grayscale and color images, either lossy to lossless or lossless to lossy, based on an internal algorithm. This behavior can be suppressed by setting ``--pdfa-image-compression`` to ``jpeg`` or ``lossless`` to set all images to one type or the other. Ghostscript has no option to maintain the input image's format. (Ghostscript 9.25+ can copy JPEG images without transcoding them; earlier versions will transcode.) -* Ghostscript's PDF/A conversion removes any XMP metadata that is not one of the standard XMP metadata namespaces for PDFs. In particular, PRISM Metdata is removed. +- PDFs containing JBIG2-encoded content will be converted to CCITT + Group4 encoding, which has lower compression ratios, if Ghostscript + PDF/A is enabled. +- PDFs containing JPEG 2000-encoded content will be converted to JPEG + encoding, which may introduce compression artifacts, if Ghostscript + PDF/A is enabled. +- Ghostscript may transcode grayscale and color images, either lossy to + lossless or lossless to lossy, based on an internal algorithm. This + behavior can be suppressed by setting ``--pdfa-image-compression`` to + ``jpeg`` or ``lossless`` to set all images to one type or the other. + Ghostscript has no option to maintain the input image's format. + (Ghostscript 9.25+ can copy JPEG images without transcoding them; + earlier versions will transcode.) +- Ghostscript's PDF/A conversion removes any XMP metadata that is not + one of the standard XMP metadata namespaces for PDFs. In particular, + PRISM Metdata is removed. Regarding OCRmyPDF itself: -* PDFs that use transparency are not currently represented in the test suite +- PDFs that use transparency are not currently represented in the test + suite Similar programs ----------------- +================ -To the author's knowledge, OCRmyPDF is the most feature-rich and thoroughly tested command line OCR PDF conversion tool. If it does not meet your needs, contributions and suggestions are welcome. If not, consider one of these similar open source programs: +To the author's knowledge, OCRmyPDF is the most feature-rich and +thoroughly tested command line OCR PDF conversion tool. If it does not +meet your needs, contributions and suggestions are welcome. If not, +consider one of these similar open source programs: -* pdf2pdfocr -* pdfsandwich -* pypdfocr -* pdfbeads +- pdf2pdfocr +- pdfsandwich +- pypdfocr +- pdfbeads Web front-ends --------------- +============== -The Docker image ``ocrmypdf-alpine`` provides a web service front-end that allows files to submitted over HTTP and the results "downloaded". This is an HTTP server intended to simplify web services deployments; it is not intended to be deployed on the public internet and no real security measures to speak of. +The Docker image ``ocrmypdf-alpine`` provides a web service front-end +that allows files to submitted over HTTP and the results "downloaded". +This is an HTTP server intended to simplify web services deployments; it +is not intended to be deployed on the public internet and no real +security measures to speak of. In addition, the following third-party integrations are available: -* `Nextcloud OCR `_ is a free software plugin for the Nextcloud private cloud software +- `Nextcloud OCR `__ is a free software + plugin for the Nextcloud private cloud software -OCRmyPDF is not designed to be secure against malware-bearing PDFs (see `Using OCRmyPDF online `_). Users should ensure they comply with OCRmyPDF's licenses and the licenses of all dependencies. In particular, OCRmyPDF requires Ghostscript, which is licensed under AGPLv3. +OCRmyPDF is not designed to be secure against malware-bearing PDFs (see +`Using OCRmyPDF online `__). Users should ensure they +comply with OCRmyPDF's licenses and the licenses of all dependencies. In +particular, OCRmyPDF requires Ghostscript, which is licensed under +AGPLv3. + +.. |image| image:: images/bitmap_vs_svg.svg diff --git a/docs/jbig2.rst b/docs/jbig2.rst index 3ee524ca..81789b6f 100644 --- a/docs/jbig2.rst +++ b/docs/jbig2.rst @@ -1,35 +1,55 @@ .. _jbig2: +============================ Installing the JBIG2 encoder ============================ -Most Linux distributions do not include a JBIG2 encoder since JBIG2 encoding was patented for a long time. All known JBIG2 US patents have expired as of 2017, but it is possible that unknown patents exist. +Most Linux distributions do not include a JBIG2 encoder since JBIG2 +encoding was patented for a long time. All known JBIG2 US patents have +expired as of 2017, but it is possible that unknown patents exist. -JBIG2 encoding is recommended for OCRmyPDF and is used to losslessly create smaller PDFs. If JBIG2 encoding not available, lower quality encodings will be used. +JBIG2 encoding is recommended for OCRmyPDF and is used to losslessly +create smaller PDFs. If JBIG2 encoding not available, lower quality +encodings will be used. -JBIG2 decoding is not patented and is performed automatically by most PDF viewers. It is widely supported has been part of the PDF specification since 2001. +JBIG2 decoding is not patented and is performed automatically by most +PDF viewers. It is widely supported has been part of the PDF +specification since 2001. -On macOS, Homebrew packages jbig2enc and OCRmyPDF includes it by default. The Docker image for OCRmyPDF also builds its own JBIG2 encoder from source. +On macOS, Homebrew packages jbig2enc and OCRmyPDF includes it by +default. The Docker image for OCRmyPDF also builds its own JBIG2 encoder +from source. For all other Linux, you must build a JBIG2 encoder from source: .. code-block:: bash - git clone https://github.com/agl/jbig2enc - cd jbig2enc - ./autogen.sh - ./configure && make - [sudo] make install + git clone https://github.com/agl/jbig2enc + cd jbig2enc + ./autogen.sh + ./configure && make + [sudo] make install .. _jbig2-lossy: Lossy mode JBIG2 ----------------- +================ -OCRmyPDF provides lossy mode JBIG2 as an advanced feature. Users should `review the technical concerns with JBIG2 in lossy mode `_ and decide if this feature is acceptable for their use case. +OCRmyPDF provides lossy mode JBIG2 as an advanced feature. Users should +`review the technical concerns with JBIG2 in lossy +mode `__ +and decide if this feature is acceptable for their use case. -JBIG2 lossy mode does achieve higher compression ratios than any other monochrome (bitonal) compression technology; for large text documents the savings are considerable. JBIG2 lossless still gives great compression ratios and is a major improvement over the older CCITT G4 standard. As explained above, there is some risk of substitution errors. +JBIG2 lossy mode does achieve higher compression ratios than any other +monochrome (bitonal) compression technology; for large text documents +the savings are considerable. JBIG2 lossless still gives great +compression ratios and is a major improvement over the older CCITT G4 +standard. As explained above, there is some risk of substitution errors. -To turn on JBIG2 lossy mode, add the argument ``--jbig2-lossy``. ``--optimize {1,2,3}`` are necessary for the argument to take effect also required. Also, a JBIG2 encoder must be installed as described in the previous section. +To turn on JBIG2 lossy mode, add the argument ``--jbig2-lossy``. +``--optimize {1,2,3}`` are necessary for the argument to take effect +also required. Also, a JBIG2 encoder must be installed as described in +the previous section. -*Due to an oversight, ocrmypdf v7.0 and v7.1 used lossy mode by default.* +*Due to an oversight, ocrmypdf v7.0 and v7.1 used lossy mode by +default.* diff --git a/docs/languages.rst b/docs/languages.rst index 5b5a8853..2b26b6de 100644 --- a/docs/languages.rst +++ b/docs/languages.rst @@ -1,16 +1,20 @@ .. _lang-packs: +==================================== Installing additional language packs ==================================== -OCRmyPDF uses Tesseract for OCR, and relies on its language packs for languages other than English. +OCRmyPDF uses Tesseract for OCR, and relies on its language packs for +languages other than English. -Tesseract supports `most languages `_. +Tesseract supports `most +languages `__. -For Linux users, you can often find packages that provide language packs: +For Linux users, you can often find packages that provide language +packs: Debian and Ubuntu users ------------------------ +======================= .. code-block:: bash @@ -20,11 +24,13 @@ Debian and Ubuntu users # Install Chinese Simplified language pack apt-get install tesseract-ocr-chi-sim -You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple -languages can be requested using either ``-l eng+fre`` (English and French) or ``-l eng -l fre``. +You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as +to what languages it should search for. Multiple languages can be +requested using either ``-l eng+fre`` (English and French) or +``-l eng -l fre``. Fedora users ------------- +============ .. code-block:: bash @@ -34,16 +40,20 @@ Fedora users # Install Chinese Simplified language pack dnf install tesseract-langpack-chi_sim -You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as to -what languages it should search for. Multiple languages can be requested using -either ``-l eng+fre`` (English and French) or ``-l eng -l fre``. +You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as +to what languages it should search for. Multiple languages can be +requested using either ``-l eng+fre`` (English and French) or +``-l eng -l fre``. macOS users ------------ +=========== -You can install additional language packs by :ref:`installing Tesseract using Homebrew with all language packs `. +You can install additional language packs by +:ref:`installing Tesseract using Homebrew with all language packs `. Docker users ------------- +============ -Users of the OCRmyPDF Docker image should install language packs into a derived Docker image as :ref:`described in that section `. +Users of the OCRmyPDF Docker image should install language packs into a +derived Docker image as +:ref:`described in that section `. diff --git a/docs/plugins.rst b/docs/plugins.rst index b962081b..e22cea36 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -1,22 +1,24 @@ +======= Plugins ======= -You can use plugins to customize the behavior of OCRmyPDF at certain points of interest. +You can use plugins to customize the behavior of OCRmyPDF at certain +points of interest. -Currently, it is possible to: -- override the decision for whether or not to perform OCR on a particular file -- modify the image is about to be sent for OCR +Currently, it is possible to: - override the decision for whether or not +to perform OCR on a particular file - modify the image is about to be +sent for OCR How plugins are imported ------------------------- +======================== -Plugins are imported on demand, by the OCRmyPDF worker process that needs to use them. -As such, plugins cannot share state with each other, and will be imported many times, -once for each worker process. +Plugins are imported on demand, by the OCRmyPDF worker process that +needs to use them. As such, plugins cannot share state with each other, +and will be imported many times, once for each worker process. Plugins currently cannot override the same hook. How plugins are invoked ------------------------ +======================= Plugins may be called from the command line: diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 13c880d1..c8fca621 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -1,205 +1,282 @@ +============= Release notes ============= -OCRmyPDF uses `semantic versioning `_ for its command line interface and its public API. +OCRmyPDF uses `semantic versioning `__ for its +command line interface and its public API. -The ``ocrmypdf`` package may now be imported. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +The ``ocrmypdf`` package may now be imported. The public API may be +useful in scripts that launch OCRmyPDF processes or that wish to use +some of its features for working with PDFs. -Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. - -.. Issue regex - find: [^`]\#([0-9]{1,3})[^0-9] - replace: `#$1 `_ +Note that it is licensed under GPLv3, so scripts that +``import ocrmypdf`` and are released publicly should probably also be +licensed under GPLv3. v8.3.0 ------- +====== -- Improved the strategy for updating pages when a new image of the page was produced. We now attempt to preserve more content from the original file, for annotations in particular. - -- For PDFs with more than 100 pages and a sequence where one PDF page was replaced and one or more subsequent ones were skipped, an intermediate file would be corrupted while grafting OCR text, causing processing to fail. This is a regression, likely introduced in v8.2.4. - -- Previously, we resized the images produced by Ghostscript by a small number of pixels to ensure the output image size was an exactly what we wanted. Having discovered a way to get Ghostscript to produce the exact image sizes we require, we eliminated the resizing step. - -- Command line completions for ``bash`` are now available, in addition to ``fish``, both in ``misc/completion``. Package maintainers, please install these so users can take advantage. - -- Updated requirements. - -- pikepdf 1.3.0 is now required. +- Improved the strategy for updating pages when a new image of the page + was produced. We now attempt to preserve more content from the + original file, for annotations in particular. +- For PDFs with more than 100 pages and a sequence where one PDF page + was replaced and one or more subsequent ones were skipped, an + intermediate file would be corrupted while grafting OCR text, causing + processing to fail. This is a regression, likely introduced in + v8.2.4. +- Previously, we resized the images produced by Ghostscript by a small + number of pixels to ensure the output image size was an exactly what + we wanted. Having discovered a way to get Ghostscript to produce the + exact image sizes we require, we eliminated the resizing step. +- Command line completions for ``bash`` are now available, in addition + to ``fish``, both in ``misc/completion``. Package maintainers, please + install these so users can take advantage. +- Updated requirements. +- pikepdf 1.3.0 is now required. v8.2.4 ------- +====== -- Fixed a false positive while checking for a certain type of PDF that only Acrobat can read. We now more accurately detect Acrobat-only PDFs. - -- OCRmyPDF holds fewer open file handles and is more prompt about releasing those it no longer needs. - -- Minor optimization: we no longer traverse the table of contents to ensure all references in it are resolved, as changes to libqpdf have made this unnecessary. - -- pikepdf 1.2.0 is now required. +- Fixed a false positive while checking for a certain type of PDF that + only Acrobat can read. We now more accurately detect Acrobat-only + PDFs. +- OCRmyPDF holds fewer open file handles and is more prompt about + releasing those it no longer needs. +- Minor optimization: we no longer traverse the table of contents to + ensure all references in it are resolved, as changes to libqpdf have + made this unnecessary. +- pikepdf 1.2.0 is now required. v8.2.3 ------- +====== -- Fixed that ``--mask-barcodes`` would occasionally leave a unwanted temporary file named ``junkpixt`` in the current working folder. - -- Fixed (hopefully) handling of Leptonica errors in an environment where a non-standard ``sys.stderr`` is present. - -- Improved help text for ``--verbose``. +- Fixed that ``--mask-barcodes`` would occasionally leave a unwanted + temporary file named ``junkpixt`` in the current working folder. +- Fixed (hopefully) handling of Leptonica errors in an environment + where a non-standard ``sys.stderr`` is present. +- Improved help text for ``--verbose``. v8.2.2 ------- +====== -- Fixed a regression from v8.2.0, an exception that occurred while attempting to report that ``unpaper`` or another optional dependency was unavailable. - -- In some cases, ``ocrmypdf [-c|--clean]`` failed to exit with an error when ``unpaper`` is not installed. +- Fixed a regression from v8.2.0, an exception that occurred while + attempting to report that ``unpaper`` or another optional dependency + was unavailable. +- In some cases, ``ocrmypdf [-c|--clean]`` failed to exit with an error + when ``unpaper`` is not installed. v8.2.1 ------- +====== -- This release was canceled. +- This release was canceled. v8.2.0 ------- +====== -- A major improvement to our Docker image is now available thanks to hard work contributed by @mawi12345. The new Docker image, ocrmypdf-alpine, is based on Alpine Linux, and includes most of the functionality of three existed images in a smaller package. This image will replace the main Docker image eventually but for now all are being built. `See documentation for details `_. - -- Documentation reorganized especially around the use of Docker images. - -- Fixed a problem with PDF image optimization, where the optimizer would unnecessarily decompress and recompress PNG images, in some cases losing the benefits of the quantization it just had just performed. The optimizer is now capable of embedding PNG images into PDFs without transcoding them. - -- Fixed a minor regression with lossy JBIG2 image optimization. All JBIG2 candidates images were incorrectly placed into a single optimization group for the whole file, instead of grouping pages together. This usually makes a larger JBIG2Globals dictionary and results in inferior compression, so it worked less well than designed. However, quality would not be impacted. Lossless JBIG2 was entirely unaffected. - -- Updated dependencies, including pikepdf to 1.1.0. This fixes `#358 `_. - -- The install-time version checks for certain external programs have been removed from setup.py. These tests are now performed at run-time. - -- The non-standard option to override install-time checks (``setup.py install --force``) is now deprecated and prints a warning. It will be removed in a future release. +- A major improvement to our Docker image is now available thanks to + hard work contributed by @mawi12345. The new Docker image, + ocrmypdf-alpine, is based on Alpine Linux, and includes most of the + functionality of three existed images in a smaller package. This + image will replace the main Docker image eventually but for now all + are being built. `See documentation for + details `__. +- Documentation reorganized especially around the use of Docker images. +- Fixed a problem with PDF image optimization, where the optimizer + would unnecessarily decompress and recompress PNG images, in some + cases losing the benefits of the quantization it just had just + performed. The optimizer is now capable of embedding PNG images into + PDFs without transcoding them. +- Fixed a minor regression with lossy JBIG2 image optimization. All + JBIG2 candidates images were incorrectly placed into a single + optimization group for the whole file, instead of grouping pages + together. This usually makes a larger JBIG2Globals dictionary and + results in inferior compression, so it worked less well than + designed. However, quality would not be impacted. Lossless JBIG2 was + entirely unaffected. +- Updated dependencies, including pikepdf to 1.1.0. This fixes + `#358 `__. +- The install-time version checks for certain external programs have + been removed from setup.py. These tests are now performed at + run-time. +- The non-standard option to override install-time checks + (``setup.py install --force``) is now deprecated and prints a + warning. It will be removed in a future release. v8.1.0 ------- +====== -- Added a feature, ``--unpaper-args``, which allows passing arbitrary arguments to ``unpaper`` when using ``--clean`` or ``--clean-final``. The default, very conservative unpaper settings are suppressed. - -- The argument ``--clean-final`` now implies ``--clean``. It was possible to issue ``--clean-final`` on its before this, but it would have no useful effect. - -- Fixed an exception on traversing corrupt table of contents entries (specifically, those with invalid destination objects) - -- Fixed an issue when using ``--tesseract-timeout`` and image processing features on a file with more than 100 pages. `#347 `_ - -- OCRmyPDF now always calls ``os.nice(5)`` to signal to operating systems that it is a background process. +- Added a feature, ``--unpaper-args``, which allows passing arbitrary + arguments to ``unpaper`` when using ``--clean`` or ``--clean-final``. + The default, very conservative unpaper settings are suppressed. +- The argument ``--clean-final`` now implies ``--clean``. It was + possible to issue ``--clean-final`` on its before this, but it would + have no useful effect. +- Fixed an exception on traversing corrupt table of contents entries + (specifically, those with invalid destination objects) +- Fixed an issue when using ``--tesseract-timeout`` and image + processing features on a file with more than 100 pages. + `#347 `__ +- OCRmyPDF now always calls ``os.nice(5)`` to signal to operating + systems that it is a background process. v8.0.1 ------- +====== -- Fixed an exception when parsing PDFs that are missing a required field. `#325 `_ - -- pikepdf 1.0.5 is now required, to address some other PDF parsing issues. +- Fixed an exception when parsing PDFs that are missing a required + field. `#325 `__ +- pikepdf 1.0.5 is now required, to address some other PDF parsing + issues. v8.0.0 ------- +====== -No major features. The intent of this release is to sever support for older versions of certain dependencies. +No major features. The intent of this release is to sever support for +older versions of certain dependencies. **Breaking changes** -- Dropped support for Tesseract 3.x. Tesseract 4.0 or newer is now required. - -- Dropped support for Python 3.5. - -- Some ``ocrmypdf.pdfa`` APIs that were deprecated in v7.x were removed. This functionality has been moved to pikepdf. +- Dropped support for Tesseract 3.x. Tesseract 4.0 or newer is now + required. +- Dropped support for Python 3.5. +- Some ``ocrmypdf.pdfa`` APIs that were deprecated in v7.x were + removed. This functionality has been moved to pikepdf. **Other changes** -- Fixed an unhandled exception when attempting to mask barcodes. `#322 `_ - -- It is now possible to use ocrmypdf without pdfminer.six, to support distributions that do not have it or cannot currently use it (e.g. Homebrew). Downstream maintainers should include pdfminer.six if possible. - -- A warning is now issue when PDF/A conversion removes some XMP metadata from the input PDF. (Only a "whitelist" of certain XMP metadata types are allowed in PDF/A.) - -- Fixed several issues that caused PDF/As to be produced with nonconforming XMP metadata (would fail validation with veraPDF). - -- Fixed some instances where invalid DocumentInfo from a PDF cause XMP metadata creation to fail. - -- Fixed a few documentation problems. - -- pikepdf 1.0.2 is now required. +- Fixed an unhandled exception when attempting to mask barcodes. + `#322 `__ +- It is now possible to use ocrmypdf without pdfminer.six, to support + distributions that do not have it or cannot currently use it (e.g. + Homebrew). Downstream maintainers should include pdfminer.six if + possible. +- A warning is now issue when PDF/A conversion removes some XMP + metadata from the input PDF. (Only a "whitelist" of certain XMP + metadata types are allowed in PDF/A.) +- Fixed several issues that caused PDF/As to be produced with + nonconforming XMP metadata (would fail validation with veraPDF). +- Fixed some instances where invalid DocumentInfo from a PDF cause XMP + metadata creation to fail. +- Fixed a few documentation problems. +- pikepdf 1.0.2 is now required. v7.4.0 ------- +====== -- ``--force-ocr`` may now be used with the new ``--threshold`` and ``--mask-barcodes`` features - -- pikepdf >= 0.9.1 is now required. - -- Changed metadata handling to pikepdf 0.9.1. As a result, metadata handling of non-ASCII characters in Ghostscript 9.25 or later is fixed. - -- chardet >= 3.0.4 is temporarily listed as required. pdfminer.six depends on it, but the most recent release does not specify this requirement. (`#326 `_) - -- python-xmp-toolkit and libexempi are no longer required. - -- A new Docker image is now being provided for users who wish to access OCRmyPDF over a simple HTTP interface, instead of the command line. - -- Increase tolerance of PDFs that overflow or underflow the PDF graphics stack. (`#325 `_) +- ``--force-ocr`` may now be used with the new ``--threshold`` and + ``--mask-barcodes`` features +- pikepdf >= 0.9.1 is now required. +- Changed metadata handling to pikepdf 0.9.1. As a result, metadata + handling of non-ASCII characters in Ghostscript 9.25 or later is + fixed. +- chardet >= 3.0.4 is temporarily listed as required. pdfminer.six + depends on it, but the most recent release does not specify this + requirement. + (`#326 `__) +- python-xmp-toolkit and libexempi are no longer required. +- A new Docker image is now being provided for users who wish to access + OCRmyPDF over a simple HTTP interface, instead of the command line. +- Increase tolerance of PDFs that overflow or underflow the PDF + graphics stack. + (`#325 `__) v7.3.1 ------- - -- Fixed performance regression from v7.3.0; fast page analysis was not selected when it should be. - -- Fixed a few exceptions related to the new ``--mask-barcodes`` feature and improved argument checking - -- Added missing detection of TrueType fonts that lack a Unicode mapping +====== +- Fixed performance regression from v7.3.0; fast page analysis was not + selected when it should be. +- Fixed a few exceptions related to the new ``--mask-barcodes`` feature + and improved argument checking +- Added missing detection of TrueType fonts that lack a Unicode mapping v7.3.0 ------- +====== -- Added a new feature ``--redo-ocr`` to detect existing OCR in a file, remove it, and redo the OCR. This may be particularly helpful for anyone who wants to take advantage of OCR quality improvements in Tesseract 4.0. Note that OCR added by OCRmyPDF before version 3.0 cannot be detected since it was not properly marked as invisible text in the earliest versions. OCR that constructs a font from visible text, such as Adobe Acrobat's ClearScan. +- Added a new feature ``--redo-ocr`` to detect existing OCR in a file, + remove it, and redo the OCR. This may be particularly helpful for + anyone who wants to take advantage of OCR quality improvements in + Tesseract 4.0. Note that OCR added by OCRmyPDF before version 3.0 + cannot be detected since it was not properly marked as invisible text + in the earliest versions. OCR that constructs a font from visible + text, such as Adobe Acrobat's ClearScan. +- OCRmyPDF's content detection is generally more sophisticated. It + learns more about the contents of each PDF and makes better + recommendations: -- OCRmyPDF's content detection is generally more sophisticated. It learns more about the contents of each PDF and makes better recommendations: + - OCRmyPDF can now detect when a PDF contains text that cannot be + mapped to Unicode (meaning it is readable to human eyes but + copy-pastes as gibberish). In these cases it recommends + ``--force-ocr`` to make the text searchable. + - PDFs containing vector objects are now rendered at more + appropriate resolution for OCR. + - We now exit with an error for PDFs that contain Adobe LiveCycle + Designer's dynamic XFA forms. Currently the open source community + does not have tools to work with these files. + - OCRmyPDF now warns when a PDF that contains Adobe AcroForms, since + such files probably do not need OCR. It can work with these files. - - OCRmyPDF can now detect when a PDF contains text that cannot be mapped to Unicode (meaning it is readable to human eyes but copy-pastes as gibberish). In these cases it recommends ``--force-ocr`` to make the text searchable. +- Added three new **experimental** features to improve OCR quality in + certain conditions. The name, syntax and behavior of these arguments + is subject to change. They may also be incompatible with some other + features. - - PDFs containing vector objects are now rendered at more appropriate resolution for OCR. + - ``--remove-vectors`` which strips out vector graphics. This can + improve OCR quality since OCR will not search artwork for readable + text; however, it currently removes "text as curves" as well. + - ``--mask-barcodes`` to detect and suppress barcodes in files. We + have observed that barcodes can interfere with OCR because they + are "text-like" but not actually textual. + - ``--threshold`` which uses a more sophisticated thresholding + algorithm than is currently in use in Tesseract OCR. This works + around a `known issue in Tesseract + 4.0 `__ + with dark text on bright backgrounds. - - We now exit with an error for PDFs that contain Adobe LiveCycle Designer's dynamic XFA forms. Currently the open source community does not have tools to work with these files. - - - OCRmyPDF now warns when a PDF that contains Adobe AcroForms, since such files probably do not need OCR. It can work with these files. - -- Added three new **experimental** features to improve OCR quality in certain conditions. The name, syntax and behavior of these arguments is subject to change. They may also be incompatible with some other features. - - - ``--remove-vectors`` which strips out vector graphics. This can improve OCR quality since OCR will not search artwork for readable text; however, it currently removes "text as curves" as well. - - - ``--mask-barcodes`` to detect and suppress barcodes in files. We have observed that barcodes can interfere with OCR because they are "text-like" but not actually textual. - - - ``--threshold`` which uses a more sophisticated thresholding algorithm than is currently in use in Tesseract OCR. This works around a `known issue in Tesseract 4.0 `_ with dark text on bright backgrounds. - -- Fixed an issue where an error message was not reported when the installed Ghostscript was very old. - -- The PDF optimizer now saves files with object streams enabled when the optimization level is ``--optimize 1`` or higher (the default). This makes files a little bit smaller, but requires PDF 1.5. PDF 1.5 was first released in 2003 and is broadly supported by PDF viewers, but some rudimentary PDF parsers such as PyPDF2 do not understand object streams. You can use the command line tool ``qpdf --object-streams=disable`` or `pikepdf `_ library to remove them. - -- New dependency: pdfminer.six 20181108. Note this is a fork of the Python 2-only pdfminer. - -- Deprecation notice: At the end of 2018, we will be ending support for Python 3.5 and Tesseract 3.x. OCRmyPDF v7 will continue to work with older versions. +- Fixed an issue where an error message was not reported when the + installed Ghostscript was very old. +- The PDF optimizer now saves files with object streams enabled when + the optimization level is ``--optimize 1`` or higher (the default). + This makes files a little bit smaller, but requires PDF 1.5. PDF 1.5 + was first released in 2003 and is broadly supported by PDF viewers, + but some rudimentary PDF parsers such as PyPDF2 do not understand + object streams. You can use the command line tool + ``qpdf --object-streams=disable`` or + `pikepdf `__ library to remove + them. +- New dependency: pdfminer.six 20181108. Note this is a fork of the + Python 2-only pdfminer. +- Deprecation notice: At the end of 2018, we will be ending support for + Python 3.5 and Tesseract 3.x. OCRmyPDF v7 will continue to work with + older versions. v7.2.1 ------- - -- Fix compatibility with an API change in pikepdf 0.3.5. - -- A kludge to support Leptonica versions older than 1.72 in the test suite was dropped. Older versions of Leptonica are likely still compatible. The only impact is that a portion of the test suite will be skipped. +====== +- Fix compatibility with an API change in pikepdf 0.3.5. +- A kludge to support Leptonica versions older than 1.72 in the test + suite was dropped. Older versions of Leptonica are likely still + compatible. The only impact is that a portion of the test suite will + be skipped. v7.2.0 ------- +====== **Lossy JBIG2 behavior change** -A user reported that ocrmypdf was in fact using JBIG2 in **lossy** compression mode. This was not the intended behavior. Users should `review the technical concerns with JBIG2 in lossy mode `_ and decide if this is a concern for their use case. +A user reported that ocrmypdf was in fact using JBIG2 in **lossy** +compression mode. This was not the intended behavior. Users should +`review the technical concerns with JBIG2 in lossy +mode `__ +and decide if this is a concern for their use case. -JBIG2 lossy mode does achieve higher compression ratios than any other monochrome compression technology; for large text documents the savings are considerable. JBIG2 lossless still gives great compression ratios and is a major improvement over the older CCITT G4 standard. +JBIG2 lossy mode does achieve higher compression ratios than any other +monochrome compression technology; for large text documents the savings +are considerable. JBIG2 lossless still gives great compression ratios +and is a major improvement over the older CCITT G4 standard. -Only users who have reviewed the concerns with JBIG2 in lossy mode should opt-in. As such, lossy mode JBIG2 is only turned on when the new argument ``--jbig2-lossy`` is issued. This is independent of the setting for ``--optimize``. +Only users who have reviewed the concerns with JBIG2 in lossy mode +should opt-in. As such, lossy mode JBIG2 is only turned on when the new +argument ``--jbig2-lossy`` is issued. This is independent of the setting +for ``--optimize``. Users who did not install an optional JBIG2 encoder are unaffected. @@ -207,930 +284,1238 @@ Users who did not install an optional JBIG2 encoder are unaffected. **Other issues** -- When the image optimizer quantizes an image to 1 bit per pixel, it will now attempt to further optimize that image as CCITT or JBIG2, instead of keeping it in the "flate" encoding which is not efficient for 1 bpp images. (`#297 `_) - -- Images in PDFs that are used as soft masks (i.e. transparency masks or alpha channels) are now excluded from optimization. - -- Fixed handling of Tesseract 4.0-rc1 which now accepts invalid Tesseract configuration files, which broke the test suite. +- When the image optimizer quantizes an image to 1 bit per pixel, it + will now attempt to further optimize that image as CCITT or JBIG2, + instead of keeping it in the "flate" encoding which is not efficient + for 1 bpp images. + (`#297 `__) +- Images in PDFs that are used as soft masks (i.e. transparency masks + or alpha channels) are now excluded from optimization. +- Fixed handling of Tesseract 4.0-rc1 which now accepts invalid + Tesseract configuration files, which broke the test suite. v7.1.0 ------- +====== -- Improve the performance of initial text extraction, which is done to determine if a file contains existing text of some kind or not. On large files, this initial processing is now about 20x times faster. (`#299 `_) - -- pikepdf 0.3.3 is now required. - -- Fixed issue `#231 `_, a problem with JPEG2000 images where image metadata was only available inside the JPEG2000 file. - -- Fixed some additional Ghostscript 9.25 compatibility issues. - -- Improved handling of KeyboardInterrupt error messages. (`#301 `_) - -- README.md is now served in GitHub markdown instead of reStructuredText. +- Improve the performance of initial text extraction, which is done to + determine if a file contains existing text of some kind or not. On + large files, this initial processing is now about 20x times faster. + (`#299 `__) +- pikepdf 0.3.3 is now required. +- Fixed issue + `#231 `__, a + problem with JPEG2000 images where image metadata was only available + inside the JPEG2000 file. +- Fixed some additional Ghostscript 9.25 compatibility issues. +- Improved handling of KeyboardInterrupt error messages. + (`#301 `__) +- README.md is now served in GitHub markdown instead of + reStructuredText. v7.0.6 ------- - -- Blacklist Ghostscript 9.24, now that 9.25 is available and fixes many regressions in 9.24. +====== +- Blacklist Ghostscript 9.24, now that 9.25 is available and fixes many + regressions in 9.24. v7.0.5 ------- +====== -- Improve capability with Ghostscript 9.24, and enable the JPEG passthrough feature when this version in installed. - -- Ghostscript 9.24 lost the ability to set PDF title, author, subject and keyword metadata to Unicode strings. OCRmyPDF will set ASCII strings and warn when Unicode is suppressed. Other software may be used to update metadata. This is a short term work around. - -- PDFs generated by Kodak Capture Desktop, or generally PDFs that contain indirect references to null objects in their table of contents, would have an invalid table of contents after processing by OCRmyPDF that might interfere with other viewers. This has been fixed. - -- Detect PDFs generated by Adobe LiveCycle, which can only be displayed in Adobe Acrobat and Reader currently. When these are encountered, exit with an error instead of performing OCR on the "Please wait" error message page. +- Improve capability with Ghostscript 9.24, and enable the JPEG + passthrough feature when this version in installed. +- Ghostscript 9.24 lost the ability to set PDF title, author, subject + and keyword metadata to Unicode strings. OCRmyPDF will set ASCII + strings and warn when Unicode is suppressed. Other software may be + used to update metadata. This is a short term work around. +- PDFs generated by Kodak Capture Desktop, or generally PDFs that + contain indirect references to null objects in their table of + contents, would have an invalid table of contents after processing by + OCRmyPDF that might interfere with other viewers. This has been + fixed. +- Detect PDFs generated by Adobe LiveCycle, which can only be displayed + in Adobe Acrobat and Reader currently. When these are encountered, + exit with an error instead of performing OCR on the "Please wait" + error message page. v7.0.4 ------- +====== -- Fix exception thrown when trying to optimize a certain type of PNG embedded in a PDF with the ``-O2`` - -- Update to pikepdf 0.3.2, to gain support for optimizing some additional image types that were previously excluded from optimization (CMYK and grayscale). Fixes `#285 `_. +- Fix exception thrown when trying to optimize a certain type of PNG + embedded in a PDF with the ``-O2`` +- Update to pikepdf 0.3.2, to gain support for optimizing some + additional image types that were previously excluded from + optimization (CMYK and grayscale). Fixes + `#285 `__. v7.0.3 ------- +====== -- Fix issue `#284 `_, an error when parsing inline images that have are also image masks, by upgrading pikepdf to 0.3.1 +- Fix issue + `#284 `__, an error + when parsing inline images that have are also image masks, by + upgrading pikepdf to 0.3.1 v7.0.2 ------- +====== -- Fix a regression with ``--rotate-pages`` on pages that already had rotations applied. (`#279 `_) - -- Improve quality of page rotation in some cases by rasterizing a higher quality preview image. (`#281 `_) +- Fix a regression with ``--rotate-pages`` on pages that already had + rotations applied. + (`#279 `__) +- Improve quality of page rotation in some cases by rasterizing a + higher quality preview image. + (`#281 `__) v7.0.1 ------- +====== -- Fix compatibility with img2pdf >= 0.3.0 by rejecting input images that have an alpha channel - -- Add forward compatibility for pikepdf 0.3.0 (unrelated to img2pdf) - -- Various documentation updates for v7.0.0 changes +- Fix compatibility with img2pdf >= 0.3.0 by rejecting input images + that have an alpha channel +- Add forward compatibility for pikepdf 0.3.0 (unrelated to img2pdf) +- Various documentation updates for v7.0.0 changes v7.0.0 ------- +====== -- The core algorithm for combining OCR layers with existing PDF pages has been rewritten and improved considerably. PDFs are no longer split into single page PDFs for processing; instead, images are rendered and the OCR results are grafted onto the input PDF. The new algorithm uses less temporary disk space and is much more performant especially for large files. +- The core algorithm for combining OCR layers with existing PDF pages + has been rewritten and improved considerably. PDFs are no longer + split into single page PDFs for processing; instead, images are + rendered and the OCR results are grafted onto the input PDF. The new + algorithm uses less temporary disk space and is much more performant + especially for large files. +- New dependency: `pikepdf `__. + pikepdf is a powerful new Python PDF library driving the latest + OCRmyPDF features, built on the QPDF C++ library (libqpdf). +- New feature: PDF optimization with ``-O`` or ``--optimize``. After + OCR, OCRmyPDF will perform image optimizations relevant to OCR PDFs. -- New dependency: `pikepdf `_. pikepdf is a powerful new Python PDF library driving the latest OCRmyPDF features, built on the QPDF C++ library (libqpdf). + - If a JBIG2 encoder is available, then monochrome images will be + converted, with the potential for huge savings on large black and + white images, since JBIG2 is far more efficient than any other + monochrome (bi-level) compression. (All known US patents related + to JBIG2 have probably expired, but it remains the responsibility + of the user to supply a JBIG2 encoder such as + `jbig2enc `__. OCRmyPDF does not + implement JBIG2 encoding.) + - If ``pngquant`` is installed, OCRmyPDF will optionally use it to + perform lossy quantization and compression of PNG images. + - The quality of JPEGs can also be lowered, on the assumption that a + lower quality image may be suitable for storage after OCR. + - This image optimization component will eventually be offered as an + independent command line utility. + - Optimization ranges from ``-O0`` through ``-O3``, where ``0`` + disables optimization and ``3`` implements all options. ``1``, the + default, performs only safe and lossless optimizations. (This is + similar to GCC's optimization parameter.) The exact type of + optimizations performed will vary over time. -- New feature: PDF optimization with ``-O`` or ``--optimize``. After OCR, OCRmyPDF will perform image optimizations relevant to OCR PDFs. +- Small amounts of text in the margins of a page, such as watermarks, + page numbers, or digital stamps, will no longer prevent the rest of a + page from being OCRed when ``--skip-text`` is issued. This behavior + is based on a heuristic. +- Removed features - + If a JBIG2 encoder is available, then monochrome images will be converted, with the potential for huge savings on large black and white images, since JBIG2 is far more efficient than any other monochrome (bi-level) compression. (All known US patents related to JBIG2 have probably expired, but it remains the responsibility of the user to supply a JBIG2 encoder such as `jbig2enc `_. OCRmyPDF does not implement JBIG2 encoding.) + - The deprecated ``--pdf-renderer tesseract`` PDF renderer was + removed. + - ``-g``, the option to generate debug text pages, was removed + because it was a maintenance burden and only worked in isolated + cases. HOCR pages can still be previewed by running the + hocrtransform.py with appropriate settings. - + If ``pngquant`` is installed, OCRmyPDF will optionally use it to perform lossy quantization and compression of PNG images. +- Removed dependencies - + The quality of JPEGs can also be lowered, on the assumption that a lower quality image may be suitable for storage after OCR. + - ``PyPDF2`` + - ``defusedxml`` + - ``PyMuPDF`` - + This image optimization component will eventually be offered as an independent command line utility. +- The ``sandwich`` PDF renderer can be used with all supported versions + of Tesseract, including that those prior to v3.05 which don't support + ``-c textonly``. (Tesseract v4.0.0 is recommended and more + efficient.) +- ``--pdf-renderer auto`` option and the diagnostics used to select a + PDF renderer now work better with old versions, but may make + different decisions than past versions. +- If everything succeeds but PDF/A conversion fails, a distinct return + code is now returned (``ExitCode.pdfa_conversion_failed (10)``) where + this situation previously returned + ``ExitCode.invalid_output_pdf (4)``. The latter is now returned only + if there is some indication that the output file is invalid. +- Notes for downstream packagers - + Optimization ranges from ``-O0`` through ``-O3``, where ``0`` disables optimization and ``3`` implements all options. ``1``, the default, performs only safe and lossless optimizations. (This is similar to GCC's optimization parameter.) The exact type of optimizations performed will vary over time. - -- Small amounts of text in the margins of a page, such as watermarks, page numbers, or digital stamps, will no longer prevent the rest of a page from being OCRed when ``--skip-text`` is issued. This behavior is based on a heuristic. - -- Removed features - - + The deprecated ``--pdf-renderer tesseract`` PDF renderer was removed. - - + ``-g``, the option to generate debug text pages, was removed because it was a maintenance burden and only worked in isolated cases. HOCR pages can still be previewed by running the hocrtransform.py with appropriate settings. - -- Removed dependencies - - + ``PyPDF2`` - - + ``defusedxml`` - - + ``PyMuPDF`` - -- The ``sandwich`` PDF renderer can be used with all supported versions of Tesseract, including that those prior to v3.05 which don't support ``-c textonly``. (Tesseract v4.0.0 is recommended and more efficient.) - -- ``--pdf-renderer auto`` option and the diagnostics used to select a PDF renderer now work better with old versions, but may make different decisions than past versions. - -- If everything succeeds but PDF/A conversion fails, a distinct return code is now returned (``ExitCode.pdfa_conversion_failed (10)``) where this situation previously returned ``ExitCode.invalid_output_pdf (4)``. The latter is now returned only if there is some indication that the output file is invalid. - -- Notes for downstream packagers - - + There is also a new dependency on ``python-xmp-toolkit`` which in turn depends on ``libexempi3``. - - + It may be necessary to separately ``pip install pycparser`` to avoid `another Python 3.7 issue `_. + - There is also a new dependency on ``python-xmp-toolkit`` which in + turn depends on ``libexempi3``. + - It may be necessary to separately ``pip install pycparser`` to + avoid `another Python 3.7 + issue `__. v6.2.5 ------- +====== -- Disable a failing test due to Tesseract 4.0rc1 behavior change. Previously, Tesseract would exit with an error message if its configuration was invalid, and OCRmyPDF would intercept this message. Now Tesseract issues a warning, which OCRmyPDF v6.2.5 may relay or ignore. (In v7.x, OCRmyPDF will respond to the warning.) - -- This release branch no longer supports using the optional PyMuPDF installation, since it was removed in v7.x. - -- This release branch no longer supports macOS. macOS users should upgrade to v7.x. +- Disable a failing test due to Tesseract 4.0rc1 behavior change. + Previously, Tesseract would exit with an error message if its + configuration was invalid, and OCRmyPDF would intercept this message. + Now Tesseract issues a warning, which OCRmyPDF v6.2.5 may relay or + ignore. (In v7.x, OCRmyPDF will respond to the warning.) +- This release branch no longer supports using the optional PyMuPDF + installation, since it was removed in v7.x. +- This release branch no longer supports macOS. macOS users should + upgrade to v7.x. v6.2.4 ------- +====== -- Backport Ghostscript 9.25 compatibility fixes, which removes support for setting Unicode metadata -- Backport blacklisting Ghostscript 9.24 -- Older versions of Ghostscript are still supported +- Backport Ghostscript 9.25 compatibility fixes, which removes support + for setting Unicode metadata +- Backport blacklisting Ghostscript 9.24 +- Older versions of Ghostscript are still supported v6.2.3 ------- +====== -- Fix compatibility with img2pdf >= 0.3.0 by rejecting input images that have an alpha channel -- This version will be included in Ubuntu 18.10 +- Fix compatibility with img2pdf >= 0.3.0 by rejecting input images + that have an alpha channel +- This version will be included in Ubuntu 18.10 v6.2.2 ------- +====== -- Backport compatibility fixes for Python 3.7 and ruffus 2.7.0 from v7.0.0 -- Backport fix to ignore masks when deciding what colors are on a page -- Backport some minor improvements from v7.0.0: better argument validation and warnings about the Tesseract 4.0.0 ``--user-words`` regression +- Backport compatibility fixes for Python 3.7 and ruffus 2.7.0 from + v7.0.0 +- Backport fix to ignore masks when deciding what colors are on a page +- Backport some minor improvements from v7.0.0: better argument + validation and warnings about the Tesseract 4.0.0 ``--user-words`` + regression v6.2.1 ------- +====== -- Fix recent versions of Tesseract (after 4.0.0-beta1) not being detected as supporting the ``sandwich`` renderer (`#271 `_). +- Fix recent versions of Tesseract (after 4.0.0-beta1) not being + detected as supporting the ``sandwich`` renderer + (`#271 `__). v6.2.0 ------- - -- **Docker**: The Docker image ``ocrmypdf-tess4`` has been removed. The main Docker images, ``ocrmypdf`` and ``ocrmypdf-polyglot`` now use Ubuntu 18.04 as a base image, and as such Tesseract 4.0.0-beta1 is now the Tesseract version they use. There is no Docker image based on Tesseract 3.05 anymore. - -- Creation of PDF/A-3 is now supported. However, there is no ability to attach files to PDF/A-3. - -- Lists more reasons why the file size might grow. - -- Fix issue `#262 `_, ``--remove-background`` error on PDFs contained colormapped (paletted) images. - -- Fix another XMP metadata validation issue, in cases where the input file's creation date has no timezone and the creation date is not overridden. +====== +- **Docker**: The Docker image ``ocrmypdf-tess4`` has been removed. The + main Docker images, ``ocrmypdf`` and ``ocrmypdf-polyglot`` now use + Ubuntu 18.04 as a base image, and as such Tesseract 4.0.0-beta1 is + now the Tesseract version they use. There is no Docker image based on + Tesseract 3.05 anymore. +- Creation of PDF/A-3 is now supported. However, there is no ability to + attach files to PDF/A-3. +- Lists more reasons why the file size might grow. +- Fix issue + `#262 `__, + ``--remove-background`` error on PDFs contained colormapped + (paletted) images. +- Fix another XMP metadata validation issue, in cases where the input + file's creation date has no timezone and the creation date is not + overridden. v6.1.5 ------- - -- Fix issue `#253 `_, a possible division by zero when using the ``hocr`` renderer. - -- Fix incorrectly formatted ```` field inside XMP metadata for PDF/As. veraPDF flags this as a PDF/A validation failure. The error is caused the timezone and final digit of the seconds of modified time to be omitted, so at worst the modification time stamp is rounded to the nearest 10 seconds. +====== +- Fix issue + `#253 `__, a + possible division by zero when using the ``hocr`` renderer. +- Fix incorrectly formatted ```` field inside XMP + metadata for PDF/As. veraPDF flags this as a PDF/A validation + failure. The error is caused the timezone and final digit of the + seconds of modified time to be omitted, so at worst the modification + time stamp is rounded to the nearest 10 seconds. v6.1.4 ------- +====== -- Fix issue `#248 `_ ``--clean`` argument may remove OCR from left column of text on certain documents. We now set ``--layout none`` to suppress this. - -- The test cache was updated to reflect the change above. - -- Change test suite to accommodate Ghostscript 9.23's new ability to insert JPEGs into PDFs without transcoding. - -- XMP metadata in PDFs is now examined using ``defusedxml`` for safety. - -- If an external process exits with a signal when asked to report its version, we now print the system error message instead of suppressing it. This occurred when the required executable was found but was missing a shared library. - -- qpdf 7.0.0 or newer is now required as the test suite can no longer pass without it. +- Fix issue `#248 `__ + ``--clean`` argument may remove OCR from left column of text on + certain documents. We now set ``--layout none`` to suppress this. +- The test cache was updated to reflect the change above. +- Change test suite to accommodate Ghostscript 9.23's new ability to + insert JPEGs into PDFs without transcoding. +- XMP metadata in PDFs is now examined using ``defusedxml`` for safety. +- If an external process exits with a signal when asked to report its + version, we now print the system error message instead of suppressing + it. This occurred when the required executable was found but was + missing a shared library. +- qpdf 7.0.0 or newer is now required as the test suite can no longer + pass without it. Notes -~~~~~ - -- An apparent `regression in Ghostscript 9.23 `_ will cause some ocrmypdf output files to become invalid in rare cases; the workaround for the moment is to set ``--force-ocr``. +----- +- An apparent `regression in Ghostscript + 9.23 `__ will + cause some ocrmypdf output files to become invalid in rare cases; the + workaround for the moment is to set ``--force-ocr``. v6.1.3 ------- - -- Fix issue `#247 `_, ``/CreationDate`` metadata not copied from input to output. - -- A warning is now issued when Python 3.5 is used on files with a large page count, as this case is known to regress to single core performance. The cause of this problem is unknown. +====== +- Fix issue + `#247 `__, + ``/CreationDate`` metadata not copied from input to output. +- A warning is now issued when Python 3.5 is used on files with a large + page count, as this case is known to regress to single core + performance. The cause of this problem is unknown. v6.1.2 ------- - -- Upgrade to PyMuPDF v1.12.5 which includes a more complete fix to `#239 `_. - -- Add ``defusedxml`` dependency. +====== +- Upgrade to PyMuPDF v1.12.5 which includes a more complete fix to + `#239 `__. +- Add ``defusedxml`` dependency. v6.1.1 ------- - -- Fix text being reported as found on all pages if PyMuPDF is not installed. +====== +- Fix text being reported as found on all pages if PyMuPDF is not + installed. v6.1.0 ------- - -- PyMuPDF is now an optional but recommended dependency, to alleviate installation difficulties on platforms that have less access to PyMuPDF than the author anticipated. (For version 6.x only) install OCRmyPDF with ``pip install ocrmypdf[fitz]`` to use it to its full potential. - -- Fix ``FileExistsError`` that could occur if OCR timed out while it was generating the output file. (`#218 `_) - -- Fix table of contents/bookmarks all being redirected to page 1 when generating a PDF/A (with PyMuPDF). (Without PyMuPDF the table of contents is removed in PDF/A mode.) - -- Fix "RuntimeError: invalid key in dict" when table of contents/bookmarks titles contained the character ``)``. (`#239 `_) - -- Added a new argument ``--skip-repair`` to skip the initial PDF repair step if the PDF is already well-formed (because another program repaired it). +====== +- PyMuPDF is now an optional but recommended dependency, to alleviate + installation difficulties on platforms that have less access to + PyMuPDF than the author anticipated. (For version 6.x only) install + OCRmyPDF with ``pip install ocrmypdf[fitz]`` to use it to its full + potential. +- Fix ``FileExistsError`` that could occur if OCR timed out while it + was generating the output file. + (`#218 `__) +- Fix table of contents/bookmarks all being redirected to page 1 when + generating a PDF/A (with PyMuPDF). (Without PyMuPDF the table of + contents is removed in PDF/A mode.) +- Fix "RuntimeError: invalid key in dict" when table of + contents/bookmarks titles contained the character ``)``. + (`#239 `__) +- Added a new argument ``--skip-repair`` to skip the initial PDF repair + step if the PDF is already well-formed (because another program + repaired it). v6.0.0 ------- +====== -- The software license has been changed to GPLv3. Test resource files and some individual sources may have other licenses. +- The software license has been changed to GPLv3. Test resource files + and some individual sources may have other licenses. +- OCRmyPDF now depends on + `PyMuPDF `__. + Including PyMuPDF is the primary reason for the change to GPLv3. +- Other backward incompatible changes -- OCRmyPDF now depends on `PyMuPDF `_. Including PyMuPDF is the primary reason for the change to GPLv3. - -- Other backward incompatible changes - - + The ``OCRMYPDF_TESSERACT``, ``OCRMYPDF_QPDF``, ``OCRMYPDF_GS`` and ``OCRMYPDF_UNPAPER`` environment variables are no longer used. Change ``PATH`` if you need to override the external programs OCRmyPDF uses. - - + The ``ocrmypdf`` package has been moved to ``src/ocrmypdf`` to avoid issues with accidental import. - - + The function ``ocrmypdf.exec.get_program`` was removed. - - + The deprecated module ``ocrmypdf.pageinfo`` was removed. - - + The ``--pdf-renderer tess4`` alias for ``sandwich`` was removed. - -- Fixed an issue where OCRmyPDF failed to detect existing text on pages, depending on how the text and fonts were encoded within the PDF. (`#233 `_, `#232 `_) - -- Fixed an issue that caused dramatic inflation of file sizes when ``--skip-text --output-type pdf`` was used. OCRmyPDF now removes duplicate resources such as fonts, images and other objects that it generates. (`#237 `_) - -- Improved performance of the initial page splitting step. Originally this step was not believed to be expensive and ran in a process. Large file testing revealed it to be a bottleneck, so it is now parallelized. On a 700 page file with quad core machine, this change saves about 2 minutes. (`#234 `_) - -- The test suite now includes a cache that can be used to speed up test runs across platforms. This also does not require computing checksums, so it's faster. (`#217 `_) + - The ``OCRMYPDF_TESSERACT``, ``OCRMYPDF_QPDF``, ``OCRMYPDF_GS`` and + ``OCRMYPDF_UNPAPER`` environment variables are no longer used. + Change ``PATH`` if you need to override the external programs + OCRmyPDF uses. + - The ``ocrmypdf`` package has been moved to ``src/ocrmypdf`` to + avoid issues with accidental import. + - The function ``ocrmypdf.exec.get_program`` was removed. + - The deprecated module ``ocrmypdf.pageinfo`` was removed. + - The ``--pdf-renderer tess4`` alias for ``sandwich`` was removed. +- Fixed an issue where OCRmyPDF failed to detect existing text on + pages, depending on how the text and fonts were encoded within the + PDF. (`#233 `__, + `#232 `__) +- Fixed an issue that caused dramatic inflation of file sizes when + ``--skip-text --output-type pdf`` was used. OCRmyPDF now removes + duplicate resources such as fonts, images and other objects that it + generates. + (`#237 `__) +- Improved performance of the initial page splitting step. Originally + this step was not believed to be expensive and ran in a process. + Large file testing revealed it to be a bottleneck, so it is now + parallelized. On a 700 page file with quad core machine, this change + saves about 2 minutes. + (`#234 `__) +- The test suite now includes a cache that can be used to speed up test + runs across platforms. This also does not require computing + checksums, so it's faster. + (`#217 `__) v5.7.0 ------- +====== -- Fixed an issue that caused poor CPU utilization on machines with more than 4 cores when running Tesseract 4. (Related to issue `#217 `_.) +- Fixed an issue that caused poor CPU utilization on machines with more + than 4 cores when running Tesseract 4. (Related to issue + `#217 `__.) +- The 'hocr' renderer has been improved. The 'sandwich' and 'tesseract' + renderers are still better for most use cases, but 'hocr' may be + useful for people who work with the PDF.js renderer in English/ASCII + languages. + (`#225 `__) -- The 'hocr' renderer has been improved. The 'sandwich' and 'tesseract' renderers are still better for most use cases, but 'hocr' may be useful for people who work with the PDF.js renderer in English/ASCII languages. (`#225 `_) - - + It now formats text in a matter that is easier for certain PDF viewers to select and extract copy and paste text. This should help macOS Preview and PDF.js in particular. - + The appearance of selected text and behavior of selecting text is improved. - + The PDF content stream now uses relative moves, making it more compact and easier for viewers to determine when two words on the same line. - + It can now deal with text on a skewed baseline. - + Thanks to @cforcey for the pull request, @jbreiden for many helpful suggestions, @ctbarbour for another round of improvements, and @acaloiaro for an independent review. + - It now formats text in a matter that is easier for certain PDF + viewers to select and extract copy and paste text. This should + help macOS Preview and PDF.js in particular. + - The appearance of selected text and behavior of selecting text is + improved. + - The PDF content stream now uses relative moves, making it more + compact and easier for viewers to determine when two words on the + same line. + - It can now deal with text on a skewed baseline. + - Thanks to @cforcey for the pull request, @jbreiden for many + helpful suggestions, @ctbarbour for another round of improvements, + and @acaloiaro for an independent review. v5.6.3 ------- - -- Suppress two debug messages that were too verbose +====== +- Suppress two debug messages that were too verbose v5.6.2 ------- - -- Development branch accidentally tagged as release. Do not use. +====== +- Development branch accidentally tagged as release. Do not use. v5.6.1 ------- - -- Fix issue `#219 `_: change how the final output file is created to avoid triggering permission errors when the output is a special file such as ``/dev/null`` -- Fix test suite failures due to a qpdf 8.0.0 regression and Python 3.5's handling of symlink -- The "encrypted PDF" error message was different depending on the type of PDF encryption. Now a single clear message appears for all types of PDF encryption. -- ocrmypdf is now in Homebrew. Homebrew users are advised to the version of ocrmypdf in the official homebrew-core formulas rather than the private tap. -- Some linting +====== +- Fix issue + `#219 `__: change + how the final output file is created to avoid triggering permission + errors when the output is a special file such as ``/dev/null`` +- Fix test suite failures due to a qpdf 8.0.0 regression and Python + 3.5's handling of symlink +- The "encrypted PDF" error message was different depending on the type + of PDF encryption. Now a single clear message appears for all types + of PDF encryption. +- ocrmypdf is now in Homebrew. Homebrew users are advised to the + version of ocrmypdf in the official homebrew-core formulas rather + than the private tap. +- Some linting v5.6.0 ------- - -- Fix issue `#216 `_: preserve "text as curves" PDFs without rasterizing file -- Related to the above, messages about rasterizing are more consistent -- For consistency versions minor releases will now get the trailing .0 they always should have had. +====== +- Fix issue + `#216 `__: preserve + "text as curves" PDFs without rasterizing file +- Related to the above, messages about rasterizing are more consistent +- For consistency versions minor releases will now get the trailing .0 + they always should have had. v5.5 ----- - -- Add new argument ``--max-image-mpixels``. Pillow 5.0 now raises an exception when images may be decompression bombs. This argument can be used to override the limit Pillow sets. -- Fix output page cropped when using the sandwich renderer and OCR is skipped on a rotated and image-processed page -- A warning is now issued when old versions of Ghostscript are used in cases known to cause issues with non-Latin characters -- Fix a few parameter validation checks for ``-output-type pdfa-1`` and ``pdfa-2`` +==== +- Add new argument ``--max-image-mpixels``. Pillow 5.0 now raises an + exception when images may be decompression bombs. This argument can + be used to override the limit Pillow sets. +- Fix output page cropped when using the sandwich renderer and OCR is + skipped on a rotated and image-processed page +- A warning is now issued when old versions of Ghostscript are used in + cases known to cause issues with non-Latin characters +- Fix a few parameter validation checks for ``-output-type pdfa-1`` and + ``pdfa-2`` v5.4.4 ------- - -- Fix issue `#181 `_: fix final merge failure for PDFs with more pages than the system file handle limit (``ulimit -n``) -- Fix issue `#200 `_: an uncommon syntax for formatting decimal numbers in a PDF would cause qpdf to issue a warning, which ocrmypdf treated as an error. Now this the warning is relayed. -- Fix an issue where intermediate PDFs would be created at version 1.3 instead of the version of the original file. It's possible but unlikely this had side effects. -- A warning is now issued when older versions of qpdf are used since issues like `#200 `_ cause qpdf to infinite-loop -- Address issue `#140 `_: if Tesseract outputs invalid UTF-8, escape it and print its message instead of aborting with a Unicode error -- Adding previously unlisted setup requirement, pytest-runner -- Update documentation: fix an error in the example script for Synology with Docker images, improved security guidance, advised ``pip install --user`` +====== +- Fix issue + `#181 `__: fix + final merge failure for PDFs with more pages than the system file + handle limit (``ulimit -n``) +- Fix issue + `#200 `__: an + uncommon syntax for formatting decimal numbers in a PDF would cause + qpdf to issue a warning, which ocrmypdf treated as an error. Now this + the warning is relayed. +- Fix an issue where intermediate PDFs would be created at version 1.3 + instead of the version of the original file. It's possible but + unlikely this had side effects. +- A warning is now issued when older versions of qpdf are used since + issues like + `#200 `__ cause + qpdf to infinite-loop +- Address issue + `#140 `__: if + Tesseract outputs invalid UTF-8, escape it and print its message + instead of aborting with a Unicode error +- Adding previously unlisted setup requirement, pytest-runner +- Update documentation: fix an error in the example script for Synology + with Docker images, improved security guidance, advised + ``pip install --user`` v5.4.3 ------- - -- If a subprocess fails to report its version when queried, exit cleanly with an error instead of throwing an exception -- Added test to confirm that the system locale is Unicode-aware and fail early if it's not -- Clarified some copyright information -- Updated pinned requirements.txt so the homebrew formula captures more recent versions +====== +- If a subprocess fails to report its version when queried, exit + cleanly with an error instead of throwing an exception +- Added test to confirm that the system locale is Unicode-aware and + fail early if it's not +- Clarified some copyright information +- Updated pinned requirements.txt so the homebrew formula captures more + recent versions v5.4.2 ------- - -- Fixed a regression from v5.4.1 that caused sidecar files to be created as empty files +====== +- Fixed a regression from v5.4.1 that caused sidecar files to be + created as empty files v5.4.1 ------- - -- Add workaround for Tesseract v4.00alpha crash when trying to obtain orientation and the latest language packs are installed +====== +- Add workaround for Tesseract v4.00alpha crash when trying to obtain + orientation and the latest language packs are installed v5.4 ----- - -- Change wording of a deprecation warning to improve clarity -- Added option to generate PDF/A-1b output if desired (``--output-type pdfa-1``); default remains PDF/A-2b generation -- Update documentation +==== +- Change wording of a deprecation warning to improve clarity +- Added option to generate PDF/A-1b output if desired + (``--output-type pdfa-1``); default remains PDF/A-2b generation +- Update documentation v5.3.3 ------- - -- Fixed missing error message that should occur when trying to force ``--pdf-renderer sandwich`` on old versions of Tesseract -- Update copyright information in test files -- Set system ``LANG`` to UTF-8 in Dockerfiles to avoid UTF-8 encoding errors +====== +- Fixed missing error message that should occur when trying to force + ``--pdf-renderer sandwich`` on old versions of Tesseract +- Update copyright information in test files +- Set system ``LANG`` to UTF-8 in Dockerfiles to avoid UTF-8 encoding + errors v5.3.2 ------- - -- Fixed a broken test case related to language packs +====== +- Fixed a broken test case related to language packs v5.3.1 ------- - -- Fixed wrong return code given for missing Tesseract language packs -- Fixed "brew audit" crashing on Travis when trying to auto-brew +====== +- Fixed wrong return code given for missing Tesseract language packs +- Fixed "brew audit" crashing on Travis when trying to auto-brew v5.3 ----- - -- Added ``--user-words`` and ``--user-patterns`` arguments which are forwarded to Tesseract OCR as words and regular expressions respective to use to guide OCR. Supplying a list of subject-domain words should assist Tesseract with resolving words. (`#165 `_) -- Using a non Latin-1 language with the "hocr" renderer now warns about possible OCR quality and recommends workarounds (`#176 `_) -- Output file path added to error message when that location is not writable (`#175 `_) -- Otherwise valid PDFs with leading whitespace at the beginning of the file are now accepted +==== +- Added ``--user-words`` and ``--user-patterns`` arguments which are + forwarded to Tesseract OCR as words and regular expressions + respective to use to guide OCR. Supplying a list of subject-domain + words should assist Tesseract with resolving words. + (`#165 `__) +- Using a non Latin-1 language with the "hocr" renderer now warns about + possible OCR quality and recommends workarounds + (`#176 `__) +- Output file path added to error message when that location is not + writable + (`#175 `__) +- Otherwise valid PDFs with leading whitespace at the beginning of the + file are now accepted v5.2 ----- - -- When using Tesseract 3.05.01 or newer, OCRmyPDF will select the "sandwich" PDF renderer by default, unless another PDF renderer is specified with the ``--pdf-renderer`` argument. The previous behavior was to select ``--pdf-renderer=hocr``. -- The "tesseract" PDF renderer is now deprecated, since it can cause problems with Ghostscript on Tesseract 3.05.00 -- The "tess4" PDF renderer has been renamed to "sandwich". "tess4" is now a deprecated alias for "sandwich". +==== +- When using Tesseract 3.05.01 or newer, OCRmyPDF will select the + "sandwich" PDF renderer by default, unless another PDF renderer is + specified with the ``--pdf-renderer`` argument. The previous behavior + was to select ``--pdf-renderer=hocr``. +- The "tesseract" PDF renderer is now deprecated, since it can cause + problems with Ghostscript on Tesseract 3.05.00 +- The "tess4" PDF renderer has been renamed to "sandwich". "tess4" is + now a deprecated alias for "sandwich". v5.1 ----- - -- Files with pages larger than 200" (5080 mm) in either dimension are now supported with ``--output-type=pdf`` with the page size preserved (in the PDF specification this feature is called UserUnit scaling). Due to Ghostscript limitations this is not available in conjunction with PDF/A output. +==== +- Files with pages larger than 200" (5080 mm) in either dimension are + now supported with ``--output-type=pdf`` with the page size preserved + (in the PDF specification this feature is called UserUnit scaling). + Due to Ghostscript limitations this is not available in conjunction + with PDF/A output. v5.0.1 ------- +====== -- Fixed issue `#169 `_, exception due to failure to create sidecar text files on some versions of Tesseract 3.04, including the jbarlow83/ocrmypdf Docker image +- Fixed issue + `#169 `__, + exception due to failure to create sidecar text files on some + versions of Tesseract 3.04, including the jbarlow83/ocrmypdf Docker + image v5.0 ----- +==== -- Backward incompatible changes +- Backward incompatible changes - + Support for Python 3.4 dropped. Python 3.5 is now required. - + Support for Tesseract 3.02 and 3.03 dropped. Tesseract 3.04 or newer is required. Tesseract 4.00 (alpha) is supported. - + The OCRmyPDF.sh script was removed. + - Support for Python 3.4 dropped. Python 3.5 is now required. + - Support for Tesseract 3.02 and 3.03 dropped. Tesseract 3.04 or + newer is required. Tesseract 4.00 (alpha) is supported. + - The OCRmyPDF.sh script was removed. -- Add a new feature, ``--sidecar``, which allows creating "sidecar" text files which contain the OCR results in plain text. These OCR text is more reliable than extracting text from PDFs. Closes `#126 `_. -- New feature: ``--pdfa-image-compression``, which allows overriding Ghostscript's lossy-or-lossless image encoding heuristic and making all images JPEG encoded or lossless encoded as desired. Fixes `#163 `_. -- Fixed issue `#143 `_, added ``--quiet`` to suppress "INFO" messages -- Fixed issue `#164 `_, a typo -- Removed the command line parameters ``-n`` and ``--just-print`` since they have not worked for some time (reported as Ubuntu bug `#1687308 `_) +- Add a new feature, ``--sidecar``, which allows creating "sidecar" + text files which contain the OCR results in plain text. These OCR + text is more reliable than extracting text from PDFs. Closes + `#126 `__. + +- New feature: ``--pdfa-image-compression``, which allows overriding + Ghostscript's lossy-or-lossless image encoding heuristic and making + all images JPEG encoded or lossless encoded as desired. Fixes + `#163 `__. + +- Fixed issue + `#143 `__, added + ``--quiet`` to suppress "INFO" messages + +- Fixed issue + `#164 `__, a typo + +- Removed the command line parameters ``-n`` and ``--just-print`` since + they have not worked for some time (reported as Ubuntu bug + `#1687308 `__) v4.5.6 ------- +====== -- Fixed issue `#156 `_, 'NoneType' object has no attribute 'getObject' on pages with no optional /Contents record. This should resolve all issues related to pages with no /Contents record. -- Fixed issue `#158 `_, ocrmypdf now stops and terminates if Ghostscript fails on an intermediate step, as it is not possible to proceed. -- Fixed issue `#160 `_, exception thrown on certain invalid arguments instead of error message +- Fixed issue + `#156 `__, + 'NoneType' object has no attribute 'getObject' on pages with no + optional /Contents record. This should resolve all issues related to + pages with no /Contents record. +- Fixed issue + `#158 `__, ocrmypdf + now stops and terminates if Ghostscript fails on an intermediate + step, as it is not possible to proceed. +- Fixed issue + `#160 `__, + exception thrown on certain invalid arguments instead of error + message v4.5.5 ------- +====== -- Automated update of macOS homebrew tap -- Fixed issue `#154 `_, KeyError '/Contents' when searching for text on blank pages that have no /Contents record. Note: incomplete fix for this issue. +- Automated update of macOS homebrew tap +- Fixed issue + `#154 `__, KeyError + '/Contents' when searching for text on blank pages that have no + /Contents record. Note: incomplete fix for this issue. v4.5.4 ------- +====== -- Fix ``--skip-big`` raising an exception if a page contains no images (`#152 `_) (thanks to @TomRaz) -- Fix an issue where pages with no images might trigger "cannot write mode P as JPEG" (`#151 `_) +- Fix ``--skip-big`` raising an exception if a page contains no images + (`#152 `__) (thanks + to @TomRaz) +- Fix an issue where pages with no images might trigger "cannot write + mode P as JPEG" + (`#151 `__) v4.5.3 ------- +====== -- Added a workaround for Ghostscript 9.21 and probably earlier versions would fail with the error message "VMerror -25", due to a Ghostscript bug in XMP metadata handling -- High Unicode characters (U+10000 and up) are no longer accepted for setting metadata on the command line, as Ghostscript may not handle them correctly. -- Fixed an issue where the ``tess4`` renderer would duplicate content onto output pages if tesseract failed or timed out -- Fixed ``tess4`` renderer not recognized when lossless reconstruction is possible +- Added a workaround for Ghostscript 9.21 and probably earlier versions + would fail with the error message "VMerror -25", due to a Ghostscript + bug in XMP metadata handling +- High Unicode characters (U+10000 and up) are no longer accepted for + setting metadata on the command line, as Ghostscript may not handle + them correctly. +- Fixed an issue where the ``tess4`` renderer would duplicate content + onto output pages if tesseract failed or timed out +- Fixed ``tess4`` renderer not recognized when lossless reconstruction + is possible v4.5.2 ------- +====== -- Fix issue `#147 `_. ``--pdf-renderer tess4 --clean`` will produce an oversized page containing the original image in the bottom left corner, due to loss DPI information. -- Make "using Tesseract 4.0" warning less ominous -- Set up machinery for homebrew OCRmyPDF tap +- Fix issue + `#147 `__. + ``--pdf-renderer tess4 --clean`` will produce an oversized page + containing the original image in the bottom left corner, due to loss + DPI information. +- Make "using Tesseract 4.0" warning less ominous +- Set up machinery for homebrew OCRmyPDF tap v4.5.1 ------- +====== -- Fix issue `#137 `_, proportions of images with a non-square pixel aspect ratio would be distorted in output for ``--force-ocr`` and some other combinations of flags +- Fix issue + `#137 `__, + proportions of images with a non-square pixel aspect ratio would be + distorted in output for ``--force-ocr`` and some other combinations + of flags v4.5 ----- +==== -- PDFs containing "Form XObjects" are now supported (issue `#134 `_; PDF reference manual 8.10), and images they contain are taken into account when determining the resolution for rasterizing -- The Tesseract 4 Docker image no longer includes all languages, because it took so long to build something would tend to fail -- OCRmyPDF now warns about using ``--pdf-renderer tesseract`` with Tesseract 3.04 or lower due to issues with Ghostscript corrupting the OCR text in these cases +- PDFs containing "Form XObjects" are now supported (issue + `#134 `__; PDF + reference manual 8.10), and images they contain are taken into + account when determining the resolution for rasterizing +- The Tesseract 4 Docker image no longer includes all languages, + because it took so long to build something would tend to fail +- OCRmyPDF now warns about using ``--pdf-renderer tesseract`` with + Tesseract 3.04 or lower due to issues with Ghostscript corrupting the + OCR text in these cases v4.4.2 ------- +====== -- The Docker images (ocrmypdf, ocrmypdf-polyglot, ocrmypdf-tess4) are now based on Ubuntu 16.10 instead of Debian stretch +- The Docker images (ocrmypdf, ocrmypdf-polyglot, ocrmypdf-tess4) are + now based on Ubuntu 16.10 instead of Debian stretch - + This makes supporting the Tesseract 4 image easier - + This could be a disruptive change for any Docker users who built customized these images with their own changes, and made those changes in a way that depends on Debian and not Ubuntu + - This makes supporting the Tesseract 4 image easier + - This could be a disruptive change for any Docker users who built + customized these images with their own changes, and made those + changes in a way that depends on Debian and not Ubuntu -- OCRmyPDF now prevents running the Tesseract 4 renderer with Tesseract 3.04, which was permitted in v4.4 and v4.4.1 but will not work +- OCRmyPDF now prevents running the Tesseract 4 renderer with Tesseract + 3.04, which was permitted in v4.4 and v4.4.1 but will not work v4.4.1 ------- +====== -- To prevent a `TIFF output error `_ caused by img2pdf >= 0.2.1 and Pillow <= 3.4.2, dependencies have been tightened -- The Tesseract 4.00 simultaneous process limit was increased from 1 to 2, since it was observed that 1 lowers performance -- Documentation improvements to describe the ``--tesseract-config`` feature -- Added test cases and fixed error handling for ``--tesseract-config`` -- Tweaks to setup.py to deal with issues in the v4.4 release +- To prevent a `TIFF output + error `__ caused + by img2pdf >= 0.2.1 and Pillow <= 3.4.2, dependencies have been + tightened +- The Tesseract 4.00 simultaneous process limit was increased from 1 to + 2, since it was observed that 1 lowers performance +- Documentation improvements to describe the ``--tesseract-config`` + feature +- Added test cases and fixed error handling for ``--tesseract-config`` +- Tweaks to setup.py to deal with issues in the v4.4 release v4.4 ----- +==== -- Tesseract 4.00 is now supported on an experimental basis. +- Tesseract 4.00 is now supported on an experimental basis. - + A new rendering option ``--pdf-renderer tess4`` exploits Tesseract 4's new text-only output PDF mode. See the documentation on PDF Renderers for details. - + The ``--tesseract-oem`` argument allows control over the Tesseract 4 OCR engine mode (tesseract's ``--oem``). Use ``--tesseract-oem 2`` to enforce the new LSTM mode. - + Fixed poor performance with Tesseract 4.00 on Linux + - A new rendering option ``--pdf-renderer tess4`` exploits Tesseract + 4's new text-only output PDF mode. See the documentation on PDF + Renderers for details. + - The ``--tesseract-oem`` argument allows control over the Tesseract + 4 OCR engine mode (tesseract's ``--oem``). Use + ``--tesseract-oem 2`` to enforce the new LSTM mode. + - Fixed poor performance with Tesseract 4.00 on Linux -- Fixed an issue that caused corruption of output to stdout in some cases -- Removed test for Pillow JPEG and PNG support, as the minimum supported version of Pillow now enforces this -- OCRmyPDF now tests that the intended destination file is writable before proceeding -- The test suite now requires ``pytest-helpers-namespace`` to run (but not install) -- Significant code reorganization to make OCRmyPDF re-entrant and improve performance. All changes should be backward compatible for the v4.x series. +- Fixed an issue that caused corruption of output to stdout in some + cases +- Removed test for Pillow JPEG and PNG support, as the minimum + supported version of Pillow now enforces this +- OCRmyPDF now tests that the intended destination file is writable + before proceeding +- The test suite now requires ``pytest-helpers-namespace`` to run (but + not install) +- Significant code reorganization to make OCRmyPDF re-entrant and + improve performance. All changes should be backward compatible for + the v4.x series. - + However, OCRmyPDF's dependency "ruffus" is not re-entrant, so no Python API is available. Scripts should continue to use the command line interface. + - However, OCRmyPDF's dependency "ruffus" is not re-entrant, so no + Python API is available. Scripts should continue to use the + command line interface. v4.3.5 ------- +====== -- Update documentation to confirm Python 3.6.0 compatibility. No code changes were needed, so many earlier versions are likely supported. +- Update documentation to confirm Python 3.6.0 compatibility. No code + changes were needed, so many earlier versions are likely supported. v4.3.4 ------- +====== -- Fixed "decimal.InvalidOperation: quantize result has too many digits" for high DPI images +- Fixed "decimal.InvalidOperation: quantize result has too many digits" + for high DPI images v4.3.3 ------- +====== -- Fixed PDF/A creation with Ghostscript 9.20 properly -- Fixed an exception on inline stencil masks with a missing optional parameter +- Fixed PDF/A creation with Ghostscript 9.20 properly +- Fixed an exception on inline stencil masks with a missing optional + parameter v4.3.2 ------- +====== -- Fixed a PDF/A creation issue with Ghostscript 9.20 (note: this fix did not actually work) +- Fixed a PDF/A creation issue with Ghostscript 9.20 (note: this fix + did not actually work) v4.3.1 ------- +====== -- Fixed an issue where pages produced by the "hocr" renderer after a Tesseract timeout would be rotated incorrectly if the input page was rotated with a /Rotate marker -- Fixed a file handle leak in LeptonicaErrorTrap that would cause a "too many open files" error for files around hundred pages of pages long when ``--deskew`` or ``--remove-background`` or other Leptonica based image processing features were in use, depending on the system value of ``ulimit -n`` -- Ability to specify multiple languages for multilingual documents is now advertised in documentation -- Reduced the file sizes of some test resources -- Cleaned up debug output -- Tesseract caching in test cases is now more cautious about false cache hits and reproducing exact output, not that any problems were observed +- Fixed an issue where pages produced by the "hocr" renderer after a + Tesseract timeout would be rotated incorrectly if the input page was + rotated with a /Rotate marker +- Fixed a file handle leak in LeptonicaErrorTrap that would cause a + "too many open files" error for files around hundred pages of pages + long when ``--deskew`` or ``--remove-background`` or other Leptonica + based image processing features were in use, depending on the system + value of ``ulimit -n`` +- Ability to specify multiple languages for multilingual documents is + now advertised in documentation +- Reduced the file sizes of some test resources +- Cleaned up debug output +- Tesseract caching in test cases is now more cautious about false + cache hits and reproducing exact output, not that any problems were + observed v4.3 ----- +==== -- New feature ``--remove-background`` to detect and erase the background of color and grayscale images -- Better documentation -- Fixed an issue with PDFs that draw images when the raster stack depth is zero -- ocrmypdf can now redirect its output to stdout for use in a shell pipeline +- New feature ``--remove-background`` to detect and erase the + background of color and grayscale images +- Better documentation +- Fixed an issue with PDFs that draw images when the raster stack depth + is zero +- ocrmypdf can now redirect its output to stdout for use in a shell + pipeline - + This does not improve performance since temporary files are still used for buffering - + Some output validation is disabled in this mode + - This does not improve performance since temporary files are still + used for buffering + - Some output validation is disabled in this mode v4.2.5 ------- +====== -- Fixed an issue (`#100 `_) with PDFs that omit the optional /BitsPerComponent parameter on images -- Removed non-free file milk.pdf +- Fixed an issue + (`#100 `__) with + PDFs that omit the optional /BitsPerComponent parameter on images +- Removed non-free file milk.pdf v4.2.4 ------- +====== -- Fixed an error (`#90 `_) caused by PDFs that use stencil masks properly -- Fixed handling of PDFs that try to draw images or stencil masks without properly setting up the graphics state (such images are now ignored for the purposes of calculating DPI) +- Fixed an error + (`#90 `__) caused by + PDFs that use stencil masks properly +- Fixed handling of PDFs that try to draw images or stencil masks + without properly setting up the graphics state (such images are now + ignored for the purposes of calculating DPI) v4.2.3 ------- +====== -- Fixed an issue with PDFs that store page rotation (/Rotate) in an indirect object -- Integrated a few fixes to simplify downstream packaging (Debian) +- Fixed an issue with PDFs that store page rotation (/Rotate) in an + indirect object +- Integrated a few fixes to simplify downstream packaging (Debian) - + The test suite no longer assumes it is installed - + If running Linux, skip a test that passes Unicode on the command line + - The test suite no longer assumes it is installed + - If running Linux, skip a test that passes Unicode on the command + line -- Added a test case to check explicit masks and stencil masks -- Added a test case for indirect objects and linearized PDFs -- Deprecated the OCRmyPDF.sh shell script +- Added a test case to check explicit masks and stencil masks +- Added a test case for indirect objects and linearized PDFs +- Deprecated the OCRmyPDF.sh shell script v4.2.2 ------- +====== -- Improvements to documentation +- Improvements to documentation v4.2.1 ------- +====== -- Fixed an issue where PDF pages that contained stencil masks would report an incorrect DPI and cause Ghostscript to abort -- Implemented stdin streaming +- Fixed an issue where PDF pages that contained stencil masks would + report an incorrect DPI and cause Ghostscript to abort +- Implemented stdin streaming v4.2 ----- +==== -- ocrmypdf will now try to convert single image files to PDFs if they are provided as input (`#15 `_) +- ocrmypdf will now try to convert single image files to PDFs if they + are provided as input + (`#15 `__) - + This is a basic convenience feature. It only supports a single image and always makes the image fill the whole page. - + For better control over image to PDF conversion, use ``img2pdf`` (one of ocrmypdf's dependencies) + - This is a basic convenience feature. It only supports a single + image and always makes the image fill the whole page. + - For better control over image to PDF conversion, use ``img2pdf`` + (one of ocrmypdf's dependencies) -- New argument ``--output-type {pdf|pdfa}`` allows disabling Ghostscript PDF/A generation +- New argument ``--output-type {pdf|pdfa}`` allows disabling + Ghostscript PDF/A generation - + ``pdfa`` is the default, consistent with past behavior - + ``pdf`` provides a workaround for users concerned about the increase in file size from Ghostscript forcing JBIG2 images to CCITT and transcoding JPEGs - + ``pdf`` preserves as much as it can about the original file, including problems that PDF/A conversion fixes + - ``pdfa`` is the default, consistent with past behavior + - ``pdf`` provides a workaround for users concerned about the + increase in file size from Ghostscript forcing JBIG2 images to + CCITT and transcoding JPEGs + - ``pdf`` preserves as much as it can about the original file, + including problems that PDF/A conversion fixes -- PDFs containing images with "non-square" pixel aspect ratios, such as 200x100 DPI, are now handled and converted properly (fixing a bug that caused to be cropped) -- ``--force-ocr`` rasterizes pages even if they contain no images +- PDFs containing images with "non-square" pixel aspect ratios, such as + 200x100 DPI, are now handled and converted properly (fixing a bug + that caused to be cropped) +- ``--force-ocr`` rasterizes pages even if they contain no images - + supports users who want to use OCRmyPDF to reconstruct text information in PDFs with damaged Unicode maps (copy and paste text does not match displayed text) - + supports reinterpreting PDFs where text was rendered as curves for printing, and text needs to be recovered - + fixes issue `#82 `_ + - supports users who want to use OCRmyPDF to reconstruct text + information in PDFs with damaged Unicode maps (copy and paste text + does not match displayed text) + - supports reinterpreting PDFs where text was rendered as curves for + printing, and text needs to be recovered + - fixes issue + `#82 `__ -- Fixes an issue where, with certain settings, monochrome images in PDFs would be converted to 8-bit grayscale, increasing file size (`#79 `_) -- Support for Ubuntu 12.04 LTS "precise" has been dropped in favor of (roughly) Ubuntu 14.04 LTS "trusty" +- Fixes an issue where, with certain settings, monochrome images in + PDFs would be converted to 8-bit grayscale, increasing file size + (`#79 `__) +- Support for Ubuntu 12.04 LTS "precise" has been dropped in favor of + (roughly) Ubuntu 14.04 LTS "trusty" - + Some Ubuntu "PPAs" (backports) are needed to make it work + - Some Ubuntu "PPAs" (backports) are needed to make it work -- Support for some older dependencies dropped +- Support for some older dependencies dropped - + Ghostscript 9.15 or later is now required (available in Ubuntu trusty with backports) - + Tesseract 3.03 or later is now required (available in Ubuntu trusty) + - Ghostscript 9.15 or later is now required (available in Ubuntu + trusty with backports) + - Tesseract 3.03 or later is now required (available in Ubuntu + trusty) -- Ghostscript now runs in "safer" mode where possible +- Ghostscript now runs in "safer" mode where possible v4.1.4 ------- +====== -- Bug fix: monochrome images with an ICC profile attached were incorrectly converted to full color images if lossless reconstruction was not possible due to other settings; consequence was increased file size for these images +- Bug fix: monochrome images with an ICC profile attached were + incorrectly converted to full color images if lossless reconstruction + was not possible due to other settings; consequence was increased + file size for these images v4.1.3 ------- +====== -- More helpful error message for PDFs with version 4 security handler -- Update usage instructions for Windows/Docker users -- Fix order of operations for matrix multiplication (no effect on most users) -- Add a few leptonica wrapper functions (no effect on most users) +- More helpful error message for PDFs with version 4 security handler +- Update usage instructions for Windows/Docker users +- Fix order of operations for matrix multiplication (no effect on most + users) +- Add a few leptonica wrapper functions (no effect on most users) v4.1.2 ------- +====== -- Replace IEC sRGB ICC profile with Debian's sRGB (from icc-profiles-free) which is more compatible with the MIT license -- More helpful error message for an error related to certain types of malformed PDFs +- Replace IEC sRGB ICC profile with Debian's sRGB (from + icc-profiles-free) which is more compatible with the MIT license +- More helpful error message for an error related to certain types of + malformed PDFs v4.1 ----- +==== -- ``--rotate-pages`` now only rotates pages when reasonably confidence in the orientation. This behavior can be adjusted with the new argument ``--rotate-pages-threshold`` -- Fixed problems in error checking if ``unpaper`` is uninstalled or missing at run-time -- Fixed problems with "RethrownJobError" errors during error handling that suppressed the useful error messages +- ``--rotate-pages`` now only rotates pages when reasonably confidence + in the orientation. This behavior can be adjusted with the new + argument ``--rotate-pages-threshold`` +- Fixed problems in error checking if ``unpaper`` is uninstalled or + missing at run-time +- Fixed problems with "RethrownJobError" errors during error handling + that suppressed the useful error messages v4.0.7 ------- +====== -- Minor correction to Ghostscript output settings +- Minor correction to Ghostscript output settings v4.0.6 ------- +====== -- Update install instructions -- Provide a sRGB profile instead of using Ghostscript's +- Update install instructions +- Provide a sRGB profile instead of using Ghostscript's v4.0.5 ------- +====== -- Remove some verbose debug messages from v4.0.4 -- Fixed temporary that wasn't being deleted -- DPI is now calculated correctly for cropped images, along with other image transformations -- Inline images are now checked during DPI calculation instead of rejecting the image +- Remove some verbose debug messages from v4.0.4 +- Fixed temporary that wasn't being deleted +- DPI is now calculated correctly for cropped images, along with other + image transformations +- Inline images are now checked during DPI calculation instead of + rejecting the image v4.0.4 ------- +====== -Released with verbose debug message turned on. Do not use. Skip to v4.0.5. +Released with verbose debug message turned on. Do not use. Skip to +v4.0.5. v4.0.3 ------- +====== New features -- Page orientations detected are now reported in a summary comment +- Page orientations detected are now reported in a summary comment Fixes -- Show stack trace if unexpected errors occur -- Treat "too few characters" error message from Tesseract as a reason to skip that page rather than - abort the file -- Docker: fix blank JPEG2000 issue by insisting on Ghostscript versions that have this fixed - +- Show stack trace if unexpected errors occur +- Treat "too few characters" error message from Tesseract as a reason + to skip that page rather than abort the file +- Docker: fix blank JPEG2000 issue by insisting on Ghostscript versions + that have this fixed v4.0.2 ------- +====== Fixes - -- Fixed compatibility with Tesseract 3.04.01 release, particularly its different way of outputting - orientation information -- Improved handling of Tesseract errors and crashes -- Fixed use of chmod on Docker that broke most test cases - +- Fixed compatibility with Tesseract 3.04.01 release, particularly its + different way of outputting orientation information +- Improved handling of Tesseract errors and crashes +- Fixed use of chmod on Docker that broke most test cases v4.0.1 ------- +====== Fixes - -- Fixed a KeyError if tesseract fails to find page orientation information - +- Fixed a KeyError if tesseract fails to find page orientation + information v4.0 ----- +==== New features -- Automatic page rotation (``-r``) is now available. It uses ignores any prior rotation information - on PDFs and sets rotation based on the dominant orientation of detectable text. This feature is - fairly reliable but some false positives occur especially if there is not much text to work with. (`#4 `_) -- Deskewing is now performed using Leptonica instead of unpaper. Leptonica is faster and more reliable - at image deskewing than unpaper. - +- Automatic page rotation (``-r``) is now available. It uses ignores + any prior rotation information on PDFs and sets rotation based on the + dominant orientation of detectable text. This feature is fairly + reliable but some false positives occur especially if there is not + much text to work with. + (`#4 `__) +- Deskewing is now performed using Leptonica instead of unpaper. + Leptonica is faster and more reliable at image deskewing than + unpaper. Fixes -- Fixed an issue where lossless reconstruction could cause some pages to be appear incorrectly - if the page was rotated by the user in Acrobat after being scanned (specifically if it a /Rotate tag) -- Fixed an issue where lossless reconstruction could misalign the graphics layer with respect to - text layer if the page had been cropped such that its origin is not (0, 0) (`#49 `_) - +- Fixed an issue where lossless reconstruction could cause some pages + to be appear incorrectly if the page was rotated by the user in + Acrobat after being scanned (specifically if it a /Rotate tag) +- Fixed an issue where lossless reconstruction could misalign the + graphics layer with respect to text layer if the page had been + cropped such that its origin is not (0, 0) + (`#49 `__) Changes -- Logging output is now much easier to read -- ``--deskew`` is now performed by Leptonica instead of unpaper (`#25 `_) -- libffi is now required -- Some changes were made to the Docker and Travis build environments to support libffi -- ``--pdf-renderer=tesseract`` now displays a warning if the Tesseract version is less than 3.04.01, - the planned release that will include fixes to an important OCR text rendering bug in Tesseract 3.04.00. - You can also manually install ./share/sharp2.ttf on top of pdf.ttf in your Tesseract tessdata folder - to correct the problem. - +- Logging output is now much easier to read +- ``--deskew`` is now performed by Leptonica instead of unpaper + (`#25 `__) +- libffi is now required +- Some changes were made to the Docker and Travis build environments to + support libffi +- ``--pdf-renderer=tesseract`` now displays a warning if the Tesseract + version is less than 3.04.01, the planned release that will include + fixes to an important OCR text rendering bug in Tesseract 3.04.00. + You can also manually install ./share/sharp2.ttf on top of pdf.ttf in + your Tesseract tessdata folder to correct the problem. v3.2.1 ------- +====== Changes -- Fixed issue `#47 `_ "convert() got and unexpected keyword argument 'dpi'" by upgrading to img2pdf 0.2 -- Tweaked the Dockerfiles - +- Fixed issue `#47 `__ + "convert() got and unexpected keyword argument 'dpi'" by upgrading to + img2pdf 0.2 +- Tweaked the Dockerfiles v3.2 ----- +==== New features -- Lossless reconstruction: when possible, OCRmyPDF will inject text layers without - otherwise manipulating the content and layout of a PDF page. For example, a PDF containing a mix - of vector and raster content would see the vector content preserved. Images may still be transcoded - during PDF/A conversion. (``--deskew`` and ``--clean-final`` disable this mode, necessarily.) -- New argument ``--tesseract-pagesegmode`` allows you to pass page segmentation arguments to Tesseract OCR. - This helps for two column text and other situations that confuse Tesseract. -- Added a new "polyglot" version of the Docker image, that generates Tesseract with all languages packs installed, - for the polyglots among us. It is much larger. +- Lossless reconstruction: when possible, OCRmyPDF will inject text + layers without otherwise manipulating the content and layout of a PDF + page. For example, a PDF containing a mix of vector and raster + content would see the vector content preserved. Images may still be + transcoded during PDF/A conversion. (``--deskew`` and + ``--clean-final`` disable this mode, necessarily.) +- New argument ``--tesseract-pagesegmode`` allows you to pass page + segmentation arguments to Tesseract OCR. This helps for two column + text and other situations that confuse Tesseract. +- Added a new "polyglot" version of the Docker image, that generates + Tesseract with all languages packs installed, for the polyglots among + us. It is much larger. Changes -- JPEG transcoding quality is now 95 instead of the default 75. Bigger file sizes for less degradation. - - +- JPEG transcoding quality is now 95 instead of the default 75. Bigger + file sizes for less degradation. v3.1.1 ------- +====== Changes -- Fixed bug that caused incorrect page size and DPI calculations on documents with mixed page sizes +- Fixed bug that caused incorrect page size and DPI calculations on + documents with mixed page sizes v3.1 ----- +==== Changes -- Default output format is now PDF/A-2b instead of PDF/A-1b -- Python 3.5 and macOS El Capitan are now supported platforms - no changes were - needed to implement support -- Improved some error messages related to missing input files -- Fixed issue `#20 `_ - uppercase .PDF extension not accepted -- Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text, - such as OCR text produced by Tesseract 3.04 -- Inserts /Creator tag into PDFs so that errors can be traced back to this project -- Added new option ``--pdf-renderer=auto``, to let OCRmyPDF pick the best PDF renderer. - Currently it always chooses the 'hocrtransform' renderer but that behavior may change. -- Set up Travis CI automatic integration testing +- Default output format is now PDF/A-2b instead of PDF/A-1b +- Python 3.5 and macOS El Capitan are now supported platforms - no + changes were needed to implement support +- Improved some error messages related to missing input files +- Fixed issue `#20 `__ + - uppercase .PDF extension not accepted +- Fixed an issue where OCRmyPDF failed to text that certain pages + contained previously OCR'ed text, such as OCR text produced by + Tesseract 3.04 +- Inserts /Creator tag into PDFs so that errors can be traced back to + this project +- Added new option ``--pdf-renderer=auto``, to let OCRmyPDF pick the + best PDF renderer. Currently it always chooses the 'hocrtransform' + renderer but that behavior may change. +- Set up Travis CI automatic integration testing v3.0 ----- +==== New features -- Easier installation with a Docker container or Python's ``pip`` package manager -- Eliminated many external dependencies, so it's easier to setup -- Now installs ``ocrmypdf`` to ``/usr/local/bin`` or equivalent for system-wide - access and easier typing -- Improved command line syntax and usage help (``--help``) -- Tesseract 3.03+ PDF page rendering can be used instead for better positioning - of recognized text (``--pdf-renderer tesseract``) -- PDF metadata (title, author, keywords) are now transferred to the - output PDF -- PDF metadata can also be set from the command line (``--title``, etc.) -- Automatic repairs malformed input PDFs if possible -- Added test cases to confirm everything is working -- Added option to skip extremely large pages that take too long to OCR and are - often not OCRable (e.g. large scanned maps or diagrams); other pages are still - processed (``--skip-big``) -- Added option to kill Tesseract OCR process if it seems to be taking too long on - a page, while still processing other pages (``--tesseract-timeout``) -- Less common colorspaces (CMYK, palette) are now supported by conversion to RGB -- Multiple images on the same PDF page are now supported +- Easier installation with a Docker container or Python's ``pip`` + package manager +- Eliminated many external dependencies, so it's easier to setup +- Now installs ``ocrmypdf`` to ``/usr/local/bin`` or equivalent for + system-wide access and easier typing +- Improved command line syntax and usage help (``--help``) +- Tesseract 3.03+ PDF page rendering can be used instead for better + positioning of recognized text (``--pdf-renderer tesseract``) +- PDF metadata (title, author, keywords) are now transferred to the + output PDF +- PDF metadata can also be set from the command line (``--title``, + etc.) +- Automatic repairs malformed input PDFs if possible +- Added test cases to confirm everything is working +- Added option to skip extremely large pages that take too long to OCR + and are often not OCRable (e.g. large scanned maps or diagrams); + other pages are still processed (``--skip-big``) +- Added option to kill Tesseract OCR process if it seems to be taking + too long on a page, while still processing other pages + (``--tesseract-timeout``) +- Less common colorspaces (CMYK, palette) are now supported by + conversion to RGB +- Multiple images on the same PDF page are now supported Changes -- New, robust rewrite in Python 3.4+ with ruffus_ pipelines -- Now uses Ghostscript 9.14's improved color conversion model to preserve PDF colors -- OCR text is now rendered in the PDF as invisible text. Previous versions of OCRmyPDF - incorrectly rendered visible text with an image on top. -- All "tasks" in the pipeline can be executed in parallel on any - available CPUs, increasing performance -- The ``-o DPI`` argument has been phased out, in favor of ``--oversample DPI``, in - case we need ``-o OUTPUTFILE`` in the future -- Removed several dependencies, so it's easier to install. We no - longer use: +- New, robust rewrite in Python 3.4+ with + `ruffus `__ pipelines +- Now uses Ghostscript 9.14's improved color conversion model to + preserve PDF colors +- OCR text is now rendered in the PDF as invisible text. Previous + versions of OCRmyPDF incorrectly rendered visible text with an image + on top. +- All "tasks" in the pipeline can be executed in parallel on any + available CPUs, increasing performance +- The ``-o DPI`` argument has been phased out, in favor of + ``--oversample DPI``, in case we need ``-o OUTPUTFILE`` in the future +- Removed several dependencies, so it's easier to install. We no longer + use: - - GNU parallel_ - - ImageMagick_ - - Python 2.7 - - Poppler - - MuPDF_ tools - - shell scripts - - Java and JHOVE_ - - libxml2 + - GNU `parallel `__ + - `ImageMagick `__ + - Python 2.7 + - Poppler + - `MuPDF `__ tools + - shell scripts + - Java and `JHOVE `__ + - libxml2 -- Some new external dependencies are required or optional, compared to v2.x: +- Some new external dependencies are required or optional, compared to + v2.x: - - Ghostscript 9.14+ - - qpdf_ 5.0.0+ - - Unpaper_ 6.1 (optional) - - some automatically managed Python packages - -.. _ruffus: http://www.ruffus.org.uk/index.html -.. _parallel: https://www.gnu.org/software/parallel/ -.. _ImageMagick: http://www.imagemagick.org/script/index.php -.. _MuPDF: http://mupdf.com/docs/ -.. _qpdf: http://qpdf.sourceforge.net/ -.. _Unpaper: https://github.com/Flameeyes/unpaper -.. _JHOVE: http://jhove.sourceforge.net/ + - Ghostscript 9.14+ + - `qpdf `__ 5.0.0+ + - `Unpaper `__ 6.1 (optional) + - some automatically managed Python packages Release candidates^ -- rc9: +- rc9: - - fix issue `#118 `_: report error if ghostscript iccprofiles are missing - - fixed another issue related to `#111 `_: PDF rasterized to palette file - - add support image files with a palette - - don't try to validate PDF file after an exception occurs + - fix issue + `#118 `__: + report error if ghostscript iccprofiles are missing + - fixed another issue related to + `#111 `__: PDF + rasterized to palette file + - add support image files with a palette + - don't try to validate PDF file after an exception occurs -- rc8: +- rc8: - - fix issue `#111 `_: exception thrown if PDF is missing DocumentInfo dictionary + - fix issue + `#111 `__: + exception thrown if PDF is missing DocumentInfo dictionary -- rc7: +- rc7: - - fix error when installing direct from pip, "no such file 'requirements.txt'" + - fix error when installing direct from pip, "no such file + 'requirements.txt'" -- rc6: +- rc6: - - dropped libxml2 (Python lxml) since Python 3's internal XML parser is sufficient - - set up Docker container - - fix Unicode errors if recognized text contains Unicode characters and system locale is not UTF-8 + - dropped libxml2 (Python lxml) since Python 3's internal XML parser + is sufficient + - set up Docker container + - fix Unicode errors if recognized text contains Unicode characters + and system locale is not UTF-8 -- rc5: +- rc5: - - dropped Java and JHOVE in favour of qpdf - - improved command line error output - - additional tests and bug fixes - - tested on Ubuntu 14.04 LTS + - dropped Java and JHOVE in favour of qpdf + - improved command line error output + - additional tests and bug fixes + - tested on Ubuntu 14.04 LTS -- rc4: +- rc4: - - dropped MuPDF in favour of qpdf - - fixed some installer issues and errors in installation instructions - - improve performance: run Ghostscript with multithreaded rendering - - improve performance: use multiple cores by default - - bug fix: checking for wrong exception on process timeout + - dropped MuPDF in favour of qpdf + - fixed some installer issues and errors in installation + instructions + - improve performance: run Ghostscript with multithreaded rendering + - improve performance: use multiple cores by default + - bug fix: checking for wrong exception on process timeout -- rc3: skipping version number intentionally to avoid confusion with Tesseract -- rc2: first release for public testing to test-PyPI, Github -- rc1: testing release process +- rc3: skipping version number intentionally to avoid confusion with + Tesseract +- rc2: first release for public testing to test-PyPI, Github +- rc1: testing release process Compatibility notes -------------------- +=================== -- ``./OCRmyPDF.sh`` script is still available for now -- Stacking the verbosity option like ``-vvv`` is no longer supported - -- The configuration file ``config.sh`` has been removed. Instead, you can - feed a file to the arguments for common settings: +- ``./OCRmyPDF.sh`` script is still available for now +- Stacking the verbosity option like ``-vvv`` is no longer supported +- The configuration file ``config.sh`` has been removed. Instead, you + can feed a file to the arguments for common settings: :: - ocrmypdf input.pdf output.pdf @settings.txt + ocrmypdf input.pdf output.pdf @settings.txt where ``settings.txt`` contains *one argument per line*, for example: :: - -l - deu - --author - A. Merkel - --pdf-renderer - tesseract - + -l + deu + --author + A. Merkel + --pdf-renderer + tesseract Fixes - -- Handling of filenames containing spaces: fixed +- Handling of filenames containing spaces: fixed Notes and known issues -- Some dependencies may work with lower versions than tested, so try - overriding dependencies if they are "in the way" to see if they work. - -- ``--pdf-renderer tesseract`` will output files with an incorrect page size in Tesseract 3.03, - due to a bug in Tesseract. - -- PDF files containing "inline images" are not supported and won't be for the 3.0 release. Scanned - images almost never contain inline images. - +- Some dependencies may work with lower versions than tested, so try + overriding dependencies if they are "in the way" to see if they work. +- ``--pdf-renderer tesseract`` will output files with an incorrect page + size in Tesseract 3.03, due to a bug in Tesseract. +- PDF files containing "inline images" are not supported and won't be + for the 3.0 release. Scanned images almost never contain inline + images. v2.2-stable (2014-09-29) ------------------------- +======================== -OCRmyPDF versions 1 and 2 were implemented as shell scripts. OCRmyPDF 3.0+ is a fork that gradually replaced all shell scripts with Python while maintaining the existing command line arguments. No one is maintaining old versions. +OCRmyPDF versions 1 and 2 were implemented as shell scripts. OCRmyPDF +3.0+ is a fork that gradually replaced all shell scripts with Python +while maintaining the existing command line arguments. No one is +maintaining old versions. -For details on older versions, see the `final version of its release notes `_. +For details on older versions, see the `final version of its release +notes `__. diff --git a/docs/security.rst b/docs/security.rst index 1db8bf7e..bcc69e8e 100644 --- a/docs/security.rst +++ b/docs/security.rst @@ -1,79 +1,162 @@ +=================== PDF security issues =================== - OCRmyPDF should only be used on PDFs you trust. It is not designed to protect you against malware. + OCRmyPDF should only be used on PDFs you trust. It is not designed to + protect you against malware. -Recognizing that many users have an interest in handling PDFs and applying OCR to PDFs they did not generate themselves, this article discusses the security implications of PDFs and how users can protect themselves. +Recognizing that many users have an interest in handling PDFs and +applying OCR to PDFs they did not generate themselves, this article +discusses the security implications of PDFs and how users can protect +themselves. The disclaimer applies: this software has no warranties of any kind. PDFs may contain malware ------------------------- +======================== -PDF is a rich, complex file format. The official PDF 1.7 specification, ISO 32000:2008, is hundreds of pages long and references several annexes each of which are similar in length. PDFs can contain video, audio, XML, JavaScript and other programming, and forms. In some cases, they can open internet connections to pre-selected URLs. All of these possible attack vectors. +PDF is a rich, complex file format. The official PDF 1.7 specification, +ISO 32000:2008, is hundreds of pages long and references several annexes +each of which are similar in length. PDFs can contain video, audio, XML, +JavaScript and other programming, and forms. In some cases, they can +open internet connections to pre-selected URLs. All of these possible +attack vectors. -In short, PDFs `may contain viruses `_. +In short, PDFs `may contain +viruses `__. -This `article `_ describes a high-paranoia method which allows potentially hostile PDFs to be viewed and rasterized safely in a disposable virtual machine. A trusted PDF created in this manner is converted to images and loses all information making it searchable and losing all compression. OCRmyPDF could be used restore searchability. +This +`article `__ +describes a high-paranoia method which allows potentially hostile PDFs +to be viewed and rasterized safely in a disposable virtual machine. A +trusted PDF created in this manner is converted to images and loses all +information making it searchable and losing all compression. OCRmyPDF +could be used restore searchability. How OCRmyPDF processes PDFs ---------------------------- +=========================== -OCRmyPDF must open and interpret your PDF in order to insert an OCR layer. First, it runs all PDFs through `pikepdf `_, a library based on `qpdf `_, a program that repairs PDFs with syntax errors. This is done because, in the author's experience, a significant number of PDFs in the wild especially those created by scanners are not well-formed files. qpdf makes it more likely that OCRmyPDF will succeed, but offers no security guarantees. qpdf is also used to split the PDF into single page PDFs. +OCRmyPDF must open and interpret your PDF in order to insert an OCR +layer. First, it runs all PDFs through +`pikepdf `__, a library based on +`qpdf `__, a program that repairs PDFs +with syntax errors. This is done because, in the author's experience, a +significant number of PDFs in the wild especially those created by +scanners are not well-formed files. qpdf makes it more likely that +OCRmyPDF will succeed, but offers no security guarantees. qpdf is also +used to split the PDF into single page PDFs. -Finally, OCRmyPDF rasterizes each page of the PDF using `Ghostscript `_ in ``-dSAFER`` mode. +Finally, OCRmyPDF rasterizes each page of the PDF using +`Ghostscript `__ in ``-dSAFER`` mode. -Depending on the options specified, OCRmyPDF may graft the OCR layer into the existing PDF or it may essentially reconstruct ("re-fry") a visually identical PDF that may be quite different at the binary level. That said, OCRmyPDF is not a tool designed for sanitizing PDFs. +Depending on the options specified, OCRmyPDF may graft the OCR layer +into the existing PDF or it may essentially reconstruct ("re-fry") a +visually identical PDF that may be quite different at the binary level. +That said, OCRmyPDF is not a tool designed for sanitizing PDFs. .. _ocr-service: Using OCRmyPDF online or as a service -------------------------------------- +===================================== -OCRmyPDF is not designed for use as a public web service where a malicious user could upload a chosen PDF. In particular, it is not necessarily secure against PDF malware or PDFs that cause denial of service. OCRmyPDF relies on Ghostscript, and therefore, if deployed online one should be prepared to comply with Ghostscript's Affero GPL license, OCRmyPDF's GPL license, and any other licenses. +OCRmyPDF is not designed for use as a public web service where a +malicious user could upload a chosen PDF. In particular, it is not +necessarily secure against PDF malware or PDFs that cause denial of +service. OCRmyPDF relies on Ghostscript, and therefore, if deployed +online one should be prepared to comply with Ghostscript's Affero GPL +license, OCRmyPDF's GPL license, and any other licenses. -Setting aside these concerns, a side effect of OCRmyPDF is it may incidentally sanitize PDFs that contain certain types of malware. It runs ``qpdf`` to repair the PDF, which could correct malformed PDF structures that are part of an attack. When PDF/A output is selected (the default), the input PDF is partially reconstructed by Ghostscript. When ``--force-ocr`` is used, all pages are rasterized and reconverted to PDF, which could remove malware in embedded images. +Setting aside these concerns, a side effect of OCRmyPDF is it may +incidentally sanitize PDFs that contain certain types of malware. It +runs ``qpdf`` to repair the PDF, which could correct malformed PDF +structures that are part of an attack. When PDF/A output is selected +(the default), the input PDF is partially reconstructed by Ghostscript. +When ``--force-ocr`` is used, all pages are rasterized and reconverted +to PDF, which could remove malware in embedded images. -OCRmyPDF should be relatively safe to use in a trusted intranet, with some considerations: +OCRmyPDF should be relatively safe to use in a trusted intranet, with +some considerations: Limiting CPU usage -^^^^^^^^^^^^^^^^^^ +------------------ -OCRmyPDF will attempt to use all available CPUs and storage, so executing ``nice ocrmypdf`` or limiting the number of jobs with the ``-j`` argument may ensure the server remains available. Another option would be run OCRmyPDF jobs inside a Docker container, a virtual machine, or a cloud instance, which can impose its own limits on CPU usage and be terminated "from orbit" if it fails to complete. +OCRmyPDF will attempt to use all available CPUs and storage, so +executing ``nice ocrmypdf`` or limiting the number of jobs with the +``-j`` argument may ensure the server remains available. Another option +would be run OCRmyPDF jobs inside a Docker container, a virtual machine, +or a cloud instance, which can impose its own limits on CPU usage and be +terminated "from orbit" if it fails to complete. Temporary storage requirements -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +------------------------------ -OCRmyPDF will use a large amount of temporary storage for its work, proportional to the total number of pixels needed to rasterize the PDF. The raster image of a 8.5×11" color page at 300 DPI takes 25 MB uncompressed; OCRmyPDF saves its intermediates as PNG, but that still means it requires about 9 MB per intermediate based on average compression ratios. Multiple intermediates per page are also required, depending on the command line given. A rule of thumb would be to allow 100 MB of temporary storage per page in a file – meaning that a small cloud servers or small VM partitions should be provisioned with plenty of extra space, if say, a 500 page file might be sent. +OCRmyPDF will use a large amount of temporary storage for its work, +proportional to the total number of pixels needed to rasterize the PDF. +The raster image of a 8.5×11" color page at 300 DPI takes 25 MB +uncompressed; OCRmyPDF saves its intermediates as PNG, but that still +means it requires about 9 MB per intermediate based on average +compression ratios. Multiple intermediates per page are also required, +depending on the command line given. A rule of thumb would be to allow +100 MB of temporary storage per page in a file – meaning that a small +cloud servers or small VM partitions should be provisioned with plenty +of extra space, if say, a 500 page file might be sent. -To check temporary storage usage on actual files, run ``ocrmypdf -k ...`` which will preserve and print the path to temporary storage when the job is done. +To check temporary storage usage on actual files, run +``ocrmypdf -k ...`` which will preserve and print the path to temporary +storage when the job is done. -To change where temporary files are stored, change the ``TMPDIR`` environment variable for ocrmypdf's environment. (Python's ``tempfile.gettempdir()`` returns the root directory in which temporary files will be stored.) For example, one could redirect ``TMPDIR`` to a large RAM disk to avoid wear on HDD/SSD and potentially improve performance. On Amazon Web Services, ``TMPDIR`` can be set to `empheral storage `_. +To change where temporary files are stored, change the ``TMPDIR`` +environment variable for ocrmypdf's environment. (Python's +``tempfile.gettempdir()`` returns the root directory in which temporary +files will be stored.) For example, one could redirect ``TMPDIR`` to a +large RAM disk to avoid wear on HDD/SSD and potentially improve +performance. On Amazon Web Services, ``TMPDIR`` can be set to `empheral +storage `__. Timeouts -^^^^^^^^ +-------- -To prevent excessively long OCR jobs consider setting ``--tesseract-timeout`` and/or ``--skip-big`` arguments. ``--skip-big`` is particularly helpful if your PDFs include documents such as reports on standard page sizes with large images attached - often large images are not worth OCR'ing anyway. +To prevent excessively long OCR jobs consider setting +``--tesseract-timeout`` and/or ``--skip-big`` arguments. ``--skip-big`` +is particularly helpful if your PDFs include documents such as reports +on standard page sizes with large images attached - often large images +are not worth OCR'ing anyway. Commercial alternatives -^^^^^^^^^^^^^^^^^^^^^^^ +----------------------- -The author also provides professional services that include OCR and building databases around PDFs, and is happy to provide consultation. - -Abbyy Cloud OCR is a viable commercial alternative with a web services API. +The author also provides professional services that include OCR and +building databases around PDFs, and is happy to provide consultation. +Abbyy Cloud OCR is a viable commercial alternative with a web services +API. Password protection, digital signatures and certification ---------------------------------------------------------- +========================================================= -Password protected PDFs usually have two passwords, and owner and user password. When the user password is set to empty, PDF readers will open the file automatically and marked it as "(SECURED)". While not as reliable as a digital signature, this indicates that whoever set the password approved of the file at that time. When the user password is set, the document cannot be viewed without the password. +Password protected PDFs usually have two passwords, and owner and user +password. When the user password is set to empty, PDF readers will open +the file automatically and marked it as "(SECURED)". While not as +reliable as a digital signature, this indicates that whoever set the +password approved of the file at that time. When the user password is +set, the document cannot be viewed without the password. -Either way, OCRmyPDF does not remove passwords from PDFs and exits with an error on encountering them. +Either way, OCRmyPDF does not remove passwords from PDFs and exits with +an error on encountering them. -``qpdf``, one of OCRmyPDF's dependencies, can remove passwords. If the owner and user password are set, a password is required for ``qpdf``. If only the owner password is set, then the password can be stripped, even if one does not have the owner password. +``qpdf``, one of OCRmyPDF's dependencies, can remove passwords. If the +owner and user password are set, a password is required for ``qpdf``. If +only the owner password is set, then the password can be stripped, even +if one does not have the owner password. -After OCR is applied, password protection is not permitted on PDF/A documents but the file can be converted to regular PDF. +After OCR is applied, password protection is not permitted on PDF/A +documents but the file can be converted to regular PDF. -Many programs exist which are capable of inserting an image of someone's signature. On its own, this offers no security guarantees. It is trivial to remove the signature image and apply it to other files. This practice offers no real security. +Many programs exist which are capable of inserting an image of someone's +signature. On its own, this offers no security guarantees. It is trivial +to remove the signature image and apply it to other files. This practice +offers no real security. -Important documents can be digitally signed and certified to attest to their authorship. OCRmyPDF cannot do this. Open source tools such as pdfbox (Java) have this capability as does Adobe Acrobat. +Important documents can be digitally signed and certified to attest to +their authorship. OCRmyPDF cannot do this. Open source tools such as +pdfbox (Java) have this capability as does Adobe Acrobat. From 11a57c7a176f49cb889ffd8847f938f892fc29f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 23 Jun 2019 16:54:43 -0700 Subject: [PATCH 101/880] Drop --mask-barcodes feature --- docs/cookbook.rst | 4 ---- src/ocrmypdf/_pipeline.py | 11 ++--------- src/ocrmypdf/api.py | 1 - src/ocrmypdf/cli.py | 7 ------- 4 files changed, 2 insertions(+), 21 deletions(-) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index bc6ca951..617a6046 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -174,10 +174,6 @@ might remove desirable content, especially from poor quality scans. - ``--clean-final`` uses unpaper to clean up pages before OCR and inserts the page into the final output. You will want to review each page to ensure that unpaper did not remove something important. -- ``--mask-barcodes`` will suppress any barcodes detected in a page - image. Barcodes are known to confuse Tesseract OCR and interfere with - the recognition of text on the same baseline as a barcode. The output - file will contain the unaltered image of the barcode. .. note:: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index db7e7464..e365fd0f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -531,16 +531,9 @@ def create_ocr_image(image, page_context): draw.rectangle(pixcoords, fill=white) # draw.rectangle(pixcoords, outline=pink) - if options.mask_barcodes or options.threshold: + if options.threshold: pix = leptonica.Pix.frompil(im) - if options.threshold: - pix = pix.masked_threshold_on_background_norm() - if options.mask_barcodes: - barcodes = pix.locate_barcodes() - for barcode in barcodes: - decoded, rect = barcode - page_context.log.debug('masking barcode %s %r', decoded, rect) - draw.rectangle(rect, fill=white) + pix = pix.masked_threshold_on_background_norm() im = pix.topil() if options.filter_ocr_image: diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 94c3c6d3..f68ba611 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -180,7 +180,6 @@ def run( # pylint: disable=unused-argument unpaper_args=None, oversample=None, remove_vectors=None, - mask_barcodes=None, threshold=None, force_ocr=None, skip_text=None, diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 6fe45275..d3011995 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -267,13 +267,6 @@ preprocessing.add_argument( help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they " "will not be included in OCR. This can eliminate false characters.", ) -preprocessing.add_argument( - '--mask-barcodes', - action='store_true', - help="EXPERIMENTAL. Mask out any barcodes that appear in the PDF so they are not " - "considered during OCR. Barcodes can introduce false characters into " - "OCR.", -) preprocessing.add_argument( '--threshold', action='store_true', From 9873d51f58ef007e0a608408e777ac59e4dbabbf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 23 Jun 2019 16:54:53 -0700 Subject: [PATCH 102/880] release notes: add next --- docs/release_notes.rst | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c8fca621..2b28d529 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,36 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +next +==== + +**Breaking changes** + +- The ``--mask-barcodes`` experimental feature has been dropped due to poor + reliability and occasional crashes, both due to the underlying library that + implements this feature (Leptonica). + +**Major changes** + +- Added a high level API for applications that want to integrate OCRmyPDF. + Special thanks to Martin Wind (@mawi1988) whose made significant contributions + to this effort. +- Added a simple plugin interface that makes certain steps of the pipeline + configurable. +- Added progress bars for long-running steps. As such, the behavior of output + messages is different. +- Dropped the ``ocrmypdf-polyglot`` and ``ocrmypdf-webservice`` images. +- Removed dependency on ``ruffus``, and with that, the non-reentrancy + restrictions that previous made an API impossible. +- Internal code reorganization. + +**Minor changes** + +- Test suite now spawns processes less frequently, allowing more accurate + measurement of code coverage. +- Updated Docker images to use newer versions. + + v8.3.0 ====== From f855bdd36b260d1206be5c170582dd3ea19e4a9f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 24 Jun 2019 01:32:42 -0700 Subject: [PATCH 103/880] Docker: Ubuntu image should be manylinux1 compatible --- .docker/Dockerfile | 1 + 1 file changed, 1 insertion(+) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index c6bd51a6..8661d9c7 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -49,6 +49,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ghostscript \ img2pdf \ liblept5 \ + libsm6 libxext6 libxrender-dev \ zlib1g \ pngquant \ python3 \ From 187283192bc212e9d62157c05134533eae5ed98f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 30 Jun 2019 15:08:10 -0700 Subject: [PATCH 104/880] Fix reporting output file size skipped Due to change to using finally for clean up --- src/ocrmypdf/_sync.py | 46 +++++++++++++++++++++---------------------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index daf332b1..b2c9461e 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -337,6 +337,29 @@ def run_pipeline(options, api=False): # Execute the pipeline exec_concurrent(context) + + if options.output_file == '-': + log.info("Output sent to stdout") + elif os.path.samefile(options.output_file, os.devnull): + pass # Say nothing when sending to dev null + else: + if options.output_type.startswith('pdfa'): + pdfa_info = file_claims_pdfa(options.output_file) + if pdfa_info['pass']: + log.info( + "Output file is a %s (as expected)", pdfa_info['conformance'] + ) + else: + log.warning( + "Output file is okay but is not PDF/A (seems to be %s)", + pdfa_info['conformance'], + ) + return ExitCode.pdfa_conversion_failed + if not qpdf.check(options.output_file, log): + log.warning('Output file: The generated PDF is INVALID') + return ExitCode.invalid_output_pdf + report_output_file_size(options, start_input_file, options.output_file) + except (KeyboardInterrupt if not api else NeverRaise) as e: if options.verbose >= 1: log.exception("KeyboardInterrupt") @@ -355,27 +378,4 @@ def run_pipeline(options, api=False): finally: cleanup_working_files(work_folder, options) - if options.output_file == '-': - log.info("Output sent to stdout") - elif os.path.samefile(options.output_file, os.devnull): - pass # Say nothing when sending to dev null - else: - if options.output_type.startswith('pdfa'): - pdfa_info = file_claims_pdfa(options.output_file) - if pdfa_info['pass']: - msg = f"Output file is a {pdfa_info['conformance']} (as expected)" - log.info(msg) - else: - msg = ( - f"Output file is okay but is not PDF/A " - f"(seems to be {pdfa_info['conformance']})" - ) - log.warning(msg) - return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file, log): - log.warning('Output file: The generated PDF is INVALID') - return ExitCode.invalid_output_pdf - - report_output_file_size(options, start_input_file, options.output_file) - return ExitCode.ok From 340e2bbac6b43b9b9aad269e69dbc672031ccd3f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jul 2019 13:10:05 -0700 Subject: [PATCH 105/880] Drop --mask-barcodes from completions --- misc/completion/ocrmypdf.bash | 2 +- misc/completion/ocrmypdf.fish | 1 - 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 50d4a25e..010fcf7f 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -69,7 +69,7 @@ _ocrmypdf() --sidecar --version --jobs --quiet --verbose --title --author --subject --keywords --rotate-pages --remove-background --deskew --clean --clean-final --unpaper-args --oversample --remove-vectors - --mask-barcodes --threshold --force-ocr --skip-text --redo-ocr + --threshold --force-ocr --skip-text --redo-ocr --skip-big --jpeg-quality --png-quality --jbig2-lossy --max-image-mpixels --tesseract-config --tesseract-pagesegmode --help --tesseract-oem --pdf-renderer --tesseract-timeout diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 4c3c5d01..6517a87a 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -9,7 +9,6 @@ complete -c ocrmypdf -s d -l deskew -d "fix small horizontal alignment skew" complete -c ocrmypdf -s c -l clean -d "clean document images before OCR" complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep result" complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR" -complete -c ocrmypdf -l mask-barcodes -d "mask barcodes from OCR" complete -c ocrmypdf -l threshold -d "threshold images before OCR" complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text" From 4dab299619fd627e9bbc9ab9907d18581e4c2d69 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jul 2019 13:27:07 -0700 Subject: [PATCH 106/880] Fix parameterization of --verbose --- misc/completion/ocrmypdf.bash | 2 +- misc/completion/ocrmypdf.fish | 8 +++++++- src/ocrmypdf/cli.py | 3 ++- 3 files changed, 10 insertions(+), 3 deletions(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 010fcf7f..d3bc6b80 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -49,7 +49,7 @@ _ocrmypdf() return ;; -v|--verbose) - COMPREPLY=( $( compgen -W '{1..9}' -- "$cur" ) ) # max level ? + COMPREPLY=( $( compgen -W '{0..2}' -- "$cur" ) ) # max level ? return ;; --tesseract-pagesegmode) diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 6517a87a..24883be1 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -46,8 +46,14 @@ function __fish_ocrmypdf_optimize end complete -c ocrmypdf -x -s O -l optimize -a '(__fish_ocrmypdf_optimize)' -d "select optimization level" +function __fish_ocrmypdf_verbose + echo -e "0\t"(_ "standard output messages") + echo -e "1\t"(_ "troubleshooting output messages") + echo -e "2\t"(_ "debugging output messages") +end +complete -c ocrmypdf -x -s v -l verbose -a '(__fish_ocrmypdf_verbose)' -d "set verbosity level" + complete -c ocrmypdf -x -s j -l jobs -d "how many worker processes to use" -complete -c ocrmypdf -x -s v -a '(seq 1 9)' complete -c ocrmypdf -x -l title -d "set metadata" complete -c ocrmypdf -x -l author -d "set metadata" complete -c ocrmypdf -x -l subject -d "set metadata" diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index d3011995..5e639155 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -186,8 +186,9 @@ jobcontrol.add_argument( jobcontrol.add_argument( '-v', '--verbose', - type=int, + type=numeric(int, 0, 2), default=0, + const=1, nargs='?', help="Print more verbose messages for each additional verbose level. Use " "`-v 1` typically for much more detailed logging. Higher numbers " From eeae6f8292eed00047b858211a4fe5b9a1864b61 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jul 2019 13:49:17 -0700 Subject: [PATCH 107/880] test: Add syntax checks for shell completions --- misc/completion/ocrmypdf.bash | 4 +++ tests/test_completion.py | 48 +++++++++++++++++++++++++++++++++++ 2 files changed, 52 insertions(+) create mode 100644 tests/test_completion.py diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index d3bc6b80..1b550da9 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -1,5 +1,7 @@ # ocrmypdf completion -*- shell-script -*- +set -o errexit + _ocrmypdf() { local cur prev cword words split @@ -84,4 +86,6 @@ _ocrmypdf() } && complete -F _ocrmypdf ocrmypdf +set +o errexit + # ex: filetype=sh diff --git a/tests/test_completion.py b/tests/test_completion.py new file mode 100644 index 00000000..8fdbb59a --- /dev/null +++ b/tests/test_completion.py @@ -0,0 +1,48 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from subprocess import run, PIPE + +import pytest + + +def test_fish(): + try: + proc = run( + ['fish', '-n', 'misc/completion/ocrmypdf.fish'], + check=True, + encoding='utf-8', + stdout=PIPE, + stderr=PIPE, + ) + assert proc.stderr == '', proc.stderr + except FileNotFoundError: + pytest.xfail('fish is not installed') + + +def test_bash(): + try: + proc = run( + ['bash', '-n', 'misc/completion/ocrmypdf.bash'], + check=True, + encoding='utf-8', + stdout=PIPE, + stderr=PIPE, + ) + assert proc.stderr == '', proc.stderr + except FileNotFoundError: + pytest.xfail('bash is not installed') From a86cb8148a0ba2eec5513914d44e3585119617e6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jul 2019 13:50:29 -0700 Subject: [PATCH 108/880] Fix jbig2 not checked for special colorspaces --- src/ocrmypdf/pdfinfo/info.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 8e29c44d..c89a47e4 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -307,7 +307,7 @@ class ImageInfo: # but encoding must be monochrome. This happens if a monochrome image # has an ICC profile attached. Better solution would be to examine # the ICC profile. - if self._comp == '?' and self._enc in (Encoding.ccitt, 'jbig2'): + if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2): self._comp = FRIENDLY_COMP[Colorspace.gray] @property From 1cc4c45b7ed1124fe4db9bb54983b0e439e1dd38 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jul 2019 00:48:34 -0700 Subject: [PATCH 109/880] docs: mention WSL works [ci skip] --- docs/installation.rst | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 73380839..38e1fb35 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -384,9 +384,19 @@ See `OCRmyPDF Docker Image `_ for more information. Installing on Windows --------------------- -Direct installation on Windows is not possible. `Install the Docker `_ container as described above. Ensure that your command prompt can run the docker "hello world" container. +Direct installation on Windows is not possible, because there are a +POSIX dependencies. Your options are: -It would probably not be too difficult to port on Windows. The main reason this has been avoided is the difficulty of packaging and installing the various non-Python dependencies: Tesseract, QPDF, Ghostscript, Leptonica. Pull requests to add or improve Windows support would be quite welcome. +* Install Ubuntu 18.04 in Windows 10 Subsystem for Linux, then follow + the Ubuntu 18.04 procedure. +* `Install the Docker `__ container. Ensure that + your command prompt can run the docker "hello world" container. + +It would probably not be too difficult to port on Windows. The main +reason this has been avoided is the difficulty of packaging and +installing the various non-Python dependencies: Tesseract, QPDF, +Ghostscript, Leptonica. Pull requests to add or improve Windows support +would be quite welcome. Installing with Python pip -------------------------- From 3ee306184b4c18d404fdcdb1f7de02007b069b7f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jul 2019 01:57:58 -0700 Subject: [PATCH 110/880] Don't overwrite input PDF when fixing NULs in metadata --- src/ocrmypdf/_pipeline.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e365fd0f..f5cc79c6 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -695,6 +695,7 @@ def generate_postscript_stub(context): def convert_to_pdfa(input_pdf, input_ps_stub, context): options = context.options input_pdfinfo = context.pdfinfo + fix_docinfo_file = context.get_path('fix_docinfo.pdf') output_file = context.get_path('pdfa.pdf') # If the DocumentInfo record contains NUL characters, Ghostscript will @@ -702,19 +703,21 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): # NULs in DocumentInfo seem to be common since older Acrobats included them. # pikepdf can deal with this, but we make the world a better place by # stamping them out as soon as possible. + modified = False with pikepdf.open(input_pdf) as pdf_file: if pdf_file.docinfo: - modified = False for k, v in pdf_file.docinfo.items(): if b'\x00' in bytes(v): pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') modified = True - if modified: - pdf_file.save(input_pdf) + if modified: + pdf_file.save(fix_docinfo_file) + else: + os.symlink(input_pdf, fix_docinfo_file) ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, - pdf_pages=[input_pdf, input_ps_stub], + pdf_pages=[fix_docinfo_file, input_ps_stub], output_file=output_file, compression=options.pdfa_image_compression, log=context.log, From 2cff6ad2d1d6b884a0f6656a6bf98650dbe79276 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jul 2019 02:22:50 -0700 Subject: [PATCH 111/880] Fixed blank pages produced when NULs removed from metadata --- .gitignore | 4 +++- docs/release_notes.rst | 5 +++++ src/ocrmypdf/_pipeline.py | 8 +++++--- 3 files changed, 13 insertions(+), 4 deletions(-) diff --git a/.gitignore b/.gitignore index b2881fb4..46ed9aae 100644 --- a/.gitignore +++ b/.gitignore @@ -3,9 +3,10 @@ .pylintrc .pytest_cache/ .ruffus_history.sqlite -.venv/ +.venv*/ *.pyc *.sublime-* +*.DS_Store # Package building .eggs/ @@ -42,3 +43,4 @@ tests/resources/private/ tmp/ /debug_tests.py *.traineddata +/private diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 6af0f263..67d4e5e5 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,11 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and ar find: [^`]\#([0-9]{1,3})[^0-9] replace: `#$1 `_ +v8.3.1 +------ + +- Fixed an issue where PDFs with malformed metadata would be rendered as blank pages. `#398 `_. + v8.3.0 ------ diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index ab11da6e..08da6ac6 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -816,15 +816,17 @@ def convert_to_pdfa(input_files_groups, output_file, log, context): # NULs in DocumentInfo seem to be common since older Acrobats included them. # pikepdf can deal with this, but we make the world a better place by # stamping them out as soon as possible. + modified = False with pikepdf.open(layers_file) as pdf_layers_file: if pdf_layers_file.docinfo: - modified = False for k, v in pdf_layers_file.docinfo.items(): if b'\x00' in bytes(v): pdf_layers_file.docinfo[k] = bytes(v).replace(b'\x00', b'') modified = True - if modified: - pdf_layers_file.save(layers_file) + if modified: + pdf_layers_file.save(layers_file + '_') + if modified: + os.replace(layers_file + '_', layers_file) ps = next((ii for ii in input_files if ii.endswith('.ps')), None) ghostscript.generate_pdfa( From fd810239b5471e36d937e99f0e6757f0fd866533 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 7 Jul 2019 01:07:48 -0700 Subject: [PATCH 112/880] Add a logo --- README.md | 3 +-- docs/images/logo.svg | 1 + docs/index.rst | 5 ----- misc/media/logo.afdesign | Bin 0 -> 23807 bytes 4 files changed, 2 insertions(+), 7 deletions(-) create mode 100644 docs/images/logo.svg create mode 100644 misc/media/logo.afdesign diff --git a/README.md b/README.md index 3509fb71..fed27f8e 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,4 @@ -OCRmyPDF -======== +OCRmyPDF [![Travis build status][travis]](https://travis-ci.org/jbarlow83/OCRmyPDF) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] diff --git a/docs/images/logo.svg b/docs/images/logo.svg new file mode 100644 index 00000000..fb5f6c2e --- /dev/null +++ b/docs/images/logo.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/docs/index.rst b/docs/index.rst index f3a28781..20b4ad79 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -1,8 +1,3 @@ -.. ocrmypdf documentation master file, created by - sphinx-quickstart on Sun Sep 4 14:29:43 2016. - You can adapt this file completely to your liking, but it should at least - contain the root `toctree` directive. - OCRmyPDF documentation ====================== diff --git a/misc/media/logo.afdesign b/misc/media/logo.afdesign new file mode 100644 index 0000000000000000000000000000000000000000..3d96656cb4f7333ed94fa93a5dd17a83dad1a3f0 GIT binary patch literal 23807 zcmZSh@9oIVz`>ALToj<}nV0rMl>rP2)W8(O3Me1KV_=XdSBM3%#TXbE{FoUS0*dlW z6d4#8l-)9OG`8yhW$^f`!}v>?FE3ocW7RS?2Bri5yTlCkd;M`=+jze^`DOgpQyeO* zTp85EbXhI(xr%bre3$IZF^giJou}k(;J7$>%Cb-0zJE0Xz2l{vCnxLO@3_QU(C9N| z!)E1Wf`Y2|nU}K(vAQddEYPH;*+=jPoH(_P{_Xhsd}FO{;&K$?c%vF z|1X>7t*og(on-W{KKTEV{->9hwmlV9JLDp`^1u9_b9?4Jid^!)J;lQAfb9~=ZLjOa z6*B)kTVf%`-{I7@x=Wtbq2KA=+Xo&$`OmEGeW<+efAPdZpC$1hR|)^$zUP)S|CM}k z?{ojY=PYksxhGa<=CEX~H)PeEsEG@EA}1|esWaDUW=&M<+OSowPHTM(6Zlsu za>{T&*jSMkBysb9E6;+=6-Ld%ZY2*>7=t5REtFZM5)#`aLkl8K7#@Dxsbpv1x`2sE zY>$FE`}B8R-VAQc3<+F%3l4mgDo;4{szdI}2f=xcJr)iI&dOH{y%pjO?Vg$ z&aVlY>7^0UmBMv?PUyqRY0Ec+2F-k^wcaZi@L= zckAmv@>%U#z|_O}CEqRK=qoRw)c=2LYo0!6T5+k__?Y}Oj>z>NCLZ$n$j|WpvnHH*`Vt_HOlt-Yk>lX^u{#8pTu;HcL{_pp1e$NR}9~N?EeW~7(sr;3J z*U9nTWBYzhsXzC1-mf{P^|E)Cf6?ss!aKgy`aeuw=(c(H*ZV7e+Wq`5`s+!i>m<#I zQOjnY|FA@9ty7@R+(T1OE#dN>bTeqH*Q(QZB*gB=*DX9@r(m3JruaYh#IM|%*M?B|1YM;`F}jKVeS9l7B}aazj1xl@Sp$mN-is% z^%*ZW%(->oaM8}D`k(I(JUa3CfBkHkxfWmk*@?+mZIW7TR#SiY(vI){ZKbA#&-n3| z_w)we)_?amh_8A3v;M~En#g~fE9#~(u84T(n`3QOUw?jy!KeR~V2gj(Ka2QQKi|^q z?|+eBXI^X8)z6pQ_Sc^4@BY)DQ>y>%w>djK^-N*-)katCHPik(hZZp?9^%VV;$S>| zoWm&Y&I|*?Z5LQ2wwkVZPc3 zfARG~f9u^>J-+lJA+k8+OZ{oi`2OW`?`J80=vgbUM$JZHYxC>7g}>`No^BO8&w9VO zeA|EB&eCi@DG0&;R=1{G9>MA4v0<^``H8 z@PFPxEAwZDrhP}Zmu!@HGwoe#Q~gXet-Alwo78-xHbqVTe|OSS|Duv^_2Q%N-;15U z^4EUF|L-fiAI^LKURrqi=l}Y($&ZST++O{z;L-Fy{D*@cpOUaTp(vWTcHyxeLECLJ z?mmdSJK-byy#N1C7{A)Fbb^Xa{i@`1X+Qogw|e)>JSV-FY2Fdf#SQ<1WM|#{zq@4B z<8_bz&$gbc^UQSC-M9azzuIA<8~^{k)MZzgwBTsUqS#ZLQdzGrfru)5@cNzrRwO|Ns1@|F^Fs=G*?yKUB0u zVs?>wOWNdPhAY?Z^(@?1pmTJy)as<;QHFds)<*X*|NB2P`(agt&410}TmzkX@y}M7 z%~<#JflS4$S;<|Mt+~ouL_U4dt#tCMJ}*^qt9p^iAN#kew`csf%~*f4Y2Sa7-M4@J z-w?}p_WwC+>9haq=az4ie0lA`zx%$y(eWR(Cx!^kbqUX!Ah7y}&k~JE zAxmnU79I*(wS~)j`87?Kl_9GyxXf~yq|xQGCthB?QfWlP6XAV!g6w_4_n>HTfu=Bga z-Q(!7bEBZ@f^Ze1rCO;0>YYKY>XUU2=bT!;L1$}_;wzW6Vau*|x`2wFn{y-9Z9KKh zYGK{t4Ke5byCquAI3?Zs^hDd!q#a+b#QJ_*!gpNcxV%ldcvA1iTTVLjW%$IEN-cI? zxLDpUTpx3%&Fc9Bahdu*^7T)y_SV)#=q>-cKXC4xPaWa6m;KLo`ENhr{GE;KZ+Jz` z7M}3*^KQjyQpP%s(+-QBFl*=iYat`^>wo{pxt80m)-Shu_4dEu1^ZuKCGMlb&Rafk2)g{Q zf77ga@Zr~wJO$f6xofSA(!OjHxYSE@k;gMFpSi1oJFL8o&a9fExmas$gT{gHt}j3O zZ3?xT=@ry@YSjjnwP6c8mHZRAx1Lhu44o9F5o)=ljsMi-x&P`V?Ii4iR^_Y>STr-$ zXSKmprBKDJR;yMePgUW0McR|qC!LXQ|N5C(*~}PR?3L6M?muwrWD@6(xAHNYRHmk^ zZVj5KYorsnEYxFN(DDGa$x8}KwWdZHEfqP_#q%$Ab+(w@wJll`mvv;NUh-74300Te z=JQ-rXYTS)4dqE1mo$@q*{Rz-DTyd5c%54^xnaQ$B`$@<(#nVVD=r*jc9T`e&`{z~ z+_88q>jK{s9FcrxGF#h&+NLO-yyT@bH7!FaICaIT)xi{os?;`OxnSL zpIugGzG1t7MB?S5<`$hNtOC(YC0pDZQnX?&NZl3B+7Q&#rsEdlFWQuL%=}i02D8Y= z11yYo)27w`kBs@Z)>^wa<8EYTNVHzu{0*fh7wJlyV&jl?jT^6)z!`rJUH0Q^nA5;>zDdl-~XTZ{=e=0{)zARzxo}czu?4%2I+Gz8kgziTZYEw zhn(}d`G4ACKQ*CA?|&Pve%d2&*_o5$#eeU#Wo!N~{u}T1_-M-izWKlVS4w$+ z%f+8tT_mT5ss)|&Il08oNN4jZ9`8vNT%oJNv}`(4yO*3=F@L_6dHI{wzy2E~KaM?< z;rwdVwtzp!7GB?RPx{B-*j@K-|DUZ}oE(v|H_>ROk6Q4{pwflqC%0|ZyE!vzRUu=W zlJ<YiT6InckHMMOdPahvqd zV_Pr$pPcV=tS7Ti)TL|21Bu+qE|!9Z4}Hn6xMiJgEI4xGRpVpDPW6O>eY=8AeEoZzr7Yp1_KPMM z9{x1bDD~77O~0O}J{&U^FqZvXAA zZ}0fBy+;1@_pN_g-q&+17hmtZCvfXs(^cDcU#b_MeaDymOZb{i5AOTiSYv5;;^gWy zL9XSiUiZdLy?Zh;ec|Nrv$=WC+rv^rzt{_uupeLieP{OLbLrC4|E)Q;>c?N(!&Zv_ z<*(YmW_d6{)oA+9|E=Ac!OMJRK4Qr^dX&-T#l$0@YIQ93y^t-@OU&d6|JAZI^3B5L za~n*$@87pvzvKUT$LBSh9_7Ep?CcmB2KtJ-$^ z=%fGr%brb*+_yjQmd?Nb(j|70VLTg^x}R@%8a4i`mLK~e(ZnRymD{&pa0=&nydHizdSYe|9{rpW!l>R(tk#6kJS0M z=>ObFJ7!}Du9^NlKWW)(Du(R!$!PNi% zf4+J&Zd3Ddv)iv|FdA;(;|H;pFT9r7pIi@Ihcv4-u~*6%PTG zkLUQ@O`CS_T~s!``|h3W=Ic5_olc%j91MBR^Q5-hS{%81Z$pSykdkQR+OVXFbIRS1 zu(WcBXasbL7(P0-(N92S$L;E8GX?LwwN(OZoY1kLV}Z---NvpNZ`3%1)HkkvEcvjr zNhI)bXgW(??;4BUv!z%#&vV?_WHryDH+(KgT9#8ak#*GHB72k{L z=QTb%_q!=isMASj{-hHvDnhMw!JaBy8*A#6yYf7g0$1GkF1vF3zuR^3WuM%xn=cEz z(W}|JS!?+kHLkN|I?__vHXCAR3rl6&7?_@Gjj)UC4b}V~c}AAI`QPRf=1SsT@RLkm0p*Way3;$NG%UX1S>kO@vaFq5^IN;5U`%V4vw1PkTFH6e&x96%_c4*~~E47&qBxWZR%zQM-o&8D!gKW(M zMUH&UC#xs*cv|@umCjo=VTw}tLd{iv3w>5iaTc1I>SuX%k)X04L&fKx_t#0sOg2i> z^awb@tGD3Df4fFbuM5shGF#h2^#Zkm)F$qn!s-3oOLUsoBh8$(ejz)QCWbA!yhN&z zBf?I5#TO4jrr9htD}$H$Ok6&>vTI2|s?pSiGtUNVC>Jj~)FTwAQZ&^u;Hdjik)VPH z%*%sRCo8QyBepWDbw#k6bMaQI)>Xl05~mpkt4*D;_=5xcf(H$qC03s5n?k!DY3WQ{ z7OJt%L)TcOxYJ~%-xA&o1!mn~H$Vov)%PyGAP3b;5u zv#s;B{?*@Fm3KbTdZzY2lO<<)_i!%Py|KRR8CzZbwP_T@U(Npt5|Py4E8saL zVBqFoq_O<$-7BTf?9cGo_74JVMd%Dl<|NKKUlk*imsXUe{dVc2r>}&teTNS_Iof3c9 zcXm&5VBOt6`_uo%cg#0Fe$QcjVZ82-683fx`v^X@-|tmx)&&#>p2w&rxX zy6FG-pP#aBU)-`_qn+C;t9t+3cJ;|0%Pxnbo)B zLaASmD`p6Za8|MGWi)MC_e*Dfd(pr9t5#2Vr1)7RKx586*%dD@H%|NYf4^|R#{c(K zTP7u3P@2oE%H(v|F3b7<|BHfB%}p-^wy=eG9N$xMCFM)o(vP<{@Z9#wO};K(bA8M9 zoB!vpmI^+pE&cqzZNopF*DY`MXJ$w+eo58y{C3~b>Nt{r7+KmJ6EPi%tH&7cF9V`QQG<@$UYA=g*(-_wE-Hji{*6 zGImVW&Gf93B}3orej+HfhY@V~#);`P~GvBt613A1xvHN|?ziv6iH zjXeE-b4skqg>2Q=OuOA4&wY@+?mJuA+LZi+>q)P;_dUCHvN2XUQq1Pe*2&ECcio%) zW7d|^#@I#6`2O5qZdLFRrT# zUTA;$8yjqV`+xi5?5s5HfAc>*-K;8{z4-t1nffcYFif(`QUqLE#d*8nOnHJA} z{?GbSFEKf_@Z698mouc!trWIi>KOmvck4dw6Q4G&%z5$OTQttmBw+rF-AyJ2l{fF@ z&1m>%esY`nf^skCb5=_o)0h6&d*Qjw@PGLilT4lDzHq;nO!1Z`(givUZ|BSmvRid) ztMF|PN$v~J`nFE|Wab}|JGpGp(#t;Ln*ZD%&tCYXl4)VO#!Z($|C{rlDb4PP=NIWc zr%~B;|NGWGJLarfXTIoYgz5K5{~bNr{v6m*cj2b%yhRhVr3`jC=3ZmAU2N=SCfNDG zvq5rhY>M#5D@O~?uUx|Zy{7*8=P6S!{4n< z|0mX{zde&AF*h^)or!PkhRY_sOv^UjcqbP8?a!Y-e=d}*=c)3#6u02!zRWY}c`;2o z@}|ez(!#!qy)vnr+u)ax`QpELu~YX^8#e2Ooo62{d>SLXZL#;&inMg!nI&1l`TPE- z%sxDC>Ec}m+D0=1@89-R;gEUk@1VUv_$W_=y;($~pv|2fhH+(!8aFmdMEGtv!~_z1 zapP0ap`*;3uIRI{97$+WbxfEm*s6Mr&$(~HQ6Z}h>slu4klG=XbkU((2+Um6ytPpy zWyh)?j4ZDi0#6i#eq`=YjM$iTv`bVwI3nXgfrSj;aiha2puX`niTFDQKrHoZoXdP> z8l`p#8_!ADwvl7XuAJNd55DO4DGt2WzvbqwOX-HoIaw!KJ?0R)ztG|4y(1SAcYb57 z*Zgn4xz}~U`}bE;Z~uSqxLogc*Vc&aV}I)3T@8^CJ3qfVm-m_e{X4HG-*wcqRn3ih zkb5mm>t0%K^n>6{JyUaX#3vtV)%07k?$_V>Q{SKaKkw|R@Bf)wey+GId`#k0l1@jD zVTVA9TSJj?v1EeCu||cCPdZ1x2{W)QX4qvT($?IPsNkH$(;6Vq%)r3lz|vsgEKqcA z;bx{|4$oWvzU)!3n8dsyVNXY@R~xT^5R-VhW7>r|S*yXOUDe@s@$kKbwNEV}tASaF@-rsJW~vlpjyvn@99*;6#*wd&5jWjZ&Rth&Ub zZ`k^9@8pd3n{e7@`E-`9KkFVmFuhS?E6lQZ?Y+wiwOL#p;q7e<3QGmA{!0qpw%^#K zMOh$(-&V-|m4MZz_7{S(MaP|gwK&dWJ@DV#`D!4Slgy>FQZ+n*RS#ayykh-=VWIL~ zzTAo0_hl-c-~5$s8J}(VVx`FJ2U)?IE4e2ne{p9iE1X>Jxn}G11^*O%+pZ?8F-!J8 zz3jl8^)KAR=G@QOmVQa=LQ!?^7l}#DWup7Iwc6}ve!G7`dAcCK=cMU@i8=B0OI9AX z-+L?JdBb|WRW-*HWCa}$85UhxvS6ZVl<3WhxB6a>4hpn7EM#hCZ%~t4dqMEWbk*lp zg1Vbabw6|Lc+|3E#Y@MvijA8*j{ZAyt>!+PSNfCc4<8~Xw(0bB>B_LZvK0EdR9Mz; z!K~lwZU@`*+cVGGDAfC}*zaIDd-=hdiBHa3y$$}Y_UiZ#+m#MWk}k;H+_om%M_u-c zu6M-cG_l_03CkY&muz0KCO;%Ar*2h0!}KYZmlq30|CIHU`wPJN9tdM{q$)5y4ZS^5P*r8D=Q+0`}iom#w#=bq1< z9Y2bU8)A&l+)~_J9LpZ|Skpp+)6KS)TR+M9uHaRD)0AF$!F^rcXJ4oOy|(xNj6;_t z*4dQl{Fz>O|EF*wTmO^!2ZXJARt2>ce-7U0`LRR(z_~^CmyWJD`Y<8tf?0><-pM9= z=k?y|Y;ic~rfZ{d;7sPWIimNrFB3@hTh?K5&X4PR^E_MQuJWfwU!R(}C@#2u&go(9 znVU{Ryuu9fdCT0i-{eo&s4)3`)^;zY4WdHcPu4}u`~8CD0()xg^$YjQR-UY}pDwsP zkI^FW)}G1CL7$q>@+^xoQTchGqFu3CnMcne!Q=m{d*#I_+)~-zUMX6o`x(RR8RA0LV#jv7Tu7SPa;1~YweRs_=G+L8 zv%Jyl)?p6v`F{_*<-EA>&`~BHJ)tS_>U&>$M{yMg7v)sxHy<=uy23`^LHel~`)T!O zxldWU4)agyJa_O_Rfl%}>BiiKm{0wdI~x*r_LK--(RZ71ii4#lF?{;v*$W&TXZ?KF zQ5&gVEPSJu>3fO2?+l~k5$-X<460jM^?Z)1ga$DzSbt*DnXh(IMmb$a^bgi>1-U$} zIdS5&%sak6O3SxT`fYXL*aLBXhW#2pr?w>|`EHum$n@l==+56lC%sHuY>bl9=2)*u zbJQujs=tEk&%wEdGmcDc5uY2w+Ay)Z^ii;JopvTp+)B&)?Nh6h7x)W0l-u^d+-a_3+Oo-A%2cbB z$9rq~%Z2feL#A{@vb68TU8=fr2W3H?-Qdd_ty@A*gHm=is7E_}RUyz}59s}~m!<(y2sQj@l2 zM-p@NYIlb<-ETj7JYM-cM?7bDSLu~S+GmQ^@|#_rrm}&H*E;8y?7C0fvUWevGNX{?t1<&G+-}QN30oz3{E=szRCda-z}ZS2-8l`{Ppi(~ zaiGAuGVW5%lSLMNCuNjHimql&^+;&b=U$O&k$Z1S-5bZIG`{QYI`aEV($c-u>b12w z3Y5|(=Ivv96ET-%vfjFrlG!XQof-FDCS*FkohsVzPL=20(~vTnz!Zh_uiSA%u8yUp#NB`<|M0$ z;*}uIZFLVnel62^IQ{9A18NaZCayR7ouI>Ndp)n(to+!T5{n$h^R-0<>bp}v$!5p} zNiAXf?8_l`z@2SLYl5Ho{lq5|6c~7z`(C*&aNhgryuZfGyv6c7Jz{AO`45@>_?{GO z%Wz#oXY+(!k>t0B3qRcA=TSM@{${D-*S@^Jvdk^XRy){(i%}Pg0?t znX_4b-rcPzZp(S}$^rcsI+qVy7bsr87_}{h_r^3QozkO{pB7CNe6a80`FMBbNga*N zcdUY}=bL>Ao364&dPZ1A)!7|-^7}T3|4^3-IHJ)Y{d-HngHy7lJ~ymI)TG6Og*;@} z6-^6lxAUGV5UQOr=jY+BiB?nAOrF3XWVxr~*OLvQGv~dUtJV-6+_$s$eCM)jnpOG# zCOUmw)@j7(^&?^i>ja)|lMP=S4ldpG(@e#6-xSqKk^OtvyMOUzcxfGy`)|bUB9gSu zm)Sbwa_`3ctjHZtT6vD$opSMq@=@touLH~+W+oL)-P5vX%EQZUf9syx#)o!@t}D7* zaB`Z0q+3ms?=GdTTJ^4MgRCm=Ptpf>2hIEurK!6 zu@9O5Vr*YG@8XIaH69YFaM3uhha~}c1j#N)LV4o z$1KyF$NH;hy;Vus?%w{;sK?(%FLAb`-pyIhB6n@tYq8USHLmROF1H<9ZhO?FW!;^! zY?IU0t)G>|Z4$G}jy1^cZmgKmQ);bX!Z^>DY15M(v9-&~lveHk_-*#1FqgN6cilcM zWc>gC|9@q-5M$6fE)a&T>B6<9#=sy8(aun8#=yYX8Q|y6CB?uj^!CKshfr-d6zCi^V9m=O>Z#I!Spue!S0=J%V;Gb{>~ z-n@CEbJxU)Vy`G4600LejvTmm@7|@e$}JPh+!+}%r&*W3Q!qCdKX3b8=2)+E`paiC&YV2S z`TP6(_T9U!pPilE{`K|s_=#ZP(@!W{>BKF27uG`Q?L0 zkCI+4+cIPJY-ur_h=zT&zmx9nDz&Kmq>`c1BaoDn zadG#hSfKjyO1-Q0`@7aH@O$`gQlwZt=qx zFABcj_ggP0I9QWIp~YsC27|(qX%8MG?D=qrd&B0Ia*AaHQP~WN`41 zWoG9)a;TO2V4+Od+NjnaKPtGGI28M3xEMHAy1Tfr{QLKP|Kd%XgkE1;dw8mLc!ouT zz$cG|3=I?5zP!6D{eI79J}xe<59{mya=(1}a$25{V#^F(R)!W6B?AM2)6@0c6B8Bf zemr2lv%miSC10f$fz1mT8YZ|I91{=}Y;5HgSJKz#f4}dy-d7d@Cppd#-U$=b42~Hj z9pPX&uzY@9m%rUl7VqhLt^olPTBF@Oo*({fV`OAh@b_0~NJvP;E=%1*pKY|XwHF3x z2*}Cx#qF(Xd~9|9vk5Ch%Z!G{7CkbSLf^i9i`baNdTniVyQ=rJ1@8TF)AFPg=bq}5 zwHA<;?w&J8CSz-qxB2p;C04ReiY)Kky=$0r!vGv&Teog4**vS2i6JR8%6Rg8!!|}{ zwuGmrrbg_mvAniEzCZQ!w1bx~U+&RWd2+Gz{XN?^x3+TMym>QXM}Z;}Gqdj9C1NMf zo;~{P?CggxUUV#4q?B`am#EG8$4{rnw}De(x)B#cN9cylo124H9$6c`Js>9L&MN-Q zemUC>IX8`JzTHf}vA5d%+4=eWGkwyIT@(70;pOAw5*H`8%zu90%gf87cUkIAn>v-1 zot^#0ty_=we!q7(I)88L-QDGf4<2OPoPNG7Xyuj?VP7XkhK*eJ?%m_y=RZD8H+ng+35bNUBKx~ry*N>eEgb-jZ8r+MOKHcJ+#odUFWV2 za}=usgCoP0EDll8tN-^4+RmMO(b#I&FZ)lQZ!au-+pIY?p{z`RWudd{ip$w&-`C)ycDO@OS!bR@4-*5^!UYU}%Gid>%|2>}FD!J{y(YJrg`r7d!HN|e zmX?+$&Yy3OTHCgGvGSZba~`~Y&24RMT^z&4upn&p#|H=|@Y(2@TIF~t@hmnynVoQdg-QO?4HGe)HulW6TyJ5ls zhVCuPLm3&)f9~k+cFxJsSr@mr>(nW)i$64zdA4oaR`F)zaRFiBX2)i>tr?p{&wuvt z@;Y?h{(n#L^K%QgY!R_JFTL(yp-hOD=%U4o4GSN+{P_L5dtGoqK*Rfezxg~pJ$d5} z-@GYlXJ_}}+c&p>0D-Enudc2d#&U4u?$57IuDxnB8o3=j9&!`@2l~)?#**^?R7srJrB+eeJ=| z7j)zK9v&(YDb;&HF}hg;d}AAY?afBe9qdgvuDp|A6Os5`S6Sris*>+uQl&>wYM% zi{F3miBI;t=Y?Nig@Wow>+*LwU!KKFC-a;;cdnqM#AIjgS5HsRgYEKl9je~b9vo<7 z*3i|>z4}uw)n{o_$~B!6-`+HSdE9^S&)?{Jzw~sE^m8pa=jA-&Y6LT$?qB}n3D25v z{m*mz1)QP|oIlSmU;9NcCnx8_QStZ(udc3s8x^}HlCdFf{qdc}&o|`VwX&)B;PAz| z(}hW2U!Ov5^tT zIvM-AJA7ERjas`E?w&3@9$n&Y5D8h+uKjFKYsgmEdKwm z@Ed>keT!ayU0GkB|M|JOhfhvc|M=md^WlA!m6aX6y{;7%7Q4#d-g=*9S^34 z#~s{J_}JyOWcIuh%}q@YUcKtNx^>^1%6Z4l@7FZXm?5!v(yAtf2fx0)wy62B;lIZ1+9T!ts_jKmB5cs8lu7dV+(s=q57Illc(sgRS0 zaEO-Zu9BBcmzVo1>+7$-`kl4YrRl?m0-d-$690c3xBu|+J+wYqNXG^CQF$9E# zwe79`-Zk%pW#8dDJBtky54EuMPxermVVuqfDrt11w@GYUeKiYI!gx>D6TQ1ZTT4qo zTwGnMc)Ev*gl(0{GT+(PUa)jt(_>EQanzcsWMw6letzE3gU#%`>$h*(Bs6p8Ob&kj z_1~>*&M#iHh^fc$#*G^iw<1GBTPIBt3JD1TweQQy%Gy*Xc`dzEpywzM5E|P0=~K~) zt65K;Jv(&tD67{}AwE970IjKC*DVlWIVhlEDk71&HEQug?pCLZdrqpEI>w*<|LSW) z9KWJ6^Pl(T*@tiJkKz0AgQ2XOI50-tNzzKWEG;{RLRw+}|(nJzcML^5n^p z=WYs2W^UjX*Hf^zj^1+i@@3)4YQ9DMsL00)YR0rH-3xV``z}}{hcsx zUf<*6{g3bLEcWpCH)k=;%gd86%@SGWJ3B36(~NoZ`WCzQKl=3aw1=msV_+btD7&yW z+MGwuM#A6rtH_@}e;7jiSDZOlo)vew(~U77Up<423?Z)+51 zY}~&6_!P}xkk-w6;t$QW7TC1;KL>*Z^XHF@udnes6&0O6SNTon$KGnimseL?EnqnC z^)>sq8_8EUe_v~utf;u;Fvw9)iY)!+T3vlIXWO$Ar%yAV|E!^-^TvYlz*Oz%ec#-9B+}FAm@I!l%>f81++&*A#I*uc=x*7&-gM>kkCKCf(PZ&$XkxFMi&r zAXojy&}W{_&7AeI8rsHnm%=p|qHl-I4VRbZepmlre2w??SG?P{B}|-np#DGil&M=^ z-DwfHB;~bKDEWBb(GEf7GrPRr%>VyKT`y+G1wHTczI+U?&K+ZKFibu+qeju%+M0!x zwUddV*RA=^otTStxt>nC0VfaXU=5J~4UshK`09JFeWXm*SFT!n_TD|Y*sNMZb62&=l3&F#wni;xP%t*W z{Nl#Cw=?F)uD_mqdt2^~f`?5Pz1lBa2$Yj-|A-eY*<_HE~tAyRokQESEc?+~hw!b`BTwMI|y}i{P-Q9<`W?x^h+Vq&tfphH z6)%@gH^{i4;GrUvVIl?YtL>}Zy`tPvZ@RLHiOA>Y=Ruw4CqH7*em#13cR6q4sj1rG6Q)m3&sZL{)-5P#(!=|GhHUck@+|D^ z=Kpz<*X#<4^WDDBhb=9e``UhcmHNM{Uueo^nYOH1qvp*$yQ%(zgUHfNda(zV1esK% ztrWQw9@qF~YWR`c+S&_iAF_U!Z8mdhY>U99n=?fiuGD(5Cw#h5pIfvu#>CTCfF&d> z%xU>$$$1{HcZc0w+Vrg8`@2|B6V3OX`cyB&loJ9rU#|q8J*W5M+3fs9;cI(}i+0L@ z20$J?dQ|ZKp6%gp6DA0(iP;(SW!|Y2BcJ(pvitx2(%!Ub6VL7U|L4iaYfVl1^5UXJ z`MWuZeU4sUTt>#m7xz||8yo{QMy5nr{z^5HJap(#PnEdOp~45p`dB}IS{29i;u+hy z>*9AVEOeN=pViJ(^v}u3XZO3+zVcgFSg>MWU9qOf)tmpPoN5bK6L9JfHPmESu{Ug= zTP62C7j13rz)0=o*2fDsY~L>alz+wgqJ-q+m${xRLy}HS(d_8#bnKV2omE>?Yy4@d zmnsu8vnyl6>ebnM8K*WkGuPDCx-vSvWyrj-Au(WONLvkSXlSTK^*5b19!aK7m!>sq zbe1e#D*0{u_U*?%JUl%6uN4DBkfw@pn#s=D1>g0)$k(zwe06nN?DBQH9K$0cOP}r& zOW0dwy3b^yfYX`|PsWD!cJ`c{oPf~K)Em=+({r|I+MHj!W{uAlxmdNy2M-^1HraUY z`0?W}ZX8oPu~pUH-d-nS!-A`&tXZoixmK=Sm=hGV@``~=;NnGLC!@s0#e3vzrEYJ_ zHO{dveI)|wdv>}stqxz`#kBRn0f#R+epmNATd}6)-xypXl^tKHLU?#nMPT(QE#WTT(nbm_Bh`Rf@DT)ip^s^_NsnR@!^j*^!` z=jPdF-+2D)?Cc$tpVdI+>E_t&%W5{RoOP@G*_oM(7#eb(77WT}&I@3b)$&)Md9agZN4-Z$%(dZ7d7Q3a; z(!nclC*$1C$9lYQt>2w%k{nGJQXX1+8&ygsBqS`*U=Uzo{P^*sgjLB3&*R}?VO_g6 zHQHA(WUba`*irFODedg6)NA$<;o;$D=G(`c>;Vn?+}T%q`@-3CpKbpAeP6%0{QbSO zx2y*0_V)HIt*sBU*6Mw~aU;UQ%IeX#x3}AJ`}+DMOtV5x>o2+Pygy&Q<;s;;k^gq? z6v(^ZHemvTqVeY3yfyRYEt!$$bEB1p|X;)vTxO@Gbc_MXvybfXJ?lj z{{H1li`P=2)UaK3f2}rc+SGGx$L#3sd3OyeKc(c{*>TYzWa6w@U3;s)CmrjNd>gfV zr=^EDM$Z}X*xAGYzgzxnZT*>v6L2`Brn9B$_~pT#kK!i0v? z)Ah|O*C=p+s!%R2E|=Sf46l(Vv|+&nDinL;0a3-AW~O;t7t z<>?T(RCnt1>gj)kl%0x<0l2GVe&#S~1X|gwsNY{dIp| z+4`-%n)LWsuSLa&1sT@I3%}Zh-oAA!>B@@0j_z(_79P`=Im}E<1~orEn8{jmad0?< zgos?ae0gEmYDEsFynA~(h1LBI96x?MBPh-t)Z9OD!sE@Yt*1AvE&KIw!h{KC?`;jG zkF7e*vp#}R&uEj?e+C8)rXQc5H%4#IkgZo=a4`eaSXr_}<;k;WQg)~P{QL^a%G}=H z+k0_+y!~MrnPLr1&4W)*PcM9b&lXf$ty;ykZr!?q^78ine*Qmy|1Qi?kqq(l^jsXf z)o3PYVB@RU(@j}lD%Z?cDEaaI*b|0^#}=TLUG490u1-!&4fbB&>wY|JulRa3Tu@L@ zao0j|F|nkpt3oY`o^)Ki{QJ(%;@7?&hYMwznwt#^9ymB(P~c$Nv}x0Zw6jtto)+1B zyOHeX=-}wc*e7Gzbn2AXnc3#+k2*H_s2yJ7Il17=i@-a3Duo@sPQN+Nwi*;>(^M~J zCiC&}b@cQcX=G;aS@n<6AtYo<P9PE ziL&#wnm4OwYV`Iz*RZf@58I-CAB!qxE-ffH@Z{uV!|HE2i*;t6E&ThdwC4Zc@1W+W zM&on^jzzP#E?=&`Xz}8YSAzYo&WT&u=}I;%6p%{39h>c9$!rq&Ds^WIohy@8sc~{iW^X-Y-H+>i<_iHg&7mw<~mb*%*lRVKCB-$+e;Q2&eb&%eKKuUK7*2@MSe717t$#b$HNkq#2( z5fK+pzP!vA)ZUr@|4(|4ob9bA=cg%foH%*%;I`b`0@BjiQ&w(yx3%Hb`48pyYukCH z%@U4wiMpkp`C7F%@Zq{$UESS*F)=+yj=0EJ6fnfseua;#{roAZ`*Ew640~FBe5~>k zMLoTQyStg4+ntykD!zt(*tqe+`gs0_&(2om-|$(u!Y(~{?}og*tCQce@iaU+sjj%^ zzulMkdfvLjY#c9N^7;#;t$A5y-q6}=`g=y-g2?`pK`TWrUc5NNuC{7#{R>uBR*T|i zJm22k?SAqkW%2C&F?z=rI5uyn`T0q0_a*l}na;?~X;-a(+u7U8AMckxzAkom!Mi(_ zpekMI|C~88GL}UwscYAWndjZHSe)3)%+A-5cDr2V-o1M-E-&xDIKe9-B4YQoAcM%W z(~rEj=dE4Q`Y7qe{-12S{Ot@Yrx|9(KYjZ2#`b*qur(1EpPb!qCn+iUj>@_qG`*Mcl9Zy;kF(xS?!u zS=*DH&lPU(m7a7xp56Fyx!C6~EnoEO8%zFHZE$QpQ+Dsxjspk$BA2YW+25))l?yc9 z-`>t%_U=yRC-!?skFpA@`8fFa@cjAnXO@h6u!hJCvs|eVEzzXp9`kK3+qF%egu? zH&;+f>XzEdDYs)nMSt9{YyNjT|KR&OoPXxMolv3^wN^~sf8LSn@%6bTC&gFr{o1>C z@59%xx%(a$JU=IUXK(fO6|48IS;O=C`g-d>hSf)US3|_+a_k& z*F1Z+^q)U}9=v(Ovo$Jr#pl3?2#$Hr4O35vOm3H160~xmuFZM#8_u9{iMO}6JFma~ zc<1wZ*`KZ6^=mzQ$9Hm(t3##bm+b4a*1y>v!)%6uett8IQn{3rl?xvpVx41KEf&5m#&A`q&O{FZ35kw_ z&FslfPE6$B;K;b<8@Vzh>CzI<7x(wuKYIMQ@Yk2jn0+-jU%ZT48FJ|IWnn=C@H;6DEl1#*3KD zdbRbRsqs5gsRdzThtlT1f9TAy|GHsDf&5k}4Go6*Q&!mH2afehzux=P+R`%d=BCsaS67R_dHYsz^2r6@ z>C}LzsIJ@F^B=#uy1L@o%ygf+TF|)CgQrhZca#+t8X6fJgGL!u)YQ6|n6^gs-rZfE ze4v3*!Z4{NRArap)PNYCzsr0DCQVvpZebO+{)?muN6dV?MN5m$9A%wk|CjU6!@~l4 zu~|QsgoZxomt~H(ekggIao)TKPR^^(NAG5z`S<{@B-2TEc}MfotjG7YP2bl25ZG0I zud(2n&tvO<^Q4z<&ir^_Lx`5^_jh+6K7QPsd3l-R`sq5)C~)x4JsJEm69=%HDFNr>7UlTz^m+6BDx` z<)qMJxt3*i3uKNz-1&Uo;Xgk=yWW1iyKVDk<7;aoFF*O2Y9#sZ%X0g~dwVK-=kc$Y zrxN=87w_BIbLaL-o99XFn;?<=;p;|^!LBnQpvy| zCh6(o7!sTH=&0{}TWu9pN9)2xA3rWR`_ZOfH+owKmmlNR`tN~(ffaAJUhfgAz3MaL z_2uRKbLP(dc(?q1XWjhClbh$){o-_Vbo_dbeMjA2tFkvY0^gslzq_lHgNG;QgRzi^ zNXy5^$IV~w)=*Gj__x(NmRZ6mg=1dLC(oQ48yKa{^F(+{r%jpi;LJ?pj-H;JlB11R zSFUO*coR{YAE#R-F7EN~{{QADPp2Quda9XN`>TOhhC@+t;X&!G3$8ZVmvH<#Z*Tr< z?mWW{cC`w(w{oxf_Nlb5Ipt5qhtJQO|NnZu?DpErjvn5t-YcDs@mb<;nCo>7&Jk1F(V)_ka24iZ?7A3rwh~4AWnDp<5lXPUVW{aGJX2uJ$q!V z=JJ`DnFVNw1ZYi_&3mNJ@c8|dCZ-qm|7Bt-EN;v&tgm5{WH!$d`EhpkY&o_S@2kB8 z6jaoljMI2h?$>d@`*{4nh52zuqwHxfS2N2c%6`yqnbGyUa>>%A4^K=~zOgNL_6yG? z(cAMRj8Zx>)r(44K7Va3cFCNCx?X0~hAf)y(`K7RbTV$B-RdWMV3{pDYjSS?w)RME~(PE%9!#EBCQb#-=c z-n@D6`gQk-6COSC_V)~0q5=aOrOopmTw5ExqxQF1-u->OlO_r6*trulM_T*)+tC$) zi(QHZwY0SZBO*F3T?!Hr7hk+(i->L2my8d4we|FlU0odxoH+xw+PbZ*CZ-q@{IL$r~9Pf4rT)|KW>^i#`1Pj~k!2NseHhGGj&uXueWa zRkdw?;9|FjYq#H9wS8`9XXlSEm;GI}rQLd^QcdQcl4*JK=GFUoQH%`1TG98l)j1rz zxPBBrSLp2g`S^~O=#B>oFEkkx=1z=Sd#sU}-Gy~l)|q*>(vOe#oBvJWV%@ROnNfJ( zj;&F;t5&T_$j-i_}$&zi;XV!G~h7y^(yGon)@86#hGug0B(l{;QyME5r)6`sf@4L2&%96F0NB zYxeaXHI-jKf6k04+npwHeo73R)m*;iesd39TX9E_VL^@9%Au8;`2KGQBzzDbUq*##m+LWygvS3{G!%Di>MK zXlK9Blkhdr#g<{NNHUMGy5EtF$;TBYpM3E4ZSV5=byA=R{CRr$(id)?T(3`_R+xM; zYD$UM(jdvRKi0owT(R(C#w*{;uOB?Pu1t_fgXWp#>wX+u;W4AR znR(IT#k23Kb-FO|^Yep-%WZ6A`1tsC)ciE+ld-(?p!)5*cYPBk2z+^eU%srY%%c9E zjo*Ab-Ks>4tfCEKWQvI$UUJX$czc zez;#U`NQYW#`WL#$MEI-sVI1Stn>KqO1VQji=SyP3y$Nv^7D7%ym<}n>=S0*e2{qN z9N(np^O^(HvsJc4ec1np>ERL1&s)wrX)wGvqNJo`Q1_<-w1nfug@s~oxm2$$E!+7{ zYTmqgx>90AB_$?Y3|?Md0byZjC!WU_KVEzK-o1U1&U|OLR(?)%zNA0buYJ-ap(oFt zDVdsH%~*5q*fF-`<9)6k9v++R+ItMYyt>K_S_R_d?99vfr$bQr!ONGO>-YbQ68UrG zOf{&zb76ORzO#NpNQj7xj0~t<@4UVxZoPX{l+>3h+Y}?orlzI`&!4wnxe~%~)mA`K zQW7+eqa%6k^;ZusFQK#BCwMKDuq;}l;gKQM$9(7R!K|>X8@HJJ{2u)J%KZIK@zu=6 zY(XRAgY)fjw_UpLRB74L#nr(k85470&fL|-y(UM_@l9Gj-|XJH>(1d}Y-YP;?)BWg+WMu4L-&8a?4kLVz2>;6<6Id-fo zjK#L%!+{U?N?%{&UAS=Jj@sYb7OKDh_xJaXvbR!AO-%u!^7?7 zuS+}Q7B@35(cjCGc&z8u#=2IfGdCq)+!9^$uB)kw>qYsSAIg~%EMI&*COjuMC(4dR zVPXR-3&Xi{4t{>GZ>KAYFq9OPd{&D{<_^QOky4Ix^&1vAc{KYxCO zxSGEH`3*W34<2M(xNxCG-JczaPN|WRl9MJ)GAMf!u_8q4$@Axri{>6N@3;R~@p_fq zts4!ktqV_CyI-UOS3X?wK|nd4%V9b^_iH4zP@{E>QYzch6fK4&djyGzQ+8Y zl+nyHD{jXMv{Y3)H~f<4$l?z*I3^$_#-^{Y@64dU!6YOsJaO(^-l<-#$;bO7|5p3V zx7+I*7oY1?X!zyzb?5y>ssZN(f>+z0Z8R zy9MuN#dW^Dy}kR|vG_d|g1^7LJ>0@6eCv-|CZC<19jI53cXwCn3rSEbAS+9YiJ7_g zlQn2kV`l0`&!&LCTVjQGyeJW<+t0{dS^0opex2UG$y2ZHdTny#+_@c(G6Dm?dNm)L zwKmN8|KomvZ*Q|_mqkjjGPF!c`~2ZU!`ijFTCYSptcsucY-Va$wMy&G?(*{)P4oNv z`!D+5J$aJTYiZDyly~p$?mlt;yt(De(BiW*3=hBfx*|lY=F3I*Me474rOgy{ba?LF zyO(>3uXxqUs|7zkBswRW?v3mJ{QUgk%a<=_q=E*%qU3L^xO(CJP3AkhLg(gg-m>Kv zrSwaF(zWx?W&kJd(S zPkMD_rP}(l=g*uu^7Qm{SO2dv#}A%Ae_n$DG}^a#@nQ{z6X(vk6%`pZH8(P!K7KgX(E`R^%`TY80-uip5e46>`t%i+_Ok7;tB9|m_{WzD@)TJ3S*YDoFJMdk@ zG=Wp659Z(H`uBW(bnUulN%!tOQ1cN;l@z&>#qsjxwV6M+h(#H@ab!+hKCkLj#_hkq zzZbr`qB*(mdWOl4ijRu|&Ah|I!$FOy4Bgs-f(?P6wAY4pe|~=c`fTL|Yb`7-71h+- zOw=70UwrW7${eHjs8smJQv^2GC`S$zC1GAT` zD3Nz%NY}w;_U>m%C0ad=y%LNkrSn7df38;KcyNABW5LTwPj6jw3}A5J$P`=|(v?2H z_L@ege^r&$z3TV2F*^zv`Q`0QO5R_;lfcEz4eFfgL~Y@SELE#r8KUJg%Vg!jn$P|A z|16l8m|SjG%sYN+sy1jo=$uJp!EKA^6x%{^T&*2qi!j*!+=u8S8hx)t;L z&9mtQ4UJoP?^iA?EIhaZv@)y~w4ChAi;IbqyA6-6QvY81^3(DDcJ>+j?IS!mRO)|A zr&ND$T0J{u#R7(3|Nk6!TKg^{gO5RRiK|85;S&>;1?A=YA0O|R&Z}xtP|(!m1T9zk z_4PHoy}kX8-Mc}vGT{+dF%!#X(E3+lq&$^KbWP9OB0 z$rNU7yD^6E+0)RyZu_In7yn`kSiSJ~17?mZfB74j7@Rz`ckJAm`0mb54Gj$kC#R+t zC(X6BPk(4xAEKqm(R6WDqm;~X&^pr}pXdL-^F(D=;^DT$b8{?zJZzVD@&Ekv=~IiM zCma&VZCO*Zw%&U2?Az1n@lv1OZf{$=RyXa;jK&>1ECO%c`SGJd#A)Ya8nT3busysMwd6O*En@Ji+ArJxu4{|MRG{*^8{vrpE^WBGwX&}jL_ zoSU26xR0*5`id?8U(rsP&1q*3EpqMlnN^@R{j_2JJsanlw|B8PHbkv;%gkJP@vJ-p zs8jRz_jmJ3+2AiPE;1XZpA*TwoMH0geEq-R{GCtwj-5Nl*KhxC$Hg*#fB%;D_UHw! zxy~ofx945edSUl-$z}ETkWyr=(nJpd8JV8-`~OMl@A<%#laur6?0eAaH<{xH@86dP zty{XZ)O&Vl`n`)61-Zrbjru0buHEDfpab~Wy_`ZGn z7Cq+T=H6KL_Lf_}@z?8 ztweu)zsEd_>(r}lzXjLVcsVn)2yC8k`f0_(R&j&OODc0Li`hPY{CMKz$ws%ujbW=< zlX;A0o(Y(mcVdE~ghhcuh*oI#I`yeuf}*0W8#Wlk?5PlZety2YmlxNTsI`fypeDW3 zLWk*ku>wLuO%s*f51l{1{@&-ab1W}sxHf6#PWDi#_%-!a@hWy)(@b>gtZ3YaH+4-~R6alMxpDE?WP_K%jvSf7GRF_zyLa!&nucc!0yGu`XuMh*=5fRb zvY2r3PDx>PzbzSss+a0O6~2J>qKg><;^OV=_x<9^%*eHtpo0=a2T2oiWwzugx{-`x|a-Wyxewu+(@d($6r$sI8?T0U44&HM2+&R9- z$NP_W>+ieq>WPQ%&Z4JWVe4W#yT$dLe*As*W5xfHlq<{*9ExFw&Y$N8HAKtb%Qd`G zZQ7iE-Yp~~WVg>89X&lzRom6oW%cLmMJZ6lpcr=OcL~j+V**PE>FsumHg?&VmXFOI{*ItdvR^8^tQaaU8`2DGMjI5MUb

5_T|m29 zh>q9S*FXOB^t6V){`Qh$D-lq`= Date: Sun, 7 Jul 2019 01:12:52 -0700 Subject: [PATCH 113/880] (Hopefully) fix logo text invisible on Github --- docs/images/logo.svg | 55 ++++++++++++++++++++++++++++++++++++++- misc/media/logo.afdesign | Bin 23807 -> 23807 bytes 2 files changed, 54 insertions(+), 1 deletion(-) diff --git a/docs/images/logo.svg b/docs/images/logo.svg index fb5f6c2e..3e2dd72c 100644 --- a/docs/images/logo.svg +++ b/docs/images/logo.svg @@ -1 +1,54 @@ - \ No newline at end of file + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/misc/media/logo.afdesign b/misc/media/logo.afdesign index 3d96656cb4f7333ed94fa93a5dd17a83dad1a3f0..9535eecebe16ff2a774b31d8fe6f89ffaa32cc10 100644 GIT binary patch delta 8845 zcmeyrlkxvf#t8=X$}1IP85kHCBp4VN{FoUS0*dlWR2Uc-l-)9OG`8yhW$^f`!}v-? zD=%EYW7RS?2BwCy=b~78EG)0tN{eoOEoHpu(Zs}eoh=6xgbc%KIC4&%kUVqgWQsA9 z*O$gA0{|93)`fG zG_?09{4d{eW1jV^?PpZ8|Ho%(aR>kWH*KfN&;QH+MY~yF{_lT%$I?%Kb-Pdg`0x97 zvfVWQ=~C0eXK)C+F8Npg!TNK3Rag4NfAW(aeqs1NA??Px|FLyTKF;ob6z-sKN6?*^=56oKmMI}J!Uwu?&xa2KleXn<=X`8kJYjKeSh;@v1OmiHm2Ot zGwSpHKl9>$T~k9p)2iK#fAu&1?@lTAtJ?S9`|;R=(ebpVw_uzjyBk~w5jA7G#r04(=##UzqnJw zyoFp7)VdxV;$&X7p(mk@%XG)09-~zcHZ(AwINPaYXW+I#s7dUA0z3QkcU|2K$Cz0h zx^x%p_$XCcaOf0={D%*MY(mGR6=%$y_-e<}`U(4=>gHz}X4;Cgcw~96F>Fu>uP9^| z*62C2T4A9QhpEv9Cbou>|FzE|X8jjG`QiWXiU_Iizv_8-n;Fg+uMSxz#5GlE<{{(s z(299aHFKthr0qDhIw*8n)HC57B1%VYD0A*m(0F+0^#5s}_CKxvem{PVGoRkFA5Bbu zrG-K^?rN%cxmjN?A8%=)uFKC8EI1xuE4xlU~13N=!lc;(d6+5S02h9HYSm6wu z!^ohm%d9$2qIyr_()BulM{YcB=QOR-VB|~YSL7+QI>?sbm)tKe8R2qb!qNZYn*T$8 z%ztcu=Ku8nI{I#V&VD<)WdHtuzpMZMz00oRq5I`Si^(tBTN@_UEnr)y@cd(ay9n>U z{AuN3$y2|$omJm?_B&_HuYIdOYz$a@^X{+wD>d`~*fUISXVx%uP%ut6Q~V!$;@UFd z&-WSA{%q2zKe$zR&Q7mS|LeExOpgC}W@G06?-$eK{6C)Au=fA=$q$(2>vuf&4Y{rV?r#uZ^Y&-`jng%e|29|DU1MCv zaA5Y0vYdba=6NPQ`Tstn`1rm@-~XT4_~yS|an7&*T3=>f5B~Gd&N%1yf01AH)1ROG z_NTsbmj3CE$Lm%xcZWsj{%>FP;71@!z1$Ty1*U1|1ygqIFpNyfY3$a|%JMnL%q+J; zfmN&`Rv{_oN*ix-&5`p=f?X2vYZU5A9<=prmpQ^Oo-gz_-hI{MOHUFai$lKDb92S_ ztINHYQu)x6E3ii0L1Ah0>$`=&>pPxq6+6#*zqowcf8EdwyB{%X8$L3JNynaP=oF|C z4vt`EJ2XS{NTx6g3|NjZ&S38zYP_e0Bm3*$D=6|qk`CsX6n=KjrA1w(yP`{Gz z?3@3wR#ywdKmI>E*H7%T)Y*66{(G0k%$m0Te|hgqrG-E4|F2Hj^tgWZI%yqO&!tXt zmj`v)M6U>#>NW9UR;g91TIRAt^&y_iLZc=w|8IPxXuVC=bzZfDhU{889*yR$l8?Js zxX!Q%M*jI9V_xCr_5b~>2tH2N>!pR>N*Q<7cq|NXL+YuEpn>Y>%?w`j%Poe%A~B73g~ zYHeP*@xPwN>SK$`bBl{U%yQjpas63=#i##VFLv(^@7=fGb35Dq)U97d|L@;(@oVe< zrCxP%ik&`l57k7ibdn6Y=`~gTvxQ!hgZ%>83p>8fp4)h6qQaLRJ$~;OkB%tyTj@$1 zWMtdWB@>?}5O|$rIV-g1on!=}o-YzD%9lumoUFj1#qKU|Q!bv=IXRF+s=oG*eC?B~y|r}_ddt7=51c#aQb+jh zW&h({{{Qbke`n+R8(vYfg(p1yyjw9$%3+qn=|t|7*@xTeWcc|0{y+NToXqW4`-5e_ ze*4e)LcZ2hOeRo5(C~+SaQ@BlGGoi+B2FDqW)@{LV{mRPswv!m;MPGW z&L6pxH*+d7o|=4>Q$x@8Nl8Re!Ry?T$qn;1Fm=f%Aq8WEV>aKXs z2CpPG9k&>N-lni)=C@K5m_^=haACBYHm&}DWX!*{*4n)pcOx@HqV?kDZ+LyCURGt2 z#qX|!%T5~I6MAQf?m4K`@~T5#;yRy>?cI7Y*`fmr?blgzHLyu}O*7afZoR_1Vu{0% zj%L9{f?iKeZ#{XC!Y;YOY4Rg3&3g5Y@O?iT+m^bCrRl6}@a7CHTd~13NK(#p(gJ7q zUUByY-aQ%zS;8*tP*v>It-En_@$O?h{e~z#{p0AJ?4q>qIX7 z&%FO%`F_3hzW;9f{;zspyt_e(nK{RxnSIu-Jx@bQ_Xb)Tr~Z%bvzqDQvG4rxHPfu? zop|`wR2KZ#pE+~Wzy5FalD*=S|IS~QY@lksCoQkJxx+u}q{p)zMOUQz^mKYSoSG$B z`OfvJKbb$#w({Wr#qaAczsV~&|J}T|eAEB;uO{Dh6m)%lT_`mqY^H~*vFc>2lN-*g zGScyUq#3d@YUKx&X}XiN7T49Sx^rja`d9x?_w?468z%FvTz$*wqjY=R8$CsT}OdQD%&z1Mr&+|rZht_1aR8;CGAbh2@CYF9M7a$FUB%%`UM zEz@&1XM7vf#gt@vU>LfB&Z$#PDS0Z=CRdN&mN1Yb#1} zg9SUK^9&M?^%w>kZTfZQwdOp}T{Y!FhHd#?~C#o-~dw#WT-l^~9 zak;zy+h3}mbu;46|0!-)d-GU0x;b)=9$hMN{+Not(Z71>U;iW1uEp?Ftv;G~@csKg zv)=uee#!pv^JnAeM2Q#nr*m2}4efa*{*U+V3GV&(w|t}R_xi&Ldf)zs|J@&(!~gE( zc5UB_|7A~q%nQjVE6IFmAXWcBaq*qR%a^*BJ!<(Uzs)Yu@xTA;o!Mu;o9nPo{=fd! z{>v55|J6&qkI}k+-$YdX@Bb^_-ctK_YD9Px1U^(gC1I>HUo&`_&&*3lBF=Rj;I#H| z?)h)~DUaR#)c53{ps*!(a5H*fCsu0Q>KV@>?&z#sqG3+k&r zY8{ady&bzaYq$2_@{K3&mN?WO)#CoQx8>|C;pRDevc#4atlD+&Md;1Bb4tamDzDyo zR#uY{8s_@{ze9O}P1XF}X76gY6}~*5wWjO-ed9I4pZ*8_-5kKc7pYV{^MCxd8Ahq6 zrfB-IyuZ1o=Kp?n+es1o`k%ImpKU&;s?+<>tA73llWv=`3zE~)KK{0TF8cky)r&d) zVSjvOzg=UFo`1mB_uuscme~{kn`g)U{Nk2rvi<*g$Ma>oBmVrqcx8{#>F?%VvycAo zZ$77fx=S%E<>Mdo4F~zX3;xtk(wX&R5nF7-{d)QP|G)pfU+1|dV&=jBmlN3j=ex4$ z$myz26VhBzP`}ajzs|h>p&Je;vPFIQf8kBa&3XU*tF7kD`>#8n^Yz|G!P{b#{;e;Y z#{7SKjltIMUH|`gl@^@;UcPJ7>;H!roGnQ``d=shwORMk_%Hi|h02X@9n{f`y8gw! z&m{Kb#a+w(M_ww4byrgQzkkb-yFHhT=JmEcP+Z*hUpG^7znjv<|MmK+gyrdj6o|9UUq^}7Gpug=Un|NnA~^qtRt|7PXy5&yD! z@5dv@_ek_LRs2jX(v@b zRI;!YX}mAGPV|~k$_R;gVERH?b_xHZj>7#2_W+%^@#eBRVRlj$tZISx5#U8q#V&QC3xp(v7 z3b%so;aA=jZJb@5`(c)=QlN(C1jYloCxRbuV{hMaTY8$8tJe~x)mOcACVhU_F=K&1 zBNMCFL?yR77S}kNSjwx<+ZZ?P-s|oH(kLKQFDN27C8M@{8AtD37RN^_Wu-RCHh~kA zvi7{{Ik0x3XnJ{Yv%*8k#^|u)!YeHqz%>BiHJ zUVmi5#$WGzS>jKob7(=$>K?})$?0x0AO7E*|{`%+_>#I9I{O7;6(_pIFHrd(f>n6SVuk*s_x#`jGtIr;N@PGLl z-giH5|B2o<<7SeW+K0O{cF26Z_U_F+8Fe$!=k+U&THn4{|GoUwtqcF-9ha}+6)n8+ zRqe^$M@Mxp+^LMe;Hwna@ibUy<^-2T7xJ|_T@#4%^wfbaV)$jW+ejN(A z_u|(v6RzEcOAEuEPPyu`@Md4Cb8;{Hw(Mf3#2SB&g8@k_0w+h zaq(0gzcS;x+2+`oD&N=j`yTyY{Ju}@b3B_dn>FL#`>b{1+NpYV8xMRgpV^(RBg0Zv zSy-5vDOhdgmYgV9z3YFy;jQnTws-I98%A|I9f{NZwq&Cg*T?_srTyP5G<|dRk;DJ! zBujI#%ay5~AO7nFyt*TK;eFE5D?7~={MUcJx;uH_{^Ipl{_f}6b>^tm#;$sPm4s$a z){2XcqSM>CW^{bqaPGm6zyHIsXa9YFgQxF*xmEMy|Nf5WLo7t+@6zX$NfzeO3y%5p zgF#AYf-}E0$3xxLn$Ssxp36PgoeLJ76!b*P#PzD?6wj$e9*Z`e(Oe#7l6ZuRA>Z!* z?b7D#pkp$woE--k*(UwCuVK;1S>@GGzxdOJQ>tFBr&Nrl8zsJsS$QbP)24Op)lRN( zl}lYEvqDrP7*B_@PMxRWlCZczzx31;%}*sp{u4!3`7ARs*<8t$nsP2=Mxauc%eo?t z6+f0uQ0n4qco4EGB}irJO{LJa0;Q)?GS-DH4=C-jJY$r0YKf4si2R2G!mQ9w1w-a=*rZ#B)y4hS38e1@@6bZJ=L+|(#oi<0Vx$(OT!dBP5d8b2CZ}o zc~!E=PdoI_q3MmGTC07GQck&8oordMaw(Vh(pMsjyi_E&MJ=6lX6Bc~1+Gdb?-+Rp zc^wm(wQMs_;Ic|CZ`DaY9!B2Fe7r+q4>0+KG%WS17yiuV?b5ht)18dR%sR=!0lxdx zc3il`aE!wwVxgMMB6ij)4&A)C=aG}QB`lfAxA3Sy)RrU_-_tb*FDY*Mq0bP~Tk5t^ z){JjyV8-4V@Bc^bx$^D5Z%Oy`TSxMy^d^2V)sXJ4o#MN==6`&^v}@fz`Sr$82t2lKik)baI8L38Ty8wW1PE zZxz*p9M##UUw(Jb;_v@W*9tb*T4e3N@_eEFyjdoF_1YQ?r}-9XEI)hqO6fEE9zJWo zZ*0xKd_K=&<}2U}D~*YLZ9kb^Or3H4WPLGZU;B>v#>ekDtS^k${ZYc+E@B_Sr}q23 zDsOXgkWXUDzxx^TJk!>kE>{=*AOG`H*6oX17HqWhJ?fdbLP*Yh&4*iti<`7tgWXsE ze}3YxJnQ87Vq$h%8BLqk{nDA=Ui9z&s`V3YDHe+cXwI>dUGegCe_y0S8{(L`o|1{BvvKlRu7gK+~7m@mM>VI^5pUhu-b}o&UUpn&zczF0^6fT_Z z?)LWP=05#>@-%T(HFk@e(mYn>xiP1-rN955n;G~2fA2s2)7sqMYyN*eb?pD`J5Rs< zpFjDbxRl2GmVf=qz12DAzt?Xu@%z7@)7Sg)ydPytKmPu|^7pBQm;TRexRh4-KYp^Z zM1f%V6{BBVR~Nj{{_;0A*!cGU^2tji^y=j$E$=KtG%Okkh!@l8s+%WZz|m#Q`Fa%gE=T6q4rz^^kK*ZlbFdsy@Qe{rY(^_usk z?+9=xNvi~ zjjLjqMM6vStseF)xAZAFN_-C&HyBy%KB4yL|NkpT3(o(a?^OkAkJL~-D2FGP_k_9&dVmH2mLOaaLqWDv^(18 z=I76!KRe&vYxF|Xv$`?u_cFuzI}15C)m)L0o)vjDdd1}*rA$@}moEGte}+?6?6bt1 z7B#c(_G!hwH~REKAJ3d?ntExa@7|yPCg<{1PwRViXw8`fx7vL5i3)l3J?@NK8~8*_ z68_$4V0C>{aQRrp&CYME>`4i>3%j^MVhfT_YjaNLKN0j_KuF*)=QNJSR5y_@2`e5) zCNZxk2X+ZLJ<5CF)e+jP<0AlO@`Up^6&~bhp78l9GuC4v@9SKLfM76`#_$1D4 zJp8c0LS{A3WKhQ~qCTmjkr~9C8=(Phh%-1I%t;p1D80GufBi-Ina>ujk`lM6)+u>-qWr&gzv0 zI`izm-;z2Lzi;Pt_1(=uRbIEY9Jsw^ZOERpQrizKPm$EVd81xWt&=-=mPhQ@-}c(~ z=i0Y?TX9+Wn8c|hjgB5e4uKT6h9cu+$pn#O6BRl>=^Xtg%)qvoVV8|aTXRdIf^!m2 zYk)vAXm)|6!N6Id=-k51Ol{83r~ZA}qhK+K@k+v;6{%j!WH$(NweH^F;}&6Y`s~C= z7i$^bZQT6V8kRpa-&+5JtM6nw$1~v3?lJvs6lMF?$(HltwzV$Peup=A>V|V>bO_oNtzkW&;&biX zhpw(8_dC^|yV|hKab0j-bIU3Cq@rxf$@TU+L)_6>cF_mQ z+3I*Ut(z#qcI|#b$fGA07dn3pv#&qJmU6)8cxs-T+s=glvjs|`uP@Htc4p&ow&rjz z1t-f{Z)ff5ztQ13eb>ZH3AP1xt@{~t)n$0*1}|9Yttwx@b81QLN5dBJP0LjdIK?gF zxyHsRrr`Pes{YZYoNjrR0}I(g`qnK>-oxvE*tU+Z|6;6BD~t0y)~o-$7hes$6YN*XTo?e_blH~p#QgjpLuo!_+Q+NTX+mgWZ~mmgi?((&{kzxln^ z;(7BUj_W$cI;>n@I*IQ^3U`XHB5VA~J8MJs-^`tF5xU^&H>;I~D(r8)s>MSrYBv?` zU+AuyEGM;u&Hd2rea{1ev+Fay?%&rCHQQpxsX(Lpre=ZVCni>ADyNvp2n*iopXzd4 zf$P2#D_aYXZ_TL%8{WH5K5weFE9G<~li-C;t;;c&4@c!++T?Ll?&P%}513BaaepuP zzvALD?SI~yHYpd1Rh~twTlu+({@?rj#g+GFb6ee+RV5^hIIr)#Vwd^eCEet~Wepas zy0czc+)HX^<<+l_{`L9Wk4+_Wn^OCxUlhKi@uTfh&gK>g=7r1KEG;Estyh)&Q`mjW zZ~lc7&$@N*n^YRe7(JC*zoD1Op4B31Mc+K53x57MnDvp7JMdRaWA?AzHmhWtN+V5YA8VLszA9+@Bc*zYdfvd^pz>)YK@SeD z4XN^cH~a4A%^xeH9$ipA&ML0FI>yHIk8-Wb!_bEJH!MX9?S$QOTs>YXUoX7YpIi7U zjZeyoZR&@7E6$IvCD->%bXbx6-txmU-+vUN=rlfsQ2U9)Y_Td+Dkg(2jef&u%F?L12?Ki|oUtCc&e!am#e;D^_nx zI{2@jm{!j8dgjvB-NwQXx(;JFrjMT}o)YK5%vb&zI zWm9wHTbH9b>>ZOztY5oDCJLT(6BTQSt1tM?vgJGDmK)7YEHA^S-upiB`?NyMTJ3wM z8w{K?Q}&71^ez&tcYPNm%EWNA)a<0o)0A3U7p~)H51BdW?Kn2)&S{4oy#W(niYU)G zz|o?)*n4|`Pe_7X<)74)Yf`GwIX4nNRPS!7_n9MPzAZUkaL1BZ7riA}4O)Q>m(1;6 zeytVbHSFnLpnqJJMMW;~-^cex{Sw9dye`YFtC=aL{JdNHhP7Jl(&U~?bvkB2ITdEE z7dwtjQZI0k*ksnOw=VGVna@?uf__i_EIrFFbLy)0^;wRs)&09a21FcA4SHF`QZUiu zOx(6(XX`by)(Tj6%|FD`-gHA~($ofFp-IzvH+?Df^1o5(og5>}+S6)Q>6CJxy-eZ4X$`=-A=y3o-@uA`$q@M7`mYr%U~-<#~rzL8bk z{PK$CxMk<232dI#A3a^$%wYDfjzr%a`+k#oEAFKwMo(w0)~;H<`H#cOv_BIW!W=E{ zE>k@inpXGL{f_%B=~~-_uaePQ?p5v$?0&4`yRpqSZRy`Oi6)yy6+g~N) z^y6O}zV29JZXG=9)AmgcZNfXJCw^9SaKGql@bb+h8ADcvj?PDO`?|QZ>eowNOmV+@ zAbR1V-dCqP{;bI^vDRVJHVBltd{Hy_ukodaRo~WY+~f=Oj8D&t{UMy1{~%5z`^T;O zX+3vyvn-kwgZA;f`!UJ5z->jXs==?;bk^wDLsn92S&AgCx0zPnys%MZO6S+_hu&N+ z-yM0ANqEt-1u_4oneKD&wlO+kamKtpcE_7ZQVc(uEM03RpIB(YAhD+JR4`xWegk># z@TMyl`xmcYefr4!%b_3jU+z_9I36MMO+`HN$gRik9{eqn@kmqC@pN8tGvmRkssFVZ zXD+^P{oq?3+rx?pIgdBgF$u8loVxIb@~oVFJ6C@`qQo4`Ei+vy^}w2ceD_bqt9y5u zo~y0jrKNrRr{{@j>dKM*&!x_;+c1?S?e6ZiV!7g7n*W7LVh%ohGi}0bK{3wOw$R{( z^L8ciRA0Q$Q4s$kM>b9>rGwGU)i}sHN#fTc?-whiSA}KlTE?gxH>pa@(@rWN#e{A4 zw|J3)lit>LEPuIIzt&kiWzw%TPS;|)>&hms*cPTRrCvy7!u}unk@tleZB!52{XCht z+sM|`+E+n;`J7#{^Mz+04f+!IQ<;?|S~yi|!UjVFaRC{z%!HSXOxa&QpB3-W3HtF# zqpqBP+6LX8T#xf|y&tXJt}x||yX#TWZ@-OM3At4opQ;VZRZo5s4ws%H z@=E(cjG50tgNO4w_U|a#khS|v`-ZH4mA2RC?`Hj+`}2vp^tr01*X}<&(dfN5yJ}aS z{<8b8(wv*mKHb!{W$Jw){@Z`UF3L427qF;mZvPa1?Fyf>_N1mOq4nD)r*P&+2Rz&% z^o?I5DlqHX@i{7%#n(QS?TTSev=031_|xf}SbcQLPN{3{bJ{JNj642qa&LGQGe@a8 zCGv#ARu_(f(8`0mr}7tU{2KyGE{NXw;Z?Vue7LUUblK0wquDm;&^XgD##k6%T zdV2A*m^SS$Ji8+17t4;fOP$T#(-R^UGq1@^HOE}K(loo(H; ziF;kzl0+_(`5(T|ZVGdGo4Bi&$(i~8|NsA$-9n5(4L=Z8Ua2^lUCmNt+HMe!fnkLz Pgf4%_@N#pA+G;ZZI{Vtk delta 8880 zcmeyrlkxvf#t8=X66Ff93=9kmVhjune#{IE0Y&*GiVO@4%5IrC8e8@MGI;#eVf>}c zmlrPJv1%C`1Ji;3U1A3Nz5ckbZMJyyWrFZn<6QUZg&q|b*bg8$fe-k4{db)&1>*-T*OppH4FRS0DU;N&nN! zOWU4`svU9>T=`%A&$&JG9!1tK`QM&mVRyiGiR8A|_2LSdf1WL|5aaJ~YFphU&+5?c zbnop0kDvT!R`)(s-uJ(FVxiBH_>ZfE|8L)OOPc>mzPR_ff8TSKx31h1t21+3chBQX z^=V(~y?uLvd+)6?sk?pj+nr0_>oa3E{g?mtU+?DJ$@AX-zx3nn%?p`xj+w;y^!d~; z?{t|PvT9D$#DzVPlNPSjnd>yOCaQI9*eX}2wLXRk{3{hXWw;+~tVj!zxcR@8XF=u) zqh?{Zl7}gb!4a+&%B)feiEWah1ra9<55MhHvNLd9z{DiBM?sx^`nxV~1~+Dg1TMV= z2R=%bCmed!A@}8j;5^433kL&d<*S9>iuI@Kqwifxx>WU!eZrL)3C;|T8-6ggs|85R z40jA_706C-Xyrb0;qUjE$(jG+RSy4u|2Q%8&g=h%COnJ==hp;I2F^?%y`yY=-S`K)#=VCv!glJAyq z^p%%T>i_z`wKY#4G_APQY%4-*gheB@_%|MB1_F5yJ|E!^vObrqHUv|sUm zyTW9l&zi+)MOTAbjn-b$@=3j-DdH-m6>@TlOOW>TJiCo2kIn#DeE;srO!Lj?le$c<=AL5VK z&-kzZKQgYbV)onFp7r&AzkmPt`!~Pmgs2Y-IkUc0Z^=~t%E0U7c<-@&zoyim`#SH} z9MgK)JIlXl_Iu$SUuyjyCNFf`y!-3@6+i8M{uiCx#jH_pr(m3JruaYh#IM|%*M?B|1YM;`F}jKVeS9l7B}aazj1xl@Sp$mN-is% z^%*ZW%(->oaM8}D`k(I(JUa3CfBkHkxfWmk*@?+mZIW7TR#SiY(vI){ZKbA#&-n3| z_w)we)_?amh_8A3v;M~E`kKgpn=9(3F|LSs=$m70R$qU9iNUA;m0+8H*FTH+RzKg; z?C*b(UuRxx*459K-1gU=>+k;4pHr&;?YB8QJ@rgs_|-;N?KRW>JBJoAC?4X=QsQ7d zeVoH6?#>JY!)+H>CAONbcu;weFQ}o3YtK8L$`=&>pPxq6+6#* zzqowcf8Edw+aEEiTOKk`la4#FfK#ABFf8&>+fo(2ibL(4e_kCdnas}`!M>IM-7Mxl ztI2t+W{gFXm$GWs2R%L|VRb@LG;!_1V?Bbl+h*K-5O;UNNA`LD|DP~^wPWc76`T51 z$>-93{9A7I?w5H^dNI?yBc6*J{s+m8!hN|4)Cl!$dd! z|9h#+u5KUq|NnlX<@kTISo4amNnWj1{!4`_w*@TJnyPg8>Z=ki-%CEN^(!a&tlpyJ z|39>~A+w8_T|SFYXbS-7u2=jdjs z)k(*r4Eb)XjqYLo_kU*g!>S0I|C+`1xduA(;-9TDo3ZZc1DT3hvy!_iTXU7Sh zfA@WZqvJnnPYe;7>k_!T`S5??O;YO@hol83|Bnp_>veyB>)DeBrrlMA>&`wbcv3HV zp}f3ay0(6D9&g?0tk>HA>MIw&=Kk+J^^cAEXG^^*2kQmW7j}G|y|D0*pu(4)Jbup) zkIpFdTj>ZiGO{dSl5)E(s(HYGQN%}x=Rjcf51%C(lR}o%I4wLBv}y~N_ws9+E-OP; zUvQb_GD)M$XUm!1#Eb3A+`A1MzU-OcZeGTEihml5VdSG}HyZf(?;P34!Y#-h&>-ab z<#;HYLRg+n&#_o(hd%*@sfN!So^&aut0XpUJjP+?cZa*j(PQUELDdD}Dn?7SQUlaG zgId)m>m1HGwS0rl)*!`KE^EV;Y`T8eUdu!_=^p=0!A2@f;r;hO3%l_xP{I{QQ{?5krH@u=|3r~3Z zdAH&;DPx_+X@|v5n6>l%wUCkd^}qk)T+3}&>z7--di!7Sg8i>aIu?sIC?$NTUv{r( z;c~MuwoOSjF;~TY{hySZY?!}Qi|@v-`+BA9|Nb+7`7ajIn)zTq`+}6)(PvILv95Vw z$SQ6$UpY!A$@@*uZW)D-t$phEBxXt|vfIvB#1+IM7`ys_s1o;4VdpI$I0Rk(*S~4j zJoxbIN1n;1oQg~h8j~Y9l^K^zuHn=X{miUvW~>9s%_TL3`w!eYnZ)_y?c}|jii};8 z?{jMC**z(VC@OfJTQa#}!44%Zg~ig!hxscm9Ab8pRmjj#;!xbNcrEJ!-xC~>d}cCR zC!2C9T1Y!M@UzS6%r|TokVw2-)ZC);gjFD#sbq_LLyA_+1*yB@SsQ|y+H~Av{6(A6 zj+x&|(O?$&cz|VcAD5)-e_54D7QZ_ZE<1U2Z|Id0J#x^fg{xCuLY!a6_HO+&*^(0r z?bq3GOmLI(nq;ucdWA*Rf`B744ht?4^m=M~>%{{Xc8Lv2lizS@)~j`dA7E%|+uA0U zuCuV=b+499h=lY?p7T;3H#RPQ#;tNe)$N`}<5ZJ8kANm_g|d$?HaQ-itoA55#fVYf zr{T@ViQgIxE<8Bc%=CPtw#RE!OTDRJ!a>ZS1U;7HE);sGULzG;9Upmo?gf~(7&8TL_zp* zoAl3PTQB^dobPk2C$mq~rEA6miQLLAmV$;4eaZE&xMiJgEI4xGRpVpDPW6O>eY=8A zeE`TbL{hzaL|NTGl-{mvM9{pc#vMR4c*?2?8fv}jEu8%oB zpHv?I-@owx{M59wM;C05k`ep!@7(&A|MpMZCtz=Hzge`c@!#e%VKc-&9~Sue|LPHo zt1s@?i{CE(Z_e{9>3{vz|F@5RdoZtV_a%|P_czMCcWX^^pFWLmGGqNbqf@eGx_dLU z{9gP|d@U^Y<^Ps>yN}h>A1*TZ`G3_v>)!Hr|CfI}xAfCr-{8Va|NZxFS@LoA`&r=_c84H+AkNtSB{JyMUNzMP*Dif{dRV%!f^-d~``Sm~e z>Q}e__SUy|eA!+jfBO5@KP~U;IhNOpuXo-Pxb?2-s%^V3)r-%*;+2Lk1zhdGyCzmbm{5;)*M^) zvE&>*%4qXq;*n3aIu`q0$d>3OX7Ysp zYFSzz`DS7BxeX@W_wQS--|_#v@-{?Fdi;NK#zPhflMhUP z{{8>)Z~y)O-@KEKcJchHXEXXyuWgv{HN$e_6d%T=+BHoM6-}F@IWBu06GC%e|ZC<%I{LlaJHO{?lNud zf9XG?wnysxTl9bKq#d&_@}+rR+_mg~@sgKbeb-F?o}aYrwUvlgRld$iL$VRuXk+qc!e=?-%Gr|QLK+%qMD@58Gf?_3c7wpZQm+0hLmIkTC)KmM(~l$Z8@z3Y6FdpcFbVv|NlQ81I`N?*PrW6`Evow?@CMC*-w zIk^vQe*F*E%I-Zib6)i~m6{$>F zazb#Thlp2@iid#8$8&t{rcJx|E-IVeefLgw^K~7ePAAVM4u(AEc~aYLEsos1w;@C; zNJ%tuZCKL8Ipyw0SXwzmGy=Lr3?Cia=qI4E<97A4nSyuT+A4uHPUu+BvB2f^Ze!Ps zH)&3CT~T~3r5Qel{~xxB(qDFe<+;}X^&8H$t~A;B_J&RI)Rh16j?*pI zYwnBB)nof_pPE~6{{FN3hmM{R+pN&|{@9VH9ox&&zdy6sxYFXkv+li{ef#zYUu*lf zpUWyWcST42;dSPU@5S`<8lRo}-IOQP>7+A%(uo!oq1L)!PZh3>HFe5ec^*oEEAD%j zUAg_=?Yj7~Pj1)Emj&MF)$HA@wS0{l*V!^1X{l_R4Y9L@rLt`dOwYAO*hTh+YW|Nr zBg@_VZ}XY5fP@Ry%Z}~YHuLSHOW9H1MgPm!mT#Hy-)M&6je5=(|8+j>j?gUr*v&d` z|LGW;+e=1 z@vlu>FUJ1szR>BN8)ux;_HO#0-Wc@!Ttl6>_R@1(8UNN>#>dKh{yVpN>AU~I7gF`E zb%xeSI7)jd9Ps8_wZk^3{({Sz_ENUv@BeR)y8Wj9cv``q{g)+W{@Zg^EjzUG$CcX5 z2NJUr3T8f<cW|4gEf?k zmmTU63REeY>KJg;{isM#!2{;yL8_CLR-O@Cnbo=?Sk1Y3t5xf&;4_KS41?9C&RG1x zfqlV)hRzZzPxVcqU5~VMCN2xrSm&W@TrX1GX|mF932%l1vu?1P%T&Lst6G%~2WxNT z@?2EO*8IQzXz8~9<%^>CTv01qoWs++-tlI{o5eX&Pk;Y!+O#F^ z%zqh;_g9XVcwbxMc|Ksv-~XC-ADxWZ_}~2Ef7Jk~sr7Fn3>SB+MkOp1RO#T~Qn2|b zV@<-54Mi$kX2%RX4zzH@9CT(<{kbKm)o8`0sankc?>8+1ImKM~f3_QQ_Rst4 zC9{s%+`c6FVRD_QS+vjN-}fzIW9DuCzc%~U%GsUvLq!<*3d+{qnne z7JvV5x>c~b)*@^FmG=wn=glhVtJhLs;O1YXvHa}aE2You&+ys!ePe6><@0%!GhYE; zSZPe`>-x$3V(N_ZC!32YGsaJj7E`FV=b5(Vbh*0d|M;JuvTk48vS6c~?@`ah6_eyF zR(yD5IJt?twb*_2|JNt}{%6_joI3w0v$L7ix8p*oUymzh2#Ii3vFv3uZCdwBXMTIp zzx%6JPk5yGStLMX&OX@{FE2Mv`}KdnaKOg@_f=aaC0tON%dE=ebl5J-`Tzfmf>O;* zF9aC3O#UsVQg7SvkLPvEoBf#?5{zF`^*q1bceFarqpu_`8I{D*%-W_M&+zO2;U6xM z@AflZ_vVY@`bE>!e(wQ&7Pk*00Ph2&F&Elpsk5zeY%qeZ@ z@Bim!#{IYN`=@_eoBMms|Ieq6{l9(Z>DT|WFI1)`e_VNvkL$<(#b583rnO%ERPW8L z+gttre9OQ2%DvS&=fBr)G4cCf&*|&^c;1h)r5}I)U-|pg!b|_BHe5<8{O>BPuB!`PXn*+|8*F_0fBWP$5_*jBlW$ANva)`P@AaSjS3{i{{Dty~RlKaB5zO55KnfZt0PA*%t^sO0zrS`9*rqX;gOI|Gst4jybE=nJ+pTVfua2e@BnDKL>Wy)nB;j zI&aYgZ7G9Yj=9&EZ5JDRnF)4&@NAHr8=E5h@ygMH^DCFIf3K;3{&~vOi$85Pd$`S4 zJ)`Tk~^NzBbme`n$wyWz4)FVnJ(H{OW_fBW<2 z&z}or>v^iYF2yalxi9lfdR|PEj=bsdwzROX^gG21Wn{kiFJA1_ebk1{dSU0; zM+=|E2ya{LeYGMj-FIe5R&f5l|0%N%&s(~9mw~p?jKKT1Jykel9{W3JFAzS;6Jc)_ z(I{whXNO^2*`mgcjS>;Q8xApn#9rL^6m;k)^QJ5MEG$P7np7PV<_fl|9^-TFn{ZUf zYQwsg`UyLvb_gY1bm$gp0WlXfZ*9~_*|F*eBg<=sz!L?bADKH8BQ_=-?Gn`vj>vdW zU?IbI+~{x$sF!w4BL2<+5KH|U=Q5v}MyXxG#&Z(3ZRD7;E9dtAgD?7hiUY6pZ@GEv zQo7-CPS%N5k2!?yFLbzh@5qJ3o!?mNHUHah&hB+x@c#Xk)Z72xJ1*C|-L*9$``DlQ zcUMCs#LmyJ&gFfkfB(+w$#)$!ZB=ul9^_sN)4G?|8~q@7Q_s|#9P!CVS~dNato!wM z{?zy9{?9vm>id7@mY*vw3m=m>m88?rW7r{(;?_`PTr8O&a;$OkUuoHT2bKl{XMv(~ z3pXW=Z|?_E$DEtMwD_ow6eiR`?O zBE6mJUCX6@xx8?2Xn$3G<(Xkc34cgpM&|R!yZn@{N=*B9J7E9B9n61~@8(H=6)34b zU%)lB)>C)>&qH>dFHU|_?fw_uT;Hh}Jhix^b#~E_T@7-}=A8dCqide{n;FGD27xak zmWD5%d-cS_6oCt@@;AgiL$Z!tu=$-U;wgWSd2iB;{~SI3m0`|BrtbNv4QoHOpZ{^G z^3)zrfoFIB8_nz6)U717vUIDzcLQ5MgUSq+%tuTPn|U-Ja zw{CF#-OCp){A3niknBk(b*9f+4L0qn4o?g$u4ZbjdOExLgqv9DiG^R=^;S7YeV(p) zFui`i&a>HdYnNYKU6pFK>-;ty50#$1IHj9yv5C)~q8YDMckV6IxyfYJ zB_4gl)`xp1XSCmh(>BYevvmDg_uzr)jS^d7mc{jJ?_E}?&Eo0^Z*OBzSSon+UsCY4 z{l+FO$^s$$wnFZ&1gti-zYvryI_~_d#c>|%f&bplR|C16WGWe4W0uYchlHs^lMw)9I{7mBKTzer4KE)(6yt<`2X^V|Ik%F_k;Jts{MOw5U| zU$XMB{oY#%&l}e3t*SYuAS>v2$gt?jk_8h@qeO33yw&%5bWotxVIfm9dxM(X+6#g| zrmH@;64c#Xs{5H^$D@`VD_%OTRczekarEDrYc==TywaakfA|nlKe0`xuS-{k?Ukj_ z*QLU;ehX&(UUxg#p5LB%-bSI`f5m=H|9F z;XdlJS9HB2E~kn0E>BqY$iHOsiZ%HmSvhs9`WdEAvAn!kF#4ygpB&GGVyWGc*YtKh z{o(tZ?fmm&k2mf=`SLChF9<4~x%bSju8Hr| z;#EBNeD3V{QDoc@V|?b8;^yL5_OQpA780Cpwzb^)NzQi#uj-qo^vVnF>+(MPI`!|h zz5iz%x-7BIrcCG0^uqf;g%jEOpUgiXY~8aesIB;O@J7#%9r6dxEwaCKbj8t!2~qVI z%sMRhPBz&)ulH7Gi^D-TT^o%9XEL|V5xuv4nLw)FvJQ)Leq7(1=h+%}l|ME5`qa!t zal!R-P7ia>+;kG+6=smnTjr+yCV#?4g~{);wtFdU5Eb%%vMyrY?-wi=*i&P#U$|ek z@??$ubiwU;j24Nv_Dp6D`qX@uXIYeq%FhE8?e&V)$~<}&2_FAn-7CLd8u0p}ddnf5 zcKOZ5)#Cf~l^b@nwsvK*KMIk)nR4U(jM@;#t_v4!+|L+h&kz@~7CW}vUeGv`K#oaK#Xw+?fV&;NViE$793hmJDw=m|}USKs^6JBq6~xG1MezxklS(iJxP z4$@D}*iYB1Kg)f}+I5(JQs=pYuc|t<`%gFKHpG1Dx7^u~xU;83@QS|Mj8hyeHHqQV zFV9}!;5h5&yN=pO^0wM^ej?0sh#9glF25oS=`%BtscR3$WsVZr(nlg@m#lQPQb zI--BDhAYVBY0Zffr)A#p{ZU%JebR5M3&$Ra^E2$%_&K#Lp+3oX)4WEeCqG4Z{uVmv zW#VFEl$18ddQF<6PT5uc6>7jxE5 zxqRbB*Dm#+e7W-v->M4R_E1Lk%e1|lGhSY}_u=cbGjZZpTHbG;TAjSWU(li4w*Tc$ za~;!`P3}^rTCF_ZThm`ItdD;jGNmJurG1ZgT3WG#`n-!CTPE1IG^;%~$*|mP>?3)4 zl85&D{-iA@C+*!cX;H|ngzFbIHlB?;CqA=H=%-54bFM>q&p-OcoamWz;o}YCod*|L zy|{QN=VaoQnzSuDl9;1cyF0Aue*4km@yh2p;yJs!O0O)^K2x-o-|X@M-mbFu|^>Da+U7K(I#-&^{)=z&UMa;SLPQuAMlo!ePO=ySxh+77svF- zOS1b-)MtkEhr4e|+9V>p*Vj>4^!O^JsNa%PmKlXCUzHhn;dWcDO4usN=n?Co7REykuQ|jJ0Hl^`h zZ`YCEUy_#YrB<)4%~7C~J~3|}+nb2FER*%tos`UGVd>1c_c9^V@$FR6-Zu*mRb&g+ zhE}F%pJAGjcl_-Jro}5-UUnR*{i8F%qy4+x>+Q)il!+nC-jOWzdc;|;TAuS%F+7vH%k@2_T8P>dTO?D?uHBR zYt51a*%UeUsNcD4pLnP$FLloUA3n3<60cM~+I8y6BN=uFodmJ%hxtoRxL0&d&G_@0 zM|-}SN5^&b_Z#MYk_z?AoXzs{?rueKTh60b4(PwoxqR5VK=Jy;sBI~{H>Nr1lpdA* zv}mH>gMAmz$Ga;}>ZosQzGD?+J>Tq0*mRXG(lf#`s?P4vli#;N{D-EBxl z9-NXb^|@g!q9!dKEaV}(u4r0dyPfw`fl%#~IX@3~O|+V_X7U6EAH__?ivQ8sLuOAUJSSRpwn{4>vaB%6apJpnq_4}r%R*LN3 z!`}UiFT+dgklcSGZWocHb-v8j8JBxE=4VCjc+$#q?Cz9{Ka`J3&w3qT<}fpTKmomlEvAvx3XWkkB*SLQzMtoqj}5-Nog z9qYAt`EP_i40|fJQ{vE}-l7vfW|`(Z)?YpAtxC#v_x6WIJ^nU&iL)K`Zq9lZxogv2 zi=76nab=Hpx$W3;+oLWm>+Y0go1C_8{j4NzlbBU@tU-2nW5tY~Qfmbh#(BO>o1W~5 ztzBNGv}*syZ?hkTxx6*J>-K3OW0LgAf`2YX_ From 98050534d6977b4a4a9950a1014566c5e8cd847b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 7 Jul 2019 01:19:23 -0700 Subject: [PATCH 114/880] logo: once more --- docs/images/logo.svg | 2 +- misc/media/logo.afdesign | Bin 23807 -> 23825 bytes 2 files changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/images/logo.svg b/docs/images/logo.svg index 3e2dd72c..57639f32 100644 --- a/docs/images/logo.svg +++ b/docs/images/logo.svg @@ -2,7 +2,7 @@ - + diff --git a/misc/media/logo.afdesign b/misc/media/logo.afdesign index 9535eecebe16ff2a774b31d8fe6f89ffaa32cc10..ac802e0d835f827bfe21dff3acc66510fcf65b77 100644 GIT binary patch delta 23454 zcmeyrlX2oM#(IYT-j3W191MBIMFGm5d1-%D8Ni@a4NNhth4LXh28O4r6k|bbDFy}x zKV}AofTH{obp{3oWw*>6jjj5B89Y>V8KqRs^1=l?RxM*=U}`vfE{df`+whvLwCLv7 zQpSrOO)Pxd(Q-gR$nde1NBw-`BDT#XMmKvJe9b2+&scEL&@1@o(PjUptXQhwyYQsp z^n90>5kf|vW;AYBUnVH1dY^d>n;82$Hs1qbF)~?6&dUF{Ur6X-|1jm_eAiyK4^uX> z_)9*lpV0kw=i;`tJ1ktBCu-WH8XOEh#^d*4N+9Drqn1C^($BE z_&og`{cO^c|MP$4OTP8_Zy)=3+Ntl`b<`gJH~;GYE5a@|D5^?F8q z-CwlhN+1277GC|h@z>j=lQ;j*d%MM8>h^Bc`pab#3uAtr@A;R%va@j2#oJSllcIM#h~;!Bf>D<*qQ4c0cAV(4vTGIeRN z#$^B18?Jh7^;j`glPyQdSfjC-(MIgN50~=q{{l=~ctV>@99_iyB@9|!Ta1GiFp8zk z2xLufm2s@k^ON}4sK_KLu;Br#rTLDM@E8^YIR%E`0zuZk>ob^z!(6{4wD%VX7(QY+ zsM8yAT(|v?ZhobAx997ffhsF?FCJrPN%+{x>$AYXX#J!mTnaZ%G;m39T>ATcW^(4g zc#*^Z-#<>wyz~0M!53~OgY#>GW_oFabfs{epA-79a@z6@^`SvCA8M`l3XRN~8StRU zLtt^g@`JhsF5LS6e|P-X{lEMFzeh49C625TA}{Xu9O$|#s&eYz@83U89sC=xls&!2 z)-Nt3$`7an*zVgK&kzyEs+Eaqp3cqyUr@yH8hKJD66P9-;a zd!G#lnK|<$Y8f0mCUcqNV{!{<+{HLHJw{!0M{QCWR z|Nigy?^ioC9bw}4jaYntaj*1C_6z}`Pjde&eJ(!upH`V}8s?dQrDf9XnuYZTmcFm% zy3b`g>+ahBZfC_${byd_KQByWrqIgNSwDIFwJvFi9KEC&v(jhP29=q*p{tkdI@6o@ z@$XafE*-9!mYaEhzIWPYHh0E|T{r$h+?sHXF zdRy+_`^Qgy+momt#Qvu~!Y}kfVx_Bh`_uX>Y;x01i~c!X&MxLI_y7OxCohgH{QiC8 zg9{U{Ml5~!TYB2Dz5m}|+_2}+k@xu#ZMj-s{~uHw)3X*Y`XBB0qsr!M z#dV3s^qj^7Q%0thGBNM_ZvS`IS--VH@5j8|oB!sobb5X8xz@pYUil68J~*-YG%)mR z;1V|1(X4C`w)1{+o$s0ShFHGd*LMpO_wBzmiFNnU>a7p{`{io={=c%fb>*L&GsOqD zXN!ovk>O!w-;=9ecJSE7ES_zvl9LaH#oy+u;oElf!T)(HU+AUH*=uet{_4)TYUcmf z4H6&At*9)xE>m#i;s5`$t4`OqTCv|x*pO-Snf3p7o@+Y|`&Sm~>|5G>G+}#*zQ~`` zCV?M+-=AbzcxlqrIe*?y(`lWwIND~Wc3u7Uie>NqJO178AItN-rvCnn9drKIzmyi9 zu6!@9v|V`rlX{73QsKugIP#t3=v*y8=o_~0uR3~cKm7UcJ}>aNZo9_ApTYIA z#bQr>&wsV!_5BTV-*Q$6sZVA4=UtkS_W%8*m3_Or{^!3d*)-#NM%niN@v9zR+EVj( zf3}Y}=h6TF?k5Y&{l8su_kpIDmgp;+sa{WRI<>9}>Ea1reOc4Q(^zzwm(kiwT{eHG zD^9C^8n`>pfV0mcAd->GM(do!b(O#b^R^3*{?|X>b`!bKcbm zQR{16g$MHPCdqvGe?Er$`r$`E{by!3Cw}~66u$Lfi7xx|1DCJJ1fS8)kzV+8!t(r+ z(nYtbuN}C2Pq=>S!~dE3H;?^!?^u0S?91QfH}78i-~P5J<6rr`n;HN9zuvR?tjBh? z@Bi~=-TKuP-??Pc3ze>0veN(NPq z!@k|OU;Lh3lKmk%EIR(<|I^a3@jv2sJ$cT!YJ$r$zeKyWAq#m{s%b0Px34K zaq_+Rallu`YXir@23D0um-?9GdWNK@%MNGU+4=A=zi7DFV_w&F3Po%}Y7*I94?Z6` zdf>-lL31-D)-PQy%R-j&yiy5V^&w?a>I~nXBGZhf~V@0;LWyIfc3x|NCn< zsG9AuyT^Dg?4O5fXm;u>^;ID&okUK?w5|*;S*~lp3;hDnHu~J5Aerp4+_C zgFUqZBSohzslTda68JesY;q`FKO!1&bu(!%UCT+5L`-Z|7b!S7%DY{dWgdy@Ncr zMQMoMR5~n_8rAwltCTC%$dfC4dC*GLV6TH~PCU53bywWOWZ%{GLYZAF!qn!b911R- zlz6&i-l0pCQLRQwol0|!mc6e&8y>E(o?^5_#B*9$(n=F&;h9>_p>cB$8HKJpG-;()@QP_YlMSCb zFjbkoNch=f-qOG^dz!RzO1+|f!wr!Ii(Q#if*&?rd$B8^IaJ6f#qd#;mVk-zzKdec zk_$ZARk=QF-16u5`A47f??z_2ygui(_*T(JuGZM}fA1A09=U1GA}!_=)slB*g#zC+ zg%Gzt2X>SsynOj@^OJ_dynhRy1~QmliI~)K@vh_1i_EQx$E3RidIPl9Uu0{qSGPLR zAZoba<$vG2oGjIlQzg@k)*4R@YE6w>>-2evRjZ2EGoIk(lV)hmP5r;V(9m%YQyagD zj@cHAW>3}_j}?cUTenU;^!6lJ_DxIA{I{74U3Mw)BbV{& z)3o5LU)`RbnesQ^@>PnG>d$@ho1@lo7}`#m(D3hkO4_Mk_P740n#B2htDn_(aN3Oz zvv&ISDcNbNDX)BdDlp$-*Cq|86FkdAEG(?=O{jNcd&>I1f8T%K&ASiU-oI~h@8rMz ztNqKwUF${W#(Pf*j7(LUdS;5xE0x1Wt50oE>0BPLG;8Uh$r0N;R`vY)xw5GE_`X&D z^^MK$9qln$vM!7JXueEU@9Vvd_wO%#z4!0_n?~ksu3|zyYLic_G;3K({-@83EH-?xR;-L#0CF#Hj6GhTK~tu^>yYrv7dFPG)@n8R!ZM5C7{Qf!<*Z<`@$9D%Vym@iy(!&`I%X^C2eZBQ(bqW8f zzroeY`qzH#^L2*z?{ARt`L}=7f4|3bfB()A?8bqozrz3g_ zBy=-JD=6KrKlSd(n#~_hu6|~={j&4wOQ?`=M(s?++=YyJk4 zZrhRzlGD;Y{!ivH{u(w)W7%of*h|G)hG z|KES#uk%_HG4bI4^A2qP^Htf_$myzIo20p@%EVHx1~%r;Fo0FKiP|4s=mI@B=+RRUCaJQUMh)oU!?SZ|CS?ndoCHx z>uq_U+1&PDH&b%Io6^Pq`l^LnE=JpYXUm!1eeOWr)|Wgd9((>g&lG%nj$o!$=Jo%2 zFW>dL|JSe1%sc=8a*Xty&wu}B<=cpVS-tn;k>h(LrZwdjCur{5?|hiUO7;6$|L!l|b*t*%|CfAx{K}@~(WP!V1%Bx)<;B&zu5U9=s!mk4u#srI zFS^d~lRd|(V^)hn4H@$}Q7OkyUV7p7zkEr+8Ho}TJ|R{qO%d4y4kgylXAArO&tDsG z>!M!z@qhUXpC(K7ib^T*sdG3<>lnpH|f|H(HTc@b(;pEk$AlT`msOr|j;VJa~O~DoA9k;c! z%6Hy4yVtDZtkOh{B`zlz9+*8@S$MnI`OaHjZ&hX0NfWPTsY;!!EO+$@>~iT8I@&Q& zIZx&_hiB97z3KC058jPaQK|>&VIbc z#q`p)RkU)}n5*fz@7Zd-+x&T_@6r07@>w6xm3{d;cazzd|DhK&vz=XK%J%YBZm%fh zHYvX+UNYH3B2KPm#rOW1@7iZD)qK}J!`?%A zKACeuEmKVcGqxq)yb$rY@$IcSAZlfy-jPoyv)0IcoPYGrMuQjjr*p(v+4d%9oyo4B zeN#_w@3PYPkN+>0yLbPTZ#&rc10ns*to^tol-fX!DQ%*L&5!U6A^gRk+~4?#4ON-7oiS zQu$FYu2K4q=SBI(pjR={7yeKGeD$c|{rfw^U;VSMXN`T9G&f>tyWoz6Elhi!Eahk&3T{J$r(#7tAsya$>ph8Bsy!+E0I%6yew4$BcDwP)|E0$Vr?*=|Nrf-gIkr4 z>9BIP?Oa7mWXnh+mF zSN25~Q?+J_c&2Xi2-S9(Ya}^q^;OM~xk-GRduAG~Ji>F7-R6T!J-g46%br#f*9M7% zCw7{wbt-htnz?X>_Y%)irNRVesf?LMN~=m%h3Se&&WlLGMCxMEhwDuJn1b5~sU z65ahFHDT&0t;swyjVAgTDQ0G^JQTFjs&$2sXWQDal`6$*wGtTX@QXweECi+ev9PVZouGC7}9cZ+d@ z;vt@p6`ZXf92>=wx@TXnytYJFB6FEjj^YuqwGy6}jp8?ibm;!$W(ZB58<^AH>%3AV zO8?pK{bKrC|JDoNeYDhUqjYMrL=Eq$V@dL;0$t~z$mOz zz^d;#MZPcK-Z6z89!$YYIU=S;@=JQ2V8~k&lzfE6?8D=ed_o-#|3fBOZuM9tqFJit zEEy#E|9;b=Epcc5e;4|n?Z%w_^Zt6ttYecaMXeJ<9{;{?5gRja^Z&Kkw^q*Xv@h4O zPMSZ*dfHP(buUMC_UV`3-Lv@nf77Lc&9xR;`>)(zXg_aGNngE|217UBB8}x|?_Mc= zX79si z{wQH@7qO4vQ~RB-#@n16aMgx$b_?tWi6v^)4E?e^V^I5 z-CwnO!Xw4cA^{q6_Q|ezdAYHE+OPlng#$MJzpvUdDIq|4F0(F^(_uR==l}mN3Q9FM zX$Y3Em3SQAQ}HC_OWV?qw>R+I_R3AZE?#qe%l4c9=dYFuKB+DJ{J(F*Kc3$$Z}z8V z%wYVIs^|IbzLS;v1o4U9Jh}!g2R|K}>d)}^e`3udo$vPi63V1>T%{;lF-lwcq4z;__kk zegE`NYjc0E`TzOUvH!R4JpKBA_JzvS2dw|zxeAN)3nyBpX$B2b$hG-pKtj$ zU%9tB=lu8jEhc{d>p6YBAJ6+yw)Erg|0{o=T6pRI)P_rGh5!90YfBUeMqDxa#dUST z3+*p|V}p%v|8JkXMnbAy-0NTc7uR^L`sX#9{?|W?n0D&FcD#iAo0Q0n@dd{kAHVru zfBkHy(|g{yhxVuIrEea6Q(De*41`E4)#*RV~YrEO{9dv}3fXSg=5`SI8H zu;%&y?34W^r3|mX_-~+g`q7+^|5sl$nG>vD=GDCKz<2K2@Ci?ogKu8=KV552Q^o?j zi{-4T8J{-oy=yXgmZVJmF`4yqF0t+dwevF=Zdn@l|5%ZA-7ly3NJfKM`I@tW`JY{` znq{g-S)DDJ_~^gMK1r{6&-ts`CixxtpHX8h{?g@7)q>0tx0Att_N{+<^l3=xUP0Ys zt9DNR!|&zvg#B2nTe^5V_HesM86~d>;JFpemL*H z{q!#n=DmMkGjG#>{l%XvpNSj&(n-|Yc>G1op%_&Sx937%9n-&@dGz04-{i71!=7_n zjLT;&)4P#5%Zu@HL{53PW_4X%U0ufRa5h`jO#KTv_f4J|o9j7F>z|eEY`nUPyHskw z{{i(aCSU${+b%qsB-cFu!jZI(7mC)^bKbtVG|Oi5=4HNCTUVOjuitbw@%*L0*cqad zXI8v_tE0ry_fdU;*oDRiLv zPD}{~jKT^V&P_NZq|C15cEX8MCL-L)^A2we=f;c$uAE@z#e1fwf%{3+k7BYP8$ukp;X8NcJXZPsmB^b^0Stk6NB^cC8SlX$eJ|kC$ zX{p}rZ}kShlrPVj@>Mx^+qN%}XEL4amz;A*I#R9`u+7}CBw}6d!T%lq=ifZ0bm4pX zt4(kJe_s$hJ@;tT8q?%|_TRI#c)E?BpWnNU?ep~fcVADwTQp_fBs1L)X4%@J`J0dF ze$b3O=4H06JMhw>DQZFCfB&67RsNj6<>%z5(n|FocG`%vH7`k2a8BYm8X(XNntNh$ zn4-vG`TRm8W1I8ysefPgBv?!myuz?&g_+kf*$u*6t-Gh#xJ9I$KKpU=i*KA#IpT6_ z82yd&vp$N-Og=AghW(d_!L){Zr`~_ryYb@%l?s&%p9k(bEk`~WM zrwaR!&C|{W9~NyYIIdUr@rmRk;l!EeSIb^zaGAiS#wg<9Ay(fgb@k=s(42=dN9MS) zDu^@PR^Aok_3&Is&^v%LC?nt^Iz}Dr<9*bMNO_ zF*_D-V_Pf~=;3!y#4Y`CP6C@vam=%{)G6G*_H#P!F158vKP^+6V^Fb_MQO_7o9kBo z?&ysuIVnA}Es^0>@C~j@^+inw+Kc6)c7?oWJbsu%T$j;VK+5b$dzACuH(zrmCNNIg z5}z9+b>i6t#uc+Ns&rmTSslCm!Q{^4=N}~Qss?)$Jc$gryx$~BR#n!eJ@49&>jJH8 z(p74qZF~Z^BCmJ8EsC7YtQn&HU#B}$_kH%%V&OT3Q9fQR5%2yNJdU+^Qd!?!dPV3; z;*?)e>Yt7nZ!^i}I`hNpS>DoLXY+nvS;_Qw+ic+;MaviAw{{n_URIa+E`2M+?DGYq zECr!^4@28dZTr^l_5XGEn$SOqD_i80Z?vp&?dzD%bMIUsZ9CA2y zVb^jW#@}@&KVy;?d2f?37hs&b!sx82ZT*_BYP#K1-cHQv<=3+?-CW@Ry@|El#i@lUBUlRhHnFvrs8MB;~?xXkQ#5Z^wB<_t&^&O_sr1W~-m+mi!0 z|F4@Yuz%jlQ&OBwR^R3^=i3>_Ra>69d%bwinfq@{jwpHae|h}K#Q4{_f3qGv+rCuj z?CZQ=kMCaG)Ov?iZNt60cV4>RJgo52a=LD;+S;I7n^IqjE?liEy2*QbisOF0bc^*{ zuN*nXcd=vocBYyIEABefGe-Q(Pv{dUk&IZ}-*+>?`IF*B0|^iNjQ89tvPDnvtaF=u zc6;jU(gmSq#+R#I6Min9*O+@f?t}Wb-!^?(eCvKs|G)Q$p4hgTB7GdUVwtvVSW{_i z8F+A!`m^hIQ-U@0ZtS?Zg6ZDhZ`|MabZ4IkS=gfD@nFy4FZ-JJeDCqj-1VXU9(#+V z*P@>rUu1Li{?IO(zG^~nh{+rkt|=O;Z!0gd*~eDEP}FIC?qw*?+02$Mr}&->NoxT= zqu9GXb&r@nXMCI@7@s`ljd4PH!Hvt2OyyoR$+C664<>~f?r7*@T$RFTv~QxX<@e1Q z2MTnW_E~f*IOKJH=&ujIb#%sO*I5dM^-r}+r7h#vy}W1jaAA0va%V^ zjp^s=lKQud_mz#o>`F?SRe86oCX^qk{&V4>*{>(h zSW7RQ?%u5068gFGwcpWa3C{D9p(Kx2*hv1Ve5lhuzBWP&^@Q(e%>%X6m}rt?w3C zFa^c(=%~*OQx9|JvD~O)R&~2g>~G@JZsYZ8tvdZ%+g4rr^KF~4g3#K_J)09G^@GBv z7#`DH+Q$9$#_H~u{`DaVI|TntFM0KJ(G3UAzJnG=vO_=F7%0qPeQu=N=d~oBP08_g z&+EfGgH`t4oOx-={;8}+b588#nXSLn%FF!V664sS3!PUSdb-Q4X6dc?y3b{@EI)I8 zsLRim_s&n!V;A2}G>mBMd^DraZ|3LuI}N7qxu0~l>|N_`tM1Z=@2AEx*MDCmbm;m& zGwnC;!kOiGLVp(BI6l2)vbob=)$=dJ4@&$f<@DYX!TD#&dfwZwHZdLxy0H1${mqr? z_ue|Eaz~t)-Ms0K?&P?}%|8w)D9_06+F-EMjNu3OGnXHml-iRR6J}Ux`CF>|I#z#f zJxidwz0>k_eggG=;YatcZJ*AtCb95F{qj{FA{W`p+23z}>|kV^=q#ml){S}onm_x6 zR>=Il$6{V=%<-7{fQa8Kbq9?TYk4lqpLhAJz1zw^CUR!f^9m_XrnIf)KMwsgt_)hk z`0jFOq-DqamYgc1p07XF2bP&ec_!T7d%Ni3w?H3BJ;&p-WX>gUT1r+<+;H)L$=_Ei zxz*}7vi@yn2oK4bc5rEQo5X|_7Rp;Ew=XQ;-4b^Unrc z=6_d4ahn$pofquiU^SQZl=jN=fgE>4FQ$wA&B|(Rh+FmMWB0#HFQf$zEZX#a@_dH@9kf5seb{F^XG~tMgL_! z^HlXOTBTN8)lW0U#rK6aDOmi)?Wn>@qkWZc&b<6Yy+?^BY5V-=Kee}?+bKK$Oi^?6#wEY3B`gb8mizi& ze|z*x^xd3{4Lf7PCT{x_{U~wweXU#01uT6FRGxT<{CY{jhJc`lEKSNT2*w*)n^vYR$Z`Wf=PDY3084Q6Ay_E)6E3JMZ@m$HQj!7_L)=U zPS$YaO9rg7{D1uRb<|$;&ET$^O5;Mt|NsC0S9S|A1~vUa7`6nd6t+ZZvZ$Id4{V{5 zfwt!1$suYQyuD@&46-2g3=9knlRMR1>nkgQukNz{@cy3UtiDX|moZWr+X6Inr&P`C zdYdYyp~Tnmf>l)H;Kau}?avolRq3s8=dar75gH=Go)VBEFjG*Y`6yGXn1+ZIf61au z;aMg(Uw?P^d6Q?gBsuC+-Mr=fn`Xb+E97x)_w}mlyDg4MFe~ORxcsu<@iE>dOO`k+ zzkIWPx22gN3u9_(s)R*>f{>8VgLm)vX8NSPPZM`TO*As!hE0r|Qg zihOc59*zu(M-&4X9Ax4uDlFovUaJ21{rmA`f4hSllaD7DNW8hLVg zp=C?MV~Zoljy+nre4bNeq-0$E-%=?~g_b*F8Ria7Q3nfU9zA}{$ngK)_x<%B|NVad zcxL)MM;DitxzTPO@0B^2ii(N~9v))7v#a#h0 z*Mya!PlM9X#^mEkA0HhpD622qR=)XG3lqbm0INwS&lk$H@yqv3)ee8OzW%Sa z-M=5ndwx9X{_y?#`y*mPMZdMQv<`iHdpj^9BI9QBVskBB-J{#@*Kzan^M{0nCSG3V zE5O3&Z~xb1&YU@K?%K=_WN?saTD@93CMKrf=cm+=@bK~%&Q=Kr81{TP#QozzGyjI< z`r~{$5gQncX08#A_CE1eQ(2igH8pj||9`&=|Ns3Ca`d$6)7i_*%Pp*}AD=b9e`NLg zeOx;PA+0o3g@KB zlRv)Qe!t1EtzEuOLa6h|J>P6WR)!;qEiEkx|Ni`(VUj6SVintdetY3#w>LL7E*6e* z<9PX&sh)}H!m+?++M2&P?S6=vUGBCvZ|3gCb4JF9;=0} zEfa3LIyf-IRlha0`FzGWVt<|Ok|51z=jY!~Zf=-&Q_G1%afx_Qaj|nzkr5L!^T)I1 z_a7XWuWyN4zrB3(EEh%w#TJ1}opa{Q`S9=e`^EMD|J8$R{dih`|CbXA4NMFm{gW#z zD5Kg5+eK_n<1Kr8>+0er z1}+AV*)DEwk4~!3cPS_^u;|;pwvIz_iE*J!`-KYudp;hMp5V2#;{D$58_M6uRhyq~ z3S$&#Fmza5&$_$x^q2p-I}V@bF87(6^~<@k(%fWSb%CDVgP7QZ60ZXVtrVAh|NQae z!Q=AvJ?isogjR>IZ}U-euBq8`m%Ag7!Qp_GmO{-p$(H2f4DLEvhYKIH@iu>1yPZL^ zR)s@x?)l4?g+V3KzmNU(3YM0VLY*wh$NO5Z^xe1olIz}3|EES`|39fWhghq(AF2EH zrt!(m^u`8;>g_E8MYBJB{(QJwe_zMd)#1r+Z*84$GDX5Tt>^Bo%bAu8SzD#9uZwk# zijoQn3MzPa$1-Mr-QLUJtJkew?Oa%BxM|a-3G?Rl&6y*!;%XL0(}k7S3pLo7FECGE zQKIe0Q1$ZOla>kh8XsHKx3sn-c-J;b0VHTd)CG&td{WV%1Q~Vl8oJx*0MMr`1<<##_a2QZ|?7xpE`Bww92T@ zrGNhXDL8NYUBS>$u(7dm(d#>VDutuB6glO$P zfBuHBxFqMv)$5yrwbs@hmtbDC>OfJ^flZqZ_|0M1U9|M*=a2e(Sv-@HHt$dGVKg;O zFwbk4YtNtf+;72-9ben{Bp6N>pKsj%xbK>Lwi<`Qo-aZ-j&^UeZx?W~Vm$wugP;HS z{eR!gb>jEQ)c-skzai_YmW#W4d3E$H(Rv069=4eMb+WIouRlJ|wmPaNCpK1g-~WHr zIrsO;8mFB(aA#++ppZ~faHh*K2`_K&jhUC#a&B$mEG;cPYiFje&Msf`fpOC0$%fV6 za%?^x5f1owvGVHm>yIxj^|q+`vf}J|Lp3$F;%8?Zb8>X{?A`m;%0HQZ&UEn~ox%ry z{@C)XJ|ZlZH}QL(MEH8P(y;?z5`uX`q?5#3A zH`khd@x>dt*YoQy=dQh&;gXf5b!|;#^Pxjd7vGv+iTW~dN_M0dAaeF|Ss@bruzEfD;>W+s)bU|F) z1e=d63X>0hxa95N>U#XVqzuo=`~UeQdJXFS7zV`O|226_ltJ&ZpZAT~`piBlwtQho zOiZkJHZy&~oH;%HcE32z+yA#|HYrya+a2si>!Y%G%%c*Z!fRx z-q#YevL!~ZJ#2Mr)Y@Yg7uUOg{CHeGdeRq(WQ)>QBH!NM@4vY@-Mn({tXWc^6quQr zx#-EheYL+IO>~!YjERw%;-xxg&YT1jsfE{H2gb(s?%8AWqQt7Dy?y(ysHUc-52v)( zJ1h)fFz^))2@6xw)#d&D{r&M?X>*tL)yMl}JG#1@0s;b}?`^v6UTe!EUt7;IZ4y&~ z%!B9bdP+=_Rt6tzZT5=MkF5+|?lO1(`t|yt z6!+st#l<_V?d|M;{{9sd6JxWtw+9vE)9!ONHGqo5M@Kq+`N!LaSej|x!B=x+Ia*UZe7PwUGi zB_%cV^~>)tG#GB!oUWL2GwGM{d1f2_hkL&(L~iApRQ+zF&6LkySnN|cqN1iBT(wH! z&;9?M{6#Vf6C0%MWG+2P`S5IsL?5%fBxl_t7m4$8GAl(I3^y!N+HmE{x4nK&9GR^_ zD_!pIt1bNUB5=yosSD%x*U4JVcTsZm@!^?PuN&P~|L^nsi0yf@an*0P`eo&Q&f#O& z5~W-A?vCaz<>?+O5nD0@g@lA=)y{KKI@lv=91s!F;cx%R+(u;kCr&PHrVXk2l4 z-RjkcPfk`3yqNGKLTz&5rzaAP`pq5ds7q5HTX?7lftqUSesd1++y6Z882U0fO>Ton}+2QFV0K6mci zqHnLhR%L9B$}o|dGIgrs%8)~6&+_KwIE``)`P0+&us@lt*C=+T5})7n_M#T=Ffl}3G%OXk^K{{GnJ z^LE_+{^$KH(~EQX7|zYN@7L~~R`vDub$($rpA9d6mftNEulw`SeTun;{O;oCerIMF zGVj~BZ(6dcsi{TTn;9?acjo2g&9ExXx^1Vq(4=5;ldspU zS%2RTCZSFiPEJk({TKfX~d-)Y8@ttPc%kwYRrdQBg6G zWLdI&`Qsy3;F{`~mmd*XHM<=1TZaGs&z@qyNs25v5<)2FiTTlUMh zZ^#x`5foh3z{FsmBeBE!y-DhO>!hPjP9FUeXUyQ(8Plii-nU>^=T!fo zD&LYice`wciB!$ktKkVf4^K_imN3tg;ghrJFit-gako}nL?q$yv0e>L&4ZVgdQY4` zzn@py?7^9t#ur|eEGnL7UoSUZKmOR&)!`pMoz_2m{5bp9uU~KMELJajf3Nq`r}`p} zriL9mEL_~&qRvW7^058+a@qgK&*$@X51#q)@v&j%r6qpVZ`W+Q_RBARA;X-x)+KR| zx8**Nua}9rzi%1yxpNO5J!*KrpFJzvx+Kqu<0XfH8~OY^R?^G-ImUn5_kET}z*-DaY%NAekh}~WG z@We#rMGOiW8Yhf&FUX+PdVHUw>Wrvi_3l#q4+2*UPK< z&+B>cAmL)NWb%(Uo6i?~e&$;s(>_tz9n{dCc7OWRsfxKK-gppcZ5 zWR-8Je7;%Q#ADC5`w0x|;#jRD7#8f4)6f)@ee~yVf{LAK)09+IS@-|_HakNjPH%ehqa&TU=f23;)mTi^jc!}$-0q|_ z@j!yXgU{#fkEhSCRa@1&AV9;Q=0^djZPwA#)8eDnd@$hvs4dScZ5HvY{@vFq+jH}5 zyIWdVR)lD^w6-dmn_myAvs6}Q78Mnp7Q;B#uf6#BxrLiH2{pWYuV7)ZqxQ0jvExX>*rAt?S$MEmn%0ofoOq`eo;H zh1z$PA3yH0x}qP)^YHb0>!qi1XviO){40; zz9__CGEayhp?IES^poe$LEZGsuZ2c4c{)2gXP9P-#l*(mF6j-At8^{7u0Pf5qqy^p04JlBIGmMjJKU%zJ42v zu%x8qiL+<7zG)2z3Tn!}zAkcFu9c-_WR~d`$@9(gO*mr`7G!cR=;!}&A>c#uF@}5h z7PWJQuXpB>#hSg7v+vyXy`Hb#dMM=Y`J+pJPUUP7c;t}7#V}{K$FoaY9(!Ne zbL6a#(!>LoF9(~*+znCB&X#;|Zdr&HsL5P^W`^P7*w;3*Peli4h)9@ZOweF>aeI6I zWXHbUyRAdCrfM)~WE!a0B^}}T@#|OAG654Q-n##P%kNY^pF6=r<;2O8f@|*-l$PcO z-)rj#;X298zvY)fSOCLg>-PJLE`N9 zf1;D_DQ;ABHuk=yBGeiB;p&I?2M!-*HcmSua5dbb;)8-*^&3ME6`_>0w61B(d8EzO z9A?#?6&lX|_~K%NbrFnZe=2rczdW|X&091wQZmBZn}L9Uxn*vpWYU9 zsAYoQ!)MRfrg~l77wB>6Weyj^k|0esez~5@%l)JKo<4ngFu~x)9G^53slec1?N^>0&73Jyrl@e$Gkp5|dE&fzeIGs)tgc;_es0dqpVg8QJOV+%i4hx_-W*`O zysmbx#qo2d5lUPPYnCi|$>fyQlyGb2hMihbcXwQ5Q*_C>v!hVLIE|-b-urF4^uDE> zJO65B&O2TuB_$We1_h2C1rM2$l9F<&_W3P$4h)?5Vup{J@Kpwe3n_)O?$_1-UVE9F zv9hWv>D!x|i-HauZsT=MOH=#w=@Y|@xW~7)W`F#4J3spF*6$*WjEo)q{o=W$d$OgV9$Bj^%XXjo^QUIXk|iB|J9q4eDE(3U`&(;&fB%bh|BrMEYiMe| z{B%~3iHT{#lqoHxudhXZ;xjTb+ED+$?#vv^&C%a8^SVz?R##S4J-R99We>ERpF)bE+^C@*fV4|v1N9~3~P7?E~ z`3~)?{g&s|f4E#{Se4E)oBHAFi4>-&#CIM}RvJS#hUWA*oUX3uR( zmMc29bv*l4Y9$LY zU+i#EIylF&xPDFB*ST}&8rJ=(kg+T}Vw4&X9^Nir|3^@3Dpz4)q3o>>%S{(9TsXrZ zk!fAr-mH_6pyb3SYt>Tu`Psphj_p^1zr4K6zSynT$Edj> zQZ0#x+e%+O_gowH_{>aW4J|Fu@M?*b?5{5`nd9T*TUuHc)c^mtNOS+PdG{{`to2db zeCt4WgI)jG#5KqFo13vS*ql#(a$;gfcQ>dp&N$=Uo6DCjefV}e|8c8$+<^&-&MvZ5 zzP`MmzOH=zpTLkG^=u5FifcpL?-TzI?h*9Q$yuYyU~v8V{}+qmtpEDkvI4D-rn@T(Qo8z=>|U z;W5zAS1Y%;l7)rD;Wpmp<;&IeVt2K8PuGjAU65iV>DDjT`}^JQ^*3Kv)K^r1hN;re z&vSKjYIKw`26hbjcvKnn}6q-@Yr;9b$vK09{=F9{{AD< z`Fk37zuy=AC1UZ#jMBxIB1%h5SFK(>apJ_r_`090^J_kFF7uh$^z!m@=cPgQUvH#5 z%scw|!JRuk)h#B+Z%fDQuRAcqZ~{ZZd3*lE+h!7zCm+;{&DwPMnF!Zq%jnaopFSF@ z-tU!QJo)BkLjGRX$Gb`e_Wcoh^Xuy~;TFqvU(3v!yZiD!7gsfzu`&pCvP@R<1?3@7 zA-Y;ufSZeJmd?}cWdRxnWp5&MA~)49x!STNaIqVx+KSkoH#hai1s#3;^OIKIk0>lO z~1BKTFK?xyp-+4>R-IFv!cx*T1>5^K#bO%HltN{(vfumBGtzh4i@T zO;^^`V9*rQZEN_^*vq?wt8XMYHJY&3rowymzS3AD0tYkRAf`s+GC(m zm8`3)#GVVibzB&rprgaHZr!?R2hN;5%j?|E_jXdysaU_|%^!ZhH>mq(*&<)h@+O3* zYs;E_`|9QP|NE6axzXxF`o{GaE(CnA>HqVmMnps;qL=5|wQEY2mXb2Z6W`t0sq6PJ z(?n`b)K)I{ez~o${#4Fu@9*ai(GuOWXU~GGS)ie)MT?X|!or*cSkBBcJ^f6RePu{i zX14;z**<|Q*RCa5D zerrC*qVQ1Gd4`=aeJfUI^vKyt6&4n*ReNai?8%cBP^D{SC6#_|&OuNg?$3|HJ9{dF zzgYPD`+s?Ty`CRrOiqrDle6=~y9pA<@AS2u7G-8)ijo!M=jJ~A=xDd=;a4{2AHTY~ zI&3imS1S`}U?HnM+2*{mj*icisrCgA4ydiypXwzjD5zMKQ*6n>*39_x=g&p@*B(4b zIPK<=d-O}y<*52%m(|yqx94sBX*2Ef-90z6T+VNaTCj+z-YHG(#(DeM+nCRtHN7gE z;a6)L5clu-PF}VJ!rz~%>7G7)`fwYsv`%%%j6#`q!{lQL_x4nN`2M|p^=fU4zGk<@ zj=8zIbw3`m=iJ`r`$cALSoiyVzxkZo`CLH*JJ*wY93*?U_r@*Z*Pvaj@Cd`>92KzK2|` z$f13;3A%R^UX=*&$+b+F_3OFeJoZ0-54yX~Ja_xUv-al)`{j5~)?2?nP;lN>w|@PM z+qIw7c-HUdyOzPZd#Zib;xA@z^|;k1dI(5JbR0O~;4|OuZqUp1yLL%g&E@l2D&)0P z$j{HuEObjY-}%oR92^b~4h+kDXR|dQY`AqRDkMDoaDu@C|M_+W<>l<>Kc9_rUm5b~ zqPzT7t9wziE&2}Ee!qJ?X`fU5;p{arJ2zeZ?7cMT%q&yx&1q+|zRuCp)0<(O&iCxb zgegH$QC*;J)29Dtt^e~%7%aH@+^lu39dGU9m3@=SKOSOW5L5GXaSY*0*&;Q+R_s{% z{H-4!ZA(AD>5R`6|M_`;Yc4%yaXfIWSK9TrFT=9)IZjSY)AZx>LZ2U=V_;tKp+2Fa zvhrcK{yu@f`<5(WxnKKT_R!(Ovw0Ti$LwfW8@=7^=Fj!9yTzE<`EtG$%gV|1WnNyk z_2q629i1~-&o5tG?4JDX%}wwe+STy*&N9Bx)mJ?{Jso3WV)~_3w`kAhd-K5Yaz-4p zTYkRBzu)^6D}QAc@OrH)Zf0+mwiME_t>1gIZ`Rf8&Np_8ALF;*@@wwI#|L=jcy4}s z`#`FW{gaK#zg6{b+m@r#j^Eo= z%I)4S=UQJ?wd?7O-Mg)YgoJYZBXVluj`>J#<*m26w9(8i+}l2PuV4Ds$xwPiThldqEpH82>=2hmgLK!igh=!fT&%54DVRbz4^z`(?e}5`hKbPordvkO1 z@hoK{etv#Xp;dmTu-$k5(!NI)ps|GF=jU8~e0r{0Uar4&?9wmXg-^Y5T16E%>Yj^M6nS*lSjNjhm>}m5q zcZy_w9C10&RwJ;w_}|C7M~#dY%(oCc^ziVb;?FiJ#m}SiO`bJa?KH4?>DQ#NV8aFh z(3t$Qv$I{}d+HZ1T)0i$sBJpe%Y@&c<1%pI6H9 z+x<}3vu96K?Y{1A?njRw7rwZlcu~Zuu+Z?*<;#MyvbwhdjrH~UU%q^~NJ^Z+pzcq_ z>fK-7EeO#Pl#}bbU;jV$%EPd*ux&eLEm*aRYvsz7D^{=Gyyl*vL_N=rvbRz}!NG-3 zPl-Nx@?^oDJu>Uot>fV3b&ZLUS>`)CEi$#>^)=m?Jr#n}_2b>j%FH%x-n=ku^~1-H zlV|?&_VE#Us&wV*)rFfjT`~%a4i0W+WM(t@eJ?aT{PCTg#fvn<^YimTUE$)-U0mFX zb1WF`6c0}ENPqVJ{zt7UOUrtLJwMe_?$_(D(mKT}#iC~?mvenx-`xGv7c(rFzG~gN zWAkjQL1S%8mM`yaQ)GOwuwAaHzrWx2`6DN1#mc{>9-B7HZf>(X&iMBiJ>{CF8_m}1)(q;hu=C5gtHZuC~zW#somUi!?BqcUpsVmE_XLE-|Wc+%U zDZo&mrza~vHJN8e5$mB&;R!0Le=GBr3e}{j^oTG_xaYC_a^mxIb5EG4Up3iK@iFO6 z>Gjxc^L-XF*r)L%92PIHt+g$&lASYmZgos2y>fZro{u}jM zdd{DT5&dvuWAdVlEnQt)8C#>g85B5}xVX612sMOiiT?WjK7V`5?njT3RzED0m!BVI zFYUHiut`C|*jV_y-EW;!r%%sbo60c5?)~lU`RC;xw ziiDz_GI4QnCr+Gj@NHJP8NA$Y;hsHrUZowZPcV3JX{on{rsl>c(K~kRP%ttI5}8se z6(0Wj+`h?WVyUXC3BSKFPFDLH>Fl7#zM>)c`4ff-@2+ppkLQsx@i6&+qEJSz{!gLJ zw;Rc^VL$V)uZx{wmMeAX(j^6sL%Cm_=k9-NWNbXUX8*0IhSPg;))!s~h>X*eLO#;QJAUtfQ>=>3zYP8|Xb_X-JxYE8{}wMex|;lZt~*}7%_(%70A-TPz?HZrqE zz4>$Y)vH$pPfiGmh>0;7)TW!<6AW1!7Ol10EF>_nv8RXU;_tVQA2&Zd++Oav;i$U{ zL$BG? zbASK-o8NC;ufHzt+{Sb8!^6W5U%l#@tnUBl#l^*otpA)j;}f&9h;`w@g$0F$2XAal zE_`xA(EE5?czFBco}66N@^I25p_n}t7eA% z+r&`z$mMzNvc(q}djI_``1h}2zWjvQx8Gd02oLuV(__gFF%|mywO*j^7w4M~56@-J zpBun%p_^CEM#4D#oJ&Q8gw*Oew^p>2bRP?KU<@`r_05O7vZ5k_6}0+c>(;H) zs{TFe+iqxXzTM2%a?4fFQlRqka)xEr1zV$fFI@_X*;{orYu&w*Cpjmp`EI%;ywp_Y zxT2C06KJA#(V|52nP%Xr-u3mVPjdHC>Qz{-#vRbRCjE_1l}`SGob-{05K!Ew6S$k^ERr&!QRk;ljT z^Xnx`=CRk*9-R4lhJ`irym=4Wd71YYEL^Emz^r2~e(2)jx!b;)U(Ps?SAVzZ#qEo? zrRPkazVez!aZIdYW|riRiyY3mx;=9xm)u}vW!O6-M(_Bir>Cp`x!FJ2cG*H$-eI@! z$5~N_vJE8Gth*R;GJdAdwWZ2&yxXtlb?$7g`SGy*_8v=tRJLi`iFmv&cq8g@xb#&xWOk&5G*_3lDC|yu2azwpn1%hhAxOo%gW;0RoE_Edr$= z$&=c8ddGf#el92{$Jh6G!R40|F0J^`yL-3wv0mxx%dV?d^}c&I!S1Wb4QBos&whz= z6~x6kI56xurp=Xix2f>;HT!P1Qntkxtq$Hix$+iKhEg_G-rcf82h^6T7{>xXyLTK8mMXX+KQ zvu-LbW4r$2zDI3sd)D`2Ma2hGybM6gkNQ+hOs>3&sI0W)VRN4S{>0g<&eA~|B8&Xa z9X-1A4N6>pgdS^#MN(28EU_k-?zB))VK?=NAXgTgKF2SNrS0 zR>#bU&qih|L$pr$gs;AOA!}>wwx8$ze!rh@xxF$3)FWF|zxCsC|M|zx+yC#`{eIu> z@Fx#iT3Qs8l$fUL$8YQ5dGh4RgI8BqOIQ>v2(4{DapFYUcZH3T$;)=<7C%TbX!v`P zaq}i4j^E#An*GmAn|__IxOD41w;iiaZMn2yXA^q)W>f>K&F}Ubte#39Wt2prLtHxBXJ)h56FS35EzyFWW#fuj?`1syUDl1X@9LM|n+gs`u*PPuRx#hDa9Cj^zvn4PfyQ9b6EDq^~cx$6@7kg?(1#! zWhE`ES8J>J&(pCA3k^*T717YsJNInTBU_>K&9e;{YjW27sECwoxwh>`DT4@ufYYti zLoJ+Pt@RSg5&P@*rp~SO2K5tDPm8U7TKD(&_v0Oc%4^!@+S=F{q@EI4?Xq3jz3;)B zo15R(h~^&uY?G9nY*_Ura|ussBAY^W=QVWPzZd1D7tnxxP)V36zj_dn^t5bG`oW^s7NvKrO7Z=J$`R z3|FGuWx@pA3x6@^Z(z2HlD^OPp2#1-pXC` zSjkyb)S&Ly%DGW@HwW-BC@zVun0Nfg$Hx zR)&1|`nC09hD%|gA!s4)j}M8fy{cr6C!U(3Im4nbY4i1xva(~!{WeYe|Nniz==t=i zQy)H^9)Ik`#l?mR2N+mbSsk^eIxW8Fur^HD-dJ-i%i(tZ=-l`-pKaLqy0AKM{5bpP=jYvBU0E+(zU*4Q zud=donstr1xVXWwHDSCnKHBupG){lC@wi;JX=mP>8yhQrJZ#@spLm$<;@v%Nix2+$ z`}<(woYd`JYs0dy?yUWC(S2IgZ`p@$-t@4`RWKaylNFwN*4NkfXd^RwL42@kO7)Br zCp=nOTO*$)R#a6fSy}B$=l_2?<-y}e4AGNCva=HYyin#S`JR8#_;?wI;$EinpFxYY zyY=_oc;sdnwRYRhbMbo9lk4Bz*%`5`WTh`>deKgqew$Ansk3h%lSsahwROeHl^b9E zVFb1HcbC7<`)ajK!XSY`uI7W|MZ>*u{r>iUx14-Csah&L{JPq9zVF|^yZiX`$VOc; zji|KbdB2za@bCBEe?79D+X+gr?H(#ZE^cl~cXyRW?5QxEGiT1MHSg~W9Ir2YP$>t> zH>aj*e|)p~{I<;bmHhntI+2@Lw8Pdo?63Qqvv1zcwQ_QDVY{P>WYWK0mYVn6u>N07 zOIuslyzt1#PS9ZS51Z)qSFb$2_36hnIqiuaYnt>vS6^9pi|gAR!zqTB`xR!Cy|u0o z`1<`0^W)k1Olw(FdvDjyeqZAi%uwGl!B27e>5Ua1LCZjxH8nLoe0^O#JvqNr*}i)H z`e30@9%r?;6YlpO2*bG z8LJYHk~1eya{m80|NqT5KckmkclPjS-=01Fsle)W?+VV=l>N2p+5d-W-BL}ZV+Zo? zGOb?2a{3(G?ak~uws)VEZQh)&X!654-8{d3o1Dp)B$>r$?|Mq-+}_3uT1XkTHmdc; zvFvPFj$|JB+Ao6pe!bFub9eXlaFv=;b4LNtXlFENeH3%w;|I^4v6YvXuQvTRcj>xy z=Qd3>NO*N%s#ofEQO;Mnmo8oUaHsfu=d$!ehYx@J`F#HEE&G{HpK}zWdt_)tjhD+(AO}}fm z*w)NS6>%x9m%qLW1O+$d=^VdvxzNw=!E!&=ihb_tLO~1;9EyAk_U)59-Y0vwQ&@eC zmZ+8pm(L7?g-1^{oPF5TthljoUDV=>51yT!9iTP!u9ayD|C|1+9JZEc`=&7pIdw2k zIQ_Jvv$Ju3{r>$gcC22lz07xZoBI5kO^1(8j4J;2CUQ;8PNC)g^VxncF5YvuzRF07 zQK3bEePPf_0a4M`MT?XemZ>&`jDZ58&`^Pm4b!?xNi?fksl;HAz* z)r(~qE~ypmlrfsgiCf}CtWtd?0^8SANrlzJ1 znU~dGlydCgyu4*XTqhI5B{gO?o&$3%i#HTLb_2D}3yO+1 zl}~CFFzbqjh`6}<&bzCW4W`e2dwDs3#k{mj z9EyASSs9j`oik^SfPg?lLjyy#d8>evhxNs^+=?v%j~YO2K>;Tj0Z?*q>R?{P&>-N% zQRn~)5e`MZ4p6Ks9$^dwMPQ4C0vAI)hhmEWJ7i%#>ADGO_SpWwf{!5}JUspRv@bqA zh7!GQ1v2dO-v4?RdFyGoLW@AX6GKZ|+o8jUnfL$s#NEat30g2$ZNBn&LeIi2TSUUw z#Wb#8ufJ#S-rav@$t3rb{tV?%Y_YJ}8`uBjNlMP$U7~q;c^^K0+*m%jil>-?fdRDp zBkk<0t52^l0eRhHJE$el$}L{-?2KfFiBz@ua`%^)m-Dyr$#&hi5uqVc-=()-4jfp! zM3Z@}=JGWu+}N#Y+ZCeK>Z7*#iC2j!sND|=qSc`72UHI#?G^#dXnOIV*~vUcTF-QK QI%uz&r>mdKI;Vst0C1=LrT_o{ delta 23400 zcmbQZi}C+X#(IYT-j3W191MBIMFGm5d1*gX8Ni@G4NNhtfbth2=F{Y0=HE zrHmInnwa>mv*mz-kYQL2M}5wz6Ov~xolG%iGP~>2m(b*PT*LRN`0QW7i>B=}X;V8M zvrlkovge5t$!z!L&2V#>Qp3NBO^p35+Xhkb_8B)WocOP6D)5-mCgo4M<6=ge6rF?4 zZa?IoBvrp#k*KZ5chw<VVjhYhV~wX|K&Sw%(H&A{fuh%|M;wWE$-l- z|EBF!`T2kOzi2n>%m4ka?^ycjuWt9rAOC&-PPUuoKV52C_zVtV*CqezKUjaR>PnyZ zPkz$FFAU!&q}^EeKelek$JyPF!W|S&N$6?SNHElLRc>!qerj*9-mIT-plrO{k9q0*q0MOm(GMrodH%Qls8rGCm3 z-4?Y{jB~5e(FWU=HkCYshU1TBdM2j)7k6ryw~%XsTGxX^oXpEM^dz)#neJHBW3=kQ zh6d&nXX`ta>g8a-!LD=bvvFg4o1#MV&qzxG+gtpDOC zKm7k)5h3;cS3M7JGs79<)gjA-xTY%2JY<|6S~2gbX3o_5khC4ARtJSni+U!!LqzGw z4Q0+93K|a&o&G=V)BdOR-|xq-apu!o_M?gEue4Ce#$8P=H|y)=<1H-|R{c28bFp33 zDNMU+i9ygG?gN!I8}=M>xxxMR(CT%I=EVNvfAwGY#F8U5r{>Iz*&4FYXZ04YV98Q0 z*NH7$p+>3`ubf&s+drqi$Pi@lr}Fah|Mh}NZx|W0b(vM?NmTDiT)JK-@W_qF?VP4n z8jO6&{E9q7i;?BDv-|Fv)Rhm8S? zZ{Gctf2C&rAA5#hPcmI6X-c0ayu z;Ry!?<8(8{|FI{oEffBHpE2#vCY}0&TZQNB^!oI_e#_3}_>X5cX8!+vF+I-z z|9`i*InVr!>#K(U{HIrPS?R3Lc)4NDt@;Cpi*`2E|9p4g(T>Oe>u1Z%wfOSSZkn{^ zCaKk?HT8!t?fCxR29yDQ{N+8pA-DD4{SD%4-u|q=ak?h*-{y+CYmCbn4$Qt$mh&wjR!GHeQ8Rz`|FY>E?`ty_D{?u2_(m&nt zc-<=I?)tC@-T&>Y9{dPok-Oriz%=c=V9Kr?hLK4*jotcLSw07undMd}u!>d0DkQ~R zY2!_z(-?9=$1Hd&PROKlfMqrvA6zy3t_j{N75vA7!nFn&vjo zf6%b~?Ge2j7Gb>y!%A~>ZoJCA-FW`sHi{GX+obBDQ< z|J^L+KFdG<>x1)m2Gl=)AkAaeo4)VC|9Ji$P> zQuB@46gBz(-APOR3rf1xi;up4FLwUQU;7pRzpw0mIPd*?Y2oRg|LfN#KPozMd-c15 zN7MiCANG3OC1HI+Q8aPo#A7~!Hd|)geGqqd!bSFZ|Noyb24!RwoBCDB_2(LD{s+sJ z|CQdh*^<%!(UQOe^(*<#zWE<(b+s`3?oqhN1zjtZOtZD22m-oI@TKMDs z|LT-YkLzczlh$$dTB5Eu(sTcM3ag_Iy0!%C&ooEaENt9!2(sZcJV)!T#o& z?lH#y|IchKu+^FOf67iX39<9~#i7zZ*NY1HY<$le9^G?jn_v`I<*#Wrip%$=^V+<% zy(n4#Z~nJQxo7I-wp`C~yk9@-Ztmau8+z@}>YvZ;efEF%Qxt34N`pNvNmkl)k+soiF9*r#JY{Ac3CZ~Tf8CW+<&)3 z%NeJnTc4h2dz!T4%avH)k4yNDi@3|%l#3^IZoK8BGhc>JT&dJz=Y{%<kaCbI^v#4M?wS39Z$%AN%Z|-(V8F6<>SQ^t|qZpI=mry~Nfn;S%xezyoaZTY~$=<>h*O|$0ChhIPP6>R(DuC+2s z`?5{oQZLa(9?!IV=B^6vu<|xKvucXwVy(4x1_!>ozWnI7Db#AFS5W7vRU1^+hAr%@ zSMpEf-g-)rGjvjzMyO>=8~>@vbN|&#+DX_2t;$&&uxMtg&uW9IN}-BbtyZl{o~pw0 zinJ%IPdX#r{*{?U*~}PR4i?oE?muwrpcChhT=|$yDpONdw+2nrHPQ)O7V5DsXnBCz zvk6s~+~)IKQ)lk-Pz~it8kaPa zf7z+qJ}HSPDtMh+GPz;i2Bt3g#nQ^m0#z57Sls5yrztRXDDGH%mUV&e3Jys=GnuLF zL2Xl%PG0iTnVOcN6r8%^)au}fWv4`S---G&E)U=lfAxcTfAnT3tCpX zENIn+Gpj>0=7lVu;BsC8$_8)F(6SX9OoJrl zOeZaHcJCE;U*O%NagZhK!VXo%PTjg2M;Grt#^XQBqkkeNYn=n*q~q>V90x3tkN0uS zS-(!?(*Ml+|CR68*GupF@3!y%s`tga8l=g7e?ad&@WdfB$OoT}MIJ=hy3nQbWRKdZ-$! zPPRI^;mj%{9nVLaAuFR+eo&dFJ4tJCUEQiXcQ&qn_5XBFZ@FPI@5Y+*vn$6{ z!N+`Rsy}XAe9V%&;^CtO&2p3X7u9=--&ij*i}%OJIR}_h5^oeJ-`^qjtNv5($r*-c zS2lSFPL^J*>pzphPvBmmPvu1A4yj2J(KZFv3J+a)&V`3_{rqPaANqa&$?rA)_vc9I z&lCJF<9n1b(^}f|+x=zho;3cq*Dg1F@!x#*(<2N1>loiU$NKkwnn4UtR{q8b|CjWC zt6#OYq9iw1uv0qEAn{m_VW82bUuRxx&hy+=Qyyg4#t&-xJ73xmGrQ}*d~l-rg1YBd z+vc76ULKdb`@j9A`dK$4{`{ZfcC|N;g`=Az=jhR;66cSp_#6GJm;UuXGVNLnPu1$9 zi3i`m|1;~|f9aR(A3uLKj!u+#VShTOHPg_ZXJY;Tc;BAj-hY3~H`;!$Kb)ZV?SJ^+ z{joXx?_O@#_PzLD_VmZRkc_gD%$EjI4-^;QNxXcid)cFwfAZVx5*`2hzuuXB=DWEL z`{e)YU+uqK@%&%C)cY8%`}a*m)&Ks#;_WT9f2T%-M?v62*II`cJym-)=RbR^Z8^X>CoG;i?eoX|1IBm@@|Pk{ZTFMe|uZb&Ju2(vnNYzX~C*p_g;kFoI9sf z%&PM0oo8h=8KGgW|NlFb7uZzI-);7;W?SLQ^I2=U?%y|FBmC)qVEy0C0StVRO2sq( z$8VcqlzM84rZ3C;n`>(R?`OB26tS=WX`A@j=5wk#y$`+SZ!qb$DZ3y!E$!oP>*u21 z|69G7;~)0NSN7XA=IHqcY<>S-KVX?X@xOU?+|Ms=nI_x+pLaZ8wmag_|BF}l7@huZ z?lt@9|NiE4>ZiLD!%{x}G2d{IKf}A=PyHmFSw9xB#Wvipm%so2`|ta8o@*jz9{hhf zf$e|3E1Qm-uKF|~%>@M;UH|LM`yaaDfFfJem;V>uq}-hM-@n>w&bur0WxVY`VZl>gZH>HdJ^;HYET#UB)&XzO1``m%LtuJ{_Ja+wgo+h(L`kHc!A872`?|fisDe181OrsdJ4t~mvM;Vk9F)w`~5(@v^>sAOR)(s*BVo#rQbj#bC37MrQF zuQ#9L<#PPwr5A4hi-%kN!|L0LJQZk*~v2h?W5;!SsZ(=@9%x5(?{2= z%ub#)i}`p#s($ZO+amRAjcax1xc~p}m0G_x(DHJze{h=F%7ssNZa(U!`P-`ZNByTc zOZ`<6SsR5|POgo0cI|7KBx0!~)am5e#lqR7a_{ED6>bIF!>_z6+Bmy9_rok#r9chO z35*AFPXs^S#@@c;w)8YFSFa^XtFLuM=9pwQ zeen6a z_|JcBr@>UUZL+h|*G+o!U+0CmllRSopRM>;my8O z=j2}YZP~?6$-V5$j_njUt-gAr)>HfLoymz8>Zjf0rBZvRdNtWhfmn%~}Km6AVcy&ke!uzD9 zS9Y2$_^MW?rO&FJ{J;oO5CfB%PN z&;I-V22bDra;xUY|NR}$hggWt-=)thlPt`m7aa5H2ZNN*1ZRG0j)%IfHKCIXJ(qi~ zI~OcEDd>roiR)F(DV|e{JQi&_qq#iFB=HCrL%!Yr+ojFfLC0iVIXez8vQ7GNU!&fl zk+aIHVezL8r&PUMPpKG9H%fdNv+_`or%mhHtDRioDwn!SW`(FoFrE%)ojOm$C1G)c ze(9+xnx9IH{3nX6@>ynNvbmBgHRW8$j6kI>mvu!PD}F4Spwz|J@E~MWN|4Ibn@XW; z1xin)WULEY9#Gn4dB!O1)Dj_M5%~`Xgk=rtC;F+sblK(=7}KhHX$#j;(UqxfNqQ62 zu67=2~WwO0EWrJQoH zI@z*h-ZAnH@;W9mYuRR=z-5(O-l~&)JdC`T z`FMxK9;j#X3u##DCH$Gq+of^QraKvrnRSwd1AO;TVTW#6mThMeM9q9J+aN z&m$*qOIR|KZ{bmas4YnK{!-v5u zw~pjZ=}r7#sv+H7JH>Z#&Hwm-Y1g`c_A@`7yK6??$yp{Z>m|Fk{kLBlZ8KB*TfN$^ z|14gMC&_B37A$n(USr(Ez}O^~x5c4>-$x>>&tc;+4mZv@3XIwn9t#*IwtO<<+Q=$2 zL(A>Jsxwu`BAnO?lBdp6R@=ZR?^iLwf#uioNoQ8N1}*hi87z5ff>-GO`%R0s#GU#7 zUFd(d8*}#0`|BmMj@eAE6}3uqdHnmnMQqHx&HvYC-l)4p8CI%)14>uGNl)q@<> z*{5HAchBPQ|4r8lHrHBY?Z5JTq5ZsBCVlnV8VslT7HKR$d-qD|Gy5JsYrk)7&A)s; z&tm2);0r5_iG6L)r?&TYp3GzkF-6ApldZ&*mFzp_8y~;tu)Z)}_eTkPyNG=RpW5&D zs=Up~K|YBo|L$kVPp%get6w1`XTIjcEyKl4+O5IvtN%Yg@mHR8vvcbFr_9b~R^N^b zrG7oGm?0#>S;exI(X?sZFP-`AMgQ)vT0h~IVzFp|<{UfO6)#UWPW$!WUL;`S|NE*f zlM*f{^)jn6IUTl(a{mASqM%fB(+hzqY#|=U_f%X-`O>!ZF`v4hQI$4Y8L5yzt8yM_Y3i;DNp~u zb^8B5GU~;Y*1%c+%^3>*y!dZ_;<$JJzw_tM_jC786OAaV(K2~4_4j)bsV}GgN4NLM z{IzH2(rEdmGhcv*hfhY~!s+gAZ*Oky)89+~=c|fjujf4B`s_~qetW*an{zMx*N&|A zoxD?AKJ0(*KmF6%+}~^de?E2W|Lr?Zzy6z_qTJM~{%UPAs&O611) zf@6)3-~6w?ezw!;J#X4W`_uK)OvF?Fezfn^?|uK@f0ORy8R6W=ovOCSYDPuMO#Nox zvaB%wi+s=fGaJ{eXi-sk$dS7tk!AkB?Z*W686V%I#Jk+)_kO8b!!C!GwxxyVj|==d zqrY*@kH5Z$HP8PScluwid0+aD@MdeRytDruO;+W}*4AGxIP>}coyr?CA-gHtoG@FnO+|Oud`T`Z%h1?- zm+p;O?^ORcPvH64Sjf0=bGMDFVwgojOY^NB_AIybDLG1f4;MEWS?xZd_UM28|0_oe z&i|k9RRwBl+40CZz}cE}L-8IF__K+UMry&!0a#-`;EVLesOlG41y#G4j1v+nk3#r3{7`t(8{&zx(TdTFKa-k<*_=kirg>w9%* z&6xzZ+I;ni3VA*5j9VM{L`)L?-f3WUeN%AxSjElGZ>;P|3APKnxItnIl22=MPUk-n z^j|c3skuV7>9!DlIuO|m~2{}E=d*Ia(+N|Rv0A})uu`lX*xaiPv{(2o2 z7sg=$ey!O+Yc;H zk<`98}Q_ z&hgbh?lJvs6lMF?$(HltwzV$Peup=A>V|V>bO_oNtzkW&;&biXhpw(8_dC^|yV|hK zab0j-bc7HhL%1^OYv@~!(Z|KDF-aYH) zrOH{moCMz8II1^qk5i43&gast=VKcIU z3Cq@rxf$@TU+L)_6>cF_mQ+3I*Ut(z#qcI|#b z$fGA07dn3pvp>a_a=_?#YMz_h&V>K71xli?FV5a}X5(?T=5Q|sC(BuHXYK00(cwCM z*ThT-wgq;r`x$iW)n$0*1}|9Yttwx@b81QLN5dBJP0LjdIK?gFxyHsRrr`Pes{YZY zoNjrR0}I(g`qnK>-oxvE*tU+Z|6;6BD~t0y)~o-$7hes$ z8ZPwh_WPqZ{i)@ISsOo{-?Zo2rww72<_9I0A6??o@$?_R`MuWSdGqTdj_W$cI;>n@ zI*IQ^3U`XHB5VA~J8MJs-^`tF5xU^&H>;I~D(r8)s>MSrYBv?`U+AuyEGM;u&Hd2r zea{1ev+Fay?%&rCHQQpxsX(KqW`X4=CRS!DrT+Cx>%J2!TMLhG&8Y<& z-n&meZ>qK{<#Z&I;Dt`D%Q2S^*GJ`F+T?Ll?&P%}513BaaepuPzvALD?SI~yHYpd1 zRh~twTlu+({@?rj#g+GFb6ee+RV5^hIIr)#Vwd^eCEet~Wepasy0czc+)HX^<*knX z_4(V6O(k=iQv0S~6uzYKqwP}8<`xO&h0EJ4EhS>DSC#xz*nP`y{)H3Ix^?fHR2sMnDvp7 zJMdRaWA?AzHmhWtN+V5YA8VLszA9+@Bc*x?-oW0V@@XYO4-T#ksq%a``|jt>A1k9C zT~I#GDz3ab#>VuIa;?h4(1!OnEJX|LgxzvnJ?dX6UoX7YpIi7UjZeyoZR&@7E6$Iv zCD->%bXbx6-txmUUtCc&e!am#e;D^_nxI{2@jm{!j8dgjvB z-NwQXx(;JFrjMT}o)YK5%vYW4EQ*-27m!moC9g|9| zU%N#n3Z8Tm6>EsAFZj)}Z4X$`=-A=y3o-@uA?LHV)5#0!FyKUo9xWKkyYLN@`~oTW#^^|Y@XF0 zJzd+(VD_($MBg0yev^4C?xiJ0PiL*xuCH3Y`H#cOv_BIW!W=E{E>k@inpXGL{f_%B z=~~-_uaePQ?p5v$?0&4`yRpqSZRy`Oi6)yy6+g~N)^y6O}zV29JZXG=9 z)AmgcZNfXJCw^9SaKGql@bb+h8ADcvj?PDO`?|QZ)=OSYald*Xdf}qpSEoDvtjR91 z)?w2&2$Z>eQ8W0j@ui1V-_~o~jXs==?;bk^wDLsn92S&AgCx0zPnys%MZO6S+_hu&N+-yM0ANqEt-1u_4o zneKD&wlO+kamGA$$D2t~3_qGIU27(vSZKi@v8L}-Fkj|=19|T7rYjfw7q4G^`pEpt zp&#{M?p0+t9wGBhrCvPp$gRik9{eqn@kmqC@pN8tGvmRkssFVZXD+^P{oq?3+rx?p zIgdBgF$u8loVxIb@~oVFJ6C@`qQo4`Ei+vy^}w2ceD_bqt9y5uo~zxZrG5OT=ZR_R z%8~ugrOvL~FqI|k?(Vf>x#C@#|Ak6o4nBM{ZNh9pG0xVu(BOshb|vvt*I&HPQ4s$k zM>b9>rGwGU)i}sHN#fTc?-whiSA}KlTE?gxH>pa@(@rWN#e{A4w|J3)lit>LEPuII zzt&kiWzw%TPS;|)>&hms*cPTRMM!1B{vZ00_k|g4R1e$zJej!L$kx=_S3!UIoL#c> zg=ZfP`V#k3nUy74I8|!G215gJ0hxNS%!HSXOxa&QpB3-W3HtF#qpqBP+6LX8T#xf| zy&tXJt}x|rv5fzl~W5xm6mUstwCkPks{)m!2Z>O8Y{Lna@Fkhx0q? z_wOj$khS|v`-ZH4mA2RC?`Hj+`}2vp^tr01*X}<&(dfN5yJ}aS{<8b8(wv*mKHb!{ zW$Jw){@Z`UF3L427qF;mZvPa1?Fyf>_N1mOq1z^>aOOt`JlrAljb9@wFzed!IVzUL z*FKf)ieXN)4*cu*)9IX8eRRrBscY?X+AW)mJO0&ga&LGQGe@a8CGv#ARu_(f(8`0m zr}7tU{2KyGE{NXw;Z?Vue7LUUblK0wquDm;&^XgD##k6%TdV2A*m^SS$Ji8+1 z7t4;fOP$Tr6CxEeugOg0_OqBV(I8=co`Rmvj$N`Yn^R|_5COUS`0Pn!%*R!Lg``qw7fChphtAvn0HAq&MgLBEp8)gvZEvG5>2j4OLz^Q&bcq&!r8u%$H3sh zo%ydH-8%blO1fLxx!QR}_k9Fd7+Xx7LqkQ2ii-=sy@?D74UODYZ>e_i_16>U&$}li zC@fmE=)r>r3^RSwUZx2;bsRO|V(=(laWyOF<|fwJX1Pr+N{vnnHw|`P(x$jr~EA1C0R!5E;IdJdZy-R16TPBpbGcsgOvo3$9U~Vpc z-uAo9v0mx)m(OOLIeC)v_xJbhyLVeZJ3G7m>+9>ukB{|EIGNJX+1c3J+q-nO2Z!R@ zuFWOP9?utDe!1ZC%Lk7hCB0m>Wyb8;(qcLh4f|?;C*9puYEk(~B}1i0ASqXaK|#0v zki_N^;i#jB4>O;)|8Mi_%gg5af1l@Dl)sbtQnh#KY*!A&w{1)eNn3Z7zwcw^7F$sM z{@#z*@&CJ$kM}t)z8JA9vPIyMo(6+Ln68Ef2e-JM%lUb>AJ^Ca?QUXl>bSaqp&?@L zf&~gX@%v=v&71e*)#~*E;^OXy8Cxc}%ds-FEU2%tNVvHvwVKN`QUiF8OY$^AxsN%WyGptaNvAVfpv( z`~Jn7Hq{HgzP9%8RPFE#iw1#D9t#;7Ca`^ZcUSuTp3i(-TwEX4*Z<{y`SRtoJR!xF z8N93vEhb6^1_Gz2>$@i=D%$;czuh0K} z-*3IIECNn)oFTjuCa4)4Ge|nZ!Ej*t{JJiGyPquH)Ad~I0|F+rM!R`DKm6Io$jGSR z@2}F3kdTO7mb!;N+h}QPFAUHSkdy0++gsK6*y{dg6IOCyO*D&XX0XWRI zZrxh4c~&bELsETcl=0;GhHZ?@Yza?KO^w)BV|i_Te1GccX$LP~zTBg$^5kOa`+K%; zZf)hhdGlt(jsis{W@g>HOTWBFkDpGDZv!XAbR#Z? zj?fL8H#Y~ZJhC==dq7OgomKpq{c^S&a&8*ce7l)`V{f(jv-9)o`Dgm19lIv3L;EWo6_p&9uW3hppf5sc#Nl9hP{!Pj}mIzCCcXyqb9RjbfuRs3m>}*92re|knGIzQpy zznu8*&rjK^QU+FrBdbn4E&B1}$Ch6r8I(OGp1u{5rD7MU) z*xSo%_w$MHkKezO54Z7J6g+UKZ|7msU{Giga5{75=1s}<`+jMGa(e#0pKR@Xvabq6 zT{Xe7w;KBU`NedjSi-}@*Tn7?tN#A(ZGa;KE5nk=rsn34m%Q~4-o7o}v(f!kyMWV~ zPD8f%`1myu8<~PuimVP>duX9^yUtx5<|tMN21kY~SsbFGSO4!9w4FP*{-Uwfu3z?_ zKHpwg__kSdYC>6=0LwyW*B31l&YeGdmiPa^@B2SK>eheMt-r6qN9}Njpt8<9haM&d zriBX_{+Qoy=r8=$ngMy0pbf(zdQM*)ICb&;&Z)cyT8{MYA z?+4TCYipZ*)DB-*=&XB9ZZiu*lfr@(D>y7IEl-?3-yXHLZSi8|IdkSbc>S8&+SZ2;?2h60>Z-0j?HXaGd78y|Lo!Ab?ChP|DNLK=N4|+B4Tr1dfmZ7nGh|} zMT-|37Cv(M@%wl8y5N9-hWGn^^LcuD^2Qy$c~i39&d%<`w{LC%0RmNDUtL{!%4(K0 zH(%S6wc8JbXf3|a*~B0s>UiQr!-P3J5ARrVXlnM(>yO?iGbwPf)StZy91fY9E$sXX zF}vC7&dWXc_jj4}t;Osr>-R9NOFzHt``Uw_FX+beJv_#HxnHqG!1#d$pQEE=#p7P{ z54W<{Km2+<{&@YXtE&aY#IC)|%?xB%a5ZbyXer+R_AuUhetYv zC5+Q}=GA_S{PFwu>@Qo+eYSabe}Dh`{r}~vzrAS;Tb)~QzpINY=f(!c+2;AjetdlV z@zZJj!)MQ)z0kI0i;0Sg%8Wl!)!*K9W?o*Fcw<9iL1}4gjGp<=8}9Xt#bs>fX*@UH zzfVkVd6e|w_jl+2N5ut%gc~0vEw0_M!Jy#r@!2)C3s-Rba&AAcK8Ew*HQfc9HvKEJ zvS^UD61j7rQEk4o4ga=WX;y|7f$kNNvKz9lYVG-UE4$+V-|rufN#`r*=&Xt5<}lrT ze$S#sOvn3V56A!eG+izK-_}eMsrom!xAV)_{ZL#NzyID7pX_FFNn=UQ^k%X!4r2xdIpzx>A&o;Bh6pXc@qI7J;ef1Y2y_KRRnPEP%Y zqvG)oUR_=NHY#>YBx6I|`r|u`pKr*!Yh_dM!QqQ_rwfz5zCMqn5sR>z&w&O;<_FK7 zrF~ga|MZmTn#j#;N59&rUMu_j%okK9tX#SB5nHj$ox67hB_%swyvP96QV$z-LI-)^pwzmN0e{C_MncBQzjZJRLt`V0QWPIfI5_!JZr7`&(HG%9fX_;gyo zaItK)+H6yn33KQ6o}F#3Y;7I=h2{OLS6!CH&k`;ya4aY(Y3c3dy^>|>EnAXMYY>Kb@$$H-Ich27Hl|McxCM73NoG~NAf{!6YOBB@La%^UETpHxK zFu-B?<%O$OaV=fC)WhH3y|U8s*SEK~pJadh_U%~w|6k!Z{_gu0z5cqgzCQo+b8`=$ zoUH!w!$ar8`zk9dJ9>LvD=I8@mA}6?XZxFr3$HqTuYWpS;M^QW31%Cc165y}o^+}o zU}E}rqyO;%-x&<=P6&ROZEjv+rD!a?>+0(NHx5Y5m@Xdi|KEdO-&j9?`SUnDFYm#; zYQ8_~_dEUgdi~h#mDip3r*eOJecgOtd|=??{pu|f)Mnf=zAJG|!pX^LL&8BOP&{de zuRHSp@BRM=PoF+r&!d|&=Oj4Z8=2WBOqigs>*wwrJ0468k2|=d@UhEl$?SP2nwy#) zyn5Ajb?d%2mGh39->+$&F+*bUq*YA{4}N`pZBg@M!;7<_D?=(i9u+q%eRYLx{pEIZDPyL^o*SriYjEAOb zKiHnX{$G9{vjyLuRiP8s#POcIva&69=49a=69hi+&f!`i_xSnsHG=smJQF+){QBNr z^0%tMbaIjev$zaT-WiD(`tfX1zb|kywN!ssIC6aZn^GYs58)6k(Oo4kn=UW+SJu~G zfAu?Sr%Tg^4+T1LdnEqWIW4;2a9DwAct zv#-5i>Aa@LoYLc{HC4&VN-F*QyrT!3*?HG*-?T|+=FFKK{QT>`TiKjnyl7E9Q;*?| z8#g3wMTUm9PMRbX5)uMx_m`EGwW&_>T6(EK&ru*CG_>{8r=k^Cvz|PAcIfC)R+<8aY*)L_n zP8~B>hG@CCxjnkH)O&_~z1^Qbf6kay`U|kUxxZiDd%9lh1hv7PshMOP@#5VZL~R$oQ;IP?N^aM zfBrCp_^&u~;)H{rAD@@E_r&Sb*~`nzdt|M*-MccW{^Q4w7S-R@D9+fpef#k#n!zBw zoA<;YnrkhvY4d*$1_|cR9~oa?<8>-3I(@G4o6e8D)rv2#uC`jhaNz4}_HQ?muWtUn z)-YL7amitjtDY2D`pvbv`ee?wXD3ddW=FK+TlIhY(JkdZd>*~zu)7$4-mn)f@i-UR*QBhLca&9)Yw6I*<;C}Yg(KTyymMmZX zxYzui!}WEsfuW(T@pV6sZm?>6UfI&#u54j(LqO%qg$$D$d+V#s`{Zo1I2h!r-x&JL zv$>hGK2}59xb9N821E4iu({##(%kRr|BJ8jp8kq=+qQ&>6A#q?=bkck>#I900+*z` zmI@^w?>pKdsC;IZ*PHqO|ETN5?6{!kecqRk;nlfg>vbR~63=S{#o<3p5434c)cVC#E zn`7Dh`}_Or(qFDDyZo}?-=E4qKOXm2{QY`8F+Cj=lNnp1oTZJ8jVI2TlOw{Al9r~V ztbBOGd>19hj0_DM8=F}*vdfb6^c?Q)l&P!~c=hMD+4O>!LM?07*fTLO)USy)zZ}ji z;N&44tRWJhA(CbtUw!YjkCe&!%2jL6-n%Ckn^kLQ?y5Ff@~c?J)~Lk{3dY8lU))&t zcE;Tvk6uKjOU_l`xUsL+TF$0|q3?0MLBRuui*~NAt{z@qMqE}x!osfaH@3I4|N8be zd!M~gAM@+G(R;qdFJ}%4ZVY2%xRBL2N9Lm1vzMGEU-KV4U?_R5_V42-rYWbcZEY8D zl2cUs$^dG4E?MHTW$n(#$9oKK+`jF+GDIpbC~B=3znl%j`Oj_YudU19NyxsA%=Vm7 zzjMXate9OToZI%72aAh~KfbrOx}&@M@Yd|>3s##R)0uqIAnlCAk6*vcUP~Q3c(CH- z(&+{n7Zf~HgfdK|!2P&=wYyi8JL*kWHZc+T{QNwqZ~dgma!ZshsHgGg&*$@PwWU_F zdNDg1PEJ-|9KJs8!`H7z@9r+=Z9FwqJG_3v^y%pt%cIu11qDrdc)!n(Oy_ZM=k$I&o1MQXd~Hv0(M}oA z5XqxQj|$%3vpxK6!UTaeF*}34%sZ80W$w&(X_^%gEUH;@;|VgJYo9$&@I|U#UithYlU8@2L{^IaK)ISRd=>PpjgXUOZzv zcU}C>g@q1t_p{oWivBq{`RsnT+E;$-3JX^3t1H$Nxq9>elv8ctY64CjqK29bEB1!% zbF1Xu=c28x9T=&--1>OohV9$MpYpF*UzCuX{4&>bWk}MgDViOfosRu-wzFz$YK=cl z^-^VGW_D$4SY5w5doSbE=4R%a+FDmehqnxwH#Q^&tPE+ZVGRupwW$84)5aso)alZ+ zW{u90rAsBhZQs8A_=kswXaBWgUZkJqKl=aJ7^*YqccT%C!q~f`V3FF>nc7yeRBsl(@KfkDRU4 z?QOZnIo74GL_i(WPM4YisS*0lWllKH~^2g9M8n|1ejUO)QE zCrsqiCk2yjEEAp|jNi-kZ+rdyo;p<(b@O>H9Rin@y?Oig;q&M0p|+uJp1+MX877=e z`SA6tYtdTMyOoud7qYf~wR?Kaefi~uD^_@zZ1mHcE`8Q5e?7y2t5-!qwceCJQ%^tL zQSwsg+&tUt8_%DeoxP*-vl^&m-5k4pS3?J2P_;!-Ip(>^lk`GX45m|GNE$ z)y)rTJ~I+N?VoBBlSXfy-`u6sATW()ppM+^v$Z7p0*PZw0%eP#) z@+$J*&Yc2z_uD2+U{EyPoSV01-n>P#7#I{%(#`*hoc~<1Sbd4@ixPpLpab#y?d-i|8*C+Z|*2mR#H~>ty*>F#0di} z`JC+R?2^OZzkF%&S}K$pwyW;1)uv6Gdamu59lbs8u0iFel$<*|E*gYPoHeU!Z}s=2 zV?B~@qn7WqRP&povFc*M$49OjB3u`}nqRzlQT050hEM%+=Kr71KiL0&-_*6Hg2@xk z%+-#IU2xrbM=AG$RXr^w98P6sCF}Pv{riyUZNBvI!#4i*H$Ofuo30x@;bi}n!|nX$ zvpA+tn9y)~y1sek8U+r}uof2=m&@(LhY$B&`(bTqsi>vpwPo#|Ju=7p<;}g$BDdiC@_Lds4>MgiIPV%$@T_#3*rxtICQZVO!B}5uCMJWL zA0NzQt+_Zj96~}wE?vI7Fl@CV2UFg?J)Oepeg}>pKb{d3=ML%yoH*g}=GNBJ8`hTn zdN^Ui1heRj@Ze2lnd3%39|DV5q z7v`u)hIo2;-&70LzKQ(%Lo@-dxw1;g`zmG)~GnW<=9C&havSIbNoW(jb z&ldjuRa*1^?{`oeRikmb0>`4+TbD0aU$l7f$1B19S7*nTfEL~BS#I0Dy?y(3^O~BP z3)}PKL5jPibLS@P`K_l?{LEyJf23r~?sCPH)W-dV%!m5zojkm=zqHk#-1|jn zN&Wxo$EI!-`*wvdTEO6xruL)u_ksMoTvt3(!u7+$J;K(tq}|WbmlV)vKa=tG~BRo;+Dm(>HXko9r(=i~9hr>8_O?)tu^ev64t+#ZR%ygUwG-qz-3=1-qKO_)8qd(9f1 z7bR9de*91{F%e;Akeop@oXw>YSxm9PCW@nKul`+K%WjvZ4{P*`yIu8(9g2M32k zK!Cuy`2Bjn=I-8Y{p;&%^;26f2o@wJPPneM;GoK+Hv1c?sSoOZasK)D_w5y{OEICL zq4l8R`r5kKY>qk7LBc#D;^N7dm-&KPK=c3qN$-)fz4he$GzE?mCr=*SmU~-3S~`2m z$}R7Y1-qdjlV?+tt^~W_Qsm;r zi!tc5oyt`uws^XRY&zU15V_C$Kx^|72dEOn1#fiPl?0g+*x64)T-MjbV z^78(R6TBiKB6eR3GKf4o{m6TJ-r5zdkCI;O|H-z?-_Eddnqg-A)2B~wY|obuTN82d z$=UsOl9G}?emw5qn0%ZsaM6ZiDe39SXJ?t-*ixSv+-&~8&BD?$@xg({vw7*YUS3`s za&Mb~QpNqc-)l7viW|xnm$g0F`CQ@lUg=5K=0H0TB@$?fmlFw4}Cg-Fo!V(e8~I7nPiiWVc53nr2^1czbK> z*EIJnDocYndi|d-pzk{hGV)al!L*vUm1YUth7he(#z!Jg={>XHVwYoOU+r z%kPDI_QkF?VwQc)vu8{H`Sa(&n>RdLqjFb#4vdK4nD^W;^_0luc9|tX zD;Mh8oHxJW3>vt2dwaX{`s+% z?EgvZ5|3}0FeBqup$&h%r^gLF4n5=?{uw>-M+p1WNPifg9qy?zTHeO z{P@WA;m;ArPdNRwqo+qCclG+|&4z6C_4R>83)?n*KAB?l<;6wkE%^=(4tsxR znMmE)UmyR4<>}L>trI3p5YvqpF`4yh>pxTDccxMc!o&`x&42&UnPdNT!;Av?tx_5q z4D+Y3B;Mcms_4MXnO7^>r50Uf`uVedPIUcp<}(qqXPfA0YjfNG`OqB5%YRIwjZe1g z=H~RnmoE#Km6av*fZ`-IHTB2O=ko=nrM0cx)`mS^KEKZC{=Qm4DXFOUUs)Uv9P5>S zz4xcJrDfvHO{p)gt`>ju_O0UNlM60`W?}=PqPlKx&wu>t>gtMTGt+(QYC(ff51u|v zt=~~rSZHWuYz!KYR8dpwVq)4F)q8h$dGdhXT)6__+>mAQpg z*!nM$CLA&I?G`O9I&+kDlKo%KKMxNJ=*4FJSP~lgpkJ0b-uj{BamIP`9ymF#J|DfC zedgl>ypl{O-Q^w4OS2x|*EW4y_d{S;`Mt)1XZ1dht^duFUb;E+dJcO&YcI(p0Tw$F&>NPPCr*9!!9o`Ur}A1{OZce3olDF z^z_{7>g={e>Fz3f%axvtNG%k2~I?WydY$G>8pO6d1ryl-dEo!cvIo+q(yf<*F%uU|z) zm9zLBK61@?weC&E(VnosmzG}8-~VPyB?E()q^FBxNNm=lqrUTPwN+FdtqT`@{J7-o zN1J}#=xrTbevDV^zXt{eR=nMMy+^3_s?UtqmzVR;nLGF6-SYeOoptjkPi~%H_lwif z(edj!_8oP9t;*iq2z-CG{_d_)4j!JI55__wA}t>uA2)x!TSGyC;ony8SY`>M6pnc{ zpFDGJY+#f&&lBM-oi=64gEKRYJ9>I@N{%*OUAd~M;7vqnew=QVxVXo^`~RDtJe__p z>#1g9?XL!284g9og$Jd#E~vlSWM9JZ>%6`BuetLKH`vuG+}_H)=G&*zzUGua6(2r7 zZ~p)5^|IS*FFSg8ua2wx!;$s%%!XOB^^O<5xU{tU#mU=qveqe2KOJ2%cN&}GS<7=* zI`&&WH%X1(8r2)T+)r_1*!F2bD~~L2Y+ewc;j@eHRovk_-VwKq+M9Qt^F6(LmTJAa z`tgiP{eZdCALZBoj$Z8S-(z@YzP-GtsOXJdrK=ydo%?K)cXwASD0B8+e*Lvd#-`%J z3W1r67b}OYkGq>Pm%CW^esYiF!hnQpYa%^7JPxei|4*xGhlhtpgWF=zoX*9JfWSb; ztx>$aZp@u7OiP0}-QAB@segL)wQ9=r>Gg~E?2)mW%V%b07N8*#pfy!C?~y*k)3 zE36IbpFL)N_Jrk?wZYwQZ*TV=`()X7_{>aW-&y60LbME%j&R7>*TtB$6#V~JtD>T! zprhk6CDlY*o12r9)9vVHw?3K7FH`Qxg=mR3H8p*>oxi_xn!K2pnAxsb3s$V)`1tYT ziZyFMDf*6ey(UN#fYUEcgP@{5tt zg6j9SC;t78uMZ3hytdo?Mu(ulw|BXBe_C5|F<7lUbjaz=&CTrUesd1=N}KoW6PaK4 zEA#M;)ZiKG^;O>QVLtix_O+SL^9>hNmmfIV+Q#p*_V)I|XJ;grELoE9@zGHYef{I-=2{oNxnY=+ zmey4zZ)9xz@pk_Hhc7NJ_VD*VZhYP*If8Y{j2RuEnNC$z)wcP8i`^cs-F|P?_PL#% zoj<-@_IK5mcI%Z&HJM+3N~Yz_n^*7WMKLl2YenDJR_Ac=;`&kiT%oh`=i@tCqB|ZW zywGG&m^(3Q?XgB?b{E!JS!d?iNtOvzWcIveNB#`k5JtzcvUboZylW7gvAftD&LM5Vh8fg~8_h`%-0FG0#0HNNAc~Y}cPZf83@UhlPeFUSAgr zntt|CJ3P}kT`JZn()i+~n>QtkpPgxp(QDtl*%&m8IC-mn%#MaXe`RKn^jMLl1LvGs)7cwJ zWbW=NZQj3se@4t?!!}9dw1ktBRCA|os(62IuY`S_jb7|7mU++RSIY1DIa^OtbLGA7 z+A1nbBA>5}+!k)0p`oKA!pf4hRm#%R(#d_EouKl|+FIJ7v}yA$?WtZ~UOin~7_Tx$Bqc3+ z@wcqF*!lmzzqeIxJgWN2^y*BcKv&lpW0jSc9Vj!q-3-TZXwJ z$vnd9en&PYA6J-s^1<7;z02p0uHh%N%bgK^edwDr| z)q8Q()YdKx(E^RX%&{sxRgk=X&6*=OH>Vrs-LW|FwCHf5rKKfkaQxwZ$>a~8KO5J7 z-yg%5_ot%Z@v+Y1zboYq?JRz#y(~D6@5;~LiSy<)w6jl`dGkTynR9%Tp3iFzP|sG` z67^yKAEt*#I6rSW@1()-;)s%xl0n^{3eYl;`WF`#ioNAhy|%P$=R2u+^XBPFi4~QU zm~b(8d3gndg{7T%9$)-;?df~>_C-4Lo!wgbInDW!{#?KINt1-0JbR{OYI-$e&AnsC z*piR;xq5hbY_@CfG5qrCDmQ2)iIcN4FXNvMLFET8Uv{qF|1V19&y_RPpcc@D-R1et z`UxQ+^&&DdGNAUp^ZJ&!_3lwoQeUcUQ;Z~=nwlOwf8KuON(jSMTLDQ)NzmMnj^wr1 zUp>6MgwAfC;I&l3vS^8hM}}A*^PRf~v%<1&++yuM0^=x_$K);bK)%QY!fH!12zm($gF69zV`r{OpY3sgIxtuKQDQr#pm)xW1P8mCuiydfv2b?y&FqF00r6%Y12BTXP^pi$jD#z-dkA`t|ECZu|46 z=0(Y_H2aC-;^NY*3|y^FC6j}-rha`UrlGIzo|?MUmAT=;gM>44t*@^!|0iWM^UR9d zu>vhs)y@sS5Z3>+3r+C~zeMpso~ctE_h*OiO+C@d&VMF+^#Lah289-r z=vvUyoE^o_{WeejbVZzD$F5z6)`ex=n3b9nv@*oRc<+)WELmH3o$Y$>GvDrR!Mj;; zoo{b%@4j{{eouwq?{9Apw{QyI`lFUv&u3?62kI{5-QAV?LK4(0$jZ`UVrK6BWDS}I znVGuLvnk;3mRR8(FG>XJ_A_!A%c;lnB(c6<= zU0JEN{_OcPXO28QJ>Av+tIY9(=g*(lU;vE@E?&G?gW<%vb8baNMorDlw{^eQ`u=*f zV~54zZM)0gKYBjD{+PG^-YcJGetN56Vxl7AY-W;xKXI#Co?#CTr0?W@%fg7?+lc z>NvR{%e?+XF@nNG*Qn}C#^NLPO$rZwzu#}Z#yH<6Elq7({{6fych29xFVD=zWALi1 ztfa)Hq-4uQm+isJ{T`m0s;yfidiHpsjFOU4PgibgTH2!v3!N9a*vLlu`1EMRH!5%} zO5|M`(si(zz57{GeTi02W3L3`N$LC${hzB9IUby!(^&9w($ibl90M2}I5GuShIFOR zuf3+x>0ec4b+7upZOo1WMt*ralalw>?<8<>bAx)WI#F9VB1_e3SB7Z$%raSdu;z2W z{XYvPCMK8L74wdtnyL+2a4jMtvS8IJuPeQ=lMkqCYjaQ6i#4kEQokdlHI?h)#fxsm z{C@LnIzc1p7T)`n3kwSmt^h4Fs|Br1`|{#q;^c0_W2@A^m%jXTyuY1&#(w(<4-S?3 z-_j}7-efktMem&Q^9JF@#&faQq&}uHX9*K*u zuG=RZU;vFw@Bj0uJJGB*F;TJX&5cITM6s7rbd-+|kKdeni^Pz9Gag%fd3kwxvVYc; z(+7QLGKE>&Zj9l3_B3>_+x}?t#lM&WRxkYhfSKdUU;YLr1}6{g9Xod>zPqziLqo&C z$*JkZNpo%O(;r&ahiEBsG+kWPC?#_ow0QN$=lTEdJW<(|c(^U`+#Jgv58LHk{69Z^ z`qZN635P^-Th`R9t@XEFJp1-^dc4%9x7*v+uGLLDGox|G4vWBBcYge+kg=^2@xMOZ zXqVL z5Hz^HG3VwcH}0b=uD)W+|5vnAW^>xvLyKIyeP$J?O+Rgzf6vBw=IveeERGFPYuz$4 zS6)0T&j9M~{Qdpiyizv!%ZrQ5#_8uoaxZ6?{5W6#FF1ealfGl;&hhoz|J!k~%-`R? zrM*3RforbwiSzAwSG8W){akWc{XL{KS*tYBLqJBRXZ`+vQu=#7Fy-XrygK_Hw3bfh z_`&=4vR4}-ws!rD8U+Q`aW=Vk#@wUq!kJw|3B1} zUfq{`e8F5R(O=*1G0);U^=jL1!Syv>&I~OAntbmYE(?n(WL+8)0zxVm<9LviYu1%V`lRZ=_KA$xQjmTXrYjtXz zG)c(fnr7MKXV2O)FE5jvre}0aK&W0Q=>5*AUaE^0EpnMDzPs%0t&C@fcdb8s`}XdG zGMrgkrH=Q@i|2xxMgRYt|8G+9Pc75rfw0hlZ|_WVgu1$;=NiX*__zN%z_{U9TL%lH zlPhb>At#RQ`OX#<0!~g%C*O*#i(!m^eLtq+$A*^+778abiI%IM1+7m9t-JpEs{U%n zirw3;b#XK`=*KdFAQ597#7y{=~I!xv4s45es*^D#iswB z1w}??YN+yr+&>Vi6}nrM^Hq*s?5>vZxJp+O8(|TVmi+yHuPyn-i`ri%WpCXHorAuCR46o!#R4PCx!W`?2DGNy-&w2M)!s zL+8))FI>2ANBMiXhF7Xho72y`g@lCc_L-xjrw6KeySlop{+zui1*#+z!w#K1$tfr( zc;fu|;~NqW3y6xc`um?(yR1LiU6_?&iK|A{^6w2Lb~48wo>ZTIWU{|qX3#R{PZ^6X z8S0lrJ}I(HF_L_DcX#?#^8;O7TvMh_J$UR`+ddD^VNg_5)HA6wg1gm; zk)NM`$BrEhk26$y1up#zU~rgq_SiAD*xhBW*Vo1B-W74;P}Ftl*J}Z-!B_-}A&!ks zphhc)Vps@i ea?ogd@t^snguqSqW6^D(?P;E_elF{r5}E)?pTSW8 From cbeddab35f0733087db77542ec310c74c03af380 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 7 Jul 2019 02:11:44 -0700 Subject: [PATCH 115/880] rename ocrmypdf.run -> ocrmypdf.ocr --- docs/api.rst | 6 +++--- src/ocrmypdf/__init__.py | 2 +- src/ocrmypdf/api.py | 2 +- tests/test_filters.py | 2 +- tests/test_graft.py | 4 ++-- tests/test_main.py | 2 +- tests/test_page_numbers.py | 2 +- 7 files changed, 10 insertions(+), 10 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index cc83b55f..2e9e7597 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -20,7 +20,7 @@ and largely have the same functions. import ocrmypdf - ocrmypdf.run('input.pdf', 'output.pdf', deskew=True) + ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True) With a few exceptions, all of the command line arguments are available and may be passed as equivalent keywords. @@ -37,7 +37,7 @@ worker processes (forking itself) - manage the signal flags of worker processes 0 execute other subprocesses (forking and executing other programs) -The Python process that calls ``ocrmypdf.run()`` must be sufficiently +The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will fail. @@ -45,7 +45,7 @@ There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. -Forking a child process to call ``ocrmypdf.run()`` is suggested. That +Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That way your application will survive and remain interactive even if OCRmyPDF does not. diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 00d757cf..ad4b3068 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -44,4 +44,4 @@ from . import hocrtransform from . import leptonica from . import pdfa from . import pdfinfo -from .api import run, configure_logging, Verbosity +from .api import ocr, configure_logging, Verbosity diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index f68ba611..9704f057 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -158,7 +158,7 @@ def create_options(*, input_file, output_file, **kwargs): return options -def run( # pylint: disable=unused-argument +def ocr( # pylint: disable=unused-argument input_file, output_file, *, diff --git a/tests/test_filters.py b/tests/test_filters.py index 5d9482ea..128001d8 100644 --- a/tests/test_filters.py +++ b/tests/test_filters.py @@ -80,7 +80,7 @@ def test_filter_from_cmdline(resources, outdir): def test_filter_from_api(resources, outdir): - ocrmypdf.run( + ocrmypdf.ocr( resources / 'crom.png', outdir / 'out.pdf', image_dpi=100, diff --git a/tests/test_graft.py b/tests/test_graft.py index c384f1f2..df764651 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -36,14 +36,14 @@ def test_no_glyphless_graft(resources, outdir): env = os.environ.copy() env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2' with os_environ(env): - ocrmypdf.run( + ocrmypdf.ocr( outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0 ) @pytest.helpers.needs_pdfminer def test_links(resources, outpdf): - ocrmypdf.run( + ocrmypdf.ocr( resources / 'link.pdf', outpdf, redo_ocr=True, oversample=200, output_type='pdf' ) pdf = pikepdf.open(outpdf) diff --git a/tests/test_main.py b/tests/test_main.py index 7e2693e4..d93227a9 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -614,7 +614,7 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): def test_masks(spoof_tesseract_noop, resources, outpdf): assert ( - ocrmypdf.run( + ocrmypdf.ocr( resources / 'masks.pdf', outpdf, tesseract_env=spoof_tesseract_noop ) == ExitCode.ok diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index a2bc9f4d..b2cd7879 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -42,7 +42,7 @@ def test_list_range(): def test_limited_pages(resources, outpdf, spoof_tesseract_cache): multi = resources / 'multipage.pdf' - ocrmypdf.run( + ocrmypdf.ocr( multi, outpdf, pages='5-6', From ee92ce871711c4df90c4bc0e6c492d0d61196783 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jul 2019 22:16:47 -0700 Subject: [PATCH 116/880] gitattributes: ensure afdesign is okay --- .gitattributes | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitattributes b/.gitattributes index 7dab7cbc..928cb783 100644 --- a/.gitattributes +++ b/.gitattributes @@ -9,5 +9,6 @@ *.png binary *.jpg binary *.bin binary +*.afdesign binary .git_archival.txt export-subst From a7b4ed9688ddd680ab799b9c8a2d4e4f01551faf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jul 2019 22:20:23 -0700 Subject: [PATCH 117/880] Double vertical bars for logo --- .gitattributes | 1 + docs/images/logo-social.png | Bin 0 -> 31298 bytes docs/images/logo.svg | 25 +++++++++++++++++++++++-- misc/media/logo.afdesign | Bin 23825 -> 26568 bytes 4 files changed, 24 insertions(+), 2 deletions(-) create mode 100644 docs/images/logo-social.png diff --git a/.gitattributes b/.gitattributes index 7dab7cbc..928cb783 100644 --- a/.gitattributes +++ b/.gitattributes @@ -9,5 +9,6 @@ *.png binary *.jpg binary *.bin binary +*.afdesign binary .git_archival.txt export-subst diff --git a/docs/images/logo-social.png b/docs/images/logo-social.png new file mode 100644 index 0000000000000000000000000000000000000000..21354b9ddb5b410de628abb816987629f6120ff5 GIT binary patch literal 31298 zcmeAS@N?(olHy`uVBq!ia0y~yU}a!nU~1%GVqjosKRYp)fq}6p)7d$oILO^e!PC{* z%+S)zOxH-ykbyyCV(G-w-cF7p?fb86Xip627SmmF-Rm&d5{{;x6$(yg6jt}f#W&u1 z8mGJ@_kx*5$E`0>SKa#f=9{Lf?L#CuMxE>P*zt^MA8@H||*H;5+w2 z$D%CDhn^2#Zk?0$sjp!5q^=_^=UeX1iV65}?#g1-I#bKYNBNVt1{Cn_0#T|x6H>kch(%?`&d@e2pV_egw+DM&N@tk;A_tmP9es7-+x#<(t^1uDhkNAA!~9#%w|%_F`z==^ zNAlS9iHRb82Y<0lzv=bwlp;@E$G*^uyZ8?5WPBbjs=JF@eD;NJE8Mnt1a9hz7UHb1 zZe3=u<&MMQq?PGUYyIRDw&=x|@>c)8p!|HUw|(~R^cRiyCVgaJVBjq9h%9Dc;1&j9 zMuu5)Bp4VJ7(87ZLn`LHxm#Hi9Qy0{$KuH)@e5j=B2NbiaB*zP-^9VOqT`%K5BCd= zo`^t6mqZ7_3xe(vu8gVff^IH_B8-fTOW(L^WH)Y_wsgw2Ef+57dKTB$8$3SWxNX{# zC-+pofBtuPx|#a<+AT+)RiEGaTw~QLP>Nw-sF~~eo{a$nE-*1lfanAcmIM&ppeWG5 z2x7E3Fo0D^j4Bun2L=WPDyD}fS8(DBP_(qPoH=vms#UMg76*ar zd#k^9ba3poZgSMTx#g24PyU+6GU3eq`~M%d%gb37r33^_xVQH} z0mlIb28MY@(>CkI>?pXn$d#|%`F$Ki`O4ts=cZ@|M{UViSgvd!#$>?2(7=B3#0d%0 ztPn4+u6w-+8x&N<~FQ zr|RDqpSLxy`BCuw3+o^D=M2mY3@_gAe!ov!{qM13$M__TmLwnVn=_A#e|!1+d(Y?B z|GT#S`r2r7JG+1L|Nlu33=E8XKi9fEFCgH;WPiJt=hzZUbQ;+h7%trZ_4W1Xs>}-u z9KEOOy}iHR{-t=_+WhcPMvAfH3)6eXFzi)Td*Q~QqO&34d2Qx4*%$;jhdMa=G z^k0`&uiv+-(6}MmQl@WZ@bYc>_xH^)3|{8*^W$-O(`?Q*Cs3($;pWA~?%sc^y2W&r zl$A~M*b2-;Lznx`zP2g#^d>e7nZE1s^|p$Nj;Gfp2!O+P?$Tw;rv3W5%y)Ly9nJ;3 zu7PQ1W(1z|ukiFdxufv0kg)L9H(YMu0`J1it692VUvJI6er<=agK^sC%4cV0@=BZO zWH(&RlC`b+^6&5Ot#7#O`avFju<*uju6?=+3J)gx+pUb=o)^j7)zuYO^>XQxCn>tS zoAMw3Uj-^lOBSrk3JAC`$Fex<%8I~iGn$&3CMvtnv#AU^eeA(u4p1Uw__A& zJ3EUPFIsd;YVz#a((>~0*A{;|kgEuaN`?!)t5yjL3Kl;*)7jCXk?m7f_U-d|`}xzR zElW-Q;0g97Lsd}d)~Kyn#m~?2a&v2E`?$Fs>kw2nGBVnd!?Uj)&3tO)c%(nn*@Q#?)RQ(+#anZD z_O*b`IiS2^)vXnQizSWIe*FA7)%4ksPGM#DzBA{~U*BZ-g9nr!7#RE)uFCrS?Cf&i z*=c68PfJanGshMZ8ARd}`;$gv^^82-m-Fmgs&$RJM->Z79Yh<7WPOl6HoGoSg z+~q2t?0h~?JN?Y}`}OvxTHY*ixoX(YcLB)@sWfL+!9NGc1eMBqjDk zWfxn@^zHxiR3DU0Et99|L~cqr$aFTUX}S``k0pHVkDpGDmorFkuuOh-Ztm`qmqBNv zn#`fnOnmLhZ*FYdTm4-x{Y;Of@xFh*vd=~}U57;c2R^>`$7jv&@A>_1_i4$`?{>dm z7q>TR^Rx%Ckd(8hZPlvxdp`U9{`U5?q;cx0DZSF>S=WU2a6rPwJTSEM*O!;N(c5&g z8xB95VUT#}+S=%?Z@A(bAzu4k^W$N=?`$*O*H;DuUZsR>Im9RC+y%`g^C$nX@J} z{6h*P^zWTIe?I@=p;ptZD;n7jj*bUUOjP!tXR|Uj{6hppMVyX;Lc+&KM*{-`Q+ZFE zJ=@#g@9*n-b(4058YCsX7n70MQ}OXpM@Ppdwv_aA`~QDFXJ%$bm9g@>fQkf$2VV~z zK5YN@%jFFlHf&-`NlEFEFjP`jc0RrCK_(=eY8+i$jyyWr-Q3)KT57_yX=3v7^FgUT z{DTI>5q68Bwq|vU>2h&!Xk;I_xj8+);$iEZJ2ASuo1Q}xtE_4EwU`YFjq~TnM>6}( zu`n!pa^laQ8c<+BRcIO-8Wul0V>tV4DzAs1pPXq{NMho`X|V@|Aq}e^v)p>6&dxGD zouXy@UPV>4SJL=c*4C_>q9yJSe>BfB&Hnc9_xtI_3E$t{eSEBU^X<2@wzAw61h>n6 zEW5TgdjIctyPxuSczDdPC|q>-;e|D-FQ5rAz4YCk%1cW;pYq(;ka+lZ{(fIJGY&`y z#+RM9{oW&OzHScV>aex5>}q$dS~Y7c%U-CncKm*~`~B1D@p|b8x;`&J1F zmwoRiy319*+xh&I&5O0**tB5 zEyU8r5@tC!7WdmlookRZPFoYRGiY<&0z-%#bLGdQ;_@{g4i>Tb%(W^#JIi#|S7u#^ zHziKbFid{8_xrsfHlKMmm5-10&icxH7UEQibFbI$zqhyg`=+>33mIWGpBuYMS5MP* zWQMrb%_Q~I6h2w2DPOZZJx|`)m@F(TyfugOog3J_2`*iW_I$rry*6sA7HW~6=L%5} z(6#8y3`1o-y>)9CU0q%E_x~v>D%vzH_Fyd}8TFh_F*-ZXc6Z|8wp88|=g-Sq7CkvR zS$*ppt~_w%#E_7Ywz=}(pP!KYke8QNQ}btM@$)TjxZXj^vxE|{?$nEmTp`(^tn6Ed zpz@_lm$tm&x&-mP!;jy;^&>Ye@tUfY%6sC(2?@g_7dN-IX|axakf7sGba8MvuqpMl zR_LmbYYkhrZ29ry$D21fy1ScPAffQ1|G|R?`ulz;si-Vj!xHsR7xz?~`$r8ilV-ah^bayv(Ljo(HE9>ei-Pm0rfm1fI zy}7kD`_7KSvfZ(#*FE5Z_+-kuu(eV1YrjQWG7AX_srk)$QD9+`&AASen;d4oy1IIM z!NWsEY&kbJJbb(Tz8Z5iD^%z5z{PI!Yd(2eGKZ~=GKCb_*_`Jfh30|x8Q0d_?2$A+ zCAlH#XxB{R^hu4jGT;)8p)UUQwYC30_Sa7_PB=M9^{M`}ws%nHD$cbkon@5jWy$>P z%*@R>H;vBzI`|e+N;|lIeQ`1Q&ySB!d44=-<_}vJvog|thb|G6c zncv)9X=kO*{&GADiOmP!Y`$D@*5C8N$&y*kf8L(L$8KkTIkG|}xADr^RNUB*n7a2> zfrXmSj18r)!*rt$y@y!ZVBE$hYn5_h!lpP~O-)JbvYe!(MbmT_)KR`JDCJTU)iW9S|*%H(dK6IgPGl%**xLn#;$4mHduej<7QxBXb5IvViMDjlaZB;z19%B zyQ~+|BDpzrV&s>+122p|A6-|RbiT4s%{@j}r`BhwL!f=omi5Y3R|Ky*ZGFoRGJ>0l znHf}HDLS{Mq@_)>QczS(d~~EUfB)ZYCy#tta$B(B>%?FUvn3s}D<74xezV^nrt)oh z=ikhvdwCZXf=apsznh6obbR+-2xQplrsiglVGEtx<8~G;y}~GMmczlrQ}g%h^^*tQ z9jlLDD|+L|^WR4{tUr2b3D?s5v*(->J3Z;hhxJD%)hf+z2Mr!EngJF&AZs(f%5T1C=Fd30+Lesq>y^n;mAN?@((e4yEAy5Z- zkHb>0sh}W{lZ(67AgUd9W|nC;&zmrogx-{=-5axy?zyWJe$VcC<7SFml>7qcx)&iU0o;cm0O?QI+JP|dl+G-K@ zOZY3OJ#FI=5_08GE4QV_^{c|xM!mXzex7ai)}oe^|1&NuIQZt~=D&ac zde>fkSg_~sx7$aL9+fJ8`kCXvyGirZm;F=m+H!Z%&4pd%^+MB&O!sOqFfbH2zrD4! zxw$$2ZE?LV&96~0@Q7j9qgtK)m=?JEWb27^t@d}mMl_4V)X@AdzFK9@3oEE=^f z=jXlZ_cMK#x)`?~-dOXq=xy|;M3#h~M(euYPcD6zc-F9RW-lmdEZF(&?d|DbFEz1p zFY}*2Z~pxCsrGRTb8m0U-}f_ZrDVnFX}V|Tc!&H7v5whj;(23h?U(OiW$iy|=1W}m zeQw6Uz>r`v*Q#_??Cx)^+~QufrB_#lt`1wf>-D2;6+32TM7X}U0~xELHaYXpkB=We zetg>F>E-3+^=RhaE~~mf7K{;M(;U3NeNB95$t|+aZb>;iC=BLp%f0>YV}JdX-Q4#J ze|$*X8~s6+C1K_oRi%$}?)m<4dsG>ibA2}_0|SG^`ELDv5u4L|_wN26`1{-2%ez_c zJ2Fgr=`c&<(~<+3x`Mwif5`(4)gBX;k@@qmT|R7WROoxR8phjgo-Pd2GOt(ewcc^m z&wtiLHU@@{9=p5j>TbdH;ny?|7BDeJNQH{-w$;ylw^Ko8-sb=#1_p*6 z6)mk-^Z);OzQl8~le6>JcSpOA9y#*jZqs`O2Jgvru`$}0c6UZU5$p8cS!E8&5{rXG zp6>ho?)R(J>)ZI{1ELIEo;0dJ4JDKz>wy|7!)}U6qT-7UCzG!MSwfRBL|EH8d5(UV0hE03o^PvkK=&R zQJ0Qa3#-gON{2eTFrSZ}$_p}bmjc7-B^(!bPXwim_Yn(U+yz<9P{F~pA%*`~XTkX! zC0aTTHdlP!tAUgsNET>N<^CpIpS7}F*oi}t?XITNpReFF!eGP37_skPdB+Qbxz#Jg zwBF?}DhAac3=aZ14t$FAd33YP>iY+u=!faoc7oC=1E@c<{Y6ALPsXa-SKsF=?YeRq zoQWA8%;Y%m?o*ES#Z}AruJ7I4CUGsU{WZv^4eiPdySs0y#J$$bYRRs;D3zEL!=LzY zH+ZyoyMorvz=(z!SE56BFTN-WZ8|!8uWLku+jOvp-a9drt)KPq(O>o>vVnQ~10Jt> z^q1XbtykWD0g%5cG?_Nsc4g@-|J6B<_ija}9Pg1U|CEew^iBuKGUTUU+|B6x!MuC_ zJzc3#(fMyhkCvUBI)9J1jfA?0t3}Ansf-nNyBAN<*J=vBz`(%Z<<4-eL+MDvvPIsu zACJ~;RCINU2yMv_QoVIe=%3ogcQbm#7%QeK)PFp6W@^pxJNQ-E<@YTmC6BaxkYgU)<~U%rBG&WI+T|zCi0?kvdH4B| zgWu7!+gn6K1l0&fJf|Vz<#XK_~`E2{~E8wN_?kl2> zb&LE|1rKZowG$ZT9M!wWbN|Vr=mF6E6x9UBBnkDed)p7L|)PY!3?kx@_3!ii*8ThM-95k!%E}G?yQ{9=S~s`q z(~^6&3n(EW|!Sv_VO+(gRVv0ua#?d$L6%Q#)BHu4d;t;?yV{3X4vxh z#um@_Vh>(|U75b}c;CyrObl;V9PfX9m&ryJ)VTbyZj#VmYla1i|4uyq4ssw#Biv1{ z@7W%_4Vk@l$Bq>Tq9OufLIPrD#K%m@-z+v;x32CVW8J^Tx;jR&`ugtphIogBf(+Ho zGvzH~SE!clzL_NGbaA&Z0|SGD`4Yi9Y}thmudea7oSAuZx_(NW*V!EfTDD(TtlhDp zDL-6fsaG#sX!f<$L3{n4u9!7-wM+OqsaVrp>pDZ;%$g)r9q0FK>zf|w`*Y{?bpPCW z^QbH9(bntT-0Q?epNWb;S-C39DB4Ney)|r=kdT<5Oqiv{mLQJ^mz1x!CSJ{Qb-BvL z#i=UC%)oG9yP(Yn>n`rk`tvMj%_@5EWZ_(Kzc=Mer>t5PnYnz^CBMJly8ksQ^3Ez> zbA3ta+qScdUmyAPlYPgv^^PmTH(%19<7Zd?=)ueE9`+eb*8^!BmPVFsP zc_rLg+x_YqdGXNIj#s4&Ctl48UmIv{ViQz$8&nX_<2ax(wd&`gnNyf=`^(;|HLEOl zJv8<58rPr4>u#33neOYvG5KcD`pn%s%D$Uq-ql?bB>Fw}inr*q=W}hgyngd>$z=9z z@n?5wZ2$3f^-N9A=%7j4{`{IM_4Sd|)>l#Q`@*(L9bsi_J+(JxWf#}F`4(Heo4N0v z%Gt)k!phFl_Jf^`orUc;JKIlomYKSByG%gknYkjv<9d-Nv9`PNU+>esA7@eN@O!nj z*PG@1?`>);p1(R;pMPmv_~~GcQv+#N|EB5=P zn@_ycCkwX(h%Bw!%*Q*syl3a*by9}Cx_RBLKX+QjpO<^S?2+FGZVoQaZXT}HGJ>I- z-Fzm$DwT4(x+Z>S7VqTFx6F*qM*2}HUEP=8RfVmdHC?@^`0AaQ!nd<#&DJg|y7DGx z;?$*b_Il^$US21>Ji_jOOy(7j=W1QuR}VP_#^`WygXjQ-O`ERtcLtV49tlbwBRCUWFj*}}DSINfveuI^rbdqK$hyjK?--|MLa>?#Z0|1avjUV?(% zzn`L_3+DOV`+a%;^ZC~M!r%LOC2H&aNxj?^Z?-n$?qmMg_wqqYt7P|DU-J0&l>7EI zy}Q3BAOCB&K~!8#{ojl^Ka!uWxE1?O`qHAudmR-MtIU^bW(WRT*?iky_D&_|t-mqhvZMJDga<2))+MhpX z&N317oi#(g%A-dRq!n&1|-mxmd6CY<6Cu{Qa?;QoECX79H;r z{bOEYzSL8AZ*;>(fd<}Jr=EVc*ZKY6$Um07(mQSU=R|ER{~Uf-Uas)_UuT`Ve^uL+ zXKvk5aJsyE`})W?*J6)9*(55S7ri&*_AA%3*5BS=cz)4pqprx&1?ifiPg37U{Jz<@ zb$;l&sMVp{*=KKiDHpq3QQe#If}=yj&hq@?_xD~!=TBV|{XAle$K73_@8|9D_2~Gy z@AsMScbEIkjrw$$zt6Jx%HDZ90?wIhY}{D2ogq1K)}aEKzDY-Xwe98ptFQOoZ@yyH zug~W5b^iR0*NeCK`g`%-<3A_2Ki~87%C@WBdo=$3yY?{FUZ>{u%=o&0e(&ABl$*YO zbpC$Lr;{!AwcqY_i~Gr5+%3y+Y^jTT^t`%T^{2)5FHyOC|DPI@MaGA1np?9X+ju|U z-*A1dX3rDJiXb6yJ2eqKEL~FlewJ# zeC6MdvahFS@6_Hj`^FmA*fWuU2Co z|K3d{<7#d~#M4z(vEi z`+@@|=bHWAeCG8!-dDAq0c4+3=iO~(H+w?4msk8g@_ubl|GUT^-#)DIo>wQ9wDO$H z-R%jdjHjQMy}nz}^48Zs&G&RG*4aw$|5I?rdG@}U;RSzZ-P_AxVBl)_{r&s>xw9*; z=>#tAx_7rib!k!R{*}SHudj8o@qT%Cv|Qutte1D0cv=|V{O;mfS8tj9=iS%SU-FK` zg0jYkZ=1#a)nDH|@z!!j@!$0KcHuD3iIHkUq@khslgwx#mKi=QZ@9Mw4TQooQ-Ra58I~o@4$1K$*tJ>Y$RAXtKv)f~3 zba=?RHRa+4Tud9zpE~Ii{plX}_4QRB*#iB3WPRK5S)Kp&U02I1x~s*0-z)m!_jlfg zh(BN69Xg#7XS4Ie+Vu0qenJd;XIhnhy1&IU^RnA}w>?uUKO~&qe!uHO!CvW)X?1m8 z;_sF?E3+irpHgo6`twWur%zK}-0ip~qx!AB*1*d8@uGXZQ}W(VyK2AoPsDpY72}z@ zy4K$>@8(_~5>xT_^2+mbEqD39cVm#48?=4h{qXos>l>#IdOoZB`{&v^(Tf+W%pV<{ za&7I`JVrN1hHs9oKX;zKCtv&YQsiFgi|&qb-}a~VXr4KLb@#;Cvfj70Wj##UyL-a= zx3>K)E$%Pwwr<}P-92rOc>ODnMfZ9c46K&c)$XtVKifHPZrZC$N8fkrM{vJ237)K0 zwVh$H0>f{q?81ki`j5q0y-tl*Iz6@W;gRt9dlI)te$9LGf3AIH#nCh2U*8?lz4fKS z|Gs;=2Q+l5%o%3#y}PsF`||as`!XB~ zBBs8m5)3re4(;UYUn3T_Gb^Ja#k2G6udQ!hU*Gh3UE0&9DRH|e)tSCtbnLv!|3B|7 zH|knl>&#vE)HQ$Us=ll9_b&hY`NqSE7X`0og|CZJw%HUF@l<2;(Ij2HO`_uezOQdR z+WIqm^QB9{{gz&D{Nk5B-rH>X*Q4tUw|bcFyk8lAqPq^&dNNdP+Vf{q@?*EJ$K}7~ z9f|(wJ1fPu)a&mn?Y+_;jl4pp?0C#q_-Dtx-UHtS8*1ItZryw@mQj9v+dJ9Zce{Q! z9$UMsrOiFWfa&JjZQaW!Pp<#@^1!zD-0S0u%iEng>d(*ra_5OEV`J64f*+^Pubw2l z^@l{_NB*4B@6Bg#fA3lor>22xxUzqa!*RtH|Z7(iFgA!k~L&MX>{W-a}I+*&j%e@>L?yrfwwzvA|%ga^U zAEo-l?y}%GP+-rrp?`yzXwBD;EW6juZgp|rz4)9rTyO5(KWF#+>+dhRuX?Kd=b@Ebz3-KN50$!q zb#dtR-$~IWB}%VmpPAONLP{{uRO-l9QKydM@#jwZ_}k7adGz31^6$D$(cPyi^Pm6o zpLJI5*Z&D#q1m&v>x?F^5)+Nl-!pIVs#Vsrf7QPJ|Fv$hwKW@C>#N&r49&)-nMqQn zr#_yut}=hLc1qS&9pkhc@-+eP-Rkbg>I z>FsasY>2EjzpCGUHfm4NIhSxTiRtGbiC=hqapGC)FI@dA)Yh*mQs&?Os%rk-u$qr& zHX6Hsm%F|F>7|*o`&V!M!Mf$e4flJcs?}L<-e_ITe=pa&YE@bE#;rRpM1_?6y)|`> z_ex1||K;mEK)%nsZ)`_~@NIKN}x$!Ut$R2Pqq*}H$&PZLyLvpKCZW^aE;KIe8&`IK;dt?R%V+1V zK7A#cx_PS2?=|;)8KRUKem5Dre}8-T%`tahAGObW`~Plre(&zHY5DIz%uSDOeajHI zp7wY8=5K$u1m%5kwA3k6Ib=$ORv--#Tb_;Dyj<)+V-F&s%?vH!8y5kS@ zO8%B(o`3S}Ne|n)2TvZ}b+5k}5;H-)CF6Q_KtYD;_t@C#u0=n7vj6(Y-qzCDu|jIU zo$b0+eLdajeeds=CEUK%wVuEH*0uM~w(tAh*s)^PyL|!-u~TyHMt%Bn`P?331&5ax z9ADnKd1X=Sr6rx3dVVZyUU~1#r&yKx)c;+byy@KX?>C>H*ROnMIo->}UE6mSsNou> zt7;XsGR!w*ZB|!zaMqOxSCVJy1TTxYzFs{2{I1_l40+s)H(uY&J(@kur#SxaEZ!&^ zlQ+Guwo31KFVuK2k@>&PyS}j4{jmm14gCMTn*QeW^KbwD#2&9^_go*-mwi)m_P%*< zmp!tTlPmnr?5tb&FK<=->JJ@Rx>nyZFK#^FTrXFyY5Mfc#l60_ORm3p^(#mu`0Fj+ z)z|YXzW;SL)|+FsC;juctKE0EFW>_4VKX^XIe%hSUcRpXL8&yt*0O*;RCT z>eQ^OI`VZH@)j3-zcO01zjfOi_h?zZk^a0A_f*%wD%)R=Hm^Tloc-(X*DT+6yMDLs zuP=0;U4198_VcO!_0Q9GFD{xH{wFQ&C;tVj^ z`=4Uo+?sDbU5i523*UatUtd}It30=Q=f}EJ?|9=Q?*7cNVTdpIRdVvj$8Bk6*Zlo8 zHRs-&K5dQCR~q#{nys~7uip_7yYY}+#fGhZvbQfRRJ~ub_}SB)kFB0DGEZTbD{xO` z@Ka`}-o83=x>>F1Y1yfNnVsj&{gJty(Lc1nI8pPER(S23rRBfv{&jshb+q;RRvWP& zcRHJY3-6m6x2Hf#p#eRCN-SzbA^{3NW zA22a(2oJulKmTRl-1)v~r-j$oL@ZAMJ=08voudm{9O?>ccs&N|W06{R(`%`_3kjKhx(wz3;&I;IPc*^yu4L zwIesX2x|UvuTl0|Vz8 zu8&HLjR%|E{c>-XHhej?(*3(gMp2jJ3~S3Du>^=QZ4ke`>zHnK)qjIWZ>^(BOWmJ#8Atz-lQdlF z)mv3`TJGjnz1Np4cbeueTJ>t%)%hpauGBb{tnY9Bz<7Jy!AH;kU)XSc_O{o$dq3?I zW?y=p<-y~w?zhcsmfLbCM_jtHbMuj`dH#2IUA?x@`Q-k8svDQB^c24At?#MBaX^$M zA%AM!yCWOp%{=vQzdsosUG`3W?;7DIXKmYWpT0CK2-uk_Y`**ThNaR${pzt6nm^Bd zJdil~m1(B#b~P2FnP>jWRo{H1{`1|xmrvrKh_NJa?A*D{T|U&`yZVXy&d+|4OZT`H zZu~nlU53->dy>IsE~yU^ED8Br|F)-I$}W?(*fV)sI_p&ZRcU)2_4lc7e>Sf-{#op9 zrR~r1SKPj)H#K_WYv%KJO25|@WjpOU6`K7;>dT@#v(&%kPmLC62wbxKcDmYMEv%=ibE48DXpHjGBY^!LqgAbuJ|}l)UsT>%sjsuYR=kK* z6xo>?D{^~JE&EF8BulS1zqi%sZ+ws+>t$P*wEv9sv8k)Cz4~~;c-}W($CbD4e{NrK zt5-$r%)Xj)fl1RN1R4r0?EXdA7CQOq+S>=)_eV%A+sZBe#<^`pWaOq-GuwG5hjKM& za~z1%EWCXEQuaOjC!w2fD(zkKde@iwLn>Q0%G&D`URGRv>3aD0sxPee{~dOkK2#I4 z|8I7e{nXmX0^9IBM+UdRn4JGVk|U2!Nwb>6^^o(*5trziJsQ)B@3if7=U?h8t+^EJ zwc4(3u5Y;$Yh1i;$2Gq?Vm3WGc$(M5tGn*}pK_$h_2q~Oy@ zhP^VH%S9N(D<1X8pPF8j(0(^cWn+RvLB#oX+p0*vt#ZJXFX$$Wc^%RYtJAqCid>u*4GaXHotuNa=L!J zpTGb4Q>Th7WVZM9^!4RkSm1cL4ZJLOUF_~xS664>d3AGhI=`Gv#m7fSw`|!WU4M30 zYHmtJMMVl2xVX5ijoP}a?Cq^hsotA!t|?sIAmAh-Zp|lSaDBzYDpggbLtmBdt$rf^ z-+*6o(TNIyTX)vSdpb0%(M)AH_we!7?>F~{e!9HBR{nQ(M#Yjlvx-{V+{>yz+&mk9 zep={<7xUyle~#njocvSz?}2Yyg;rnfel2i&(c&Xb7aA66oVEILb6?fQHBELh4VyP@ z5`q9>VQq`uv*&@pzU1S5(*JKP@tpkb?(XkzZW^1JU7I_%Ys;1`U>|N!T*`F&;ge?| zFweI7+cA5+$W2RvmU_kRE_-`#Z}ss$+2-cvySKCogiFuMDL(!fyL8h=|AqUDmZltJ zI{kY6?l{lRzA!evknlAHdqo(Q3pVUMb9IfkI{zt~`|HCFTfcwxigh;6_V-_pzs;U5 zBL2j4<+4?;KK{KOegAL6lq{!?J9ExVS)#jX6{ljAy^&Mkjo+t!U+LKCoV4|I-kpsx z@+()aoNbnSYg6j!iOTLVI;C%JY)n4hcXqDz_SoHJ=k0#y%r?t?@^9kF6DMx$D16+; zE6pcw_r_~(_VsmZBR8j=on`v(-@muFwn|%-WJt>~wS>t?n|GPtn-Fl~hVcQXj!Bu9 z&(#0h9JG4c>!oM-?Q_%_n0F~LRL@kduA3CS;s50I6-TXZFPQK_j{k4!-~InIS3mfd zo2|n3Ao+Jn-oHCba`pRbRivMC^Q~3cILo^Kjn?IPw$;;gf6ooRwyX5@w|4tK54W9~ zXH}YYZ%^fHvy=9FYtDm$_|4AG=d9iP zYS!I-wZFf;y$xE{Ha)&>XOQYldo9uT4<8-ve*W+Ezu)ikudbS!7Jh$U?eFmT+NV`> zIfd0+TwQm@9p|{x!eVM1T=>&|`l93gtJ6=O-FklZbMwkguc!LhR&j(mGVl7-^rZQI zz<2MvX)D7HTfa8xnH+IpZB)q{WkZJLa(9b$trrKSm}*|-dwWY^={erDyY3vn=pN^l z$8hfKxd{<#n~aQ&x99$zo2s_E{C(clRiVvnyjnpYzP!Br`T6<#I|>)yv8kNg%+Ajz zZT4nrcwFU^iSBb|b8WksbHD!o-$kz7mo8sky8V0f>aew59v%|5RVDxb{k^;W!r_Mn zF}q4MO-;Z4`T2R}%9Z9q`BhcF^5?JL7PU5t?ZH&-@PF_6uf5N@u)y)^s?eohk_BG< z{Jf4m!I44ikYVkvUuTTHe0sV|BUpJ~^fA2gU6-1kGG}_3+HQeGrQaLlM7A8#3eUZ; z=km+L)~TXEb&elI_EZ&m4|Bb}?m*UvLZbkdzJ9=dAPteG<_-_?T@m6tF5_j39CWBu~- z0s;?`m_;YfpT9r*y59BIJHD=K%s(~x!;1+*a_5#T)zH|WANR&QbJFXnXZY<-Xk|FK zvn1qCnpNbGWIFleoh`oRovfx8_BhLTGX8k<@`+mc-M3Y9xtGV(&ic6h(XFqXJX}*F z1fsaOIHyPWethsd*u76Ca$C;L3k#jSYfY`KfB*e{KYoARUbBnew_jTqTm9%rXXfQ) zSJ%bvR(k9|-|p|V=zLkLk`s4(?`jzu7T(y9IKS@KN@43LbrlsEyP6-zSl;g}e%{qJ z=cda31q%{xZ_C|V^|k2H-tF11udNL_x9QdW{rg|L^0=OzS~zFNO&uu{0r!3{ji+a2 zPEYIP6mFS6zq@pb#rnt^_8%(e7{944?|(BfLU)~W{Lb47*$(fHomhE%@{%i-)j9L0 zSecvT-gZth&F~R5`W#m3{`lr15K0Qs%%!`X$XPf1Ed3qM={k^xL z?r+ud`_<8Ad3Rn^O*G5B<#O7h>sZ7RmITEqyGoZ{`Ix8fZ^h>^qszcFOT{Fzl5s%@ zOTzkx1yf{h?~Y75IBl+~Oz(n?UK6jrGR@Uree35GmDw$uTdzL$-2d5f_ah&*;=^fg zQ$m}+aeZ3nxkgOn(M84^SzG;PnXFv0#E;P zO-lnUWh#DtZtKmQD6_KF!OQ*je!G>u{r0iH-|y=`eRX?V?(Vw3yQTy^)zHxR@%4{Z z_?`0k;&UP&WIj2=z^%wrCv>UavTo8w@yv`BMcZ3gGZ^MLGklYJ`P^SDZu+;IRvJ2s z9Aly*uWq@rCs{snsZ8Ib)8c-+Bc4Q+xrN=*)cnh#_%7pe@H?rfj5QATV;&)4^u;{J+00E_QF{lbtnZ&YD{9rMEVxpO4Pp zTN*9=Y~#j_z0&4oe=mit{;IAi|Kjx4+}mc?>$U#;{kt`C?c3}W)`ai(H9m6nt=}(Z zbZwi)T&taNEDgmR2jU+8=?tz|aP#KPj~_qY|L2qU-rGOr!&ZNt8Xo8A?fv_$xOwg^ zlbjnH&YbbtyG!fpt*zPOIuQlY`4de!*BsQ!(0o0Acj;>}sZ;+rw)6$oZR$?wKK1?tvwaGDEs)fDWPY--sYOFKWECNle}x{ zzm;8II`5A5_W)J%jZtfb)qEt3QcmovHIi?euBxt{{^&^O{&{~lCmwDqe0)r_`bpY0 zZgIUmnyaU;kKbRXHn}DKQD7Ev2yXV6MJ?) zzbX^GZAY9-=A|QFl_2pd?k_tn^i;~Wkkyw1!c5;@?VMV6|I3$&clF*~o;*+MZNjQk zq1lypx25bjx$^LC)v(oJtDV^2zS2`wb^YF6cY0&}|9|=We!6jNto{AX{9c9g`n5aj zrr*u`|KoALw%tCjsamquUN^)qJS^~-xls2s$VvMj2doTQdTy?@_sioMBGb>`{MElk z$Fg+K{KvoF$Lc3#-3VY|-yO%^Ak1;#-ohOQ_MaDRnR(2rH!4zRUEZVhi?g;aUE1_I z+<1MgXZga0s(&q8Hm}Sw%TQY_QW7|0etAj%8HK>ud;Pl()_%F@{`&g*`ma~R^KWfB zeR)mv_IEcn-mcJdzNYn}^z}8*mBCt?FL&kN-zUb+Zf@l~aZ~=byLvMi|LK>0PUCb} zVrSL-Uhz80=D=;ihP{WUe)%l4`s$}o3q(UV&pVPeJ!ZzCv%9CoPEgN%nef$W<0dah z@#TFgRx+19g}%Kp?O)>={u6-#0S1YOSkB*^yIMxR_DkSV_VV}lY$J}TTsqjyt{=PW zN!8qFQBl#{+uP25wOUhO7F4oDNc3ryq^YvA)05AVtgNq&cCVXmSo-2Yl^}zk0z-BC z8Siw}u+@UfiSGksLs#lO?_4IWH?Ocxe6{xUbDnQhR;AT;UR_}M^?~2a*_>B%?|3Dd zR&*aZawK+lS!h8)LPEm(dwZA0lpk*64SIDa<#?a$I+@@a=exT~XPbV0_j_BFA+MZ{ z$kDG>uio0s&131i5^}v;<4rT`ix!4AvX{=!oe*(0T24pDyC7oehK{UFtL7Fb%C9-& zEmX()Z&lWoGYR~?b1inf-f?B}u58ne$q~G9Dd*-`ipsxvSncQ?`Tx%T`g&KlE@Oq@ z{dGaFJZ66QqWWs4@jCIqcRLP)(xL;yH@?VEP16=>gf4o0YSl8~rD2D~Hf25C^?~QZ zR&673Ubm3RTS7Ta!Y7n}z0De|B4#pEH?AtkJf{4|hlj<_wRY`#vTpagu4x}9qyzaWmj~+c*8@+v3T>IDaeq8+Kp0_N0)_&k$;E>pTsW#!ey=iJH}bcHbWi351y!3EFU_wl zXHqll9Bf~_lI(Kg5R}kZ6*4h^{Xnh~Lp5{g*?l^1pJl&|$ShuNHR;EWHEVZV$lGlB z{obM1*Hg9zZmtWCT)JSX(C*y3;>P@W@w(e{Ra&btvbpVac{Z8`&ZYC@7t6X?7p*U;zh$! zY00EwkTi*H>59tXcE<`T5;1Lmpnaa%IYtDb`xwIuiC)eSP%k(WJ?9@1K9)#x1U=qON}W zR^F+st!gSNJ0gxGZT#}`vb&ePTXI*wdFd;arBfb1@hl9w<=A8Q=vAnxuJ6j%pO{mx zF3HVcTJVVDK%Bz&S!WJ^cRHl35_Z|U;mYI0F78jxhvxh|A2dPzZS>))@chN|>??oF z-kI+eGxy`_Z+qEi`s~|dqi$xl&40e#-m0&wqPM?W`+pf<`{NfE7e{Z;`}?q6Uhmbe zuYD&|UftUInwj55Kw3I_bK2SH^9w=#KR4HUd*0n$nU~dGs`IRm-oCDhF@odc0cQRk zat*BludlD4F4|E3_Eu;>z=NcT>tc7ec}E?|)|sjueXMtKsE@DLq|j5-SA?(g;5e{L zzcEbY;XeDMdF9%|I~p0@uoe}s^H?J$AfdEzR@@&ZCe<))tyAUidVgMDQk(sU+2Z5X z=oR9A@^?SoJv3#B?zhY!owez{op_t(|+@%O6OC!f5s2HdZ) z{rzTheBIBde|~=c{CVyl;a6K$g|4>y^WiYR?U#VBudbe!4BJ`s^xt1l2i)G&^y^`M z`;yv2pSNFIAOHW7w|;2B_s7-m_kN%M|IhN*iZzGZco)0%PAdBEdg|Z#ocgCSpRcSG zx_-^7?9GJGQ_FYM)Vp&mbaC62%IkTQm6bJS^0sMjx65d1nr++Jwp@v!x-t9uCe^Ud z{GVDEsFv?v@u=yw-yZK!*`GT9SCsbO^-l}2x)Zx*xsYU(PyFCA^$9>%c)_ z7FJfzaJa zq@<*K_5c5tzP`5i=d;gn_4HGj8m$+B}Pt!y?|Y*`T#=_J^n zvMM^9g{9+B(q3ssFGq%NOsqf89of8z<;M}9#q+N)#5@UM*d`|GRr9A`cY$qKV%@u& zkAL^S^|shly?1TIG4*+yOtY@8^?aR|x3}fknY7Oj4mQuP`}Oj~36JMmA#0;b-`&}% z?A|wJ(~<2mj*gCp+xhF?Y&`zw&!0J~FRYK>|KP!cjmgJv&AA)?UhR0l{Q3k2F(;c3 z2bhDFEx5im`gs+@y}i}jUw{5q`|Hcc@8xObY|6&IGZHHQmMXr#@nn+kY?oPEiZyEH zwC*%zfSwD`HAV7)*S_Chf)f^8S~>aG=krzH8K*QcykUJ4HuL-Wnq;;cNgHRJ5p(Zs zv6X1u%(&ym&DP6HCrQ6l3<>;rd2LAOjohQx9yvu9P6W+{#7vnzcjC;lZ=Ze&i~Gy( zoK<;nZtp+)$tUl;-xw$w^XZjnQgX7o|Gb!w?2)I1ThcvJOJgZ*4C#_cUGC*es*^D@jltt zUt$}im%RCZ=GXJv!U75NOiq4!wc6?*YfO#+CGl zT%+`cr+M;A7cU6N^gk!^)OvGZV&K9nk_q+lKldH5)ZNYZ;?8^js&d7%kB)Xfe*8Ff z>3!*4WpAtA@BQwjqtWu>=H};{QctTai&tz}q*4F-_Wh3_SQdWWRKau zLLlkyudnO&)SjPbyUyR``}_O%w|{q3dUSIe2-|y5h>FMdE*T2XbO>^L$ z&v#&1M7%}0uwCbUy|@EmI&TFYJlNpj`f*}@RlnP@_jKRr#iRZr)A%K!9S4QBSJeZTb-EN>bWE#a6_&mC%8zc1xg$jj7P zcH;y4-B*VDzZYw`^z+>#R+g-&~n|IQ8SqnU#0e zP57YI96#rj?1Zk2OG`W-AM1@>AG|hd>(Op;ekqd^wf-^uG8O?|bEiz1(#0Pp;_8*B zqoq|;Wv{#Y?!V2a{r?_4di3a#BRe9Fu(Psg*jCR{^V{<0_x*qBen0Aeul;$Dy|rg6 z+tN+v=bW@$_cUPr=^s1-T5l)!um@#b|MoC<-NUODY&W<)Vwhwl9$9~v@_utkuD9%+ z^M)r3Z{!s=O6_rV2zc>$vHgk|)u{JwdyXAwWM1w&`&rUNHeM-{%u6m__Nkfo_Ei4; z_4WK5%g0G`wZrq?ui5(h{r>uUdn)zfLO3?o{jGYgw(bgxpRMQdIPcn8Cpt>g4R|(z594JkP$x?y~Z=OU`cH6FRZyYGvMxUEdYNx-Tv5 z-j&vSua?1FnL)byRl$q9O(sI0mlr<_ekAE(e(K~JwtN8*v5!}>H)kL5m~qLZuk7uu zopDdff|m-5xMp5oH}^H;k6*dxj6l09Qcq7~=aVV;xXZfi&55&TOOtAMy#8%J+ehss ze|5D;@Z)1n$DG*AH{E>m_;_*C>-QYf3YWD_Vehn^acPw(ql8?C>V-H3m5^Jq_ZKnU zk39F}&biQ~MS0At4lEFe;by&M4tgkBVkIkj`Z&!t{k6UoMSWSv0L9u0u z=d;_q2b?Z_S$Q|FTuq_n8sCHmr89^hZ0j$tS-aZ&)9<_f^6Fq8ZDK`WOn87{0%|d;9+Vng5sHJ<_>% z!S}bf%iqpgy+8HFKd<<{*xg?LzM1dc{edkgt)bSIL7P|l)m^3l(A*IR*VOHI3*-8p zwC=VF3=Mt6v*GAU)eA!94~2rh|7vT`zP0Xq@npNTL3{h|^;WEonBIS_z(U3~Xx@G8 z@MbpNnENTGI~io;S^*rR&kPPkaNH~Xk*{c)!|z{qx4n9$oQ~bhX@~)Zx=ub@1V^HF z+3(|ayf@Yu{n!2TXU@+(PvtdAIZccY?9(%{O8c?o{H^c57VFRbzPdX=lD$1cprN%V zY%ceKxz^?DbRK`1So-=}Xw8RxwZDssiemOCy`KMd(X%Hvaszbv=8+ zLJhCy=c>#Z@ARBcj8pJ<6%@VNpCw`Sm8ac}2hZI};^0VPlMsC*E^y`h$MAXe*^y_f z-q}_iS6IT5z&p)>TXL9JEOK{t&QGZwvl7b zg0%G8^Xj7Z&gHE4sAP*M{W-Jpe0tVJr}ttFvq6Du!1zO~D80d1JGAgoY|(Uw-^aeQ zZqK^^eB<=x^|yE4PyWYfS9@=9o6{1Ts^y%r(NjyOyldl?)-pQtupsB&o|{LrzrMeJ zKjl$!=QWv=C%5gZy&AguOWu*w8bN6d>%DHDyPT{kSNB4kr}MrE$b;@0*%S3ybXQ+t zJ-l$L(%g#aHy^1Nif+lTi7%ME-zTc*|Ag>Eb9NnW+O%We<|%0)z{V+J%8Qw*Zha3eV%J+TpV}4{r`lgfB!NteV$i*?z?r^-Hp!g9c?D`Fs;e@ z7yS6&lzFya!Z%J?rKP`(&zJbj*NZ9A*(N@z zi1mhS-N)Ov=l+_1Y3=NL_RbH2L=yNDHvaP1bE)FrnXh&io}SBEt);SzwWQQL?*9Tk zv*)Kn=jlxqW&Iww%MEq`JB_ckbW+M|KgIBE3Y0uzViRp()a5=pBLQ0!+OJ*rOoYBmsNC`OW2gY zbMvNr+swh&ZyLQO_IF%o`%RBOk_T(In3?B9{F`?#_`rgsP$YVl6qY)W^(fRdD~uIwoW@Q=KjOoz4=0bs-B;dU*yad0S5C9rZF>(rx7>RI!q;4=7VBT$Bc-Y%wpR1( zS+^T3+y%}e3Fif*G#~f6DYgit`%fv^vi+3T<<`RwHLN#^(`V0q{rdLQ;&b|iQ+F>|dguR{kNeMEy_8h{|L?XlKF9Y| zF1oroo?!=P>IH?jidobBezGl?`)vFA2SS{C+27pPd9E8B8oFBR)Z*RVH)KQfw?04b zb%#aUY~yR`FUQqR-np~?SIo{Mq z=O=$rQPH{9%QN}y?d|u^nIogixkKpd3Bli&mVPt8FLJ!^rEPb(;nHPU{DJF#Psn8` z3T&`--Y!?Gx@@j}@u9`s1y5hTS6!C*b7_a~3*Ev=pVLoIx~yWefnD*)-~ILKN@|YD z9=9eG=GzvxPpJ&6{SdUiE;3A>^ZKI(>7ams1$!C7!^87$Z_~B5{{6{6^Zx(JLoJ;D z{u%X$wrAvR6BoNBt`{P%f2~^Ny~@)J`Kjfq408*(x1@J?zEJ&WbFwS=c&h8mM=?9k z{8WA{wENFq!Hu0_tG9R+|2xAO88~54bkY5TkJG$5S6Njj{ob`SXit|o1M4luT}r0#qHk&zFP4~FhAfCNjU%Dad!86ir4)aUF*Aa zipGOqGtWLdIaxjXeB}1L*y!lns`7u{?S8MNWs(~wASh^ubKJ#4hx$ey7WRpi<+2++z3mOx7Pi8sG9&x34j| zZe9AtIx>*6mW^xS%S%;G^LZi<&#~FEdgH#RFsUURy*<*!pZ^9ex)nY9di^(k&A>d;F}?z1cSce-u93G5e|A zCH|P;`F6EIK|y@7R!a=ipPrg3tmd=hhjVD?R`CSQr*}46@X3XAuS+hs-5tAjweZ!e zS-!n1Rt0HmAHBKx)yc_^zGRjaU7MbHfZ?rTmMg3HE5^M64YqD`uJE>dm*s!Cz*JMQ z|4g~hzpN`J=s;QH>`$Te=GVGSN|G!{QB>(^K_Wx8=bDu8yzvuhO{hD0m_YW*x z<~MiMUWWWzTUKt|XsDp@;1GXzcenq1yS1ttKnG3y`f|}?>hZ#yn@Rhh%k5q6a9uN5 zCFkY(X;U)z884Xb;rJ|}D7euv@#3aS-)c7>dbr`VSMk!?`!$-{r(R`WuKR1IqjgS9 zLMc^F(dFYer}o2&+Ps^1q-;%B?VcmMf4}~{x`gA`gBCTsHn04YGWpqTv)i|BU0WYt z|FQkn_WfEK8WVQxSAM|N+4=0;Hs}3;eDA{9zc@2~ljvIX<&AkvSX6E4(Up(qeSepn z8+!ENUg^}OOL^qdH~z}t2^0wqy1y{!nV^%&wYm17Qd=rzW5k-D#r2xnxmOoWzx6~m zg*T@E#EBDv%5E9wckU41vTBtS_l89pL1oXGVosJ^k3DqWzq-%)^KW;CT|G>1JReni zSG;+$$hlhY?ztm>r@pzRV7Rut-ZEc9#p7?SrtUpeL*7kaPM&yoGyc%-e@-2@PJLRN zJ>8>|Y2ExOm)o}G+zg71)oszx)YR11zb-U&=1fUp;l;Ho2d17e`X{q8H1rjJ-ZF*< zNg@gUO6Bk8?7hDxAoTpM+hKS2Chk14W62c9prRzrll za76A>UCMXz=fo*1`o3K&^}at*h&4vt-hTg`J27u9SBI^=RQG>#e}Dg!M22vq`?cFQ zL~VUukRZB1pKF8q#$P}9WK4aN9`u!Zt1h#=-6;E7bs4XudF-`k+KmSvuAC{n?r-V! z-0#QEt1fd~H0R~_*}6h;e>oJ(>i2H(xz~B^z}E>y&o0cftv1WM6A>1+OjSi){rRK3 zOAX5nJS($eLuKdJswuKNgae( znD`-Mrh8;w?A{eycLpXYq;Ph2NLv40zUfj}-3O!Db%&?6@klB?-FQ2Fer;GX(|7yB zk2R8y&tl86wF=~TsG$|Ke`@6dp#QGU_PjH&hY z3?DVkvzu3~@;be}y|vZz<|L^C+4%U=e||cBj!bgmIDTsCEoX)dN5*dkX`53&?06}} z+`aGbqCeH;^+G#VJqmgF=(-9UpIr510VDqTTYmlElQa)bdhksD{_|T4sy}_^KQmi5 z*L=;@?zm@J_Aw`A_4Yn{{ybVoZ0~Z0yoQ$>kNZSM3N^hc`+cZYsE8rMf$^Ko%fl&l zd0H(I`*RuvR^9RG3<>@F>96=(*|(dReAB$|A52bJ8arj8t!L7$ox$#2-WL2PeR*Xp zEw%RbPg(KnRabs^y^1Q=4!M&jPriKl^4B?qr-3O4N`r!R`|b5qPj1YfuAwzah;_kf zjRW)4^$eFrpZ_Ps%>VAst0#ir|GxflpzmvYS-q~ILGl%o*?VqJ)$j6tC4Ala+38bj zv&FrAnqAVmOE2AtKk4gN5`=jTqF$`&(u-(O=xLqlC%-{Xu8(y24;<2kt}3#G z>K88-yy5%xq{Kb#=DZF*I^bLL|1cax-lI5Qqh{JS|d&mcHDa^J7T;dg&><|=o4`<#^dpB$N~ZC`N{ zR9rK!^IN}r0XwhMm9^34vD4V{rc9do^7GHkjn_`ML4$?|rels$4sk873cNVP)M}7h8U#S!#!zSJ}4Gn?|ado?#3Ap4N9Q z+_=VsAu6E3cItZFgexH@#T%_}McGyS{McsY%WmWKu;-h0i$wKh+aiDaT`@&bX#o*i z?wmO<_Gjfjw`a8vCa9{|q}}OZPg%M)?n{xkaQU9YqBS)&ptFb`xvmY%j*FYuEzY{j zLDKr#ub0cc_UuVNE5r9&d%cG@^8#Mh8_lUt?+FRWD4K767IX8__4@l4?`fHTYqMDS z_nV}Jnr7*SFE_kr=w7R-{LLJHAtwIs&qF`o?evyx^PeLsA*89gYU8^<=Gx+(olNUa zel2(Gmo!$hi1^3%ry<6xZeQt1Ayu=mq-C{NwMze1bv80=S-?zDS3bI?o{of|HgM>OSb%cU;jOFUrcA$B+1)XQtE$Y&TM4Z zvXG&AThE(6JD${CO8%Z7_wm^K_XQufZCR`9+d6;GQ*h9le@>scvGbhAzBwmTW>x)O zd?|VRnm_BGzCX9Qe2cgLzQ9$h%&vvLwr!O@DPgIpW*F)<<#1yB?jO9{Y@4|?x*nbC zK5+2hL1p*8lAA>bdW~%VY&ie!?zH*;LVo-@<@>JUdu-|Vyqz<_em9$Z)956pcw?)8 z>GMm{ptC?ps-twxvMSzORjr( z%6pTry5E51t(4lYp+`BRr>tJ?cXNKFTa^wtU+G)s7F31pn|#Q; z=G!;x`d^*F+uPPz)<0W(ecrzZN8_)C%qzMi6#TyC(DCcfmVD4YIkRs6dfkAkHw*8^ z|9@Lv%=aeZ`Kj&}U0>ec+t$SW__IFl*Vipm-`m%}yE1XrX7z9RiKid^zq-3VY<^>M z%F?xe9;EKx>l`Lkp!)ISM`m`ulCx2F{_~VC=8;PI*Z+S*-rZNh5gA3dPTr8M|GV0v z{+{PUhF!-NeP_7fH?!!s^6`DQFI|dw-t)!yzWDy1zi+$WFNxp3=FW|lIHB^uysQ;> zw^v@B)gS*dBYWc1%AJhL+(B$<*82awn4AxNe)gy6Xm$Oxd3Wvv{JyB||3!9|b=s}F zA8jws*|L7?Dl09&sde=Q-yhb_?tk<9SYf1m^!Jz1e>Zc;Z^mByQy*KTA@23_g zT%Gy8`+LqWqvAWhL6;gbb;UD@s>CKNEX#tw;cDvuHa*H?8cGcB~)AQYrewVaKi@JX> zSvl$P?mN4sUa#sG$+g{(eD~no_?4@Q^^@C2EwyRUEePJ>C&E|cL-cO!BJv-li|K7Vh{Y`x4T3t1* z7C7{K1{;rn{+Y-C!Unxw56b`$K`MNeCymaF}KL_V~$F&wP%buX{FYE5jWD zt_{;)%r|}bU`6KdeLwH4Ts=qIdQa6w@73aaMebKWI1m^-If$K|t^Fobv#;9ctNSFj&s$8tdC&Vxnphc`87f$u5fiAHyWTWoU>*riI`1x41Kg3(CmYl)bre=#bO$ zbw;VDM3z50+&y#CX5X7ucaANcdiUPHACDh(2+vnwzK|-CuzsP>&cD}>^XKgT6>1%M zs_%`i^{4;uW(A9=sw|4m2>Jft!;1$SUL3fP9T45fy(RqRRsHNs zZ~I@{w*GrG>G$1|@4|VXPDEy;+&MG<@9ulS7T&?NSI@1V4l12LZe{hE1F9nU-bIx! z$(X&yWI=G)YE%3Bk?x!`&Za$YWzO@kt=dw1PUhm3CEpbo)^#ww$=&q%d5Z6x)!#*` zx?WuS{LMZ7%jfHFFWx`>_53MW>%*P?Os`+RK7TJSbJg`~E^{LVYkReg>(^|V`hNeB ziKh#;WnNxpsabhtMc~4P3;np*@AOtZ^Z7l;(&iWU4oj{L+XF>XKds&W$~a@qz2+Cn z7B=6`9sg&nq2l!3QO~sQS8Z+kvsbpQ!COqO<^IoFc`tZ};O3h~FR#U)^0F>}bz&Xw ze_d_f7a>tmQT=kZb0(hod)|O=>$Yuw&skrcYyDM@<$!5G!`=xulm32euAfrDdn)(_ zlk>Y@XR{Ln3#<4qaZbKz^!VR@;j4SM-)puw+V+%t)~mT+85XL)x|$zx>OtYAOV{fD zWF^;bU1Rc~sB4kM;+1SHEIpFOZdE!D`lO_9*Z&GGeH*p+IYU9NNW%Lk&*e*gFV?Tf zzWZv*`uYoV{r8AnUE6-IIcB1+!i#I}&)D5}B->wKv36-;_Cxh5|A%2y_C#LVwm3U? z)%_p$xl@-i&Mk6dJa_u^=dWL<%5m?I-nwW}?z@=Mk4byk84F@H4#a7z*gV^JfA7p0 zvsQjrSSKJi`^y32_oe=Rm#ZgacXW8}`2TXhd&%pS-xWUP{`zyb?UvTw)lLm&i-k3 zdSCNKHNW=Dmql_f57&SCU2gMq$CV2wc2v)=QeNl2>t3_RwTss_rJg-r<^M45*xj1- zN00l5-I~=OC+akHDqD zKTmd+b@Y43IjJ9Z%q+M0mE;gG<6iTfh6NKkcGv%o3c6*mmwjT+XYv1QR!=MVxAr;* zSMQmL+kbz5xL2D0?ccC4se+w%?!-f(!I-2sxlx@%S!-@6(|9<;Z6BxDXy9PrZ z&wq<>_Aw4V|d4Tc`Nr-ely1Jf)-Td~22HJf6pD1tt z|K)c(%kSa(H<+Bi%v;&F=??>ggcInF0M}W%+W%KS`y&a8wS9H}gr9p@3%&gF@{;Up zyUCzHy%u``wDA1qWo~n8hL%;grt4%%+r`9g@pvD|P%?4(H1-n5l$b|*{rB1xzEkd6 z^upHr`ph%aJh$7H-4D(`a-{m;YHd~B;ye2DqMw*PH?x*0<6FD;NW;TQA^D$rF`u^o zy7hU}r3UYawMBU%Nx!u#RFD7r`}X^t%ZAz0H1w}mi!sEyGn$EQzG+nTN|yiM%VxQk zhriBSs$I2&|Nqfq`Mx#elq<@cAX z1->U4MSptkwVUJ4wM#n^f9?iZxF%wwQ+3D5lvyU3lfKCusI}(Hbzagi$;HK`<(td_S&OdjxA}XYR($rocuI_?Vfq3F z>1mgYYcqp z^PRu@yxiur?$QG%nj9A|`|`s2$m`Yb`B_<7pZ(;Ka=f6^)`^jFcUi82J zWX?HWzsgN`3+U9{5keIbJdE31T4pwG$7nI|)6PJaIH(}#b$ z)%$-wy6|p&>73=^$5uEl<_cm<`~8#srA6DmCMV_1>Stvq+ArU;Wl6^CTX&_WnVMgD zW#J#w|7FU-_xSbl$13x(-}sil|5cUg9^E}W1055Cid49` zdiQ+py#8vHRtl@xmFt{>n%Zv{efjd`{-0;&my%T{86}2H(f&7m-`{VyP4n1AB{lzV z-w_;N^Rac+s$0c3{!6}Fu{Sn1Q&3lT(Nfp7tAqXL2Kmnm@%FxYxc#ba^^t?kRp;3c zEY&#h!lfYM%Y%!o&FQj(xbVesqM1g_##)gxE)nr&hPCdD{_Bx zyi3lalSk$~2~XMh>zkaxqVElYPBu3#g~Z2IeRwi)(WObU%lm4t-wM%B?L08y#Jj!5 zNy1g`rQUr#-dz3f-eq%7m{jxk&mU&vO`l`;tkC}IA**6|_Su}(`)#YfytufyJ$R9- zC+K8FuS?ZkuZ7^5I-P6<3I#T)G#`fICio)B^&PEA^v^!rjXkcH+@c7k~4J&qRSh1t-uo{P= z44->)Y-nz%>)W5krc;-%*7Ca=8oFhx$=0oATg}Y3n3--dF^}1E`0{c61Ly5eTo-$h z&Hdry=k2U*E=N>aTuZrod$>9KIr;iG{VGt;4gc|z{i~1sE#ul3J7&KZbjrC?mSZNbmyKc{~J1eg}Yuh&M*|lfKu03nx{vZ3}Si^&kh6x=17a06MAQ0`aL0931 zzJi8+Y9oBrv^$$S69o-K41(wid9zCXsV^xvLW zOYh1RXmJ_jHJnW0;_Bt$>g{fJY!LPhfHC7*~*;SM!ZTljtIKBv&o zYIo`V$_(pVPCbo3`0qRegTDh~TZ-D`lPQk^WXu<7bR9kQo}=O1)~#FDPAa_M{r=wG z;2?{6QVb8??NDN2XeeA8yL;QZb$-=+x8oTZY?xRj7#Nz1EM!2-e7?!J2CDhbTVu=4 zz>vO}ftev8K-xUdrtHm(Z*vk4x4F8zXP;+eIB-+r00VHf|$-rPC z4RWq?Mov!6!&dQW`tf>(#oylCOxqlpw4kPI(RUVxm=60-JH)sQ7#Qy5e}8xPyv=8y z%F3NbeB|Zl%m01hF2p?7@#RJ5_ks*N`X(-o!^{>}%w^9z%yFA0tYkl0S zi;JuHnGU?1@}(s8!~{Qk)xWY16?$OzEZe&E>*9X9s_%EpU%!4mcgv9z6P2sKy?ME$ zVCNd`ur0i@T2fM9QbFds041JDF3-GBS)23NfuigD_*Xdxr|8WacR3gN5HthM_SM_4TinUj(Wf(rSv})G=iuvokkbDSTW@ufu z#`sX-#r^-jt}lIkE%x8X+j+Zl54CWvo6pX$M-*gezj^X8o^+7}dHMO3pPyZY->;ecjho?v8YunQ% zcXVt#a$v@c8LZr5DJdx)XIK~Z9d z?*99AegC=H=K2;EHzb1@86Mbyih+uL>04Z<(>B|Fx!~+QO=qS-V$&jxbF;Y^7|ahk z+c7Xy^k2DhWnb;@X;Y@eeAedX;*zWT@o>co4RMeQXFuRFV5pc;w)@9Ew>KKE|7;HR zeRR6{(TmE3yUMh7^skO**%r0d{C>^ncKNy=$L;@3{Km~-AhnKxxgk9xPkgEAqAgKV zu5$gJdONptbH(3i#g+~pfy2T_I$SQ=hecFuzwDvrKkv^aZ~d)Xw#0n;c;UhY@LFtB zZ5x}I=?57YF37SYDa*G`Ua!j~@}{ZjJqbo|d_0s;aP z=Fi_>^Yat8{gw!w$jHdb%F4-VzMsB+y;^R@!0_ehYDI>E_}<*4-k1TgxbXHG&!z#}`yCQY0733s!gOr}89LnXDFj%l*!-eC= z-QPPhFcj#EB$x|HFZs^GaK|}#cSYgi8+Ac-pH51Bysdxau(E4j{?CU6@?DEA)G;#b zkSRaJ*ida5tUoW};goOHj_uBS_IUnDZZFv2Z^FP}Ai-t8uw&xksgEw6KiVm^Tz|_J z51a2(xBOns$N;G)?zp-(|2P~Kc>mR+r`zU=tj~S6BY}Y-K|v&ep@94B+eiMbi{4#Y z96dMtZs5nviQeHX3=9_pStS_0EMN7gwRh1MHveB!AI-O(^J9-810;Di{F_#*zHQ!x z^vuwX=jI)!+-k$v85jywxeORytXURWv%&p5A=sdP*?8v{du97spnZ_o0bf2t4m8C`tn_nCo# z;e;zA8^hh4(#^~j(zlP4r+<}WVqiFssd0c|L4Voq8N}b^%4K;N7#x&A9y^@Nz2P<=8w0}vaJ&fSU9N7@1{EX> z`5+(Gy0C%VSEH+R!2P`<0|UdHW{~9`otLVc&VA_=>AGhOav>+kg%@ojO8Bo;cbx9= zxKPc+z`(G4Ap>(mo}0=m+k>fXyYG82FfeQiXkc_Oe|OESx~21m0whYLK<-d8J++tb zuIyFk_n>pL=X5ghFzhnaOiHWY+x|noF3eCh6k0#nwu#%pxQ|0=?=J4C*mieL2B|%;RwKS}PMLwPm zJal}`Z3f9gC41NHS3wch2J+- zcT@$=4*g>n&hm3tRV)JoLu^0;^5vt|)<@?R`3CM2Gu%9X#?AYY;Zs4XK+V28 zQR_KAF0a}6eoI89#~Tgdvv1A6z5S?|6vA(~bv?&+%XH9b>E|68*%9%_;kz3n8$*AhX}99tqZ8gaaa>Fj zzj}7jD$hCjLR)nYB?&$Jr*tU3<+e-5{D3*Tx=k1u8qz^&*Zq@W_Y4O?;T7LyKsU2B z$1VV!vdz%uzz9AcTms};(8=u_ph}5>p+ON;-7_$ZDi{q1&;i%vriTN*mNI;M-5D6x zg9>Mc3p4xs{g=zzm%f_PtnZY~HeEi=X|z`v3n{t3;q(Y3}~(91IK$;j>IKg9-}^m(Tz8 z<8goa+gqi_Wy@32)7x8HcUBlMoV$GSV&R7e2m9@QWn5hq8vbk6n>RUT2Pb(1hK9Ze zm61K`_U)^?wkC4w)T!bBH@$fA;=zN2>pd(7R?6_XM@B|QN8i4>Kj-10);mnPp!$R1 zfNsu>4G%B6%U`YLV#rH*cV}n6{l6Xef-6KpZGDCU?&RZrpgSDy1v3R>+kTmF zFStS)90G1O`2r?|L1woB{|gs40=K$B3}+L^Pib%d_3imTWmV0mCUe1Gc>fjt7}`{ z-KZEa2n<~K-GxEr^T&^#zP`TR-qHpM2Y!5f{P)kFFY`{E4g}rGW#e30`gLdV^NNpb z6}+?0URxV&-hFh@cLg?ZYGPoJU^0gsme143!~;sF1}q?tGcX(w0JWSM7#N(vRWAd> lr~>ena--ouz4Y+#Ka;6k@PYH@E~TKP=jrO_vd$@?2>{s-I`;qo literal 0 HcmV?d00001 diff --git a/docs/images/logo.svg b/docs/images/logo.svg index 57639f32..5d5c7a7f 100644 --- a/docs/images/logo.svg +++ b/docs/images/logo.svg @@ -1,7 +1,7 @@ - + @@ -30,6 +30,9 @@ + + + @@ -41,7 +44,7 @@ - + @@ -49,6 +52,24 @@ + + + + + + + + + + + + + + + + + + diff --git a/misc/media/logo.afdesign b/misc/media/logo.afdesign index ac802e0d835f827bfe21dff3acc66510fcf65b77..023ff1850aa66d0fcbbd288f021bf35fb0eff694 100644 GIT binary patch literal 26568 zcmZSh@9oIVz`>ALToj<}nV06L#Q+A!p!7dYFc(g=E2+eSI7$o*41UZE3;{*?CB_U4 z49aeqIT~B_|1x;2^=GUK*Q*Q@@L1Ww!nL45$4$zw%z0+kbiM z+dIB&F1hPT&HAi?h13Lvjn3{?|1%i{g8+E+TdB{@E)c3zGsSho4mv z^ZUBZxGFS^J-xTqH_mjs*do@-GuQ9;Vm|2N-A=;8b@;bXljr{Dv{#hh|R`~EKw z+x@uj*Vb!~0(e~(eVFi&gfIR0{gP|N7iY+k=}u{_`Kqt7)h{ z_9^@Ne_i`aKemQ`)YD*C%Au=MqrqUyExEjS@sawtm|Py^i+^LYdyKi<|YJFZ(=KmC}>@Bi`t z^CulOefjdAUhn<6QdYf}Bc^p9pZLLj^)6vqUeIA1hv z^M)y#WW!lJGG=)!W^mZ`p^?{bf&L1A$EH?+jGP0kycaI~{hpbe`Y+z(@c(y@;xqSL z{-2t`#IwL=vsbFtTtCf8FRhEiYM*HNq?J8%o z=zll$y7&(+2j&N%;evd6|JlF(mk7`_n7_ifUpx5g!ZVj%c|Q&Q(Wvn9%gO-N2`SGm zd8?b6&A7zr!}W9b?%n@)^C*6+X!JVW(5ZPuESNp1T7c58gx-}>M3-~K=4KleXsIkxlW!|KCr4&8q(Q|EbpfssDB6&EMLf9cM9FvhVEO`(Br>|M+jc>&Qm^8-ML{ zb$;*nm!AEvp1bzhSC-xT&-0xBH^24Y{U?>1Z2$Yud%9cc(T>&64lEK4JNbX1rbV5} zLUt1s3C7dUIaf$sm}a7QOuBhzOHvi8vc2mzX1&=u z%Q$&&*nP=!yLsmwz4yOwr3Rl{&t8-3Vp+4#d4K+Q{=tD`)(gHJj4iYYZ2$j1Q|qi) zmb6dv-f2%yUH^UG@W-=5Hg3JiAD-!|eY>@1XTbY)n_LdQ+s~PN=t#>&gP-@+bV{2R zpOrbox_kfGn9ASv0sr5B7CUTv_r6&w@ALn$x0bDW_`ss^|N0H*Rv5AI7kIcvWXxIk zghMXjNKV%K+>Rg2=l{n)5iE`g@(`M5e^xWuq~!l)nexBCTW6bow7a(?U{85wpU<2B z(N<>*!$17@zQfB_EF~%Pwto7v9Y#8_|KHEtWZL$i{(pYq$-e)w+cGwYs!n}X(zf(c z#jNH2Q@wicXNk7OTTWZ4AL=x7=~an;{6fjaPXo7Se0b4m<&nt9y;J*)K)8y3f@#-} zNA=ei9rT*=;=iqMvvJ#kkAzw#9)OH4U<@!CJ@()e5d*MIr1R~!*_{{M9S z**weIj&M%Ay2Q}ODmycDj=oIz$2G^aZWyN3Jl^WroxkV5eQw&?_M@NY8hZIdV3!}XO!?z z>d=b%Qm?&A!}FnsQLAQ9<*##wTYtF~i`Ff_#A5$h$5(ywsgjqwG__~8@z2$%aS4cy z|6kg&K!1@Fn?T313?o;r@I;GAMKgYdXa;a^hT@#@Rwo zMgQVETmSnN$9(>8xuMZ*<*6XMsVhHB375}ojjCP!L8R{CO0U|b#i^5y$;kCOP8JPt z+;+LXTs+h9j)kzysiR$+c^0t9Zm6ic#;u!Rs^H2Nb?0D~G8d1@%4*(6v%&djL4#`{ zU!U%``K{_wb%h^uQm=pc&mHB;mi~lg!X{`^ji0jb#j(B(hXhr=bmZ{|tvLA7RjfG5 zp@~%_fz|Pdq`0mCL)YbQmN`2Dt6%sib4&_ZA~Vk~UNbbRcIgG4{gb`6{_KY0@wx1fNNQvDQ#faHaCzry#)};sH%elKwJ*&Gnz-V!<-Cyih`uM& zeJ%f-m^Aa3-}B6^Mr-}lq@Eo!IPu7#s5kW5S8WHtTQsti_l9)p!28Se{}lSSu>n{NHeD8n^A-jRyTsAFtYc;s4|uY3Ab1 zq4_qS>$iu<{`fC@;eTX+=(I{D9?1(y?V^bXY9?gt?nrcUKk`TLhm~PO~g+QV-Gm zK$r7x7I0~+IC3WGPU=vcwVd}ruyG5Ml|su^H@=W{5>7#kZyR~TeFB^Lrt*AD<-X!N zXR3t%ffE-xH-&7yX5q+Ycr;kqTH2|FW6^*6Wz}~RnpdCJePB}TuC?+^XyiQgpryb3 zmK1$m>KFQBu3yogOrbm;dK42rf`w>QP+Ebh^iA?Y@w#klN)}HUBtF;`-Wp^Gvdb)870D zdpCoi#Vx8w`un2#^xiRSDA|+1RQH>`T*dQA#w_)#8GZ7VMr-#KPUW%>k@LK-r8Cth zMDyz+qh|T*_oUkTOxJoPbIv>C!xeacrPcBCCUYN#TOGdI&U9S(rOV7+>pJrY6fXkAKIOzqMj*JXZcU7a(k4s^&$7j39HBI8!z!rSdDv>|}E?u^ug#kQZ6 z<3BP5moOJJUlnrPBam&qN??z%{hONkEWaErb!VD1p+$O$ z!Y-@j<`o+PT4pc{EfUl^s(S6kg%)<{3r=iCf9pGad3|3_F`BD#$|hLzs{6C6ue6;* zfAp$f`MLCv&PWYLk_r)9r){*6yeoC@q3t9R@?__XDZ zSAt*M`~NF`-#4<5x^wxj?&-O2{%`-4Qt0n+@#f4$HK(+MSB5MLSXun%#2O|0Ni1#i zy;7I$6HxJ;q&2^;ZrPqA8`rc|XpqFaa`ic0K)ieZ`RaTBzW<>-$?? zsa$;T|8b8Po?rKmCRrH1`0wS@R_yqH_l?zJ%~4T0PwZDt-kZfO7kgX8<&v9pa6r&8 zot=&;UuRwywwY}5^VO0CJa@I%>wjon62Hxi^}oGa>+FVKbFWKQY2LkGIrrhe{~rG{ z(~>{@50?CDwo6bd!X++SSMTVPtWzIl9{+cD`2W9a(p9nMylv)duK%5T|I*)j{S4%k%$YvTu&HOT78# z7yI$;|C=xVzwOE|@c&-EYt!rh?h8+sq#pdQJ=e4Q=mY&<_g_YAmVB5{CbBmBm%RI| zb&oS*FaOuh+_g?waN_@1uWLDOnv%!eTs}+)JX}9*N>8lv#259``%G`XNISRn4HD;l`D16W$4V@^7{WYk?+SA|8L)GwEg-2)OS4ZUj8fFxJ^^b1RN+JgB{Pnb5_;O-9=<){6(Jr}8*GR&bc$G53SQVFBGK`F(|d z>T^mj^%YoHe6#O9BewbJ+TK(xKAB+U@=Fp$=~p_74#IH#xyhmxU@09xi?^ev7^yovLJVjE!TG+Q#bBk|#SGL;@dw zO=HRHU1M%Bi;Knc9LJqa+vj-nUY`w;mgQ7UWL&s5&3D0>6r+GkUt1R)nYMXa9*^d; z5M~z6vuaG2TDdMUG%eZsaE`MyE3m&{9klfjqU8>ckgYwPyS!;_S{IVd3opM z8^5`Ilixq*3AOz=SBXpUSrnJ*iT~*lKHn}ao(}@@>LzD;6ECRqn*4gE#@?*mo^N8t zzi(mTg#YdNn`CFyEjJhccmJTd__RwcYK8ZNYIa9Ozxi8#>fD2Q`%lkT?J+UyPPA}L z558955Vg9wx^UK+E3^MA8kcWd{Oozj~Ydynf$*@#|2~y%)a@nQ-kkP&G7r+H$eO z$x!BWn~T&u@pVTzJ56l#O}DkiZIW8;^(S7lP>1K&exbv*jgIdnOwX12@`z8jySz*P z$Nv}IlMnuxzp$rqcEdlvuKlfCnccAlAO85>dL+Ta-WU@T6%`S2U|!$EwQCw|@BZH{ z5qZ7QcE5MenuCf-F}&YCcqEI;Jo+EIMdpU`>*=f4HvZduwNt*G2Uf?PG>@M0$un_HXQkHw|Mdej~BPRCSNjc(&N_Ib$N-FW9SV3 zG_J2R!xVnbJQw<7q1TE_;?vSro?3ZTuqR+zT7}WVQGP zOY0uI;&H#cg^NibY?u6x1?(m(f-cW>`xMfu9+Vlic3-HcY2C6*l6C276ZdV>X?^m_ z z%~YAHE*dH2ec63h*yHI|{tv?^E&p^n{iMjO)R(OBI}W%^TP>0)b^6kuFy$4$&Rv$M zOVcu`U0i9k_ME|S#us}6QYXbsylLXU;g8o6udj_M5_Os)oPYe52gl9y3BJxWHPh9| zJ5i!Z?UVJ~<*$rA6MoD*G5L?jl$BGAJ582y>qT7nAAL$@+O@+=jyMVjlsK$0yS5<7 zbxKp8rgg}|$LyK~hj^z6uj*(LuwOFaz_M@m+SV-SWH@}Rbe{3bpHio%a2Rwj9PW`! zJ#*=+S0}^c{|o`9=OmB5*ifj^s&u>Y<-c=hmTvwZ?l^mA30v9X9G>d=jyEOVEY9It z`uly;rcJUl{##tQQ`wv;e5}Ruc)*sw|3$XUY~Gph-@VZ{^6<<5>KU^)wOzhd5cgH> zQ0CqL+8yRE0yLd@)RGcp89fA)Iu5e&u>II0(0FR1z!F8S#BQyKhHjZj>W(grOFErf z8lR+c39Z-MI8VtVk=eBHl#h`3i-vwR+ZhSnbrDXjYMxV-mO6cMp6GGaT`KtD^utqM zInVVlc5QoFxQ9jM!W z!86@IG0VN^lW@Gc;9zr!*91ML|L2<)fjstJtuEZ>+0jq?V-2^oRL)!KwdZ)4#kW)Q z=FgumAMSm+G49nK%^&|WSMRc|f0&^6<8+Uy$$Ql~+v0XdshX_+9qqPk!+&1a)0f2A z{=N8D{#I;q`kQ+Di<>@+G1oteC=)r_DSUjB_=biRty}>KOcuKDdN2Ri6I!9jsEiA_v%hh!uL%FlmE-Q?0->S-BuL2p8j*sg51U1_jxSJiI-_IbF6j9&EV^5zV=H;zGN@=7TL{W7o^1kw*KFL zx$)bD|M#Cd%q-Dfz?s3h&}Pp9Ym-&Xvz-6`zo`26zkBWlsVQu89aA%I73#_Tt(V&G zSVb=ooO z|F>WNx42Z%{`TcRziG!_{+GM7$>bj6{k0}_|Lsd-ErNc%XI`)=;!Ztx=9g0M4VkB= z^<7t;y({kW61~@d_v^SbC;zOUl;h22#d~two@EoC`n@+lim#=v<+fUx6yWVrlp348XuO8DqbMd>*-o^QmRk72apUs|Zwr|h> z^Oj|Yww&3ram{^JU$@wWk|L$8yY{DNocdNTy&yf~l-Qs7O?vayc$P^1wO>cyx2wO=r~C;3_Yb^c$j z7%z24c(YYj?%Ds2Mz3;ZOY2YPo%&qA|HA)xuW5TNAOHWmIMw_)@0$>|(#m}CU+d&f zr!U*I@V|O!1?x$N+C}eQu|^hcT6$B5@$dVIxw#JSL(+OmwVIE4{g1v-oRj$P-HXUm zVJwsp?U?*5vd)xNo~l z!fM{~{=S9E!XNVGV%L}&Fz6 zC3(H-8ZF$MJw*<}j~jPgkd&R-z^m|+VL_*OnuI~4=KuR&`_8SM$ob>HeZJwH&;OtE zT7&u%cK$d1=PmqPnIEopt5;DAPZ+`y#`E%mK7|AasQ+DpzR2w|w_?;PFBJ5W6@@htgMydtEn^|mv|rN` zT6okCICAP3FWAtr;jn;4#laT7HCl=)dF-oNBFugqVgWOsBvdkZ2}VWyU}!qZpy8U4 z_>h@HEnuNjE0?HNV1)aMt65v4dP5NSQn6PPETOfhQJ5pjWZ;W8o?<|GShl-}I- zzwTw8Pj0}qyg6rYT}n2X&dEB_=&^^;^@9vIFCDlLzwO)CI@N#WsnXpJckiz}z3soW z`*OY8O(ur9oS*+!w!76Pe*9Cpd}-y+ljiTgoP0EI%Dg_)W#4CA^*MNb&Bm9uvL_`b z^&UK^!sswXJt+L`zvCyupVuExdh&a}qwb`zU42*Gc9(dkf0}ygz@_KfXA?yBUivvT zZNqQt=Ucz+-+k_9%0ZcF*FHzhJz(qe@B9Ia?1}&N(_<`a78^<4`v0!)VS(X~{gY=F zKArk5KSZqi|99*8I^6%hU;JCY`+x5A7Ott1fBv7B{{P-hG{Ppr=^AgC&B8+>Rwt*l zZdjqr@=WApkVZz2A%{e=OG1gUvt)qCv57JsOomTC2{Rm9#IVUigw44nQNcM$;7Ec% zGXn<$g8;(;9tRGij(1vZX3txtANFk5U@}}?zGsE67n@kA@Uhde=gk}h)}Ab#6RESq zOG<+Ooj}#Rge;|Pv-?j<&v`oe1S?hV; z1YW%1{J-cNN5`%GUr#ZX_aC2qwdX;~iASuD_a1(CF5?n8tSFF0E z@nnn5@r~cCL#KMM&ibCQ|Ek7aPQ9G%uANLpFLl_`D?^- zKdofjNijdM83Np?^Gd{TCARU`zW8vrEhpa7)_ks~hWL`|W3g%L^M7iZzPMu+_~RNk z@26C^_oeq1zudB!iPP6J*7OsN8{$iEyhKhE2=878>H#|I5 z6ukQNnL?qMGdCEH`rj6azxurW<9mkYddJMw8;Uf4?-1la>A0?YWr;<0P5;EOlEw<- z=j%%CFSqxl_pfb}RE>3)>16X=o2sE?6wWg3QP}>XCt_RQ{qeBnT9~IQ*xGq$gK7am8HAj=| z^X=uLn`iZ!>+&yRgej!oS5=WX!Dm8{H+wyH`mXSde*=O+Jr zenE;y;_2$e>-Wwn=`_`_(~xiaQ{nmCx#e(p!dAH#XWN)IBrOZOI8D%?(zs$Cvvhy2 z?>pH8+kSrIMBIQ`+mg!aCii*C#OITvlbFyUF2@dK70?VYmmo6|BU{uXd- z=8ZjaY=P&3eX0Hqa^@a-w=A~rDmA#!FPiDwzjevR8TV$Fw%E>jsOmSJ_uI(@mt7u9 zCA2ZGt6x{n0e}61&Hh#O_atudmEY11(5A%GIAh&L>(&Pkr|YdTz8&Ki7qj!nUQ|jejXZcZwfnBzu3naL6=I`L^vh{`O~%Q8&X} zeEVKMR8d{s;r|1XFf|u5MtbV?zYu%g({U6NYmIXb35IM?U6;h@d1>SN z$bk09WS_MEo7m*arPHSWpZ?ROUS|KAx#=r*FwV43UvSHIO0&xTxq;7byX2Jitt?mP z&1&stu1oCNOl$9=OoJxX#V!o%*^w0)3VKfgGj=Wnfaf~jGG)}*aZ6*-hDj3z|J z%3sVnTD#`Kww5J^-YScC-ap^$&7j)rZlUp@|NR8dO)SrjH%(pkbdkGTjwHX}0^R#p zSz7agm)!Sto^`i_d;7>`YAI#S#EIj=CfXA)apKWvXf4zu0!SC^=AaHunUopO| zwHEg$ow4G5rfiXJcJSif3s#&8`P({}^p5$-ZMWa7t7~@YPu2r92`}Af=^VF)sV75t z!?a}|+1_6+yJN}p<2LJ)ea~MFnz`YEH^&$A#Vpg`3NT-`^Vs3OVS2M;j@hea*O@g| z&~VC+OlkWqyM?s>Z#6%;KH|RW;!~W__Vb(XJu>*ZVo|5EUt7W!Z-<>- zjTw{W8nq)oZ`5p;V6Zs7?AfH!*^S~}Ew?9hb|%~3E!eOUP2D!@0ZnFPU)4b>oTiGR<$6Y^_Q(G%%UXs`$pN)oSVe+zAWT z9{sxT*W6j}KKPu>zH_STn9q|fN0xDH?^~iiqr^jb`HV?cEEg)5PyW^Vz(u-Zd={=l4J3$e?7d(XAh_B%TebDooJRFTX9H|{c9YHkqXZjQT=879zwwQ2Sk#KKhoxu{)e9UaZ zjiN8Ff5+Nv_%V53?;mLc9)X4*7H-At?&svv<0`7&GO|zl>i;|Q#7ee37uS7?cDH=6 z=)UorlO7C=6OX!wNZJ>6HpQIGsN3=~@S9~SfKLc=H%b8{A_){_RX8xGX3uE<=y*k9aXt1@P(D_u(DRa`GeP|AM@P1pJuSD zwK?QL{k`|qEF3#ZSFY;Fd|9+}Wv+zGuO>g1>%0fTE?Aj;+g-b+C&-}ML_zP`3xk7? z5?-WH7afPQ2M8-;?p>pHNBab9>hNnv-N@x2?Dy zYV+(J8$^vcJ5!)zHmarMLE&WatnUeJ)307tzq9a z`A@>xjJ|G3=9`P(zpxdMP4v;*z;D#M{>A29n{oy zc7{lr^}gi0Z@Fg0oe0OQ7<;43558oy+<&!1`}Pa#o6?U@f4T59UG0IF%1`xYKC(Y2 zuFu$;b6Yg+P!I2)x2qFWV`uo^?ECTj(;J2p-?sJdIr%}i$+SGR-rZJvlhxEq1$UqB zT=q(WPuFA0>ex$R96mPPEAnT)oojJSOCxUAR(9j%eFj?=Mxattp$bYq*F8`tN)_W_b_AsaI5ISp0&4!Tu~KKpa<*PMkPZ6!_$&k>xcs>L~1 zy6RffIlUE+CwadwI{ zu1utiZP9(X%C(bNr{4+|Uw+hy`Tzg_|CQZBK#M^cKmfK<6xT{pMvx)~1_oOO1_nk3 zCNS3nD#r8_!ho$l4eqE*Vq#-pU|?rR$xqfxNi1PvU|?7V*2BP%q7Sk=BqKKoB=E_G zfk76co#Cks0|R4cfS)@rmlPKR0|T$8hf5Fx14uJN8wbe1)qF-`3=9mM1s;*b3=Din zK$vl=HlH*Dg93x6i(^Q|oVRz&D`KwB{PFSq-eA+k_ts>coU|q_x@ao5o$Cg&zkd?Z zmoh@!Hl>`r_S2?c?!w&Yb0;p%`981u+;6X0%Q88f^l!W^J9YZ>%{`UIWxH!P%zi8! z8oF}D3I$cw*3n^CPR`EkyESXYV8zr&Gg!Je!@*w%fBw zfkS)_2Sbb9M3u}9vkT(fdL$0^*Z)a2GBS$TRiYWaEypo%%E7#83LJ-6m6;e4Z%A1d zsg&QZHGlc?rOm$|kJrTSmuo+4c;l=Pi{lOj7X}5M?3F85-YLIdyJX3d4~Mw*H{{*5 z5@g|MGSKN@aFF;G6dYXn<)V9MSJ$KZf6wc0Y{?YXo1VR4wkk)H!2%Zs1&$r^?0+7K z?^iG|nDBgl{k%JOVqCg-90l}Jl^6tCHVX*}?fC!iHzyCzlauQ6m#kRf!NS<25VHxS z^KD96TG4si?+@O;KfnL~@BN$0-^Yc7%DzeCY*P4QEy%#(#6DTw|JlCZ@0_Ql&a3;C zS@!0JViSX-fSx7DY(Jk_CY|^HJTt$MV-{EWRP@`syV`Q>98CqDf{YCcC5pDTbJgcn zG_6{t6<71owNKXC?1rxpi{lG(kmfv)Z(l6#4+;)Gd}*n-rmn8+8#R!<+#q|+RXmd} zE%Cfl{eEvqNXUmH!u}l-LH2Skl3Jkf#bJ_4+SysG2R5I#J8gX4<}fqAjYCw_Ersnq z0(x=Xdcih=`t&dg!vY zdWM3X#3JSgO6wVwIck+o8*NTMucn~T0CISEeC^cV@AuDt^5n_Eyjcn{Q?#_TD}TLQ zuBoB1VBx}ti+CAg^u#Br2ue$zzO%D<=hth|oSdAJZ_<7)xy-QOCTITJru_b<&u?C- zH}hj(x>-F~MmMl?` ztNW2Sul`?U*qVrgvgLOg4<@`gevRwU()_=_zN)CH75#p@J^o7nWga$WwaK0H|9x5h zICH~qksNDI?qEm#WddB{ojZ0|9P5`q z|LN)J%1k+6gc*#Y`(c6LWhNgg{Lg|muo?R zf&Gtz`~?;=I?>yBLPH}rrEvcH_pkZx7QIRqhB8(&x%dAX#KfOWJ$S~2L!_UnrMu-| z!h>1a>kj6aZLa-YR`&j0?Y-w~R;^mJZCh9fp9l+MgP^io!SlK0i!`|M_x)5`wW{dz zS#$oI3dfZhR%EuhRi9^{YTo25xk8Qe(1DKr{^#Fr=cgZRV!iR!?$4s{2|qtQT@$f! zk$jXI$DyT4R#vlqzuV34d-=z4`+tpb>(ft8QgtrsYj$T4NMM@FfB4rwxBK6OTNWP6 zFcFfMufO;FqRHDuGhb?R9BLJnlRM|kZ+irkQfKGy^K@`vxE@o?TlM!sROS1Nz0tiB zWUd`%6zbt(-M{dw9gCv?OX3L*F0M!Ss^6Z#Xuc9uNv z&NZIrTCm{GTmg=AW&acn?2rECSg_`ylX>Zjj6$bU?jD}T4-3CXxC^j2N+<;d1?~9# zZg=I&rPC|k?R;KfArn*ibgHvba>8=vpOzmDj3mCzZS?;S_lM%#-;?hA7u5@##dh{=hOX|r z|L>2zm>}S$ANS_pvQ3|uOx_)upd+TTTGaQli-H2jp#~04&W9h5%RlF@|G|9T<}=T~ zfB$@D7$jcETXrO-j&bcOuCKl30p__)4`*C)n>1n5rVIbC*B=y!UZ%`(XsN2EX6O9B zZ_@Yt`E>fvzwi5>Z#*uSd})bivyqU3-CGt7J-5o=xm)(^Ihc3y!OvIPCH!_A7cVy7 z6>=2ND~ybc1eN$Uzu#=O`FbVT=Ff-2XXaRbZt7?G63uYoQi#8^uc|NaRtp*38Grw5 z{J}0HBG!8G$@Tc@2VZWu7!vR79n9?4&TF5o@XvAK4BdYolTP?uQP&gaS-4cQ*(r44 z($o#niX4Xw6g-7)-n{wenfd;Q8;{F9emXtgZTV%x+FxI|rC8>KGc3rII?nxl!|&VY zEoN;Ot@gfr^Z9Zq1w~IGwQhUK$O#c1lnThYxJx?U6&DK6S&}V958q&OM zVdgveB`TgQEjkoN=ZT- z{eCX%<@cTBTXM#yWdC2?pnwV2vu7=H^yR(v@o`2w-@@FxU1dKCwm+XNUn_F=&d%6< z>oTQGOcqR?zWD9^{$D?z7d-Ph&+F^u^!b1zPG< zJcU-ST&WYiZB6OS1P2F($H#hiC*-~eW>}!5cgkNj`Eb|%iTmHK&ih_!mGdUzvs|@# zo!hrFt7q=}5O&Dzi1N=5dz>$uyjp$w$Gg|xZ|-|m8^aep{dd?@x#zE+p8tH-eZA4g zPtrU7{E|=pSY95dH+}M@$?TSU@)w_fZ&>%GLM^^fc(IY%LwS)dKl45FrgPsRUhnl_)mv^4JbiJLc1gvY)7{B`RrsVP@i2hOkJTDw6YXYT*o z-i(YVjMEhS=e_y7Gca@sznuVMg91mho|012@_AKV_y4{txB36)^PgAY`;|;hzc$6} zzMlNdZ~5HjYO{@hj=ERB`D5F+dHaUqx9n?bUiZ(jEL?GX=CxVNie5!9%J2Cjy{+bB z*!H>mj~8al(ET^#%=76Bm;d+>XBYP^`d;ln!B5-QAI;0QTmL!s)RF}XCEM?HJ^XzB z;{TWa9*f=P{>owE7ud523_Uha8?Qpn;=BckY^iEqf zDL?M$@H-+U(R4WP@U;;Cqc4-Mgw9p8(3y4R&FKqY4QljXfBEr~edlxD45_kcW53h;flFkaj9U)&-2%#@!mj_7>U^HfxPQTHd=K%&6*mqtkF-2ax3A6@u+yuM-40-b0! zzx4C$rsj)HO-ujoTBg|PWLLY4(OoO?d}eu&j%%v{cZ%!b8yXu^A_S_ur==DZ75#YW zUw^9jyzTK|f7_$V{WeZ+ZpRwpwukSH%uUt2zqYP_mGN=gcMqBmEt*;Q^li_nIXy@1 zIrbjxIDE~=JJ^=JJpak;{^<)d-_1ERYgyrz1nJ+~*Cw92BW<;}Z|%-TTWPz;DG&2Z zLUZrR2?!nR(XMw0dM5I{v_~z(_2AQz3ztINQ(t<<@L2~cSIgOPcJ@5owY#qQU4$^5MDE!g(V3xRR^u z*1aNfeKpVLuDHE@aqj(lwbr7nsXvRRgvYb8?K9h~_y3Rn{|_d*%ROurk9%;@ zU0!wP-{;x(pDh^zLR*=4?|T^W&#eEPy@AT4qyK+#ycAAXJe`|=@BFJcMghs`J^?p4 z>?+@$yms9rH1+K7`ox%XHD|)!2R2Mv%=+g3FHybp+hmsiWZae?yR%iUj2wvN%MKsz5Sm{ZE56d@8x~ZE(b(g)-W^pI^W(c z{`D~b-TSAFP8hwK|G($m&gU0m>uX=H3o0=9G{bO3>>RE;=YL$?|LXWUlp6|5%+q}l_#S7P5EZ!a5c!}*t zx*W^zgQm>YkCix_-Q3*Rc9*|TyS**<%skuQd(NqzHoAH9=ANI=X5T43Z@VUD=cM27 z_tzgNV)-G%V3Bk7&D}4RujlUBz0oQyzTSN1Wc~M_EF!NNCSJQ55Q!0?4<3xr5o2Ov+Ew$uYte#LT2^z{ovzI> zo9+JK+P+>p%knon|7&q+aNYX&ctyJP>uc%%-IR|UZ@$VE%X>6?)^8cvoiQ2rYujHh zw?F#f)oPFAsyzOzq|d#QaCNvHy|VBwOO?{bNjcvPZrmnJomr$ z<*2w$@FW#a7xg8|Cyi!Z68jJ`ed>Z$T60+${FYDloozO8`gHf8AR%rsodZvcHrD%xp3LCwpXvRQXYQb zefQq3+-GNI_Q>1si?3T2RcLr?WpFNIgNowOmEt0jmvb{L(X49@B)m>Xm${iR_ zL`Hg8TE4sa`>botP_UGPq2*!Xr6r!AzHH8|EuE)MovQOxU6(0! z_3Bji_>uNlW|GU;n51>gw?7x7Cl1aQe-$IJnTc9n@}(+>#;4(9SQPc5_pz zjCEPhvu9~ePEH#lbX;6rE1yPbXlOi`TYfL|uPYZ9SJI;+oqwklIA!O2*>~A|_QCgm zgd8;I*EIcEEU)P5vcvqxPetFEOz-!y_ujqJ#>IT-T@gE@?YreW+__m2EheaR?ks+8 zlzfawN=oXNDnmf%RQvxw&tJK68uIMeqA_g*~Y@hZvDIe_w?`x2@C($WtiaPyzuavsAjHfzRtdxr*RqgWi>|$Egv;uwc!7M9<5$=OTS;J>e1~V z>)BLY=T$yh+n5-^)Kt5Fg5P`7s0T$>E7pEF^DQj+>D<4%8gC9gO%D%kwUrJJ+L7LT zyr%s@N5mrG%e8Vw>sPMKeEWx)nR%vZw%Fz6{@-V*Mny$=IXXIKKhNJ7F~c~WFL=4% z*Hxmip`k}_ZOvxv=3}@BExyBYkbV zgu=b={cMf`Y@Pnf3|)^FZQOP!Il3a}{iMAflUnARa%@%=h%UB$z{ikvs{5kJ*01Mw z?OJo>^^JxJ^EO}nI_>baGba!KzH^$vl(*Vjxb$kO%&pxslUZ*UTHekp{=%l5T=Yy> z%yd)ubq;Rs$9>lC9=uw;UQOY^-dhV7ENF=T|LgjcDO>jXdrtE3@ku#3NmVCq&x{2N z6ij@RpP!qn6T555qeqXf)$afID?4w`N4J#7w2O;e*Fi0DK4GwIcay$y`2v=ni)@?=g&OEQqphxtZ=$tvGJ{w zlmAYVb<@@V-VwC4gF|KSJiYU4t}m{Cq4X@@J%+pfspue<5mm%tsY&r}Xh zT(z$KN2$>nDb|&L|Ep_rIN5*JWGL9VXL>9v%h!MUG5H3G1~a3X8GM5`Za0}f*Yf?p z8SB^UeYZN7^!B(;L3-$P(a+y5^MBbf-!EUG_EL=6*`0iw=Wjj!_jdG#1jhfL^#4EF zsB3CE_4T^lebeKrR+i2_nDF4^ary5nxOF{+l8^U2?b2TN;Qzn(|D(RIUA=no(xt4c zuX=@riCw&SaV=xhs#Q6+xAo36PERtBD6pFQJ$jF;j=XUI{xb$lhtN*l#;yMmp zorU_0JpAr{9uD&inVp<%%R>8Q9RGg1eJ_5ovFv3oc9tFM??4!;J4P4hwr(uFR`Fq^{=9<=pDe=aS`s`s8qauGgxTI$D^pb> zR8x7w_Q#heO`!h!j-Mh7FTS$*)P?n*+r4bhj@in4GG}$IU3pLW?z(#kDk?_j7C+|h z?fGoAZR0wrN7bb}>jZ7JCf~e%uk+-c-PZr#c}%+Ua?QDOYTWZ~U(Ybv#>TK})uW8? zUF9Di9o_Tkl=j-me?SAuTeHQv#r2l#*fHbKp+nbdb;V8}xH@Os%>c7Jm&jcrxu>Vu zm0yzeRCcSVSsxm*pmaM+Kxp9ix6XIBo_?+R_JZR5AWrVPMz0c*be6a~dCJaN*ZwAh z!@x0_Uznv`doNGB*6+9HA5AKFw`_U%Pmh+mr@;&gCxtF{O3K)8*jyT2`qDGxXzQEg zsWzd38JVAAlolFGaPobh^FUnY;z_G-x4hQpKG!xVVYw`Jx6Lo#Ws*us74vcKvJL-t z>qmdLoVLHW>+h`cufJQv{4VR|iJe8qV>$WxpLc1mOSrl!bnR`spHGCJotx{O zn5g*lX>NZAs0cRub6<8kcjAjW=~xkpzfvxZEc%W~&^;oJ801+SUw4?KB(KmYo>{P37U*45$b^JHCb z?5nkAWo3PHbMx`!{l$~nf4c+*?EAs=_JrVfo&TGRwk%TGawlesusYkaxYz z%(c4uId;#guD&<{y)D10uDsv(`s=6r-jx9^o1!MF`8arayr_EgN#TDq4wMuYZx_sHw&+l2l?lsl#_kNeF z{{AlaubqlYOXlTeU;irUiNA7}U-<8B{_~5M;wGq+Y`_0*=Prl93tZd>71L&L9tv2z z=eORv-|vK1=iHi{TC)9a+q;?R6O@$N?r1ce-`>2~NKMf8RI$b$jePFqeX?I3X7v8O zzjoQW!?XS`*?9Q_f8U22Xql+rn9#j~W!;oXO*LPyUeMdvR`yCH z%KvBsi=%`{lw;$IKh{B^Ta42eY%ALkeMxh^JA=+Og=J^)WN1CaDxYof^&;_PNS^(w)?{`87h7r>1OscP=#XJ)D{Ecy-slhZ9m|*%tqtR9|*i`_cPN#m+yTKPqNW zIBB%lNZnYgn9H^D{!a5ODRsVg&pz+}+jw-ltxwJ+iGIGlOIJ3ozhB>Jd%y0El*t_9 z|IdC)3O-9e-zs7rvUY`(NYfMD_B|T0xuB`A*JanWQpt&K#etEUo*s z-{)R`4VsO(vf`d<-mV8cY)ek-InT0}|NHg2a%x&tNZA!HuYl?L?$))tW>4mD_Wk+t zan7$lx?E*7y^prc|ErVpzfq&{$+_)@R@Gv~`8#%BKF=C&dNq7=^4;Taetk=p%GSDY z;ar?w_TFhNEvgC~oW*$x)yJl`xc7cp;c+V@{;>VL-$!egJ{4`5tbVai*7^4{=_}{$Ue`^@Kf=+& z%4!?GXDRF2Ra|Sgh?L&7y7K*g`{U!auO@ue-_uZYJ9qDC%SR=PbYikZ+xPL zkAb1d$=yILU2b{5{DP$)W17TzCn@jT^Y8yV6&0nv9~UEk{A4eFcEr)kU*<^S=QSxy zE=zmJSpAmQ4Y$i3(Pm>+oZCZ~rtcV>#p+;y)XIrjFhc)zdvR^HNP zgObcyZneIw)n+xv%H#KXdk4psS2sW7+r$|ABbhdWpnzc_} z+>~Cl)iJSWekE`6vNNhidN=2$mh3Q=xU$9C@D`91E++uMP0m8_+2 zqsp~D_*vzwbpEta@4T!ucVy%XF+G>xU)KlP{}Fh)a`~N^Ub|;|r)=G|{^DuR%Ryn& ztPP{S9-YXuIGnLz&1FNibjIFN)11Gr=9W1&MEkmOnu^|AS##b{;_Q-?*kgNCzb6`$ z#y`|Zy<@UWe_wCDedXL`Rp;elTB4JI$y zXeo1R^TyKK^LXb^vaHlQ_F|3Q{PG1azE%|#6-8`L<6VCHuByA6n^E2!i}kyhn*F9t z{8Jqlx$x>$P&vF~Ugfd9M*<{1tvc=f-rU*SE4xH&F8glT((8It%f?<-}xAm0WPWY6_p` zk<_{K&NZD=VPKHy2CW5HupvX~@gz`Q6d0{qI`lb9n(F&y*OhMnyplJ@swR*s`?zf-8S%%vv^aR_)Ai zy-Dj{@VGtM+%Fdz>RMPTcgN&f*oKFf%`-CBq#yiV-*(Sni`#uYE-~(1+3Q`oxDNgN z``$VA|I?<4lO|uZ|EZ$ae)#XrN2i|@`4(Tdc&W!J*|2}V_3am~i~P5@rghstPLDgf zqUWTH{f^m-{=5}=d*AEx&Av}EbMJY?oU#Zy=Bu80W=fv=y^>=ICWd*hoZR}(J^OLc zR_5gIzD0L*P0Xv#t@!epUri@VWwDW(u~xCr!$8g-?Av$+U-G&oO+Kc0GB@8`#b(#> z)Y^%KC(rDCDAy;Oaav2=(`f3_jZ`rbCtBt{q)4_d`#>@OS|2J(vy!YDwJDV-qnWKKi zJ&9etw)A}4+r0VD(pIc;($#SI{fN8V@`14Wol5pyk4l_UQ*ZB#)yv3uQB|MFkeL(t zm7nijO#7w8A9tiA4&9mkPFbpkDD;xZrwdpHNQ`0%T+qf;gkMtwl*Q; znaDApnYW*c_RHToe6M%@!wr?o%D1t!7<@lAf5HSa#?7h>%@Tm4F8!J{<-C3Q^H&B^6O)oBZrXI=?99g-U$IBJ&9jxhd%mvi)9(AF_2=%yT=BR2 zct=8{#U@WC$gRzKzQ^u~Dw?XSASGcD9n+?}_e)6^+J0cj%FFxy`{vts=ht+*%s<|m z8tLxg!lNi?*fYOUbk(Ya?7Rg#_wg;i+^4F*(OoaVwwRN1pV936jYczPIsGJ zdvofFF7E1LW#8UXjjry72L^w17}M21mdrCf|49DGn$`6$YTwD-Qt=cDey*#SKJR%= z>n63%W?4DoSw3ocyIz9uGP(P~US3@OwqK|C=gu@pWO{tO|9xS+?`6C--WVwMUu$Gu$cvnNRA1X-7`)Em-J^1Q8>_~Pk;XOuVlKV>;&Mbq zX!4AE4@G-BH?L2<|17lTkNv!x>bYmXKFN!8{nZkwz`xPGpxHxL%PDr-nc_wBRTlcp z?|azHKVi}B-V-lMPMF_q*;JCbU6^_6?V!Thx|Ci{$e2Ok5qUOHSyjzsIV6^37BI^}nCp>{%H5zef17TET7odG6N}ckEfYZ`Pf)@ufvAZbw%xmN|R> zVWq8oeOk@?GEnYFeSUvc{J*_->nG0)erA+<@9aF?D<-e9Ppy2tFK)iw3?k6$}loE;jPnU$4w_y3uh##gRgtGdD`X>^YD=kt9bqPd^bV)p+M zyxzBfp4-++T~F>`xaaS-=R14dO0LH~Usqt|)4Ab`()AlRPMn)- z9l5W@^7r@m`FG#Fe}Dez^!UE3)1L=6Hogm)di3bwZBd;f^8F$*HY%4(<9M{Z{Rh) z+}zaK)Wp;$Z|~>n$yv2`o}jXuLE$5pXJ==-CnYIu(pjv*b-YhjSzo_j*}X3yB;?4- z;N=|r{OWpoeP6zmi0MW>31q#Jw>>yI+B-H@mY0|J%GIkAXUw=!b*@juHSzJWUb&M; zORQwM#dKa=7j?UJYwHSiKZWJno=%iUD=0?jIvt_=C z_^xrlK!vHH`$nIv^^Y%?{cmp1mru8{iI=ypvzZ=SHq-z4L?NLIwjWt`En4(&rgQwH zgeM-~KgoXlvB2M}>r-atr#aPv0zxNuWOCbHyHV~wNo83~p|b1m?G-u)S8Lk4=7vfN z8g{O|9pl%2{Eke)`XH}UXB5{=?pnm`JL$=})ooc9g3o`wmAyVNBElmjMTOt)heFuu zss8qVr@Xzr{ql_$k~VW}Dktr&{yu5?^yAL_wk}CYO8@^nxBvO+w7#Z}PR?2`u7Y1* zGK-3fBX^hSR#sMmRw?z!yqwoCcmDkIyWj8gc6DX_`t|FTt5-K>TvU3u|9_p1*!$zU zp`oFFzSsZX?#^(b@LX{Hg*kp2`t4dHI?VfA@_oe@f601Xp58vFJ z&cF3-(3QM9J0@Dc+p+jEL&E20XXSn~#ngViI>YDh9LdmWdp|X1%(`ZO`fH5dlzuy> z^5e2{vzG=gzGKdK_(T5w|EFGair@M2^77uc*Fix+6%Sj**>pbNKQ&c*=ciNJH}_VX zuUZus6u;>6d+FV0{gP|vUN_ovV1q}`vU|GIO(lF-+AG5vLPA@2tqSVTow`&#dUDBJ z<&#E@b3=sp3gq;D6?qeLJ#lr*{Q3pwZ!((tF5a_e&+T*nWu~7#dTy?D<@>$gcfI{{ zJ}`9Z-s6ti&Spjyeap4j%9J%9lf-j>mxQEdR)Eu;)&AhvDY7$R(l&s-IXn> z?rI7Pz50`J?~=E7UT*b0!zfVmEac!btKB+RIQfpo3fCGeS-#vj?TiE$7gt9|2dKgE z=R>=_ijvZz=gsGvUud*`i_uf>dSvT-`;mN$U;c}$J5%P$WhcxyZg%wPqFslKSxbT! zmzOzuPTKH;-{+C6@8yCUxi`+InO!vz4Gk=~l^Zg3Zpvrz{NhIoo-1=soI|0RaM;nVC2Oq!hDSGPl?Ck94hBtTY*l{DdPVm5;oyE1Y zXE^%wP%R2G-e~xF1mSqIQa95@l>|%_Xvtb{{tT_hBDo6O74(Ufu z+xdFk?i+in%{Ql=RkF3+ds>FE$sj;v`f26ACOf8|jr`Mm;cJo6yfv2-c{#q_Sf-^1 z9=Bh+HF=Y1<^8WxzG7$Um?yU!P0UWZZZ03ZBmMAAGtC>bq*t!hTKeQe-1J z{^Xi;{K=PJQH_m_fBrnTPrtXPQtt4_%ggG>#~5TD6f*;ODnRkt@+3y z#KXo6DmK6V&Pv^?v~E>F4vmO$BF!1vXx>W9aB8 zurJGaleTHu&7M5Z=PzGRJ#fI=n1RFOAaAn6)dTqw{h}EM-UV;?sv@&8uR5e;15;nf z!Fj<2=R-VXTC5Z-EiHX!o4uV>EvFyjVsbWn(d=^xF8L zp|6a8{`_t8aKm;`3v_$k%gg&V`T2vE$?Si!{BD}z8?TfUmFqFZz1P2;Ipd?|Kkv*e z(`=h<_v?PY^_g#X_wl;e_17m)oqF`e#l<(buFg$av~%aoUAwH-glVfy?p(cYS6Axk zX$+Y1YU5dlAMRARdVc+Lh6U0=RmblLi1+91E{$wbD0DtKvnryP zrEAsR=bbx3Zp3rtziWt(XMDyjqan$&S2{_w<=--MiHp)U^&jV!Nq-YN?{`~#)v78{ z{z)pFi%&i>ysv6-qtE|b-QudE2i}md;VRBjN{+D^Yf-7 zpc!rN{|96lu3Wz^T>Ne4_a;{EAHVPae>c&wsz6P%kAR!KUQXIXMVLyJL2l$pOYpZOn9-c>7jsW z&->T4%`dx{m-o_7wZ=E9a5SIFFiL(C@YC;dNRIE<6tKQn|Jbxof8 zykNb`>kTXweCt`;tNlb+zAxs8E>@n#_@bhA>I}2S!eDWp2x+V8clQcKFIZOXy_Y?0 z-?B`Nx6^*Pxwsglo)X#j_uK7jxl5NWeRF$z|H;Yf)w-@cJUl6Xetev}PUMBTne?e$ zr5gKxiAL4co&Wv4?PiaN+`EmQ*?I2@e{Q~Zsp;3pRxI zD@*(0>(Yf=wNEV0)avCmWxn)N%12BswvPE=;>8`tvrn&Bb?9ENwuZLl9V?TF9p%EA z!p|od7M^Rc?R(*$ws~gc=CqBKpPyZ;l;LY{=9M;^FlUZWT--eOyv8qIzUXDo`EllH zQTdMNMqe#u-lf**>aO_xZQ;IuRBhA1qtNwY*)s{U1SG99|l(R-I=iJPnn-{1Y^juHZift(V)V*LyYet77ms=BhYR`OpjSIxgI zHS4vlJmZW%e>!_Uzqyh+cixE|D*X%&g_fFk{^}+dKU--2rMPL9pU0#lH&%W=aew2q z#g)0`ZO8fEDAfOsTBi45-bFi+qsNX-nmYBW6a&K?tJ1E1yI&obE^Yh6otT*DGsEEF zp69YFcbDy4{X6zW`ZGtz7kPUwW^dk>e0+ls6N{4I1plBz2|4`p6O$J!)Fw>7&wXfu z=+WX4&kPUxdKaIP6+$Xl-=>5-ks+v2me4JkC^W(A!Xl`bO zMeUyN`3w5bOk;R<_lt1(#m$>Tjvh6iud{gDrY*CT+xr?C7(mVCgxyOEO-)VX4*w4f z3`{9qBTywcQC#PP6yt?crvvx@@?v3dx_)SF!QqGw{!enxrZ?;k+q(-W5>2iu~n<8o|Ij9 zS>iL->Z-EGju{@2kuTouzrXj{e%aF3Vda8b61`X$>`wKxsEZ$%nmIv;k5h4iAm{eJ z_Vjg42^(XAB%Z0A-8XFlUu%Vl@8iSM<&sZNe}5?=y!CQozOUBem0D%d#_4_{pB|px zuULNix!mznR<_3LcXmJcaBjXOd#me2qdcj$GHn5o>67oQjoubik#s$b$Sr}NgYzm-_oYDOAcY#v5%$4X{3 zC#;xo>GDPU-#NG0=JIJ82|ao|F*#|^YR+3ZE9C2YcBS=JhaNIeWooc9czLv7ze0(c zW--_1bvq~bYX7^g%u#S-w&jLmUG}nF+D~Tw`IEV7z1wS*#q(7>dt_|qonF6x_i@$Q zhZi}0VhgyhrXDrfI@=~Ad)4y;=SpR6?Br`c$-nrft>eY3i7pEHx3hwd$GdsBys6xq zzxSZWs~*@34=lPZr)E~!bNtD!b1zD&jMUCQt+uZaS-kV}(>A|qZr{?-9WrKCc9ZQx zryQU7_x{(@2DK4aJdBE;C2!ketDc{^M!o*G)T-0|rNvJ!2uMhrcsf1)-Bw#8qe+r> zQ@HkPSj^QpJ$>tB9lJLVUA;BWnph91q&HM{3M*%w!GN#f6+=Dlg( zy(T^R`1D-h{n};xbN$ZG&HalM#|I@e=t*?cX_wyH@Vdg*fZ?9s~8x;yfApSAt?UTMZ`!wYTSW=v7w z5Lfk_)Zu&TL0|mdGbW^>y^VTBgtVS2NZ}7#JUtaLKH62>qDQVhv@$oWYxycVg zH(p3EGOoNQ)Ia42+v1rG6Xwl-%+AtY{FC>4K+Kf)6GbncxBrndm+AdZw(!_eQA^9* zQ0BwdOIGFTD=RnO&fkAGIXZo=S;NVcPp59n%gMb9m46`9e)!<@_$o0QqnSQuW}B}+ z{q2j3wzhNZ)>FlwV&m@pEZxV){_50J+xO|)7hk=$Xs%W3uDZRumlW)t#eO%ixOn2e zAB(nHOuTlb<=R57$4ebp7@8D#B=%l(ae7rD6m)#?jyX4S&syZ1-7&|nC(Yg`jhk(9 z{C?Z`YtlVO)caNKDs?um|7R}qP3(>CJB0&J(vO~(x{%zNH2LPCC6?E}?Μq-Lmf zxVUmrT-%?$?`B`@%iI0dZ2o!M!bdJ@esf-2-lsOXvz1$XQ^7;0`QPsEF4uQ;b(Omf z9)ExP`wvI%@3J}9<8D8HY#kch{Q1@D-{0rRuF&7t@#|4Hb5-v`uxn!ktoAZ?*9(aK z-xogV_1mw1zOemSBmVr>SMHZbI)!!O_uX0i0koENzMbsh=3RX=&ydNPx@ArDloO#%4bDVong2JUMD?KJo zjGdM%=cB{`+RV{*ZEf`NRiUf*Y@09rYyY!Z*%w|{Uw1vRIsNyZ{K!bjNh*R;Qny~e zj>*}gxa*OF4zmc0qePpSPJ~0oiu84C%?A(6u`I6maFD&TzhC~s>;3-U2L+FG4ahG(JB+|NQUw`|JvOdVSLQdnUGWiwlW9t>%2b z&#LZ^#U`D{1r}{N?3F8p!J)yhDz{O1YeM<1&jS1Ab6b9V*Ob(3oByv(Ixpe*xw&U% zoA>W5em-f&j2EAm*gt%v&cv|L-{pv29h;S;+zy_QvwP3gZm{0G(CuQvZOWPx$du%)Vclp56}Ja zS+pwk%8I}v$Bupa_xpYF_S>8s91?HTVhd(1EBy5Zv`Oyb>e5SxJ}g}EVf);<<^P_W zSU#9*YoK_{c9s%{_$&^FmN-u@uSd&f=Oz98^mNaML)-sF^~sZzyuDwq zeb%0$6f^0R^ur1KQ`5RvEm$L%rQJ{mt;f3hrlSjVq`)<3;clNQF#_2Xj?Y!z~ zn`fRl;c@8Dp|+B_C5Z(h*GqM5?pX3##DR7o7Jzh_PCn^UQetuuv;{@PQDTx(t?jpvaeR?p=(vxS@nKkp{H)AJt%=~2&hghr|L|Mk14%Ecd^|B;NfHPE+0*^B|r33ti?KKgwCJUzrfy6+%4L>!$HUK$&uVVohS?@5A>rYh z^Y7bj+Ps;Qk56vHYjAdT-0?t6H|ofD2DTgK%1OB MUHx3vIVCg!0FwU0C;$Ke literal 23825 zcmZSh@9oIVz`>ALToj<}nV0rgl>rP&)xZ?PS|}gFV_N{eoOEoHpu(Zs^H z9W4hGgbW{BdCWI1V%uC|bhD?y*L|(Pb;u4j7L?-_FfAIa8*P8R(e${)l9cI4c zJS%bczjDS&$MXHUk{KB_Q+BNSk;w2xG+KfQf^i~ao?^5TBafv&5fDyRPa{{7?B!M_1Z+0%P$4J0?kJydR;`6!;@ z-s9${qH2d?UWmn&iavewsXpL;x}%!vv*2fEo~&BRl^V4)bjGO_!5UpEAuA?nb}k9i zyW5g9x!Z8zfu|Gp@813Uzqi0*eujvb5*iI4VGf$$H!Leg9 zr&)ni?VlqDq?Fin|LvFgXy5wZ>fhl%^*_&l3OaH-=f2Oc->>)Y|9<~|wL{YpCVtLdI@FCeF0d~kcVh}at$9#-}}x$0#Hk8RB2 z*|sV<`CwT5ZN3`5ZATycpSSXbUfP_!=H}wB?wqS;{(s#d@v+>B%7W`M1xFtK|3ACx zbgLEn{e%sfHlJDlf9JWj)3ASKq0YXg-A5C)m*|W9Ic*a7@%Q~nmW7ulU7hpi{WP7{ zNsFUxW@^{fZ?9PP?!V*T{r<5$-)rjc&)6~NfBj2o;pxiv;!4|v_dltZxF!{T?1CfT zNsjJb)3lBk8;`zW`~Iq<$M(aY|L*ewkL$K;Jp36fTP*hE_xx8oUfwo^cl1($NXOwOKAHVAHr7bmo_h*xkS?C^)t5C*JdH({c^R#})MfK`y5h9zr-8fk3^@BN0wNi?Y_!fv zTvrK9FmJo?=>PLw2aVmo{9o7kbV5MMo3o1ZGdAUHeUO^R(`~Z*_O1Hm@2Y>-Py2U& z8DI75pY_7i79YJaWw)ZqY?D6E+_%16pIehP=Ut5uwZ7I>cp&d?lFWzy=VQ39AAa=H ze`a=b;>SNm;ad-u=(0aQaQTW%@EPqK>4i@xEYCkFU39DZ+JVdWgr`3IpQ(TI*q`@~ z)n~=N{9S(Y?zR8zZ;LYimEXIW@$diZJ)6&ZY-juaKX2BpUtRH?OD4Tg>AEE={crw+ zs}-_Kjb=;ze&4hy%cknOnL7LW$Sd8&xqR))|5rEc+kN}R@7X2UAELve<3IjCEgc*G zBYxMD=Zvc+xGeKav|AgpkY`oS(o>Uv%7{BIkiW2iS!#RTW)nY$PtEf#-c{s`n-lpY zu$@)O;s8U4!r_eFO0D7=4NZ=V7%g7$&F>U(ozykOUn3;KYid~H%PUr`8dI}GXL(Ha z5OOW-oVL?Y=huZV8HQ{V_TNNoGiT zy6kYqot+O4^NWUyJ?3>?r%=Quq$ZKw_2Bc7qX&K*7Bn|gV*S$PvMgjN&nuO{RUc9& zrOxpEDKc$on3l`j6<1GfvFgp3DScRJxA1Axu*h)rEBPYRPXD~Gp~D_rxp;%j%?&X; z)93A4tvT%# zKYVcE|Nj#AXC_DgEnl_x-amWBzl-%xHYsySGqL?of2Av4m6e>TaQb*??!y22H_gO8 zhpm3pu=fAkpz;U*-5vhtJBH3+>(h=nz$F)4p!ivH(NU#kN6zgy-6$CUqJW7rd9u=z z1?<5gPHh5JXDtsC)6hzjlMF*&e%ljOW7sd8meFr_NGe6|&MvRM4Rlg&MuqRWDoXt}HnSRSCVGUZTkUg{^8NtKT`)LpPh zGCs`oD4pH!xcGMNHFI^QG~9o8P}Mugb6b>#=uM@=La9-$Pqa$8QjI*h!j}iFR1Nkz zxaP!z`&)O#JxumpEtJ`{B1~;=%Aw%WNr|UR<{i3J8P#f})TuPrXxaPfL(YEfs_fhE zxrKAS=+s`oV6@B7aKS-lF$IHHKNJoN*s~mz&DCYdb#gh8x8dOm>nTP{L_DX3C9O1Z z7M`i)92z(GkWuKmLz7l&1+SRaGuiN|15=gRi-ey&<}D2zv!_Wbrzq+-+z?r?*p*2o z_+itv7rO$QLxqe|3?Ef#378n~yC~)?xxk}cmFvUCEq{KWfAlH;Ze*s*>vLXZj6$lR)UOu9>;H$ZFsMYeWzs}l{Ph6`T)_sz@6QVls(GR%>zZHWaqVT4(JuU`j1w z;Z)tQ=-68h&w_`C_2llFr8Hb}t?BXAc)-jx!Pf55A`WF6s~i^DSScD5?J3C%-vr9fzUqlnD+0 z&ZnfE`elFXf2v8G&$s$neFvxA_%LgyZ=aH#rke7~$EO1GEp~0va5}-WOvJ*%`rd?k zH@2s&|NHm-_uahvpzZzpCihPM+rQesOx#suZoK!Dz{pgksb{A6yiz%AwEENrmCoe> zOS6_9njEpsV^zn&=*L(l&ziDLN z<|-!Sqc-_uidO6@y)Sw<%dV+f2YZ|9?nz)_SSF>T42_l^s~HnMG`I5$?>X33?aWu@ z`|p%yY{LGc46_HviwHcK`F=|C=)RRJi>qHW6iC@-|21X8r74li2_L zUvuZ^g8%n(rU^Iy+k7U&TKvoZ*#`>@SL{9N{BN?|t<~EeUEJz>RHfj+!WK#6YbhsB zyfR+Dyh7{cr@Nl65_X_QzK-YNqFk+i^**ih8-C3Vm#)&ff4_3;+kf(&|1;B)KmHGv z3N`;VQDKeJ22t_pNjB1xZBG99pZ(&0`pn2;!xwAC+FAGS|D3(+fBh2vBj?Yb&uL9u z@Ly-M=(3}K3|#-!n;QC={`!9Z=(jund7F3M_#glEf7wRc9n0^pGjaW2u5)~M;KG|1 zmo7b=(XhOysNL6Fe^!_9ulgHYt*n3T*FIloc>n$e8J~aqSN->UJooqi%ss^`Yw9nD z%=z*^SYJ2u&$A^7k_SVN@@uA?*-*30XQolAmvFMBFq7JggkHmczfbIzouj<(zn-ns zRL9T|)yr-tE%g5WU-_zP@&EaAb7Ou*@9Tg1JF4!`cl(yh-PaZFt+@40D)e^j%l+NH z@0KzD(hiIK@Lo0N+8l{5PeM0yw1SfC`cv-lX+Qo?}<@@hn1-P~*T(f{?$=hRPkDTbwd{A0GFk>982 z&we4@Sq#l=aSiwX%isV1{rCMkuQd@95B@*z!1h01m2HijuKKk}nu|&fy8hRh_dj&g z4o0@EG|AMn6sYn0o#J@J{J{tdJf3Q%w@vVb8no-xk)c2Xhp1ing+5gB(C9&>{l>YDE za^!B$C8K%0Ee|xC+y3ijO73@4y7*sTwQ$SDXq)eBIn%q(9jM#-lIO%@&!6X+f^W|e z%(Tk9{$KCqyI%MI`qi0v=l@@hk-qc!@87I^8}Toz_kKKbe2>JmrrhEL&3*g*Z=HSf zf9s3M$7XlFj}kGtenDc{r~mfxu`-|ko2<%{{hKc}=QHcyda3<+I`{8arhEV2&-!gzkcbaYb)lw+__xScG%6T%cIXs(o?@ga4d+=_Y z3Rok@N{&?=UR(BA2e%x{Yf-5b+HG}DaGpnpplMvG`-5*Mx;F1#=``s>-vQm#k2|ka zT-|APPf%{s1fgjYZlC(*($(vxtQs4yX>2+7YDw}EiBoPKlcN1pSh_{FD5`kQJ{?); z^6}S^WxH&|CdsEA_vJae`2G8w?x+9HFWM}5`cZbFRCD{nOrC#R7HV1S3w2XC*5keO z#E1WpW{VGgdGQ*Dh8W{`D)>`}4oK zRL|#VXYTQ5Ggqtp{?E1ROr`I~zgfkGKmKcGH&_2N`!A-KwymO-v&LLa&wbBU>)qzh zJAIG-l+XHjuI$U-xtq+s{13gLneFTrD3Sn|gYCmzBnU{C~0Bz5Azpvt;x9 zhJWvw_IFP;65k*8;OFi$M>mV|G40*6W5<>)9k#NIH*V;#z59Q^#M|FT=DmA2U1IG~ zg{1Xj)t4efn}7Vj-mCuYg4DOH!Ug|zH_nmnez|9p%8z<+jna2KFUmIty^4{(@PGQ} zt49s*-`^Sj>YqJp?6ah~5lh>oO)vw-q_Fs7p+wuR_ za}R#}zj#5q)|@W?yVKeFHojz?uBlg9bAX5QKqI?s!-r|k`?OBZSaMk<{P{{QPt79H zS&LqYoLb^#sS+6ZY)Y`MlwlHUgZcdbZ+9Kss(eg`m9uRJBip1O_tgy=SFqTs90)9r zcq+71DOG5u_sk6?>p~J&s?0ld?UhTbwos<4mG3ISj)te3o4oo(6gLDOm~NGNN~F?i z=5i0NP_xT-be6Jh9Vcty7_E*35-7yq9>EDitO$OJ&S7 zQd(8ADoj^Ia$eNRP!FNA!4h0X-(#tX*AK#NHH^O<)NUJ zR;??9JlocWtyC#itCh%@=wY<|%%mxk3|p7I)N!7YIqy(Vr;6~TGm|tg3x$erXb=`n zSo!G4mV-h;i&p4Z-21_BdI!st$@!eRTZ|hN5AlSo;B5Wi*eI6NJ^Om)wI#X|naiAV z6px6lmGHc56u%*)L-!vyLum5cz?}A8=anK+`pNSJk%Y?SKE5V*by%{?70GXWw{fN6PUo$ykkp zOxnDP_AUnxofYHw&?tDAOGx9E#9;+SVU+?_ea|WKeF685DeUlI3SP<)F*TB3((?pE z-kPA~BP?bg9-rhB>TviUGRbnQ$0`xcQY~l6Aj$vtn-*<}JM;g$(En^V=Io#M*Gpy{ zv$=gq^227)KF7_V9RB?3y^001t@E}1)!$l`cRta2ruILRC1-i}a4y%qvA*mXTV4I- zQz75>-(I$Ky-&#F-}fzIW9DuCzc%~U%GsUvT^a)*@^FmHP|r=gleUtJl(C=;m9bvHa}aE2YouefVtrzOgm`^7-6#n74p0tTZO} zb$y@O-rIR9Mb~}*zu&W|!2f^l<-NE2rXAb+Z}+wT^|Pjm_n(tJ-RJgy{-K%4`3j#@ z9!nKHKl6X~wg2a>ir?@~iNEYSyC*rY?(U!c>3`)r<{KZs=divoUiU`{d%K8z1fSaP zd^O(Yxr-w=CFb=X=yM@x>%LvlSm68BT8EZY_3S z{r~lezyDb_JEzWn%Is`r_3gM&>Q{HA6+$MQT`X%EZJXBp(wX00^zZ(v)e{~meijMP zn6poI#mmc$(|-NmFC4J(|9#b#NeKbUbD4FSoDSQ0IsgBEQBbP6Nkg!Nt;FN_o{A?a zU)q*_yuE?vwpVWQb@7_(Tejc)KYz7U@JVgy=l^{h{_*^7d9yz?V+P}wR6Wmc_noZV zCx}n<=Fv51Ir!<&RDXuQ{}XE#>3qM>_~L(tc+`}q|L;2e{~sCgqNFu&)_-$`B8Hd$ z?Oz=4?*DiG{P}+GelgLAiW*Im7gK-di%5Mr^}qT!pX}fGHqht+2=V=xIVj6zu#Uc@aEhL|MerQ{TwP&AOE}U`s~kt`@jF$EpAHl zSe56-oYI#5{(o*}+<*JNfBL7jxxd%^|9tA$|J!$-e*HiDLS<_5$Cc;wxPJU!{Pm7$ zTI9wZ6hRmu*TD zw&$tv{P{m8J-x@fd~^EXbF8-&t|5fAwEnpoDADudycrE&j{6E~(J!4uy^Y6T#2kuI)o^<*^wlx_%b7?24fahgOEc^_x5c=8)-t^tnX|kYFGu8* zcWYMH)z#Hy><(wMRn64DkaOSUnX$Q^nBJeIL~qh+Sw* zVp}6Gy{3UfHt(K@`b7rvLj+VuAS_XWYzbB{)?F-`tw z|2<2Kr~CQ&z1!G6PtSk%_2j!nQ|3)F)BRwUtu30r`Iznp&B$Y3X4|?0FCChq78L&X z-}zJJ&-q(^uDGmxOyX3MMn{h!hd_#3Ly>W^WP-@Ci3%N`bdG)#W?)9Zd>|&yMs@kHaV-@O7nUl`thOr zi$4qQaGq?mtm+kXs<02)JndZYVbP|7<9cNupGZCuPMmpuwd`dEmkDfYj3OQ$VvSN) zUrr9qc_?#ajw`EzIMZ$AT`^t{&$SiXi|<@!`EE`j&)>a5D-663Z8y5kWu&@1V7}Md z?>DWoHWxYfex4PxWAQe&#X^A|e)mM&(jVs}u-O#HJWETR!u@MMr{nHYTdVZbGPOAd z6-!x^rYyd>ZsqTe-iVTu(lgr<8D0h7;JQ@QbfCRhE^1fEd&c93ImC4todu-KjE2B#1rIgjN+aFBsJbwN`;;w42N5PZGfXn+$qGVNN zUE1@m{kSgB$|hZ<7TU%qa4Yh9=i8#l+02?D+W&RBGj-o*Pc0UnQyAsr)e`aUf5GEe zizk)crB{TmBu@DirT*!N@ivoet}{Qpp5-n5bvEz!m6c3?x6KyrQM7y!ertC@>t%JB z@6xwI%syW*%2E)z_b{~W)V6Q!UjJWruL=E=xUxk~`9{kc*S?PFOb%|Jodp*D_3$i_ zSoA^n$svbR7j`Z8VfZtIB;^wZ6rfL(1{+ga4OLf99ASl~H`? zN$Rv8kN+RI(!(6lvD{yx?(*@cftMB6onFOhVk9wh(W_c*JtOD)1s65GCK#%-#$C%= zI{n?gn|JG^_b*rFvGw}1ImGDf`sGLeUvi%m#QLLecCg@c=foq&T9(-r$+p+!bT{Wc zU$8wnfb;*l$pZW5tvn^g*<|%?9&^5(aa^_KnY-7E_nf)^#^i{SH~*K%k4%hzo%=WI z(X;JKh0eat`}O$l#Z9euSk*S%yL;!Q`_01&FD<9*#;UCiy0t0wrRc)dx}uxBr>8jX z*GsopzxB$IV|*7orf+AeS+L@+17pO`{DeM%63K|g{e3qRoIfdEG?4JH&v?(hB3twn z&pNlsXSb)mE?p2>W_-EYHR0#td5yW(<36Z=`)$*w#kcPF^#6O0=!tEcDbmMrE0$@? zhBcMOmVpNssXx1ZHzim@@5YXsE12&6{l@)$Pj~i-kcBNO9uM{${<5!m&-Whh%v~Su zvA0NiE&93fMK(w85ACAqt0n}8n9Nb(nxe7#w(=sIeQX5`MV;2?UWW3V&1~s%itovg zv=;C)ioNSo_lW6p#>Xjw@yS!(7$>9`+_)UcRPI%iEL-RMU{aXjj)pGARVj={`zHEY ze&3vNpg^Z-pGCKVLtf{H{`&A+M`wI?ouyFtRJ&B#GJf65dsZ)3OD9kL_4>w(J%Tqx zjw%aHp64K(cw$=3t_HWWX>+g7?k>=IzTVotKqCC{)339dkL;~tufAb)(r{%=OsOa< zoAKP3ey%R5f6I7Z*%-{8ps?-18LKA?TAFqTi|Ne0FI@I-lX18!bMi`FX1m+pF9}9R zt~h_;%MDK%kJwLE6VwtpOWMkXKC&|X7W8e}JYoBlnKzwQ9O5?S?0v)6xBOAd)kkTw z+dK~xy!!txZte8Ffv-;s3rr4r{oL!q@@pSA?X)oAn9i=Gq*;}ByJ|xDk?KDe9-94n z@{G0g!s+hKsx6_PJ74=9eU{)nFI`CJw!U(&_lD=uH>Ndh+rFsW>a@B~_Az4#g9~7ELeBZ>Fw0-TH2U z1yfKgkB<7xF!eBZ9?Oj?W>vS_#Qr8e?KWPo)~eILwQbdns zis3QMrET0_Z>;Wq=^v7?L-619l2=a`-EiRSJ7{qvJM@!{fx;Zt=SI4HUQ6QHlpJsO zygs}$SY_|cnU|*QpUP@9=fqx~+4@VZyvz?SF^(;|(0Rq7r@P#0mfni5`&=f=@-yd$ zy8LW;@BAb^cJb{*!-&StM>G2TW`3T((_s3Z`$=cZ-nIU=>Mniwerhc9_eDa7uKzRB ze)BGzS&k?4XVH!0(_1E+JN;EX|3dtr#E()=?=2CWf0nH0z5QwvoBrrdj%(cf#jQp+*220HtesDi?`LRi9<-`pa519OYwUS$H zBkSLGhVYP_X$O}^w@FM`VWGTra{I#a-7Rr<11eju-Q<<~sK0Qt$B)H~uAYaNWd3($ z6t{Wt(0RfB4OVkmPie0_AINb>^kTZ$->j_GhPYL4K6d}R^g>$jz@kmxC(oDJrhPeh z^JQJ5(o?Grf3(~fsqgy!uYHi)9@D^k%$(B9#f%3NC$deGv)d46f4`>jgi@=1|D*En z2{Z3I-97g8O(DzvG#ASQ+dLE4b#HxtroC54Z;>MN-u@+*`WNsxf39d!^k3#PPgVb- zRcggmJ=Ob>f3-a~Hkt44V>kJ1$*;_|$uoRT#(m8&-Zj4bu64FgkcN2N^vawVgxB-U|_IiU|?WmU;=YJ zpkmW@Lm05dVg}lphnd(I7#P?YQu34aQW8s;7#J9qf%Pyjr09d}4#~((0txh*F)+wN zv@C@^@sIEGZrd3(3ABKYbq`w#E$NzUra^nMv5rLiqQLw8Em%&xboVj4<(9WPi# zMGj7Uywm=Cp;eXM3U~ghogSefBJ3#vDFQPEHJXnywTfwoSn-!E$`qbua`W|fcb_+T zR!fqjKGn@z-oI(~o4rCF*LGj8y1v`um;|$8?t;rN3mzZiU9x0}!}7~FcUzhXvM{Em zrb<{8CCS$`!qqPj=KdS3=_8PD19xq@Ao_Fn!jJKPw-F)&=C2SH_7)) zmF<-*(kJfgScTOPy9*iD-@ZAd>a$0K3DusQvF z+wR@F_s&k4eCJL~&b>X7e}8{(kI_56e108Q^TCG5&1r!G4jhFZA`BDEPM$ySo|&mB zrWez3aXQAt#UR zkKW#8xU^e>hfRWqt)ja6_HNC$&mKQ+cI_5Rd~so6#O^X(K6$&G_iBPp9@`f)G;HBM zUidA~sr|>#pAS3L=N))(uz6$BQ7#$#IvKac7w?@_YMGGd%*ap}@%Pu)!;4(I4Kgk$ z?D_NQv_;7afiG2i_s(|ZP`um3#PI0Cp~HtarkoVwk+Bevulu3MCuif~$e?&cF@V8A zCa$8wBChJC>W|;QA5ZqTJGe3Vc!Gh%o4Yzr94~`J7zCIummKKU-^anv&tLQ5AiIaZ zKl{`Z3N3feX)q`};a|9LVZ^o^NgErR567hQ6|AjiH!(P!$n9ogaFXii={fS~X!pZs z&)W9?d8%(w{Y{6(aYOlL4#g#x0~s7Vl$G84j!aVZKJeM*-`Dl^il(Mh!J6Bd7@TGa zNK1Ez$5pbHmX?A7=G;76@5_uW6K->{GPIZ^Us&K+@%QWXz`(%9|9|iQKl=6c^|yCb zoH$;Zh%g9TYHDO=f3$RZoKtA1XxN&F#z~Wes?9qEoUB+`8Cte9JhnJ;?AW7~%jY>o zMoPxj|1Fi`RA{*)mSOJT6m_sr=F#KFj12$(ec%7_-|zR2XQt0{ba82!8|~)tUYUcb zsHmvm;UU&LyGl>rJDVx)C=d`H&fe^y%e~z;X_C;D>(_7ZwtSmrAfYh*G${RSOg^6U z@zK$Ova)UEn{Tx+F+2*ensoAfp-daUeBV^<@JH+G|7zR)`;olo$D{5K-@m^DCER6p4e@*7hne*nZ&Fnx1 z2breTtF>cdVhVnKN(~7QFMr`|m2iMz&xb?YKOQvmZ%97QrxUS(!D!|h;b`v@Z#9*b znNw3!cl`hNyYT&(1b?cX#id zogvJ{(BaI%$LHqb!(*0rr^EbS1#@Cz;sc9ycb^+e^>Xm=fU=EY`MVe{29Ax6wY9b^ zjt&V43-->wlxfj-`1Shzec$g@^G9#bYn?n<_}%XJ`{E2`9T*u3IgDoV7^k0WaawpF zfB)aMEnBwm&ELFe5tDiTy(M#*ISUKq?f07u+uG&pB!oJT-1E&A zWMw##*wWIH@bAyh877%RC04QR=eHL=c6)PU<6_|`H;$KYnV6V9TnYA1yt*p%$EVZ! zi3SoI5)L-W?X_wE#mAE(OD88Ms~Wju5_|USv0CWbGU2wXg9Afc^;=V$&u5Gy_Se}i z3DSIae*XRB=7xDUwVXH`h`|G{zj`j)u$+sil4a$#gpY!SHB zIcLtC5C49@UtItHAIQ>=r}g)LIib+N!~oJhxw5kI;m_yu-S^l1{ctmVe&YLkdu{XF z1GyMD6k8_f2?z-_E%lzh$bY_F#O5^KvbVRcE^cDrV(^&l;^y|~r22f9f&v4JzTIo< zI24x{7s|9>xDc@C<1y(8UP~+9@BO}^{C!-t`RS%GMu7%Hht;gROHY6Kue;;$Y3_2L zxmmxQD=W=S)>Rki={<;vJt*-yP|!+o$@kA6KOQ_TU*DrXzeZ?v`1&>s3@tRprC?xcPwM}*X_Ohy?Wj1)y{>5 zhMP8RnlNu(-<&xzE3Rg7G+kJEy-6cruV zwCRA~9EROROOJm3sK1xRGbw5F{`4M3Q_}?VyoR~<{E5%~7VOyZwT(}L;bigo#{G}` zuE}SsaTx6RB6Q0VgZQ^Pf5R`H$cK_sv`gw$BH6Iu!O`dF6{Vm7l;}PM2e-|sSUcdhM(o%1WsxK?ft~XRu zV=I1k#xW;HXV2cfZ>{{3`R7a*|IsOY@aK;$zaqk7c@w|qNrbOwJAHy-)#|l!huz(q z=bN2<_ImRwEs5{%S~nD_-gtW2!MOhWoZZ$x43iJ^%Vd7usmSrbEPok8gFuq_v#X5n z_kNd~rXSz8{azJ!{qNiN1!QIKu9fE4TFK?G{PM#a8H{%ttlwpwiczF%4(lbPB1 z99CZ~dr`C9*l%S^%Db2oW$#+P+$}%&_5J;i#Z{G#zaBJCczcKQGD>{-{s!i_IjXU^ZBbd+ny?%n3qYqPJfOMH84tB1e8xa@Qf z6%IbWHm{{k)2EBSdi82S*y_T+ze+{K#g!El7?O|o<(B@4)0*nEGNkEwrNOZUSF>in zd7>Y;2b7(f4cj_})vfM$I7Ao3#Z9pJ$f7X$;D<}z4z8}p&r8bioV@>^PomeL?vG(W z{QX~(w?rBAF8g`kn61z3lVZykhQ!3gif1#^C(N1C({J~SIe#Te{ctg7d?I zf|}m%|1W20X}QSyM?(X{nfdne%I`RBq;27c=#Z{EZ4dF|If$&{Lv7H!GqEEjykxIO?L-5#> zC)4D^L$7{%&#AaX^y{`e6Sg!xw&0O8V(AvsJv7TSdqdGvuRpKj|2rilB&cuMkZQ)x z;I>$>onQXg%HZWLbN8=buMbLWKYmnPywlp=&i?1`UqLZ3HhX(}P~ko8K4(({s33fF zq|>*LDJ3l}>HIv~MQ4R;YHBu=ybSW0X|(j@?&|vb`j+sQl9Ri60s#iP)JB}5h#BGe`)etL3p$M1K$qu<=Ij=D7Uv4w|< z5U5?I?ls>(<=W z#ib#_RZ&rK;PPeRbLY-2`u6H;RmRq+3=^p-Q>Qwv3^{c6EN@<3UVw(k*?ne*<&Rd+ znmx~;f0D_QCr_?yvt4!8asJQ!ZpsNJO;4XV{8+qs+nK+opE`JOJb1-rVlHxi#!H!N z`fP8WcuLHxliKtorNFs;787$TQ%#Uii>&p5bMv?Fz1t#i>EskI)klvWO_(;Vjg?!> zVQEll)F-)Qp55i|k8M70$L;Tb-p?|T9mz+@nUCQUfv9=(yZH7Yxo{m@HMmXGM)dt zsA2hn1q@9JCO7$d-J13H{a_O6WZ~rGlrYH<;Ns>k{P7`ij$N%(WMt&D>vAF@Eu6w? z3YwZHg)GHHL>_#2c-W%;pN)~RabQpoQ+RkdsFPKGuTp)|q)9r9`M-Ys3JUeiOie9q z?ZD7bR(pGU6%`c|NtPwcmp?wzDLl>L;OW!clhys(=Fg8`z9(MCUVhDn59b*g9v^6J zY2fB!I(;hZzGc6B`-W_B6+yve4NMI7ITAap-{(G;b1N&U@-3-zx65XjNY#A38lKSe@YGao3G+M|J~^8X($WIJa}oT_r&@0`+23!9-NtJeBouuqT+e>^>Wko2rdgsRE(xuF1SckbMI;?${4Tnt@Z zT@&Wc&8?i{qvo8FqEeF+aX(-C+->QMDWT_2ojM@(Bw!1>w>R_Y)4z=w7!I(qGV;YN z=oGlLSwn=&+o&@AMPJ|jGs_H1k4fm6tP;Mz+<(3qufNCApw$e`U_rk1SzpA{SpZ{#*?A_NY z+jH}5yIWdVR)lD^w6-dmn_myAvs6}Q78Mnp7Q;B#uf6#BxrLiH2{pWYuV7)ZqxQ0< zO!_U&^EPYrCpQ+Cv6jvExX>*rAt?S$MEmn%0ofoOq z`eo;Hh1z$PA3yH0x}qP)^YHb0>!qp z8;h`{q~wXSXScp-4G0Qq%D%oXa$2sHrDbH6=@!ZJ&GSt-V-gl*axUoS|8XJUL-H|( zd-oQ#bA_*W`tkMpw)n-Gy_2)=-1WVluibhm;@HXk;3w*d-m|`0?vk)G`4RDc-vOf6MPwKA$_mL*>NDlY(pS6qJ_c z2H$Jz2;n-(%)jNAL0AC8W9#<~bFHuMdu?*JLE`N9 zf1;D_DQ;ABHuk=yBGeiB;p&I?2M!-*HcmSua5dbb;)8-*^&3ME6`_>0w61B(d8EzO z9A?#?6&lX|_~K%NbrFnZe=2rczdW|X&091wQZmBZn}L<_(9U9ky8m}yh3jaa-WGJI zWrE(rXV2KCdR^TY=yB;~4j03cAWb%Yxt`0*{iFMyK7D#H!QjRmpEMJxz~ErzRZ1Gu zPp=jFGt28oc7*kU%|CwrOuVD7S4=nhMX7a3Dx$?);>G3-KmZh&mii(O1Qcg_p z+q-S;SDqZroGDYLsBkfS`uus~ym@^eJ`}93U6y`s&ds0Ik`g=uLBWX;8=2l5V7$Dp zcCN+obEXkWTnuZLEP2V~l-87RYvzWXT2Xg*Tx3&p$+@$mP{KHkr()jwZM*corJOte zYGuwlUL_?Z7sdt!jvWOLnUa!{a;ow$(^ zVww{x^7HeL96h@AzmuDCkNtn1vahQ&Pw(F%qGK(sV=Jp;zCGJ2tiJx@JO+jXT-?$7 zyTw^vz2JPw$HshPsgV9$Bj^%XXjo^QUIXk|iB|J9q4eDE(3U`&(;&fB%bh|BrMEYiMe|{B%~3 ziHT{#lqoHxudhXZ;xjTb+ED+$?#vv^&C%a8^SVz?R##S4J-R99PEKCicBo(8zR!EQ-qv3+$0WYJyPLiBSpAh1fiG@< z;Feo+TlUkD503<=3h;aeJjsPuI^c5FXwxU;jr?YbsY^VWI4;56ev#E?hXnAdzWZ+}^B{ zk)V{sCu`MG`T5zwm5%LKg1@}H%)Z#I*U8VX?`g#0TrKxLnVZ#RhEFUmn;p*I!)T{> zE~K7;L87_flS;u)qqUFUvwo?vT^F~v>&zLSDbuD!nSKBCEdj>QZ0#x+e%+O z_gowH_{>aW4J|Fum}!ZX?5{5`nd9T*TUuHc)c^mtNOS+PdG{{`to2dbeCt4WgI)jG z#5KqFo13vS*ql#(a$;gfcQ>fj%{b%To6DCjefV}e|8c8$+<^&-&MvZ5zP`Mm?yP+M zpTLkGYz&~PYD3%a6aNqH5%kZ=S)?!U}Rbs~^w&mPx@||s_sIPzi z(n7O05ohQ7mY0{GnPVyZ`Ptdcr&zPPxyAJo{{H%EQTJy@hAVHf&llSV@89=p1}}SX zuljwi(anl$Ya)H?>ZCxUdIbdrPEJk+B`*SeW|??~Oq|T{;ZxHS`Tq@19=6Y}_sP|b zaAb^#lWXY~Ke+EF+nZZk#jJh)tz529`*iBG?-GJ6mshyBvPR_Twal+?+mI=I!6~w= z+pbn2a;MNF%i^f|9OYyDb`2lbZol=%raC*=YsdWg@@hUa8o=poy5TXK*Zce3?)5icS5#Dh2B6Z<&vSKjYwtzJEG;>5=Ix}U7`Yd&!<^O@Q7^73-$r9oeBq&&0m65|bw%)Qip9boiMF*JaD-)2W|68miv!m0&#i=4L|vUe?FE zN(J`)5qk6M>oegN%XMGN%$vLW@;(<=HJPz82z9bdR`Uhr9#GM_T33LZi))t7)9hsd z8U|%=B6K1*ExFpVC2+ADsG^G4o;Nr3#|0gI{qvJn-j66OG~|=9XkcV!E75qFduK=C zj?d?;U2lH1|NBK)NLbh~=f;Mz`P*0j`2Aa1Nr~zCxw*o3SzX=T5AP3DtL^`Ae7Z|^ zM5$@Z*Vl;^|7&M&?(zJ7DZ}K(?sEM}lP9mWTfObX$&&|nzu(tA+dTi#z3TT{Gn1zt zI^?t_cDLB?Z*Q}U^Cm1`uI|<+bMsl5nd@3`SNg`@YV&t{zsKDvzh4_WZ%vK$qdj{R zB6kS<`SKDp{?oEG`(V@!g|)uVzkP52vcA66pvr`2{ob@y!Iw_{Xk_-CJ>~1BKTFK? zxyp-+4>R-IFv!cxzqzyXa@N|);y-`>fa;2s!OL%j^tkCwSJu?zeE06%jh)5nesir- zF9&h;JzfyDdSTdVYY_$uOUuNUmzM4*c-XX5WK-1IW1u0EtgEZUo(sKoTo|CBqr)g)wc2dx(Sij}XAAY|#sQYKxB45w)CWNPJ%bI=rO=a* z^%pJ#e6Z>N^QT5cL?oh@=i0SvN|u(AGRG6&-Px(@_b}5$YE9HuF86-9t*`!6&TH@Q z=MT{m-Lq%Uf~#4eQKm(UltRM7oCH|T%rZUwOp|?ONLFUI0>{}tfh*UpC7qt8YarDN zYBx=P|F~}VqncX9$Xy~AG7@4oFZ;ah@3hkg)zzGSz1wYI^HU{4`1WmcPk{qlZe5!$ zYHiJY@Zf>3uVwSI&9lNoCCtsuedgQAD!cc&L`6x}{eEjc$D;61)_I1VF?}mmX!OY0 zN);9su2p+z^6bfz7Emo~WhIq*39_x=g&p@*B(4bIPK<=d-O}y<*52%m(|yqx94sB zX*2Ef-90z6T+VNaTCj-8DNXIhdHdPhn9rRxy(*jGS8E#(_wV^mUbY3o-=C@Jo<4p0 za2v0*PIbtPLYa2M8VeYFX?cN1Qf2=K|ZOqliSx#2wa zKYtIpyUsj!`@^&L=Lh@ccu!itKTvSqR=0lrjN7%J)p*wL=ew4{xqGU8*5WT_Z}qs< zCwd4-NOT-H;NUag?rzY_^}BXSS5R`6|M_`;Yc4%yaXfIW zSK9TrFT=9)IZjSY)AZx>LZ2U=V_;tKA)%tO@?p3BK7qgcmMmepU;AD5(BZ?gc^2r$ z>}XgUz1{5Q&-Jmp#hBUoa=sPI%E|R*US78K?!?Cjc;$F*etY{ss*e4WjmqU^ju9Ie4(%#kyZv?o!|t-P zTmJkoyP~P6b-Zwb*U}H4J~eqQb#ifG`SbVh!#8hoUe2(7ZDlm`Owrs%X7)p;Pjd&Y z+;a8xjOOMx`)zTTuRXt2^>5pjqtlMx+f~Z#-Y@4`RkiEsjNQAfg@lB1{3CK|;*R-9 zZso1Fy0p>EF5KHbcduXi*2&Y#w|aVda(;VrGr6?p$A^a%Kc7yYz2;Tsu|gR!ors2= z#m~FmO<{FB@bvWb!he4%S3j5Nb$fGj^YJWYBYu8&?9wmYL*T;IL z16E%>Yj^M6nS*lSjNjhm>}m5qcZy_w9C10&RwJ;w_}|C7M~#dY%(oCc^ziVb z;?FiJ#m}SiO`bJa?KH4?>DQ#NV8aFh&>;J>v$I{}dloKSxJ})vvZ8^Tm-Y1d_gj2x zs-*Q)SmrEO&v(;H~;N^9ViIG|6J3B2hwczzN z-IzTUg46Zm-O9?$Hf`R#Fl_b1$B&a|{_^(m5qYX~JlC3W^R6Zf0a=Gx>cl zG(7zAot?#tG{f`r^Ff{6;?G@N+=_E780{1fPVq>8_Wu4yttv}PgFQdhQtsF5uhKfj zE5)K`Czo@5UEkdO(-$)=n7(S=x?}Tft3iWkOO`M1Zc}7@u&`aOslUJ9_xU3yXT{3D zr5>9$%WiJ7JI?s`7vsHqCbbF5i!W;2vN)}yrNxzetY_m(8JXjW4-PbT^z|LvSNmJX zb6?e7&;-khl`93Geu~*!C2F2`M_}&Z!-tt4AM4$GyI^J>{CF8_m}1)(q;Iu=C5gO0hOJGWzho{(tnA zcJHJlB{p8EE6c8DbB9G_{Cbxuz)+y4Co4ZSnP*24>!D8J2`Z|8EAy5L)ugEOh%ijJ z=dt{9;`4KJPnf7*HQ7+{G3id}_1JCmeHJpa9_ zz@$%`@R?vciqMb5vad9V3oN(}MR=F9x+;8EYJ$GKE9ZWEIaA~Qx zhNkAmC(%20>`*W=3KE%8Dit37`rN+BWn!tSstLcpF-}(d8|mzz$G)N=`1uot3Gc3N z&yVMkGVw6^f1*%EuKrJ<&9@uLv0*>+udj=pVU{a(>Cz=>3zYP8|Xb;R*?bYE8{}wMex|;lZt~ z*}7%_(%70A-TPz?HZrqEz4>$Y)vH$pPfiGmh>0;7)TW!<6AW1!7Ol10EF>_nv8RXU z;_tVQA2&Zd++Oav;i$U{L$BG?ery0C#%T8aRPL{g5 zx{l6H(B#pbyLStposoQVfB*iQ-)~*NF7DjMbMV8%!w+A*>YA+X|LDcV#fz-}oH^qY zv$KeG;lhOlg@p%iY)meEazfDicwBgR`{U#N$)BE_T-5S#(j=jnJrx%}rT0Bv5VX>& z{ztqXyM3dG4~s7a4m0{Vn+SuVKFYgxR;>T($@g_Yl)#$qq3U z`ubI%?ic5q4-d~}&Yv5=aG{%5&PKvG{hUihg@s$MRO^BT3b$6Ylyn~pbzlrOJ@w6p zyRxDpf)%uOVe8hd)2jYG>)UQ2nP%Xr-u3mVPjdHC>Qz{-#vRbRCjE_1l}`SGob-{05K!Ew6S z$k^ERr&!QRk;ljT^Xnx`=CRk*9-R4lhJ`irym=4Wd71YYEL^Emz^r2~e(2)jx!b;) zU(Ps?cem-q?Tfdi=S-iz@|s6+OsryNmgJ6$9L~AAJ#!_O++bv7*gGRe@A#*ur>p*k=Bixw>cB_7F>+Io7&etv!~C@06)_jtkOmlH0n_|Ut1xAn1J>Fdj`t5)^CdpE)E ztH=#z{u$4HiE^P>)m3OzP@bxwOZnjdk#TTs(-aNVTZ!m)p=Z*~q2TpTq zXuhoY{icPfA+FD=K8fFV>ZRNm&MRT2rlw)5BP!-?7ZGIk`G4+g`6J$|SFbv*zh3@Q z!y!8+CT2s%MJ1oPR-$%xc5|(`o

8X6X%HPvi(@`9k1Yupk~ojRp+TIlZGyMm&k ztxGkrcf82h^6T7{>xXyL zTK8mMXX+KQvu-LbW4r$2zDI3sd)D`2Ma2hGybM6AhWb=YOs>3&sI0W)VRN4S{>0g< z&eA~|B8&Xa9X-1A{=Fw=Pu?8sJ$HNc0Y41}g_bRm!JrY;6X(z87YEK; z#?)X}`|H3~$IOV&MrJERv`+biufBRAYisPbpXdI5zn^cpy)p#U?OL?;<8uG`$IjdT z@7eu+-|p}y4_aDU6qJ;hrt8OV>*0CwZ9j41MA~ndy#STCL@mD-)5Tq&rF+sov*lb>piy}t4?jXv|wpk_Ev+s&%8Iz+n4+9e$&Im z;N&s;=g*%9Z{C#LE|n`Vb<(6u%NQJ%UtYLk#R;v0`YI|%Rt10m@};G_yF24#jH4st zsne%-=WyR+m#aAN>Z`_7uRWj7Sue7Ft-t?|(8Y@vIr#YAO)4u<`y9vn``cURxANs1 z+!hO}`_F6HzkmPLQ*DyQX(curE@ma(*(toE=4MFl=4Ho9*khx(rcF&<<#a3bX}2g> z!NP^h4mqWL{hAQI{@W?u8Y^K|hL#CtlTW6ArWUTod|c)`8?*xH#vU_!b93?Zb8`gm zUj1ytKh-N$l;Ov(UxyAKb{1g}5ff9=*5;l#aboPgKUdGh?E7(JWAe0QR#w&@58LHI zOX9ZL2k(Nk10|9rRD*tr=q*RNl{KJ;zFw*)EP7T?(le?B<3pT8Ow zQ(*98_j|EW$&@JF8(&}dMgH8vsKw&KkvYBk{od=ZKw}?$GL}h~KYy*V^_y>}`>KWE z@`wK$5)ZGrcFEn%Z4qenarebvzpCcgRBk#c#Ta|^@?~L9PtQejSoX&C$JhTAeSU84 z>uqHvEvr{+tNG83*8Hf5lx(@S?MEqt2!nvr zt<*y;oMEl?63G$!>-MJ3t@H-<1yfIpt$tef_xJbX9fHbh+UDBY*chaq5?SrCUD>_w z!JC_#-`0rc9{+5Ul$>l>^(Eut$2)go-rU_C9@4Q~-P_Ac=&E)?PvD zj=l)B;ESoTQK|nuJM-jx$z*|`paYjKy}7FGuWx@pA3x6@^Z(z2HlD^OPp2#1-pXC` zSjkyb)S&Ly%DGW@HwW-BC@zVun0Nfg$HxK%<;riQ#5B-6eexHUQ$+eOu65tY5)Jf?-xCvK6UED zr_Nm(7ai7yDcjr2i|fTWI65-ktNWdMn@MLXTZsA|qit_f=L_ zPP48N7Z*1;wkC{s#z&j}na1glHXfJDHto!Nb7N!0kB99W6A!apyt~J3@xgz8e;+KI zle*n&ZCLizowZ*sx=*Y6E&K4zn;v$#3Wno-vcgl(`uh4FZDeLIh!0jxsh)A-ghxwj zYvj|!imEClE2~}U{Qpmfl7s?zZ-}5gTA1~uj+{<+SGiXJ1xBk8x zkK7ES)^59bE?#eX^1C}bBX*Un^yN%1+9}g-^Qj|s_U&U5$rrM=u2{Kp&P z^7naPt+q)RBrwR;d~m#IxHqof-~R8GlW!+gONEDDSKH3_{rh)!ADx}N49f2L20$!Lq*8N%`NHfuF{A-6^3)>%$c?3{e6Msg%2v_KpE!LRPB#% zHlN>?Ilq#hpI;|(6N`4(8i)OLe{=TD+qqUwPA+VBRFO>j*UM7#o*UNxt7&O#>zWrH z8QBRM0RCYUz5eQz$G1NHm?oz^(PK@M-skEo3vY3KyJI-T@N&PxjIy`Z6#`$s-(h|{ zJD+JSYijT9+S%`Gyn-28Cip2%KfSTyBWMW+v! z#_4^jFO`lLN*JYhOnzQgSm@~F)U@JLl7oXo_f&rA{e2G}JV?t{$=DhtV^!i&a^~bo z&i_B>|G)X>XY|tR&K~~k+q0)X6_FaKrqyd$PM>4D zy_sFd_U^N?&70E|O@3IXo9AznGx?Gvv-s>?PwAZ7+jv2%9>dl~wca?Eoh{3e%p+g> zMR4D*SK4px?%p1*Qd4T~C;%G6j0UZ0V(xqV;Mp^_^78W4rvK(HUAOMsrl|%AuMSN0 zO5HBX`6~C)rAr^~6rb;0mVW5);g3I`&%eE8Khx@5S8y>D!Ca%^+C+~7~M)>*h zU0WBMJ$1HW?XQyEdFxzu@v}0t2t0E5Qf1pGYu)B=_mk!Nx>(`AdI|~*o}QlD=AXGL z8sovy%V46abEb;@kX##RMRXxUk0GczpM7o3!IiG08sQN`chM6QY1DYV>wKHKlb#e447R~bn$Dzpf&FAQ2KAS&9r zXps`bGS#Nf&(1nKIXP`t@e2zNSJu|%4h;>p-B(<~0;-xinm>H`;*ygC>Vg`Uy@^m! zQ8{q`{{E|~cZV2)++hQh~gpk{bMQPHOI$!`v2m?-J!@Dx8k zw{%q|OsSur-;Uq!b{l40(GU?AH{W@8m9oL~*>5i|=dYNTc8NoAFFz~8lCyK>%n=X} zXlQ6)s5WmEaPqLexRzV7Mc`2bsNE;vBqIPy3Qir&ix?UNoHz;{K;glm$kzdiam6Ey zfuPuHu}}atUt0v&Aq(eepxI;l0}DQefbj72=hMFU^cYI?x)sQ<&wKytUF5B&;R-DR zP7Ez=ZHEpYX5Rnj6L%YrBxr?OwfV~92|Ww9Y!L}x7t^?Yz5brPdw2htC6nA!`ZJV6 zvBkn>Z(RSACn-61cZufZ<$d`0abx-9DxP8n1_sd1jI^_}u0FlK1mtm#?Vy^!m0P^v z*%`?U6RB$RDz)#;$^XP&NpF6*2UngAY>rnmqA From 12769b96e560cc97d6c414255280099d022096e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jul 2019 13:36:06 -0700 Subject: [PATCH 118/880] Drop support for omitting pdfminer.six --- docs/release_notes.rst | 8 ++++++++ requirements/main.txt | 2 +- setup.py | 5 ++--- src/ocrmypdf/pdfinfo/__init__.py | 9 +-------- tests/conftest.py | 10 ---------- tests/test_main.py | 1 - tests/test_metadata.py | 11 ++++++----- tests/test_pdfinfo.py | 5 ----- tests/test_weave.py | 1 - 9 files changed, 18 insertions(+), 34 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 67d4e5e5..b0d1976f 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,14 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and ar find: [^`]\#([0-9]{1,3})[^0-9] replace: `#$1 `_ +v8.3.2 +------ + +- Dropped workaround for macOS that allowed it work without pdfminer.six, + now a proper sdist release of pdfminer.six is available. + +- pikepdf 1.5.0 is now required. + v8.3.1 ------ diff --git a/requirements/main.txt b/requirements/main.txt index 383d7189..787072fe 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ chardet == 3.0.4 cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 -pikepdf == 1.3.0 +pikepdf == 1.5.0.post0 Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 diff --git a/setup.py b/setup.py index b5c959ea..d2564c1e 100644 --- a/setup.py +++ b/setup.py @@ -98,15 +98,14 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six == 20181108 ; sys_platform != "darwin"', - 'pikepdf >= 1.3.0, < 2', + 'pdfminer.six == 20181108', + 'pikepdf >= 1.5.0, < 2', 'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"', # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels 'reportlab >= 3.3.0', # oldest released version with sane image handling 'ruffus >= 2.7.0', ], - extras_require={'pdfminer': ['pdfminer.six == 20181108']}, tests_require=tests_require, entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run_pipeline']}, package_data={'ocrmypdf': ['data/sRGB.icc']}, diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index aaad8ebe..ec6e13c2 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -30,6 +30,7 @@ from pikepdf import PdfMatrix import pikepdf from . import ghosttext +from .layout import get_page_analysis, get_text_boxes from ..exceptions import EncryptedPdfError, MissingDependencyError @@ -564,14 +565,6 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): ) pageinfo['bboxes'] = bboxes else: - # pdfminer required for this section - try: - from .layout import get_page_analysis, get_text_boxes - except ImportError: - raise MissingDependencyError( - "pdfminer is required for this feature. Your distribution " - "may not have installed it." - ) pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') miner = get_page_analysis(infile, pageno, pscript5_mode) pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) diff --git a/tests/conftest.py b/tests/conftest.py index 19679fdf..dad59987 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -62,16 +62,6 @@ def running_in_travis(): return os.environ.get('TRAVIS') == 'true' -@pytest.helpers.register -def needs_pdfminer(fn): - try: - import pdfminer - except ImportError: - skip = pytest.mark.skipif(True, reason="pdfminer not available") - return skip(fn) - return fn - - @pytest.helpers.register def have_unpaper(): try: diff --git a/tests/test_main.py b/tests/test_main.py index 248a9003..0ec9d346 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -219,7 +219,6 @@ def test_skip_ocr(spoof_tesseract_cache, resources, outpdf): assert pdfinfo[0].has_text -@pytest.helpers.needs_pdfminer def test_redo_ocr(spoof_tesseract_cache, resources, outpdf): in_ = resources / 'graph_ocred.pdf' before = PdfInfo(in_, detailed_page_analysis=True) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 442c71a1..6b85faa9 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -22,7 +22,7 @@ import logging import mmap from os import fspath from pathlib import Path -from shutil import copyfile +from shutil import copyfile, move from unittest.mock import MagicMock, patch import pytest @@ -308,10 +308,11 @@ def test_metadata_fixup_warning(resources, outdir, caplog): assert record.levelname != 'WARNING' # Now add some metadata that will not be copyable - graph = pikepdf.open(outdir / 'graph.repaired.pdf') - with graph.open_metadata() as meta: - meta['prism2:publicationName'] = 'OCRmyPDF Test' - graph.save(outdir / 'graph.repaired.pdf') + with pikepdf.open(outdir / 'graph.repaired.pdf') as graph: + with graph.open_metadata() as meta: + meta['prism2:publicationName'] = 'OCRmyPDF Test' + graph.save(outdir / 'graph.repaired.modified.pdf') + move(outdir / 'graph.repaired.modified.pdf', outdir / 'graph.repaired.pdf') log = logging.getLogger() context = MagicMock() diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index a4ab14f6..ff811488 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -183,12 +183,7 @@ def test_ocr_detection(resources): @pytest.mark.parametrize( 'testfile', ('truetype_font_nomapping.pdf', 'type3_font_nomapping.pdf') ) -@pytest.helpers.needs_pdfminer # pylint: disable=e1101 def test_corrupt_font_detection(resources, testfile): - try: - import pdfminer - except ImportError: - pytest.skip("Needs pdfminer") filename = resources / testfile with pytest.raises(NotImplementedError): pdf = pdfinfo.PdfInfo(filename) diff --git a/tests/test_weave.py b/tests/test_weave.py index 06fa2a0a..0d7e3a48 100644 --- a/tests/test_weave.py +++ b/tests/test_weave.py @@ -44,7 +44,6 @@ def test_no_glyphless_weave(resources, outdir): ) -@pytest.helpers.needs_pdfminer def test_links(resources, outpdf): check_ocrmypdf( resources / 'link.pdf', From 0c781faf8998f4209c2a83b790d0128f15da33a6 Mon Sep 17 00:00:00 2001 From: jbarlow83 Date: Thu, 11 Jul 2019 00:35:39 -0700 Subject: [PATCH 119/880] Create funding.yml [ci skip] --- .github/FUNDING.yml | 12 ++++++++++++ 1 file changed, 12 insertions(+) create mode 100644 .github/FUNDING.yml diff --git a/.github/FUNDING.yml b/.github/FUNDING.yml new file mode 100644 index 00000000..8dfc9e79 --- /dev/null +++ b/.github/FUNDING.yml @@ -0,0 +1,12 @@ +# These are supported funding model platforms + +github: # Replace with up to 4 GitHub Sponsors-enabled usernames e.g., [user1, user2] +patreon: # Replace with a single Patreon username +open_collective: https://opencollective.com/james-barlow +ko_fi: # Replace with a single Ko-fi username +tidelift: # Replace with a single Tidelift platform-name/package-name e.g., npm/babel +community_bridge: # Replace with a single Community Bridge project-name e.g., cloud-foundry +liberapay: # Replace with a single Liberapay username +issuehunt: # Replace with a single IssueHunt username +otechie: # Replace with a single Otechie username +custom: # Replace with up to 4 custom sponsorship URLs e.g., ['link1', 'link2'] From b601cb0cba9422d5b3e7930c2390e11090bd2d81 Mon Sep 17 00:00:00 2001 From: jbarlow83 Date: Thu, 11 Jul 2019 00:36:09 -0700 Subject: [PATCH 120/880] Fix funding.yml --- .github/FUNDING.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/FUNDING.yml b/.github/FUNDING.yml index 8dfc9e79..f9e58e51 100644 --- a/.github/FUNDING.yml +++ b/.github/FUNDING.yml @@ -2,7 +2,7 @@ github: # Replace with up to 4 GitHub Sponsors-enabled usernames e.g., [user1, user2] patreon: # Replace with a single Patreon username -open_collective: https://opencollective.com/james-barlow +open_collective: james-barlow ko_fi: # Replace with a single Ko-fi username tidelift: # Replace with a single Tidelift platform-name/package-name e.g., npm/babel community_bridge: # Replace with a single Community Bridge project-name e.g., cloud-foundry From 7117dc10de671a69de1c167b47ba21fa4347b5c2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 11 Jul 2019 01:23:01 -0700 Subject: [PATCH 121/880] Suppress noisy empty debug messages --- src/ocrmypdf/exec/ghostscript.py | 2 +- src/ocrmypdf/optimize.py | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index b744cdbf..b39ff628 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -166,7 +166,7 @@ def rasterize_pdf( p = run(args_gs, stdout=PIPE, stderr=STDOUT, universal_newlines=True) if _gs_error_reported(p.stdout): log.error(p.stdout) - else: + elif p.stdout: log.debug(p.stdout) if p.returncode != 0: diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 0ee1ef98..405f48af 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -276,7 +276,8 @@ def _produce_jbig2_images(jbig2_groups, root, log, options): ) as pbar: for future in concurrent.futures.as_completed(futures): proc = future.result() - log.debug(proc.stderr.decode()) + if proc.stderr: + log.debug(proc.stderr.decode()) pbar.update() From 6189910c74b70e4c2c141117fe4b57ace08a5b32 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 11 Jul 2019 02:20:04 -0700 Subject: [PATCH 122/880] Fix text-image registration when mediabox contains an offset Cropbox, trimbox not addressed... should look at those. Also rotation. --- src/ocrmypdf/_graft.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 3bdb79ae..6a7cf23a 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -21,8 +21,6 @@ from pathlib import Path import pikepdf -from .exec import tesseract - MAX_REPLACE_PAGES = int(os.environ.get('_OCRMYPDF_MAX_REPLACE_PAGES', 100)) @@ -117,6 +115,7 @@ def _graft_text_layer( translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2) untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2) + corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) # -rotation because the input is a clockwise angle and this formula # uses CCW rotation = -rotation % 360 @@ -134,8 +133,9 @@ def _graft_text_layer( scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) # Translate the text so it is centered at (0, 0), rotate it there, adjust - # for a size different between initial and text PDF, then untranslate - ctm = translate @ rotate @ scale @ untranslate + # for a size different between initial and text PDF, then untranslate, and + # finally move the lower left corner to match the mediabox + ctm = translate @ rotate @ scale @ untranslate @ corner pdf_text_contents = b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' @@ -221,8 +221,9 @@ class OcrGrafter: text_rotation = autorotate_correction text_misaligned = (text_rotation - content_rotation) % 360 self.log.debug( - '%r', - [text_rotation, autorotate_correction, text_misaligned, content_rotation], + f"Rotations for page {pageno}: [text, auto, misalign, content] = " + f"{text_rotation}, {autorotate_correction}, " + f"{text_misaligned}, {content_rotation}" ) if text and self.font: From 016a2a01d9d55cf575e1eca60617dd2e417c0b06 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 13 Jul 2019 02:06:11 -0700 Subject: [PATCH 123/880] docs: Notes on WSL --- docs/installation.rst | 35 +++++++++++++++++++++++++++++------ 1 file changed, 29 insertions(+), 6 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 38e1fb35..48af600d 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -105,6 +105,8 @@ sources`_. detect it on the ``PATH``. To add JBIG2 encoding, see `Installing the JBIG2 encoder `_. +.. _ubuntu-lts-latest: + Installing the latest version on Ubuntu 18.04 LTS ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ @@ -384,13 +386,28 @@ See `OCRmyPDF Docker Image `_ for more information. Installing on Windows --------------------- -Direct installation on Windows is not possible, because there are a -POSIX dependencies. Your options are: +Direct installation on Windows is not currently possible, but it works well in +Windows Subsystem for Linux: -* Install Ubuntu 18.04 in Windows 10 Subsystem for Linux, then follow - the Ubuntu 18.04 procedure. -* `Install the Docker `__ container. Ensure that - your command prompt can run the docker "hello world" container. +#. Install Ubuntu 18.04 for Windows Subsystem for Linux, if not already installed. +#. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 18.04 `. +#. Open the Windows command prompt and create a symlink: + +.. code-block:: powershell + + wsl sudo ln -s /home/user/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf + +Then confirm that the expected version from PyPI (|latest|) is installed: + +.. code-block:: powershell + + wsl ocrmypdf --version + +You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing +``wsl``, and call it from Windows programs or batch files. + +Why no native Windows? +^^^^^^^^^^^^^^^^^^^^^^ It would probably not be too difficult to port on Windows. The main reason this has been avoided is the difficulty of packaging and @@ -398,6 +415,12 @@ installing the various non-Python dependencies: Tesseract, QPDF, Ghostscript, Leptonica. Pull requests to add or improve Windows support would be quite welcome. +Docker +^^^^^^ + +You can also :ref:`Install the Docker ` container on Windows. Ensure that +your command prompt can run the docker "hello world" container. + Installing with Python pip -------------------------- From f83de20c3796e66246825ebd609c2759aae0f97d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 01:41:14 -0700 Subject: [PATCH 124/880] Remove plugins (for now) It's holding up too many other useful, releaseable changes. --- docs/index.rst | 1 - docs/release_notes.rst | 2 - src/ocrmypdf/_pipeline.py | 12 ------ src/ocrmypdf/_plugins.py | 79 ---------------------------------- src/ocrmypdf/api.py | 9 +--- src/ocrmypdf/cli.py | 9 ---- tests/test_filters.py | 90 --------------------------------------- 7 files changed, 1 insertion(+), 201 deletions(-) delete mode 100644 src/ocrmypdf/_plugins.py delete mode 100644 tests/test_filters.py diff --git a/docs/index.rst b/docs/index.rst index 20b4ad79..29136779 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -23,7 +23,6 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat docker advanced api - plugins batch security errors diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 2b28d529..0488ee43 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -27,8 +27,6 @@ next - Added a high level API for applications that want to integrate OCRmyPDF. Special thanks to Martin Wind (@mawi1988) whose made significant contributions to this effort. -- Added a simple plugin interface that makes certain steps of the pipeline - configurable. - Added progress bars for long-running steps. As such, the behavior of output messages is different. - Dropped the ``ocrmypdf-polyglot`` and ``ocrmypdf-webservice`` images. diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index f5cc79c6..02042112 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -28,8 +28,6 @@ import pikepdf from pikepdf.models.metadata import encode_pdf_date from . import PROGRAM_NAME, VERSION, leptonica - -from ._plugins import load_plugin from .exceptions import ( DpiError, EncryptedPdfError, @@ -160,12 +158,6 @@ def validate_pdfinfo_options(context): pdfinfo = context.pdfinfo options = context.options - if options.plugin_validation: - validate = load_plugin(options.plugin_validation) - result = validate(context) - if result is not None: - return result - if pdfinfo.needs_rendering: log.error( "This PDF contains dynamic XFA forms created by Adobe LiveCycle " @@ -536,10 +528,6 @@ def create_ocr_image(image, page_context): pix = pix.masked_threshold_on_background_norm() im = pix.topil() - if options.filter_ocr_image: - filt = load_plugin(options.filter_ocr_image) - im = filt(im) - del draw # Pillow requires integer DPI dpi = round(xres), round(yres) diff --git a/src/ocrmypdf/_plugins.py b/src/ocrmypdf/_plugins.py deleted file mode 100644 index 2a52b105..00000000 --- a/src/ocrmypdf/_plugins.py +++ /dev/null @@ -1,79 +0,0 @@ -# © 2019 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -import logging -import importlib -import os -import sys -from pathlib import Path - -log = logging.getLogger(__name__) - - -def _load_object_from_module(location): - """Load a object given a module location - - For location=a.b.c, will effectively run "from a.b import c" - - Example: - _load_object_from_module("a.b.c") - - """ - module_parts = location.split('.') - module_name = '.'.join(module_parts[:-1]) - object_name = module_parts[-1] - module = importlib.import_module(module_name) - obj = getattr(module, object_name) - log.debug(f"Loaded object: from {module_name} import {object_name}") - return obj - - -def _load_object_from_pyfile(location): - """Load a object from a file - - Example: - _load_object_from_pyfile("test.py::blur_filter") - """ - filename, object_name = location.split('::', maxsplit=1) - log.debug(f"Loading object {object_name} from {filename}") - - module_name = Path(filename).stem - spec = importlib.util.spec_from_file_location(module_name, filename) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - obj = getattr(module, object_name) - return obj - - -def load_plugin(plugin): - if callable(plugin): - return plugin - - if not isinstance(plugin, str): - raise TypeError() - - if '::' not in plugin: - plugin = _load_object_from_module(plugin) - else: - plugin = _load_object_from_pyfile(plugin) - - return plugin - - -def check_plugin_loadable(plugin): - load_plugin(plugin) - return plugin diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 9704f057..8486632f 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -122,12 +122,7 @@ def create_options(*, input_file, output_file, **kwargs): for arg, val in kwargs.items(): if val is None: continue - if (arg.startswith('plugin') or arg.startswith('filter')) and ( - callable(val) or isinstance(val, str) - ): - deferred.append((arg, val)) - continue - elif arg == 'tesseract_env': + if arg == 'tesseract_env': deferred.append((arg, val)) continue cmd_style_arg = arg.replace('_', '-') @@ -203,8 +198,6 @@ def ocr( # pylint: disable=unused-argument user_patterns=None, keep_temporary_files=None, progress_bar=None, - filter_ocr_image=None, - plugin_validation=None, tesseract_env=None, ): """Run OCRmyPDF on one PDF or image. diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 5e639155..2d5b6788 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -18,7 +18,6 @@ import argparse from . import PROGRAM_NAME, VERSION -from ._plugins import check_plugin_loadable def numeric(basetype, min_=None, max_=None): @@ -467,14 +466,6 @@ advanced.add_argument( help="Specify the location of the Tesseract user patterns file.", ) -plugins = parser.add_argument_group("Filters and Plugins", argparse.SUPPRESS) -plugins.add_argument( - '--filter-ocr-image', help=argparse.SUPPRESS, type=check_plugin_loadable -) -plugins.add_argument( - '--plugin-validation', help=argparse.SUPPRESS, type=check_plugin_loadable -) - debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" ) diff --git a/tests/test_filters.py b/tests/test_filters.py deleted file mode 100644 index 128001d8..00000000 --- a/tests/test_filters.py +++ /dev/null @@ -1,90 +0,0 @@ -# © 2019 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -import os - -from PIL import Image -import pytest - - -import ocrmypdf -from ocrmypdf.filters import invert, whiteout -from ocrmypdf._plugins import load_plugin - - -check_ocrmypdf = pytest.helpers.check_ocrmypdf - - -def filter_42(): - return 42 - - -def test_pyfile(): - obj = load_plugin(f'{__file__}::filter_42') - assert obj() == 42 - - -def test_pyfile_notexist(): - with pytest.raises(FileNotFoundError): - load_plugin('thisfile.doesnot.exist.py::filter_42') - - -def test_pyfile_noobject(): - with pytest.raises(AttributeError): - load_plugin(f'{__file__}::no_function_with_this_name') - - -def test_module(): - obj = load_plugin(f'os.getuid') - assert obj() == os.getuid() - - -def test_module_notexist(): - with pytest.raises(ModuleNotFoundError): - load_plugin('thismodule.doesnot.exist') - - -def test_filter_from_cmdline(resources, outdir): - (outdir / 'temp.py').write_text( - "from PIL import Image\n" - "def whiteout(im):\n" - " return Image.new(im.mode, im.size)\n" - ) - - check_ocrmypdf( - resources / 'crom.png', - outdir / 'out.pdf', - '--image-dpi', - '100', - '--sidecar', - outdir / 'sidecar.txt', - '--filter-ocr-image', - f"{outdir / 'temp.py'}::whiteout", - ) - - assert (outdir / 'sidecar.txt').read_text().strip() == '' - - -def test_filter_from_api(resources, outdir): - ocrmypdf.ocr( - resources / 'crom.png', - outdir / 'out.pdf', - image_dpi=100, - sidecar=outdir / 'sidecar.txt', - filter_ocr_image=whiteout, - ) - assert (outdir / 'sidecar.txt').read_text().strip() == '' From 5304c631ec23e96bdcdc586ebbff2afb80a6b46b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 02:14:59 -0700 Subject: [PATCH 125/880] Don't warn about --user-words in Tesseract 4.1 or later --- src/ocrmypdf/_validation.py | 7 +++++-- src/ocrmypdf/exec/tesseract.py | 9 +++++++++ tests/test_validation.py | 9 +++++++-- 3 files changed, 21 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 4163a58d..02bf043b 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -253,10 +253,13 @@ def check_options_advanced(options): "--pdfa-image-compression argument has no effect when " "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" ) - if tesseract.v4(options.tesseract_env) and ( + if not tesseract.has_user_words(options.tesseract_env) and ( options.user_words or options.user_patterns ): - log.warning('Tesseract 4.x ignores --user-words, so this has no effect') + log.warning( + "Tesseract 4.0 ignores --user-words and --user-patterns, so these " + "arguments have no effect." + ) def check_options_metadata(options): diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index c16a9202..863047a1 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -88,6 +88,15 @@ def has_textonly_pdf(tesseract_env=None): return False +def has_user_words(tesseract_env=None): + """Does Tesseract have --user-words capability? + + Not available in 4.0, but available in 4.1. Also available in 3.x, but + we no longer support 3.x. + """ + return version(tesseract_env) >= '4.1' + + def languages(tesseract_env=None): def lang_error(output): msg = ( diff --git a/tests/test_validation.py b/tests/test_validation.py index c151ec16..138c8820 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -81,8 +81,13 @@ def test_optimizing(caplog): def test_user_words(caplog): - vd.check_options_advanced(make_opts(user_words='foo')) - assert 'ignores --user-words' in caplog.text + with patch('ocrmypdf.exec.tesseract.version', return_value='4.0.0'): + vd.check_options_advanced(make_opts(user_words='foo')) + assert '4.0 ignores --user-words' in caplog.text + caplog.clear() + with patch('ocrmypdf.exec.tesseract.version', return_value='4.1.0'): + vd.check_options_advanced(make_opts(user_patterns='foo')) + assert '4.0 ignores --user-words' not in caplog.text def test_pillow_options(): From 4d011c28ea885742d28e8d9fc7dc8117b4fe6506 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 03:02:04 -0700 Subject: [PATCH 126/880] Improve completions --- misc/completion/ocrmypdf.bash | 2 +- misc/completion/ocrmypdf.fish | 59 +++++++++++++++++++++++++++++------ 2 files changed, 50 insertions(+), 11 deletions(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 1b550da9..64df612d 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -76,7 +76,7 @@ _ocrmypdf() --max-image-mpixels --tesseract-config --tesseract-pagesegmode --help --tesseract-oem --pdf-renderer --tesseract-timeout --rotate-pages-threshold --pdfa-image-compression --user-words - --user-patterns --keep-temporary-files --flowchart --output-type' \ + --user-patterns --keep-temporary-files --output-type' \ -- "$cur" ) ) return else diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 24883be1..645f6f98 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -1,8 +1,8 @@ -complete -c ocrmypdf -l version -complete -c ocrmypdf -l help +complete -c ocrmypdf -x -n '__fish_is_first_arg' -l version +complete -c ocrmypdf -x -n '__fish_is_first_arg' -s h -s "?" -l help -complete -c ocrmypdf -l sidecar -r -d "write OCR to text file" -complete -c ocrmypdf -s q -l quiet +complete -c ocrmypdf -r -l sidecar -d "write OCR to text file" +complete -c ocrmypdf -x -s q -l quiet complete -c ocrmypdf -s r -l rotate-pages -d "rotate pages to correct orientation" complete -c ocrmypdf -s d -l deskew -d "fix small horizontal alignment skew" @@ -17,8 +17,14 @@ complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)" -complete -c ocrmypdf -x -s l -l language -d 'language' -complete -c ocrmypdf -x -s l -l language -a '(tesseract --list-langs)' +function __fish_ocrmypdf_languages + set langs (tesseract --list-langs ^/dev/null) + set arr (string split '\n' $langs) + for lang in $arr[2..-1] + echo $lang + end +end +complete -c ocrmypdf -x -s l -l language -a '(__fish_ocrmypdf_languages)' -d "language" complete -c ocrmypdf -x -l image-dpi -d "assume this DPI if input image DPI is unknown" @@ -36,7 +42,7 @@ function __fish_ocrmypdf_pdf_renderer echo -e "hocr\t"(_ "use hocr renderer") echo -e "sandwich\t"(_ "use sandwich renderer") end -complete -c ocrmypdf -x -l pdf-render -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options" +complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options" function __fish_ocrmypdf_optimize echo -e "0\t"(_ "do not optimize") @@ -53,6 +59,13 @@ function __fish_ocrmypdf_verbose end complete -c ocrmypdf -x -s v -l verbose -a '(__fish_ocrmypdf_verbose)' -d "set verbosity level" +function __fish_ocrmypdf_pdfa_compression + echo -e "auto\t"(_ "let Ghostscript decide how to compress images") + echo -e "jpeg\t"(_ "convert color and grayscale images to JPEG") + echo -e "lossless\t"(_ "convert color and grayscale images to lossless (PNG)") +end +complete -c ocrmypdf -x -l pdfa-image-compression -a '(__fish_ocrmypdf_pdfa_compression)' -d "set PDF/A image compression options" + complete -c ocrmypdf -x -s j -l jobs -d "how many worker processes to use" complete -c ocrmypdf -x -l title -d "set metadata" complete -c ocrmypdf -x -l author -d "set metadata" @@ -66,10 +79,36 @@ complete -c ocrmypdf -x -l png-quality -d "PNG quality [0..100]" complete -c ocrmypdf -x -l jbig2-lossy -d "enable lossy JBIG2 (see docs)" complete -c ocrmypdf -x -l max-image-mpixels -d "image decompression bomb threshold" complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file" -complete -c ocrmypdf -x -l tesseract-pagesegmode -d "set tesseract --psm" -complete -c ocrmypdf -x -l tesseract-oem -d "set tesseract --oem" + +function __fish_ocrmypdf_tesseract_pagesegmode + echo -e "0\t"(_ "orientation and script detection (OSD) only") + echo -e "1\t"(_ "automatic page segmentation with OSD") + echo -e "2\t"(_ "automatic page segmentation, but no OSD, or OCR") + echo -e "3\t"(_ "fully automatic page segmentation, but no OSD (default)") + echo -e "4\t"(_ "assume a single column of text of variable sizes") + echo -e "5\t"(_ "assume a single uniform block of vertically aligned text") + echo -e "6\t"(_ "assume a single uniform block of text") + echo -e "7\t"(_ "treat the image as a single text line") + echo -e "8\t"(_ "treat the image as a single word") + echo -e "9\t"(_ "treat the image as a single word in a circle") + echo -e "10\t"(_ "treat the image as a single character") + echo -e "11\t"(_ "sparse text - find as much text as possible in no particular order") + echo -e "12\t"(_ "sparse text with OSD") + echo -e "13\t"(_ "raw line - treat the image as a single text line") +end +complete -c ocrmypdf -x -l tesseract-pagesegmode -a '(__fish_ocrmypdf_tesseract_pagesegmode)' -d "set tesseract --psm" + +function __fish_ocrmypdf_tesseract_oem + echo -e "0\t"(_ "legacy engine only") + echo -e "1\t"(_ "neural nets LSTM engine only") + echo -e "2\t"(_ "legacy + LSTM engines") + echo -e "3\t"(_ "default, based on what is available") +end +complete -c ocrmypdf -x -l tesseract-oem -a '(__fish_ocrmypdf_tesseract_oem)' -d "set tesseract --oem" complete -c ocrmypdf -x -l tesseract-timeout -d "maximum number of seconds to wait for OCR" complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence" -complete -c ocrmypdf -x -l pdfa-image-compression -a 'auto jpeg lossless' -d "set PDF/A image compression options" + +complete -c ocrmypdf -r -l user-words -d "specify location of user words file" +complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file" complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)" From 85c90404d7ac17478d07c648f707ba208498ec0e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 03:23:56 -0700 Subject: [PATCH 127/880] Update release notes --- docs/release_notes.rst | 27 +++++++++++++++++++++++---- 1 file changed, 23 insertions(+), 4 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index e7dbb793..e09b2179 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -21,24 +21,43 @@ v9.0.0 - The ``--mask-barcodes`` experimental feature has been dropped due to poor reliability and occasional crashes, both due to the underlying library that implements this feature (Leptonica). +- The ``-v`` (verbosity level) parameter now accepts only ``0``, ``1``, and + ``2``. +- Dropped support for Tesseract "4.00.00-alpha" releases. Tesseract 4.0 beta and + later remain supported. **Major changes** - Added a high level API for applications that want to integrate OCRmyPDF. Special thanks to Martin Wind (@mawi1988) whose made significant contributions - to this effort. -- Added progress bars for long-running steps. As such, the behavior of output - messages is different. + to this effort. OCRmyPDF is GPLv3-licensed. +- Major internal code reorganization. +- Added progress bars for long-running steps. ■■■■■■■□□ +- When the number of pages is small compared to the number of allowed jobs, we + run Tesseract in multithreaded (OpenMP) mode when available. This should + improve performance on files with low page counts. +- Pages with vector artwork are treated as full color. Previously, vectors + were ignored when considering the colorspace needed to cover a page, which + could cause loss of color under certain settings. - Dropped the ``ocrmypdf-polyglot`` and ``ocrmypdf-webservice`` images. - Removed dependency on ``ruffus``, and with that, the non-reentrancy restrictions that previous made an API impossible. -- Internal code reorganization. +- Added a new ``--pages`` feature to limit OCR to only a specific page range. + The list may contain commas or single pages, such as ``1, 3, 5-11``. +- Output and logging messages overhauled so that ocrmypdf may be integrated + into applications that use the logging module. +- pikepdf 1.6.0 is required. **Minor changes** - Test suite now spawns processes less frequently, allowing more accurate measurement of code coverage. +- Improved test coverage. +- Fixed a rare division by zero (if optimization produced an invalid file). - Updated Docker images to use newer versions. +- Fixed images encoded as JBIG2 with a colorspace other than ``/DeviceGray`` + were not interpreted correctly. +- We have a logo. 😊 v8.3.2 ====== From e4cfcec5f3ddf6abb46384dab140d6692ec91e07 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 04:04:33 -0700 Subject: [PATCH 128/880] docs: some cleanup --- docs/api.rst | 14 ++++++++------ docs/introduction.rst | 2 +- src/ocrmypdf/api.py | 6 +++--- 3 files changed, 12 insertions(+), 10 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index 2e9e7597..09a68539 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -31,11 +31,13 @@ Instead, output should be managed by configuring logging. Parent process requirements --------------------------- -The :func:`ocrmypdf.run` function runs OCRmyPDF similar to command line -execution. To do this, it will: - create a monitoring thread - create -worker processes (forking itself) - manage the signal flags of worker -processes 0 execute other subprocesses (forking and executing other -programs) +The :func:`ocrmypdf.ocr` function runs OCRmyPDF similar to command line +execution. To do this, it will: + +- create a monitoring thread +- create worker processes (forking itself) +- manage the signal flags of worker processes +- execute other subprocesses (forking and executing other programs) The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently privileged to perform these actions. If it is not, ``ocrmypdf()`` will @@ -88,7 +90,7 @@ When OCRmyPDF succeeds conditionally, it returns an integer exit code. Reference --------- -.. autofunction:: ocrmypdf.run +.. autofunction:: ocrmypdf.ocr .. autoclass:: ocrmypdf.Verbosity :members: diff --git a/docs/introduction.rst b/docs/introduction.rst index 0643c7e7..acc23aee 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -161,7 +161,7 @@ OCRmyPDF is also limited by the PDF specification: the spaces between words must be derived heuristically. Some PDF viewers do a better job of this than others. - Because some popular open source PDF viewers have a particularly hard - time with spaces betweem words, OCRmyPDF appends a space to each text + time with spaces between words, OCRmyPDF appends a space to each text element as a workaround (when using ``--pdf-renderer hocr``). While this mixes document structure with graphical information that ideally should be left to the PDF viewer to interpret, it improves diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 8486632f..bc4a4e5a 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -64,10 +64,10 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= similar to ocrmypdf command line interface. If not used, the external application should configure logging on its own. - ocrmypdf will perform all of its logging under the `"ocrmypdf"` logging namespace. - In addition, ocrmypdf imports pdfminer, which logs under `"pdfminer"`. A library + ocrmypdf will perform all of its logging under the ``"ocrmypdf"`` logging namespace. + In addition, ocrmypdf imports pdfminer, which logs under ``"pdfminer"``. A library user may wish to configure both; note that pdfminer is extremely chatty at the log - level logging.INFO. + level ``logging.INFO``. Library users may perform additional configuration afterwards. From 0c066d1d533d9a21422411f6df9234042d213d7d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 04:26:23 -0700 Subject: [PATCH 129/880] Expand scope of --pages testing --- src/ocrmypdf/_validation.py | 5 ++++- tests/test_page_numbers.py | 30 ++++++++++++++++++++++++------ 2 files changed, 28 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 02bf043b..5c9bc2a1 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -185,7 +185,10 @@ def _pages_from_ranges(ranges): except ValueError: pages.append(int(g) - 1) else: - pages.extend(range(int(start) - 1, int(end))) + try: + pages.extend(range(int(start) - 1, int(end))) + except ValueError: + raise BadArgsError("invalid page range") if not monotonic(pages): log.warning( diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index b2cd7879..46643037 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -20,14 +20,32 @@ import pytest import ocrmypdf from ocrmypdf._validation import _pages_from_ranges from ocrmypdf.pdfinfo import PdfInfo +from ocrmypdf.exceptions import BadArgsError -def test_str_ranges(): - assert _pages_from_ranges('43') == {42} - assert _pages_from_ranges('1, 2, 3') == {0, 1, 2} - assert _pages_from_ranges('1-3') == {0, 1, 2} - assert _pages_from_ranges('1-3,5,7,42') == {0, 1, 2, 4, 6, 41} - assert _pages_from_ranges('3, 3, 3, 3,') == {2} +@pytest.mark.parametrize( + 'pages, result', + [ + ['1', {0}], + ['1,2', {0, 1}], + ['1-3', {0, 1, 2}], + ['2,5,6', {1, 4, 5}], + ['11-15, 18, ', {10, 11, 12, 13, 14, 17}], + [',,3', {2}], + ['3, 3, 3, 3,', {2}], + ['3, 2, 1, 42', {0, 1, 2, 41}], + ['-1', BadArgsError], + ['1,3,-11', BadArgsError], + ['1-,', BadArgsError], + ['start-end', BadArgsError], + ], +) +def test_pages(pages, result): + if isinstance(result, type): + with pytest.raises(result): + _pages_from_ranges(pages) + else: + assert _pages_from_ranges(pages) == result def test_nonmonotonic_warning(caplog): From b0f1a555375520b5c3030afae05cdcdb6e6af6b4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 04:26:38 -0700 Subject: [PATCH 130/880] completions: --pages --- misc/completion/ocrmypdf.bash | 2 +- misc/completion/ocrmypdf.fish | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 64df612d..9ea30f3d 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -58,7 +58,7 @@ _ocrmypdf() COMPREPLY=( $( compgen -W '{1..13}' -- "$cur" ) ) return ;; - --sidecar|--title|--author|--subject|--keywords|--unpaper-args) + --sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages) # argument required but no completions available return ;; diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 645f6f98..81d24e6c 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -78,6 +78,7 @@ complete -c ocrmypdf -x -l jpeg-quality -d "JPEG quality [0..100]" complete -c ocrmypdf -x -l png-quality -d "PNG quality [0..100]" complete -c ocrmypdf -x -l jbig2-lossy -d "enable lossy JBIG2 (see docs)" complete -c ocrmypdf -x -l max-image-mpixels -d "image decompression bomb threshold" +complete -c ocrmypdf -x -l pages -d "apply OCR to only the specified pages" complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file" function __fish_ocrmypdf_tesseract_pagesegmode From 1a91cd46526fdab2debc2913f47a3601030a7687 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 04:36:48 -0700 Subject: [PATCH 131/880] pikepdf 1.6 --- requirements/main.txt | 2 +- setup.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 97c692ba..121c441d 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ chardet == 3.0.4 cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 -pikepdf == 1.5.0.post0 +pikepdf == 1.6.0 Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 diff --git a/setup.py b/setup.py index 8881e22a..6b2b7d11 100644 --- a/setup.py +++ b/setup.py @@ -96,7 +96,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six == 20181108', - 'pikepdf >= 1.5.0, < 2', + 'pikepdf >= 1.6.0, < 2', 'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"', # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels From 5f00e4f9d8dc91f72baa53b8e32b2039c93646f4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 04:51:52 -0700 Subject: [PATCH 132/880] Sort imports --- misc/webservice.py | 17 ++++++------ src/ocrmypdf/__init__.py | 25 +++++++----------- src/ocrmypdf/__main__.py | 6 ++--- src/ocrmypdf/_jobcontext.py | 2 +- src/ocrmypdf/_pipeline.py | 5 ++-- src/ocrmypdf/_sync.py | 2 +- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/exec/__init__.py | 5 ++-- src/ocrmypdf/exec/ghostscript.py | 1 - src/ocrmypdf/exec/jbig2enc.py | 2 +- src/ocrmypdf/exec/pngquant.py | 2 +- src/ocrmypdf/exec/tesseract.py | 4 +-- src/ocrmypdf/exec/unpaper.py | 6 ++--- src/ocrmypdf/filters.py | 2 +- src/ocrmypdf/optimize.py | 6 ++--- src/ocrmypdf/pdfa.py | 3 +-- src/ocrmypdf/pdfinfo/__init__.py | 2 +- src/ocrmypdf/pdfinfo/info.py | 9 ++++--- .../pdf.bin | Bin 0 -> 3610 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 13 +++++++++ tests/cache/manifest.jsonl | 1 + 24 files changed, 64 insertions(+), 54 deletions(-) create mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin diff --git a/misc/webservice.py b/misc/webservice.py index 677561d5..ed2a0374 100644 --- a/misc/webservice.py +++ b/misc/webservice.py @@ -23,21 +23,22 @@ to emphasize that SaaS deployments should make sure they comply with Ghostscript's license as well as OCRmyPDF's. """ +import os +import shlex +from subprocess import PIPE, run +from tempfile import TemporaryDirectory + from flask import ( Flask, Response, - flash, - request, - redirect, - url_for, abort, + flash, + redirect, + request, send_from_directory, + url_for, ) -from subprocess import run, PIPE -from tempfile import TemporaryDirectory from werkzeug.utils import secure_filename -import os -import shlex app = Flask(__name__) app.secret_key = "secret" diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index ad4b3068..9bbd2aa3 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -24,24 +24,19 @@ __version__ = pkg_resources.get_distribution('ocrmypdf').version VERSION = __version__ +from . import helpers, hocrtransform, leptonica, pdfa, pdfinfo +from .api import Verbosity, configure_logging, ocr from .exceptions import ( - ExitCode, BadArgsError, - PdfMergeFailedError, - MissingDependencyError, - UnsupportedImageFormatError, DpiError, - OutputFileAccessError, - PriorOcrFoundError, - InputFileError, - SubprocessOutputError, EncryptedPdfError, + ExitCode, + InputFileError, + MissingDependencyError, + OutputFileAccessError, + PdfMergeFailedError, + PriorOcrFoundError, + SubprocessOutputError, TesseractConfigError, + UnsupportedImageFormatError, ) - -from . import helpers -from . import hocrtransform -from . import leptonica -from . import pdfa -from . import pdfinfo -from .api import ocr, configure_logging, Verbosity diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index c24796af..bcfd6528 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,12 +21,12 @@ import os import sys from . import __version__ -from .cli import parser -from .api import configure_logging, Verbosity from ._jobcontext import make_logger from ._sync import run_pipeline from ._validation import check_closed_streams, check_options -from .exceptions import ExitCode, BadArgsError, MissingDependencyError +from .api import Verbosity, configure_logging +from .cli import parser +from .exceptions import BadArgsError, ExitCode, MissingDependencyError def run(args=None): diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 96ad6c7e..e50aa75a 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -16,9 +16,9 @@ # along with OCRmyPDF. If not, see . import logging +import os import shutil import sys -import os class PicklableLoggerMixin: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 02042112..0a846bfd 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -22,18 +22,17 @@ from datetime import datetime, timezone from shutil import copyfileobj import img2pdf -from PIL import Image - import pikepdf from pikepdf.models.metadata import encode_pdf_date +from PIL import Image from . import PROGRAM_NAME, VERSION, leptonica from .exceptions import ( DpiError, EncryptedPdfError, InputFileError, - UnsupportedImageFormatError, PriorOcrFoundError, + UnsupportedImageFormatError, ) from .exec import ghostscript, tesseract from .helpers import re_symlink diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index b2c9461e..5cb7bd8a 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -28,6 +28,7 @@ from tempfile import mkdtemp from tqdm import tqdm from . import __version__ +from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files, make_logger from ._pipeline import ( convert_to_pdfa, @@ -59,7 +60,6 @@ from ._validation import ( create_input_file, report_output_file_size, ) -from ._graft import OcrGrafter from .exceptions import ExitCode, ExitCodeException from .exec import qpdf from .helpers import available_cpu_count diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 5c9bc2a1..c957d989 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -41,7 +41,7 @@ from .exec import ( tesseract, unpaper, ) -from .helpers import is_file_writable, re_symlink, is_iterable_notstr, monotonic +from .helpers import is_file_writable, is_iterable_notstr, monotonic, re_symlink # ------------- # External dependencies diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index bc4a4e5a..2c862d81 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -23,9 +23,9 @@ from pathlib import Path from tqdm import tqdm -from .cli import parser from ._sync import run_pipeline from ._validation import check_options +from .cli import parser class TqdmConsole: diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 28193a01..c1a18c9a 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -21,9 +21,10 @@ import logging import os import re import sys -from subprocess import run, STDOUT, PIPE, CalledProcessError -from ..exceptions import MissingDependencyError, ExitCode from collections.abc import Mapping +from subprocess import PIPE, STDOUT, CalledProcessError, run + +from ..exceptions import ExitCode, MissingDependencyError log = logging.Logger(__name__) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index b39ff628..48cceff2 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -28,7 +28,6 @@ from PIL import Image from ..exceptions import SubprocessOutputError from . import get_version - gslog = logging.getLogger() diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index 771e58a3..696c899c 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -18,8 +18,8 @@ from functools import lru_cache from subprocess import PIPE, run -from . import get_version from ..exceptions import MissingDependencyError +from . import get_version @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index 536dca1c..f22bb68e 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -21,8 +21,8 @@ from tempfile import NamedTemporaryFile from PIL import Image -from . import get_version from ..exceptions import MissingDependencyError +from . import get_version @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 863047a1..3b5a647a 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -24,13 +24,13 @@ from functools import lru_cache from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run -from . import get_version from ..exceptions import ( MissingDependencyError, - TesseractConfigError, SubprocessOutputError, + TesseractConfigError, ) from ..helpers import page_number +from . import get_version OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 9c9db5d5..d3ee1ea4 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -26,11 +26,11 @@ from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory -from . import get_version -from ..exceptions import MissingDependencyError, SubprocessOutputError - from PIL import Image +from ..exceptions import MissingDependencyError, SubprocessOutputError +from . import get_version + @lru_cache(maxsize=1) def version(): diff --git a/src/ocrmypdf/filters.py b/src/ocrmypdf/filters.py index e51381bf..7e904876 100644 --- a/src/ocrmypdf/filters.py +++ b/src/ocrmypdf/filters.py @@ -1,5 +1,5 @@ -from PIL import Image import PIL.ImageOps +from PIL import Image def invert(im): diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 405f48af..eb559eb9 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -22,15 +22,15 @@ from collections import defaultdict from os import fspath from pathlib import Path +import pikepdf +from pikepdf import Dictionary, Name from PIL import Image from tqdm import tqdm -import pikepdf -from pikepdf import Name, Dictionary from . import leptonica from ._jobcontext import PDFContext -from .exec import jbig2enc, pngquant from .exceptions import OutputFileAccessError +from .exec import jbig2enc, pngquant from .helpers import re_symlink DEFAULT_JPEG_QUALITY = 75 diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 684254e7..93dfd849 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -36,9 +36,8 @@ from binascii import hexlify from pathlib import Path from string import Template -import pkg_resources - import pikepdf +import pkg_resources ICC_PROFILE_RELPATH = 'data/sRGB.icc' diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index 83cf9a48..093fea5e 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -16,4 +16,4 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from .info import PdfInfo, Colorspace, Encoding +from .info import Colorspace, Encoding, PdfInfo diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 4d93d4f7..964c3036 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -16,23 +16,24 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging +import re from collections import namedtuple from decimal import Decimal from enum import Enum -import logging from math import hypot, isclose from os import fspath from pathlib import Path from warnings import warn -import re -from pikepdf import PdfMatrix import pikepdf +from pikepdf import PdfMatrix from tqdm import tqdm +from ocrmypdf.exceptions import EncryptedPdfError, MissingDependencyError + from . import ghosttext from .layout import get_page_analysis, get_text_boxes -from ocrmypdf.exceptions import EncryptedPdfError, MissingDependencyError logger = logging.getLogger() diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..ce1b614413bc4b38a780e5352273c55960f2d342 GIT binary patch literal 3610 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fw_^nfr*Kwsim%gxw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0Gx{& Av;Y7A literal 0 HcmV?d00001 diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..61f78d82 --- /dev/null +++ b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..25fdded2 --- /dev/null +++ b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,13 @@ +Portez ce vieux whisky au juge +blond qui fume sur son Ile +interieure, a cöte de l'alcöve +ovoide, oU les büches se +consument dans l'ätre, ce qui +lui permet de penser & la +caenogenese de |'etre dont il +est question dans la cause +ambigu& entendue a MoY, dans +un capharnaüm qui, pense-t-il, +diminue ca et la la qualite de son +ceuvre. + \ No newline at end of file diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index c9a1fc25..73520981 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -65,3 +65,4 @@ {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.5.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.4", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} From eb104b405d7d914b2d007b2a1c51241b22dabfc7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 05:02:19 -0700 Subject: [PATCH 133/880] Avoid circular imports for __version__ --- src/ocrmypdf/__init__.py | 10 +--------- src/ocrmypdf/_pipeline.py | 4 +++- src/ocrmypdf/_sync.py | 1 - src/ocrmypdf/_version.py | 24 ++++++++++++++++++++++++ src/ocrmypdf/cli.py | 7 ++++--- 5 files changed, 32 insertions(+), 14 deletions(-) create mode 100644 src/ocrmypdf/_version.py diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 9bbd2aa3..8d779a95 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -15,16 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import pkg_resources - -PROGRAM_NAME = 'ocrmypdf' - -# Official PEP 396 -__version__ = pkg_resources.get_distribution('ocrmypdf').version - -VERSION = __version__ - from . import helpers, hocrtransform, leptonica, pdfa, pdfinfo +from ._version import PROGRAM_NAME, __version__ from .api import Verbosity, configure_logging, ocr from .exceptions import ( BadArgsError, diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 0a846bfd..7af59c02 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -26,7 +26,9 @@ import pikepdf from pikepdf.models.metadata import encode_pdf_date from PIL import Image -from . import PROGRAM_NAME, VERSION, leptonica +from . import leptonica +from ._version import PROGRAM_NAME +from ._version import __version__ as VERSION from .exceptions import ( DpiError, EncryptedPdfError, diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 5cb7bd8a..ccedbdc6 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -27,7 +27,6 @@ from tempfile import mkdtemp from tqdm import tqdm -from . import __version__ from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files, make_logger from ._pipeline import ( diff --git a/src/ocrmypdf/_version.py b/src/ocrmypdf/_version.py new file mode 100644 index 00000000..430f76e7 --- /dev/null +++ b/src/ocrmypdf/_version.py @@ -0,0 +1,24 @@ +# © 2017 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + + +import pkg_resources + +PROGRAM_NAME = 'ocrmypdf' + +# Official PEP 396 +__version__ = pkg_resources.get_distribution('ocrmypdf').version diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 2d5b6788..d8a55481 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -17,7 +17,8 @@ import argparse -from . import PROGRAM_NAME, VERSION +from ._version import PROGRAM_NAME as _PROGRAM_NAME +from ._version import __version__ as _VERSION def numeric(basetype, min_=None, max_=None): @@ -54,7 +55,7 @@ class ArgumentParser(argparse.ArgumentParser): parser = ArgumentParser( - prog=PROGRAM_NAME, + prog=_PROGRAM_NAME, fromfile_prefix_chars='@', formatter_class=argparse.RawDescriptionHelpFormatter, description="""\ @@ -167,7 +168,7 @@ parser.add_argument( parser.add_argument( '--version', action='version', - version=VERSION, + version=_VERSION, help="Print program version and exit", ) From ce13431ecfd494a2221888bd5ec81a1028775c49 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 15:47:09 -0700 Subject: [PATCH 134/880] Remove experimental filters.py --- src/ocrmypdf/filters.py | 10 ---------- 1 file changed, 10 deletions(-) delete mode 100644 src/ocrmypdf/filters.py diff --git a/src/ocrmypdf/filters.py b/src/ocrmypdf/filters.py deleted file mode 100644 index 7e904876..00000000 --- a/src/ocrmypdf/filters.py +++ /dev/null @@ -1,10 +0,0 @@ -import PIL.ImageOps -from PIL import Image - - -def invert(im): - return PIL.ImageOps.invert(im.convert('L')) - - -def whiteout(im): - return Image.new(im.mode, im.size) From db4598f76a2f6fd2423bf57bde8f5c251ab2c53e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 16:15:48 -0700 Subject: [PATCH 135/880] Add PDF linearization --- docs/release_notes.rst | 23 ++++++++++++----------- src/ocrmypdf/_pipeline.py | 20 ++++++++++++++++++-- src/ocrmypdf/api.py | 1 + src/ocrmypdf/cli.py | 12 ++++++++++++ src/ocrmypdf/optimize.py | 26 +++++++++++++++++--------- tests/test_main.py | 32 ++++++++++++++++++++++++++++++++ 6 files changed, 92 insertions(+), 22 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index e09b2179..26223b64 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -23,33 +23,35 @@ v9.0.0 implements this feature (Leptonica). - The ``-v`` (verbosity level) parameter now accepts only ``0``, ``1``, and ``2``. -- Dropped support for Tesseract "4.00.00-alpha" releases. Tesseract 4.0 beta and +- Dropped support for Tesseract 4.00.00-alpha releases. Tesseract 4.0 beta and later remain supported. +- Dropped the ``ocrmypdf-polyglot`` and ``ocrmypdf-webservice`` images. -**Major changes** +**New features** - Added a high level API for applications that want to integrate OCRmyPDF. Special thanks to Martin Wind (@mawi1988) whose made significant contributions to this effort. OCRmyPDF is GPLv3-licensed. -- Major internal code reorganization. - Added progress bars for long-running steps. ■■■■■■■□□ +- We now create linearized ("fast web view") PDFs by default. The new parameter + ``--fast-web-view`` provides control over when this feature is applied. +- Added a new ``--pages`` feature to limit OCR to only a specific page range. + The list may contain commas or single pages, such as ``1, 3, 5-11``. - When the number of pages is small compared to the number of allowed jobs, we run Tesseract in multithreaded (OpenMP) mode when available. This should improve performance on files with low page counts. -- Pages with vector artwork are treated as full color. Previously, vectors - were ignored when considering the colorspace needed to cover a page, which - could cause loss of color under certain settings. -- Dropped the ``ocrmypdf-polyglot`` and ``ocrmypdf-webservice`` images. - Removed dependency on ``ruffus``, and with that, the non-reentrancy restrictions that previous made an API impossible. -- Added a new ``--pages`` feature to limit OCR to only a specific page range. - The list may contain commas or single pages, such as ``1, 3, 5-11``. - Output and logging messages overhauled so that ocrmypdf may be integrated into applications that use the logging module. - pikepdf 1.6.0 is required. +- Added a logo. 😊 -**Minor changes** +**Bug fixes** +- Pages with vector artwork are treated as full color. Previously, vectors + were ignored when considering the colorspace needed to cover a page, which + could cause loss of color under certain settings. - Test suite now spawns processes less frequently, allowing more accurate measurement of code coverage. - Improved test coverage. @@ -57,7 +59,6 @@ v9.0.0 - Updated Docker images to use newer versions. - Fixed images encoded as JBIG2 with a colorspace other than ``/DeviceGray`` were not interpreted correctly. -- We have a logo. 😊 v8.3.2 ====== diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 7af59c02..a485e447 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -716,6 +716,13 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): return output_file +def should_linearize(working_file, context): + filesize = os.stat(working_file).st_size + if filesize > (context.options.fast_web_view * 1_000_000): + return True + return False + + def metadata_fixup(working_file, context): output_file = context.get_path('metafix.pdf') options = context.options @@ -749,11 +756,14 @@ def metadata_fixup(working_file, context): context.log.info( "The following metadata fields were not copied: %r", not_copied ) - pdf.save( output_file, compress_streams=True, + preserve_pdfa=True, object_stream_mode=pikepdf.ObjectStreamMode.generate, + linearize=( # Don't linearize if optimize() will be linearizing too + should_linearize(working_file, context) if options.optimize == 0 else False + ), ) original.close() pdf.close() @@ -762,7 +772,13 @@ def metadata_fixup(working_file, context): def optimize_pdf(input_file, context): output_file = context.get_path('optimize.pdf') - optimize(input_file, output_file, context) + save_settings = dict( + compress_streams=True, + preserve_pdfa=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + linearize=should_linearize(input_file, context), + ) + optimize(input_file, output_file, context, save_settings) return output_file diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 2c862d81..00d5334c 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -196,6 +196,7 @@ def ocr( # pylint: disable=unused-argument pdfa_image_compression=None, user_words=None, user_patterns=None, + fast_web_view=None, keep_temporary_files=None, progress_bar=None, tesseract_env=None, diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index d8a55481..b11e9a56 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -466,6 +466,18 @@ advanced.add_argument( metavar='FILE', help="Specify the location of the Tesseract user patterns file.", ) +advanced.add_argument( + '--fast-web-view', + type=numeric(float, 0), + default=1.0, + metavar="MEGABYTES", + help="If the size of file is more than this threshold (in MB), then " + "linearize the PDF for fast web viewing. This allows the PDF to be " + "displayed before it is fully downloaded in web browsers, but increases " + "the space required slightly. By default we skip this for small files " + "which do not benefit. If the threshold is 0 it will be apply to all files. " + "Set the threshold very high to disable.", +) debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index eb559eb9..550f2473 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -449,7 +449,7 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) -def optimize(input_file, output_file, context): +def optimize(input_file, output_file, context, save_settings): log = context.log options = context.options if options.optimize == 0: @@ -479,11 +479,7 @@ def optimize(input_file, output_file, context): target_file = Path(output_file).with_suffix('.opt.pdf') pike.remove_unreferenced_resources() - pike.save( - target_file, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, - ) + pike.save(target_file, **save_settings) input_size = Path(input_file).stat().st_size output_size = Path(target_file).stat().st_size @@ -497,8 +493,11 @@ def optimize(input_file, output_file, context): log.info(f"Optimize ratio: {ratio:.2f} savings: {(100 * savings):.1f}%") if savings < 0: - log.info("Optimize did not improve the file - discarded") - re_symlink(input_file, output_file) + log.info("Image optimization did not improve the file - discarded") + # We still need to save the file + with pikepdf.open(input_file) as pike: + pike.remove_unreferenced_resources() + pike.save(output_file, **save_settings) else: re_symlink(target_file, output_file) @@ -535,7 +534,16 @@ def main(infile, outfile, level, jobs=1): with TemporaryDirectory() as td: context = PDFContext(options, td, infile, None) tmpout = Path(td) / 'out.pdf' - optimize(infile, tmpout, context) + optimize( + infile, + tmpout, + context, + dict( + compress_streams=True, + preserve_pdfa=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + ), + ) copy(fspath(tmpout), fspath(outfile)) diff --git a/tests/test_main.py b/tests/test_main.py index 0b508936..52ef343a 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -28,6 +28,7 @@ import pytest from PIL import Image import ocrmypdf +import pikepdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import ghostscript, qpdf, tesseract from ocrmypdf.leptonica import Pix @@ -1093,3 +1094,34 @@ def test_version_check(): with pytest.raises(MissingDependencyError): get_version('echo') + + +@pytest.mark.parametrize( + 'threshold, optimize, output_type, expected', + [ + [1.0, 0, 'pdfa', False], + [1.0, 0, 'pdf', False], + [0.0, 0, 'pdfa', True], + [0.0, 0, 'pdf', True], + [1.0, 1, 'pdfa', False], + [1.0, 1, 'pdf', False], + [0.0, 1, 'pdfa', True], + [0.0, 1, 'pdf', True], + ], +) +def test_fast_web_view( + spoof_tesseract_noop, resources, outpdf, threshold, optimize, output_type, expected +): + check_ocrmypdf( + resources / 'trivial.pdf', + outpdf, + '--fast-web-view', + threshold, + '--optimize', + optimize, + '--output-type', + output_type, + env=spoof_tesseract_noop, + ) + with pikepdf.open(outpdf) as pdf: + assert pdf.is_linearized == expected From df320086672cb6eac9c6916cc1afc08eb0f979e1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 27 Jul 2019 16:47:53 -0700 Subject: [PATCH 136/880] Ensure test_optimize passes Linearization sends it over the edge --- tests/test_optimize.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 6884ee8e..248f906e 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -35,7 +35,7 @@ def test_basic(resources, pdf, outpdf): infile = resources / pdf opt.main(infile, outpdf, level=3) - assert Path(outpdf).stat().st_size <= Path(infile).stat().st_size + assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size def test_mono_not_inverted(resources, outdir): From c4afc5c242d9eeb601cf1761c612e8c57c4a398a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jul 2019 00:39:14 -0700 Subject: [PATCH 137/880] Add missing item from v9.0.0 release notes --- docs/release_notes.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 26223b64..d07108d8 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -59,6 +59,8 @@ v9.0.0 - Updated Docker images to use newer versions. - Fixed images encoded as JBIG2 with a colorspace other than ``/DeviceGray`` were not interpreted correctly. +- Fixed a OCR text-image registration (i.e. alignment) problem when the page + when MediaBox had a nonzero corner. v8.3.2 ====== From a6805ed343a032b8cd135de234a59697996088de Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jul 2019 00:42:38 -0700 Subject: [PATCH 138/880] Travis: remove vestiges of pdfminer being optional on osx --- .travis.yml | 111 +++++++++++++++++++++------------------------------- 1 file changed, 44 insertions(+), 67 deletions(-) diff --git a/.travis.yml b/.travis.yml index 7ca3206e..95f48ea5 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,7 +1,7 @@ cache: pip: true directories: - - $HOME/Library/Caches/Homebrew + - $HOME/Library/Caches/Homebrew matrix: include: @@ -16,23 +16,23 @@ matrix: apt: update: true sources: - - sourceline: 'ppa:alex-p/tesseract-ocr' - - sourceline: 'ppa:heyarje/libav-11' - - sourceline: 'ppa:vshn/ghostscript' + - sourceline: "ppa:alex-p/tesseract-ocr" + - sourceline: "ppa:heyarje/libav-11" + - sourceline: "ppa:vshn/ghostscript" packages: - - ghostscript - - libavcodec56 - - libavformat56 - - libavutil54 - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - qpdf - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra + - ghostscript + - libavcodec56 + - libavformat56 + - libavutil54 + - libexempi3 + - libffi-dev + - pngquant + - poppler-utils + - qpdf + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra - os: linux dist: xenial sudo: required @@ -44,41 +44,22 @@ matrix: apt: update: true sources: - - sourceline: 'ppa:alex-p/tesseract-ocr' + - sourceline: "ppa:alex-p/tesseract-ocr" packages: - - ghostscript - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - qpdf - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - unpaper - - os: osx - osx_image: xcode9.2 - language: generic - addons: - homebrew: - update: true - packages: - - exempi - ghostscript - - jbig2enc - - leptonica - - openjpeg + - libexempi3 + - libffi-dev - pngquant - - python + - poppler-utils - qpdf - - tesseract + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra - unpaper - os: osx osx_image: xcode9.2 language: generic - env: - - ADD_PDFMINER=1 addons: homebrew: update: true @@ -95,7 +76,7 @@ matrix: - unpaper before_cache: -- rm -f $HOME/.cache/pip/log/debug.log + - rm -f $HOME/.cache/pip/log/debug.log before_install: | mkdir -p bin @@ -113,20 +94,16 @@ before_install: | fi install: -- export PATH=$PWD/bin:$PATH -- pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251 -- pip3 install -r requirements/main.txt -- pip3 install --no-deps . -- | - if [[ "$ADD_PDFMINER" == "1" ]]; then - pip3 install --no-deps .[pdfminer] - fi -- pip3 install -r requirements/test.txt + - export PATH=$PWD/bin:$PATH + - pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251 + - pip3 install -r requirements/main.txt + - pip3 install --no-deps . + - pip3 install -r requirements/test.txt script: -- tesseract --version -- qpdf --version -- pytest -n auto + - tesseract --version + - qpdf --version + - pytest -n auto deploy: # release for main pypi @@ -134,13 +111,13 @@ deploy: # a race and all versions will try to deploy # OTOH if we ever need separate binary wheels then each version needs its # own deploy -- provider: pypi - user: ocrmypdf-travis - password: - secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo=" - distributions: "sdist bdist_wheel" - on: - branch: master - tags: true - condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" - skip_upload_docs: true + - provider: pypi + user: ocrmypdf-travis + password: + secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo=" + distributions: "sdist bdist_wheel" + on: + branch: master + tags: true + condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" + skip_upload_docs: true From 77bbc22c5056ca448558486f9608dd7e7e479fd0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Aug 2019 01:07:45 -0700 Subject: [PATCH 139/880] Ensure --image-dpi on non-image produces a warning --- src/ocrmypdf/_pipeline.py | 2 +- tests/test_main.py | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index a485e447..cc7142ba 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -129,7 +129,7 @@ def triage(input_file, output_file, options, log): if _pdf_guess_version(input_file): if options.image_dpi: log.warning( - "Argument --image-dpi ignored because the " + "Argument --image-dpi is being ignored because the " "input file is a PDF, not an image." ) # Origin file is a pdf create a symlink with pdf extension diff --git a/tests/test_main.py b/tests/test_main.py index 52ef343a..d41750de 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -1125,3 +1125,14 @@ def test_fast_web_view( ) with pikepdf.open(outpdf) as pdf: assert pdf.is_linearized == expected + + +def test_image_dpi_not_image(caplog, spoof_tesseract_noop, resources, outpdf): + check_ocrmypdf( + resources / 'trivial.pdf', + outpdf, + '--image-dpi', + '100', + env=spoof_tesseract_noop, + ) + assert '--image-dpi is being ignored' in caplog.text From f276c4ef1eeea871833588b1c79b184bbf76b6f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Aug 2019 01:09:18 -0700 Subject: [PATCH 140/880] Alpine Docker: jbig2enc moved from testing to community --- .docker/alpine.dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index d8472c11..8f0b383e 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -56,7 +56,7 @@ RUN \ # Add runtime dependencies && apk add --update \ python3 \ - jbig2enc@testing \ + jbig2enc@community \ ghostscript \ qpdf@community \ qpdf-dev@community \ From 7bfcd0a9d5b9a80ba85456901b1fa8844266c399 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Aug 2019 01:12:13 -0700 Subject: [PATCH 141/880] Use pikepdf 1.6.1 --- requirements/main.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/main.txt b/requirements/main.txt index 121c441d..24087d27 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ chardet == 3.0.4 cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 -pikepdf == 1.6.0 +pikepdf == 1.6.1 Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 From a1a7b973e9d92caac78f044117b7a9602cd39bb4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Aug 2019 01:23:49 -0700 Subject: [PATCH 142/880] tests: split out stdin/stdout tests --- tests/test_main.py | 110 +---------------------------------- tests/test_stdio.py | 137 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 139 insertions(+), 108 deletions(-) create mode 100644 tests/test_stdio.py diff --git a/tests/test_main.py b/tests/test_main.py index d41750de..35234368 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -1,4 +1,4 @@ -# © 2015-17 James R. Barlow: github.com/jbarlow83 +# © 2015-19 James R. Barlow: github.com/jbarlow83 # # This file is part of OCRmyPDF. # @@ -18,10 +18,9 @@ import logging import os import shutil -import sys from math import isclose from pathlib import Path -from subprocess import DEVNULL, PIPE, run, Popen +from subprocess import PIPE, run import PIL import pytest @@ -84,11 +83,6 @@ def spoof_no_tess_gs_raster_fail(tmp_path_factory): ) -@pytest.fixture(scope='session') -def spoof_tess_bad_utf8(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') - - def test_quick(spoof_tesseract_cache, resources, outpdf): check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) @@ -550,68 +544,6 @@ def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): assert out_pageinfo[0].images[0].enc == Encoding.jbig2 -def test_stdin(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): - input_file = str(resources / 'francais.pdf') - output_file = str(outpdf) - - # Runs: ocrmypdf - output.pdf < testfile.pdf - with open(input_file, 'rb') as input_stream: - p_args = ocrmypdf_exec + ['-', output_file] - p = run( - p_args, - stdout=PIPE, - stderr=PIPE, - stdin=input_stream, - env=spoof_tesseract_noop, - ) - assert p.returncode == ExitCode.ok - - -def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): - input_file = str(resources / 'francais.pdf') - output_file = str(outpdf) - - # Runs: ocrmypdf francais.pdf - > test_stdout.pdf - with open(output_file, 'wb') as output_stream: - p_args = ocrmypdf_exec + [input_file, '-'] - p = run( - p_args, - stdout=output_stream, - stderr=PIPE, - stdin=DEVNULL, - env=spoof_tesseract_noop, - ) - assert p.returncode == ExitCode.ok - - assert qpdf.check(output_file, log=None) - - -@pytest.mark.skipif( - sys.version_info[0:3] >= (3, 6, 4), reason="issue fixed in Python 3.6.4" -) -def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): - input_file = str(resources / 'francais.pdf') - output_file = str(outpdf) - - def evil_closer(): - os.close(0) - os.close(1) - - p_args = ocrmypdf_exec + [input_file, output_file] - p = Popen( # pylint: disable=subprocess-popen-preexec-fn - p_args, - close_fds=True, - stdout=None, - stderr=PIPE, - stdin=None, - env=spoof_tesseract_noop, - preexec_fn=evil_closer, - ) - out, err = p.communicate() - print(err.decode()) - assert p.returncode == ExitCode.ok - - def test_masks(spoof_tesseract_noop, resources, outpdf): assert ( ocrmypdf.ocr( @@ -993,36 +925,6 @@ def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf): assert pdfa_info['conformance'] == f'PDF/A-{pdfa_level}B' -@pytest.mark.skipif(sys.version_info >= (3, 7, 0), reason='better utf-8') -@pytest.mark.skipif( - Path('/etc/alpine-release').exists(), reason="invalid test on alpine" -) -def test_bad_locale(): - env = os.environ.copy() - env['LC_ALL'] = 'C' - - p, out, err = run_ocrmypdf('a', 'b', env=env) - assert out == '', "stdout not clean" - assert p.returncode != 0 - assert 'configured to use ASCII as encoding' in err, "should whine" - - -@pytest.mark.parametrize('renderer', RENDERERS) -def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): - p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', - no_outpdf, - '--pdf-renderer', - renderer, - env=spoof_tess_bad_utf8, - ) - - assert out == '', "stdout not clean" - assert p.returncode != 0 - assert 'not utf-8' in err, "should whine about utf-8" - assert '\\x96' in err, 'should repeat backslash encoded output' - - @pytest.mark.skipif( PIL.__version__ < '5.0.0', reason="Pillow < 5.0.0 doesn't raise the exception" ) @@ -1051,14 +953,6 @@ def test_text_curves(spoof_tesseract_noop, resources, outpdf): assert len(info.pages[0].images) != 0, "force did not rasterize" -def test_dev_null(spoof_tesseract_noop, resources): - p, out, err = run_ocrmypdf( - resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop - ) - assert p.returncode == 0, "could not send output to /dev/null" - assert len(out) == 0, "wrote to stdout" - - def test_output_is_dir(spoof_tesseract_noop, resources, outdir): p, out, err = run_ocrmypdf( resources / 'trivial.pdf', outdir, '--force-ocr', env=spoof_tesseract_noop diff --git a/tests/test_stdio.py b/tests/test_stdio.py new file mode 100644 index 00000000..76c99d56 --- /dev/null +++ b/tests/test_stdio.py @@ -0,0 +1,137 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import os +import sys +from pathlib import Path +from subprocess import DEVNULL, PIPE, run, Popen + +import pytest + +from ocrmypdf.exceptions import ExitCode +from ocrmypdf.exec import qpdf + +# pytest.helpers is dynamic +# pylint: disable=no-member,redefined-outer-name + +run_ocrmypdf = pytest.helpers.run_ocrmypdf +spoof = pytest.helpers.spoof + + +@pytest.fixture(scope='session') +def spoof_tess_bad_utf8(tmp_path_factory): + return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') + + +def test_stdin(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): + input_file = str(resources / 'francais.pdf') + output_file = str(outpdf) + + # Runs: ocrmypdf - output.pdf < testfile.pdf + with open(input_file, 'rb') as input_stream: + p_args = ocrmypdf_exec + ['-', output_file] + p = run( + p_args, + stdout=PIPE, + stderr=PIPE, + stdin=input_stream, + env=spoof_tesseract_noop, + ) + assert p.returncode == ExitCode.ok + + +def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): + input_file = str(resources / 'francais.pdf') + output_file = str(outpdf) + + # Runs: ocrmypdf francais.pdf - > test_stdout.pdf + with open(output_file, 'wb') as output_stream: + p_args = ocrmypdf_exec + [input_file, '-'] + p = run( + p_args, + stdout=output_stream, + stderr=PIPE, + stdin=DEVNULL, + env=spoof_tesseract_noop, + ) + assert p.returncode == ExitCode.ok + + assert qpdf.check(output_file, log=None) + + +@pytest.mark.skipif( + sys.version_info[0:3] >= (3, 6, 4), reason="issue fixed in Python 3.6.4" +) +def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): + input_file = str(resources / 'francais.pdf') + output_file = str(outpdf) + + def evil_closer(): + os.close(0) + os.close(1) + + p_args = ocrmypdf_exec + [input_file, output_file] + p = Popen( # pylint: disable=subprocess-popen-preexec-fn + p_args, + close_fds=True, + stdout=None, + stderr=PIPE, + stdin=None, + env=spoof_tesseract_noop, + preexec_fn=evil_closer, + ) + out, err = p.communicate() + print(err.decode()) + assert p.returncode == ExitCode.ok + + +@pytest.mark.skipif(sys.version_info >= (3, 7, 0), reason='better utf-8') +@pytest.mark.skipif( + Path('/etc/alpine-release').exists(), reason="invalid test on alpine" +) +def test_bad_locale(): + env = os.environ.copy() + env['LC_ALL'] = 'C' + + p, out, err = run_ocrmypdf('a', 'b', env=env) + assert out == '', "stdout not clean" + assert p.returncode != 0 + assert 'configured to use ASCII as encoding' in err, "should whine" + + +@pytest.mark.parametrize('renderer', ['hocr', 'sandwich']) +def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): + p, out, err = run_ocrmypdf( + resources / 'ccitt.pdf', + no_outpdf, + '--pdf-renderer', + renderer, + env=spoof_tess_bad_utf8, + ) + + assert out == '', "stdout not clean" + assert p.returncode != 0 + assert 'not utf-8' in err, "should whine about utf-8" + assert '\\x96' in err, 'should repeat backslash encoded output' + + +def test_dev_null(spoof_tesseract_noop, resources): + p, out, err = run_ocrmypdf( + resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop + ) + assert p.returncode == 0, "could not send output to /dev/null" + assert len(out) == 0, "wrote to stdout" From 8ad034a678097ad1b54985e7d75720dd2c09ab8b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 15:49:28 -0700 Subject: [PATCH 143/880] docs: update install on FreeBSD to point to ports --- docs/installation.rst | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 2cf1d11b..85136488 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -87,7 +87,6 @@ Fedora 29 or newer .. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg :alt: Fedore Rawhide - +------------------------------+ | **OCRmyPDF version** | +------------------------------+ @@ -403,14 +402,19 @@ The command line program should now be available: Installing on FreeBSD ===================== -FreeBSD 11.2 is known to work. Other versions likely work but have not -been tested. +.. image:: https://repology.org/badge/version-for-repo/freebsd/python:ocrmypdf.svg + :alt: FreeBSD + :target: https://repology.org/project/python:ocrmypdf/versions -In general it should work to: +FreeBSD 11.2, 11.3, 12.0-RELEASE and 13.0-CURRENT are supported. Other +versions likely work but have not been tested. -#. `Install and build - pikepdf `__. -#. Install the equivalent list of dependencies for Linux. +.. code-block:: bash + + pkg install py36-ocrmypdf + +To install a more recent version, you could attempt to first install the system +version with ``pkg``, then use ``pip install --user ocrmypdf``. Installing the Docker image =========================== From b241f6691988fe603c950c196fed7534acb8e3d7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 15:34:40 -0700 Subject: [PATCH 144/880] travis: Add a minimal Ubuntu config --- .travis.yml | 62 +++++++++++++++++++++++++++++++++++++---------------- 1 file changed, 43 insertions(+), 19 deletions(-) diff --git a/.travis.yml b/.travis.yml index 95f48ea5..8695ea6e 100644 --- a/.travis.yml +++ b/.travis.yml @@ -12,7 +12,8 @@ matrix: python: "3.6" env: - DIST=trusty - addons: &trusty_apt + - MINIMAL=true + addons: apt: update: true sources: @@ -24,15 +25,49 @@ matrix: - libavcodec56 - libavformat56 - libavutil54 - - libexempi3 - libffi-dev - - pngquant - - poppler-utils - qpdf - tesseract-ocr - tesseract-ocr-deu - tesseract-ocr-eng - tesseract-ocr-fra + before_install: | + pip3 install --upgrade pip + pip3 install --upgrade wheel + - os: linux + dist: trusty + sudo: required + language: python + python: "3.6" + env: + - DIST=trusty + addons: + apt: + update: true + sources: + - sourceline: "ppa:alex-p/tesseract-ocr" + - sourceline: "ppa:heyarje/libav-11" + - sourceline: "ppa:vshn/ghostscript" + packages: + - ghostscript + - libavcodec56 + - libavformat56 + - libavutil54 + - libffi-dev + - qpdf + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra + - libexempi3 # --- optional extras from here --- + - pngquant + - poppler-utils + before_install: | + mkdir -p bin packages + pip3 install --upgrade pip + pip3 install --upgrade wheel + wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb + sudo dpkg -i packages/unpaper_6.1-1.deb - os: linux dist: xenial sudo: required @@ -74,26 +109,15 @@ matrix: - qpdf - tesseract - unpaper + before_install: | + pip3 install --upgrade pip + pip3 install wheel before_cache: - rm -f $HOME/.cache/pip/log/debug.log -before_install: | - mkdir -p bin - if [[ "$TRAVIS_OS_NAME" == "linux" ]]; then - pip3 install --upgrade pip - pip3 install --upgrade wheel - if [[ "$DIST" == "trusty" ]]; then - mkdir -p packages - wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb - sudo dpkg -i packages/unpaper_6.1-1.deb - fi - elif [[ "$TRAVIS_OS_NAME" == "osx" ]]; then - pip3 install --upgrade pip - pip3 install wheel - fi - install: + - mkdir -p bin - export PATH=$PWD/bin:$PATH - pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251 - pip3 install -r requirements/main.txt From 793348a47cb7fd85cd2aa0a229c43835800f6845 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 16:15:49 -0700 Subject: [PATCH 145/880] tests: mark test as requiring pngquant --- tests/test_optimize.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 248f906e..bde0a279 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -55,6 +55,7 @@ def test_mono_not_inverted(resources, outdir): assert im.getpixel((0, 0)) == 255, "Expected white background" +@pytest.mark.skipif(not pngquant.available(), reason='need pngquant') def test_jpg_png_params(resources, outpdf, spoof_tesseract_noop): check_ocrmypdf( resources / 'crom.png', From 7755c5c5a768aaabdb8a19deba64be9d13ebd3e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 16:16:10 -0700 Subject: [PATCH 146/880] tests: fix interpretation of None as omitted argument --- tests/conftest.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index e693139a..b443ac1f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -176,10 +176,9 @@ def no_outpdf(tmp_path): def check_ocrmypdf(input_file, output_file, *args, env=None): """Run ocrmypdf and confirmed that a valid file was created""" - # p, out, err = run_ocrmypdf(input_file, output_file, *args, env=env) - options = cli.parser.parse_args( - [str(input_file), str(output_file)] + [str(arg) for arg in args] + [str(input_file), str(output_file)] + + [str(arg) for arg in args if arg is not None] ) api.check_options(options) if env: From 2eeaca11686d34182ddeaedc007e6ff0e5138810 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 17:13:55 -0700 Subject: [PATCH 147/880] travis: make minimal config even more minimal --- .travis.yml | 4 ---- 1 file changed, 4 deletions(-) diff --git a/.travis.yml b/.travis.yml index 8695ea6e..85a58776 100644 --- a/.travis.yml +++ b/.travis.yml @@ -18,13 +18,9 @@ matrix: update: true sources: - sourceline: "ppa:alex-p/tesseract-ocr" - - sourceline: "ppa:heyarje/libav-11" - sourceline: "ppa:vshn/ghostscript" packages: - ghostscript - - libavcodec56 - - libavformat56 - - libavutil54 - libffi-dev - qpdf - tesseract-ocr From e9bc093842ed7329c0319f4408e8d540349b6e3c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 17:14:11 -0700 Subject: [PATCH 148/880] v9.0.1 release notes --- docs/release_notes.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index d07108d8..0005c48e 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,15 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.0.1 +====== + +- Fixed test suite failing when either of optional dependencies unpaper and + pngquant were missing. +- Fixed Alpine Docker image build. +- Documented that FreeBSD ports are now available. +- Changed to pikepdf 1.6.1 (also for Alpine Docker). + v9.0.0 ====== From 707ebeb1513646e670aa24d8346890f556517837 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 11 Aug 2019 18:48:56 -0700 Subject: [PATCH 149/880] docs: installation updates --- docs/installation.rst | 52 ++++++++++++++++++++++++------------------- 1 file changed, 29 insertions(+), 23 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 85136488..4dee61dc 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -21,7 +21,7 @@ installing the Python binary wheels. Installing on Linux =================== -Debian and Ubuntu 16.10 or newer +Debian and Ubuntu 18.04 or newer -------------------------------- .. |deb-stable| image:: https://repology.org/badge/version-for-repo/debian_stable/ocrmypdf.svg @@ -33,27 +33,29 @@ Debian and Ubuntu 16.10 or newer .. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg :alt: Debian unstable -.. |ubu-1710| image:: https://repology.org/badge/version-for-repo/ubuntu_17_10/ocrmypdf.svg - :alt: Ubuntu 17.10 - .. |ubu-1804| image:: https://repology.org/badge/version-for-repo/ubuntu_18_04/ocrmypdf.svg :alt: Ubuntu 18.04 LTS .. |ubu-1810| image:: https://repology.org/badge/version-for-repo/ubuntu_18_10/ocrmypdf.svg :alt: Ubuntu 18.10 +.. |ubu-1904| image:: https://repology.org/badge/version-for-repo/ubuntu_19_04/ocrmypdf.svg + :alt: Ubuntu 19.04 -+-------------------------------------------+ -| **OCRmyPDF versions in Debian & Ubuntu** | -+-------------------------------------------+ -| |latest| | -+-------------------------------------------+ -| |deb-stable| |deb-testing| |deb-unstable| | -+-------------------------------------------+ -| |ubu-1710| |ubu-1804| |ubu-1810| | -+-------------------------------------------+ +.. |ubu-1910| image:: https://repology.org/badge/version-for-repo/ubuntu_19_10/ocrmypdf.svg + :alt: Ubuntu 19.10 -Users of Debian 9 ("stretch") or later or Ubuntu 16.10 or later may ++-----------------------------------------------+ +| **OCRmyPDF versions in Debian & Ubuntu** | ++-----------------------------------------------+ +| |latest| | ++-----------------------------------------------+ +| |deb-stable| |deb-testing| |deb-unstable| | ++-----------------------------------------------+ +| |ubu-1804| |ubu-1810| |ubu-1904| |ubu-1910| | ++-----------------------------------------------+ + +Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later may simply .. code-block:: bash @@ -64,7 +66,8 @@ As indicated in the table above, Debian and Ubuntu releases may lag behind the latest version. If the version available for your platform is out of date, you could opt to install the latest version from source. See `Installing HEAD revision from -sources <#installing-head-revision-from-sources>`__. +sources <#installing-head-revision-from-sources>`__. Ubuntu 16.10 to 17.10 +inclusive also had ocrmypdf, but these versions are end of life. For full details on version availability for your platform, check the `Debian Package Tracker `__ or @@ -81,19 +84,22 @@ For full details on version availability for your platform, check the Fedora 29 or newer ------------------ -.. |fedora-29| image:: https://repology.org/badge/version-for-repo/fedora29/ocrmypdf.svg +.. |fedora-29| image:: https://repology.org/badge/version-for-repo/fedora_29/ocrmypdf.svg :alt: Fedora 29 +.. |fedora-30| image:: https://repology.org/badge/version-for-repo/fedora_30/ocrmypdf.svg + :alt: Fedora 30 + .. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg :alt: Fedore Rawhide -+------------------------------+ -| **OCRmyPDF version** | -+------------------------------+ -| |latest| | -+------------------------------+ -| |fedora-29| |fedora-rawhide| | -+------------------------------+ ++-----------------------------------------------+ +| **OCRmyPDF version** | ++-----------------------------------------------+ +| |latest| | ++-----------------------------------------------+ +| |fedora-29| |fedora-30| |fedora-rawhide| | ++-----------------------------------------------+ Users of Fedora 29 later may simply From 6460a7eb3e2911e29a541fcb10ccc5196277d22f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 26 Aug 2019 12:07:34 -0700 Subject: [PATCH 150/880] docs: leptonica.com -> .org --- docs/cookbook.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 617a6046..75e8a52d 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -165,8 +165,8 @@ might remove desirable content, especially from poor quality scans. - ``--deskew`` will correct pages were scanned at a skewed angle by rotating them back into place. Skew determination and correction is performed using `Postl's variance of line - sums `__ algorithm as - implemented in `Leptonica `__. + sums `__ algorithm as + implemented in `Leptonica `__. - ``--clean`` uses `unpaper `__ to clean up pages before OCR, but does not alter the final output. This makes it From 09457edad35e05e60d3c750b3bd473ae23218bcc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 26 Aug 2019 12:49:47 -0700 Subject: [PATCH 151/880] alpine: use jbig2enc@community --- .docker/alpine.dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile index 8f0b383e..b2879e27 100644 --- a/.docker/alpine.dockerfile +++ b/.docker/alpine.dockerfile @@ -14,7 +14,7 @@ RUN \ && apk add --update \ python3-dev \ py3-setuptools \ - jbig2enc@testing \ + jbig2enc@community \ ghostscript \ qpdf@community \ qpdf-dev@community \ From fdefcd8af277817d03b39dc948f32ec598a1d322 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 26 Aug 2019 13:30:07 -0700 Subject: [PATCH 152/880] travis: Make 3.7 the build leader/deployer --- .travis.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.travis.yml b/.travis.yml index 85a58776..9098888a 100644 --- a/.travis.yml +++ b/.travis.yml @@ -127,7 +127,7 @@ script: deploy: # release for main pypi - # 3.6 is considered the build leader and does the deploy, otherwise there is + # 3.7 is considered the build leader and does the deploy, otherwise there is # a race and all versions will try to deploy # OTOH if we ever need separate binary wheels then each version needs its # own deploy @@ -139,5 +139,5 @@ deploy: on: branch: master tags: true - condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" + condition: $TRAVIS_PYTHON_VERSION == "3.7" && $TRAVIS_OS_NAME == "linux" skip_upload_docs: true From 638eb556effb9e753a81b8e3b20bcfbef7a3b816 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Aug 2019 14:52:59 -0700 Subject: [PATCH 153/880] Reactivate user-words test that was always skipped --- tests/test_main.py | 26 ++++---------------------- 1 file changed, 4 insertions(+), 22 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index 35234368..caf4a0e8 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -652,26 +652,11 @@ THIS FILE IS INVALID assert p.returncode == ExitCode.invalid_config -@pytest.mark.skipif(tesseract.v4(), reason='arg has no effect in 4.0-beta1') -def test_user_words(resources, outdir): +@pytest.mark.skipif(not tesseract.has_user_words(), reason='not functional until 4.1.0') +def test_user_words_ocr(resources, outdir): + # Does not actually test if --user-words causes output to differ word_list = outdir / 'wordlist.txt' - sidecar_before = outdir / 'sidecar_before.txt' - sidecar_after = outdir / 'sidecar_after.txt' - - # Don't know how to make this test pass on various versions and platforms - # so weaken to merely testing that the argument is accepted - consistent = False - - if consistent: - check_ocrmypdf( - resources / 'crom.png', - outdir / 'out.pdf', - '--image-dpi', - 150, - '--sidecar', - sidecar_before, - ) - assert 'cromulent' not in sidecar_before.open().read() + sidecar_after = outdir / 'sidecar.txt' with word_list.open('w') as f: f.write('cromulent\n') # a perfectly cromulent word @@ -687,9 +672,6 @@ def test_user_words(resources, outdir): word_list, ) - if consistent: - assert 'cromulent' in sidecar_after.open().read() - def test_form_xobject(spoof_tesseract_noop, resources, outpdf): check_ocrmypdf( From 11ef78a8912cfacb72063bccbde69826d4859d85 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Aug 2019 14:54:03 -0700 Subject: [PATCH 154/880] Fix running without eng.traineddata installed raises exception --- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/exec/tesseract.py | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index c957d989..c5dcb6a4 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -107,7 +107,7 @@ def check_options_output(options): options.pdf_renderer = 'sandwich' if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( - options.tesseract_env + options.tesseract_env, languages ): raise MissingDependencyError( "You are using an alpha version of Tesseract 4.0 that does not support " diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 3b5a647a..baefba89 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -61,13 +61,13 @@ def v4(tesseract_env=None): return version(tesseract_env) >= '4' -def has_textonly_pdf(tesseract_env=None): +def has_textonly_pdf(tesseract_env=None, langs=None): """Does Tesseract have textonly_pdf capability? Available in v4.00.00alpha since January 2017. Best to - parse the parameter list + parse the parameter list. """ - args_tess = ['tesseract', '--print-parameters', 'pdf'] + args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf'] params = '' try: proc = run( @@ -358,7 +358,7 @@ def generate_pdf( if pagesegmode is not None: args_tesseract.extend(['--psm', str(pagesegmode)]) - if text_only and has_textonly_pdf(tesseract_env): + if text_only and has_textonly_pdf(tesseract_env, language): args_tesseract.extend(['-c', 'textonly_pdf=1']) if user_words: From 462bfb84fb525aeea0f551d34bf6501ee0cb8999 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 31 Aug 2019 01:24:31 -0700 Subject: [PATCH 155/880] install: affirm that we now require Tesseract beta --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4dee61dc..27e27671 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -513,7 +513,7 @@ manager. ``pip`` cannot provide them. - Python 3.6 or newer - Ghostscript 9.15 or newer - qpdf 8.1.0 or newer -- Tesseract 4.0.0-alpha or newer +- Tesseract 4.0.0-beta or newer As of ocrmypdf 7.2.1, the following versions are recommended: From b0d9775343a7c04a9271fc1470d415bbb14642f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 31 Aug 2019 01:25:36 -0700 Subject: [PATCH 156/880] Attempt to resolve black-inversion issue --- src/ocrmypdf/optimize.py | 30 ++++++++++++++++++------------ tests/test_optimize.py | 2 +- 2 files changed, 19 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 550f2473..0b1c5089 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -401,12 +401,14 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): # 1 - no predictor # 10-14 - there is a predictor # Leptonica's compdata->predictor only tells TRUE or FALSE - # From there the PNG decoder can infer the rest from the file. - # In practice the predictor should be Paeth, 14, so we'll use that. + # 10-14 means the actual predictor is specified in the data, so for any + # number >= 10 the PDF reader will use whatever the PNG data specifies. + # In practice Leptonica should use Paeth, 14, but 15 seems to be the + # designated value for "optimal". So we will use 15. # See: # - PDF RM 7.4.4.4 Table 10 # - https://github.com/DanBloomberg/leptonica/blob/master/src/pdfio2.c#L757 - predictor = 14 if compdata.predictor > 0 else 1 + predictor = 15 if compdata.predictor > 0 else 1 dparms = Dictionary(Predictor=predictor) if predictor > 1: dparms.BitsPerComponent = compdata.bps # Yes, this is redundant @@ -417,9 +419,14 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): im_obj.Width = compdata.w im_obj.Height = compdata.h + log.debug( + f"PNG {xref}: palette={compdata.ncolors} spp={compdata.spp} bps={compdata.bps}" + ) if compdata.ncolors > 0: # .ncolors is the number of colors in the palette, not the number of - # colors used in a true color image + # colors used in a true color image. The palette string is always + # given as RGB tuples even when the image is grayscale; see + # https://github.com/DanBloomberg/leptonica/blob/master/src/colormap.c#L2067 palette_pdf_string = compdata.get_palette_pdf_string() palette_data = pikepdf.Object.parse(palette_pdf_string) palette_stream = pikepdf.Stream(pike, bytes(palette_data)) @@ -431,20 +438,19 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): ] cs = palette else: + # ncolors == 0 means we are using a colorspace without a palette if compdata.spp == 1: - # PDF interprets binary-1 as black in 1bpp, but PNG sets - # black to 0 for 1bpp. Create a palette that informs the PDF - # of the mapping - seems cleaner to go this way but pikepdf - # needs to be patched to support it. - # palette = [Name.Indexed, Name.DeviceGray, 1, b"\xff\x00"] - # cs = palette + if compdata.bps == 1: + # PDF interprets binary-1 as black in 1bpp, but PNG sets + # black to 0 for 1bpp. Use Decode to ensure color is + # correct. + log.debug("Inverting photometry") + im_obj.Decode = [1, 0] cs = Name.DeviceGray elif compdata.spp == 3: cs = Name.DeviceRGB elif compdata.spp == 4: cs = Name.DeviceCMYK - if compdata.bps == 1: - im_obj.Decode = [1, 0] # Bit of a kludge but this inverts photometric too im_obj.ColorSpace = cs im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index bde0a279..03f5d03a 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -48,7 +48,7 @@ def test_mono_not_inverted(resources, outdir): xres=10, yres=10, raster_device='pnggray', - log=logging.getLogger(name='test_mono_flip'), + log=logging.getLogger(name='test_mono_not_inverted'), ) im = Image.open(fspath(outdir / 'im.png')) From c8d6ea6b10d431224493559bfa8966eb9903fe65 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 31 Aug 2019 14:55:40 -0700 Subject: [PATCH 157/880] Fix tests broken by --print-parameters change --- tests/spoof/tesseract_badutf8.py | 2 +- tests/spoof/tesseract_big_image_error.py | 2 +- tests/spoof/tesseract_crash.py | 2 +- tests/spoof/tesseract_noop.py | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/spoof/tesseract_badutf8.py b/tests/spoof/tesseract_badutf8.py index ba21bb17..472a47a9 100755 --- a/tests/spoof/tesseract_badutf8.py +++ b/tests/spoof/tesseract_badutf8.py @@ -52,7 +52,7 @@ def main(): elif sys.argv[1] == '--list-langs': print('List of available languages (1):\neng', file=sys.stderr) sys.exit(0) - elif sys.argv[1] == '--print-parameters': + elif sys.argv[-2] == '--print-parameters': print("Some parameters", file=sys.stderr) print("textonly_pdf\t1\tSome help text") sys.exit(0) diff --git a/tests/spoof/tesseract_big_image_error.py b/tests/spoof/tesseract_big_image_error.py index d3ee53d8..8b710bee 100755 --- a/tests/spoof/tesseract_big_image_error.py +++ b/tests/spoof/tesseract_big_image_error.py @@ -44,7 +44,7 @@ def main(): elif sys.argv[1] == '--list-langs': print('List of available languages (1):\neng\n', file=sys.stderr) sys.exit(0) - elif sys.argv[1] == '--print-parameters': + elif sys.argv[-2] == '--print-parameters': print('A parameter list would go here\ntextonly_pdf 0\n', file=sys.stderr) sys.exit(0) elif sys.argv[-2] == 'hocr': diff --git a/tests/spoof/tesseract_crash.py b/tests/spoof/tesseract_crash.py index 8b6f90d8..03c7dbde 100755 --- a/tests/spoof/tesseract_crash.py +++ b/tests/spoof/tesseract_crash.py @@ -50,7 +50,7 @@ def main(): elif sys.argv[1] == '--list-langs': print('List of available languages (1):\neng', file=sys.stderr) sys.exit(0) - elif sys.argv[1] == '--print-parameters': + elif sys.argv[-2] == '--print-parameters': print('A parameter list would go here\ntextonly_pdf 0\n', file=sys.stderr) sys.exit(0) elif sys.argv[-2] == 'hocr': diff --git a/tests/spoof/tesseract_noop.py b/tests/spoof/tesseract_noop.py index cdb71664..857c3e2a 100755 --- a/tests/spoof/tesseract_noop.py +++ b/tests/spoof/tesseract_noop.py @@ -76,7 +76,7 @@ def main(): elif sys.argv[1] == '--list-langs': print('List of available languages (1):\neng', file=sys.stderr) sys.exit(0) - elif sys.argv[1] == '--print-parameters': + elif sys.argv[-2] == '--print-parameters': print("Some parameters", file=sys.stderr) print("textonly_pdf\t1\tSome help text") sys.exit(0) From feff1e38bb09334a2a78fbd5e16f8fe503edd4bf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 17:19:12 -0700 Subject: [PATCH 158/880] Use context managers to ensure Pillow images are closed --- src/ocrmypdf/_pipeline.py | 5 ++-- src/ocrmypdf/exec/pngquant.py | 3 +-- src/ocrmypdf/exec/tesseract.py | 4 ++-- src/ocrmypdf/exec/unpaper.py | 43 +++++++++++++++++----------------- tests/test_lept.py | 3 ++- tests/test_main.py | 10 ++++---- tests/test_optimize.py | 12 +++++----- tests/test_rotation.py | 10 ++++---- 8 files changed, 45 insertions(+), 45 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index cc7142ba..56c17e35 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -54,9 +54,9 @@ def triage_image_file(input_file, output_file, options, log): # Recover the original filename log.error(str(e).replace(input_file, options.input_file)) raise UnsupportedImageFormatError() from e - else: - log.info("Input file is an image") + with im: + log.info("Input file is an image") if 'dpi' in im.info: if im.info['dpi'] <= (96, 96) and not options.image_dpi: log.info("Image size: (%d, %d)" % im.size) @@ -89,7 +89,6 @@ def triage_image_file(input_file, output_file, options, log): elif im.mode == 'CMYK': log.info('Input CMYK image has no ICC profile, not usable') raise UnsupportedImageFormatError() - im.close() try: log.info("Image seems valid. Try converting to PDF...") diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index f22bb68e..5dbe5b64 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -40,8 +40,7 @@ def available(): def quantize(input_file, output_file, quality_min, quality_max): if input_file.endswith('.jpg'): - im = Image.open(input_file) - with NamedTemporaryFile(suffix='.png') as tmp: + with Image.open(input_file) as im, NamedTemporaryFile(suffix='.png') as tmp: im.save(tmp) args = [ 'pngquant', diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index baefba89..03bf037d 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -233,8 +233,8 @@ def _generate_null_hocr(output_hocr, output_sidecar, image): the same size as the input image.""" from PIL import Image - im = Image.open(image) - w, h = im.size + with Image.open(image) as im: + w, h = im.size with open(output_hocr, 'w', encoding="utf-8") as f: f.write(HOCR_TEMPLATE.format(w, h)) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index d3ee1ea4..e186dea8 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -42,33 +42,30 @@ def run(input_file, output_file, dpi, log, mode_args): SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'} - im = Image.open(input_file) - if im.mode not in SUFFIXES.keys(): - log.info("Converting image to other colorspace") + with TemporaryDirectory() as tmpdir, Image.open(input_file) as im: + if im.mode not in SUFFIXES.keys(): + log.info("Converting image to other colorspace") + try: + if im.mode == 'P' and len(im.getcolors()) == 2: + im = im.convert(mode='1') + else: + im = im.convert(mode='RGB') + except IOError as e: + im.close() + raise MissingDependencyError( + "Could not convert image with type " + im.mode + ) from e + try: - if im.mode == 'P' and len(im.getcolors()) == 2: - im = im.convert(mode='1') - else: - im = im.convert(mode='RGB') - except IOError as e: - im.close() + suffix = SUFFIXES[im.mode] + except KeyError: raise MissingDependencyError( - "Could not convert image with type " + im.mode + "Failed to convert image to a supported format." ) from e - try: - suffix = SUFFIXES[im.mode] - except KeyError: - im.close() - raise MissingDependencyError( - "Failed to convert image to a supported format." - ) from e - - with TemporaryDirectory() as tmpdir: input_pnm = os.path.join(tmpdir, f'input{suffix}') output_pnm = os.path.join(tmpdir, f'output{suffix}') im.save(input_pnm, format='PPM') - im.close() # To prevent any shenanigans from accepting arbitrary parameters in # --unpaper-args, we: @@ -95,10 +92,12 @@ def run(input_file, output_file, dpi, log, mode_args): log.debug(proc.stdout) # unpaper sets dpi to 72; fix this try: - Image.open(output_pnm).save(output_file, dpi=(dpi, dpi)) + with Image.open(output_pnm) as imout: + imout.save(output_file, dpi=(dpi, dpi)) except (FileNotFoundError, OSError): raise SubprocessOutputError( - "unpaper: failed to produce the expected output file. Called with: " + "unpaper: failed to produce the expected output file. " + + " Called with: " + str(args_unpaper) ) from None diff --git a/tests/test_lept.py b/tests/test_lept.py index 804ca215..2504c600 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -38,7 +38,8 @@ def test_colormap_backgroundnorm(resources): def crom_pix(resources): pix = lept.Pix.open(resources / 'crom.png') im = Image.open(resources / 'crom.png') - return pix, im + yield pix, im + im.close() def test_pix_basic(crom_pix): diff --git a/tests/test_main.py b/tests/test_main.py index caf4a0e8..11e02993 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -118,8 +118,8 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): def test_remove_background(spoof_tesseract_noop, resources, outdir): # Ensure the input image does not contain pure white/black - im = Image.open(resources / 'congress.jpg') - assert im.getextrema() != ((0, 255), (0, 255), (0, 255)) + with Image.open(resources / 'congress.jpg') as im: + assert im.getextrema() != ((0, 255), (0, 255), (0, 255)) output_pdf = check_ocrmypdf( resources / 'congress.jpg', @@ -145,8 +145,8 @@ def test_remove_background(spoof_tesseract_noop, resources, outdir): ) # The output image should contain pure white and black - im = Image.open(output_png) - assert im.getextrema() == ((0, 255), (0, 255), (0, 255)) + with Image.open(output_png) as im: + assert im.getextrema() == ((0, 255), (0, 255), (0, 255)) # This will run 5 * 2 * 2 = 20 test cases @@ -792,6 +792,7 @@ def test_compression_preserved( assert pdfimage.color == Colorspace.rgb, "Colorspace changed" elif im.mode.startswith('L'): assert pdfimage.color == Colorspace.gray, "Colorspace changed" + im.close() @pytest.mark.parametrize( @@ -853,6 +854,7 @@ def test_compression_changed( assert pdfimage.color == Colorspace.rgb, "Colorspace changed" elif im.mode.startswith('L'): assert pdfimage.color == Colorspace.gray, "Colorspace changed" + im.close() def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 03f5d03a..85bcc1a5 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -51,8 +51,8 @@ def test_mono_not_inverted(resources, outdir): log=logging.getLogger(name='test_mono_not_inverted'), ) - im = Image.open(fspath(outdir / 'im.png')) - assert im.getpixel((0, 0)) == 255, "Expected white background" + with Image.open(fspath(outdir / 'im.png')) as im: + assert im.getpixel((0, 0)) == 255, "Expected white background" @pytest.mark.skipif(not pngquant.available(), reason='need pngquant') @@ -110,10 +110,10 @@ def test_flate_to_jbig2(resources, outdir, spoof_tesseract_noop): # This test requires an image that pngquant is capable of converting to # to 1bpp - so use an existing 1bpp image, convert up, confirm it can # convert down - im = Image.open(fspath(resources / 'typewriter.png')) - assert im.mode in ('1', 'P') - im = im.convert('L') - im.save(fspath(outdir / 'type8.png')) + with Image.open(fspath(resources / 'typewriter.png')) as im: + assert im.mode in ('1', 'P') + im = im.convert('L') + im.save(fspath(outdir / 'type8.png')) check_ocrmypdf( outdir / 'type8.png', diff --git a/tests/test_rotation.py b/tests/test_rotation.py index b8bc07ca..7aaead96 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -224,12 +224,12 @@ def test_rotate_deskew_timeout(resources, outdir): @pytest.mark.parametrize('image_angle', (0, 90, 180, 270)) def test_rotate_page_level(image_angle, page_angle, resources, outdir): def make_rotate_test(prefix, image_angle, page_angle): - im = Image.open(fspath(resources / 'typewriter.png')) - if image_angle != 0: - ccw_angle = -image_angle % 360 - im = im.transpose(getattr(Image, f'ROTATE_{ccw_angle}')) memimg = BytesIO() - im.save(memimg, format='PNG') + with Image.open(fspath(resources / 'typewriter.png')) as im: + if image_angle != 0: + ccw_angle = -image_angle % 360 + im = im.transpose(getattr(Image, f'ROTATE_{ccw_angle}')) + im.save(memimg, format='PNG') memimg.seek(0) mempdf = BytesIO() img2pdf.convert( From 19ba3ae011ab0323f4833a005e4c512d94d6c5b1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 17:38:54 -0700 Subject: [PATCH 159/880] Allow test_german to xfail if deu language is not installed --- tests/spoof/tesseract_cache.py | 2 ++ tests/test_main.py | 7 +++---- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/tests/spoof/tesseract_cache.py b/tests/spoof/tesseract_cache.py index db06aa8e..528e5a1e 100755 --- a/tests/spoof/tesseract_cache.py +++ b/tests/spoof/tesseract_cache.py @@ -100,6 +100,8 @@ def main(): # Convert non-standard but supported -psm to --psm sys.argv = ['--psm' if arg == '-psm' else arg for arg in sys.argv] + if '_OCRMYPDF_TEST_INFILE' not in os.environ: + real_tesseract() # test not properly set up source = os.environ['_OCRMYPDF_TEST_INFILE'] # required args = parser.parse_args() diff --git a/tests/test_main.py b/tests/test_main.py index 11e02993..e87e5163 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -349,10 +349,9 @@ def test_german(spoof_tesseract_cache, resources, outdir): sidecar, env=spoof_tesseract_cache, ) - print(os.environ) - assert ( - p.returncode == ExitCode.ok - ), "This test may fail if Tesseract language packs are missing" + if 'deu' not in tesseract.languages(): + pytest.xfail(reason="tesseract-deu language pack not installed") + assert p.returncode == ExitCode.ok, "Requires tesseract deu language pack" def test_klingon(resources, outpdf): From b2cfaedf917c4a93c88dff500791c0bdfe11cba6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 23:26:13 -0700 Subject: [PATCH 160/880] optimize: Don't reinsert 1bpp images There seems to be version to version inconsistencies between Leptonica's photometric interpretation of 1bpp images, in particular commit a0692307 introduces a change to force transcoding in this situation. However, I never entirely got to the bottom of where the problem is, and in any event 1bpp images are probably better optimized by JBIG2 than pngquant, so we're going to stop running them through pngquant. --- src/ocrmypdf/optimize.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 0b1c5089..18accf90 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -392,6 +392,13 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): ) continue + if compdata.bps == 1: + # Discard 1bpp images due to issues with preserving the photometric + # interpretation. In particular Leptonica changed behavior in + # version 1.77.0, such that it transcodes 1bpp. + log.debug(f"discarded optimized image {xref} because it was 1bpp") + continue + # When a PNG is inserted into a PDF, we more or less copy the IDAT section from # the PDF and transfer the rest of the PNG headers to PDF image metadata. # One thing we have to do is tell the PDF reader whether a predictor was used @@ -440,12 +447,6 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): else: # ncolors == 0 means we are using a colorspace without a palette if compdata.spp == 1: - if compdata.bps == 1: - # PDF interprets binary-1 as black in 1bpp, but PNG sets - # black to 0 for 1bpp. Use Decode to ensure color is - # correct. - log.debug("Inverting photometry") - im_obj.Decode = [1, 0] cs = Name.DeviceGray elif compdata.spp == 3: cs = Name.DeviceRGB From 671c88d3b5975947727598a7a18d431ff140716c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 23:37:23 -0700 Subject: [PATCH 161/880] optimize: exclude images with custom Decode tables --- src/ocrmypdf/optimize.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 18accf90..9ed26d9f 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -73,6 +73,9 @@ def extract_image_filter(pike, root, log, image, xref): if filtdp[0] == Name.JPXDecode: return None # Don't do JPEG2000 + if Name.Decode in image: + return None # Don't mess with custom Decode tables + return pim, filtdp From c6caff90a128db8028240a2d7e24919cbabb26d7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 23:46:25 -0700 Subject: [PATCH 162/880] optimize: only re-insert pngs after pngquant Previously we attempted to reinsert all PNGs, but it appears to be unlikely that Leptonica's API is actually capable of optimizing the PNG before it inserts it. In any event qpdf has gained image optimization capabilities as well which we coudld borrow. --- src/ocrmypdf/optimize.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 9ed26d9f..5358bd17 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -346,6 +346,7 @@ def transcode_jpegs(pike, jpegs, root, log, options): def transcode_pngs(pike, images, image_name_fn, root, log, options): + modified = set() if options.optimize >= 2: png_quality = ( max(10, options.png_quality - 10), @@ -366,6 +367,7 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): png_quality[1], ) ) + modified.add(xref) with tqdm( desc="PNGs", total=len(futures), @@ -375,7 +377,7 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): for _future in concurrent.futures.as_completed(futures): pbar.update() - for xref in images: + for xref in modified: im_obj = pike.get_object(xref, 0) try: compdata = leptonica.CompressedData.open(png_name(root, xref)) From a650caa599600208f6655697f09eb40ac1c508ce Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 23:47:20 -0700 Subject: [PATCH 163/880] optimize: don't consider 1bpp images for PNG optimization --- src/ocrmypdf/optimize.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 5358bd17..e4b4e208 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -107,6 +107,10 @@ def extract_image_generic(*, pike, root, log, image, xref, options): return None pim, filtdp = result + # Don't try to PNG-optimize 1bpp images, since JBIG2 does it better. + if pim.bits_per_component == 1: + return None + if filtdp[0] == Name.DCTDecode and options.optimize >= 2: # This is a simple heuristic derived from some training data, that has # about a 70% chance of guessing whether the JPEG is high quality, From 0d80fab339231abd3fd095a3d9cebeae4cef8acc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Sep 2019 23:47:55 -0700 Subject: [PATCH 164/880] Remove restriction on pytest < 5 --- requirements/test.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/test.txt b/requirements/test.txt index ad5ec593..f6ce8d84 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,4 +1,4 @@ -pytest >= 4.4.1, < 5 +pytest >= 4.4.1 pytest-helpers-namespace >= 2019.1.8 pytest-xdist == 1.28.0 pytest-cov >= 2.6.1 From c728836956529fa2a6f82b4296a12ad7bd80daae Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Sep 2019 00:50:48 -0700 Subject: [PATCH 165/880] Adjust test requirements --- requirements/test.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements/test.txt b/requirements/test.txt index f6ce8d84..975325bd 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,6 +1,6 @@ -pytest >= 4.4.1 +pytest >= 5.0.0 pytest-helpers-namespace >= 2019.1.8 -pytest-xdist == 1.28.0 +pytest-xdist >= 1.29.0 # For DumpError fix pytest-cov >= 2.6.1 python-xmp-toolkit # requires apt-get install libexempi3 # or brew install exempi From 1c3e90a89232d435a5a7ad1f7e69a3c7e8d4e319 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Sep 2019 00:51:47 -0700 Subject: [PATCH 166/880] optimize: solve monochrome by converting to G4 --- src/ocrmypdf/optimize.py | 138 ++++++++++++++++++++++----------------- 1 file changed, 78 insertions(+), 60 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index e4b4e208..2976554c 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -384,7 +384,11 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): for xref in modified: im_obj = pike.get_object(xref, 0) try: - compdata = leptonica.CompressedData.open(png_name(root, xref)) + pix = leptonica.Pix.open(png_name(root, xref)) + if pix.mode == '1': + compdata = pix.generate_pdf_ci_data(leptonica.lept.L_G4_ENCODE, 0) + else: + compdata = leptonica.CompressedData.open(png_name(root, xref)) except leptonica.LeptonicaError as e: # Most likely this means file not found, i.e. quantize did not # produce an improved version @@ -400,69 +404,83 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): f"{len(compdata)} > {int(im_obj.stream_dict.Length)}" ) continue + if compdata.type == leptonica.lept.L_FLATE_ENCODE: + return rewrite_png(pike, im_obj, compdata, log) + elif compdata.type == leptonica.lept.L_G4_ENCODE: + return rewrite_png_as_g4(pike, im_obj, compdata, log) - if compdata.bps == 1: - # Discard 1bpp images due to issues with preserving the photometric - # interpretation. In particular Leptonica changed behavior in - # version 1.77.0, such that it transcodes 1bpp. - log.debug(f"discarded optimized image {xref} because it was 1bpp") - continue - # When a PNG is inserted into a PDF, we more or less copy the IDAT section from - # the PDF and transfer the rest of the PNG headers to PDF image metadata. - # One thing we have to do is tell the PDF reader whether a predictor was used - # on the image before Flate encoding. (Typically one is.) - # According to Leptonica source, PDF readers don't actually need us - # to specify the correct predictor, they just need a value of either: - # 1 - no predictor - # 10-14 - there is a predictor - # Leptonica's compdata->predictor only tells TRUE or FALSE - # 10-14 means the actual predictor is specified in the data, so for any - # number >= 10 the PDF reader will use whatever the PNG data specifies. - # In practice Leptonica should use Paeth, 14, but 15 seems to be the - # designated value for "optimal". So we will use 15. - # See: - # - PDF RM 7.4.4.4 Table 10 - # - https://github.com/DanBloomberg/leptonica/blob/master/src/pdfio2.c#L757 - predictor = 15 if compdata.predictor > 0 else 1 - dparms = Dictionary(Predictor=predictor) - if predictor > 1: - dparms.BitsPerComponent = compdata.bps # Yes, this is redundant - dparms.Colors = compdata.spp - dparms.Columns = compdata.w +def rewrite_png_as_g4(pike, im_obj, compdata, log): + im_obj.BitsPerComponent = 1 + im_obj.Width = compdata.w + im_obj.Height = compdata.h - im_obj.BitsPerComponent = compdata.bps - im_obj.Width = compdata.w - im_obj.Height = compdata.h + im_obj.write(compdata.read()) - log.debug( - f"PNG {xref}: palette={compdata.ncolors} spp={compdata.spp} bps={compdata.bps}" - ) - if compdata.ncolors > 0: - # .ncolors is the number of colors in the palette, not the number of - # colors used in a true color image. The palette string is always - # given as RGB tuples even when the image is grayscale; see - # https://github.com/DanBloomberg/leptonica/blob/master/src/colormap.c#L2067 - palette_pdf_string = compdata.get_palette_pdf_string() - palette_data = pikepdf.Object.parse(palette_pdf_string) - palette_stream = pikepdf.Stream(pike, bytes(palette_data)) - palette = [ - Name.Indexed, - Name.DeviceRGB, - compdata.ncolors - 1, - palette_stream, - ] - cs = palette - else: - # ncolors == 0 means we are using a colorspace without a palette - if compdata.spp == 1: - cs = Name.DeviceGray - elif compdata.spp == 3: - cs = Name.DeviceRGB - elif compdata.spp == 4: - cs = Name.DeviceCMYK - im_obj.ColorSpace = cs - im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) + log.debug(f"PNG to G4 {im_obj.objgen}") + if Name.Predictor in im_obj: + del im_obj.Predictor + if Name.DecodeParms in im_obj: + del im_obj.DecodeParms + im_obj.DecodeParms = Dictionary( + K=-1, BlackIs1=bool(compdata.minisblack), Columns=compdata.w + ) + + im_obj.Filter = Name.CCITTFaxDecode + return + + +def rewrite_png(pike, im_obj, compdata, log): + # When a PNG is inserted into a PDF, we more or less copy the IDAT section from + # the PDF and transfer the rest of the PNG headers to PDF image metadata. + # One thing we have to do is tell the PDF reader whether a predictor was used + # on the image before Flate encoding. (Typically one is.) + # According to Leptonica source, PDF readers don't actually need us + # to specify the correct predictor, they just need a value of either: + # 1 - no predictor + # 10-14 - there is a predictor + # Leptonica's compdata->predictor only tells TRUE or FALSE + # 10-14 means the actual predictor is specified in the data, so for any + # number >= 10 the PDF reader will use whatever the PNG data specifies. + # In practice Leptonica should use Paeth, 14, but 15 seems to be the + # designated value for "optimal". So we will use 15. + # See: + # - PDF RM 7.4.4.4 Table 10 + # - https://github.com/DanBloomberg/leptonica/blob/master/src/pdfio2.c#L757 + predictor = 15 if compdata.predictor > 0 else 1 + dparms = Dictionary(Predictor=predictor) + if predictor > 1: + dparms.BitsPerComponent = compdata.bps # Yes, this is redundant + dparms.Colors = compdata.spp + dparms.Columns = compdata.w + + im_obj.BitsPerComponent = compdata.bps + im_obj.Width = compdata.w + im_obj.Height = compdata.h + + log.debug( + f"PNG {im_obj.objgen}: palette={compdata.ncolors} spp={compdata.spp} bps={compdata.bps}" + ) + if compdata.ncolors > 0: + # .ncolors is the number of colors in the palette, not the number of + # colors used in a true color image. The palette string is always + # given as RGB tuples even when the image is grayscale; see + # https://github.com/DanBloomberg/leptonica/blob/master/src/colormap.c#L2067 + palette_pdf_string = compdata.get_palette_pdf_string() + palette_data = pikepdf.Object.parse(palette_pdf_string) + palette_stream = pikepdf.Stream(pike, bytes(palette_data)) + palette = [Name.Indexed, Name.DeviceRGB, compdata.ncolors - 1, palette_stream] + cs = palette + else: + # ncolors == 0 means we are using a colorspace without a palette + if compdata.spp == 1: + cs = Name.DeviceGray + elif compdata.spp == 3: + cs = Name.DeviceRGB + elif compdata.spp == 4: + cs = Name.DeviceCMYK + im_obj.ColorSpace = cs + im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) def optimize(input_file, output_file, context, save_settings): From 944d59e5ad75a0ef337fbb8fdbdfe2e109839630 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Sep 2019 01:17:52 -0700 Subject: [PATCH 167/880] Fix --print-parameters issue when chi_sim is not installed --- tests/test_validation.py | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index 138c8820..3914986d 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -37,15 +37,21 @@ def test_hocr_notlatin_warning(caplog): def test_old_ghostscript(caplog): - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.19'): + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.19'), patch( + 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + ): vd.check_options_output(make_opts(language='chi_sim', output_type='pdfa')) assert 'Ghostscript does not work correctly' in caplog.text - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.18'): + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.18'), patch( + 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + ): with pytest.raises(MissingDependencyError): vd.check_options_output(make_opts(output_type='pdfa-3')) - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.24'): + with patch('ocrmypdf.exec.ghostscript.version', return_value='9.24'), patch( + 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + ): with pytest.raises(MissingDependencyError): vd.check_dependency_versions(make_opts()) From a2a197ce4ca0681834d90f2e9eb5c68b1cb5df0f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Sep 2019 02:34:21 -0700 Subject: [PATCH 168/880] v9.0.2 release notes --- docs/release_notes.rst | 21 +++++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 0005c48e..b65fa9a2 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,14 +13,31 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.0.2 +====== + +- The image optimizer now skips optimizing flate (PNG) encoded images in some + situations where the optimization effort was likely wasted. +- The image optimizer now ignores images that specify arbitrary decode arrays, + since these are rare. +- Fixed an issue that caused inversion of black and white in monochrome images. + We are not certain but the problem seems to be linked to Leptonica 1.76.0 and + older. +- Fixed some cases where the test suite failed or produced unexpected if + English or German Tesseract language packs were not installed. +- Fixed a runtime error if the Tesseract English language is not installed. +- Improved explicit closing of Pillow images after use. +- Actually fixed of Alpine Docker image build. +- Changed to pikepdf 1.6.3. + v9.0.1 ====== - Fixed test suite failing when either of optional dependencies unpaper and pngquant were missing. -- Fixed Alpine Docker image build. +- Attempted fix of Alpine Docker image build. - Documented that FreeBSD ports are now available. -- Changed to pikepdf 1.6.1 (also for Alpine Docker). +- Changed to pikepdf 1.6.1. v9.0.0 ====== From 17ac9d7a9a296ae3d50146fbefad5281e2851b0f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 5 Sep 2019 13:17:26 -0700 Subject: [PATCH 169/880] Embed ICC profile in .ps (fixing Ghostscript 9.28 compatibility) Previously we included the filename, which required Postscript to run with file access enabled. For security, Ghostscript 9.28 enables ``-dSAFER`` and as such, no longer permits access to any file by default. This fix is necessary for compatibility with Ghostscript 9.28. We use ASCII85 for a slightly more compact representation. --- docs/release_notes.rst | 10 ++++++++++ src/ocrmypdf/exec/ghostscript.py | 1 + src/ocrmypdf/pdfa.py | 21 +++++++-------------- 3 files changed, 18 insertions(+), 14 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index b65fa9a2..37611b71 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,16 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.0.3 +====== + +- Embed an encoded version of the sRGB ICC profile in the intermediate + Postscript file (used for PDF/A conversion). Previously we included the + filename, which required Postscript to run with file access enabled. For + security, Ghostscript 9.28 enables ``-dSAFER`` and as such, no longer + permits access to any file by default. This fix is necessary for + compatibility with Ghostscript 9.28. + v9.0.2 ====== diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 48cceff2..14a1332a 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -259,6 +259,7 @@ def generate_pdfa( "-dQUIET", "-dBATCH", "-dNOPAUSE", + "-dSAFER", "-dCompatibilityLevel=" + str(pdf_version), "-sDEVICE=pdfwrite", "-dAutoRotatePages=/None", diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 93dfd849..408fd155 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -31,6 +31,7 @@ Ghostscript's handling of pdfmark. """ +import base64 import os from binascii import hexlify from pathlib import Path @@ -48,12 +49,10 @@ SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPA # files, from the Ghostscript documentation. Lines beginning with % are # comments. Python substitution variables have a '$' prefix. pdfa_def_template = u"""%! -% Define entries in the document Info dictionary : +% Define an ICC profile : /ICCProfile $icc_profile def -% Define an ICC profile : - [/_objdef {icc_PDFA} /type /stream /OBJ pdfmark [{icc_PDFA} << @@ -67,7 +66,7 @@ def (ERROR, unable to determine ProcessColorModel) == flush } ifelse >> /PUT pdfmark -[{icc_PDFA} ICCProfile (r) file /PUT pdfmark +[{icc_PDFA} ICCProfile /PUT pdfmark % Define the output intent dictionary : @@ -104,16 +103,10 @@ def generate_pdfa_ps(target_filename, icc='sRGB'): else: raise NotImplementedError("Only supporting sRGB") - # pdfmark must contain the full path to the ICC profile, and pdfmark must be - # also encoded in ASCII. ocrmypdf can be installed anywhere, including to - # paths that have a non-ASCII character in the filename. Ghostscript - # accepts hex-encoded strings and converts them to byte strings, so - # we encode the path with fsencode() and use the hex representation. - # UTF-16 not accepted here. (Even though ASCII encodable is the usual case, - # do this always to avoid making it a rare conditional.) - bytes_icc_profile = os.fsencode(icc_profile) - hex_icc_profile = hexlify(bytes_icc_profile) - icc_profile = '<' + hex_icc_profile.decode('ascii') + '>' + # Read the ICC profile, encode as ASCII85 and convert to a string which we + # will insert in the .ps file + bytes_icc_profile = Path(icc_profile).read_bytes() + icc_profile = base64.a85encode(bytes_icc_profile, adobe=True).decode('ascii') t = Template(pdfa_def_template) ps = t.substitute(icc_profile=icc_profile, icc_identifier=icc) From d7b7ca05744825ef23a9220687939e30508eca06 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 5 Sep 2019 13:39:43 -0700 Subject: [PATCH 170/880] v9.0.3 notes; Remove test_tesseract_config_notfound from suite --- docs/release_notes.rst | 4 +++- tests/test_main.py | 1 + 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 37611b71..f90f096d 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -22,6 +22,8 @@ v9.0.3 security, Ghostscript 9.28 enables ``-dSAFER`` and as such, no longer permits access to any file by default. This fix is necessary for compatibility with Ghostscript 9.28. +- Exclude a test that sometimes times out and fails in continuous integration + from the standard test suite. v9.0.2 ====== @@ -33,7 +35,7 @@ v9.0.2 - Fixed an issue that caused inversion of black and white in monochrome images. We are not certain but the problem seems to be linked to Leptonica 1.76.0 and older. -- Fixed some cases where the test suite failed or produced unexpected if +- Fixed some cases where the test suite failed if English or German Tesseract language packs were not installed. - Fixed a runtime error if the Tesseract English language is not installed. - Improved explicit closing of Pillow images after use. diff --git a/tests/test_main.py b/tests/test_main.py index e87e5163..e84fc052 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -613,6 +613,7 @@ language_model_penalty_non_freq_dict_word 0 ) +@pytest.mark.slow # This test sometimes times out in CI @pytest.mark.parametrize('renderer', RENDERERS) def test_tesseract_config_notfound(renderer, resources, outdir): cfg_file = outdir / 'nofile.cfg' From 078bc2abe9c7c724b79d390f439b5e298287fdaf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Sep 2019 12:55:38 -0700 Subject: [PATCH 171/880] pdfa: assume 3 RGB channels always --- src/ocrmypdf/pdfa.py | 13 +------------ 1 file changed, 1 insertion(+), 12 deletions(-) diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 408fd155..5f2a192e 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -54,18 +54,7 @@ pdfa_def_template = u"""%! def [/_objdef {icc_PDFA} /type /stream /OBJ pdfmark -[{icc_PDFA} -<< - /N currentpagedevice /ProcessColorModel known { - currentpagedevice /ProcessColorModel get dup /DeviceGray eq - {pop 1} { - /DeviceRGB eq - {3}{4} ifelse - } ifelse - } { - (ERROR, unable to determine ProcessColorModel) == flush - } ifelse ->> /PUT pdfmark +[{icc_PDFA} << /N 3 >> /PUT pdfmark [{icc_PDFA} ICCProfile /PUT pdfmark % Define the output intent dictionary : From cf4b04c5d163a3b661f45ae8783ca1bbf9baf3ab Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Sep 2019 12:56:27 -0700 Subject: [PATCH 172/880] optimize: work around pikepdf 1.6.3 limitation with indexed ICCbased colorspaces --- src/ocrmypdf/optimize.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 2976554c..bfe027af 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -111,6 +111,11 @@ def extract_image_generic(*, pike, root, log, image, xref, options): if pim.bits_per_component == 1: return None + try: + pim.indexed # pikepdf 1.6.3 can't handle [/Indexed [/Array...]] + except NotImplementedError: + return None + if filtdp[0] == Name.DCTDecode and options.optimize >= 2: # This is a simple heuristic derived from some training data, that has # about a 70% chance of guessing whether the JPEG is high quality, From ff860e8362f315e80b92e8a47c0a0c39d0dcda2b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 15 Sep 2019 01:46:13 -0700 Subject: [PATCH 173/880] Fix black settings in pyproject.toml --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 8a3375ef..4d61be1a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,7 @@ build-backend = "setuptools.build_meta" [tool.black] line-length = 88 -py36 = true +target-version = ["py36", "py37", "py38"] skip-string-normalization = true include = '\.pyi?$' exclude = ''' From 6e8b0c31947e83b254641e14763ca35cd45bb89a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 15 Sep 2019 01:47:10 -0700 Subject: [PATCH 174/880] Fix py36 test including 37 --- src/ocrmypdf/api.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 00d5334c..d279d6f4 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -33,7 +33,7 @@ class TqdmConsole: def __init__(self, file): self.file = file - self.py36 = sys.version_info >= (3, 6) + self.py36 = sys.version_info[0:2] == (3, 6) def write(self, msg): # When no progress bar is active, tqdm.write() routes to print() From a8565bac6ec2ce1efad5da0732af7b2da42c0f2a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 15 Sep 2019 01:47:31 -0700 Subject: [PATCH 175/880] Fix any False in the ocrmypdf.ocr() API being set to True --- src/ocrmypdf/api.py | 14 ++++++++++++-- tests/test_validation.py | 21 ++++++++++++++++++++- 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index d279d6f4..b735f398 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -122,13 +122,23 @@ def create_options(*, input_file, output_file, **kwargs): for arg, val in kwargs.items(): if val is None: continue - if arg == 'tesseract_env': + + # These arguments with special handling for which we bypass + # argparse + if arg in {'tesseract_env', 'progress_bar'}: deferred.append((arg, val)) continue + cmd_style_arg = arg.replace('_', '-') - cmdline.append(f"--{cmd_style_arg}") + + # Booleans are special: add only if True, omit for False if isinstance(val, bool): + if val: + cmdline.append(f"--{cmd_style_arg}") continue + + # We have a parameter + cmdline.append(f"--{cmd_style_arg}") if isinstance(val, (int, float)): cmdline.append(str(val)) elif isinstance(val, str): diff --git a/tests/test_validation.py b/tests/test_validation.py index 3914986d..e5e3c8f9 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -16,13 +16,14 @@ # along with OCRmyPDF. If not, see . import os -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock, patch, call import pytest import ocrmypdf._validation as vd from ocrmypdf.api import create_options from ocrmypdf.exceptions import MissingDependencyError, BadArgsError +from ocrmypdf.pdfinfo import PdfInfo def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): @@ -119,3 +120,21 @@ def test_report_file_size(tmp_path, caplog): os.truncate(out, 50000) vd.report_output_file_size(opts, in_, out) assert 'No reason' in caplog.text + + +def test_false_action_store_true(): + opts = make_opts(keep_temporary_files=True) + assert opts.keep_temporary_files == True + opts = make_opts(keep_temporary_files=False) + assert opts.keep_temporary_files == False + + +@pytest.mark.parametrize('progress_bar', [True, False]) +def test_no_progress_bar(progress_bar, resources): + opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) + with patch('ocrmypdf.pdfinfo.info.tqdm', autospec=True) as tqdmpatch: + vd.check_options(opts) + pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) + assert tqdmpatch.called + _args, kwargs = tqdmpatch.call_args + assert kwargs['disable'] != progress_bar From 68c852acecff7a0ee52ba9ef6518d393b79d1b64 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Sep 2019 13:28:02 -0700 Subject: [PATCH 176/880] Remove test_tesseract_config_invalid from suite Also causes problems in CI --- tests/test_main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_main.py b/tests/test_main.py index e84fc052..399d0662 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -630,6 +630,7 @@ def test_tesseract_config_notfound(renderer, resources, outdir): assert p.returncode == ExitCode.ok, err +@pytest.mark.slow # This test sometimes times out in CI @pytest.mark.parametrize('renderer', RENDERERS) def test_tesseract_config_invalid(renderer, resources, outdir): cfg_file = outdir / 'test.cfg' From c149f860b5a1372d95cd558315c5c0f8b8d30180 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Sep 2019 17:02:22 -0700 Subject: [PATCH 177/880] Add contributing guide --- docs/contributing.rst | 38 ++++++++++++++++++++++++++++++++++++++ docs/index.rst | 7 ++++++- 2 files changed, 44 insertions(+), 1 deletion(-) create mode 100644 docs/contributing.rst diff --git a/docs/contributing.rst b/docs/contributing.rst new file mode 100644 index 00000000..2a7c81ff --- /dev/null +++ b/docs/contributing.rst @@ -0,0 +1,38 @@ +======================= +Contributing guidelines +======================= + +Contributions are welcome! + +Big changes +=========== + +Please open a new issue to discuss or propose a major change. Not only is it fun +to discuss big ideas, but we might save each other's time too. Perhaps some of the +work you're contemplating is already half-done in a development branch. + +Code style +========== + +We use PEP8, ``black`` for code formatting and ``isort`` for import sorting. The +settings for programs are in ``pyproject.toml`` and ``setup.cfg``. + +Tests +===== + +New features should come with tests that confirm their correctness. + +New Python dependencies +======================= + +If you are proposing a change that will require a new Python dependency, we +prefer dependencies that are already packaged by Debian or Red Hat. This makes +life much easier for our downstream package maintainers. + +Python dependencies must also be GPLv3 compatible. + +New non-Python dependencies +=========================== + +OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for +its functionality. In general we prefer to avoid adding new external programs. diff --git a/docs/index.rst b/docs/index.rst index 29136779..b050c6da 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -22,11 +22,16 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat cookbook docker advanced - api batch security errors +.. toctree:: + :caption: Developers + :maxdepth: 2 + + api + contributing Indices and tables ================== From de61530d4d4fa62cd57afc5f98eabd0f028bf04b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Sep 2019 17:02:35 -0700 Subject: [PATCH 178/880] docs: fix intermediate file list for v9 --- docs/advanced.rst | 26 +++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/docs/advanced.rst b/docs/advanced.rst index fbc14d1a..6f8567ba 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -322,15 +322,23 @@ working files on a per page basis have the page number as a prefix (starting with page 1), an infix indicates the processing stage, and a suffix indicates the file type. Some important files include: -- ``.page.png`` - what the input page looks like -- ``.image`` - the image we will show the user if we are in a mode that - changes the final appearance; may be in one of several image formats -- ``.text.pdf`` - the OCR file; this will load as a blank page but - should have visible text if checked with a tool like pdftotext or - pdfminder.six -- ``.ocr.png`` - the file that is sent to Tesseract for OCR; depending +- ``_rasterize.png`` - what the input page looks like +- ``_ocr.png`` - the file that is sent to Tesseract for OCR; depending on arguments this may differ from the presentation image -- ``layers.rendered.pdf`` - the composite PDF, before metadata repair - and optimization +- ``_pp_deskew.png`` - the image, after deskewing +- ``_pp_clean.png`` - the image, after cleaning with unpaper +- ``_ocr_tess.pdf`` - the OCR file; appears as a blank page with invisible + text embedded +- ``_ocr_tess.txt`` - the OCR text (not necessarily all text on the page, + if the page is mixed format) +- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo + data structure +- ``graft_layers.pdf`` - the rendered PDF with OCR layers grafted on +- ``pdfa.pdf`` - ``graft_layers.pdf`` after conversion to PDF/A +- ``pdfa.ps`` - a PostScript file used by Ghostscript for PDF/A conversion +- ``optimize.pdf`` - the PDF generated before optimization +- ``optimize.out.pdf`` - the PDF generated by optimization +- ``origin`` - the input file +- ``origin.pdf`` - the input file or the input image converted to PDF - ``images/*`` - images extracted during the optimization process; here the prefix indicates a PDF object ID not a page number From 78e8bf9cbf4a69f71ff5b2d627e4ea291fd3ee52 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Sep 2019 17:11:29 -0700 Subject: [PATCH 179/880] Use at most 3 Tesseract threads Based on a user suggestion and tesseract-ocr/tesseract#2611, I reviewed thread limits and found that thread limit of 3 is still beneficial, but not 4. > time env OMP_THREAD_LIMIT=2 tesseract omp4.png stdout >/dev/null Warning: Invalid resolution 0 dpi. Using 70 instead. Estimating resolution as 143 116.67user 1.67system 1:26.26elapsed 137%CPU (0avgtext+0avgdata 356752maxresident)k 2213inputs+0outputs (18major+131059minor)pagefaults 0swaps > time env OMP_THREAD_LIMIT=3 tesseract omp4.png stdout >/dev/null Warning: Invalid resolution 0 dpi. Using 70 instead. Estimating resolution as 143 136.89user 1.63system 1:19.56elapsed 174%CPU (0avgtext+0avgdata 356784maxresident)k 821inputs+0outputs (0major+131080minor)pagefaults 0swaps > time env OMP_THREAD_LIMIT=4 tesseract omp4.png stdout >/dev/null Warning: Invalid resolution 0 dpi. Using 70 instead. Estimating resolution as 143 161.31user 1.51system 1:18.80elapsed 206%CPU (0avgtext+0avgdata 356632maxresident)k 8477inputs+0outputs (12major+131074minor)pagefaults 0swaps > time env OMP_THREAD_LIMIT=8 tesseract omp4.png stdout >/dev/null Warning: Invalid resolution 0 dpi. Using 70 instead. Estimating resolution as 143 160.30user 1.62system 1:18.01elapsed 207%CPU (0avgtext+0avgdata 356640maxresident)k 821inputs+0outputs (0major+131078minor)pagefaults 0swaps --- src/ocrmypdf/_sync.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index ccedbdc6..b51238a1 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -225,15 +225,15 @@ def exec_concurrent(context): if max_workers > 1: context.log.info("Start processing %d pages concurrent", max_workers) - # Tesseract 4.0 is multithreaded, and we also run multiple workers. We want to - # avoid the situation where we end up trying to run NxN jobs on N CPU cores, - # as that gives poor performance. Performance testing shows we're better off + # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want + # to manage how many threads it uses to avoid creating total threads than cores. + # Performance testing shows we're better off # parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we # get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the # input file is small, then we allow Tesseract to use threads, subject to the - # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers and limiting - # Tesseract to 4 threads. - tess_threads = min(4, context.options.jobs // max_workers) + # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers. + # As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system. + tess_threads = min(3, context.options.jobs // max_workers) if context.options.tesseract_env is None: context.options.tesseract_env = os.environ.copy() context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads)) From 4d26867dee317ccbc8e0920549b414b15baaa054 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Sep 2019 17:17:11 -0700 Subject: [PATCH 180/880] Delinting --- src/ocrmypdf/_sync.py | 4 ++-- src/ocrmypdf/exec/tesseract.py | 2 -- src/ocrmypdf/optimize.py | 6 +++--- tests/test_main.py | 2 +- tests/test_validation.py | 3 ++- 5 files changed, 8 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index b51238a1..c02543b7 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -189,7 +189,7 @@ def worker_init(queue): root.addHandler(h) -def worker_thread_init(queue): +def worker_thread_init(_queue): pass @@ -301,7 +301,7 @@ def exec_concurrent(context): class NeverRaise(Exception): """An exception that is never raised""" - pass + pass # pylint: disable=unnecessary-pass def run_pipeline(options, api=False): diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 03bf037d..05e3266e 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -17,10 +17,8 @@ import os import shutil -import sys from collections import namedtuple from contextlib import suppress -from functools import lru_cache from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index bfe027af..bb1269db 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -549,11 +549,11 @@ def main(infile, outfile, level, jobs=1): """Emulate ocrmypdf's options""" def __init__( - self, input_file, jobs, optimize, jpeg_quality, png_quality, jb2lossy + self, input_file, jobs, optimize_, jpeg_quality, png_quality, jb2lossy ): self.input_file = input_file self.jobs = jobs - self.optimize = optimize + self.optimize = optimize_ self.jpeg_quality = jpeg_quality self.png_quality = png_quality self.jbig2_page_group_size = 0 @@ -564,7 +564,7 @@ def main(infile, outfile, level, jobs=1): options = OptimizeOptions( input_file=infile, jobs=jobs, - optimize=int(level), + optimize_=int(level), jpeg_quality=0, # Use default png_quality=0, jb2lossy=False, diff --git a/tests/test_main.py b/tests/test_main.py index 399d0662..79625d12 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -217,7 +217,7 @@ def test_skip_ocr(spoof_tesseract_cache, resources, outpdf): assert pdfinfo[0].has_text -def test_redo_ocr(spoof_tesseract_cache, resources, outpdf): +def test_redo_ocr(resources, outpdf): in_ = resources / 'graph_ocred.pdf' before = PdfInfo(in_, detailed_page_analysis=True) out = outpdf diff --git a/tests/test_validation.py b/tests/test_validation.py index e5e3c8f9..5413e375 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -16,7 +16,7 @@ # along with OCRmyPDF. If not, see . import os -from unittest.mock import MagicMock, patch, call +from unittest.mock import patch import pytest @@ -135,6 +135,7 @@ def test_no_progress_bar(progress_bar, resources): with patch('ocrmypdf.pdfinfo.info.tqdm', autospec=True) as tqdmpatch: vd.check_options(opts) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) + assert pdfinfo is not None assert tqdmpatch.called _args, kwargs = tqdmpatch.call_args assert kwargs['disable'] != progress_bar From 6e99e7b3467064ce08f86c50a458596a4d1bf9c2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 1 Oct 2019 00:42:58 -0700 Subject: [PATCH 181/880] Use lstm_use_matrix for --user-words,patterns --- src/ocrmypdf/exec/tesseract.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 05e3266e..8f88b563 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -269,6 +269,9 @@ def generate_hocr( if user_patterns: args_tesseract.extend(['--user-patterns', user_patterns]) + if user_words or user_patterns: + args_tesseract.extend(['-c', 'lstm_use_matrix=1']) + # Reminder: test suite tesseract spoofers will break after any changes # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) From b55d7e57af3095fb8564aa559ea15cddc919f86c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 20 Oct 2019 03:20:54 -0700 Subject: [PATCH 182/880] Python 3.8 updates --- README.md | 6 ++++-- docs/installation.rst | 2 +- docs/release_notes.rst | 7 +++++++ requirements/main.txt | 2 +- setup.py | 3 ++- 5 files changed, 15 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index fed27f8e..e5f1d319 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ OCRmyPDF -[![Travis build status][travis]](https://travis-ci.org/jbarlow83/OCRmyPDF) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] +[![Travis build status][travis]](https://travis-ci.org/jbarlow83/OCRmyPDF) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] [travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status" @@ -10,6 +10,8 @@ [docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD" +[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "Supported Python versions" + OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched or copy-pasted. ```bash @@ -120,7 +122,7 @@ If you detect an issue, please: Requirements ------------ -Runs on CPython 3.5, 3.6 and 3.7. Requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. +In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. Press & Media ------------- diff --git a/docs/installation.rst b/docs/installation.rst index 27e27671..77702791 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -517,7 +517,7 @@ manager. ``pip`` cannot provide them. As of ocrmypdf 7.2.1, the following versions are recommended: -- Python 3.7 +- Python 3.7 or 3.8 - Ghostscript 9.23 or newer - qpdf 8.2.1 - Tesseract 4.0.0 or newer diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f90f096d..fbc7c06f 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,13 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.0.4 +====== + +- Fixed compatibility with Python 3.8. +- Fixed Tesseract settings for ``--user-words`` and ``--user-patterns``. +- We now require pikepdf 1.6.5. + v9.0.3 ====== diff --git a/requirements/main.txt b/requirements/main.txt index 24087d27..5fac1105 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ chardet == 3.0.4 cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 -pikepdf == 1.6.1 +pikepdf == 1.6.5 Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 diff --git a/setup.py b/setup.py index 6b2b7d11..e9d167f2 100644 --- a/setup.py +++ b/setup.py @@ -68,6 +68,7 @@ setup( classifiers=[ "Programming Language :: Python :: 3.6", "Programming Language :: Python :: 3.7", + "Programming Language :: Python :: 3.8", "Development Status :: 5 - Production/Stable", "Environment :: Console", "Intended Audience :: End Users/Desktop", @@ -96,7 +97,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six == 20181108', - 'pikepdf >= 1.6.0, < 2', + 'pikepdf >= 1.6.5, < 2', 'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"', # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels From 3660007fc88a1a0c48dfefc10441dde5365cbcff Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 20 Oct 2019 04:06:13 -0700 Subject: [PATCH 183/880] travis: Python 3.8, osx_image --- .travis.yml | 25 ++++++++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 9098888a..c8524373 100644 --- a/.travis.yml +++ b/.travis.yml @@ -88,8 +88,31 @@ matrix: - tesseract-ocr-eng - tesseract-ocr-fra - unpaper + - os: linux + dist: xenial + sudo: required + language: python + python: "3.8" + env: + - DIST=xenial + addons: + apt: + update: true + sources: + - sourceline: "ppa:alex-p/tesseract-ocr" + packages: + - ghostscript + - libexempi3 + - libffi-dev + - pngquant + - poppler-utils + - qpdf + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra + - unpaper - os: osx - osx_image: xcode9.2 language: generic addons: homebrew: From b332d76782b5f110ed1f5d39f457c936400aef68 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Oct 2019 01:49:38 -0700 Subject: [PATCH 184/880] Mention when we default to English and the system locale is not English Closes #337 --- src/ocrmypdf/_validation.py | 9 ++++++++- tests/test_validation.py | 26 +++++++++++++++++++++++--- 2 files changed, 31 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index c5dcb6a4..5a533fa9 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -17,6 +17,7 @@ # along with OCRmyPDF. If not, see . +import locale import logging import os import sys @@ -47,6 +48,7 @@ from .helpers import is_file_writable, is_iterable_notstr, monotonic, re_symlink # External dependencies HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por']) +DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony log = logging.getLogger(__name__) @@ -58,7 +60,12 @@ verify_python3_env() def check_options_languages(options): if not options.language: - options.language = ['eng'] # Enforce English hegemony + options.language = [DEFAULT_LANGUAGE] + system_lang = locale.getlocale()[0] + if system_lang and not system_lang.startswith('en'): + log.debug( + "No language specified; assuming --language %s" % DEFAULT_LANGUAGE + ) # Support v2.x "eng+deu" language syntax if '+' in options.language[0]: diff --git a/tests/test_validation.py b/tests/test_validation.py index 5413e375..5465a580 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -15,6 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import locale +import logging import os from unittest.mock import patch @@ -27,9 +29,9 @@ from ocrmypdf.pdfinfo import PdfInfo def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): - return create_options( - input_file=input_file, output_file=output_file, language=language, **kwargs - ) + if language is not None: + kwargs['language'] = language + return create_options(input_file=input_file, output_file=output_file, **kwargs) def test_hocr_notlatin_warning(caplog): @@ -139,3 +141,21 @@ def test_no_progress_bar(progress_bar, resources): assert tqdmpatch.called _args, kwargs = tqdmpatch.call_args assert kwargs['disable'] != progress_bar + + +def test_language_warning(caplog): + opts = make_opts(language=None) + caplog.set_level(logging.DEBUG) + with patch( + 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') + ): + vd.check_options_languages(opts) + assert opts.language == ['eng'] + assert '' in caplog.text + + with patch( + 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') + ): + vd.check_options_languages(opts) + assert opts.language == ['eng'] + assert 'assuming --language' in caplog.text From cdcdd1686537e0fac5d67292d95c387f0f28d513 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 23 Oct 2019 12:27:29 -0700 Subject: [PATCH 185/880] Require Pillow 6.2.0 based on security vulnerability report in older versions --- requirements/main.txt | 2 +- setup.py | 4 +--- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 5fac1105..5db7550a 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -6,7 +6,7 @@ cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 pikepdf == 1.6.5 -Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" +Pillow >= 6.2.0 pycparser == 2.19 python-xmp-toolkit == 2.0.1 reportlab == 3.5.13 diff --git a/setup.py b/setup.py index e9d167f2..b7d2e2d7 100644 --- a/setup.py +++ b/setup.py @@ -98,9 +98,7 @@ setup( 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six == 20181108', 'pikepdf >= 1.6.5, < 2', - 'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"', - # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 - # block 5.1.0, broken wheels + 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling 'tqdm >= 4', ], From 775b958c555afb066dbc23ad7173639f4ec0c46a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Oct 2019 16:58:39 -0700 Subject: [PATCH 186/880] Update release notes --- docs/release_notes.rst | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index fbc7c06f..4f9a265e 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -18,7 +18,10 @@ v9.0.4 - Fixed compatibility with Python 3.8. - Fixed Tesseract settings for ``--user-words`` and ``--user-patterns``. -- We now require pikepdf 1.6.5. +- Changed to pikepdf 1.6.5 (for Python 3.8). +- Changed to Pillow 6.2.0 (to mitigate a security vulnerability in earlier Pillow). +- A debug message now mentions when English is automatically selected if the locale + is not English. v9.0.3 ====== From a58209e89545b798332ab1764ea6152c3edab0af Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Oct 2019 18:16:47 -0700 Subject: [PATCH 187/880] Disable Py3.8 for now --- .travis.yml | 48 ++++++++++++++++++++++++------------------------ 1 file changed, 24 insertions(+), 24 deletions(-) diff --git a/.travis.yml b/.travis.yml index c8524373..eba47aaf 100644 --- a/.travis.yml +++ b/.travis.yml @@ -88,30 +88,30 @@ matrix: - tesseract-ocr-eng - tesseract-ocr-fra - unpaper - - os: linux - dist: xenial - sudo: required - language: python - python: "3.8" - env: - - DIST=xenial - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - packages: - - ghostscript - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - qpdf - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - unpaper + # - os: linux + # dist: xenial + # sudo: required + # language: python + # python: "3.8" + # env: + # - DIST=xenial + # addons: + # apt: + # update: true + # sources: + # - sourceline: "ppa:alex-p/tesseract-ocr" + # packages: + # - ghostscript + # - libexempi3 + # - libffi-dev + # - pngquant + # - poppler-utils + # - qpdf + # - tesseract-ocr + # - tesseract-ocr-deu + # - tesseract-ocr-eng + # - tesseract-ocr-fra + # - unpaper - os: osx language: generic addons: From 80651fe12c2a817e6f51a2d0a17961f256dca91f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Oct 2019 18:17:03 -0700 Subject: [PATCH 188/880] Fix test suite error --- src/ocrmypdf/_validation.py | 4 +--- tests/test_validation.py | 1 + 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 5a533fa9..fe15a84e 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -63,9 +63,7 @@ def check_options_languages(options): options.language = [DEFAULT_LANGUAGE] system_lang = locale.getlocale()[0] if system_lang and not system_lang.startswith('en'): - log.debug( - "No language specified; assuming --language %s" % DEFAULT_LANGUAGE - ) + log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE) # Support v2.x "eng+deu" language syntax if '+' in options.language[0]: diff --git a/tests/test_validation.py b/tests/test_validation.py index 5465a580..7d70f367 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -153,6 +153,7 @@ def test_language_warning(caplog): assert opts.language == ['eng'] assert '' in caplog.text + opts = make_opts(language=None) with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') ): From 7f8018ffdef6b22b395a1b63295f6d5009815535 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Nov 2019 01:49:36 -0800 Subject: [PATCH 189/880] Mention that v9.0.4 requires a source install for Py3.8 for now, due to lack of CI availability --- docs/release_notes.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 4f9a265e..42b19c28 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,7 +16,7 @@ licensed under GPLv3. v9.0.4 ====== -- Fixed compatibility with Python 3.8. +- Fixed compatibility with Python 3.8 (but requires source install for the moment). - Fixed Tesseract settings for ``--user-words`` and ``--user-patterns``. - Changed to pikepdf 1.6.5 (for Python 3.8). - Changed to Pillow 6.2.0 (to mitigate a security vulnerability in earlier Pillow). From ad48fc641578f1d55a1272c6975a018c65b77455 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Nov 2019 22:35:15 -0800 Subject: [PATCH 190/880] Remove Alpine Docker image --- .docker/alpine.dockerfile | 91 --------------------------------------- docs/batch.rst | 2 +- docs/docker.rst | 22 ++++++---- docs/introduction.rst | 2 +- 4 files changed, 15 insertions(+), 102 deletions(-) delete mode 100644 .docker/alpine.dockerfile diff --git a/.docker/alpine.dockerfile b/.docker/alpine.dockerfile deleted file mode 100644 index b2879e27..00000000 --- a/.docker/alpine.dockerfile +++ /dev/null @@ -1,91 +0,0 @@ -FROM alpine:3.9 as base - -FROM base as builder - -ENV LANG=C.UTF-8 - -# Normally: -# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories - -RUN \ - echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\ - >> /etc/apk/repositories \ - # Add runtime dependencies - && apk add --update \ - python3-dev \ - py3-setuptools \ - jbig2enc@community \ - ghostscript \ - qpdf@community \ - qpdf-dev@community \ - tesseract-ocr \ - unpaper \ - pngquant \ - libxml2-dev \ - libxslt-dev \ - zlib-dev \ - libffi-dev \ - leptonica-dev \ - binutils \ - && pip3 install --upgrade pip \ - # Install pybind11 for pikepdf - && pip3 install pybind11 \ - # Install flask for the webservice - && pip3 install flask \ - # Add build dependencies - && apk add --virtual build-dependencies \ - build-base \ - git - -COPY . /app - -WORKDIR /app - -RUN pip3 install . - -FROM base - -ENV LANG=C.UTF-8 - -# Normally: -# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories - -RUN \ - echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\ - >> /etc/apk/repositories \ - # Add runtime dependencies - && apk add --update \ - python3 \ - jbig2enc@community \ - ghostscript \ - qpdf@community \ - qpdf-dev@community \ - tesseract-ocr \ - tesseract-ocr-data-deu \ - tesseract-ocr-data-chi_sim \ - unpaper \ - pngquant \ - libxml2 \ - libxslt \ - zlib \ - libffi \ - leptonica-dev \ - binutils \ - && mkdir /app - -WORKDIR /app - -# Copy build artifacts (python site-packages) -COPY --from=builder /usr/lib/python3.6/site-packages /usr/lib/python3.6/site-packages -COPY --from=builder /usr/bin/ocrmypdf /usr/bin/dumppdf.py /usr/bin/latin2ascii.py /usr/bin/pdf2txt.py /usr/bin/img2pdf /usr/bin/chardetect /usr/bin/ - -# Copy -COPY --from=builder /app/misc/webservice.py /app/ - -# Copy minimal project files to get the test suite. -COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ -COPY --from=builder /app/requirements /app/requirements -COPY --from=builder /app/tests /app/tests -COPY --from=builder /app/src /app/src - -ENTRYPOINT ["/usr/bin/ocrmypdf"] diff --git a/docs/batch.rst b/docs/batch.rst index 555879b2..d27836c1 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -47,7 +47,7 @@ where the PDFs are stored): .. code-block:: bash - find . -printf '%p' -name '*.pdf' -exec docker run --rm -v : jbarlow83/ocrmypdf-alpine '/{}' '/{}' \; + find . -printf '%p' -name '*.pdf' -exec docker run --rm -v : jbarlow83/ocrmypdf '/{}' '/{}' \; This only runs one ``ocrmypdf`` process at a time. This variation uses ``find`` to create a directory list and ``parallel`` to parallelize runs diff --git a/docs/docker.rst b/docs/docker.rst index 3b34ebba..4cd34d83 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -23,11 +23,11 @@ If you have `Docker `__ installed on your system, you can install a Docker image of the latest release. The recommended OCRmyPDF Docker image is currently named -``ocrmypdf-alpine``: +``ocrmypdf``: .. code-block:: bash - docker pull jbarlow83/ocrmypdf-alpine + docker pull jbarlow83/ocrmypdf Follow the Docker installation instructions for your platform. If you can run this command successfully, your system is ready to download and @@ -63,7 +63,7 @@ To start a Docker container (instance of the image): .. code-block:: bash - docker tag jbarlow83/ocrmypdf-alpine ocrmypdf + docker tag jbarlow83/ocrmypdf ocrmypdf docker run --rm -i ocrmypdf (... all other arguments here...) For convenience, create a shell alias to hide the Docker command. It is @@ -103,7 +103,7 @@ on the public one: .. code-block:: dockerfile - FROM jbarlow83/ocrmypdf-alpine + FROM jbarlow83/ocrmypdf # Add French RUN apk add tesseract-ocr-data-fra @@ -117,17 +117,16 @@ The OCRmyPDF test suite is installed with image. To run it: .. code-block:: bash - docker run --entrypoint python3 jbarlow83/ocrmypdf-alpine setup.py test + docker run --entrypoint python3 jbarlow83/ocrmypdf setup.py test Accessing the shell =================== -``bash`` is not installed in the image. To use the busybox shell in the -Docker image: +To use the bash shell in the Docker image: .. code-block:: bash - docker run -it --entrypoint busybox jbarlow83/ocrmypdf-alpine sh + docker run -it --entrypoint bash jbarlow83/ocrmypdf Using the OCRmyPDF web service wrapper ====================================== @@ -137,7 +136,12 @@ service. The webservice may be launched as follows: .. code-block:: bash - docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf-alpine webservice.py + docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf webservice.py + +This will configure the machine to listen on port 5000. On Linux machines +this is port 5000 of localhost. On macOS or Windows machines running +Docker, this is port 5000 of the virtual machine that runs your Docker +images. You can find its IP address using the command ``docker-machine ip``. Unlike command line usage this program will open a socket and wait for connections. diff --git a/docs/introduction.rst b/docs/introduction.rst index acc23aee..efe1de91 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -208,7 +208,7 @@ consider one of these similar open source programs: Web front-ends ============== -The Docker image ``ocrmypdf-alpine`` provides a web service front-end +The Docker image ``ocrmypdf`` provides a web service front-end that allows files to submitted over HTTP and the results "downloaded". This is an HTTP server intended to simplify web services deployments; it is not intended to be deployed on the public internet and no real From c3719d3b721bdcd7300c79192d6d18e23173f0b5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Nov 2019 23:39:40 -0800 Subject: [PATCH 191/880] Dockerfile: remove venv from Ubuntu image; tweak reqs --- .docker/Dockerfile | 34 ++++++++++++++-------------------- docs/docker.rst | 2 +- requirements/main.txt | 9 +++------ requirements/test.txt | 2 +- requirements/webservice.txt | 1 + tests/test_completion.py | 5 +++++ 6 files changed, 25 insertions(+), 28 deletions(-) create mode 100644 requirements/webservice.txt diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 8661d9c7..ceb69bb0 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -1,6 +1,6 @@ # OCRmyPDF # -FROM ubuntu:19.04 as base +FROM ubuntu:19.10 as base FROM base as builder @@ -10,16 +10,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential autoconf automake libtool \ libleptonica-dev \ zlib1g-dev \ - ocrmypdf \ - pngquant \ + python3-setuptools \ python3-pip \ - python3-venv \ - tesseract-ocr \ - unpaper \ wget \ git - # Compile and install jbig2 # Needs libleptonica-dev, zlib1g-dev RUN \ @@ -31,15 +26,15 @@ RUN \ && cd .. \ && rm -rf jbig2 -RUN python3 -m venv /appenv - COPY . /app WORKDIR /app -RUN . /appenv/bin/activate; \ - pip install --upgrade pip \ - && pip install . +RUN pip3 install \ + -r requirements/main.txt \ + -r requirements/webservice.txt \ + -r requirements/test.txt \ + . FROM base @@ -53,7 +48,6 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ zlib1g \ pngquant \ python3 \ - python3-venv \ qpdf \ tesseract-ocr \ tesseract-ocr-chi-sim \ @@ -62,10 +56,13 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ tesseract-ocr-fra \ tesseract-ocr-por \ tesseract-ocr-spa \ - unpaper \ - wget + unpaper + +WORKDIR /app + +COPY --from=builder /usr/local/lib/python3.7/dist-packages/ /usr/local/lib/python3.7/dist-packages/ +COPY --from=builder /usr/local/bin/ocrmypdf /usr/local/bin/ocrmypdf -# Copy COPY --from=builder /app/misc/webservice.py /app/ # Copy minimal project files to get the test suite. @@ -74,7 +71,4 @@ COPY --from=builder /app/requirements /app/requirements COPY --from=builder /app/tests /app/tests COPY --from=builder /app/src /app/src -COPY --from=builder /appenv /appenv -COPY --from=builder /usr/local /usr/local - -ENTRYPOINT ["/appenv/bin/ocrmypdf"] +ENTRYPOINT ["/usr/local/bin/ocrmypdf"] diff --git a/docs/docker.rst b/docs/docker.rst index 4cd34d83..ed118dfc 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -117,7 +117,7 @@ The OCRmyPDF test suite is installed with image. To run it: .. code-block:: bash - docker run --entrypoint python3 jbarlow83/ocrmypdf setup.py test + docker run --entrypoint python3 jbarlow83/ocrmypdf -m pytest Accessing the shell =================== diff --git a/requirements/main.txt b/requirements/main.txt index 5db7550a..eec5dc7a 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -1,13 +1,10 @@ # requirements.txt can be used to replicate the developer's build environment # setup.py lists a separate set of requirements that are looser to simplify # installation -chardet == 3.0.4 -cffi == 1.12.2 +cffi == 1.13.2 img2pdf == 0.3.3 pdfminer.six == 20181108 pikepdf == 1.6.5 Pillow >= 6.2.0 -pycparser == 2.19 -python-xmp-toolkit == 2.0.1 -reportlab == 3.5.13 -tqdm == 4.32.1 +reportlab == 3.5.32 +tqdm == 4.37.0 diff --git a/requirements/test.txt b/requirements/test.txt index 975325bd..58fd0fdd 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -2,7 +2,7 @@ pytest >= 5.0.0 pytest-helpers-namespace >= 2019.1.8 pytest-xdist >= 1.29.0 # For DumpError fix pytest-cov >= 2.6.1 -python-xmp-toolkit # requires apt-get install libexempi3 +python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi PyPDF2 >= 1.26.0 #PyMuPDF == 1.13.4 # optional diff --git a/requirements/webservice.txt b/requirements/webservice.txt new file mode 100644 index 00000000..f6e3c4e6 --- /dev/null +++ b/requirements/webservice.txt @@ -0,0 +1 @@ +Flask >= 1, < 2 diff --git a/tests/test_completion.py b/tests/test_completion.py index 8fdbb59a..8837fbe2 100644 --- a/tests/test_completion.py +++ b/tests/test_completion.py @@ -19,6 +19,11 @@ from subprocess import run, PIPE import pytest +pytestmark = pytest.mark.skipif( + pytest.helpers.running_in_docker(), # pylint: disable=no-member + reason="docker can't complete", +) + def test_fish(): try: From a492e3b4720e35717108ddbe38fbfee06ebf2eac Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Nov 2019 23:51:55 -0800 Subject: [PATCH 192/880] Dockerfile: fix errors are trying to build unneeded cached wheels --- .docker/Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index ceb69bb0..dd13a3f0 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -30,7 +30,7 @@ COPY . /app WORKDIR /app -RUN pip3 install \ +RUN pip3 install --no-cache-dir \ -r requirements/main.txt \ -r requirements/webservice.txt \ -r requirements/test.txt \ From 3a4490ee363e9cfa9373a8e293acb7512fc578a0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Nov 2019 23:52:08 -0800 Subject: [PATCH 193/880] Dockerfile: fix jbig2 not copied over --- .docker/Dockerfile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index dd13a3f0..95edba47 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -60,8 +60,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ WORKDIR /app -COPY --from=builder /usr/local/lib/python3.7/dist-packages/ /usr/local/lib/python3.7/dist-packages/ -COPY --from=builder /usr/local/bin/ocrmypdf /usr/local/bin/ocrmypdf +COPY --from=builder /usr/local/lib/ /usr/local/lib/ +COPY --from=builder /usr/local/bin/ /usr/local/bin/ COPY --from=builder /app/misc/webservice.py /app/ From 99db5d91ae87e86ce73fe3ba8aa77093b0b2b15f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 00:03:49 -0800 Subject: [PATCH 194/880] Fix issue "MANIFEST.in exists" by removing MANIFEST.in MANIFEST.in is always an issue --- MANIFEST.in | 43 ------------------------------------------- 1 file changed, 43 deletions(-) delete mode 100644 MANIFEST.in diff --git a/MANIFEST.in b/MANIFEST.in deleted file mode 100644 index 2021f662..00000000 --- a/MANIFEST.in +++ /dev/null @@ -1,43 +0,0 @@ -# requirements -recursive-include requirements * - -# git -include .git_archival.txt - -# docker -include .dockerignore -recursive-include .docker * - -# tests -include .coveragerc -recursive-include tests *.bin -recursive-include tests *.jpg -recursive-include tests *.jsonl -recursive-include tests *.png -recursive-include tests *.pdf -recursive-include tests *.py -recursive-include tests *.rst -recursive-include tests *.txt -recursive-exclude tests/resources/private * - -# documentation -include LICENSE -include *.rst -recursive-exclude .github * -recursive-include docs *.py -recursive-include docs *.rst -recursive-include docs *.svg -recursive-exclude docs/_build * - - -# support files -recursive-include src/ocrmypdf/data * -include *.py -exclude tasks.py -recursive-exclude .travis * -exclude .travis* - - -# code -exclude src/ocrmypdf/lib/_leptonica.py -exclude scratch.py From 1ee829dd598094cc7a5077941a1e8b9b8f4b1026 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 00:05:18 -0800 Subject: [PATCH 195/880] Travis: enable Python 3.8 testing --- .travis.yml | 53 +++++++++++++++++++++++++---------------------------- 1 file changed, 25 insertions(+), 28 deletions(-) diff --git a/.travis.yml b/.travis.yml index eba47aaf..887471db 100644 --- a/.travis.yml +++ b/.travis.yml @@ -88,30 +88,30 @@ matrix: - tesseract-ocr-eng - tesseract-ocr-fra - unpaper - # - os: linux - # dist: xenial - # sudo: required - # language: python - # python: "3.8" - # env: - # - DIST=xenial - # addons: - # apt: - # update: true - # sources: - # - sourceline: "ppa:alex-p/tesseract-ocr" - # packages: - # - ghostscript - # - libexempi3 - # - libffi-dev - # - pngquant - # - poppler-utils - # - qpdf - # - tesseract-ocr - # - tesseract-ocr-deu - # - tesseract-ocr-eng - # - tesseract-ocr-fra - # - unpaper + - os: linux + dist: bionic + sudo: required + language: python + python: "3.8" + env: + - DIST=bionic + addons: + apt: + update: true + sources: + - sourceline: "ppa:alex-p/tesseract-ocr" + packages: + - ghostscript + - libexempi3 + - libffi-dev + - pngquant + - poppler-utils + - qpdf + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra + - unpaper - os: osx language: generic addons: @@ -138,10 +138,7 @@ before_cache: install: - mkdir -p bin - export PATH=$PWD/bin:$PATH - - pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251 - - pip3 install -r requirements/main.txt - - pip3 install --no-deps . - - pip3 install -r requirements/test.txt + - pip3 install -r requirements/main.txt -r requirements/test.txt . script: - tesseract --version From 4da5214ca9d18936308e25d57e494a2eb1d874bd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 00:09:04 -0800 Subject: [PATCH 196/880] Drop support for unpaper 6.1 on Ubuntu 14.04 --- .travis.yml | 2 -- docs/installation.rst | 14 +++++--------- 2 files changed, 5 insertions(+), 11 deletions(-) diff --git a/.travis.yml b/.travis.yml index 887471db..de391a8a 100644 --- a/.travis.yml +++ b/.travis.yml @@ -62,8 +62,6 @@ matrix: mkdir -p bin packages pip3 install --upgrade pip pip3 install --upgrade wheel - wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb - sudo dpkg -i packages/unpaper_6.1-1.deb - os: linux dist: xenial sudo: required diff --git a/docs/installation.rst b/docs/installation.rst index 77702791..1f89922d 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -272,15 +272,11 @@ Now we need to install ``pip`` and let it install ocrmypdf: curl https://bootstrap.pypa.io/ez_setup.py -o - | python3.6 && python3.6 -m easy_install pip pip3.6 install ocrmypdf -These installation instructions omit the optional dependency -``unpaper``, which is only available at version 0.4.2 in Ubuntu 14.04. -The author could not find a backport of ``unpaper``, and created a .deb -package to do the job of installing unpaper 6.1 (for x86 64-bit only): - -.. code-block:: bash - - wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O unpaper_6.1-1.deb - sudo dpkg -i unpaper_6.1-1.deb +The optional dependency ``unpaper`` is only available at 0.4.2 in Ubuntu 14.04, +and no backports are available. Previously the author maintained a backported +.deb package for unpaper 6.1, but since Ubuntu 14.04 is now end of life, this is +not supported. As such, ``unpaper`` is not available on Ubuntu 14.04 or must by +compiled by hand. To add JBIG2 encoding, see :ref:`jbig2`. From 05eb85ee770858421001f206cd9f84cbf19ce79f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 01:23:54 -0800 Subject: [PATCH 197/880] Docker: try adding automated test --- .docker/docker-compose.test.yml | 3 +++ 1 file changed, 3 insertions(+) create mode 100644 .docker/docker-compose.test.yml diff --git a/.docker/docker-compose.test.yml b/.docker/docker-compose.test.yml new file mode 100644 index 00000000..ca4881db --- /dev/null +++ b/.docker/docker-compose.test.yml @@ -0,0 +1,3 @@ +sut: + build: . + command: python3 -m pytest From 031b800aac03c4acebbc4b4e278bbbfbce7f856e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 02:04:07 -0800 Subject: [PATCH 198/880] Docker autotest: fix, maybe? --- .docker/docker-compose.test.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.docker/docker-compose.test.yml b/.docker/docker-compose.test.yml index ca4881db..30ff6455 100644 --- a/.docker/docker-compose.test.yml +++ b/.docker/docker-compose.test.yml @@ -1,3 +1,3 @@ sut: - build: . + build: ../ command: python3 -m pytest From d656b2b3f2c9294f641aa3396b1113d5454900eb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 02:08:42 -0800 Subject: [PATCH 199/880] docs: remove comment about Ubuntu image [ci skip] --- docs/docker.rst | 31 +++++++++++-------------------- 1 file changed, 11 insertions(+), 20 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index ed118dfc..8c356787 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -22,21 +22,20 @@ Installing the Docker image If you have `Docker `__ installed on your system, you can install a Docker image of the latest release. -The recommended OCRmyPDF Docker image is currently named -``ocrmypdf``: - -.. code-block:: bash - - docker pull jbarlow83/ocrmypdf - -Follow the Docker installation instructions for your platform. If you -can run this command successfully, your system is ready to download and +If you can run this command successfully, your system is ready to download and execute the image: .. code-block:: bash docker run hello-world +The recommended OCRmyPDF Docker image is currently named ``ocrmypdf``: + +.. code-block:: bash + + docker pull jbarlow83/ocrmypdf + + OCRmyPDF will use all available CPU cores. By default, the VirtualBox machine instance on Windows and macOS has only a single CPU core enabled. Use the VirtualBox Manager to determine the name of your Docker @@ -51,6 +50,9 @@ CPUs: docker-machine start "yourVM" eval $(docker-machine env "yourVM") +See the Docker documentation for +`adjusting memory and CPU on other platforms `__. + Using the Docker image on the command line ========================================== @@ -166,14 +168,3 @@ also licensed in this way. In addition to the above, please read our :ref:`general remarks on using OCRmyPDF as a service `. - -Ubuntu-based Docker image -========================= - -A Ubuntu-based OCRmyPDF image is also available. The main advantage this -image offers is that it supports manylinux Python wheels (which are not -supported on Alpine Linux). This may be useful for plugins. - -.. code-block:: bash - - docker pull jbarlow83/ocrmypdf From 6c23b137e21aa2022579b8f882c1f1ccaf124a00 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 02:27:30 -0800 Subject: [PATCH 200/880] Docker: relocate dockerfile --- .docker/docker-compose.test.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.docker/docker-compose.test.yml b/.docker/docker-compose.test.yml index 30ff6455..d6492fac 100644 --- a/.docker/docker-compose.test.yml +++ b/.docker/docker-compose.test.yml @@ -1,3 +1,5 @@ sut: - build: ../ + build: + context: ../ + dockerfile: .docker/Dockerfile command: python3 -m pytest From 983835cce4634f7d9ecda55140c8b36a8de26fac Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 02:32:29 -0800 Subject: [PATCH 201/880] docs: add remark about optimizing without OCR --- docs/cookbook.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 75e8a52d..4ec4ff4d 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -216,6 +216,15 @@ processing or PDF/A conversion. ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf +Optimize images without performing OCR +-------------------------------------- + +You can also optimize all images without performing any OCR: + +.. code-block:: bash + + ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf + Redo existing OCR ================= From 69e80f1545ac4741a0a0ddc89cb84dee7deb4853 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 02:58:57 -0800 Subject: [PATCH 202/880] docker-compose.test does not seem to be ready for production use --- .docker/docker-compose.test.yml | 5 ----- 1 file changed, 5 deletions(-) delete mode 100644 .docker/docker-compose.test.yml diff --git a/.docker/docker-compose.test.yml b/.docker/docker-compose.test.yml deleted file mode 100644 index d6492fac..00000000 --- a/.docker/docker-compose.test.yml +++ /dev/null @@ -1,5 +0,0 @@ -sut: - build: - context: ../ - dockerfile: .docker/Dockerfile - command: python3 -m pytest From 681fa039cc3df04dd2b33a1fd42843a9eb8b1f9a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 03:00:15 -0800 Subject: [PATCH 203/880] Update release notes; disable Py3.8 test again --- .travis.yml | 48 +++++++++++++++++++++--------------------- docs/release_notes.rst | 9 ++++++++ 2 files changed, 33 insertions(+), 24 deletions(-) diff --git a/.travis.yml b/.travis.yml index de391a8a..7291c685 100644 --- a/.travis.yml +++ b/.travis.yml @@ -86,30 +86,30 @@ matrix: - tesseract-ocr-eng - tesseract-ocr-fra - unpaper - - os: linux - dist: bionic - sudo: required - language: python - python: "3.8" - env: - - DIST=bionic - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - packages: - - ghostscript - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - qpdf - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - unpaper + # - os: linux + # dist: bionic + # sudo: required + # language: python + # python: "3.8" + # env: + # - DIST=bionic + # addons: + # apt: + # update: true + # sources: + # - sourceline: "ppa:alex-p/tesseract-ocr" + # packages: + # - ghostscript + # - libexempi3 + # - libffi-dev + # - pngquant + # - poppler-utils + # - qpdf + # - tesseract-ocr + # - tesseract-ocr-deu + # - tesseract-ocr-eng + # - tesseract-ocr-fra + # - unpaper - os: osx language: generic addons: diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 42b19c28..958b4512 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,15 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.0.5 +====== + +- The Alpine Docker image (jbarlow83/ocrmypdf-alpine) has been dropped due to + the difficulties of supporting Alpine Linux. +- The primary Docker image (jbarlow83/ocrmypdf) has been improved to take on + the extra features that used to be exclusive to the Alpine image. +- No changes to application code. + v9.0.4 ====== From 3438afaffe590029f0f759a5dcb2eb815e1c9826 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 4 Nov 2019 03:15:59 -0800 Subject: [PATCH 204/880] Support pdfminer.six 20191020 --- docs/release_notes.rst | 1 + requirements/main.txt | 2 +- setup.py | 2 +- src/ocrmypdf/pdfinfo/layout.py | 71 ++++++++++++++++++---------------- 4 files changed, 41 insertions(+), 35 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 958b4512..1cd7d7ac 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -21,6 +21,7 @@ v9.0.5 - The primary Docker image (jbarlow83/ocrmypdf) has been improved to take on the extra features that used to be exclusive to the Alpine image. - No changes to application code. +- pdfminer.six version 20191020 is now supported. v9.0.4 ====== diff --git a/requirements/main.txt b/requirements/main.txt index eec5dc7a..f7c2275f 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,7 +3,7 @@ # installation cffi == 1.13.2 img2pdf == 0.3.3 -pdfminer.six == 20181108 +pdfminer.six == 20191020 pikepdf == 1.6.5 Pillow >= 6.2.0 reportlab == 3.5.32 diff --git a/setup.py b/setup.py index b7d2e2d7..1dcebd40 100644 --- a/setup.py +++ b/setup.py @@ -96,7 +96,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six == 20181108', + 'pdfminer.six >= 20181108, <= 20191020', 'pikepdf >= 1.6.5, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 9bb7f3a5..31d0e8eb 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -20,6 +20,7 @@ from math import copysign from pathlib import Path from unittest.mock import patch +import pdfminer import pdfminer.encodingdb import pdfminer.pdfdevice import pdfminer.pdfinterp @@ -36,51 +37,54 @@ from ..exceptions import EncryptedPdfError STRIP_NAME = re.compile(r'[0-9]+') # -# Unconditional pdfminer patches +# pdfminer 20181108 patches # +if pdfminer.__version__ == '20181108': -def name2unicode(name): - """Fix pdfminer's name2unicode function + def name2unicode(name): + """Fix pdfminer's name2unicode function - Font cids that are mapped to names of the form /g123 seem to be, by convention - characters with no corresponding Unicode entry. These can be subsetted fonts - or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode, - barring a ToUnicode data structure. - """ - if name in glyphname2unicode: - return glyphname2unicode[name] - if name.startswith('g') or name.startswith('a'): - raise KeyError(name) - if name.startswith('uni'): - try: - return chr(int(name[3:], 16)) - except ValueError: # Not hexadecimal + Font cids that are mapped to names of the form /g123 seem to be, by convention + characters with no corresponding Unicode entry. These can be subsetted fonts + or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode, + barring a ToUnicode data structure. + """ + if name in glyphname2unicode: + return glyphname2unicode[name] + if name.startswith('g') or name.startswith('a'): raise KeyError(name) - m = STRIP_NAME.search(name) - if not m: - raise KeyError(name) - return chr(int(m.group(0))) + if name.startswith('uni'): + try: + return chr(int(name[3:], 16)) + except ValueError: # Not hexadecimal + raise KeyError(name) + m = STRIP_NAME.search(name) + if not m: + raise KeyError(name) + return chr(int(m.group(0))) + pdfminer.encodingdb.name2unicode = name2unicode -pdfminer.encodingdb.name2unicode = name2unicode + original_PDFFont_init = PDFFont.__init__ -original_PDFFont_init = PDFFont.__init__ + def PDFFont__init__(self, descriptor, widths, default_width=None): + original_PDFFont_init(self, descriptor, widths, default_width) + # PDF spec says descent should be negative + # A font with a positive descent implies it floats entirely above the + # baseline, i.e. it's not really a baseline anymore. I have fonts that + # claim a positive descent, but treating descent as positive always seems + # to misposition text. + if self.descent > 0: + self.descent = -self.descent + PDFFont.__init__ = PDFFont__init__ -def PDFFont__init__(self, descriptor, widths, default_width=None): - original_PDFFont_init(self, descriptor, widths, default_width) - # PDF spec says descent should be negative - # A font with a positive descent implies it floats entirely above the - # baseline, i.e. it's not really a baseline anymore. I have fonts that - # claim a positive descent, but treating descent as positive always seems - # to misposition text. - if self.descent > 0: - self.descent = -self.descent +# +# end of pdfminer 20181108 patches +# -PDFFont.__init__ = PDFFont__init__ - original_PDFSimpleFont_init = PDFSimpleFont.__init__ @@ -97,6 +101,7 @@ def PDFSimpleFont__init__(self, descriptor, widths, spec): PDFSimpleFont.__init__ = PDFSimpleFont__init__ + # # pdfminer patches when creator is PScript5.dll # From 979b0bcaed8e6fcfcf73dc716a4b99fc382c60ef Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 5 Nov 2019 15:38:09 -0800 Subject: [PATCH 205/880] tesseract: refactor logging --- src/ocrmypdf/exec/tesseract.py | 27 ++++++++++++++++----------- 1 file changed, 16 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 8f88b563..0b9dcaf3 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -19,6 +19,7 @@ import os import shutil from collections import namedtuple from contextlib import suppress +import logging from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run @@ -50,6 +51,11 @@ HOCR_TEMPLATE = """ """ +class TesseractLoggerAdapter(logging.LoggerAdapter): + def process(self, msg, kwargs): + return '[tesseract] %s' % (msg), kwargs + + def version(tesseract_env=None): return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env) @@ -177,15 +183,14 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env= return oc -def tesseract_log_output(log, stdout, input_file): - prefix = "[tesseract] " +def tesseract_log_output(mainlog, stdout, input_file): + log = TesseractLoggerAdapter(mainlog, extra=mainlog.extra) try: text = stdout.decode() except UnicodeDecodeError: log.error( - prefix - + "command line output was not utf-8. " + "command line output was not utf-8. " + "This usually means Tesseract's language packs do not match " "the installed version of Tesseract." ) @@ -198,25 +203,25 @@ def tesseract_log_output(log, stdout, input_file): elif line.startswith("Warning in pixReadMem"): continue elif 'diacritics' in line: - log.warning(prefix + "lots of diacritics - possibly poor OCR") + log.warning("lots of diacritics - possibly poor OCR") elif line.startswith('OSD: Weak margin'): - log.warning(prefix + "unsure about page orientation") + log.warning("unsure about page orientation") elif 'Error in pixScanForForeground' in line: pass # Appears to be spurious/problem with nonwhite borders elif 'Error in boxClipToRectangle' in line: pass # Always appears with pixScanForForeground message elif 'parameter not found: ' in line.lower(): - log.error(prefix + line.strip()) + log.error(line.strip()) problem = line.split('found: ')[1] raise TesseractConfigError(problem) elif 'error' in line.lower() or 'exception' in line.lower(): - log.error(prefix + line.strip()) + log.error(line.strip()) elif 'warning' in line.lower(): - log.warning(prefix + line.strip()) + log.warning(line.strip()) elif 'read_params_file' in line.lower(): - log.error(prefix + line.strip()) + log.error(line.strip()) else: - log.info(prefix + line.strip()) + log.info(line.strip()) def page_timedout(log, input_file, timeout): From e13a673b1a5c0494958f36af0de63e4e458ac732 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Nov 2019 02:59:02 -0800 Subject: [PATCH 206/880] docs: mention how to suppress progbar --- docs/api.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/api.rst b/docs/api.rst index 09a68539..ef429066 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -71,7 +71,8 @@ Progress monitoring OCRmyPDF uses the ``tqdm`` package to implement its progress bars. :func:`ocrmypdf.configure_logging` will set up logging output to ``sys.stderr`` in a way that is compatible with the display of the -progress bar. +progress bar. Use ``ocrmypdf.ocr(...progress_bar=False)`` to disable +the progress bar. Exceptions ---------- From 1273e7aeda19e62ebb7d71da098b36a54d9cd141 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Nov 2019 03:22:28 -0800 Subject: [PATCH 207/880] docs: document optimization --- docs/index.rst | 1 + docs/optimizer.rst | 71 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 72 insertions(+) create mode 100644 docs/optimizer.rst diff --git a/docs/index.rst b/docs/index.rst index b050c6da..c3759efb 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -12,6 +12,7 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat introduction release_notes installation + optimizer languages jbig2 diff --git a/docs/optimizer.rst b/docs/optimizer.rst new file mode 100644 index 00000000..f6336f77 --- /dev/null +++ b/docs/optimizer.rst @@ -0,0 +1,71 @@ +================ +PDF optimization +================ + +OCRmyPDF includes an image-oriented PDF optimizer. By default, the optimizer +runs with safe settings with the goal of improving compression at no loss of +quality. At higher optimization levels, lossy optimizations may be applied and +tuned. Optimization occurs after OCR, and only if OCR succeeded. It does not +perform other possible optimizations such as deduplicating resources, +consolidating fonts, simplifying vector drawings, or anything of that nature. + +Optimization ranges from ``-O0`` through ``-O3``, where ``0`` disables +optimization and ``3`` implements all options. ``1``, the default, performs only +safe and lossless optimizations. (This is similar to GCC's optimization +parameter.) The exact type of optimizations performed will vary over time. + +Optimizations that always occurs +================================ + +OCRmyPDF will automatically replace obsolete or inferior compression schemes +such as RLE or LZW with superior schemes such as Deflate and converting +monochrome images to CCITT G4. Since this is harmless it always occurs and there +is no way to disable it. Other non-image compressed objects are compressed as +well. + +Fast web view +============= + +OCRmyPDF automatically optimizes PDFs for "fast web view" in Adobe Acrobat's +parlance, or equivalently, linearizes PDFs so that the resources they reference +are presented in the order a viewer needs them for sequential display. This +reduces the latency of viewing a PDF both online and from local storage. This +actually slightly increases the file size. + +To disable this optimization and all others, use ``ocrmypdf --optimize 0 ...`` +or the shorthand ``-O0``. + +Lossless optimizations +====================== + +At optimization level ``-O1`` (the default), OCRmyPDF will also attempt lossless +image optimization. + +If a JBIG2 encoder is available, then monochrome images will be converted to +JBIG2, with the potential for huge savings on large black and white images, +since JBIG2 is far more efficient than any other monochrome (bi-level) +compression. (All known US patents related to JBIG2 have probably expired, but +it remains the responsibility of the user to supply a JBIG2 encoder such as +`jbig2enc `__. OCRmyPDF does not implement +JBIG2 encoding on its own.) + +OCRmyPDF currently does not attempt to recompress losslessly compressed objects +more aggressively. + +Lossy optimizations +=================== + +At optimization level ``-O2`` and ``-O3``, OCRmyPDF will some attempt lossy +image optimization. + +If ``pngquant`` is installed, OCRmyPDF will use it to perform quantize paletted +images to reduce their size. + +The quality of JPEGs may be lowered, on the assumption that a lower quality +image may be suitable for storage after OCR. + +It is not possible to optimize all image types. Uncommon image types may be +skipped by the optimizer. + +OCRmyPDF provides :ref:`lossy mode JBIG2 ` as an advanced feature +that additional requires the argument ``--jbig2-lossy``. From df4a8faecda4f94eac048aa29ecb95537ac9d38d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Nov 2019 03:24:54 -0800 Subject: [PATCH 208/880] docs: mention systemd for batches --- docs/batch.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/batch.rst b/docs/batch.rst index d27836c1..e3fe9920 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -244,6 +244,9 @@ Caveats Alternatives ------------ +- `systemd user services `__ + can be configured to automatically perform OCR on a collection of files. + - `Watchman `__ is a more powerful alternative to ``watchmedo``. From db914d4cd1b2f0aafb1ef72b7e571edc242605c2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Nov 2019 14:21:57 -0800 Subject: [PATCH 209/880] Report missing optional dependencies as possible cause of file size increase --- src/ocrmypdf/_validation.py | 14 ++++++++++++++ tests/test_validation.py | 12 ++++++++++++ 2 files changed, 26 insertions(+) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index fe15a84e..b9df4b12 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -418,6 +418,20 @@ def report_output_file_size(options, input_file, output_file): f"The argument --{arg.replace('_', '-')} was issued, causing transcoding." ) + if options.optimize == 0: + reasons.append("Optimization was disabled.") + else: + image_optimizers = { + 'jbig2': jbig2enc.available(), + 'pngquant': pngquant.available(), + } + for name, available in image_optimizers.items(): + if not available: + reasons.append( + f"The optional dependency '{name}' was not found, so some image " + f"optimizations could not be attempted." + ) + if reasons: explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n" else: diff --git a/tests/test_validation.py b/tests/test_validation.py index 7d70f367..1c9bc1bc 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -117,11 +117,23 @@ def test_report_file_size(tmp_path, caplog): opts = make_opts() vd.report_output_file_size(opts, in_, out) assert caplog.text == '' + caplog.clear() os.truncate(in_, 25001) os.truncate(out, 50000) vd.report_output_file_size(opts, in_, out) assert 'No reason' in caplog.text + caplog.clear() + + with patch('ocrmypdf._validation.jbig2enc.available', return_value=False): + vd.report_output_file_size(opts, in_, out) + assert 'optional dependency' in caplog.text + caplog.clear() + + opts = make_opts(in_, out, optimize=0) + vd.report_output_file_size(opts, in_, out) + assert 'disabled' in caplog.text + caplog.clear() def test_false_action_store_true(): From 45bea1c0e096197084288c56d3edc71864a1eb9d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Nov 2019 14:56:30 -0800 Subject: [PATCH 210/880] Import and docstring cleanup --- src/ocrmypdf/exec/ghostscript.py | 9 ++++++++- src/ocrmypdf/exec/jbig2enc.py | 2 ++ src/ocrmypdf/exec/pngquant.py | 2 ++ src/ocrmypdf/exec/qpdf.py | 2 ++ src/ocrmypdf/exec/tesseract.py | 2 ++ src/ocrmypdf/exec/unpaper.py | 3 ++- src/ocrmypdf/pdfa.py | 2 -- tests/test_validation.py | 1 - 8 files changed, 18 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 14a1332a..44fed271 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -15,8 +15,11 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Interface to Ghostscript executable""" + import logging import re +import warnings from functools import lru_cache from os import fspath from shutil import copy @@ -193,7 +196,7 @@ def generate_pdfa( output_file, compression, log, - threads=1, + threads=None, # deprecated parameter pdf_version='1.5', pdfa_part='2', ): @@ -216,6 +219,10 @@ def generate_pdfa( """ if not log: log = gslog + if threads is not None: + warnings.warn( + "use of deprecated parameter 'threads'", category=DeprecationWarning + ) compression_args = [] if compression == 'jpeg': diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index 696c899c..dff450c8 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -15,6 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Interface to jbig2 executable""" + from functools import lru_cache from subprocess import PIPE, run diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index 5dbe5b64..17721065 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -15,6 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Interface to pngquant executable""" + from functools import lru_cache from subprocess import run from tempfile import NamedTemporaryFile diff --git a/src/ocrmypdf/exec/qpdf.py b/src/ocrmypdf/exec/qpdf.py index 653cc4ff..e96848c0 100644 --- a/src/ocrmypdf/exec/qpdf.py +++ b/src/ocrmypdf/exec/qpdf.py @@ -15,6 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Interface to qpdf executable""" + from functools import lru_cache from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, run diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 0b9dcaf3..f6b13d59 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -15,6 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Interface to Tesseract executable""" + import os import shutil from collections import namedtuple diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index e186dea8..4515a33f 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -18,10 +18,11 @@ # unpaper documentation: # https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md +"""Interface to unpaper executable""" + import os import shlex import subprocess -import sys from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 5f2a192e..617da444 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -32,8 +32,6 @@ Ghostscript's handling of pdfmark. """ import base64 -import os -from binascii import hexlify from pathlib import Path from string import Template diff --git a/tests/test_validation.py b/tests/test_validation.py index 1c9bc1bc..78181e41 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import locale import logging import os from unittest.mock import patch From 0c4b69ec5a6b266e82b51504f751318266113259 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Nov 2019 14:56:43 -0800 Subject: [PATCH 211/880] Fix lint warning about missing cur_item --- src/ocrmypdf/pdfinfo/layout.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 31d0e8eb..af9d7961 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -212,6 +212,7 @@ class TextPositionTracker(PDFLayoutAnalyzer): super().__init__(rsrcmgr, pageno, laparams) self.textstate = None self.result = None + self.cur_item = None # not defined in pdfminer code as it should be def begin_page(self, page, ctm): super().begin_page(page, ctm) From 9b2ab92913306143f2fac2f33019e33e71cccc06 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:19:01 -0800 Subject: [PATCH 212/880] tesseract: fix exception when logger is RootLogger --- src/ocrmypdf/exec/tesseract.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index f6b13d59..afa0e4a6 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -186,7 +186,9 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env= def tesseract_log_output(mainlog, stdout, input_file): - log = TesseractLoggerAdapter(mainlog, extra=mainlog.extra) + log = TesseractLoggerAdapter( + mainlog, extra=mainlog.extra if hasattr(mainlog, 'extra') else None + ) try: text = stdout.decode() From 11a5c809174f4275caa77324f1f36ca4642ec554 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:19:15 -0800 Subject: [PATCH 213/880] travis: enable Py 3.8 --- .travis.yml | 48 ++++++++++++++++++++++++------------------------ 1 file changed, 24 insertions(+), 24 deletions(-) diff --git a/.travis.yml b/.travis.yml index 7291c685..de391a8a 100644 --- a/.travis.yml +++ b/.travis.yml @@ -86,30 +86,30 @@ matrix: - tesseract-ocr-eng - tesseract-ocr-fra - unpaper - # - os: linux - # dist: bionic - # sudo: required - # language: python - # python: "3.8" - # env: - # - DIST=bionic - # addons: - # apt: - # update: true - # sources: - # - sourceline: "ppa:alex-p/tesseract-ocr" - # packages: - # - ghostscript - # - libexempi3 - # - libffi-dev - # - pngquant - # - poppler-utils - # - qpdf - # - tesseract-ocr - # - tesseract-ocr-deu - # - tesseract-ocr-eng - # - tesseract-ocr-fra - # - unpaper + - os: linux + dist: bionic + sudo: required + language: python + python: "3.8" + env: + - DIST=bionic + addons: + apt: + update: true + sources: + - sourceline: "ppa:alex-p/tesseract-ocr" + packages: + - ghostscript + - libexempi3 + - libffi-dev + - pngquant + - poppler-utils + - qpdf + - tesseract-ocr + - tesseract-ocr-deu + - tesseract-ocr-eng + - tesseract-ocr-fra + - unpaper - os: osx language: generic addons: From 1c303afe21971b695eb7d70dbe94f7f1cf723bda Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:22:30 -0800 Subject: [PATCH 214/880] docs: fix installation instructions for pikepdf manylinux2010 wheels --- docs/installation.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 1f89922d..46bee8d3 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -152,7 +152,8 @@ user's ``PATH`` to check for the user's Python packages. .. code-block:: bash export PATH=$HOME/.local/bin:$PATH - pip3 install --user ocrmypdf + python3 -m pip install --user --upgrade pip + python3 -m pip install --user ocrmypdf To add JBIG2 encoding, see :ref:`jbig2`. From 5bd6665b499a6facf3a52bf138e3c50f992137ae Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:36:38 -0800 Subject: [PATCH 215/880] Use pikepdf 1.7.0 to improve Python 3.8 support --- requirements/main.txt | 2 +- setup.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index f7c2275f..c353ed51 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -4,7 +4,7 @@ cffi == 1.13.2 img2pdf == 0.3.3 pdfminer.six == 20191020 -pikepdf == 1.6.5 +pikepdf == 1.7.0 Pillow >= 6.2.0 reportlab == 3.5.32 tqdm == 4.37.0 diff --git a/setup.py b/setup.py index 1dcebd40..1a1d026f 100644 --- a/setup.py +++ b/setup.py @@ -97,7 +97,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six >= 20181108, <= 20191020', - 'pikepdf >= 1.6.5, < 2', + 'pikepdf >= 1.7.0, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling 'tqdm >= 4', From 000040d4970fe44396c3003b145bf2b550d30108 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:39:33 -0800 Subject: [PATCH 216/880] v9.1.0 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 1cd7d7ac..4681d44a 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,14 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.1.0 +====== + +- Improved diagnostics when file size increases at output. Now warns if JBIG2 + or pngquant were not available. +- pikepdf 1.7.0 is now required, to pick up changes that remove the need for + a source install on Linux systems running Python 3.8. + v9.0.5 ====== From 703b6db95c5e838a841e39674cfce0850018ec03 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 11 Nov 2019 22:58:48 -0800 Subject: [PATCH 217/880] test: fix test_report_file_size --- tests/test_validation.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index 78181e41..7f2ece47 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -120,8 +120,9 @@ def test_report_file_size(tmp_path, caplog): os.truncate(in_, 25001) os.truncate(out, 50000) - vd.report_output_file_size(opts, in_, out) - assert 'No reason' in caplog.text + with patch('ocrmypdf._validation.jbig2enc.available', return_value=False): + vd.report_output_file_size(opts, in_, out) + assert 'No reason' in caplog.text caplog.clear() with patch('ocrmypdf._validation.jbig2enc.available', return_value=False): From 5f5421f23d23070d1e0d63a0b92673c567e94528 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 Nov 2019 01:14:21 -0800 Subject: [PATCH 218/880] test: further fixes to test_report_file_size --- tests/test_validation.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index 7f2ece47..0ac16e24 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -120,12 +120,16 @@ def test_report_file_size(tmp_path, caplog): os.truncate(in_, 25001) os.truncate(out, 50000) - with patch('ocrmypdf._validation.jbig2enc.available', return_value=False): + with patch('ocrmypdf._validation.jbig2enc.available', return_value=True), patch( + 'ocrmypdf._validation.pngquant.available', return_value=True + ): vd.report_output_file_size(opts, in_, out) assert 'No reason' in caplog.text caplog.clear() - with patch('ocrmypdf._validation.jbig2enc.available', return_value=False): + with patch('ocrmypdf._validation.jbig2enc.available', return_value=False), patch( + 'ocrmypdf._validation.pngquant.available', return_value=True + ): vd.report_output_file_size(opts, in_, out) assert 'optional dependency' in caplog.text caplog.clear() From f517efe81921ee05586e57f06f6c82115422f3b2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 Nov 2019 15:01:15 -0800 Subject: [PATCH 219/880] docs: wsl - get-pip.py --- docs/installation.rst | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 46bee8d3..ae29b6a0 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -134,8 +134,7 @@ first install the system version to get most of the dependencies: sudo apt-get update sudo apt-get install \ - ocrmypdf \ - python3-pip + ocrmypdf There are a few system dependency changes since ocrmypdf 6.1.2. Let's get these, too. @@ -146,13 +145,18 @@ get these, too. libxml2 \ pngquant +We will need a newer version of ``pip`` then was available for Ubuntu 18.04: + +.. code-block:: bash + + wget https://bootstrap.pypa.io/get-pip.py && python3 get-pip.py + Then install the most recent ocrmypdf for the local user and set the user's ``PATH`` to check for the user's Python packages. .. code-block:: bash export PATH=$HOME/.local/bin:$PATH - python3 -m pip install --user --upgrade pip python3 -m pip install --user ocrmypdf To add JBIG2 encoding, see :ref:`jbig2`. From 0a08d6ce1f27cb9f6aa428cea25c896e945fcd4b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 13 Nov 2019 01:45:06 -0800 Subject: [PATCH 220/880] Update version of pdfminer.six supported --- requirements/dev.txt | 2 -- requirements/main.txt | 2 +- setup.py | 2 +- 3 files changed, 2 insertions(+), 4 deletions(-) diff --git a/requirements/dev.txt b/requirements/dev.txt index 4faf987d..ab2e5ff6 100644 --- a/requirements/dev.txt +++ b/requirements/dev.txt @@ -1,4 +1,2 @@ -check-manifest >= 0.35 twine >= 1.8.1 coverage >= 4.5 -GitPython == 2.1.3 diff --git a/requirements/main.txt b/requirements/main.txt index c353ed51..2b6f1fca 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,7 +3,7 @@ # installation cffi == 1.13.2 img2pdf == 0.3.3 -pdfminer.six == 20191020 +pdfminer.six == 20191110 pikepdf == 1.7.0 Pillow >= 6.2.0 reportlab == 3.5.32 diff --git a/setup.py b/setup.py index 1a1d026f..7e0302eb 100644 --- a/setup.py +++ b/setup.py @@ -96,7 +96,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20191020', + 'pdfminer.six >= 20181108, <= 20191110', 'pikepdf >= 1.7.0, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 9fb8b267af701d878bff7d6198d110c4ac347199 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 Nov 2019 15:21:45 -0800 Subject: [PATCH 221/880] docker: use get-pip to install pip Smaller download, needed for manylinux2010. --- .docker/Dockerfile | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 95edba47..29de00c6 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -10,16 +10,21 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential autoconf automake libtool \ libleptonica-dev \ zlib1g-dev \ - python3-setuptools \ - python3-pip \ - wget \ + python3 \ + python3-distutils \ + ca-certificates \ + curl \ git +# Get the latest pip (Ubuntu version doesn't support manylinux2010) +RUN \ + curl https://bootstrap.pypa.io/get-pip.py | python3 + # Compile and install jbig2 # Needs libleptonica-dev, zlib1g-dev RUN \ mkdir jbig2 \ - && wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \ + && curl -L https://github.com/agl/jbig2enc/archive/0.29.tar.gz | \ tar xz -C jbig2 --strip-components=1 \ && cd jbig2 \ && ./autogen.sh && ./configure && make && make install \ From b787a369ee1c2c93d6b428f1cab209df7527f5d2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 18 Nov 2019 15:13:42 -0800 Subject: [PATCH 222/880] Fix reference to Alpine apk add --- docs/docker.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 8c356787..392b82d9 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -108,9 +108,9 @@ on the public one: FROM jbarlow83/ocrmypdf # Add French - RUN apk add tesseract-ocr-data-fra + RUN apt install tesseract-ocr-fra -You can also copy training data to ``/usr/share/tessdata``. +You can also copy training data to ``/usr/share/tesseract-ocr//tessdata``. Executing the test suite ======================== From 7691ba8535cf65da2c790f48c9cba69203d05504 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 18 Nov 2019 15:17:00 -0800 Subject: [PATCH 223/880] v9.1.1 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 4681d44a..9f1a23b5 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,13 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.1.1 +====== + +- Expand the range of pdfminer.six versions that are supported. +- Fixed Docker build when using pikepdf 1.7.0. +- Fixed documentation to recommend using pip from get-pip.py. + v9.1.0 ====== From 11afe3507f1daf38c92a123246626ff36ce688ed Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 Nov 2019 14:20:59 -0800 Subject: [PATCH 224/880] black: don't reformat _leptonica.py --- .pre-commit-config.yaml | 6 +++--- pyproject.toml | 1 + 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index b268b628..c86af76e 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,6 @@ repos: -- repo: https://github.com/ambv/black + - repo: https://github.com/psf/black rev: stable hooks: - - id: black - language_version: python3.7 + - id: black + language_version: python3.7 diff --git a/pyproject.toml b/pyproject.toml index 4d61be1a..a28f55c0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,5 +28,6 @@ exclude = ''' | docs | misc | \.egg-info + | src/ocrmypdf/lib/_leptonica.py )/ ''' From 4e4bcaf243c06c0f3f8b8d7c518da565909ae3fc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 Nov 2019 14:38:23 -0800 Subject: [PATCH 225/880] Improve pre-commit checks --- .pre-commit-config.yaml | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index c86af76e..37240976 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -4,3 +4,13 @@ repos: hooks: - id: black language_version: python3.7 + - repo: https://github.com/pre-commit/pre-commit-hooks + rev: v2.4.0 + hooks: + - id: check-case-conflict + - id: check-merge-conflict + - id: check-toml + - id: check-yaml + - id: debug-statements + - id: name-tests-test + args: ["--django"] From ad9a3b530266051fc95e0d55aefed691386ae2ce Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 13 Nov 2019 01:45:06 -0800 Subject: [PATCH 226/880] Update version of pdfminer.six supported --- requirements/dev.txt | 2 -- requirements/main.txt | 2 +- setup.py | 2 +- 3 files changed, 2 insertions(+), 4 deletions(-) diff --git a/requirements/dev.txt b/requirements/dev.txt index 4faf987d..ab2e5ff6 100644 --- a/requirements/dev.txt +++ b/requirements/dev.txt @@ -1,4 +1,2 @@ -check-manifest >= 0.35 twine >= 1.8.1 coverage >= 4.5 -GitPython == 2.1.3 diff --git a/requirements/main.txt b/requirements/main.txt index c353ed51..2b6f1fca 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,7 +3,7 @@ # installation cffi == 1.13.2 img2pdf == 0.3.3 -pdfminer.six == 20191020 +pdfminer.six == 20191110 pikepdf == 1.7.0 Pillow >= 6.2.0 reportlab == 3.5.32 diff --git a/setup.py b/setup.py index 1a1d026f..7e0302eb 100644 --- a/setup.py +++ b/setup.py @@ -96,7 +96,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20191020', + 'pdfminer.six >= 20181108, <= 20191110', 'pikepdf >= 1.7.0, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From b7f63bc93d524cef99f9c445bdded50b091218ac Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 17 Nov 2019 15:40:09 -0800 Subject: [PATCH 227/880] Make devnull check compatible with Windows --- src/ocrmypdf/_sync.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c02543b7..4c25b300 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -304,6 +304,13 @@ class NeverRaise(Exception): pass # pylint: disable=unnecessary-pass +def samefile(f1, f2): + if os.name == 'nt': + return f1 == f2 + else: + return os.path.samefile(f1, f2) + + def run_pipeline(options, api=False): log = make_logger(options, __name__) @@ -339,7 +346,7 @@ def run_pipeline(options, api=False): if options.output_file == '-': log.info("Output sent to stdout") - elif os.path.samefile(options.output_file, os.devnull): + elif samefile(options.output_file, os.devnull): pass # Say nothing when sending to dev null else: if options.output_type.startswith('pdfa'): From 84cc49b14b1cb4c395615dbaef78a38c7ab78dde Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 Nov 2019 14:20:59 -0800 Subject: [PATCH 228/880] black: don't reformat _leptonica.py --- .pre-commit-config.yaml | 7 ++++--- pyproject.toml | 1 + 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index b268b628..2ccebb79 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,7 @@ repos: -- repo: https://github.com/ambv/black + - repo: https://github.com/psf/black rev: stable hooks: - - id: black - language_version: python3.7 + - id: black + language_version: python3.7 + exclude: ^src/ocrmypdf/lib/_leptonica.py diff --git a/pyproject.toml b/pyproject.toml index 4d61be1a..a28f55c0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,5 +28,6 @@ exclude = ''' | docs | misc | \.egg-info + | src/ocrmypdf/lib/_leptonica.py )/ ''' From 17c419dfcb6dd50bbe0310233ae47b60cab401d5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 29 Nov 2019 04:00:40 -0800 Subject: [PATCH 229/880] compile_leptonica: move to correct location --- src/ocrmypdf/lib/compile_leptonica.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index c76dd324..5cfbb6f2 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -491,3 +491,8 @@ ffibuilder.set_source("ocrmypdf.lib._leptonica", None) if __name__ == '__main__': ffibuilder.compile(verbose=True) + if Path('ocrmypdf/lib/_leptonica.py').exists() and Path('src/ocrmypdf').exists(): + output = Path('ocrmypdf/lib/_leptonica.py') + output.rename('src/ocrmypdf/lib/_leptonica.py') + Path('ocrmypdf/lib').rmdir() + Path('ocrmypdf').rmdir() From 72d3ee3a87bbcd4c2bf9ef66d608d40d42513cdb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 17 Nov 2019 15:56:45 -0800 Subject: [PATCH 230/880] Refactor symlink usage to support Windows --- src/ocrmypdf/_pipeline.py | 6 +++--- src/ocrmypdf/_validation.py | 4 ++-- src/ocrmypdf/exec/tesseract.py | 4 ++-- src/ocrmypdf/helpers.py | 12 +++++++++--- src/ocrmypdf/optimize.py | 6 +++--- tests/test_main.py | 1 + 6 files changed, 20 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 56c17e35..445d0faa 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -37,7 +37,7 @@ from .exceptions import ( UnsupportedImageFormatError, ) from .exec import ghostscript, tesseract -from .helpers import re_symlink +from .helpers import safe_symlink from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps @@ -132,7 +132,7 @@ def triage(input_file, output_file, options, log): "input file is a PDF, not an image." ) # Origin file is a pdf create a symlink with pdf extension - re_symlink(input_file, output_file) + safe_symlink(input_file, output_file) return output_file except EnvironmentError as e: log.error(e) @@ -701,7 +701,7 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): if modified: pdf_file.save(fix_docinfo_file) else: - os.symlink(input_pdf, fix_docinfo_file) + safe_symlink(input_pdf, fix_docinfo_file) ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index b9df4b12..b02e4532 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -42,7 +42,7 @@ from .exec import ( tesseract, unpaper, ) -from .helpers import is_file_writable, is_iterable_notstr, monotonic, re_symlink +from .helpers import is_file_writable, is_iterable_notstr, monotonic, safe_symlink # ------------- # External dependencies @@ -374,7 +374,7 @@ def create_input_file(options, work_folder): else: try: target = os.path.join(work_folder, 'origin') - re_symlink(options.input_file, target) + safe_symlink(options.input_file, target) return target except FileNotFoundError: raise InputFileError(f"File not found - {options.input_file}") diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index afa0e4a6..cb837dfd 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -30,7 +30,7 @@ from ..exceptions import ( SubprocessOutputError, TesseractConfigError, ) -from ..helpers import page_number +from ..helpers import page_number, safe_symlink from . import get_version OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) @@ -324,7 +324,7 @@ def use_skip_page(text_only, skip_pdf, output_pdf, output_text): # Substitute a "skipped page" with suppress(FileNotFoundError): os.remove(output_pdf) # In case it was partially created - os.symlink(skip_pdf, output_pdf) + safe_symlink(skip_pdf, output_pdf) return # Or normally, just write a 0 byte file to the output to indicate a skip diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 80eff55f..b719d4e7 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -18,6 +18,7 @@ import logging import multiprocessing import os +import shutil import warnings from collections.abc import Iterable from contextlib import suppress @@ -27,14 +28,14 @@ from pathlib import Path log = logging.getLogger(__name__) -def re_symlink(input_file, soft_link_name, *args, **kwargs): +def safe_symlink(input_file, soft_link_name, *args, **kwargs): """ Helper function: relinks soft symbolic link if necessary """ if len(args) == 1 and isinstance(args[0], logging.Logger): - log.warning("Deprecated: re_symlink(,log)") + log.warning("Deprecated: safe_symlink(,log)") if 'log' in kwargs: - log.warning('Deprecated: re_symlink(...log=)') + log.warning('Deprecated: safe_symlink(...log=)') input_file = os.fspath(input_file) soft_link_name = os.fspath(soft_link_name) @@ -60,6 +61,11 @@ def re_symlink(input_file, soft_link_name, *args, **kwargs): if not os.path.exists(input_file): raise FileNotFoundError(f"trying to create a broken symlink to {input_file}") + if os.name == 'nt': + # Don't actually use symlinks on Windows due to permission issues + shutil.copyfile(input_file, soft_link_name) + return + log.debug("os.symlink(%s, %s)", input_file, soft_link_name) # Create symbolic link using absolute path diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index bb1269db..72022741 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -31,7 +31,7 @@ from . import leptonica from ._jobcontext import PDFContext from .exceptions import OutputFileAccessError from .exec import jbig2enc, pngquant -from .helpers import re_symlink +from .helpers import safe_symlink DEFAULT_JPEG_QUALITY = 75 DEFAULT_PNG_QUALITY = 70 @@ -492,7 +492,7 @@ def optimize(input_file, output_file, context, save_settings): log = context.log options = context.options if options.optimize == 0: - re_symlink(input_file, output_file) + safe_symlink(input_file, output_file) return if options.jpeg_quality == 0: @@ -538,7 +538,7 @@ def optimize(input_file, output_file, context, save_settings): pike.remove_unreferenced_resources() pike.save(output_file, **save_settings) else: - re_symlink(target_file, output_file) + safe_symlink(target_file, output_file) def main(infile, outfile, level, jobs=1): diff --git a/tests/test_main.py b/tests/test_main.py index 79625d12..c55186b0 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -946,6 +946,7 @@ def test_output_is_dir(spoof_tesseract_noop, resources, outdir): assert 'is not a writable file' in err +@pytest.mark.skipif(os.name == 'nt', reason="symlink needs admin permissions") def test_output_is_symlink(spoof_tesseract_noop, resources, outdir): sym = Path(outdir / 'this_is_a_symlink') sym.symlink_to(outdir / 'out.pdf') From d5bb9929f390fc8d2607e3ce1588ad6a969496c8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 17 Nov 2019 15:41:48 -0800 Subject: [PATCH 231/880] leptonica: Use Windows name for DLL Thanks to @dibu28 --- src/ocrmypdf/leptonica.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 1cdec312..ef2a2b82 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -39,7 +39,11 @@ from .lib._leptonica import ffi logger = logging.getLogger(__name__) -lept = ffi.dlopen(find_library('lept')) +if os.name == 'nt': + libname = 'liblept-5' +else: + libname = 'lept' +lept = ffi.dlopen(find_library(libname)) lept.setMsgSeverity(lept.L_SEVERITY_WARNING) From 9baccee8c5c7fdbbef0a7b48b81bc5ab285ea94c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 20 Nov 2019 00:29:48 -0800 Subject: [PATCH 232/880] leptonica: Handle API change for pixFindPageForeground --- src/ocrmypdf/leptonica.py | 13 +++++------- src/ocrmypdf/lib/compile_leptonica.py | 30 +++++++++++++++++++-------- tests/test_lept.py | 4 ++++ 3 files changed, 30 insertions(+), 17 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index ef2a2b82..831a0695 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -506,17 +506,14 @@ class Pix(LeptonicaObject): display=0, pdfdir=ffi.NULL, ): + if get_leptonica_version() < 'leptonica-1.76': + # Leptonica 1.76 changed the API for pixFindPageForeground; we don't + # support the old version + raise LeptonicaError("Not available in this version of Leptonica") with _LeptonicaErrorTrap(): cropbox = Box( lept.pixFindPageForeground( - self._cdata, - threshold, - mindist, - erasedist, - pagenum, - showmorph, - display, - pdfdir, + self._cdata, threshold, mindist, erasedist, showmorph, ffi.NULL ) ) diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index 5cfbb6f2..e6d3c9f3 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -74,6 +74,17 @@ struct Pixa }; typedef struct Pixa PIXA; +/*! Array of compressed pix */ +struct PixaComp +{ + l_int32 n; /*!< number of PixComp in ptr array */ + l_int32 nalloc; /*!< number of PixComp ptrs allocated */ + l_int32 offset; /*!< indexing offset into ptr array */ + struct PixComp **pixc; /*!< the array of ptrs to PixComp */ + struct Boxa *boxa; /*!< array of boxes */ +}; +typedef struct PixaComp PIXAC; + struct Box { l_int32 x; @@ -294,14 +305,12 @@ pixCleanBackgroundToWhite(PIX *pixs, l_int32 whiteval); BOX * -pixFindPageForeground(PIX *pixs, - l_int32 threshold, - l_int32 mindist, - l_int32 erasedist, - l_int32 pagenum, - l_int32 showmorph, - l_int32 display, - const char *pdfdir); +pixFindPageForeground ( PIX *pixs, + l_int32 threshold, + l_int32 mindist, + l_int32 erasedist, + l_int32 showmorph, + PIXAC *pixac ); PIX * pixClipRectangle(PIX *pixs, @@ -414,7 +423,10 @@ pixExtractBarcodes(PIX *pixs, l_int32 debugflag); BOXA * -pixLocateBarcodes ( PIX *pixs, l_int32 thresh, PIX **ppixb, PIX **ppixm ); +pixLocateBarcodes ( PIX *pixs, + l_int32 thresh, + PIX **ppixb, + PIX **ppixm ); SARRAY * pixReadBarcodes(PIXA *pixa, diff --git a/tests/test_lept.py b/tests/test_lept.py index 2504c600..b74f417b 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -63,6 +63,10 @@ def test_pix_otsu(crom_pix): assert im1bpp.mode == '1' +@pytest.mark.skipif( + lept.get_leptonica_version() < 'leptonica-1.76', + reason="needs new leptonica for API change", +) def test_crop(resources): pix = lept.Pix.open(resources / 'linn.png') foreground = pix.crop_to_foreground() From fe7c69ce95639dfb150916671c8b2780cfe1badb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 14:41:37 -0800 Subject: [PATCH 233/880] leptonica: don't open files by name; use memory buffers Avoids encoding issues and makes error trap unnecessary in some cases. --- src/ocrmypdf/leptonica.py | 26 ++++++++++++++++++-------- src/ocrmypdf/lib/_leptonica.py | 10 +++++----- src/ocrmypdf/lib/compile_leptonica.py | 8 ++++++++ tests/test_lept.py | 16 +++------------- 4 files changed, 34 insertions(+), 26 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 831a0695..ca0c546e 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -296,9 +296,11 @@ class Pix(LeptonicaObject): Leptonica can load TIFF, PNM (PBM, PGM, PPM), PNG, and JPEG. If loading fails then the object will wrap a C null pointer. """ - filename = fspath(path) - with _LeptonicaErrorTrap(): - return cls(lept.pixRead(os.fsencode(filename))) + with open(path, 'rb') as py_file: + data = py_file.read() + buffer = ffi.from_buffer(data) + with _LeptonicaErrorTrap(): + return cls(lept.pixReadMem(buffer, len(buffer))) def write_implied_format(self, path, jpeg_quality=0, jpeg_progressive=0): """Write pix to the filename, with the extension indicating format. @@ -306,11 +308,19 @@ class Pix(LeptonicaObject): jpeg_quality -- quality (iff JPEG; 1 - 100, 0 for default) jpeg_progressive -- (iff JPEG; 0 for baseline seq., 1 for progressive) """ - filename = fspath(path) - with _LeptonicaErrorTrap(): - lept.pixWriteImpliedFormat( - os.fsencode(filename), self._cdata, jpeg_quality, jpeg_progressive - ) + lept_format = lept.getImpliedFileFormat(os.fsencode(path)) + with open(path, 'wb') as py_file: + data = ffi.new('l_uint8 **pdata') + size = ffi.new('size_t *psize') + with _LeptonicaErrorTrap(): + if lept_format == lept.L_JPEG_ENCODE: + lept.pixWriteMemJpeg( + data, size, self._cdata, jpeg_quality, jpeg_progressive + ) + else: + lept.pixWriteMem(data, size, self._cdata, lept_format) + buffer = ffi.buffer(data[0], size[0]) + py_file.write(buffer) @classmethod def frompil(self, pillow_image): diff --git a/src/ocrmypdf/lib/_leptonica.py b/src/ocrmypdf/lib/_leptonica.py index 17a2f757..549098c1 100644 --- a/src/ocrmypdf/lib/_leptonica.py +++ b/src/ocrmypdf/lib/_leptonica.py @@ -3,9 +3,9 @@ import _cffi_backend ffi = _cffi_backend.FFI('ocrmypdf.lib._leptonica', _version = 0x2601, - _types = b'\x00\x00\x01\x0D\x00\x01\x33\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x34\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x37\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x3F\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x38\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x1A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x3C\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x05\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x60\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x13\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x10\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x4E\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x50\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x13\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x3A\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x3A\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x3A\x0D\x00\x00\x13\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x9C\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x45\x0D\x00\x00\x10\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x10\x11\x00\x00\x00\x0F\x00\x00\x45\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x3E\x0D\x00\x00\x45\x11\x00\x00\x00\x0F\x00\x01\x3E\x0D\x00\x00\x00\x0F\x00\x00\x60\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x32\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x60\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xC3\x11\x00\x00\xC3\x11\x00\x00\xC3\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xC3\x11\x00\x00\xC3\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x60\x11\x00\x00\x60\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x60\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x35\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xC3\x11\x00\x00\xC3\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x1A\x11\x00\x00\x1A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x4F\x03\x00\x00\x8E\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x10\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xEC\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x10\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x4D\x03\x00\x01\x04\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x23\x11\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\xEC\x11\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x1A\x11\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x13\x03\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x9C\x11\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x00\x45\x03\x00\x00\x00\x0F\x00\x01\x53\x0D\x00\x01\x53\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x01\x36\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x05\x09\x00\x00\x04\x09\x00\x01\x3B\x03\x00\x00\x06\x09\x00\x00\x07\x09\x00\x01\x3E\x03\x00\x01\x3F\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x60\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x39\x03\x00\x01\x4E\x03\x00\x00\x04\x01\x00\x01\x50\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', - _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x1B\x23boxDestroy',0,b'\x00\x01\x1E\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x00\xB6\x23getLeptonicaVersion',0,b'\x00\x01\x21\x23l_CIDataDestroy',0,b'\x00\x01\x06\x23l_generateCIDataForPdf',0,b'\x00\x01\x30\x23lept_free',0,b'\x00\x00\xB8\x23makePixelSumTab8',0,b'\x00\x00\x29\x23pixAnd',0,b'\x00\x00\x36\x23pixBackgroundNorm',0,b'\x00\x00\x2E\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x20\x23pixClipRectangle',0,b'\x00\x00\xEE\x23pixColorFraction',0,b'\x00\x00\x7D\x23pixColorMagnitude',0,b'\x00\x00\x1D\x23pixConvertRGBToLuminance',0,b'\x00\x00\x74\x23pixConvertTo8',0,b'\x00\x00\xC0\x23pixCorrelationBinary',0,b'\x00\x00\xDA\x23pixCountPixels',0,b'\x00\x00\x90\x23pixDeserializeFromMemory',0,b'\x00\x00\x74\x23pixDeskew',0,b'\x00\x01\x24\x23pixDestroy',0,b'\x00\x00\x42\x23pixDilate',0,b'\x00\x00\x1D\x23pixEndianByteSwapNew',0,b'\x00\x00\xC5\x23pixEqual',0,b'\x00\x00\x42\x23pixErode',0,b'\x00\x00\x94\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xD5\x23pixFindSkew',0,b'\x00\x00\x47\x23pixGammaTRC',0,b'\x00\x00\xE7\x23pixGenerateCIData',0,b'\x00\x00\xCA\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x4E\x23pixGlobalNormRGB',0,b'\x00\x00\x42\x23pixHMT',0,b'\x00\x00\x25\x23pixInvert',0,b'\x00\x00\x17\x23pixLocateBarcodes',0,b'\x00\x00\x78\x23pixMaskOverColorPixels',0,b'\x00\x00\x56\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xDF\x23pixNumSignificantGrayColors',0,b'\x00\x00\xF7\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x62\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\x98\x23pixProcessBarcodes',0,b'\x00\x00\x89\x23pixRead',0,b'\x00\x00\x9F\x23pixReadBarcodes',0,b'\x00\x00\x8C\x23pixReadMem',0,b'\x00\x00\x74\x23pixRemoveColormap',0,b'\x00\x00\x78\x23pixRemoveColormapGeneral',0,b'\x00\x00\xBA\x23pixRenderBoxa',0,b'\x00\x00\x25\x23pixRotate180',0,b'\x00\x00\x74\x23pixRotateOrth',0,b'\x00\x00\x6F\x23pixScale',0,b'\x00\x01\x01\x23pixSerializeToMemory',0,b'\x00\x00\x29\x23pixSubtract',0,b'\x00\x01\x0C\x23pixWriteImpliedFormat',0,b'\x00\x01\x15\x23pixWriteMemPng',0,b'\x00\x01\x27\x23pixaDestroy',0,b'\x00\x00\x12\x23pixaGetBox',0,b'\x00\x00\x84\x23pixaGetPix',0,b'\x00\x01\x2A\x23sarrayDestroy',0,b'\x00\x00\xAC\x23selCreateBrick',0,b'\x00\x00\xA6\x23selCreateFromString',0,b'\x00\x01\x2D\x23selDestroy',0,b'\x00\x00\xB3\x23selPrintToString',0,b'\x00\x01\x12\x23setMsgSeverity',0), - _struct_unions = ((b'\x00\x00\x01\x33\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x50\x11refcount'),(b'\x00\x00\x01\x34\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x50\x11refcount',b'\x00\x00\x23\x11box'),(b'\x00\x00\x01\x36\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x4D\x11datacomp',b'\x00\x00\x8E\x11nbytescomp',b'\x00\x01\x3E\x11data85',b'\x00\x00\x8E\x11nbytes85',b'\x00\x01\x3E\x11cmapdata85',b'\x00\x01\x3E\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x8E\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x37\x00\x00\x00\x02Pix',b'\x00\x01\x50\x11w',b'\x00\x01\x50\x11h',b'\x00\x01\x50\x11d',b'\x00\x01\x50\x11spp',b'\x00\x01\x50\x11wpl',b'\x00\x01\x50\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x3E\x11text',b'\x00\x01\x4C\x11colormap',b'\x00\x01\x4F\x11data'),(b'\x00\x00\x01\x39\x00\x00\x00\x02PixColormap',b'\x00\x01\x31\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x38\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x50\x11refcount',b'\x00\x00\x1A\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x3B\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x3D\x11array'),(b'\x00\x00\x01\x3C\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x48\x11data',b'\x00\x01\x3E\x11name')), - _enums = (b'\x00\x00\x01\x41\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x42\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x43\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x44\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x45\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x46\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x47\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), - _typenames = (b'\x00\x00\x01\x33BOX',b'\x00\x00\x01\x34BOXA',b'\x00\x00\x01\x36L_COMP_DATA',b'\x00\x00\x01\x37PIX',b'\x00\x00\x01\x38PIXA',b'\x00\x00\x01\x39PIXCMAP',b'\x00\x00\x01\x3BSARRAY',b'\x00\x00\x01\x3CSEL',b'\x00\x00\x00\x32l_float32',b'\x00\x00\x01\x40l_float64',b'\x00\x00\x01\x4Al_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x49l_int64',b'\x00\x00\x01\x4Bl_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x52l_uint16',b'\x00\x00\x01\x50l_uint32',b'\x00\x00\x01\x51l_uint64',b'\x00\x00\x01\x4El_uint8'), + _types = b'\x00\x00\x01\x0D\x00\x01\x50\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x51\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x55\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x57\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x56\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x52\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x5B\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x05\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x5E\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x70\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x72\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x11\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x59\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x59\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x59\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x9E\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x47\x0D\x00\x00\x8C\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x8C\x11\x00\x00\x00\x0F\x00\x00\x47\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5D\x0D\x00\x00\x47\x11\x00\x00\x00\x0F\x00\x01\x5D\x0D\x00\x00\x00\x0F\x00\x00\x62\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x34\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x62\x11\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x53\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x18\x11\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x71\x03\x00\x00\x90\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xF9\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x6F\x03\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x26\x11\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x26\x11\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x25\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\xF9\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x11\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x9E\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x47\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x01\x75\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x00\x0A\x09\x00\x01\x54\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x06\x09\x00\x00\x07\x09\x00\x00\x04\x09\x00\x01\x5A\x03\x00\x00\x08\x09\x00\x00\x09\x09\x00\x01\x5D\x03\x00\x01\x5E\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x62\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x58\x03\x00\x01\x6D\x03\x00\x01\x6E\x03\x00\x00\x05\x09\x00\x01\x70\x03\x00\x00\x04\x01\x00\x01\x72\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', + _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x38\x23boxDestroy',0,b'\x00\x01\x3B\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x01\x13\x23getImpliedFileFormat',0,b'\x00\x00\xB8\x23getLeptonicaVersion',0,b'\x00\x01\x3E\x23l_CIDataDestroy',0,b'\x00\x01\x16\x23l_generateCIDataForPdf',0,b'\x00\x01\x4D\x23lept_free',0,b'\x00\x00\xBA\x23makePixelSumTab8',0,b'\x00\x00\x2B\x23pixAnd',0,b'\x00\x00\x38\x23pixBackgroundNorm',0,b'\x00\x00\x30\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x22\x23pixClipRectangle',0,b'\x00\x00\xFB\x23pixColorFraction',0,b'\x00\x00\x7F\x23pixColorMagnitude',0,b'\x00\x00\x1F\x23pixConvertRGBToLuminance',0,b'\x00\x00\x76\x23pixConvertTo8',0,b'\x00\x00\xCD\x23pixCorrelationBinary',0,b'\x00\x00\xE7\x23pixCountPixels',0,b'\x00\x00\x92\x23pixDeserializeFromMemory',0,b'\x00\x00\x76\x23pixDeskew',0,b'\x00\x01\x41\x23pixDestroy',0,b'\x00\x00\x44\x23pixDilate',0,b'\x00\x00\x1F\x23pixEndianByteSwapNew',0,b'\x00\x00\xD2\x23pixEqual',0,b'\x00\x00\x44\x23pixErode',0,b'\x00\x00\x96\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xE2\x23pixFindSkew',0,b'\x00\x00\x49\x23pixGammaTRC',0,b'\x00\x00\xF4\x23pixGenerateCIData',0,b'\x00\x00\xD7\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x50\x23pixGlobalNormRGB',0,b'\x00\x00\x44\x23pixHMT',0,b'\x00\x00\x27\x23pixInvert',0,b'\x00\x00\x15\x23pixLocateBarcodes',0,b'\x00\x00\x7A\x23pixMaskOverColorPixels',0,b'\x00\x00\x58\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xEC\x23pixNumSignificantGrayColors',0,b'\x00\x01\x04\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x64\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\x9A\x23pixProcessBarcodes',0,b'\x00\x00\x8B\x23pixRead',0,b'\x00\x00\xA1\x23pixReadBarcodes',0,b'\x00\x00\x8E\x23pixReadMem',0,b'\x00\x00\x1B\x23pixReadStream',0,b'\x00\x00\x76\x23pixRemoveColormap',0,b'\x00\x00\x7A\x23pixRemoveColormapGeneral',0,b'\x00\x00\xC7\x23pixRenderBoxa',0,b'\x00\x00\x27\x23pixRotate180',0,b'\x00\x00\x76\x23pixRotateOrth',0,b'\x00\x00\x71\x23pixScale',0,b'\x00\x01\x0E\x23pixSerializeToMemory',0,b'\x00\x00\x2B\x23pixSubtract',0,b'\x00\x01\x1C\x23pixWriteImpliedFormat',0,b'\x00\x01\x2B\x23pixWriteMem',0,b'\x00\x01\x31\x23pixWriteMemJpeg',0,b'\x00\x01\x25\x23pixWriteMemPng',0,b'\x00\x00\xBC\x23pixWriteStream',0,b'\x00\x00\xC1\x23pixWriteStreamJpeg',0,b'\x00\x01\x44\x23pixaDestroy',0,b'\x00\x00\x10\x23pixaGetBox',0,b'\x00\x00\x86\x23pixaGetPix',0,b'\x00\x01\x47\x23sarrayDestroy',0,b'\x00\x00\xAE\x23selCreateBrick',0,b'\x00\x00\xA8\x23selCreateFromString',0,b'\x00\x01\x4A\x23selDestroy',0,b'\x00\x00\xB5\x23selPrintToString',0,b'\x00\x01\x22\x23setMsgSeverity',0), + _struct_unions = ((b'\x00\x00\x01\x50\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x72\x11refcount'),(b'\x00\x00\x01\x51\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x72\x11refcount',b'\x00\x00\x25\x11box'),(b'\x00\x00\x01\x54\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x6F\x11datacomp',b'\x00\x00\x90\x11nbytescomp',b'\x00\x01\x5D\x11data85',b'\x00\x00\x90\x11nbytes85',b'\x00\x01\x5D\x11cmapdata85',b'\x00\x01\x5D\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x90\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x55\x00\x00\x00\x02Pix',b'\x00\x01\x72\x11w',b'\x00\x01\x72\x11h',b'\x00\x01\x72\x11d',b'\x00\x01\x72\x11spp',b'\x00\x01\x72\x11wpl',b'\x00\x01\x72\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x5D\x11text',b'\x00\x01\x6B\x11colormap',b'\x00\x01\x71\x11data'),(b'\x00\x00\x01\x58\x00\x00\x00\x02PixColormap',b'\x00\x01\x4E\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x6E\x00\x00\x00\x10PixComp',),(b'\x00\x00\x01\x56\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x72\x11refcount',b'\x00\x00\x18\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x57\x00\x00\x00\x02PixaComp',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11offset',b'\x00\x01\x6C\x11pixc',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x5A\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x5C\x11array'),(b'\x00\x00\x01\x5B\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x67\x11data',b'\x00\x01\x5D\x11name'),(b'\x00\x00\x01\x52\x00\x00\x00\x10_IO_FILE',)), + _enums = (b'\x00\x00\x01\x60\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x61\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x62\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x63\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x64\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x65\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x66\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), + _typenames = (b'\x00\x00\x01\x50BOX',b'\x00\x00\x01\x51BOXA',b'\x00\x00\x01\x52FILE',b'\x00\x00\x01\x54L_COMP_DATA',b'\x00\x00\x01\x55PIX',b'\x00\x00\x01\x56PIXA',b'\x00\x00\x01\x57PIXAC',b'\x00\x00\x01\x58PIXCMAP',b'\x00\x00\x01\x5ASARRAY',b'\x00\x00\x01\x5BSEL',b'\x00\x00\x00\x34l_float32',b'\x00\x00\x01\x5Fl_float64',b'\x00\x00\x01\x69l_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x68l_int64',b'\x00\x00\x01\x6Al_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x74l_uint16',b'\x00\x00\x01\x72l_uint32',b'\x00\x00\x01\x73l_uint64',b'\x00\x00\x01\x70l_uint8'), ) diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index e6d3c9f3..4d50943d 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -16,6 +16,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +from pathlib import Path + from cffi import FFI ffibuilder = FFI() @@ -221,9 +223,15 @@ ffibuilder.cdef( """ PIX * pixRead ( const char *filename ); PIX * pixReadMem ( const l_uint8 *data, size_t size ); +PIX * pixReadStream ( FILE *fp, l_int32 hint ); PIX * pixScale ( PIX *pixs, l_float32 scalex, l_float32 scaley ); l_int32 pixFindSkew ( PIX *pixs, l_float32 *pangle, l_float32 *pconf ); l_int32 pixWriteImpliedFormat ( const char *filename, PIX *pix, l_int32 quality, l_int32 progressive ); +l_int32 getImpliedFileFormat ( const char *filename ); +l_ok pixWriteStream ( FILE *fp, PIX *pix, l_int32 format ); +l_ok pixWriteStreamJpeg ( FILE *fp, PIX *pixs, l_int32 quality, l_int32 progressive ); +l_ok pixWriteMem ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 format ); +l_ok pixWriteMemJpeg ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 quality, l_int32 progressive ); l_int32 pixWriteMemPng(l_uint8 **pdata, size_t *psize, diff --git a/tests/test_lept.py b/tests/test_lept.py index b74f417b..116060c3 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -94,16 +94,6 @@ def test_leptonica_compile(tmp_path): ffibuilder.compile(tmpdir=fspath(tmp_path), target=fspath(tmp_path / 'lepttest.*')) -def test_with_stderr(capsys): - # pytest redirects stderr too; we must disable this for the test to be valid - with capsys.disabled(): - with pytest.raises(FileNotFoundError): - lept.Pix.open("does_not_exist1") - - -def test_without_stderr(capsys): - # pytest redirects stderr too; we must disable this for the test to be valid - with capsys.disabled(): - with patch('sys.stderr', new=None): - with pytest.raises(FileNotFoundError): - lept.Pix.open("does_not_exist2") +def test_file_not_found(): + with pytest.raises(FileNotFoundError): + lept.Pix.open("does_not_exist1") From 17d20309c7f06baec8a53b3cefd8fda961bfa804 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 Nov 2019 14:17:32 -0800 Subject: [PATCH 234/880] leptonica: fix missing Leptonica error message for Windows Since it has the unintuitive fix of adding Tesseract to PATH. --- src/ocrmypdf/leptonica.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index ca0c546e..6fbf2d99 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -34,6 +34,7 @@ from os import fspath from tempfile import TemporaryFile from .lib._leptonica import ffi +from .exceptions import MissingDependencyError # pylint: disable=protected-access @@ -43,7 +44,12 @@ if os.name == 'nt': libname = 'liblept-5' else: libname = 'lept' -lept = ffi.dlopen(find_library(libname)) +_libpath = find_library(libname) +if not _libpath and os.name == 'nt': + raise MissingDependencyError( + "Please ensure that 'tesseract' is on your PATH environment variable. " + ) +lept = ffi.dlopen(_libpath) lept.setMsgSeverity(lept.L_SEVERITY_WARNING) From e63503d64bbd4505a802509c7ee0eb13e7fbb866 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 27 Nov 2019 01:12:21 -0800 Subject: [PATCH 235/880] Fix difference in Windows error message breaking test_no_languages --- src/ocrmypdf/exec/tesseract.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index cb837dfd..4b0560d8 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -127,9 +127,10 @@ def languages(tesseract_env=None): except CalledProcessError as e: raise MissingDependencyError(lang_error(e.output)) from e + for line in output.splitlines(): + if line.startswith('Error'): + raise MissingDependencyError(lang_error(output)) header, *rest = output.splitlines() - if not header.startswith('List of available languages'): - raise MissingDependencyError(lang_error(output)) return set(lang.strip() for lang in rest) From 3f92867ae678da3270722bc24e4ec9b3b26383c4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 19 Nov 2019 18:01:10 -0800 Subject: [PATCH 236/880] Fix TypeError "environment can only contain strings" Apparently Windows Python doesn't coerce pathlib.Path to str. --- src/ocrmypdf/api.py | 2 +- tests/conftest.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index b735f398..a05edbc1 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -158,7 +158,7 @@ def create_options(*, input_file, output_file, **kwargs): # If we are running a Tesseract spoof, ensure it knows what the input file is if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env: - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) return options diff --git a/tests/conftest.py b/tests/conftest.py index b443ac1f..e875315a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -183,7 +183,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): api.check_options(options) if env: options.tesseract_env = env - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) result = api.run_pipeline(options, api=True) assert result == 0 From 37f6f72df3aad5e24fb1db1792371ef87ed038a5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 19 Nov 2019 18:07:33 -0800 Subject: [PATCH 237/880] tests: a few Windows fixes --- tests/test_main.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index c55186b0..a2ba24cb 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -586,7 +586,7 @@ def test_overlay(spoof_tesseract_noop, resources, outpdf): def test_destination_not_writable(spoof_tesseract_noop, resources, outdir): - if os.getuid() == 0 or os.geteuid() == 0: + if os.name != 'nt' and (os.getuid() == 0 or os.geteuid() == 0): pytest.xfail(reason="root can write to anything") protected_file = outdir / 'protected.pdf' protected_file.touch() @@ -872,7 +872,7 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): pdfinfo = PdfInfo(resources / 'multipage.pdf') num_pages = len(pdfinfo) - with open(sidecar, 'r') as f: + with open(sidecar, 'r', encoding='utf-8') as f: ocr_text = f.read() # There should a formfeed between each pair of pages, so the count of @@ -888,7 +888,7 @@ def test_sidecar_nonempty(spoof_tesseract_cache, resources, outpdf): resources / 'ccitt.pdf', outpdf, '--sidecar', sidecar, env=spoof_tesseract_cache ) - with open(sidecar, 'r') as f: + with open(sidecar, 'r', encoding='utf-8') as f: ocr_text = f.read() assert 'the' in ocr_text From 4ab0a8ff35a5a96728ac6b6cca0e711b6c640d05 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 27 Nov 2019 02:26:13 -0800 Subject: [PATCH 238/880] Fix test_single_page_inline_image - remove temp file --- tests/test_pdfinfo.py | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 81669a62..a775e950 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -91,21 +91,21 @@ def test_single_page_image(outdir): def test_single_page_inline_image(outdir): filename = outdir / 'image-mono-inline.pdf' pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) - with NamedTemporaryFile() as im_tmp: - im = Image.new('1', (8, 8), 0) - for n in range(8): - im.putpixel((n, n), 1) - im.save(im_tmp.name, format='PNG') - # Draw image in a 72x72 pt or 1"x1" area - pdf.drawInlineImage(im_tmp.name, 0, 0, width=72, height=72) - pdf.showPage() - pdf.save() - pdf = pdfinfo.PdfInfo(filename) - print(pdf) - pdfimage = pdf[0].images[0] + im = Image.new('1', (8, 8), 0) + for n in range(8): + im.putpixel((n, n), 1) + + # Draw image in a 72x72 pt or 1"x1" area + pdf.drawInlineImage(im, 0, 0, width=72, height=72) + pdf.showPage() + pdf.save() + + info = pdfinfo.PdfInfo(filename) + print(info) + pdfimage = info[0].images[0] assert isclose(pdfimage.xres, 8) - assert pdfimage.color == Colorspace.rgb # reportlab produces color image + assert pdfimage.color == Colorspace.gray assert pdfimage.width == 8 From a3726e4ce3ba092b8f9981814fa690aa89ca6670 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 27 Nov 2019 02:34:53 -0800 Subject: [PATCH 239/880] Fix test_metadata: use mmap in a Windows and POSIX compatible way --- tests/test_metadata.py | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index e0650a34..6d33db12 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -21,6 +21,7 @@ from datetime import timezone import logging import mmap from os import fspath +import os from pathlib import Path from shutil import copyfile, move from unittest.mock import MagicMock, patch @@ -330,10 +331,8 @@ def test_prevent_gs_invalid_xml(resources, outdir): str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context ) - with open(outdir / 'pdfa.pdf', 'rb') as f: - with mmap.mmap( - f.fileno(), 0, flags=mmap.MAP_PRIVATE, prot=mmap.PROT_READ - ) as mm: + with open(outdir / 'pdfa.pdf', 'r+b') as f: + with mmap.mmap(f.fileno(), 0) as mm: # Since the XML may be invalid, we scan instead of actually feeding it # to a parser. XMP_MAGIC = b'W5M0MpCehiHzreSzNTczkc9d' From fde550f9a708d62bde595b4a056cac5d51f3f1a6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 14:03:18 -0800 Subject: [PATCH 240/880] test: Replace many instances of run_ocrmypdf in subprocess with inline --- tests/conftest.py | 16 ++++++++++ tests/test_main.py | 69 +++++++++++++++++++++--------------------- tests/test_stdio.py | 1 + tests/test_userunit.py | 8 +++-- 4 files changed, 57 insertions(+), 37 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index e875315a..1a63a5d2 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -193,6 +193,22 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): return output_file +@pytest.helpers.register +def run_ocrmypdf_api(input_file, output_file, *args, env=None): + "Run ocrmypdf and let caller deal with results" + + options = cli.parser.parse_args( + [str(input_file), str(output_file)] + + [str(arg) for arg in args if arg is not None] + ) + api.check_options(options) + if env: + options.tesseract_env = env + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + + return api.run_pipeline(options, api=False) + + @pytest.helpers.register def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=True): "Run ocrmypdf and let caller deal with results" diff --git a/tests/test_main.py b/tests/test_main.py index a2ba24cb..77ea154e 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -39,6 +39,7 @@ from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api spoof = pytest.helpers.spoof @@ -197,8 +198,8 @@ def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): def test_repeat_ocr(resources, no_outpdf): - p, _, _ = run_ocrmypdf(resources / 'graph_ocred.pdf', no_outpdf) - assert p.returncode != 0 + result = run_ocrmypdf_api(resources / 'graph_ocred.pdf', no_outpdf) + assert result == ExitCode.already_done_ocr def test_force_ocr(spoof_tesseract_cache, resources, outpdf): @@ -300,34 +301,34 @@ def test_maximum_options( ) -def test_tesseract_missing_tessdata(resources, no_outpdf): +def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir): env = os.environ.copy() - env['TESSDATA_PREFIX'] = '/tmp' + env['TESSDATA_PREFIX'] = tmpdir - p, _, err = run_ocrmypdf( - resources / 'graph_ocred.pdf', no_outpdf, '-v', '1', '--skip-text', env=env + returncode = run_ocrmypdf_api( + resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env ) - assert p.returncode == ExitCode.missing_dependency, err + assert returncode == ExitCode.missing_dependency def test_invalid_input_pdf(resources, no_outpdf): - p, out, err = run_ocrmypdf(resources / 'invalid.pdf', no_outpdf) - assert p.returncode == ExitCode.input_file, err + result = run_ocrmypdf_api(resources / 'invalid.pdf', no_outpdf) + assert result == ExitCode.input_file def test_blank_input_pdf(resources, outpdf): - p, out, err = run_ocrmypdf(resources / 'blank.pdf', outpdf) - assert p.returncode == ExitCode.ok + result = run_ocrmypdf_api(resources / 'blank.pdf', outpdf) + assert result == ExitCode.ok def test_force_ocr_on_pdf_with_no_images(spoof_tesseract_crash, resources, no_outpdf): # As a correctness test, make sure that --force-ocr on a PDF with no # content still triggers tesseract. If tesseract crashes, then it was # called. - p, _, err = run_ocrmypdf( + result = run_ocrmypdf_api( resources / 'blank.pdf', no_outpdf, '--force-ocr', env=spoof_tesseract_crash ) - assert p.returncode == ExitCode.child_process_error, err + assert result == ExitCode.child_process_error assert not os.path.exists(no_outpdf) @@ -340,7 +341,7 @@ def test_german(spoof_tesseract_cache, resources, outdir): # properly. It is fine that we are testing -l deu on a French file because # we are exercising the functionality not going for accuracy. sidecar = outdir / 'francais.txt' - p, out, err = run_ocrmypdf( + result = run_ocrmypdf_api( resources / 'francais.pdf', outdir / 'francais.pdf', '-l', @@ -351,16 +352,16 @@ def test_german(spoof_tesseract_cache, resources, outdir): ) if 'deu' not in tesseract.languages(): pytest.xfail(reason="tesseract-deu language pack not installed") - assert p.returncode == ExitCode.ok, "Requires tesseract deu language pack" + assert result == ExitCode.ok, "Requires tesseract deu language pack" def test_klingon(resources, outpdf): - p, out, err = run_ocrmypdf(resources / 'francais.pdf', outpdf, '-l', 'klz') + p, _, _ = run_ocrmypdf(resources / 'francais.pdf', outpdf, '-l', 'klz') assert p.returncode == ExitCode.missing_dependency def test_missing_docinfo(spoof_tesseract_noop, resources, outpdf): - p, out, err = run_ocrmypdf( + result = run_ocrmypdf_api( resources / 'missing_docinfo.pdf', outpdf, '-l', @@ -368,7 +369,7 @@ def test_missing_docinfo(spoof_tesseract_noop, resources, outpdf): '--skip-text', env=spoof_tesseract_noop, ) - assert p.returncode == ExitCode.ok, err + assert result == ExitCode.ok def test_uppercase_extension(spoof_tesseract_noop, resources, outdir): @@ -379,24 +380,24 @@ def test_uppercase_extension(spoof_tesseract_noop, resources, outdir): ) -def test_input_file_not_found(no_outpdf): +def test_input_file_not_found(caplog, no_outpdf): input_file = "does not exist.pdf" - p, out, err = run_ocrmypdf(input_file, no_outpdf) - assert p.returncode == ExitCode.input_file - assert input_file in out or input_file in err + result = run_ocrmypdf_api(input_file, no_outpdf) + assert result == ExitCode.input_file + assert input_file in caplog.text -def test_input_file_not_a_pdf(no_outpdf): +def test_input_file_not_a_pdf(caplog, no_outpdf): input_file = __file__ # Try to OCR this file - p, out, err = run_ocrmypdf(input_file, no_outpdf) - assert p.returncode == ExitCode.input_file - assert input_file in out or input_file in err + result = run_ocrmypdf_api(input_file, no_outpdf) + assert result == ExitCode.input_file + assert input_file in caplog.text -def test_encrypted(resources, no_outpdf): - p, out, err = run_ocrmypdf(resources / 'skew-encrypted.pdf', no_outpdf) - assert p.returncode == ExitCode.encrypted_pdf - assert out.find('encrypted') +def test_encrypted(resources, caplog, no_outpdf): + result = run_ocrmypdf_api(resources / 'skew-encrypted.pdf', no_outpdf) + assert result == ExitCode.encrypted_pdf + assert 'encryption must be removed' in caplog.text @pytest.mark.parametrize('renderer', RENDERERS) @@ -415,8 +416,8 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) -def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): - p, out, err = run_ocrmypdf( +def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf, caplog): + result = run_ocrmypdf_api( resources / 'ccitt.pdf', no_outpdf, '-v', @@ -425,9 +426,9 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): renderer, env=spoof_tesseract_crash, ) - assert p.returncode == ExitCode.child_process_error + assert result == ExitCode.child_process_error assert not os.path.exists(no_outpdf) - assert "ERROR" in err + assert "SubprocessOutputError" in caplog.text def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf): diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 76c99d56..a0f07ddb 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -29,6 +29,7 @@ from ocrmypdf.exec import qpdf # pylint: disable=no-member,redefined-outer-name run_ocrmypdf = pytest.helpers.run_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 81282431..83ad01d4 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -24,6 +24,7 @@ from ocrmypdf.pdfinfo import PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api spoof = pytest.helpers.spoof @@ -32,9 +33,10 @@ def poster(resources): return resources / 'poster.pdf' -def test_userunit_ghostscript_fails(poster, no_outpdf): - p, out, err = run_ocrmypdf(poster, no_outpdf, '--output-type=pdfa') - assert p.returncode == ExitCode.input_file +def test_userunit_ghostscript_fails(poster, no_outpdf, caplog): + result = run_ocrmypdf_api(poster, no_outpdf, '--output-type=pdfa') + assert result == ExitCode.input_file + assert 'not supported by Ghostscript' in caplog.text def test_userunit_qpdf_passes(spoof_tesseract_cache, poster, outpdf): From 0cd424ffcbecf3b85e30c6476f1145e68b278549 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 14:44:32 -0800 Subject: [PATCH 241/880] Enforce str-only environment for Windows since it's more strict --- tests/conftest.py | 2 ++ tests/test_main.py | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/conftest.py b/tests/conftest.py index 1a63a5d2..9f796d3a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -205,6 +205,8 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if env: options.tesseract_env = env options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + if options.tesseract_env: + assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) return api.run_pipeline(options, api=False) diff --git a/tests/test_main.py b/tests/test_main.py index 77ea154e..17946965 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -303,7 +303,7 @@ def test_maximum_options( def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir): env = os.environ.copy() - env['TESSDATA_PREFIX'] = tmpdir + env['TESSDATA_PREFIX'] = os.fspath(tmpdir) returncode = run_ocrmypdf_api( resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env From 8a1dddc3eeec9e7ca3bdbe9921ba72346d40bf48 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 14:45:03 -0800 Subject: [PATCH 242/880] Don't worry about closed streams on Windows --- tests/test_stdio.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_stdio.py b/tests/test_stdio.py index a0f07ddb..53c0398f 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -77,6 +77,7 @@ def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): @pytest.mark.skipif( sys.version_info[0:3] >= (3, 6, 4), reason="issue fixed in Python 3.6.4" ) +@pytest.mark.skipif(os.name == 'nt', reason="POSIX problem") def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): input_file = str(resources / 'francais.pdf') output_file = str(outpdf) From ca9669742d632ff1c773e45921d1c98401727441 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 16:19:58 -0800 Subject: [PATCH 243/880] Move gs tests to test_ghostscript --- tests/test_ghostscript.py | 64 +++++++++++++++++++++++++++++++++++++++ tests/test_main.py | 57 ---------------------------------- 2 files changed, 64 insertions(+), 57 deletions(-) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 8756e8a8..c659fd68 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -18,12 +18,47 @@ import logging from decimal import Decimal + import pikepdf import pytest from PIL import Image +from ocrmypdf.exceptions import ExitCode from ocrmypdf.exec.ghostscript import rasterize_pdf +check_ocrmypdf = pytest.helpers.check_ocrmypdf +run_ocrmypdf = pytest.helpers.run_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +spoof = pytest.helpers.spoof + + +@pytest.fixture(scope='session') +def spoof_no_tess_gs_render_fail(tmp_path_factory): + return spoof( + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' + ) + + +@pytest.fixture(scope='session') +def spoof_no_tess_gs_raster_fail(tmp_path_factory): + return spoof( + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' + ) + + +@pytest.fixture(scope='session') +def spoof_no_tess_no_pdfa(tmp_path_factory): + return spoof( + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py' + ) + + +@pytest.fixture(scope='session') +def spoof_no_tess_pdfa_warning(tmp_path_factory): + return spoof( + tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' + ) + @pytest.fixture def linn(resources): @@ -79,3 +114,32 @@ def test_rasterize_rotated(linn, outdir, caplog): with Image.open(outdir / 'out.png') as im: assert im.size == (target_size[1], target_size[0]) assert im.info['dpi'] == (target_dpi[1], target_dpi[0]) + + +def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): + p, out, err = run_ocrmypdf( + resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail + ) + print(err) + assert p.returncode == ExitCode.child_process_error + + +def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): + p, out, err = run_ocrmypdf( + resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_gs_raster_fail + ) + print(err) + assert p.returncode == ExitCode.child_process_error + + +def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf): + p, out, err = run_ocrmypdf( + resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_no_pdfa + ) + assert ( + p.returncode == ExitCode.pdfa_conversion_failed + ), "Unexpected return when PDF/A fails" + + +def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf): + check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_pdfa_warning) diff --git a/tests/test_main.py b/tests/test_main.py index 17946965..73f91e5e 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -56,34 +56,6 @@ def spoof_tesseract_big_image_error(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_big_image_error.py') -@pytest.fixture(scope='session') -def spoof_no_tess_no_pdfa(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py' - ) - - -@pytest.fixture(scope='session') -def spoof_no_tess_pdfa_warning(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' - ) - - -@pytest.fixture(scope='session') -def spoof_no_tess_gs_render_fail(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' - ) - - -@pytest.fixture(scope='session') -def spoof_no_tess_gs_raster_fail(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' - ) - - def test_quick(spoof_tesseract_cache, resources, outpdf): check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) @@ -557,19 +529,6 @@ def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop, resources, out check_ocrmypdf(resources / 'epson.pdf', outpdf, env=spoof_tesseract_noop) -def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf): - p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_no_pdfa - ) - assert ( - p.returncode == ExitCode.pdfa_conversion_failed - ), "Unexpected return when PDF/A fails" - - -def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf): - check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_pdfa_warning) - - def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): "Checks for a Decimal quantize error with high DPI, etc" check_ocrmypdf(resources / '2400dpi.pdf', outpdf, env=spoof_tesseract_cache) @@ -718,22 +677,6 @@ def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): ) -def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): - p, out, err = run_ocrmypdf( - resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail - ) - print(err) - assert p.returncode == ExitCode.child_process_error - - -def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): - p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_gs_raster_fail - ) - print(err) - assert p.returncode == ExitCode.child_process_error - - @pytest.mark.skipif( '8.0.0' <= qpdf.version() <= '8.0.1', reason="qpdf regression on pages with no contents", From 43ab7c88d7604d7c9b6f59d5eb15a21699879c42 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 16:52:56 -0800 Subject: [PATCH 244/880] Remove os_environ() context manager --- src/ocrmypdf/_graft.py | 2 +- tests/conftest.py | 25 ------------------------- tests/test_graft.py | 7 ++----- 3 files changed, 3 insertions(+), 31 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 6a7cf23a..a535d492 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -21,7 +21,7 @@ from pathlib import Path import pikepdf -MAX_REPLACE_PAGES = int(os.environ.get('_OCRMYPDF_MAX_REPLACE_PAGES', 100)) +MAX_REPLACE_PAGES = 100 def _update_page_resources(*, page, font, font_key, procset): diff --git a/tests/conftest.py b/tests/conftest.py index 9f796d3a..3a2850a4 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -18,7 +18,6 @@ import os import platform import sys -from contextlib import contextmanager from pathlib import Path from subprocess import PIPE, run from ocrmypdf import api, cli @@ -107,30 +106,6 @@ def spoof(tmp_path_factory, **kwargs): return env -@pytest.helpers.register -@contextmanager -def os_environ(new_env): - old_env = os.environ.copy() - if new_env is None: - new_env = {} - - for k, v in new_env.items(): - if k != 'PYTEST_CURRENT_TEST': - os.environ[k] = v - yield - new_keys = set(os.environ.copy()) - set(old_env) - for k in new_keys: - if k != 'PYTEST_CURRENT_TEST': - del os.environ[k] - for k in old_env: - if k != 'PYTEST_CURRENT_TEST': - os.environ[k] = old_env[k] - - for k, v in os.environ.copy().items(): - if k != 'PYTEST_CURRENT_TEST': - assert v == old_env[k] - - @pytest.fixture(scope='session') def spoof_tesseract_noop(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_noop.py') diff --git a/tests/test_graft.py b/tests/test_graft.py index 2fd3d480..52aa336c 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -16,14 +16,13 @@ # along with OCRmyPDF. If not, see . import os +from unittest.mock import patch import pytest import ocrmypdf import pikepdf -os_environ = pytest.helpers.os_environ - def test_no_glyphless_graft(resources, outdir): pdf = pikepdf.open(resources / 'francais.pdf') @@ -33,9 +32,7 @@ def test_no_glyphless_graft(resources, outdir): pdf.pages.extend(pdf_cmyk.pages) pdf.save(outdir / 'test.pdf') - env = os.environ.copy() - env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2' - with os_environ(env): + with patch('ocrmypdf._graft.MAX_REPLACE_PAGES', 2): ocrmypdf.ocr( outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0 ) From d249aef57d60fd38b07613c483aca5b669bbb7fa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 13 Nov 2019 03:33:40 -0800 Subject: [PATCH 245/880] ghostscript: don't use NamedTemporaryFile Temporary files are more awkward for Windows. --- src/ocrmypdf/exec/ghostscript.py | 192 ++++++++++++++++--------------- 1 file changed, 99 insertions(+), 93 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 44fed271..65d15794 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -20,11 +20,12 @@ import logging import re import warnings +from contextlib import suppress from functools import lru_cache +from io import BytesIO from os import fspath -from shutil import copy -from subprocess import PIPE, STDOUT, run -from tempfile import NamedTemporaryFile +from pathlib import Path +from subprocess import PIPE, run from PIL import Image @@ -141,54 +142,57 @@ def rasterize_pdf( if not log: log = gslog - with NamedTemporaryFile(delete=True) as tmp: - args_gs = ( - [ - 'gs', - '-dQUIET', - '-dSAFER', - '-dBATCH', - '-dNOPAUSE', - f'-sDEVICE={raster_device}', - f'-dFirstPage={pageno}', - f'-dLastPage={pageno}', - f'-r{res[0]:f}x{res[1]:f}', - ] - + (['-dFILTERVECTOR'] if filter_vector else []) - + [ - '-o', - tmp.name, - '-dAutoRotatePages=/None', # Probably has no effect on raster - '-f', - fspath(input_file), - ] - ) + args_gs = ( + [ + 'gs', + '-dQUIET', + '-dSAFER', + '-dBATCH', + '-dNOPAUSE', + f'-sDEVICE={raster_device}', + f'-dFirstPage={pageno}', + f'-dLastPage={pageno}', + f'-r{res[0]:f}x{res[1]:f}', + ] + + (['-dFILTERVECTOR'] if filter_vector else []) + + [ + '-o', + '%stdout', + '-sstdout=%stderr', + '-dAutoRotatePages=/None', # Probably has no effect on raster + '-f', + fspath(input_file), + ] + ) - log.debug(args_gs) - p = run(args_gs, stdout=PIPE, stderr=STDOUT, universal_newlines=True) - if _gs_error_reported(p.stdout): - log.error(p.stdout) - elif p.stdout: - log.debug(p.stdout) + log.debug(args_gs) + with Path(output_file).open("wb") as output: + p = run(args_gs, stdout=PIPE, stderr=PIPE, check=False) + stderr = p.stderr.decode('utf-8', errors='replace') + if _gs_error_reported(stderr): + log.error(stderr) + elif stderr: + log.debug(stderr) - if p.returncode != 0: - raise SubprocessOutputError('Ghostscript rasterizing failed') + if p.returncode != 0: + with suppress(OSError): + Path(output_file).unlink() # no unfinished files + raise SubprocessOutputError('Ghostscript rasterizing failed') - tmp.seek(0) - with Image.open(tmp) as im: - if rotation is not None: - log.debug("Rotating output by %i", rotation) - # rotation is a clockwise angle and Image.ROTATE_* is - # counterclockwise so this cancels out the rotation - if rotation == 90: - im = im.transpose(Image.ROTATE_90) - elif rotation == 180: - im = im.transpose(Image.ROTATE_180) - elif rotation == 270: - im = im.transpose(Image.ROTATE_270) - if rotation % 180 == 90: - page_dpi = page_dpi[1], page_dpi[0] - im.save(fspath(output_file), dpi=page_dpi) + with Image.open(BytesIO(p.stdout)) as im: + if rotation is not None: + log.debug("Rotating output by %i", rotation) + # rotation is a clockwise angle and Image.ROTATE_* is + # counterclockwise so this cancels out the rotation + if rotation == 90: + im = im.transpose(Image.ROTATE_90) + elif rotation == 180: + im = im.transpose(Image.ROTATE_180) + elif rotation == 270: + im = im.transpose(Image.ROTATE_270) + if rotation % 180 == 90: + page_dpi = page_dpi[1], page_dpi[0] + im.save(fspath(output_file), dpi=page_dpi) def generate_pdfa( @@ -256,50 +260,52 @@ def generate_pdfa( # https://bugs.ghostscript.com/show_bug.cgi?id=699216 compression_args.append('-dPassThroughJPEGImages=false') - with NamedTemporaryFile(delete=True) as gs_pdf: - # nb no need to specify ProcessColorModel when ColorConversionStrategy - # is set; see: - # https://bugs.ghostscript.com/show_bug.cgi?id=699392 - args_gs = ( - [ - "gs", - "-dQUIET", - "-dBATCH", - "-dNOPAUSE", - "-dSAFER", - "-dCompatibilityLevel=" + str(pdf_version), - "-sDEVICE=pdfwrite", - "-dAutoRotatePages=/None", - "-sColorConversionStrategy=" + strategy, - ] - + compression_args - + [ - "-dJPEGQ=95", - "-dPDFA=" + pdfa_part, - "-dPDFACompatibilityPolicy=1", - "-sOutputFile=" + gs_pdf.name, - ] + # nb no need to specify ProcessColorModel when ColorConversionStrategy + # is set; see: + # https://bugs.ghostscript.com/show_bug.cgi?id=699392 + args_gs = ( + [ + "gs", + "-dQUIET", + "-dBATCH", + "-dNOPAUSE", + "-dSAFER", + "-dCompatibilityLevel=" + str(pdf_version), + "-sDEVICE=pdfwrite", + "-dAutoRotatePages=/None", + "-sColorConversionStrategy=" + strategy, + ] + + compression_args + + [ + "-dJPEGQ=95", + "-dPDFA=" + pdfa_part, + "-dPDFACompatibilityPolicy=1", + "-sOutputFile=%stdout", + "-sstdout=%stderr", + ] + ) + args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs + log.debug(args_gs) + with Path(output_file).open('wb') as output: + p = run(args_gs, stdout=output, stderr=PIPE, check=False) + + stderr = p.stderr.decode('utf-8', errors='replace') + if _gs_error_reported(stderr): + log.error(stderr) + elif 'overprint mode not set' in stderr: + # Unless someone is going to print PDF/A documents on a + # magical sRGB printer I can't see the removal of overprinting + # being a problem.... + log.debug( + "Ghostscript had to remove PDF 'overprinting' from the " + "input file to complete PDF/A conversion. " ) - args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs - log.debug(args_gs) - p = run(args_gs, stdout=PIPE, stderr=STDOUT, universal_newlines=True) + else: + log.debug(stderr) - if _gs_error_reported(p.stdout): - log.error(p.stdout) - elif 'overprint mode not set' in p.stdout: - # Unless someone is going to print PDF/A documents on a - # magical sRGB printer I can't see the removal of overprinting - # being a problem.... - log.debug( - "Ghostscript had to remove PDF 'overprinting' from the " - "input file to complete PDF/A conversion. " - ) - else: - log.debug(p.stdout) - - if p.returncode == 0: - # Ghostscript does not change return code when it fails to create - # PDF/A - check PDF/A status elsewhere - copy(gs_pdf.name, fspath(output_file)) - else: - raise SubprocessOutputError('Ghostscript PDF/A rendering failed') + if p.returncode != 0: + # Ghostscript does not change return code when it fails to create + # PDF/A - check PDF/A status elsewhere + with suppress(OSError): + Path(output_file).unlink() + raise SubprocessOutputError('Ghostscript PDF/A rendering failed') From bf99587aa1024255907eff74cb7bdbf532da91f7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 Nov 2019 16:29:53 -0800 Subject: [PATCH 246/880] ghostscript: use correct executable name on Windows --- src/ocrmypdf/exec/ghostscript.py | 20 +++++++++++++++----- 1 file changed, 15 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 65d15794..3613b43f 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -19,6 +19,7 @@ import logging import re +import os import warnings from contextlib import suppress from functools import lru_cache @@ -29,15 +30,24 @@ from subprocess import PIPE, run from PIL import Image -from ..exceptions import SubprocessOutputError +from ..exceptions import SubprocessOutputError, MissingDependencyError from . import get_version gslog = logging.getLogger() +GS = 'gs' +if os.name == 'nt': + GS = 'gswin64c' + try: + get_version(GS) + except MissingDependencyError: + GS = 'gswin32c' + get_version(GS) + @lru_cache(maxsize=1) def version(): - return get_version('gs') + return get_version(GS) def jpeg_passthrough_available(): @@ -84,7 +94,7 @@ def extract_text(input_file, pageno=1): args_gs = ( [ - 'gs', + GS, '-dQUIET', '-dSAFER', '-dBATCH', @@ -144,7 +154,7 @@ def rasterize_pdf( args_gs = ( [ - 'gs', + GS, '-dQUIET', '-dSAFER', '-dBATCH', @@ -265,7 +275,7 @@ def generate_pdfa( # https://bugs.ghostscript.com/show_bug.cgi?id=699392 args_gs = ( [ - "gs", + GS, "-dQUIET", "-dBATCH", "-dNOPAUSE", From c5fa72bd4ec7d46dad864263ab1e2883dc1a5282 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 Nov 2019 16:30:28 -0800 Subject: [PATCH 247/880] ghostscript: use run(check=True) for more consistent error handling --- src/ocrmypdf/exec/ghostscript.py | 74 +++++++++++++++++--------------- 1 file changed, 39 insertions(+), 35 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 3613b43f..4ce625fd 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -26,7 +26,7 @@ from functools import lru_cache from io import BytesIO from os import fspath from pathlib import Path -from subprocess import PIPE, run +from subprocess import PIPE, run, CalledProcessError from PIL import Image @@ -103,14 +103,15 @@ def extract_text(input_file, pageno=1): '-dTextFormat=0', ] + pages - + ['-o', '-', fspath(input_file)] + + ['-o', '-', fspath(input_file), "-sstdout=%stderr"] ) - p = run(args_gs, stdout=PIPE, stderr=PIPE) - if p.returncode != 0: + try: + p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True) + except CalledProcessError as e: raise SubprocessOutputError( - 'Ghostscript text extraction failed\n%s\n%s\n%s' - % (input_file, p.stdout.decode(), p.stderr.decode()) + 'Ghostscript text extraction failed\n%s\n%s' + % (input_file, e.stderr.decode(errors='replace')) ) return p.stdout @@ -167,7 +168,7 @@ def rasterize_pdf( + (['-dFILTERVECTOR'] if filter_vector else []) + [ '-o', - '%stdout', + '-', '-sstdout=%stderr', '-dAutoRotatePages=/None', # Probably has no effect on raster '-f', @@ -176,18 +177,19 @@ def rasterize_pdf( ) log.debug(args_gs) - with Path(output_file).open("wb") as output: - p = run(args_gs, stdout=PIPE, stderr=PIPE, check=False) - stderr = p.stderr.decode('utf-8', errors='replace') - if _gs_error_reported(stderr): - log.error(stderr) - elif stderr: - log.debug(stderr) - - if p.returncode != 0: + try: + p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True) + except CalledProcessError as e: with suppress(OSError): Path(output_file).unlink() # no unfinished files + log.error(e.stderr.decode(errors='replace')) raise SubprocessOutputError('Ghostscript rasterizing failed') + else: + stderr = p.stderr.decode(errors='replace') + if _gs_error_reported(stderr): + log.error(stderr) + elif stderr: + log.debug(stderr) with Image.open(BytesIO(p.stdout)) as im: if rotation is not None: @@ -290,32 +292,34 @@ def generate_pdfa( "-dJPEGQ=95", "-dPDFA=" + pdfa_part, "-dPDFACompatibilityPolicy=1", - "-sOutputFile=%stdout", + "-o", + "-", "-sstdout=%stderr", ] ) args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs log.debug(args_gs) - with Path(output_file).open('wb') as output: - p = run(args_gs, stdout=output, stderr=PIPE, check=False) - - stderr = p.stderr.decode('utf-8', errors='replace') - if _gs_error_reported(stderr): - log.error(stderr) - elif 'overprint mode not set' in stderr: - # Unless someone is going to print PDF/A documents on a - # magical sRGB printer I can't see the removal of overprinting - # being a problem.... - log.debug( - "Ghostscript had to remove PDF 'overprinting' from the " - "input file to complete PDF/A conversion. " - ) - else: - log.debug(stderr) - - if p.returncode != 0: + try: + with Path(output_file).open('wb') as output: + p = run(args_gs, stdout=output, stderr=PIPE, check=True) + except CalledProcessError as e: # Ghostscript does not change return code when it fails to create # PDF/A - check PDF/A status elsewhere with suppress(OSError): Path(output_file).unlink() + log.error(e.stderr.decode(errors='replace')) raise SubprocessOutputError('Ghostscript PDF/A rendering failed') + else: + stderr = p.stderr.decode('utf-8', errors='replace') + if _gs_error_reported(stderr): + log.error(stderr) + elif 'overprint mode not set' in stderr: + # Unless someone is going to print PDF/A documents on a + # magical sRGB printer I can't see the removal of overprinting + # being a problem.... + log.debug( + "Ghostscript had to remove PDF 'overprinting' from the " + "input file to complete PDF/A conversion. " + ) + else: + log.debug(stderr) From e51e21c6b6422fdf96287d296f1f1bae5efe7a96 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 Nov 2019 12:54:55 -0800 Subject: [PATCH 248/880] ghostscript: Refactor checking for executable name on Windows --- src/ocrmypdf/exec/ghostscript.py | 12 +++++----- tests/spoof/gs.py | 40 +++++++++++++++++++++++++++++++ tests/spoof/gs_feature_elision.py | 5 +--- tests/spoof/gs_pdfa_failure.py | 6 +---- tests/spoof/gs_raster_failure.py | 5 +--- tests/spoof/gs_render_failure.py | 5 +--- 6 files changed, 50 insertions(+), 23 deletions(-) create mode 100644 tests/spoof/gs.py diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 4ce625fd..434859ad 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -27,6 +27,7 @@ from io import BytesIO from os import fspath from pathlib import Path from subprocess import PIPE, run, CalledProcessError +from shutil import which from PIL import Image @@ -37,12 +38,11 @@ gslog = logging.getLogger() GS = 'gs' if os.name == 'nt': - GS = 'gswin64c' - try: - get_version(GS) - except MissingDependencyError: - GS = 'gswin32c' - get_version(GS) + GS = which('gswin64c') + if not GS: + GS = which('gswin32c') + if not GS: + raise MissingDependencyError("Ghostscript (gswin64c or gswin32c)") @lru_cache(maxsize=1) diff --git a/tests/spoof/gs.py b/tests/spoof/gs.py new file mode 100644 index 00000000..978f346b --- /dev/null +++ b/tests/spoof/gs.py @@ -0,0 +1,40 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +"""Find Ghostscript executable""" + + +import os +import shutil + + +def real_ghostscript(argv): + if os.name != 'nt': + gs = shutil.which('gs') + gs_args = [gs] + argv[1:] + os.execv(gs_args[0], gs_args) + else: + gs = shutil.which('gswin64c') + if not gs: + gs = shutil.which('gswin32c') + os.execv(gs, argv[1:]) + + return # Not reachable diff --git a/tests/spoof/gs_feature_elision.py b/tests/spoof/gs_feature_elision.py index ad65a619..0ae46b46 100755 --- a/tests/spoof/gs_feature_elision.py +++ b/tests/spoof/gs_feature_elision.py @@ -30,10 +30,7 @@ from subprocess import check_call PDF/A creation.""" -def real_ghostscript(argv): - gs_args = ['gs'] + argv[1:] - os.execvp("gs", gs_args) - return # Not reachable +from gs import real_ghostscript elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 diff --git a/tests/spoof/gs_pdfa_failure.py b/tests/spoof/gs_pdfa_failure.py index 730fa5b5..b8559192 100755 --- a/tests/spoof/gs_pdfa_failure.py +++ b/tests/spoof/gs_pdfa_failure.py @@ -27,11 +27,7 @@ import sys """Replicate Ghostscript PDF/A conversion failure by suppressing some arguments""" - -def real_ghostscript(argv): - gs_args = ['gs'] + argv[1:] - os.execvp("gs", gs_args) - return # Not reachable +from gs import real_ghostscript def main(): diff --git a/tests/spoof/gs_raster_failure.py b/tests/spoof/gs_raster_failure.py index b404cca8..f7269b3f 100755 --- a/tests/spoof/gs_raster_failure.py +++ b/tests/spoof/gs_raster_failure.py @@ -27,10 +27,7 @@ import sys """Replicate Ghostscript raster failure while allowing rendering""" -def real_ghostscript(argv): - gs_args = ['gs'] + argv[1:] - os.execvp("gs", gs_args) - return # Not reachable +from gs import real_ghostscript def main(): diff --git a/tests/spoof/gs_render_failure.py b/tests/spoof/gs_render_failure.py index 3027a684..5bb6ce7c 100755 --- a/tests/spoof/gs_render_failure.py +++ b/tests/spoof/gs_render_failure.py @@ -26,10 +26,7 @@ import os import sys -def real_ghostscript(argv): - gs_args = ['gs'] + argv[1:] - os.execvp("gs", gs_args) - return # Not reachable +from gs import real_ghostscript def main(): From 06a1f987d499f855d1576c3c67a4e49483c3e40d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 Nov 2019 16:40:04 -0800 Subject: [PATCH 249/880] Use _OCRMYPDF_TEST_PATH for testing and .py stubs to simulate symlinks --- src/ocrmypdf/api.py | 1 + src/ocrmypdf/exec/__init__.py | 23 ++++++++++++++++++- src/ocrmypdf/exec/ghostscript.py | 5 +++-- src/ocrmypdf/exec/jbig2enc.py | 4 ++-- src/ocrmypdf/exec/qpdf.py | 4 ++-- src/ocrmypdf/exec/tesseract.py | 4 ++-- src/ocrmypdf/exec/unpaper.py | 5 ++--- tests/conftest.py | 37 ++++++++++++++++++++++++++----- tests/spoof/gs_feature_elision.py | 1 - tests/spoof/gs_pdfa_failure.py | 1 - tests/spoof/gs_raster_failure.py | 1 - tests/spoof/gs_render_failure.py | 1 - tests/spoof/tesseract_cache.py | 2 -- tests/test_main.py | 10 ++++----- 14 files changed, 70 insertions(+), 29 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index a05edbc1..af47c969 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -18,6 +18,7 @@ import logging import os import sys +import warnings from enum import IntEnum from pathlib import Path diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index c1a18c9a..1f5656e7 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -21,14 +21,35 @@ import logging import os import re import sys +import shutil from collections.abc import Mapping -from subprocess import PIPE, STDOUT, CalledProcessError, run +from subprocess import PIPE, STDOUT, CalledProcessError, run as subprocess_run from ..exceptions import ExitCode, MissingDependencyError log = logging.Logger(__name__) +def _get_program(args, env=None): + program = args[0] + test_path = env.get('_OCRMYPDF_TEST_PATH', '') + if test_path: + program = shutil.which(program, path=test_path) + return program + + +def run(args, *, env=None, **kwargs): + if not env: + env = os.environ + program = _get_program(args, env) + if os.name == 'nt' and program.lower().endswith('.py'): + args = [sys.executable, program] + args[1:] + else: + args = [program] + args[1:] + log.debug(args) + return subprocess_run(args, env=env, **kwargs) + + def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): "Get the version of the specified program" args_prog = [program, version_arg] diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 434859ad..1d13122c 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -26,13 +26,13 @@ from functools import lru_cache from io import BytesIO from os import fspath from pathlib import Path -from subprocess import PIPE, run, CalledProcessError +from subprocess import PIPE, CalledProcessError from shutil import which from PIL import Image from ..exceptions import SubprocessOutputError, MissingDependencyError -from . import get_version +from . import get_version, run gslog = logging.getLogger() @@ -43,6 +43,7 @@ if os.name == 'nt': GS = which('gswin32c') if not GS: raise MissingDependencyError("Ghostscript (gswin64c or gswin32c)") + GS = Path(GS).stem @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index dff450c8..5218edbd 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -18,10 +18,10 @@ """Interface to jbig2 executable""" from functools import lru_cache -from subprocess import PIPE, run +from subprocess import PIPE from ..exceptions import MissingDependencyError -from . import get_version +from . import get_version, run @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/qpdf.py b/src/ocrmypdf/exec/qpdf.py index e96848c0..9be8692b 100644 --- a/src/ocrmypdf/exec/qpdf.py +++ b/src/ocrmypdf/exec/qpdf.py @@ -19,9 +19,9 @@ from functools import lru_cache from os import fspath -from subprocess import PIPE, STDOUT, CalledProcessError, run +from subprocess import PIPE, STDOUT, CalledProcessError -from . import get_version +from . import get_version, run @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 4b0560d8..34bd8983 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -23,7 +23,7 @@ from collections import namedtuple from contextlib import suppress import logging from os import fspath -from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run +from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired from ..exceptions import ( MissingDependencyError, @@ -31,7 +31,7 @@ from ..exceptions import ( TesseractConfigError, ) from ..helpers import page_number, safe_symlink -from . import get_version +from . import get_version, run OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 4515a33f..1143c0e9 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -22,7 +22,6 @@ import os import shlex -import subprocess from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory @@ -30,7 +29,7 @@ from tempfile import TemporaryDirectory from PIL import Image from ..exceptions import MissingDependencyError, SubprocessOutputError -from . import get_version +from . import get_version, run as external_run @lru_cache(maxsize=1) @@ -77,7 +76,7 @@ def run(input_file, output_file, dpi, log, mode_args): # their unpaper arguments (whether intentionally or otherwise) args_unpaper.extend([input_pnm, output_pnm]) try: - proc = subprocess.run( + proc = external_run( args_unpaper, check=True, close_fds=True, diff --git a/tests/conftest.py b/tests/conftest.py index 3a2850a4..08b45fb2 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -80,6 +80,19 @@ PROJECT_ROOT = os.path.dirname(TESTS_ROOT) OCRMYPDF = [sys.executable, '-m', 'ocrmypdf'] +PY_FILE_TEMPLATE = """ +import os +import subprocess +import sys + +args = [sys.executable, {spoofer}, *sys.argv[1:]] +p = subprocess.run(args, check=False, stdout=subprocess.PIPE, stderr=subprocess.PIPE) +sys.stdout.buffer.write(p.stdout) +sys.stderr.buffer.write(p.stderr) +sys.exit(p.returncode) +""" + + @pytest.helpers.register def spoof(tmp_path_factory, **kwargs): """Modify PATH to override subprocess executables @@ -97,12 +110,24 @@ def spoof(tmp_path_factory, **kwargs): for replace_program, with_spoof in kwargs.items(): spoofer = Path(SPOOF_PATH) / with_spoof - spoofer.chmod(0o755) - (tmpdir / replace_program).symlink_to(spoofer) - - env['_OCRMYPDF_SAVE_PATH'] = env['PATH'] - env['PATH'] = str(tmpdir) + ":" + env['PATH'] + if os.name != 'nt': + spoofer.chmod(0o755) + (tmpdir / replace_program).symlink_to(spoofer) + else: + py_file = PY_FILE_TEMPLATE.format( + python=sys.executable, spoofer=repr(os.fspath(spoofer.absolute())) + ) + if replace_program == 'gs': + programs = ['gswin64c', 'gswin32c'] + else: + programs = [replace_program] + for prog in programs: + (tmpdir / f'{prog}.py').write_text(py_file, encoding='utf-8') + env['_OCRMYPDF_TEST_PATH'] = str(tmpdir) + os.pathsep + env['PATH'] + if os.name == 'nt': + if '.py' not in env['PATHEXT'].lower(): + raise EnvironmentError("PATHEXT is not configured to support .py") return env @@ -178,7 +203,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): ) api.check_options(options) if env: - options.tesseract_env = env + options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) diff --git a/tests/spoof/gs_feature_elision.py b/tests/spoof/gs_feature_elision.py index 0ae46b46..f9856311 100755 --- a/tests/spoof/gs_feature_elision.py +++ b/tests/spoof/gs_feature_elision.py @@ -38,7 +38,6 @@ not permitted in PDF/A-2, overprint mode not set""" def main(): - os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH'] if '--version' in sys.argv: print('9.20') print('SPOOFED: ' + os.path.basename(__file__)) diff --git a/tests/spoof/gs_pdfa_failure.py b/tests/spoof/gs_pdfa_failure.py index b8559192..6dd90e29 100755 --- a/tests/spoof/gs_pdfa_failure.py +++ b/tests/spoof/gs_pdfa_failure.py @@ -31,7 +31,6 @@ from gs import real_ghostscript def main(): - os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH'] if '--version' in sys.argv: print('9.20') print('SPOOFED: ' + os.path.basename(__file__)) diff --git a/tests/spoof/gs_raster_failure.py b/tests/spoof/gs_raster_failure.py index f7269b3f..c07b881b 100755 --- a/tests/spoof/gs_raster_failure.py +++ b/tests/spoof/gs_raster_failure.py @@ -31,7 +31,6 @@ from gs import real_ghostscript def main(): - os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH'] if '--version' in sys.argv: print('9.20') print('SPOOFED: ' + os.path.basename(__file__)) diff --git a/tests/spoof/gs_render_failure.py b/tests/spoof/gs_render_failure.py index 5bb6ce7c..a43833a8 100755 --- a/tests/spoof/gs_render_failure.py +++ b/tests/spoof/gs_render_failure.py @@ -30,7 +30,6 @@ from gs import real_ghostscript def main(): - os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH'] if '--version' in sys.argv: print('9.20') print('SPOOFED: ' + os.path.basename(__file__)) diff --git a/tests/spoof/tesseract_cache.py b/tests/spoof/tesseract_cache.py index 528e5a1e..83c09535 100755 --- a/tests/spoof/tesseract_cache.py +++ b/tests/spoof/tesseract_cache.py @@ -59,8 +59,6 @@ import subprocess import sys from pathlib import Path -if '_OCRMYPDF_SAVE_PATH' in os.environ: - os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH'] __version__ = subprocess.check_output( ['tesseract', '--version'], stderr=subprocess.STDOUT diff --git a/tests/test_main.py b/tests/test_main.py index 73f91e5e..c18a23f1 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -297,10 +297,10 @@ def test_force_ocr_on_pdf_with_no_images(spoof_tesseract_crash, resources, no_ou # As a correctness test, make sure that --force-ocr on a PDF with no # content still triggers tesseract. If tesseract crashes, then it was # called. - result = run_ocrmypdf_api( + p, _, _ = run_ocrmypdf( resources / 'blank.pdf', no_outpdf, '--force-ocr', env=spoof_tesseract_crash ) - assert result == ExitCode.child_process_error + assert p.returncode == ExitCode.child_process_error assert not os.path.exists(no_outpdf) @@ -389,7 +389,7 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf, caplog): - result = run_ocrmypdf_api( + p, _, err = run_ocrmypdf( resources / 'ccitt.pdf', no_outpdf, '-v', @@ -398,9 +398,9 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf, renderer, env=spoof_tesseract_crash, ) - assert result == ExitCode.child_process_error + assert p.returncode == ExitCode.child_process_error assert not os.path.exists(no_outpdf) - assert "SubprocessOutputError" in caplog.text + assert "SubprocessOutputError" in err def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf): From 66d04dd6e32c7930baec14ebf3fbb008992c1c1a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 13:33:00 -0800 Subject: [PATCH 250/880] Don't expect filenames to be replicated on NT --- tests/test_main.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_main.py b/tests/test_main.py index c18a23f1..fa7473c4 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -363,7 +363,8 @@ def test_input_file_not_a_pdf(caplog, no_outpdf): input_file = __file__ # Try to OCR this file result = run_ocrmypdf_api(input_file, no_outpdf) assert result == ExitCode.input_file - assert input_file in caplog.text + if os.name != 'nt': # name will be mangled with \\'s on nt + assert input_file in caplog.text def test_encrypted(resources, caplog, no_outpdf): From cff37bf6814d3ea44c96f1bd4f43cee3e77e1113 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 13:40:48 -0800 Subject: [PATCH 251/880] Make test_german more Windows-friendly --- tests/test_main.py | 26 ++++++++++++++------------ 1 file changed, 14 insertions(+), 12 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index fa7473c4..e703f07f 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -313,18 +313,20 @@ def test_german(spoof_tesseract_cache, resources, outdir): # properly. It is fine that we are testing -l deu on a French file because # we are exercising the functionality not going for accuracy. sidecar = outdir / 'francais.txt' - result = run_ocrmypdf_api( - resources / 'francais.pdf', - outdir / 'francais.pdf', - '-l', - 'deu', # more commonly installed - '--sidecar', - sidecar, - env=spoof_tesseract_cache, - ) - if 'deu' not in tesseract.languages(): - pytest.xfail(reason="tesseract-deu language pack not installed") - assert result == ExitCode.ok, "Requires tesseract deu language pack" + try: + check_ocrmypdf( + resources / 'francais.pdf', + outdir / 'francais.pdf', + '-l', + 'deu', # more commonly installed + '--sidecar', + sidecar, + env=spoof_tesseract_cache, + ) + except MissingDependencyError: + if 'deu' not in tesseract.languages(): + pytest.xfail(reason="tesseract-deu language pack not installed") + raise def test_klingon(resources, outpdf): From d0301813cc30f4282f2d360de04426e544da7ec5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 14:58:46 -0800 Subject: [PATCH 252/880] ghosttext: mention page number differences --- src/ocrmypdf/pdfinfo/ghosttext.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/pdfinfo/ghosttext.py b/src/ocrmypdf/pdfinfo/ghosttext.py index 43156154..9626fad7 100644 --- a/src/ocrmypdf/pdfinfo/ghosttext.py +++ b/src/ocrmypdf/pdfinfo/ghosttext.py @@ -96,6 +96,7 @@ def extract_text_xml(infile, pdf, pageno=None, log=gslog): page_count_difference = len(pdf.pages) - len(page_xml) if page_count_difference != 0: log.error("The number of pages in the input file is inconsistent.") + log.error(f"Expected {len(pdf.pages)}, txtwrite says {len(page_xml)}") if page_count_difference > 0: page_xml.extend([None] * page_count_difference) return page_xml From 9db01c7ff5cdea61f5b5d807fd9d5493754ad8b9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 15:00:12 -0800 Subject: [PATCH 253/880] Remove test_bad_utf8 Due to difficulties of getting this to work on Python 3.8, Windows, and high probability that this behavior is now gone from Tesseract 4.0+. Originally added in 2017. --- src/ocrmypdf/exec/tesseract.py | 7 +------ tests/test_stdio.py | 18 +----------------- 2 files changed, 2 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 34bd8983..a4a42b7d 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -194,12 +194,7 @@ def tesseract_log_output(mainlog, stdout, input_file): try: text = stdout.decode() except UnicodeDecodeError: - log.error( - "command line output was not utf-8. " - + "This usually means Tesseract's language packs do not match " - "the installed version of Tesseract." - ) - text = stdout.decode('utf-8', 'backslashreplace') + text = stdout.decode('utf-8', 'ignore') lines = text.splitlines() for line in lines: diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 53c0398f..0f2609af 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -18,7 +18,7 @@ import os import sys from pathlib import Path -from subprocess import DEVNULL, PIPE, run, Popen +from subprocess import DEVNULL, PIPE, run, Popen, CalledProcessError import pytest @@ -115,22 +115,6 @@ def test_bad_locale(): assert 'configured to use ASCII as encoding' in err, "should whine" -@pytest.mark.parametrize('renderer', ['hocr', 'sandwich']) -def test_bad_utf8(spoof_tess_bad_utf8, renderer, resources, no_outpdf): - p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', - no_outpdf, - '--pdf-renderer', - renderer, - env=spoof_tess_bad_utf8, - ) - - assert out == '', "stdout not clean" - assert p.returncode != 0 - assert 'not utf-8' in err, "should whine about utf-8" - assert '\\x96' in err, 'should repeat backslash encoded output' - - def test_dev_null(spoof_tesseract_noop, resources): p, out, err = run_ocrmypdf( resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop From cb3cfaa055e2f6b2fca99657daf4f60ec4b1dbd7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 15:56:18 -0800 Subject: [PATCH 254/880] Add Windows install advice --- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/exec/__init__.py | 12 +++++++++++- 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index b02e4532..af518b2b 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -446,7 +446,7 @@ def report_output_file_size(options, input_file, output_file): def check_dependency_versions(options): check_external_program( program='tesseract', - package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'}, + package={'linux': 'tesseract-ocr'}, version_checker=tesseract.version, need_version='4.0.0', # using backport for Travis CI ) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 1f5656e7..62ea5c47 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -132,23 +132,33 @@ On RPM-based systems (Red Hat, Fedora), search for instructions on installing the RPM for {program}. ''' +windows_install_advice = ''' +If not already installed, install the Chocolatey package manager. Then use +a command prompt to install the missing package: + choco install {package} +''' + def _get_platform(): if sys.platform.startswith('freebsd'): return 'freebsd' elif sys.platform.startswith('linux'): return 'linux' + elif sys.platform.startswith('win'): + return 'windows' return sys.platform def _error_trailer(program, package, **kwargs): if isinstance(package, Mapping): - package = package[_get_platform()] + package = package.get(_get_platform(), program) if _get_platform() == 'darwin': log.info(osx_install_advice.format(**locals())) elif _get_platform() == 'linux': log.info(linux_install_advice.format(**locals())) + elif _get_platform() == 'windows': + log.info(windows_install_advice.format(**locals())) def _error_missing_program(program, package, required_for, recommended): From d4abe88452e07f919406246c9d1e4b1926c8472b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 19 Nov 2019 12:52:48 -0800 Subject: [PATCH 255/880] docs: sketch Windows install procedure --- docs/installation.rst | 29 ++++++++++++++++++----------- 1 file changed, 18 insertions(+), 11 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 46bee8d3..70a015b9 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -431,8 +431,24 @@ See `OCRmyPDF Docker Image `__ for more information. Installing on Windows ===================== -Direct installation on Windows is not currently possible, but it works well in -Windows Subsystem for Linux: +You must install the following for Windows using their installers: + +* Python 3.7 (64-bit recommended) +* Tesseract 4.0 or later +* Ghostscript 9.50 or later +* QPDF 9.0.2 or later + +You can install all except Tesseract with the Chocolatey package manager: + +* ``choco install python3`` +* ``choco install ghostscript`` +* ``choco install qpdf`` + +Modify your ``PATH`` environment variable so that Tesseract, Ghostscript and QPDF +executables on the ``PATH``. + +Installing on Windows Subsystem for Linux +========================================= #. Install Ubuntu 18.04 for Windows Subsystem for Linux, if not already installed. #. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 18.04 `. @@ -451,15 +467,6 @@ Then confirm that the expected version from PyPI (|latest|) is installed: You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing ``wsl``, and call it from Windows programs or batch files. -Why no native Windows? -^^^^^^^^^^^^^^^^^^^^^^ - -It would probably not be too difficult to port on Windows. The main -reason this has been avoided is the difficulty of packaging and -installing the various non-Python dependencies: Tesseract, QPDF, -Ghostscript, Leptonica. Pull requests to add or improve Windows support -would be quite welcome. - Docker ^^^^^^ From b8b7ecfe7f7d05d30037f4f2d9ac97d24c952c94 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 21:10:27 -0800 Subject: [PATCH 256/880] Fix DecompressionBomb related errors due to Windows process differences --- src/ocrmypdf/_sync.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 4c25b300..27262135 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -26,6 +26,7 @@ from collections import namedtuple from tempfile import mkdtemp from tqdm import tqdm +import PIL from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files, make_logger @@ -176,7 +177,7 @@ def post_process(pdf_file, context): return optimize_pdf(pdf_out, context) -def worker_init(queue): +def worker_init(queue, max_pixels): """Initialize a process pool worker""" # Ignore SIGINT (our parent process will kill us gracefully) @@ -188,9 +189,15 @@ def worker_init(queue): root.handlers = [] root.addHandler(h) + # In Windows, child process will not inherit our change to this value in + # the parent process, so ensure workers get it set + PIL.Image.MAX_IMAGE_PIXELS = max_pixels -def worker_thread_init(_queue): - pass + +def worker_thread_init(_queue, max_pixels): + # This is probably not needed since threads should all see the same memory, + # but done for consistency. + PIL.Image.MAX_IMAGE_PIXELS = max_pixels def log_listener(queue): @@ -261,7 +268,9 @@ def exec_concurrent(context): unit_scale=0.5, disable=not context.options.progress_bar, ) as pbar, Pool( - processes=max_workers, initializer=initializer, initargs=(log_queue,) + processes=max_workers, + initializer=initializer, + initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS), ) as pool: results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) while True: From 5607429d9a5e6e2bbfdee50b001681def84db5ed Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 21:31:01 -0800 Subject: [PATCH 257/880] tests: error message from tesseract change --- tests/test_main.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_main.py b/tests/test_main.py index e703f07f..8df33562 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -612,7 +612,10 @@ THIS FILE IS INVALID '--tesseract-config', cfg_file, ) - assert "parameter not found" in err.lower(), "No error message" + assert ( + "parameter not found" in err.lower() + or "error occurred while parsing" in err.lower() + ), "No error message" assert p.returncode == ExitCode.invalid_config From 51abd791363116b0fd4150e247a4e2bd99e067df Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Dec 2019 21:35:28 -0800 Subject: [PATCH 258/880] Tesseract no longer posts an error message if config file not found --- tests/test_main.py | 17 ----------------- 1 file changed, 17 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index 8df33562..a5af5b10 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -576,23 +576,6 @@ language_model_penalty_non_freq_dict_word 0 ) -@pytest.mark.slow # This test sometimes times out in CI -@pytest.mark.parametrize('renderer', RENDERERS) -def test_tesseract_config_notfound(renderer, resources, outdir): - cfg_file = outdir / 'nofile.cfg' - - p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', - outdir / 'out.pdf', - '--pdf-renderer', - renderer, - '--tesseract-config', - cfg_file, - ) - assert "Can't open" in err, "No error message about missing config file" - assert p.returncode == ExitCode.ok, err - - @pytest.mark.slow # This test sometimes times out in CI @pytest.mark.parametrize('renderer', RENDERERS) def test_tesseract_config_invalid(renderer, resources, outdir): From f6510e2b1512a2c760256c42fe57b5e0b8e68612 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 6 Dec 2019 15:00:12 -0800 Subject: [PATCH 259/880] Document function of symlink shim --- tests/conftest.py | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 08b45fb2..1b1ece82 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import ast import os import platform import sys @@ -80,7 +81,9 @@ PROJECT_ROOT = os.path.dirname(TESTS_ROOT) OCRMYPDF = [sys.executable, '-m', 'ocrmypdf'] -PY_FILE_TEMPLATE = """ +WINDOWS_SHIM_TEMPLATE = """ +# This is a shim for Windows that has the same effect as a symlink to the target .py +# file import os import subprocess import sys @@ -92,6 +95,8 @@ sys.stderr.buffer.write(p.stderr) sys.exit(p.returncode) """ +assert ast.parse(WINDOWS_SHIM_TEMPLATE.format(spoofer=repr(r"C:\\Temp\\file.py"))) + @pytest.helpers.register def spoof(tmp_path_factory, **kwargs): @@ -114,8 +119,8 @@ def spoof(tmp_path_factory, **kwargs): spoofer.chmod(0o755) (tmpdir / replace_program).symlink_to(spoofer) else: - py_file = PY_FILE_TEMPLATE.format( - python=sys.executable, spoofer=repr(os.fspath(spoofer.absolute())) + py_file = WINDOWS_SHIM_TEMPLATE.format( + spoofer=repr(os.fspath(spoofer.absolute())) ) if replace_program == 'gs': programs = ['gswin64c', 'gswin32c'] From 66bda3420a2fbd13fe6ae2aa5089fee556e3deea Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 6 Dec 2019 15:03:20 -0800 Subject: [PATCH 260/880] docs: cause about using Windows in production --- docs/installation.rst | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 70a015b9..414239f1 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -431,6 +431,13 @@ See `OCRmyPDF Docker Image `__ for more information. Installing on Windows ===================== +.. warning:: + + Native Windows support is new. Consider it "beta" software. Some + functionality is missing or may be more difficult to enable. If you need a + production-ready solution, use Windows Subsystem for Linux or a Docker + image. + You must install the following for Windows using their installers: * Python 3.7 (64-bit recommended) @@ -438,12 +445,17 @@ You must install the following for Windows using their installers: * Ghostscript 9.50 or later * QPDF 9.0.2 or later -You can install all except Tesseract with the Chocolatey package manager: +You can install these with the Chocolatey package manager: * ``choco install python3`` +* ``choco install tesseract`` * ``choco install ghostscript`` * ``choco install qpdf`` +Also consider adding: + +* ``choco install pngquant`` + Modify your ``PATH`` environment variable so that Tesseract, Ghostscript and QPDF executables on the ``PATH``. From 8077718804c5e7978291f4d24c2f05d1f1029a58 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Nov 2019 01:41:24 -0800 Subject: [PATCH 261/880] Possible fix to loss of log adapter state --- src/ocrmypdf/_jobcontext.py | 8 ++++---- src/ocrmypdf/exec/tesseract.py | 5 +++-- 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index e50aa75a..42c96282 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -100,14 +100,14 @@ def cleanup_working_files(work_folder, options): class LogNameAdapter(logging.LoggerAdapter): def process(self, msg, kwargs): - # return '[%s] %s' % (self.extra['filename'], msg), kwargs + # return '[%s] %s' % (self.extra['input_filename'], msg), kwargs return '%s' % (msg,), kwargs class LogNamePageAdapter(logging.LoggerAdapter): def process(self, msg, kwargs): return ( - #'[%s:%05u] %s' % (self.extra['filename'], self.extra['page'], msg), + #'[%s:%05u] %s' % (self.extra['input_filename'], self.extra['page'], msg), '%4u: %s' % (self.extra['page'], msg), kwargs, ) @@ -116,9 +116,9 @@ class LogNamePageAdapter(logging.LoggerAdapter): def make_logger(options=None, prefix='ocrmypdf', filename=None, page=None): log = logging.getLogger(prefix) if filename and page: - adapter = LogNamePageAdapter(log, dict(filename=filename, page=page)) + adapter = LogNamePageAdapter(log, dict(input_filename=filename, page=page)) elif filename: - adapter = LogNameAdapter(log, dict(filename=filename)) + adapter = LogNameAdapter(log, dict(input_filename=filename)) else: adapter = log return adapter diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index afa0e4a6..ac97ac2d 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -55,6 +55,7 @@ HOCR_TEMPLATE = """ class TesseractLoggerAdapter(logging.LoggerAdapter): def process(self, msg, kwargs): + kwargs['extra'] = self.extra return '[tesseract] %s' % (msg), kwargs @@ -194,8 +195,8 @@ def tesseract_log_output(mainlog, stdout, input_file): text = stdout.decode() except UnicodeDecodeError: log.error( - "command line output was not utf-8. " - + "This usually means Tesseract's language packs do not match " + "Tesseract's output was not utf-8. " + "This usually means Tesseract's language packs do not match " "the installed version of Tesseract." ) text = stdout.decode('utf-8', 'backslashreplace') From fbf271a3ecdbb6c3303bc869becb2a737ea75578 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Nov 2019 01:42:10 -0800 Subject: [PATCH 262/880] Remove Tesseract < 4.0 specific check --- src/ocrmypdf/exec/tesseract.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index ac97ac2d..3b4e67c2 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -138,7 +138,7 @@ def tess_base_args(langs, engine_mode): args = ['tesseract'] if langs: args.extend(['-l', '+'.join(langs)]) - if engine_mode is not None and v4(): + if engine_mode is not None: args.extend(['--oem', str(engine_mode)]) return args From 1c1b60fa9f4a5392be917518b32ac1cb0296bfbf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 6 Dec 2019 15:10:54 -0800 Subject: [PATCH 263/880] Add typing hints for ocr() function --- src/ocrmypdf/api.py | 85 +++++++++++++++++++++++---------------------- 1 file changed, 43 insertions(+), 42 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index af47c969..29b25a38 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -21,6 +21,7 @@ import sys import warnings from enum import IntEnum from pathlib import Path +from typing import List, Optional, Dict from tqdm import tqdm @@ -165,52 +166,52 @@ def create_options(*, input_file, output_file, **kwargs): def ocr( # pylint: disable=unused-argument - input_file, - output_file, + input_file: os.PathLike, + output_file: os.PathLike, *, - language=None, - image_dpi=None, + language: List[str] = None, + image_dpi: int = None, output_type=None, - sidecar=None, - jobs=None, - use_threads=None, - title=None, - author=None, - subject=None, - keywords=None, - rotate_pages=None, - remove_background=None, - deskew=None, - clean=None, - clean_final=None, - unpaper_args=None, - oversample=None, - remove_vectors=None, - threshold=None, - force_ocr=None, - skip_text=None, - redo_ocr=None, - skip_big=None, - optimize=None, - jpg_quality=None, - png_quality=None, - jbig2_lossy=None, - jbig2_page_group_size=None, - pages=None, - max_image_mpixels=None, - tesseract_config=None, - tesseract_pagesegmode=None, - tesseract_oem=None, + sidecar: os.PathLike = None, + jobs: int = None, + use_threads: bool = None, + title: str = None, + author: str = None, + subject: str = None, + keywords: str = None, + rotate_pages: bool = None, + remove_background: bool = None, + deskew: bool = None, + clean: bool = None, + clean_final: bool = None, + unpaper_args: str = None, + oversample: int = None, + remove_vectors: bool = None, + threshold: bool = None, + force_ocr: bool = None, + skip_text: bool = None, + redo_ocr: bool = None, + skip_big: float = None, + optimize: int = None, + jpg_quality: int = None, + png_quality: int = None, + jbig2_lossy: bool = None, + jbig2_page_group_size: int = None, + pages: str = None, + max_image_mpixels: float = None, + tesseract_config: List[str] = None, + tesseract_pagesegmode: int = None, + tesseract_oem: int = None, pdf_renderer=None, - tesseract_timeout=None, - rotate_pages_threshold=None, + tesseract_timeout: float = None, + rotate_pages_threshold: float = None, pdfa_image_compression=None, - user_words=None, - user_patterns=None, - fast_web_view=None, - keep_temporary_files=None, - progress_bar=None, - tesseract_env=None, + user_words: os.PathLike = None, + user_patterns: os.PathLike = None, + fast_web_view: float = None, + keep_temporary_files: bool = None, + progress_bar: bool = None, + tesseract_env: Dict[str, str] = None, ): """Run OCRmyPDF on one PDF or image. From 17d97b354adcfe36d99d88ba7dd11603bf9c50b4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 Nov 2019 01:16:50 -0800 Subject: [PATCH 264/880] Ignore mypy cache --- .gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitignore b/.gitignore index 46ed9aae..bf6e896c 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ *.pyc *.sublime-* *.DS_Store +.mypy_cache/ # Package building .eggs/ From cac4a8b9b67b3b513422274df439e5b158cce1ab Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 29 Nov 2019 03:46:26 -0800 Subject: [PATCH 265/880] Suppress duplicate error messages from Ghostscript --- src/ocrmypdf/exec/ghostscript.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 1d13122c..1bc96f6e 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -313,7 +313,17 @@ def generate_pdfa( else: stderr = p.stderr.decode('utf-8', errors='replace') if _gs_error_reported(stderr): - log.error(stderr) + last_part = None + repcount = 0 + for part in p.stdout.split('****'): + if part != last_part: + if repcount > 1: + log.error(f"(previous error message repeated {repcount} times)") + repcount = 0 + log.error(part) + else: + repcount += 1 + last_part = part elif 'overprint mode not set' in stderr: # Unless someone is going to print PDF/A documents on a # magical sRGB printer I can't see the removal of overprinting From 65855dc14c1b1c4573f47f5cd14fb5552fd664c3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 7 Dec 2019 12:44:23 -0800 Subject: [PATCH 266/880] Fix close_fds=True on Windows Python 3.6 --- src/ocrmypdf/exec/__init__.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 62ea5c47..d5a7de44 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -47,6 +47,10 @@ def run(args, *, env=None, **kwargs): else: args = [program] + args[1:] log.debug(args) + if sys.version_info < (3, 7) and os.name == 'nt': + # Can't use close_fds=True on Windows with Python 3.6 or older + # https://bugs.python.org/issue19575, etc. + kwargs['close_fds'] = False return subprocess_run(args, env=env, **kwargs) From 7be293f628f0f13fbb2c2ba45443fce115d1050d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 7 Dec 2019 13:09:25 -0800 Subject: [PATCH 267/880] Address tests that fail on Windows with Python 3.7 or 3.6 --- tests/test_stdio.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 0f2609af..ce2076b1 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -105,6 +105,7 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): @pytest.mark.skipif( Path('/etc/alpine-release').exists(), reason="invalid test on alpine" ) +@pytest.mark.skipif(os.name == 'nt', reason="invalid test on Windows") def test_bad_locale(): env = os.environ.copy() env['LC_ALL'] = 'C' @@ -115,6 +116,10 @@ def test_bad_locale(): assert 'configured to use ASCII as encoding' in err, "should whine" +@pytest.mark.xfail( + os.name == 'nt' and sys.version_info < (3, 8), + reason="Windows does not like this; not sure how to fix", +) def test_dev_null(spoof_tesseract_noop, resources): p, out, err = run_ocrmypdf( resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop From b354511ac92c0ec988eef556dd145177b8151564 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 13:02:49 -0800 Subject: [PATCH 268/880] ghostscript: document need to write to stdout when using txtwrite --- src/ocrmypdf/exec/ghostscript.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 1bc96f6e..0f2499e2 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -93,6 +93,9 @@ def extract_text(input_file, pageno=1): else: pages = [] + # Note due to bug https://bugs.ghostscript.com/show_bug.cgi?id=701971 + # Ghostscript <= 9.50 will truncate output unless we write to stdout, so + # don't write to a file. args_gs = ( [ GS, From fd9550acdacfe8d63bf54f6a5d2675313e9d8ba4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 6 Dec 2019 16:04:01 -0800 Subject: [PATCH 269/880] Add Azure Pipelines CI/CD --- .travis.yml | 4 + azure-pipelines.yml | 274 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 278 insertions(+) create mode 100644 azure-pipelines.yml diff --git a/.travis.yml b/.travis.yml index de391a8a..4aa53def 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,3 +1,7 @@ +branches: + except: + - azure + cache: pip: true directories: diff --git a/azure-pipelines.yml b/azure-pipelines.yml new file mode 100644 index 00000000..fe2d6bd4 --- /dev/null +++ b/azure-pipelines.yml @@ -0,0 +1,274 @@ +trigger: + tags: + include: + - v* + branches: + include: + - "*" + exclude: + - "travis" + +stages: + - stage: "Test" + jobs: + - job: Windows + pool: + vmImage: "vs2017-win2016" + strategy: + matrix: + Python36: + python.version: "3.6" + Python37: + python.version: "3.7" + Python38: + python.version: "3.8" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - pwsh: | + choco install --yes --no-progress --pre tesseract + choco install --yes --no-progress python3 + choco install --yes --no-progress ghostscript + choco install --yes --no-progress qpdf + choco install --yes --no-progress pngquant + displayName: "Install system packages" + - pwsh: | + refreshenv + $env:path = "C:\Program Files\Tesseract-OCR;C:\Program Files\gs\gs9.50\bin;" + $env:path + pip install --upgrade pip wheel + pip install -r requirements/main.txt -r requirements/test.txt . + tesseract --version + qpdf --version + displayName: "Install Python packages" + - pwsh: | + refreshenv + $env:path = "C:\Program Files\Tesseract-OCR;C:\Program Files\gs\gs9.50\bin;" + $env:path + $env:pathext += ';.py' + # -n auto helps Windows + pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() + - task: PublishCodeCoverageResults@1 + inputs: + codeCoverageTool: Cobertura + summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" + - job: "Ubuntu_1804" + pool: + vmImage: "ubuntu-18.04" + strategy: + matrix: + Python36: + python.version: "3.6" + Python37: + python.version: "3.7" + Python38: + python.version: "3.8" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - bash: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + python3-software-properties \ + curl \ + ghostscript \ + img2pdf \ + libexempi3 \ + libffi-dev \ + liblept5 \ + libsm6 libxext6 libxrender-dev \ + pngquant \ + poppler-utils \ + qpdf \ + tesseract-ocr \ + tesseract-ocr-deu \ + tesseract-ocr-eng \ + unpaper \ + zlib1g + displayName: "Install system packages" + - bash: | + curl https://bootstrap.pypa.io/get-pip.py | python3 + pip3 install -r requirements/main.txt -r requirements/test.txt . + displayName: "Install Python packages" + - bash: | + tesseract --version + qpdf --version + displayName: "Record versions" + - bash: | + # -n auto is slower on Linux and breaks on Python 3.8 + pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() + - task: PublishCodeCoverageResults@1 + inputs: + codeCoverageTool: Cobertura + summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" + - job: "Ubuntu_1604" + pool: + vmImage: "ubuntu-16.04" + strategy: + matrix: + Python36: + python.version: "3.6" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - bash: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + software-properties-common + sudo add-apt-repository -y ppa:alex-p/tesseract-ocr + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + ghostscript \ + img2pdf \ + libexempi3 \ + libffi-dev \ + liblept5 \ + libsm6 libxext6 libxrender-dev \ + pngquant \ + poppler-utils \ + qpdf \ + tesseract-ocr \ + tesseract-ocr-deu \ + tesseract-ocr-eng \ + unpaper \ + zlib1g + displayName: "Install system packages" + - bash: | + curl https://bootstrap.pypa.io/get-pip.py | python3 + pip3 install -r requirements/main.txt -r requirements/test.txt . + displayName: "Install Python packages" + - bash: | + tesseract --version + qpdf --version + displayName: "Record versions" + - bash: | + # -n auto is slower on Linux and breaks on Python 3.8 + pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() + - task: PublishCodeCoverageResults@1 + inputs: + codeCoverageTool: Cobertura + summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" + - job: "macOS_Mojave" + pool: + vmImage: "macos-10.14" + strategy: + matrix: + Python37: + python.version: "3.7" + Python38: + python.version: "3.8" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - bash: | + brew update + brew install \ + exempi \ + ghostscript \ + jbig2enc \ + leptonica \ + openjpeg \ + pngquant \ + qpdf \ + tesseract \ + unpaper + displayName: "Install system packages" + - bash: | + pip3 install --upgrade pip + pip3 install -r requirements/main.txt -r requirements/test.txt . + displayName: "Install Python packages" + - bash: | + tesseract --version + qpdf --version + displayName: "Record versions" + - bash: pytest -nauto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() + - task: PublishCodeCoverageResults@1 + inputs: + codeCoverageTool: Cobertura + summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" + + - stage: "Artifacts" + jobs: + - job: "sdist_wheel" + pool: + vmImage: "ubuntu-18.04" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "3.7" + - bash: | + python -m pip install --upgrade pip wheel + python setup.py sdist bdist_wheel + - publish: dist + artifact: sdist_wheel + + - stage: "Deploy" + jobs: + - deployment: "PyPI" + pool: + vmImage: "ubuntu-18.04" + environment: "deploy" + strategy: + runOnce: + deploy: + steps: + - download: current + artifact: sdist_wheel + - script: | + mkdir -p dist + mv $(Pipeline.Workspace)/sdist_wheel/* dist + displayName: "Move dist files" + - task: UsePythonVersion@0 + inputs: + versionSpec: "3.8" + architecture: x64 + - script: | + pip3 install --upgrade pip wheel twine + python setup.py sdist bdist_wheel + displayName: "Generate artifacts" + - script: | + cat <.pypirc + [distutils] + index-servers = + pypi + + [pypi] + username: __token__ + password: $(TOKEN_PYPI) + + FILE + displayName: "Generate PyPI auth file" + - script: | + python -m twine upload --config-file .pypirc dist/* + displayName: "Upload to PyPI" + condition: and(succeeded(), startsWith(variables['Build.SourceBranch'], 'refs/tags/')) + - script: | + curl -X POST -d "token=$(TOKEN_RTD)" https://readthedocs.org/api/v2/webhook/pikepdf/39557/ + displayName: "Trigger ReadTheDocs" + condition: and(succeeded(), or(startsWith(variables['Build.SourceBranch'], 'refs/tags/'), startsWith(variables['Build.SourceBranch'], 'refs/heads/master'))) From 5e2a7f8a56bb010de44fd96b12d96dc5b900c7cb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 15:04:37 -0800 Subject: [PATCH 270/880] tests: speed up several slow tests --- debian/copyright | 4 +- .../pdf.bin | Bin 0 -> 3501 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 30 ++++++++++++++ .../pdf.bin | Bin 0 -> 2962 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 2 + .../pdf.bin | Bin 0 -> 4068 bytes .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 27 +++++++++++++ tests/cache/manifest.jsonl | 3 ++ tests/resources/3small.pdf | Bin 0 -> 165787 bytes tests/test_main.py | 37 +++++++++++------- tests/test_tess4.py | 24 ------------ 17 files changed, 90 insertions(+), 40 deletions(-) create mode 100644 tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin create mode 100644 tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin create mode 100644 tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin create mode 100644 tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin create mode 100644 tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin create mode 100644 tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin create mode 100644 tests/resources/3small.pdf diff --git a/debian/copyright b/debian/copyright index 8c190b4d..62a52737 100644 --- a/debian/copyright +++ b/debian/copyright @@ -79,7 +79,7 @@ Copyright: held by the contributors to the Wikipedia article "Optical character (epson.pdf generated from Wikipedia article as of 2016-09-14) License: CC-BY-SA-3.0 -Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf +Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf tests/resources/3small.pdf Copyright: (C) 2005 Ellywa License: GFDL-1.2+ or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0 @@ -87,7 +87,7 @@ Files: tests/resources/overlay.pdf Copyright: (C) 2017 Max Anderson License: Expat -Files: tests/resources/baiona*.png +Files: tests/resources/baiona*.png tests/resources/3small.pdf Copyright: (C) 2014 Euskaldunaa License: CC-BY-SA-4.0 diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..bb37acb288eface63d97baf21c3c041f339cebdc GIT binary patch literal 3501 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+q zpl58VU}R{lXKDbhhn@5DN>cMmiWLk&@6O{IZ($3%|LSlH!R|e*WUoeC6`;Tsa?4RQa}V zhStWX@1HXczVJrEK8`8V^X=#U@b6zAX4Y}!{d>IRujibON&Za>9FA6-@88A0Uv1tE zRrCKvk`Jfd{d2vxSbpYWBlSeHD<0}|&i|14&~s*<{m%D!{~pQteRh@9o>E&Bq~y6} zq1U?5F0PsLRx4E6XS|d!7FEb?*s_b}lQziukho@vRUu$~er#-LJc@K-)tqC)a zzq#~#qUR~zNfir=ICa&v3LCC1T*9M!#8sSa?v2ORB@<`;a5=E9W#i`n?M{g=N>m%l9v*_HmECn3y40}@2d3v ztHdee6;XA~kqcAwR6O+gY_}e{!M3e=uGe;#t!}S2EPXFwZNV_hAUbpV+$~>MT0cv@ zoOF8ct#jYs-!j!c{XZ|;CN~OXq*R>ajGoQB=#u z-sh&@?VWo=6U#ERgU^U}6o0=pOIY^!bH(^f?Zs2g51+dD*H?ktI&@L5s^lHJE`<8{x?EpggMYotXk^ zU4x2+AaKUhcg`=(D^V~+sI{udG3+(8w8YSBWMqILW^7<)hA^+BC^0i9wFug!3eK!b1%-lw rf__kbeu)Ao62N6FsF$GtjuUX#s<5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a|w zqGw>BU}&LdU;wU$o%8cbQu9iR6%0YrK_C-?pao$uQY{Fo2w{a~w1R$si<^ReNNPoi zg1%>NVtT3*l=g(u&fpBG0I9zeY-|+t({v4V4HQ64hagZ%tNqfd9RAbUe;|+LcCi&Q$A%+j+md!cD8XN+b+ZM_@iuopD^Op2uHkvVnW|L zGX>Oy0>yn0IC%A)^Gowe6bzBdB}3E(T|iNON@;Rxk%C4@YH@LDQDSn5f{C7?o&mTz zaxO|uEXmBzb4e^oRnTy;GBPlG&D9eG&IvSuuwNJPzS3F$t=l9Rj_e$1=THx zSgR;XP2)09Ff`%5QUM$%;P!KINn%k6IM9sDjLo@JRbBnvxB#;%*Si1! literal 0 HcmV?d00001 diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..d4784957 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.0 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..2df091b8 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,2 @@ +Covfefe is a perfectly cromulent word. + \ No newline at end of file diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 0000000000000000000000000000000000000000..54fb87cb2fafd22517036072e9d27cb488663273 GIT binary patch literal 4068 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+q zpl4v9U}$2jXJi1bhn@5DN>cMmiWLk&@6O{IZ($3%1n7G)H8Fg=VfREVAe{@y$Ny50YG-+bI3ZB(}Tp46{TzrS7H9$wMP zyZ!g}v%6AaxNiU5|F-{adP4#80jK|V442k_vEFumUR}N1e}9Gt52n98|MuT)u?r%* z9{u~?X5ii!aOtZ-o%_0MgO*X zxc~bi7<(;1x9iZyy!CoQXV$OG%U^g@N~((afO_TnL#K6ZUN8P=H%D(uM}Ka}oJ-q9 zx9aM0sV0Bf$MmV|>pia-r*FLJIO}b+`GgbAKZB~@Kat<6ubz5Ww(dZXd4O(*ZB z^d0^3I@jdp;(Nb-TDaem)Hoh>dZ$QdaH&@4^>tj=Sl<6wwY^LOIzED6!P@^o?5!xgvm?jCVm z^>~q~KqTinp9xz8Qar5Ya2?$!d%{B~|Nha>H;uUbS^Zm&IBiPWc|1*qZDOailxW1s zI*pr_ixyX}dD3^VU2pa4pOe?#-ViV&f7+fK(TjZIt|Y~Ix7}`yF+P=)Idj^yyt7Xp zcI>&Zcl)niW8I&*$traRgI)HWY@5tdb+`Oo3%gLkyO!Wy|4H|iOukK@>^WtRwYz{z+3_(}GPKFQ^G79Yzo+dhx=W>Uoy zuh>hQ%j2$3te)}h?CElX}sc~RF_)>@KdfHj`c-X2- zty_}pfs?Y8`L^Kv{Meai?aE<5wHx_ow;_U$>ZbGJ@CZe+OY zd#3wpMcyf?%zNi#y0jXZUp&ortMJ>nuHL6HD>o%UGB^e>|4QkZL!4CNlWEDPp+HAC@ST0chhM$45WcRsFjB-1_)q^GtQGw@0qB zTKp<}x~L>S!%lOaZFq3pj1yPmPwU*3ZeRYYY0Z*p3(ktTPcDc`y}LJ@W3u(e6OT7N zk8g|oZEWn2TYsQY{@F5F{T-X!s_S-_UG9;*w0+gGyN1uMd(8X$A?VxTj#cYxQ&sYc z<*$90DX36fD95mMZ}-K*tAF>J`sH+r8^>gC{?f7R&)zdTtv|Ph9kB9wwn%J|VXy3p zx%JO4=AQcDyeImHd{oP1A&i>B2uDo;D%$nEGgCmFGEi+21TL@jo%2icN)!x{S~`ZP zUBrN*{FKt<)FK6qkksPh)S|@X5(N`ILp=j<1J}7IHL)Z!KhGtxBvnDf#mdOQ(9+Pz zz|zpz(9qCA*T6#Ez(5_WG9W>Du;0URgb9(i#| UVo?b=(2UJZjJZ@*UH#p-0LlCvz5oCK literal 0 HcmV?d00001 diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..d4784957 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.0 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..522e4174 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,27 @@ +Linzensoep a la Waterman + + + +4 ons linzen + +3 liter water + +3 uien + +bloem, boter + +2 kopjes melk + +laurier, kruidnagel, kerrie, zout + +De linzgen wassen en in -l liter kokend wa- +ter 1 dag laten weken, 2 liter water bij +de linzen voegen, zonder het water waarin +ze geweekt zijn af te gieten, De helft van +de uien bakken met laurier en Kruicdnagel. +Alle uien, kerrie en gout bij de linzen +voegen, Alles aan de kook brengen,. Van de +bloem met boter en melk een papje maken en +verder afmaken met de soep, Als de linzen +gaar Zijn is de soep klaar. + \ No newline at end of file diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 73520981..05a0e86c 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -66,3 +66,6 @@ {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.4", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} diff --git a/tests/resources/3small.pdf b/tests/resources/3small.pdf new file mode 100644 index 0000000000000000000000000000000000000000..7d2824969672d463c127c24d4fcc8280e93252d5 GIT binary patch literal 165787 zcmY!laBjI^|a=L@R(y zG%_$yFfugOGc+<)fSMhUSd^4JdnRFz;4xd&d+v`I~Rn)?%t1{ckhpZyB& zDQYP>pG$PSUbyB3zlTe6@jbSLPqmZ7-^$7N$g>I`$lG9Yza)X-!@R=o=zC`~_lYg*>(kYxvVsG6Vuv_qIB&}@zWc>`^yDZUG~w;ym7br3TTSrU5ScS#&` z!Pm8!^G``>sCXB>I?!OU$os@8fz3O&T|9DMEiXXmX1;y#(OHR8xMoyDTkdN4V_#L% zAaHTTt&XX8uU(7PF)nOcl)~_Wf8Bu$=0i?*0vM!g*2$i`C3omp|9+*oWAge>HiTHE z-ePE3I!ADR;qEoO{Fh%@i#J##)!gJU6hw$~!TW2P)?EZP>IJI+;I?Llf z*)e=@zi#>^N9FyGFs1CPF%{O5TnRj}(cgK+kDn>Gk@+h9e~Mpe?t9U?Kfn4mMBn<& z_~F+V-vg_6OQ>F-l630Uv#X~q=h)QWFf144NL1+hb)DOcS?J)sQ%-CCG?X+=&D;34 z&S68m$Mzo%y|W4f1aeq(mFB5EDd=vT`Ks@jQ|K%=qpEP0n)1!p7YdncU2!x0t=z2U z_K~sRpWF{i1~CJfy$rhAUP~qFi$#j7o1eAKsBg%g=w=f$Hye^ssg7x}*G`YmgrLve?*&IIl`t@Zk{|NZDse^bq$9=lQ> zcW8b1--k8*6%22s{{7v!{QcK$!GBkLTI$!1fep9i->hEoh7Ro>VpLe!~LFeg{4HF7NJJ|x_#9qt3 z`1ksR>fG<__txK^ru#A4oPl{`fPicId%4pL2N-s%u!hfmpjX8g@?HBwH~@0XTH$7CoYLC00e=(%>Q4u59<4IU(^~ zRQ2&)j&8dfxLHN_KD7LNHhT7qoe4Y-tBY>kB47D&mLTtfDxQ~fzq5<$9aNRxwMLprUqs{!Q@T#v zuN!NHY9>~mDOFRv6aD|=q3F^bRxb0~GApjmwc_9L_Qtj?oZ{cu<1Y8t1<#ne;tnk2 zb+Z{{osM3+tRyC(^7G_1HedYaG#Uzoe9huWn38&&(Xz}&Z$s3(9r{x*YV$lSTlBwn z_wr3%E>R4NuT+WG%&~gxn8>x~z{~x+E*^QIYxSvL>x=nw!}O2GJh&4?!uNc+bgZi6 zk`voJj|PkDIxQ(mZto@i)@@xp-*oG-Xg2MMq6ajdU3vMg|9ZuR#9JAz=a^(>&%dj} z;3i-2kH0_1$$oWjf9BbqC2|I*0>AVoJZ^HIc21cwJasY`gK2w?SDl!Qe9a7NUnhCb z&3Qo*e^;BFdwH&BL1;|K>*M$JcF8et@T~eF&6VD#uxGQYD7W~bO;+iBofp%!US3#l zlg}IOHT{HOFyrDK8Q1yK1Q`sN=hQu)@mTs!hUL{~e)G7@?(cu#^@%-F=ta0$c~b5k z8Anlrtl3uHPV7RwQ*`93clwI|ub6gVYO%@k*BA2ra_0X0`eWurIRmDG-_vIntCSp? zGR$8aAtE>1yd64U&$ZE!X`)>@DNXt)4vRdzf8r zPCuT%XYPakRr9ys*)!v!y`RN#0S<XzKdiPc_T63*@&*P5xBk5jkny ziw?!u`Rl!2HgPasJ$`snQ=Q1#hINyl#jsp`IQesOfnejs)?JT0ZpjNt{O!vT)%~`G zY0!0X#I~&w)c)ZcOq`c%s z>h~$_7vEbovE{AHT~e6Vrmyw$ykU2cgkJWfmUTI+)l5zPgmkfOkQZI+_bb+{Df8?$ z_jUuO1*fJ2E#qE} zgKEZ1i=J(lKH|@2QJ+xaZEl~voP|ejWAgLb#}D3%Ow8r@<#_bB>dyV9?#B%j4J!k! zm=s>u&po31Sz@7L?_T}6mZc{T7oUBxC~tzd*KyI!bzcq{mapMc6y#0Fdk}v6ivRb- zFS%S!ZtM+91?^9WZkYD=?!~Tdo3ks=a;n9we0;t>Eqq<=0Ri5G4g;nI`rAD!zaKGj z>XF>A|MT?g_L&d*m|l3TI61K}RVsXM3ENHY_x@ey7@ZyFXirO56!@-Z`tJEk9^QnN z1s-KFyA-w_nWM~TA*^f~+pPWQ-uI7|CoAqU89OOR9k{kAwlBN2g~7&Wa+{%v<4o2D z#^35&jh`=doT3^To4G@1tNfeLiYxv_fReqyLQsgzDz@oT^S6ve9E0y+Do{U6_(!?X6{`$=jh~6C)K0VpLxF6 z{OP-&M$)VPOM*EUy*cE6KbR`MT0%tIm}#?BV$-z^Hk&56R_?lGB~zPP`~7H;U+k|0 z_kVA`DoGt+JhoK2yy`&T+1;|q!VCs_OK-_dXL_61wtCgZ5KpK39e4HDY?pkdKYNk0 z_37#B?Y3K2v9KTT{pKSXzAR_y)`XskU(2f|eNBjIlXB1E+R~A=_G)|ng~v)_Mtb3Q zZ_S-Fr~0}_MWds`)9`0su4S;kiM;&${Jy(~b|}3&d(8b^Z>mD>1i9(Qz0!o8yH^>B z2dv~3_;$;U;nnr`Hw!CmqW|p)mJqYARIL9t$^ZN2fY0*(&#qr5I#Jf4jPuLX?6bSB z+ zk5ulSxBUm!-6-jk^x71-d_%xnnVpO!dp}?MHaW~K-s7=c-Hr(deCO}qakkY?$a3bH z7rfJ_CW&SUY`wbl8t>m!o0w}~JdJsrG&b9)7f8y?*Rx-v_UHUk%Y=T08@apQcJ8eC z*k5%2^wWvAnWOs5^7>Eu_@6)5ds(>3al(R?32Zw)9Q4h;?7om+XVb)s-YIKEtulgE zPYN)z=xco$Hw2{RK60_kvsluS|3pH}lsg;V9lYVOwjq6O zLiNvc{{KGy@t*(2Hdk;5%hZQQe_y+yH*eOd8|>55IIpIYgI zRbTC=?a%*zikJ2?v%F|r9^B2&us?ag%E-g5iw3!kmk3?XYO?ix<3 zif6E&E|l&#dDW!YA}K5Hes_np^X>Q7OqMbZ+Wu5w2K(KwH!eB3i3g_#E|ELf=T#zL z7Og7T8<}i0NhtY(Znw7tV|AqAFzy}M>e9Il^In{nInpobr$@vel= zNvc825w&dlcC20dt|95nsnb&Py4Jn9@S4FQI$!$Ohu{gjQ)^@bg5OQ#3Tk|o+gobc zT~hFFjn~ObhM&K!wcNbyXw&!li4$g@?AcMtuz324q`f8{i&+CokD0tK`FAy5eWCig zug68UzwC=-^O@}+lfbdw<&Ex+TW#xZx9h!`;J74FMl*_3G-_cc$s>jyr#*N|s4TFp6=tgsq2$gSp?k8B;kK zJ~ABU_m@u>n~*25TH+F`#e2iZmzMWS)H!EVgBEw|KGcTlX{g!p>Pt6cq*%ZRM z?~xH-UHNM!3ww*Cq8@_-fg$OBGiLIvIIgXG|HhR$(Q2G@%~ib2DbkOoGNj|uaa&z3x9!^YqiKQXu2;O;sDlatCSi4FXI zzh1t%%chnXF-Ly2FvF}>nIT3x@p;+IPkx%K8a*!BvTn2fk`vqswYpUa^BiBeHCysC zT#0(a|wn#Z7$<`P?O>>^`FfWpc`ImfE%%L+kcEnA(t6XQ` zmJ}lyIEUF6L^bf&ztK>4!AxpRP5#e&qiBYqRIK6xYW}RN5x|mHd+MM5E;R=LIQj3Te(sEV6HZ z`kk3^;DW}r(@xtS3#}+Ud@U^Hj*;*Ap128{H9s*$TEmiVnZ`Pfqtq%7~m4`aH+=`-zV~i+9_4ez!H^>9KqGMcS*+C5Yi?OIO~> zoh(=Oex9$twvg%3=l|Oy(^sdjtNHwQYjnb+E%Hwkew_F>Q@yN_$JaY6_)b_!d%(X_ zv&4MS`{^&-Dv2oM- zJz;arXDrN26F<6bk#x$<4QIQySgT1$)E-}Ez0`v%p_X}0Jt?|92H>rafy4dG=s#$1VN`C&fzqx0LePJ1-aa zZ94Fw=Ulj;Qf}59K`HUYx6X$BSp4h!zQoiMiays(L@QBl6o($7%oIe!4PkVYh^zx(QTnGPt5_`OA8V6@Wv1HYw|K2hSuS794 zv@UV(xY2r#AvZa;Nq5~@#}{wToZ7B?X~yyAvnBpIdj5V_>z%@~VV|V!$9jEDS4C}x z53jbXv~+lNow;CE88G{1I-m2m((~_K|LpC5Vz@Z|!dZK}EtW=15l>zCSDf^GaYn^^ z>xW-&a!p$prz|M-EQ{RssQL(dcumUH^R-Fa=3R_0xo&Hssl@ccu1)UdpG}!5fjkV8 z9u+CNv@8+WqIXfXJUol%h&*Fdk+c;K`{89Ctj*AM*}G>#!emMIwJg1- zi}fom8vw(aN=l=`G{@*>MxZ{cf8 z_UhNXnUMG6Zq7{h4LA3s7VHRjV6}bBcJ;rMj~GM7uIKhP920AHoL*R%{-z^3eBL&y3shb6t8EkG?b3 zeJJZ7X{)?^_5%`gyKJl8R zUfaSNlDVdA=9tQ}^&5-HxzDjvO*eZUZFoH8wf4Og-B-9cmb30_lP=X|_^>&YlVj?i z6s`7c7hfJ;x9ZC;w?hq!XL{c{rSy%X%(Qf6!y*}mpf>dt9K7c|4%V&ZPugbvc4<{C zXH$cko~-yr_YJFA_a%u29%Hznwwgn*vMDFKZCYS*=N$RV)1sd_2?XqNe9B?lw3FA) zLT$E8Uc+RS)vW?btX1DNGj7exI^UBR+ANjL>3E>w#_svb8<`IA#h()RWWi}r^YMUS zi`S%cduK0S0knk+&{d6a?%+Savr6zE1E2sIPaEF#`GI6{N}Xn zIAq-Uh@(m6a9nH9!_CyB`&9Q)Q^}r6Tw|P_Cv)KBOx4JFNi3|=`BEb5Y!~z} ztO@&ZX{P9G28ZChO>w-Pk*_{*U~lR{l;sKlYba2 z_dCvNa`c+bvy2IRUD<5)MS3#K9!3X_tu*A(|07&eu$L=dEv)59NQ!x zU0`~r;eEOP*pqjGf;>UnEMxi;N{?%<6pQ=u z_~8F})>Rk33h5~<=iZkgU@6POAS^Ha%3(#1X-2WY=UC=vQWp}Z>?=vRTj*r4;eo;0 z2D#|mCcd~8>>sYWyU*Xl&Z9qFy6(}*i%bl2G)|vw5s+MVT(0ffgxcvRq}%yqj2DN< zZaUrhpkUhWLz|eKvieMo_DwH4Z~rv?G1GMBx-+ZZitH)jTYN}S`bf``C~L=TFaN^E z&WnxiJeGzm?VJZ4uGZg|n>Srx$84Vuhm2mj*_h@XmY>ha$WXbU+W~1-JA$X0Ke@wum1*aO~%`T$wz^D~C_dx-0jI zx4!Ceg0x=H>xbQEC9k|tV3@Px!I#9AQx9}jI<+3l*_v#0P~0~&SFYN|LUls>J5^&c}%8sYbBJ;(@D?6Au zxGKM~w4}3^%oX_XOZ?@Y)RyiXrn*b1(b{_qKK>K*bL!~1W_jK*+H2O6n}$5~*DYqb z$z&+7Ffwu+=eqIm+|%d1F=neSt4P(g9n}h}C_E+qp7&w$MXL<$h9@_?G=(p+s`p#Y zno_}XFrhSH27AG~w$B9_8vD|>3fC+>k+HsEQ!wvcBNetcEz4x=RgcYX7x*0`3J) zu~i2?^~nj`;i>=l;mV^X!DO@cEXhL(oKl-sPkjF3qOG4y{H!ECH}Q-P&fRhWN+qB2 zgQB9MHr@Z<`7OEVw%7jq#=1IdTr)Tvn#y+Y2VblBlCfv&!UboXBnr+XhBIEgx%c$^ zs#hz2IGlH3{>XW!rL26NN2#L942KImR&mSx{2t5`H~*=6@>s1z#g&*A&B#)rHcPlzg#B^TLBlI<9Dt6DL}* zs)g_D;&a~{PG9W2%vzt8xKU`&0txeVMG7J_Iu=M;&;4ZMq`*4u+QRl2iHte11zU92 zFm-&K`9k2c^?$hwZ(H{7x$$eWLR0vkQ_rW#G;Wl6WO}F2-a<8s!;>ej7A1q?q^!FfljM{_#=jr z@H0G*AAL-^&enIeGU{&l$HN>hzjJ?C8tHkST;aXtnjo9h@#pW?aj#~0G;7LxW{XR= zna}Omb9~RuigG=>pJE?>PpkLlFyn9DGQlqJ*GIlsKfA}5ie{I1F|*Axx*8=J$9QHA z1H+w_>?t+XwhT)vStXvoN0&W%!}u z#-ueLxDcIh5CliB&GVsqI$esdAU8#W8yF?PP${r_YAeVuLIa|)~< z8cEomtI%#bV4y4e*XtmIwBZ|x0}OQ^**>(NRq$k~P}{}qzy3zR=YP)qbzkrOad8V+ zskpX*#r^KQ+N%wAicgnhJE}c4df}SgBYN<_f(@%0I_9opShSbDXFj9Oao0m_^E|tk zUHoS{`QO{`V&AV`W)@mlce%ShE`04Qb(=p|DvtObdYa(++{7Z3h1qz|1IyY2a(s*~ ziClXwyr`9WZJNNW_sD@U<*im7B{{0De%J(m^<<*j~aEZB=|7UUGfpS3xma_~33cp3|JNs|lzGSxPUBICWIsSE$ zp2>R`R;3<|J9<6O$vz)=f zLdUI1f$1z8`>}1gr*=$Hyv$k{mJ|N2a;e(xC%$3foGpu_W9r?_X1|!?98t(#QSrs1 zHpDll$k2`X_XNSlIc&D>nuUCc&wlE#I(`kPp^)iRxaoevC(|$!GjBS@1LwYJw3m9=7A44QfsbeTv_~M z;jKk2DXL4}&RukPZsOxs$B)yWD#trLW32GDcyNtL{fct7*!0_627kpGK4?y8XporU zpp?Pk5V71$_tt|amu~3X7QFs%{r?Gif{dTbLX9+e7Zev6dN_%jubaD{z5i^3tYM3# z`NaDLR~Com2j+O$vK%N**rL5S^j-g!SxJ#X%P$#=_gXmi+A1>4<6=0-u>O$RJBi$@ zEgl)!ztU!Ut6ElP#ogF$eqx)p&k!>GI42t@T*jT&C+tuj_1#=Tz276pI8A;LjQW%_9q?xMHn)qLtAF;%4SxW za_q+;KQ;OE|193@uDpWAH?KJ{p3F`+>3L}^$-oe_R7`xTW@wS8uc`2*$cC~{g&VdO z7C#o^l{j?7#6*qf>y%e(y-xZmGQNByYz``}h?k zN_L-ol_B3U-|FCA;dOHis&jAd4lTPnZ^qO=EB)h|&pciH#fk6vXI4gr3sIgsH`uKg z$(U+d+Mx8a&u4i+y0p98l(`qV(hkhKkS1{A^Sfu=1+yNSDIF4(^4J}k;G4ysZP^)_$ts{Ae_z9%&w=ge>BMvY zg=bthw>ntpA9=#rR+sT=b@Bb;cfT2$w!}|yyCAaXU8fIo$&*J{ILzC-F8{Xgce^IC z{qA%w+xQReZn4cYamqEu4Ac2+?qGe5s= zI&JCoH*EhEWZpU~qwBi4){Wg-=!8W`bWrm>v&ouo`vM|1FEDQ^_f-2P!oZPKx44(# z#Wy$g5|KLHir7~t1M0uMd?KqiO=!Ze-7Gw@C*7af}5wJLSdD~=ne%Js0dkc2WWbpdt8rnQ_FSA0+ zhV{`087z1$^L^Uxmq;WBb!>j~a&dXnH;cl1>%Xil+1%i;VsFK~M`xtnXVh~QZCI^& zO@P6nsX@0obJit~)A^}l|1DC5#3OyrlD^}87zHs08F>TXWEiOBpN{JGPk&03lK z=d1pE)X9^Tar5SHEv=k=`%8a#c3qP9%s8@L-r~+=1OL8vJ5T@Z-W=(E=u&rb@^AOW zaf}<@>{k42S-X#CO8Sn#fG6rc`b?%%@*nrsf0R(Np1n)^|3rt+SD3W!*NN}U+g8G* z-?DV2kqm454{!6Ghu7rmO}>yM>nbY$_toy#Up)*Sezn*CbaY~6h^xX)W#iowgcr=? zuUe|OeMNNp)?K_;rZXOCH}DmxIM1S{Q6N|(Z8A%-^+aR&@7?vDiO;`WT*_bnUx)ik z_0{8-+8;423~ZMDQlG8jTw9XPx~0P|K=SkD=i;Xtb3e@b9{xRf*8h1mF%EpW_ZDt7 zQ+c5(%JYHqF)N4S*PoO7xlY)vZfg9pJmzuFN3lMmM$uh0GHv?ZdcClY3nVQ9|*QN_&C*!<4as~m$0VSZC^Qt3=9H`+RLR0eR$K;^&d$G!b@p3Uyqa=)zRbi<#qX`v4BZho zRyud+T`bC#c)C$iiM7qThv(ZO<8q}owtPuuF8J8E z_p?am8&q#nmI6Yky2>~EgU_okN5lXv!uYKEpYx?YnT!?_nPD|RpC&u>EJ%zsV_AQ6io$+vbJzbfgm%qLoW=e1?CGQjw)qnZC98i6oRs%kXF!}wSaWacumzio=-KsVFcD@VE7i&6IUoZUg z?gg_62mk1Mp4cEMC3WIm(glV)wUhr&G=G&-Ql`V5a9Leiy62viPhq?S4?|Pzo*DPw zTkgGFxxjYMlCQNp4m_-9D7(~BeaT07N5u|K$%uVhSJk~dqor=wbV~QX_@-Ketmd=1 z2Bx<Y3YL4`O}u3luUE8fSKK09(eS-&QRrl=y%Jvo>wUjE ztvUa}oiS}bi$KVmvu3^z_FPosICER8(O4~f<6eP~wxc=uz%!XKNZ{+CD0l?|=S%h~k#h@CRS#jm$o z*Tl`@HoJRA{LKo(%rkRS_>CV)oS1HhAmcm^~ZNSYn{YaNOAn~QJ!4L@JppZ zq2Q(5Mop%r?TwsFhr?tTrf=z7eJ?K6yCQVcN^9ka6@iNP4zn=tTDvNHUr?5tg3%+r zSM|FL>=z%NDAB#$Y|-u`fmd`t6{dG8mNZUodU`YcM{`!B!Rme{4z{z_s|;2a^*IQO zX1aIzXUP8B^f`IErddn%*O}+<)$cT{UAshv_t~O%lP-JA(Kpbx%$e|hX9e@`Nbg;} z_D^gTH0!kFXY#oVwuW1O{N2atB)D#`rD*!Di{j@eEPGw3Bb)Rn{``E$d5`7RC1l0S zd~ogAq5Wple13dYeD5!JmqkJ6Z-3Sw?q`B5E30yk?|qXuXKL$o{zv+*v)Mkxom){I z$oz}1p=7G^_VY{}SDx$7R=be*&qMgz)s3ekPH5>)jy=NJ;+dcF*P_3XLFZqTU0&(- zyBt5A-5$(-cJ9$Z^%!rtu)0Y5dtdfu_E#0jt=;7OAoI&12Bt3s$!FKp<=pyUeYDSH zW6mZOv&Sc*c7C|9a{syBAJ0NcHmXZER%ptx?OJOX6U4o`wz}pnQ*vbVpDMx5*)Ack zraip*)S)FkJ<>x>dge8aMb4ea?z2~CTKHH!RQND?@1G>67ikY-7B+q5H+cW73`1mG|DikDm+e^U3)pS$++^4W1eUvVAfeSK{!M@`_SUHu=HpFL{pbN!@$!IM;0>IbPcNeYiH0ZGJ4wZ$^)^VFF z@A2n!cw%;b+j*uHd%L83td7hvTqZ0TQEeb8v9zmI(2O^FRfLzu$FzBG{Sr9LSM6Ho z&%^ZDBYN^9-Q8AOW|(t)*}81n!hM_xfsZQeugu`sUeU_$q}G1DI!}6Xwnnepo?~6R zOrPwSaBD{Yx)ODRhkH)#GvH|1Wb)#5suM$(yvqg2i*NQlZBdh#`SJ6(kjrGQE6*83 z&SdfZa6S{b-e=2FUw5@Px(;kRm5v^WT|YD7q32X@m6G)*|8H3&&G}`^(O)g<8x~0X ze5khNZtb$_ZR_LjKjPO)GB$aq!*Xi-lCPEwZIS%+23=H==!Z=k&=7zTaj(A({8+XNtI6SXjTZ!~Coh&yJRx zz2$SVnVOLv0*_+&xPV;)( z4=l@XzhYS|JdF_9`^z!7@+AiB%^7rq%;XUJf(qg}ih}zPIzl$#zu>8ssmfd${`mu&9 z3pTnP=b5K{opr{Q!(H0Z@3va7wyG*<^8VOkx^OG++nUth&nxdLGiAG8*u4Fxw>?X4 zoPfkr|GJx3)sH5sygB?wvHxJX(QknU#nrvprblG{eENKDS&M>(^$zXbCp9DdH_h4D zx&GI7-ly;RCvCfYcd}!D@u8}j`(HJC)ikaT{AVJb7?Ji@M2x9+$HU~x&JuYer-I81 zpBtO;9QCzgeY}1C^cS2V_YW+*RTySHZ{|go$7$bpBza}DEhzsP{8qo+*j?&q_r75B zC7TjvInRH_udtqXazZ?#z{R<;3^mu+7_DtTn!erkniGTk5=G9y-KrbsmZ$v_m;Tpz zGSPg;r^${P@6WvtPQT=ErR4E;p~;e}p7Wkv-u|=MLhix6aE3J&ny;qt9FhC_DWp2s zNg(yDT@2UJ>ze)Lha66B6aKovm+S4LM8&?VcNerA4O$cb{#nGGJv>cflKj^^{AKti zx3PaO{MS-hGJj^5vi@<6vtEqR8x#*skvkT$y?$5SPNu$|8#@|2=lpu6BiU z&vA=2>zO;ewW>uAeCo9~>3(2!>He8Mf&3?Ile-TlIK>>i%D}*6elBU1evWS5Gv0Z& z>LvTOJWZP3d^76d)vGZ(Q}w|_X^eB7`>jxE4>du4xS zd`#Z0Z!ywsXU<5uo=Cme#wgL-vsdoO#3i{qS2uO+PngVq=Et|7;tfvce!8ET9{YbL z-|iQTCzV-y;!GU+R1R-+6MxvcGrieX{?oHhzbxt*4419l^8V%}8P@jZdHHX%cK&&- z?JRTaoIkTm=4;JxHix-2)k2duKW+0|yE$~_T#aW7IvpjN7JdHiVN&7OY{`+Fw*HS? zPVM{V{ttSLGj_CA7_Pd0Pk&Eofb@=&FLebq|=mKuJ)&V9)-!FV86osoAXw*m~_q zz)F!R)`GsJKid2Y*(Y?_>M=I8HgmGL^c-T43w@U>F=_n_Dy^d5zyIc` z=RT0X{kQ)tp;HXDoQgWPXZL=d&TB5FT)0@R>o3Rq%YPgh7?uWVtGKdmiCCggT5)#P zePf1uH%?X+D=^$TRJ3%-8lP6hNlx2Ot68ru6e~%x{JHysrH)uPV`kvDw@$pr9`MS) zOOE1kQaN2Y<7SmX#M=c|x7pM`u}Ku+Z#b3QDIsHcpKryh8$$L9`g?V2rZ?&wSpV+S zH;Js@S}fi5wsCro{VS50n$+ukwLja*sXV-SKrTt((noifkP}7z%KvATozHAgQcGZB z{~Gy4V5!KpZZEA=uc_NyE6+yUFlM-SuWkP&drS8Sr8I-DU->l-xt4T%PPF*G?MbQn zji7TpOu2qlPNHr%S$JfUs_v$7T=}WVS!!}EW6p{>T63c%-*NEE-k;BF!hh}>^ZJ?2 zAETzE*z;W3Tl4mRoS}Ir&!Snv?eDpl#qe4s)qi;XWy8#?Tl?g{Ix}Q&JyPMeFulLP zP&np;T!Mhs`B?SN!p%7o6F-)3H@dM}q54FC+NUhS+$626ws&DCKgQ6AwA7?-DKkRU5*%f}Xn={+vKR%5uoVWRU z>;L`v`|f|}Z_b`x`CsbsC-v$p786#U`+vRmX~X{h)WZ2DjT=6B2py}g(`n;zOG>eS zz_aYqlgG`!WNJM$r+dCWeOjf+GV{i1v(k@~*#oq_&iYL+@|?Aen@i?EvT?__|` z>F3qY8(zB4Q`=vVw8-zHw)H0C%f)A3)k%2FSN(Co?1*&MzOy}bOP{)UO{;o&(EiYC zj__l`t^&*6taOjlZszH>|9Ey%r3(lm@1zgW(w>LUd)hlUg30c zLC2%KjT2ZDG;5Et^OapKjhoUUP~iJ7a7uNm`>g*>SN|5uX0Cttqe{3f;idT^Gwl$k zhO8$%(jR}m{P%*hE&i_U(ze6m5%Liu|CVw^di&qkZ=VRb zNG0C+)$!hWzDVPSe;zuF&wbM*p7@ukr!OdvTy?#o`bs+w-&zA}?QpMIkM-tmc;xAQ zK>50D`|j#f`vlh-JovcKt6#PDSyRvC(7FE=+?i+eIlJwexo_6}q^;+QwoZI6?bNcl zUSjn-N6DAg%lt!(P4`}* z_C|G{1!3n>nF78pEx4$|%)a89{J#h5eNw!piQAsgXFZX2qk-|5so|M_ai&#?Q9;x0 z+}O^iI4yjKx|gob#`+m63Vvk#n``WOzi0LGM=?q}H|%xhkBgA1GT2bLpdj_W>1xt zoln)?z=Qn)<^>zVOOvvuZuoTdsM(F@)78%l?3ra{kQAu#L~yNU%cszCeHYW5>#@I{ z&bxa^yv^vvryHMt%rIpO47e|;Q~qe-JNbXIyE_lQ+&q)bdTPw7?nf$I$9WF~$S+)P zqPjn(S!b2A+|GsR3?X$Q$(y_&&t&piyD$V)yIliKT|G6i9ohWukmN$(Cusug{lOcJo-S=-(Urpm&Z{tS9RwRl|*5U7KWW z&Mi*fBK@`4=y$FBl@wchonWWONfTb0YPigjRKJ|C!~2PWxSQ)PW&Z>J{sgoC_dK2Z z#-udB^`^nr*WBmbE=^rt-(5a?znN3?M~!__PHhQiB=nft3ImjV6U-MgeBK{yne)ZC zr{=5V?t@D;ofk>$JtyyyRT$`%sLQ$Vsk;0Aj_(`|v-G>wb&NOs54^+WHq(2le&Vr| zklI&kg`|>MdgM-IukK7d6jt;l>PF4gu0OwL8fC=rB-yM#IB(Vr9YHopJ!R(@L#~UF z*C!Pp{(tiDq;T&oeeQc!G}qU@e7kt}roG|oKTSVkzv6<@t;~Qe?VQt#Uq1hN+h(^Y z+lqgh%=?UvaMT~)+g!}GoaJDgxs}P&N=;EC-YxIE7G3+w)9@smH91U@Dfn<>lH}qn z29xJ$Q&+Am)Z+N#R)a87BbxQqJ%2L3Q94g|rn@94MZ;6;j^lp=~_n+@_u;}~@diibM!I!)% zEcg~1Yt1ir+qSYx;TGx+ujrt)gRZlKg%n0NjvqY>&cNAhnpLwzHYD?1&ilBh>wlHS&Ez_$ z*T=}V|B^97ll&#Ne3J(%Dl@CrAOCG_|3f-&+k@~W_jM*aoGGr^VC}-7mLKdJ+r_i$ z&7Jm2rxYf~7EU&vPn(u64Qk!A%E2R0YN}T&hli3G_j1{qyHCUqHe5d&^KQP+r|G+m zcJnrDR^O=U-NUl|X`kq^^L%ey|1B_kT4Q)<*S&)a!mBwB=xQeBOK?nC7v1vGYE^{z z$5rMg|G%xgRNcQ+;A30gSEZ2q>AUwlZR(MUm^>k18SAnm3J- zf7mL=RFGYB;dT!n(}(I~ZhI8x&pNvB?nKpP^`Y0+8gbpI@SXH_=`wvY!`bS3RdW>0 z8WkK495YrqG}U3jOO35A%R)}6JUwdUvzAR;ZFY5HsKcr9_hNbr{w$CAbG4D}=W0u- zS0%+SI^vG>d+iBn*>(J%+=kAJ#teDYPd>{nKg|6*?Zc!tyMUE@Z`@;xOli^#dY7)m zQgGnU=Gls>o)dE#q?)82P0IGj2ok;B-p6w&@b(hPsI}KL%BH&K8H8V5u%w>#Px}io z27Nik8kZc z#}3b|NqOk7M8VF@?ZK=LL&j}y1zTP{UO92`ov+^+Uhdj+qsG!im2G0|kBWGnt;=Qp zOmO+5$uP6i_w211_vbs`6>R#j_|9YA<}%4$aTaFViwf7X{&%!`9AkGk_ucV-`3BW& z2kyVS@!$1A-Eynz)4OMto|`^@{l-6;SMHoY!#?4goW!N|zH6AS%-PX&@sHv|<_O-T zfT`2xggPcK6AL^P0Ddx5*a1G?EVZRm}I;w0WLpL`9j(RD-+g zum4!RzdE?5bY{|=gU>%&$mrcRW_m5jP=5E(pSxE>63ryG2h}C?9JO^?;8Q(kT|>u1 z+oFbIW4j49Rjlk1tOoZtHcQRX5V*R`PA72om(x3XWiO`5Fgz}sbmW)fGvTGR?{D!> z-o^J|r?4B-(*=(s)LU8Z0`p@T6N;1D3-%?mmQuLr`M~xl#;i`2;OgD4|rpe@IP#_k8n^^>P+F8rol( zyxpihtMxys+e&5MeZimpnUyi_oZgyao2wZQOjbX5pXv77 z)&2cjoY%hh&r$!jr7FoIsLp&fUpvqB6^(xG`A*lT*G|YQuYI@GN$9Yf{_)vuy_@D7 zk2HFmBv5!*{a5m?;^49shCe%Q#tB3wDQPk6&Q=nbXYVTMDRxPjkVmc%kDK8i~chclVtxFsN9Cq)0 z(k)et`>za8tWIIN1 z-9O95d?lgg(TbR#pLMQS|B>z8`hlnE@uSyD1%f7=G>(Xgw{j$(W^+@LR@I3*mf1kJ?cD_b;%h!3kjKBLvD*Gk7bJwq9KmGUovP&oG zZfVFF*|(QBK5`N-Q3=WSH5O`Nb!BJ!^j_zSi+|&vW*+zO?R9$F?y!BUl0UHH?UU$- zx#bh@8=5D*b+iBfuk!bwbuaZ5e9wD7So1ws+;B(lmzukrTbu$)RL;4ubCgf?{_s5Q zLCUAMB@Vy)dyT5^@c;kZuWwe~IPtUk#VM+7#S_k5@2yao!B*|gINvrm@3_SI)7NjF zJUZ_}{VQdLwA%HRE9Y8v=6$-=zvbayRmb+2z;ONOj?#c9%g+Zlw;f8N&(0l+Dhqvo z?of|i-p1W6b%jQaUkvW;)j9a!fMpt+)PlXD#%z(bFZ;I{Uz6lJbV+6_{~kHhEz@-- zoXU6EP<-*Z@Qw4cFKe#YyD??|TJ12mZOhL(Oy?HAp}A{KazORLt}Z^Y$1-AN&uYBq zpGb{VFgi7Vw#!NT@({kSX&=9Qe_pimNd7v-&vLhmU*BEBFRIyG*67CdrTgz^7Ut^H ztg+W`WnVatPWFMHuFZ=1W4YWe@gZg z%ib$pCTg{35{u)j%YO#wZ+jO!pG9Ct=F+LTk=oNGD(33gv8;OcRR8;J`J@n+7svYl zec|Juo8!CSypi)Wt&nS$tB?3eG`u)?ZW0^Q#P7Sl=KT6m*lWqM>4t}9bJ(w{)&tiI zYCI-g(!RR;`m=_YLAQ3in7!lS+;C~v`1`CE?nh3$me!t8J?Y@rNY80ZW$N{d?r(m= z=WM>$OvgeaXRoeR?b5A#VwD_ZpT_U)j(XGc+CJsISH8`Zj=GbQF?Jc}wkjT%u0Op^ zs3C?|BIO3h366n6S0#=s@XcGUgT z(HD|;e_q`i(%L)g4d>U#hZwWVHT;a!Oq^TTbne~v<8v;Z%dJ{=XtvJzyPw;Ht2{0` zwC9GFr<^|XVT$pI;N-R$|BIsY(=%_JW9#>7eAoYL_t#krKC2qqJ_z^7X=dDb^O@n% zH--2o7qZ+H3yd_%I(0LJcja%ARlBpKT_;_MMM~k;j@T**znsL303W8D)AKL5MnAvw zrs&`+zAJvorUEa%9(ymzxmEhif`k>%Ctm#~o0{#O-05br=!(|VprvY_U5ifSD4m!f z&E@sz@axvg4*6S;yFX!Lc+xxLK~v*$7KaB;T-i~~CwNLtWfocdDQvC(?)rY-lFH!d zJ@pzXSMy&j?g>1r^PtSv;6=G(?q1mhzqIhgU$=vdsl{>PNGV((oIuyF| z_FSK-82n3+!N!zX85RiuS;l^a&nj~uCsdfrR-1dCu~u!m)p#~k+GtSq2!q3$rINq zq-TGb8}ax`?a|!ziyO8r;&5Gad#Rm^IeX+zLlcGt7g&ER&Yl_&t=_P;>D$p20efoV z5)O;yFTLw}alN+HS2ljh6&H^Cxke#RD8~<;F)uvaL&T~xBs~!TQ>8m9t-fyE4j;aLz`id`|+c)KZ>jM zCUMR)Vo1zbVpVMKhsbU1rtGT5zUfN^g`St)^SR3V;73<(w2!q)P-pkVZ1;bR zYu3xgzIxGG<^Espox=YI4XpQUFP551TQDs+ng6ro?}BR*-@oRpnLTOu`y{z!UIn`p zNnzPq9w~+rCx%l?7#QA872>-+yOW1qS$6lDtQ!90@S+vP>AwEk66ehC56acvJ7rNp z^p#zbFFblS9Ata`ykbsL{GRY9_YW_Y{h{o(?*Z>TH-%>)DwPvuK^4ld*;6Vu!Eamz&PZVlRDJ zuYOCRYhCt_$5!*-Us5V}uGcvoyztX|(SJ$&*AIV-@QKOepXbJKmq~K#*&xOBdZ#kn zHvQWBc2-{Jp>uAF5Ai%|aa*Uh#rkn+i-WbSo)CkvzF}jS+_z6m)rWJEMmNs?I9+^x+QGbl1LlVIi;J}uPkY?OTqVDSnU7%-^HtR)SGek}9Q~^; zCmHzUzFIfOjcKnr+u|+pr+b7ua++KYT{C*KZj$89oo7`J)$smgJ7u`dfaP`NzboQ? zrC;p%>-w1I`7t;OEL@j;tFP){MuMu4OUU)|x%O8JC1*WO*fQBryWnJ>3P%GsJL3V? z^2;kGo;RHOyzT@?RpG|XiHbA+S2HyTI5ot?NxD4$7nJ10))%!bGuB9vk^9~Kl^oh@ z6nU!_yo#A{ZLxS#!1=g<(g)`&#n^4uGO*oF*y?**CtCO~pPWnC-tFx#6P9#tnx1im zsjaMDi-k{yL1FnnzLL7SgKYKzyPoc-I&@l@$Im_CDtrF2dlu1*jGJl`dE!16PYYyv zsac=9cEzJBJ)Lu<1)pY3Q#_Y3(b@L;w&#-_`eLt^?b$!!+ReKiA>OAhCFjTPobcU* zA>(xO&b3;LrSgl*J_d%%$N!YPQ>vHvs$*sGf*0Ro4kR%*7_OW8=)C1($JyydTgCVJ ze~#OG=UvNZbk z3^ScJ%~-UK$z!hA!sovIw=Z5fx+5jq?AQ{)xJq$unLdV1Gyj(fX#FjmpLAuJX3VV% z8|Jh`{90GUAz)a~Ipu%V;W-lG0^zRx?@c!Hd2OE9ym0XnmF?#iE`J(-bY4ZX&y$)z z2mk+m{ye_+`{i|)`8RcYpI+_CHS1tG*RlCb3L-id9{=)>KaZ*% zT+!jD(KcDk;pKY!r{1oi;vsJ=ePpK?6kKMqtNeZ2y^s4^E2Hw6l=u403=f5`Z+Njd z-Q%!{=-r9ixzdi=w3YHT%urL}Jt@P==wiF3G4bL&6X)L(w}dPS(-QI8+O_Gcq=aeS zt2?Utdp7-?%Rl{qot^Na{N$t0m)yR_$C_Dr*!T8nuC7D7r#ThRDVlM;QtGzIs~J1w z$`@(snl8KO=5XhxqIC*JRxXjNfQeLqT^<$=bx_os}zSlMc?Pv7@@Z>6<1 zw|ugjQS{Pj53?_@DQ28nmj3g!WkAx4bCE_(Cpn7BgbHTzh_#>2s`k0hKEbfRJm~Z6 z45v-SJ5E0SA@5!PD(IJesb9~s*@2PbevhUEzWgb2_+Tbej+G2+-o8@TD%qiD+= zhUgia=P=$4`+jHF4gaK1zuzh9R{6cjvD)OZTjTd4`Bz;`_gvlf`OJ{)tNE$%^53Gx z&e83sH^#K|&U&b{zk2tv1w{+lbv`;+g?_E(l-;=LaYFEJwQpg`cNBDf#2FQ)K4(~A zulfG_w#IHZ_T0O{VFp!3e|~Pbz)(M@AXTd<{yqD=F!P>$y_Hj${j%JgFE6^jaM?C( zE|-d?XXlb;%s%+?%mLXwvbW#yRwOe{-}U*c^0hC$fo--%FL!$@GMxB+^ zwyojH5R2NbQkJ|>QuAf$7Y64Uie0NyzbufJvb_ zU3V&vU*hZG`#XD?Hdyen{hm>1derF5-3MQpJ-?rG@gEk} ziim3#d!{p&Q`O$SAvY;~&XzmJ&i)cvGvAW?SGHGXv~|h}CEMR8x4VX%f586qxLzjj zhv%C=q{;By%&-cvefD_Ca>fI1bT^d-uK4jh<}r&CTfl@c?FZdI3!)ls{yBKdhJoj| z@PzN)t65$hd7uANs!Zwdp5`juGc&B`2fW#T@m8GQiD2`Y$NFErFw`m5dh=X}*?{@? z`uaCkMsI$t58=PyE_(mZ+djpAf70*xus+l`OHX+B=g0oH^G|K_&whS%-@CSp=ARe~ zwnQG@8NRvntU`)Ntw!PIP31GHc)!n0ILO&)D)6C~HQk@Xc2>-G{U!AiIUJd5Q{P6e z4L*Ov=l0)O&8swduBf!^Uw5~4;ju`ITee4J7*?6S`Nvjdp1s{if!iiZrmwAf*zg-Ym zR{hF9L?^pzsn?EuJ5Ihiq353b=1ihGC&Rurw=Hh-m#(~+_42yUX^#YQjYY>Q>T_kPW8PN zZPdH`f`iKkADM=j>-(N~q~GV3NYBtT)cIdt8=}Bfb(s5D&ic!_hc~S{d3e^860uni z8qR3t`F+sx@;ZHbb?eulonGF_EZy0!6*$fWzFHnPN5cQan+s3o{VY5ERO!@>N1Tnp z=R_|ZzCLqC&oaB;KOW1UWYhiqzW+P(m-~uIlk(o%pZYSbvrfI^CiA()#kYi~`X*oO-F$;Dk{jndDbluHan0ygbWu&O7=zmFC!)fK@6BoP&I{?`HM7Xw zZyxWz;J7xkp<%PDz?%dNmlUPlDt#ekJqj_xEnrAa4Zqd84w!PrReA-oPb)=8Lj*=lqx)S09|wAI|W6f8C#t_bq<>d3L$#7RSP9F6O4DVw9}?Bs?$VwOSIclMf^6Ebhh9{fq z{F4rz&;G^8*>L|BB|yV) z@3WGGFF(C%?g#~I+?jonvC;X_WhI6aH$1-V4r6w3{)FDgzOgR7 zo~fbu)XzE7E?0>hn04mFvnA|L{=a&Ev+z(UTT#Qt&}(i>vbq{KtA2J=ZxJJxnRS$BVG=#Ju6`&SGK-HEXrYExFfEbwN^)R?>K zxzOBsjc=xXJa=_|l=uAb(kYEg*{8owRmiq`+*R~^$0I|=Ub`7x?K75_r~eU+tz+4+ zV5g&m(WLt(2ioVhCv2Z3d#LM5PnOUjzc)VrPZSFoe72nIel=LXdYddu>9-U8R=yb) zYoGTD6wa!D&E!@E2>OO-Zw^w{VyhOK74Fo)6Z&@-QdMc?|X`!XF8 z#NX&P{o45Yb9ceh{r~4&NGZEBd0}Turr4C{`)XM#PT%8bJGS`u)Co&=pMS-0>da#M zoV{_5$#Wgw+CIs$ong9v^73XKZW+h@tDg(=F8^xNaB~sIp0h^_!p<~(-5*-65HK|% z)xu}ip|guBOs6t2y~w&Y_tE*E!ZRf&FMGr`FXZHV%K-nEPrj~-k-KLQ)N;v0C#4}X zaLT8%nvybq|FH-k+$<4%?m|X_`9<}!mj4dxPd~kN!)JSmAMyVlY_e3V<=U9~?TBp9 z^Ay8!DZOnK^ZdS;&cC*pq3Vo$gF`ga{P~Ti`b#7tPMZJsi#BChb0k)0mfN~;riPi~ zZ#yl@IOd4?_Zztco&0x)?S<1soi_)!`RqMa6XWE*Y2wNw-OT$%F8-4atD3PVrE%86 zX=Thc0l_lGIj5(2?<_haGpDv@y|+ZSkNW4p%ylz1pPRC@&w8uuo7W|Vxei?%>sj3y z9PVb@t#x0U(9g%lekd*6-fx%1_PcrPdk)Q7P!loN`}V_W3=xYP8XnBt8CGnzz^i}% zmpvA17auut>f0Jl7lsR=-FsHtIf#dRwA`u$_P3Pw|n8HrL5tzEC6c$zB}~bEe}9p5@G9vh0@*-Q{ZgyfQ_k zbCGu;N6Q($cUQd5?YI2(tb-%Pw#FkgFm6`(+C>Y0cHKFv9@)I!z$kLp<{g>Y`~2qA zu4GAG)^PbF0}Fc$%YjVgg|o%xt>0sj)>W7OMULUacZ=nZH-1&+73ALaSla)!CxhnU zyD2%LoMzjn_TIXBw7q5xL&0_1bNA=yGJZIb+Vkyhj@auDUm~AfTI}_^JT{qMVwS|R zz2>_*ru?bW=XpBkoa>V5tM0wC6;H|G`*luKvSfzgs^ZDAVRK}~-&=Djv|eh<(|{LFA;l}!mNW1B%z2kN&S8NlBSX&4w7E{p zE+{HLzqDxcl@Fa4EPZetmiL%))&A(#0R9 zeSZALsEr1ljiY6^331$<=Q@7ds*Lgq0T%}KV)2kp6t=Q)Z_d% zSMmNsO~wtYsu~`e_s;lSd*JIf#iLq!iM`?hO*+v9?Da{Nbe{C( zbM!v@0`2gX!nM=3SxL&b>=Wc-_~E?rfz-6^7n9OX%RV^y=bFp|o@}25C8`1uB~4Ci z*9&C$ExyEgy^f($M=6VM<(~9%osa(QH!99%F$!}|+b7wYsaIPa$#=VB*`N4tR}OwK ze|B?5ZF7xfyz2H7d+V}KpI>hDS}^UXcFPojT?bd{?{2b}YH-x<(fioMG~fBrwQR9O z(f3Q8PA4;J7siL5W>8T~u$;O5kq2MNc4r1&d6uBOLxReQ)&JV|Z%XqC3p<_1@@wx7 z&5pprAG43^*7<4d{1Gc8wC08Y&zbO5+mv`VGdO72)tbaJ8)n2t?vCnf*j&@8`{OX< z3!57|!dw@tFJ3e`Mce3`Xrv(b4DZT_ZIAZ%uymx$ZQd#(xb|^J#EZswJ%weg7jn;b z8JKqn+2nu*ZCDdtUM%kQa2g{5j_Hn~j@wXBOq$;bK^=aCy@q)e=#r1GjZD zU7la;xEYaBZ=$h>*~#r-*q^)oKfkc?GCCOk+@Q_)cQA??bnzb!i3NLc=MI%13SxxcT--*lswtmHQ{|8pSM8HmOC7m zmnody)8G)z%y5z4{B>ea;)F=v*+2F;6sWQKWt{xBc6X%IBhS~26XveXcULHi{2HdR z&gS>Y`fpboPLw~p9g!0&SToaMsR_HbhE0K6o@u>DYX0$9`}wT@IvAG4yg$G4^6c=hV#eWGn@G}#1my3buW#qemy?D7u`(XXnQSx)4cIqx>VRu`uHcFSLx{|66S z`bH{zf5G?p?m_*v-xs;0N=+Ai&AfKQyM?DC1vje3!JQ&LOS<(}U&Pxj-2C51(o%b!**syoiQzs*DN z$*#OdJ35|e@|K)-IA#0fqJZJC{iYKe*Ek+_?b_^;Rx@AjozHeLd5^0TYmP*};XfcR znUb``}MM|K~ou++w7dxlml7BlAh(zg->4lDDMK zc9mVQcGP%yif6y|%+lpYa_5UwTJmq3_w&cg?NtK*-o;3-j@M1P-68W@blFR<`(o{D zuI=Y)SE4J2oR$#KbGP>qXd&#NFvuyX*v1}K#3)<(F)1~lr z>s{Bd$L%L4F!+1f#m*wT2vH$TlwpV5julU8E+_+(W z+;I0dG1j6v0+Q++T>dia53hOVf9`?cYO&8>F6zF%UuDTt{%nOy$;FSh_OXiobsrx) zKCaC3_nug!buFzSl0)hD)aWP2=Q0Zz8+j%!6gHjw;D*v^J#RJ^u^GA=IvTGJN*sS? zC-QfBxXmS-IDw*tOxvb6x^pet%^Xl_=*QWl<}MH`s&sBiTd>Ii4u_JTkCzy&e$Do< zrH6G+n91KlcXx&ZIWEjLZ+K^(68(JT#H-K)C9i+Go^p-nnVYXB{o$IU?-e$N&7wZt zF&6{AY^?e4nqlIB9DnvHeuibuO)5;Zz(yH}=f7bp=k(IEFFSQHgwbpoZ z^KbCEFz0_eR{mLM@BX$ipztZznFXJ8d%XpsdgT)wHMzgU=N*#t3TCOf605QhWh2Q$5YFrUl{*jar0{QdVU%%n9pN_^k)LTz#E$LdwbW}MCC?rY!Gf9DCW z+#iqik*}^e@=nkD`}}>W!JD@~_pJNOE@88AauhGGNV(_z&yPbDx7_;mU~@`N)v?#_ zmKh`*SbmSgbMt3o`K2GtN=2Lt`cll4N%oZGignC&o7nC1 zyQfC?${(1k$=Uq(l8u_d-Fa0EhtsWR{*{~1Z!4$FZsxGK?b${5t9S4HIvLoq^5;56 z=D!Xv*GU<3PRXw9o7QO1uz2U{tNE^fJ|8HOV>*#})hs@^o>;7u{e9hj@79-V4OjeFuJw7>=Q~JDnGDf1kFg@P)20t-C5-e`CID8KeD~i%XrF7v1A} z@#n(wdmI}x;>5#h^yGJn$<;i3+#jH>*`AW<7dVUS!jAsx1E~+^STZQAuqcuLweLlu zTmOp}Pal{^i0M80KGo`w?V7i@E*pGYAmFN(Dpz&qL!kCTSp&h;^DYb*)?IEdKkl{q zi=%q`@zt|yZcfnh?OMHLw~8W1w8|mL$>}ezyqAw-igGb3lJ{92+N{xk+$)32-u2_t z^ZTuZmZw?z)V+QCRXL)8jf3$>tmQPlA7^DGJ+9Bp)qgFmFg0o0{14rt>9-qgeuUcV zysUmWsV&B`lJ|CK?AfA=(T^j(Hag#Z=yv$;5wk5TGHYgT|G#{*Xl#}JgoA5VxBs78 zliXx)*Ko+)>$Hd~hlnb}cE*pY$0{$R&HZ)$`Ef20pYxjee@lNR=s9f=EItxr#3=GN z;#%eOC0A@N@-*(yc=&DV=HL5%|GsZ$S-!F4)~9cau9ROc*!Ai6nx%>yldfj*r?npI zysxw7U4Es{;dQSQEc4W_+}~|g`Mkf}WrKIUP?D6STmO&XmdbYfr+qJO^P3mzN*nD8 zJ=*hlUvU3~iHmM+_4EwY;z|C{-om<3y6LatHI;P1>WNz0lCFENn=959;@sNRY&BQC zw?0IBdS-#lV!2IsYEiY{Ywz_pqRmcjTb*@7^d#$3 zL5WGJ_9ueMdhHk%hMhgSqucG3%NhB2CM_ddmT4~&WJ&-MBzSY|s`}bSL zuSxo8^Zi}CSo`JxeS5DcncJ*?9hbMR5 z?qwTuiT-{xgVi zy&2GKFVoBTLC533rb{!YwI%ZHT(eV}CMqR~Iw-O#X8XW>I-}*KRS1>Hl3wRv#Vy#_VkWW(V5jG(YSBHC(41E~p@iM$s=ihoI zt?=V%pS4@3XCJ9s`*OA3>CVX+49`;RRnE-4$T6Rh;p^+|jn+r^@4dm<$Z??O(7zKd zJ)WOVW*v}~t6A;h5WBg52?q;9a?!pg*FNIPMIt#zg^Dh zy~&aN&gG)^4I3v+VRaI#X8f>*;f2llNA2fiW-9v3iuqjn#9H_1+ae+3BNDtf|2iwn zlzb^QnfcQ_xJ`S7RCVABV$i;EI(B{xpwX<=(i zS>lr^f15`&eBERV*G-@In0PtsEdHUgZSnfu{nG7>x(fwfnP>Q(oSU`OKW6%2rW-bj zNk;SKk29R-YPj}tZo&3fZ#tW9OcH1DaE}j-Oq=7xUC91BQ8CZOxb~RJf_Vp8XU$mk z|CEEU_<;l^j-XAwM_xKqzOm%oGSi~7S>SO0LxVZH3LdAu_)%ALZNryUzdzq%TI6d# zpZ%t@x@yd|Cx$%}TuwhdIwAYH(35uw$B#2Oq)liy(Q5Tl6b}!b>UmRi-nLj4f$fRc z9G+}nk$dBo{9QrcXh#K`GS2$@`IrBBD>JBi`ktNS%y8rHVfI@>#|kdhO*gJn&01{# z{ciiY#ee2l=&Nl^-jy?Z4yW`oWNS22J4-P)P zU6>fn_ww#{!-k64%o7+NKlYi!we!66371{`@^4eU+unWG@7=H3*m64JYT>gf21_?= z@@mZp^3J-FC9`?1l_gQvUnq6&?AU+&C$raqRr$xyZD2|{ zYrZp2US*Dq$a)u!&!wFFTMzrZ*7@{%jbZjJImS@+%}YvEgfB6$tK2?(^IBW-Yqf+& z8!j?g)(bH;q@8^|L+RTOC&^>-rlIlgWv*SXnH{9{JCU7IS^*;4QFA1TXKc2jx;;~rVvxOsquQSJE6>sL=+ZER>zdz!Q|Ms4zyYb#BnxVBEX zeyM_6^7yN@bu(qQg>lVHPQSTjU&sBfgMT{G#LR{29h6m^-zD?@e)HnYE&nBt-R+L$ zvbD2)W^`C_xvQsy{rMfC(AcFefsqQWB1?62*tUpm-+nOn&z_$)(b7CS1WcF|>Ux5$ z3$*=gk0m~M^i@jQ`ZZ|U`C$N?2>6!V6oj=%j`bfZJ z*8i6}r9(1TBr~RRIoifAahbzClTYv6H>b`RrNE#P<%%}hvsr=m)`DAt{uE{?_SsD1 z$!LGtGP^fB)mCcGk9Ha6OZ%tpKj}X0n&~x>Cri^enJ&5hx_r;?+X*M;^zAz@vGLr8 zj;^rPQ%zl+wgxGRt(1^(l2~^1?wz+etMwn}$Ru^EZx1^YQ24!z!^%(Q`h(^c(;`RH z*4SD1O(rxvYX9ldHEHv?sR>3Lixw|rH4WEa=`g{$Zf)fT1s20qjC=f|o-C2(yf(G# zSf-T1?~RjxzTI`t^r_d&D*wCJ-nB+oNEhyPS+XxAY1iCcfe&YB_C5-xv&_*gyv zzqrc7`hJh!+3_muIwzHxxST67D`x4Wgb6(x?f!1*(@BwgX}`>My}Mei9oxfRT2tQ>rLdFnV(miIIwElb;d_n7Z zOtVfL`ttYciA;|`IabH*+s+<+zV+H3cNa%Cj`$ZB8~?^$_R?cp&%p5HsNPlmCb{Ck z<&*r?|A{+1y!JxVimf@EWBdM-CvQLAl9+G!qW^aci^KQVpQ_3Zzw3304?9@Ev}w(< z>b$f)5#E<`%4Cb*nY?9S|$GbHsY<9fks#lp!tOBe4aJtWE zUvGQ;=2Uh2`hFL^&q@+20%z6kx_@@_F}D}ZQU47k4uq`sov|zNy-aOLxpC1W=fdyj z%0hR4a-H^h`NZNKPwIKKniufwpBnv)^TONAU;etUW0(s1<#<_`C*RACyJ3_Zdr$76 zTM|pgpVD2Kdr$9s%ii+rv|XlSlIub?<|7RM6^@$S;e60_d2#!UyZpKHyeFN!l-sx5 zGxF@p2sVePBO4t<%J@7#%jBd!l}JD2(!NYk{nqE-U&>$k@f7Dx4_oQDrT3(hQFgQ6 zY5A|x^Jk|zDxA?;@9$f2)_#AL;FRiJkz5KYsylR#%{DoA;_IWw`e_wSR=ukqsy>KW z@uuS%cXXQogUP?Fn|G%(Y?&97npf2n(HtBzU6I*FQXwoepZh?@_06w;p77!9n4mhZ z@I*n0OsIqMlB=7RO>C9wjaYT%n!}m;!hp>``L!oA|9|@J=lyH7m)Dt#x^D6OX z?)?V4t0q0S-RykPz`TBW)ZD7i`&azWy#3-k1B2)ZUE#=OwW?auaXChr28vx9QYvmd|+Rzow?#JLQjAmX7q+ z`X4N-g-k5=8j%;%kF*KLcQti1p(=^Hm_1Lj$6gLWAoh$0pFK(n(4L~ zGbEgQf8I51{lT6Ett?i*uRAs{RLQ(`y>W}}RoTI?JrhcsZh45N-(0Ww_Pow=mOp;4 zcj|t%(%mO*aM3)DiC6w$%E_pMN%2!HKK9A%6PJM)?SEz2$9e^K z&5xhlxai)^eWz;vYzbP+@VsYc-mV8fUQF?s#-G@5#<>DC5qh1?>wxx4esa z_}5jSRKROPlXhXs)C!)S7YnjZWjyy=&z1O?i7!n|ZpQaVQ+1l38C1Kt^5v|X_cnxG z&y`he@%n$(H!k!3nDd|Q#rc12wFh2)KlOZ<{M=LOr~l5?UU>6Xc;0gt-kN>J_G=On z{+#`8wWjzb=cgq5yeqorjz*gcUHK^L+rH)g-Eu~qFxQ7CSZc3YZY)(2dN8Mhab8}9 zac!@V06T-_lp800q@6Opl*G`uet}|+Y2Fm^R~EA08UNe*Pn{+s_~-_onHSfaGb_F| z?TOo6^elCb z@YDmnWpjluR^B=NLZsgBG1H8-*LV7yzQ!!E=g^5m)6VYxZP3tnnW^WJdspyfqp2Lu ze=b}o*e&)eis8yve}kD*nGX4A-*}`F^Tg!GrhhRP|GLU@T2|)t@~=rqW8Jl#eYVqt z^f;N%GWTm2{aAHWT3;peR{4V7$IquaE97lBXmIm1gU`HU#<3@M?7P>i+d9{~?7Kpx z(&T2}8F~4yHq1SmVEs?-_{HQsbwPYJ&$djA*z!C+Vc+SahQa?%FTMAC##|A_9~%-A zPtWLnkoU}O(MR!&TX)+TdAL|k7}QRU?0@B|lJY_6clAywgM^7J304m|-)=iz_qy*4 z(}X3uWpk4k3rnY5KCI~Y_FU`C`@i-bW}9BF%Nwv|buPzMl_#?@pS$dk^VRvAdTY}| zajx6^f!*f?lP+w@cd}!6aO}*+v)9%KJf8UG#=R?HQ`$c)Jy4RogNH}>{>kNzKh8K@ zUb8NH&s=fFhThu>0s7zEUL@4}+3z^5&A{-V;l=m9jKAL7cdw5+xXf7Qu-alT{~m+O3vC~jUfs|+;rsObBBl?qyIbcM3W?WwnE%PC_x$^7 z_8ecm?63CPl?)S{mUVls;CTO0f1kBls{giQsaKCFaIDEcw3>ICAx~}B^CnGa;atgn zZ4rxh@*x zbvr9X6^X~oN(%)~&sb);OVmHt#Ao9Y`)hh96!ew-7A?(~cDunt+|K0~ugE63gKd-B zUGJXZc6;XGu<}-y`74v;{&`$aw>~giIoWGxv);`9Nt^*r?3@}`1!}l|aontVw{Xph zDR0H2v|M6;hh68GyCcZC)*<75enEpQqmlp9q=W-D^VhvvFZg#`-J0!=2QG6h_+`P9 zu=U)twVpmJLy{#f*F|Iqte(uEJyYuQp$Yqxu3AlvJf_RPc+0gDQ#zhV++_=sFy7kB zwrTz4_Rss*oV|9!>W)-ia%YVrpY}1eBOW4|5|b|pRfr3;XeDk)c4x@EbFH>ug*2bT z->cao8|E7>lP^!~F7$JKyLji0eay~tEp}B`RIK8AD7@UtZsyBN%b6Xf zY@44fZYtQ6qO+vYPUWQc8l6i?4TA6FMGml}`P^C*(c2yVSZ#Nc{!;~G2A43#jyq{@ z!yM*%78J&Mn_u4_Y3AC_*}J0SefyU8v5XNhr-g1$n0C{CiIJ`8X}*)GT4M7LG>8gl zX|8g5EZkqfk&qmqePl1&kDmn$D<3qi>6Yy1njv$IdBvAhhTpnclM6CA{ywWvtToFI zYjR;Oh&5X;eRxj?qlPBiLSHxK*^XB%51wO8xv->ET*mOH&D#D)yX{x5tiP!L#`Kqi zOK4O2{(#P|yL>Dx2Af|lUUKxPsp`VSmbG`%e!ZCT$Es`ny3`FEd@gf6tW7&NM=l9l zy;Q@?TbPH9`S?^W#xHYy5+wQ(1(|-7)1A@sU2t^x!6IfcK^eZ#5j8 z8v-&k7973lIaA2P+3)`bM?MyVYLg#!X>SAH$j>`dGy5m`5s?36ImZ zWyD8mP8G8IVa%#fxv)%=U6H}j*(~3!DTte+y8Yon-kW>Uwlm#Q+mI=JlQE-hpOS#? zrfu&|ect4=Vsdr(bn}cb)%v$yPWw}xrzAv8|F-My+;wNmjG66Aza4h`E0FK?#h|>L z*J)pOchJ<+dzoE3Q@3>VhA8#;Zm?eR%W=iE3t5+r&2v>Y^Ly|=SAba}F*!f|cyd?> zi`A7hljpmVqULAJJz27H-h#&ea}0d^mpU=OunwQR>e`y?cF_!KF_NcUc&ps!Oj>vP znd_c^51Dgy7V|q)$~?PMRkY3N?d_d1&L_MW7>gn|?R9d0ki0&Vw^E^b&)q=gEW4#% zueue_Z(l9{JSIAit3tR#{xFdwn<(Xpj?&BK zIkE_}J(g$;SJXZB*xAGCcSY6h4IE7B1wQNZOML?vY&I7JdN_A6%xxEHPj>j{aCG^P z^Sx;kqDm9P>i+8FS3kN^=-9D-W5EskjHIfg-_K>g6=VrXz4q8@qL|T-=HJz?HdX4L z*M^2u#Z_L_k7!^ifYQoiSK?OxWadsZ9_o8qP07y>$P+&cCs?+ds2x(Acl zWOKt`{cBI@|DH23^H0m9?eBYIERS__24q*eXtTeJw_hdkdvieCJiUtvSwUU}H(8E7 zEWUZBON%Y&p~j=M`=fy1?9Sv^1{o zOxM(x2XdnYUIw^YaaJ(c1T$Q_)ANZfrLZh<7MJ3k zev92lw2U4sY2|TP`TG65RAsZLPdlE+=uh|iJL%{o*&|H9es7w|>vqYQO~EOpvMOe3 z&w?3d6Bqtnv(CM3->FCSIrq5^T)DYW-Rkp^3_IO#zc+vFmyNace8nNNch1s^(4Xh8 znQngYGW2s@X0*n-NA{~awR@O3TyO8leqE@p=HQn5*osxTF!#Wjj|*&EorI!X9?qC{ zaI5r_XO9mx_$Kj7{a7}IXRqZe7eTYjYR+c^y!1o0L*@4^RrITWZCp{na=GtnN~ujV z!>Z4pJ{y}b2=2=hbTku)yqkYke9?-pYvQ6BkLq_6`j&fWKjy!ClH>o?-ZS?C z52yUzWFS51Z0o;36CLY?sT+*~@7c|~$6{UYk;*?0TK&53 zyxaHsE8}YJy;}YE*%%b=ebQ~`KbO6_@crJHkKcRWT=SV{d#>v4jZfuW(X(6i_Xu(X zJpU~9bmP;_F(0~qzU+B1M57C_a7OzYv1waKl0M-#`ZnmZ!g%^ulR85mHIub(Qi-xaSU zpB!&xH)YN;-Q%)1c^@ssaq|OA+E|uL^^ot_8gc!}raH;y z&)3uP>#qc#Y3#lqX}nJUqV5mI37nE_L7rPwvu8l(<*;nCGf(N>49RB1LuS}tKzM99@KuvlBYe-m0|1ifBEh{ zf+buLzoz#Z|7|~)SavyCU0Cq^#fH^Cb}}zJpL}7-w(eNvgQpl@I2umdr`?-(>p5r8 zEPjuD_y4GJJZo|e&l3MSlxC<>Ol05TLv-#Tr6|kCb-@_uhei= zeW{3g(oy#MS+5uwUTyP~*Oxp zArTqNs~g^KdUo}pjeujo^%v>92d0V5Hz=ID;PyP5HRq4d<*f{!aCg6+zd@?HZpe`EbnzRfmd90(LxWVdrmnO6?ajix zP_EFABgCh3RiVlOjwp#%CbvaXZlBj_$T*RG-m>iRWc}4l`z{pRh}*G`&(@?!HpeWe zWTJg<)#Lkf1V6;zi@zTDsjzlM-PdmMD650NUOwJZH_ztl_X=G`hCTm(#WOnm|Nd8k z@72odj~1kzyt;Nz3&WX0#^v7gTp6x9buD}Q_rtrIZ=Dm2Z%jG;<{oQnB8wrLTWPO- zuIBQ4yi5&yrvHA)(X)T~>x)~>nsbisoVS@Z|3`Pxm(zAj=5**Ezgsht<4eTjY>9xo zpC*Pe9KX5c2hZ0xj1e2Z|1M(K75;HWal>k3qtLH42Y)a9;K{%-btj*WQ+~bjdGS9P z3@fB}&6AmX`BlPilS%b&qb$-F&D@R!v*m5qgdb2`=EU{k`mwh)hIa)TEtr*eU)pSNUs>7hLW9(&m=|3P zYxepw9r?I^{dxp^#iE8`5zp7su zGbgP5-4=OwUei>aMSFeuEx*^#sJN-%$Z{nh>@oY}u=y6BxbBL4YHFH(Q<)*CMZnhvD^P#sg06(mQ;Hdw{VH8qf>BYX>W>Ei8224E3myj@ML~%&i9jlXHAd(Iq~6%Iqniut){UW&U)t{CAH<#imd%r zLKnGtj2L>&9hOaHWV`;*HF|}1`OKxe1X8%499{S(-HB;c(N_OO=eZ;WliXg|GX&JW zSo$(wbxY@Lx1^~Py9y>u3Y#7(lJY!Vglnow!&<3;u&3dTg7vrTKR4?u2V`$Qm%wyj z6Bh?VP_S&H-1cegeKNt$7qorv>-PM8HT!*XVSV~t_Wza|-+368ZmK)qCOB#T=0Bb` zzmraInfiy{uKDc0KQ?-+ZvNs$4I2`r8~lUA)w^z*WFM;PUCdpR`sNAi^sLsemHm(9 zQaM}~O|q?z+_T_{`BP znxOWDp{lLwR&dD{x7PAtc^youNwmVe|aeyY!-xo2_)8HfMA#+p=U@nWDpWX@aZO=rG4XA@KIhja!9p~MBhqeEq7d}i#u*YlUBp=9HacnjWnHlMPzWaK)d zm>ZUSs#X8IDqc3`p~ag?2d1$Lmi*$@6f2&p(o@5lcrb|XNlvQ8j$2MF@6&ufN>LC(dAPs`|h=K-{133 zTaL@@sa$WhdDFhE3k6$C(vGBrH!<#IaH!HKY;@RenJ>q{k;qWNbeC;Pex-qh$X4Ow zUK|Z`4((j$wDsXX+qhfPCv4Z9vn=rOf2n_qH72QPc7=GFO7JoEW^gDj(3n+x(00}9 zux>W@sA+Ynk0NugzWs7$k#%#t)|n@ZvR67fDjt!SI`FT~Kj9naR7ZdRZhtA+&`FUK zybeESF$`CJyOP72<-yPJ*BOuIGoHS-^F{FEiHgz^ho<#E*VB=>CC58+&BKCg7J??1 zw*KwDvwP`L+3)5hp2{79JRJgQuU}*wPs!5L=C@)io#`Ozm^Fiep>p1aw~S`p+co<4 z$x3lesDF9*O8A`r7Z{TK6+ZD^3l02Nb9!eQ3tODmQZ-(-etSoW^!p}r88r_#3^=4P| z>mRWy0Vgk9vSPT|_xA09rl6}o*xnVex7)81{PZMkL6PBZhisMQe?&UZZ*@FS_Unt< zs|l`4CEkP^T0MEVP>Y3e!$ARu07Zj3qvH(Ij$Jv{)79>IJpAA~$%9GLrX6Ns*tzf4 zz8@EAcRt>ct}3PWPv+)ZsfbxTIco&@UnH5$Fk>)%b|Grvu>^A1>TyVcfO63G~|D|gztdQN-|FH>j*E|y)a(sc-y0}hS?>e z-)CK7{ZRa|YI&b|bo5&N^IQyNm38GyV-$XGWqReJ^(N@bnzr?uQcWytMAl05Oq?`% zLprO2>f+z|rh%X6P7R+vucT3b(?`_!Rbu0A#&DyxziBWlX zLL#a>Qf(WB*a9rfOAp3BDqprQQ1kA>TsC`a#h;9K6)*Fb z2`m2PvuASZ8vNh><=SWEd#~qb|CV)hPQMUZof;+>@sP)4-7no&X&0A>tv2T^m*SbH z$gp$HrgvvsuBBW^Tqm~p$brhgyMu(kZjPw7c4qo8&xJitx@IR~l2wuXI(^tvFS_I^DD)v7j;8TX60om8uO7BAG563%|40Q*px%w#sNb z-b0L1nj7}XT5nHlRBZY5s8o8*Lf$(9I-SQBac<9)K74y4tKq6%v2;Bir}DKYR|yBM z6V2;&`+wr8&4QKFUv7WDDsu6`%`6Ink~j9Id2D2fGVDLO$Ykm@=XgQkO*bzv1eH27 zGCJfnE)aCs_>;kjVWwc5?Aou!xdqN;`nw*h%SY)kZDxsBnI7Df#j)Qo3bAj z+V`h(IS6cRxVU!jLzCV$OdD?fI^p-}{C#%m7b3@&zdUik?S@%K!m&lEJ9hr?6bhZ; zsCQMI`<8^#y45>xDOX;&v}1nE{i<#8##cFZJPs9MP;hkVPqm3=Xv#ZR^U%;^=F{}T z9Uf`9{VPtjb(nEnwB5ST#y#co+;x@RiR(UB-dCLXu-&q;()Ds)Y--G&{r8-hicQLn zHtGsrH_5(MU@iNB)#twB14(9qvh~u(Cc0;=IdF|reQnr*9bJhTuln~)TE72lhuVLh zWk=3DSzi=>P{v$&>rr*B+fuR_Ph3K3CT-CXQZtSSmA&?7;v&XbTO`;W0zG#Y%r|8S z-MYfMx_HCEt6^74$_`xFQu=4*zMsB%8a8rjub6!%d_VDwUD54=k8opz<*gaUd4GPD zDVpDyQ}E%zu2f}*jpco-g|key$-mpfRTw?jQ)i{h%mYhEQRE?n7ui?g1FaUM8QU-9oQqvVA} z?Aj$B7yhzseY3jqVdC?|DpxL^B%NRjM;^0|b-~eJuCafLI;!?PY^MK%&c=cuL4cm8D zFc51*41k}GdC?&mt1jl{r|zkZFOM`ECP$Y*X958l;7RAON(>j=Zswn=N{P~atc$)nG^SK z>9M8BIbRPRUes27Onuv0;lmss7W+hrn>0jRDBA94{f#$mo2dr_*F0a&@0tnER|bb4 z`q{5s`}MkO+}^iqcWivU=~n)W7o5&poJ%)dIKFd*%C}c5oA>|U^}i@5YhpzAyw*~u z8{YTgY>zS~eK^;5x^k7?H}g}R-J2~AG&W{%GqyZAY5pYH#@b-Y#?SM%33i?6jM%N} z!C-WdO|;yhjWOfqogG^ma$e*dl9*85_o{@U;_x~3PU)K;7CxE9|2g1Z!$-%K6$;wn zYwI&a-inIZR^6GA$UG-y=Jo#f_Bu6cUd0<~z8cmDTuoabdRWjwq>nKr_1L2Y(wdsi z6Ixf^vsrq<>7QKPi5o`J+itJkQ5t${NB_53XI*nO+eLDd>b_mdtv~Z=MU%64+skT2 zvC!u+AKxy?5-WAKTjlVtCRxMoSnyQdjqCkaZQ59z@_f zcVd^%{^5MC`}TU5-uOEqIfs<%^RIAEyS|FM%yRm*7ygkc{Qa}|ydA!OYU|jf_L_a! z1i#&~CWcYf_O1ryR}Tq>eq=luU97GDroQ~}p_>=~nWf}j`)kerXw#B2=?N$Q^{xF; zG_~Rd6HkECp=+0pBy#k{B<(l4uxayyg{!yyf1W8X_w#IeQRwi$ICsNBTaIyz5J|pnD2S**Ng8`Pb{!I zZjgOgDbw;GBk$In3n#e~o=Dw!k(wUcRkZW5sUQ=>*8X+B-4}g1)sa}kppyPhDfhx8 zhb21q7d@zX@vUb^qR;#zWe*q`<{YcMuA*hopvC24Vxn)6`dfB-!HlWN3*S9Ux)3AU zV|2gvZ`!whZY^=BQ<{@JM3Z|op4IY=r+BfH!#?d#*M4;d8v0-a;8u1OFm7M5BY zTdaF%*@ZfWW%|xV0`d{H>^f5C*-L(HzpioY%-KeUqGmD1o0)I_GvwsG@m#;>dl$o_ zTfz5hx3A}6dMuai@|yj}h7$jk+%rzFZujJ7Y)NA?bjbZS+j{xinP;4GkHo+0d#rRr zYgS#~!c*)2-Z!1{Y3Bwd-eb^mJDyBx`m_58+t%uWj#W`9ioZ)TnNb*J&t1vBkunU@P) zesFT}k3QM&Y4c=uT-uz?aBG?x%gxty)yqzO-o$cFR?vmh{IG43rSQ_M@~dBuG}M*t z{C9b;qC=teWr6H}O4pAWFR0$M%;L^7epZDya^Y&ef&vY*)f*50DY$ZRe^n6E9kC+D z69<_T99-j1I>)C+h1_}?yd%Nvl}*@!1GBYLEfaH(%(waHru~PZXuj#00-IVsZ|NHg z`F+A(oo9>9X|?yuGsrjcd!BUW=fQ1%TCnJF*pB^~p1QL0S*%;{p>_juf=rCk0SY+Y&kN@v2m zbIYHc&k2u{{q^u{)hE{p)VN9kcjN z*4Kh-&gW$GPMp^(kPVso!gwY>gJ}I8wohsO`OVT>{|ZK!eynGZ$~Q?OT}?i1?X*bH>*bTPkD!{Iz?w$^5p?8eYe*dw)MlyUpy_m+!J=K-jH zr16!^DbT9Ia!;I1)qD){O(HOSvMhPW3xTG zGs`sCCEttX3j}|8EYBDA(r0>i-E5~_8x4|@^JE_BR$XQnsVfeW7w3H4x4e(j$3~fZ!P`2dtXcJ zj=9rBlk}Q4yi=6ONP61aBk#9&U)Rf$a{IYCGlsDZ4tXZpWn29 zhqkf>L&d%~*Ls(FtTav#UESy)vFGP6-lsq3SSm`)ExTX1<%?ML5~DEf#Xr9VSJnwI zoSl1m>w(f|9?kFB>i2E9gTWn zP&?)0$x{y(<}Cj6*PlsD-7jalr1$DdwmJX#%MQE{KO39)kh|!SmNG}^YMHD6hv@V2 zGpm(DFE`A8eBk1}9W&=l7r6BA%DgK_FWwMVG|8#8=i?yYj&Y%3^Y%izUTVim5Ib+>~w2$ihyJhz9ep>#o@~zoM z|LA`%>Gk`TKJQO+-@og~=^D=dEnF*BuaVYu^^beMB7a_))at#Sez70^%!$0L*3;Q1 zbKq%}#m%CKHJtBDJz4(eZ*|rfD03pZ{ZK*c9J0Z_hQ&-J0E-&gymaWO+q9 zb#!f%c2RU;E}Q=0_k1ok2kB}@+aftW)1^F(h1Eeq)iKgr@5wCKe*D9~C#^gl$99JDcF&mkM059(_re*|W%zL4c1Zdajl zdB58%`;tk!k3O3^E92X5)%5D3J0KNpM4 zN&BwXbu;$fTcLnuYz#rJ>hIWl1MW>dCw_iy{oy{&8J z7d?9Im-~t&RmFYhI1bFdo^gBXLWe4atHlz`fA|l~%k%N6P0me^xiYP(&gP1u-Rr-f z^Q5E77BAvH8FKgtyK>F4{N`^6Cuj)ZF5|>)iUYL zyR*Fa1b99l5H1NnQI)quw!%%?W$!i9y}#{_)n1pK$@bw+=9jwP6P3=~sCoaiUp8>F zWzBc5>Yf28~?xWQSpv@D%#w8&}_z4iR3cr0wxs|2L*uzs*2p4CbgQ#Jh|JD zSl52z(w&(*?JWNnyO+P4t9@s^#2v#kDdj=cOly_SF8}e6&pdn8=k@=em1h52weL^d z8TqNcah9Q;ik}MBPcqK1x0eeO@l}=C#eMbZd!Kz$7NrFz=dD`F-LCuh>?TjI85>>x z6ta6vf52$Ce5ThHz0FgkS1s7!x_al2@9x#X5l`yoJ0JP;)84{0vp02_|04Ta>&;3Z z6hz#;5&By@zl43?wna;JXeval)!Z56am|InZU?uAuEL#Ze2J@OaYh}SyUgz5hVtg% ztNz>iUd#ViaPN_CTQlQfS+R)0_nIRs1QkWzmX(H@Wq)os@jG1D=0fTowf_O1o@e$d zFYCV~>u)_%+GfA}JFy^c4s-c+JC4b&n>dTdr+D}p15gR>U?EKwFjfi zj59YHu04@>8vJ(ot~rsB(>6@lnxnK!o?qg~pP%O>-yb{RRJ`eZ#I z&(;6`E%7{a|3PK*!2?Y57GGWPLf~Av6T=#j7dpAVnTgF4KYo;Zy3oWVR^Rj#)8gMO zI}U1cHd}jC?`b^Wkgc-4qdzz}B2;|y>uD!fEZt%BQ$qdTAMXk^E}bK@3pzh4U+uL#-k&bncBZoE-i!&mjp~EASDb$K#MDhP=F}C<8T5ZIXox=-V79Q~Z z>h8wl?k>T-{_cfJKE8z6tGi9+T&k{^F~95RU0*JR)j>|T-*ASn6Fc#E?MIbTd7ta- z+L|sq<3)`gmFSkV-(o8l-TG2iv_zp{i=p-6b?+16_ZhCfwSY@r^v%&l;uGwcPoQVPeJ{!AqI-1hj`}P~55_&rR~nywEqLmrdu)S<(xlG||JS=m zd|K{(_pT3vSKA-PMe*zZPrbD5;-cMu87J&tu-B`0QGzSGQHHqPdx73>>+S9d*sYu| zqo=c6KDOSD<-`BruhkE|88^vRY=G&#`~nd^&Mze%GnnA9fc@*M)z-*0r_GebxWR;a^Mk%Dr`(<=y=IP}hQ56V2a; zmQ7qdU3uC$Sp|3P4*7R>LEJYYRz2v-{uOB9$hAIc0eh>ef^Xf%g8~BYUh#)>GG;tj z*SGYWBa^4UJPDtf0Y`K`Znh8<@)qO zWO~YWl6>8E)>&u-`ddMrqQ2$Eg2hGPkA_ zXIG1fQAwQjXe)|QNTB0ax*+Ctr{)`nM0LiSl~nAXbvrvS#Q$d3K(?j#c50?MHX(rT=!QbvT)nPEvQ2&Un{&;NjBg zZhk=vb07Ii6lcBtw_u@}VR{?mn+@EiiUl_oFZY<9koGO*T6fG> zjpwI0@MOxznh0vifBdpp`Hsde;qnL8%sL&Hx)<=?c{znqd0m?Aj1^AhT#r^SiTm*O z$RiKFU#B!=VlSUM79F;6ZRme}VF%C0>r9lLJx}yRlzUxV+o5;(IKPt91+~Km-*DPJ zHz{_$cy_)vvuaY9QKDv@kyCS={Th zlP;Y<$Iwwx=BSYK?Ri4Cb9&~pbAs8<5BslXIKSM!tK#p&1FU!D#Pp1VSQiL6w0=#> zxvRvwVe7%Kebzf4{&fmYl4%t_{x!vb#pU<6>wEZDZGR>(!Bg8^h%4NpMC$WycMcYl z{;EZ#mGQrOwJoO_oaXg=anZ9zYs2YeLxX^+sfYSr|9F=x%u*KpZa4eoa*@3^`lN;4 zOSbyOxqqD9Cv4_6BmO~yNZyK#HA~fQ?oPRuw)EfWWEP)CGW=ZCIoqG~hiOQ$2XLIr zpKr#zc7ZrQ|LgjtX}1zAuD-eVEu}^E|COf7eHN@76@kZQ-fvNOZ+zN>_4DfETQxb2 z{W_{|&+*SSc>wTBBCwRjIl->WxmlgO{1w z(Js}QEBpOij@mK*VC?7q`TE_zc8MRh;vyo86!iNJiT*qC^5ehbzTElV|CTUZ>N(!h zFK@fJvj5_b7aCmdPaY|z9Z1$;k@1p%EyBby1AFPc% z{xJJ}4&NK99>2g$eB!JJsgncCTAKLeoga2=c9{Moly7R|mWCIv+O4+~q*#@xIu`1b z^s(o0Yc{wmrz-ylDc%(n@b#7q-mddFLqTveYObtvNR{r#R%VPEIXaCogzRXt-< z`Mbq`<)?eO*?fzX%v*YY zvh_?ES)1U#+fG~x<)(63mGi=U!k>LM+WLp9U`l^`y~*!WK{B;2ttSja!xgx4{?D+S z{_P##j{kSs4+&aUcJ=wScwBe+Go!6UuI5L=7m3)n3j`)S&fmB5p9FKz=daV|-rLN5 z&o;M7_U*Yvo_8)T>@>O~cPZe`%Y9luzUrRpG+dtD#girJu=($;%c~NS%hFy2>0MpV@37iR(xI|pPT!34K^BXCZ!-Lu zX5zd1&T^;euAEXOxIzsy|eg@-8=zxiu*+d{*t@k`E!WN8c*4PRQdAc3d#+`I5uOrp{40C^n(;)2i!} zuN`eGNjY<8%kQ(F?}rxj9}BQZ{PT=$X7*mov%B?cS{58)IMH;mS<67cZ=$09F}YhG zPEI`7l{SB+zklYxqGaZL7yni3r-q#r|9jZRO^>fd`P63_0n>-Z4YfOe+A6d3A9s-O z@SmahN9f+R^fPz9JpQ*aMyj_y$njx^%-@#o*vicv8LMiZ%=_~BFi*&S-sLY8pM2P} z^!*{$sE@hR3NzkR{yQkBQ*zy8vEQ6diM#P_)^5LMwb}Ug{Jg$Z^2Ls2{Y}4AUOR{V zv_2buaJKKh)y^6^9=?<9TvBgU%v<wrqA<)gF~frys<0kg%6-fsR^7^ro}?^4f;ni;XWzGmrxaqi{Ti3T;8{+^&R7~^_wSrU}zP7a;V?uKC97e zqge~zu$!1Md`Q%&@>;r4jbXb%$g?Le4+)D*%{=x|Ib_Gw?}s8Q=k4;}c|*c9PyeM5 z&wu-c#h3Qiyl7mkZcx#8*#&_qS#!YmF6{V#K*9gA3VKJO9yWI^O62PiHT0Ez z@83C(S>)dfPL`(C^KEub2s#y9yE$xie%_k(%j2B?%vkls>c3`lNI_?Bw_YM=6{lvy z^P8?`!elFzS`Kj*eNd77wV+n;|Dr%X3C$H+3w{@~7gR02mvBL}?plITz`W&Ziy5`@ z-ycxaTK6qT)L6r?!*R99!xyGY82mb3fA$otEZOq;&s^Wczw;*Co_@9Z15d(&hD&>Y zFFHPZOK=OcZ^2EA1@ZMUs>$Pds-g#A(8c*ldX)Tpeh!T=YdtlM* zw?5bC!yZmY!GnKAcC>piF0om7!1ae-i5=_1Q)zSmtSl5M%gA(AS(w45ki1LE?!(rl zu1DvbIctB$_uwf8ztxw`CBOKd{dH;nMQ)wLn|fxyXW%zq8niN@z<(l>M}J)HA!b+c z7YhWI|2+N6NBRFv7cOg;@9%!zWmwOukgRFd|KaS>R@1Xa?=|98d3aOLx-q+V9`b$o z?A`NaPqdDie`QtVc^4YQpkFV3E~seB%()Ee9{1Y6MhLtY*!byk?O(=KOP6r#DlpFK zUMJ=(yQ<7Qbo!NR8??;(1?U~?XAc~==$xm_1$6H&aejNMtvG?cKD*pc+HyF06o?Q?U({Xj~!X48+ z8B^GO-+7<>nKkKw+gu%%ABmRB&ZiYE-8C5x!M> zYW`?ARDR65Y1p#frZ!g8C@QQj@V)t-{DwnUEi*3YpWhqaTRm08A$i*_d4><^5AANA zWcbj|>QNEn_t4?EgWB?GcV<}0>@T+bmwaVclhe@$59>BvW76U03!e7io{DZjrp4aG zU_FKjlZs~u_$2mUsVm`qEWR_)LgD)P*zM<}Oo}q>o!007U#3|%NAE3%eR5`!z-_nR zk^-T(TB;{qVG>jQeXgVNL58ukx5k?Z$IcsjyqPQ~()Ysr{~enyri2IvH@Ua20afd6 z)NlH9%+;mmL%PF7wpW|it-r_oT|%>AO1k53o9pLi2Q^HYuhy)Te>VPe`pYvXmd}0G z|D;|>OKo}}gG;`W$Xh2V{Uz)^VRud}EZdnx4OpLg}sp5Foa zT#jYUItO{nI?7I0TGi|nZ#9m8vR5EYPtl9PO7>RkjC1vI2Ab<$Ie7lrw4-d@()bdm z#ZN2l1TPVrU_VvHPt!qBVVB7xd0uOm{cl-}pGZFcsr>a!&rgHoJsbudgE0}hmxK89;fvNt# zvzJm6nHU%8y58>ic*E=`^S|ZidE%_1{4{TAHb|7~+VoU+i8%DnwBH{AB==+5_gb?fq5oc8TfWLh)8)#3 ze%8!YZrU96nY(kO`b^Jx2LzVPQA|7iCGC#%5{6H%sy1(A@)ulwvZ3%t^s&z~HMpL? zcwRk!UY>3VH$%dy6A!r`>XNyg z6=GvP=a6O4pWrX+b3S=7aW?SqHgYd{vHF$E`sAm@%QQ5sR4?q<-)Hi0o&xIxjk^-R zzb$Q$&b_R!s=oipo_U>RGxw|g`CHF+dsbSzwFmCu(Go;p+R z{|0u~^ofPWv+6(YuewtwWX5^->}6>)kCT3vcW7SQ{!%_W)%*D2rAhK@cbn&&)cVAE zz!p3gX*UB%X%m(?Ua`O zsV=vqpfTis=6pqo?JvX~7%z9KF;yR#uuDHJrL5{v$?uidcz1>u&*o&f)TnaoJio@# zy$)Onzfx~l#Gmd-QdA1J&q)%q;|TY-d+V=u{jmdbUuFFmmuWD6^EmsRmGj1VwP zB4Tqo1!jHv)>dA`p;-F+(x=%M_bCLGUDDd__N6X}!S{FnhieluF3;R|(LlK^dG_j^ zH~9s3IPe_3TKw;Uh5anamwt?QEtPX7{#iTY%+kuUn`Hwmzc5SOeB$aoPs&BYZS()r z+xMF?)qGod`uZ96V1Y+U6M2GcS6}??d3mFTf8x}cr`0L~7><;DOfWO3p7#3B%lwwL zU*9?xYD)>$wKk;0%I4`RFT4CwP_Xr8VPVni-cWZZPJ;x7Gfv^5*M6Rql=z-%EO*{L zsOaU6**O(F4zDfT9{MpTEfZ3haQ04(?gAZ&y>mqz0-9#~urHM2ZTN8g+x&+e^^Zy{ zTe+8c?N597c%Jgr8#-KzJi9{oWM zM=TCMZm{C-mMLE=janu>%3ianUZFwloNmsukKB&+6CZAyB+!+2?0v9LBkRsj(+odq z3QYN5a)_n>2m4=*n1}XH(>5L~Xq)nXcUQx%{T5T4LN-m8&9->_bHd-BC(e1Vu6ZE! zv|_RoTfy|xvr;+t9XKfOQ*tbM%9WbH_rhj->neZoKfZX{lab}T;Rc0hzKV5yE_d%u z{^}xTZ+fLP_sEZf*^AA;w-ys^ZE)O}zKi77~(t^(WKjtrd@3T~| zz1sYWki)k(7F-jaF|#KBHSUs^I2LdtnZv5RXv&ENzn?BV%m3h>B5T3%vYV6bc6a%z zyqWk=i)(tv$6xhx?@Vxb$o2oCC)CCx=sU&vz#FyKD_)HO#9jeRPa5gZcZ>Kqj?> zq>!VnA6z#6%ns7Hcz4xv@%Iv~_q=zpu)F;{w8(L_e~c+h?$h^g?OoSOcWgVc>}Sx; z50%Y&%k2$L&#^Ch{5SYT*}wf7)u&tTd2>BDAo^Fs#kc6GhNSE2j)fM*M~-od-`PzqoMWII`p4}C(SR?yW`@dd!(Eeb}qc)RE zCDUr3`76F)+*~Ic*cIn9t1q?bL`m_{jV7Bf>SL1Tmom~ zcz14G&S|^!`~qR?3*MfLExS9LBllf; z_=+1Qw~TjJe@hs`=HJ+|l55efpvk8^ z|CorXn&^CdAN{;CcuU)bJ5Mg1OpIK<=sD}*V>8-CMI4U#F<0>=a#qP}HY{6jzy8JI z7LVi!Ut49Fg1`N?_YanS6&H8;n$@l)b=Up>3iGg+azDPSk+Z=#{gj47zwF`-$`hGR z99th@>+hPlEb;E)-0gVTp!kZR_woH>*FDDo?>9M)) zghPxM>+3s|9e#XD@L0ld>-l-DO8*b_RRNPbi@wgXt-hT7KYL=|slOo$a{Kqp-D8x{ zVHPWIKV@k_Q^tNc)U!_K z=}Jm%ntMW)$bK|_GS|rOi{1%-?+*(_&IMe3^NX9|lD{28=N1Q5p0785f7+9sd2Q2o zyZe^PA`!n|P3ylK?k6*O|FevRg_HPAd;WgyVx6cZdCMi>Q}D3`foIjX8XsGw5fjL( zvqtfOH~Z?s_{n+)Ly|9<$leMF`+cE()2m7K;eL{4H~Wj)H(Q#mT4q^RF!9dfR%t~B z*V`Pg`ozBSKi4mr1jtNCmoH-k8L6%|2=2a z77>{u@Zc2N;qGL`1sU>9ZKowyJ$zT`swmua%c|@C&e=?I`aMVfdT~AYx@7$Byc1M{2!Zwd`os zml8Ag(rpi|uz0Gt>Y44H3wbQN-gw_y`1h56)?>Go8m`eR;@`+ISgKCk;qR%t;NSzt z2Mtn{mJFZn-8?b<&MjH{+=DlFCH|dVe7ddgGo!2Zggd;xEHi#feTs9QUFJ z|K=M06W=oTWA>>X4JQO@!zx4O$U9%O+f{BXsZi6)C6uXguC4rBcxX$8wC}sOJ@H=~ z*BdDDv9d|GQ{=!yZmzOf0db38v~Lp(dHZ`+h0C7*OGD1g zyUH)})@={tf{g-ZDnSgrXPech=l#vd-u#fd@}AFmV<1M4hhub-b=MUIm(~#y>uz!-? zLDrN1_rA;j|KoV{$>08Ip|d1fPo1qlb7y&6&Cma~FHYkr&zmd(GXiX6E0x~PSZ65X5lW;@i&%HouDk>MtL!{>=5trkzd7+xV*^|M&2QppseB zgc$O6Q&&}8dU(;V^`4&jvyV)#*KXP?tUh_#$Jtw7oL%DmVXIg2mhQ*gf{&7UB_6)w z-mJy!{#$z0CDYnv{2`4&C8F0^7~VN?4$Ta+yx7TW1`|5A8 zfMvFRmAPhvRjA6GM%1@^(xMpS$j3zvP3%0bT2qcd;>i;A`Mj+pyx5 z7Y|E}R?X9bh37aD9=Uu;x}-Ye)#ucM4+~bEbl9Vy&=Mx|d-c@wvB47?iX>m=a3lo`{n2OxuzQnR;_V+pV=3m{Ow$Rm(p^V zD>BxHZ9cd>Vwe_GGC7Ku@kD@FnbP4RZ-$1nGpmWKWSNplvAK{U(KL!UdsMLzgL{+tf`av z{@d#we(Y&~|Bs~B`E!dxr=2Y+=32F{E;aHr--^0?r%SIU>z>?Qd1h}{bxRKOr@d23 z3$}FXY`lH(;a294A-0kZX$#zMu9V+a^YiT?+(GLuVnhU!NwH9(NXy0zleV-F14X3g4e|Kf1-F}YhNeu}{wQ@vQVlQ|jFjb?o5^_A3m8W{gIL4Mh@ znx*$&IKFfGu}l2oG=YZB(1&7gwoExAa7E&mh1@0?i-}Cp&Z*mGGaI_HZRJ@I${=v$ z>5&Mo_jdn(U7z(eFEuN{HRREwwXs%$AJ$(rbhDT+eYx7L_Zfy~O^$~$9S#1P!tLRG zZi%ebtlI}xGBoTEp1EzE%I`Z5=aS<|Pk>jUOxfykswOdam2| zk->HQ>Q$@)7D9Z-R`z8VmMr4w`!wm(^sVW}oC{1DW-w$wyqj+KY$ct4OT z_b~19*OgV1qGuCNkulamVhIzuPJr?#m=WJZZXf91 zv1e|aH>>{{*V1wKa)^=NbgeJGExDgQ8+=pLjo#47k$-0C&p0Q|D6bY>rq6yUojd!p z@A+?Wv$(WraYW}1$2D$iE>!j0U0GG8l5*=scQ13>rho&T2kT0=PFeI^;#`L0j^1zv zH)iI0zHDAOvknx9GF*{L-LP<)e1CXNZ&#G_e6QuR3olQdqr%V;l~v61&hA*i?bq*@ z#n0_}`b;No&s5=G9JBi_Tsbmx{ra^j&gxTF9{qQuE8RI;VOm{OrE*%@tfI?a*-TH4 zC&xu<&oS{??c*y`uXRmk&C~z4wgkm4+xDizSNrzeBSr8S77}# zhbPA~uB{b~h#h&x^bHBF{7tjc$IgA;?hEPy7^_d~ z9zD-^5W}4r!psbn*VnX`FPxjKC3fFO;d}W_1%@}P zHE+jOZ;tn!!RwdEP5Nc7ZxByYBZ3|ES6v2r5+RRQ3Iz z-@>5BTIJ8gt}>&nlQ(#a(A)!WII4Ut1iwd|E&RIn?L8Y)orRa*x3U_${paQ|%N6e5 zXn$3fZ^3VcWs&`I4qbLOyi8x_u5LM^*7#<1#a=##jnDYKrYonky8G-baGck(^m*$u zzU{ZxSBonB4BE4>xg)mWNugE3+ zed6s~FP?W-!Ma|3!XDop+?!R!#PV6@R($rS z*O;7IlQhde&)E3><{You=S&@McYi4s4=7Q0Dh^)vfrVkwyzk8}ReC!_mG5|lmVdYV zmb=S?li|ntY}Us0@ zk1R~v-H$mav2NgPzLO!li&b&gXZbzFLH-HPzWwl#yrnRaNn%~XrwhCsS$zBUTkh?2 zSQ2LwdTRa_Bd6PYM0{`BIEo7f-&}gx+@h_q<4b*xNTbWIc@wtnEnoD1wO0w}!>-GE zW&zocY>wHAaCDY%=Jj((l(NJ+{8;qOH z&OQ3}Hdlcqp{9+e`%|joL@PzvZ4YZ+ytm0rm$$kk$Y6IlnK@;G$J`0$0%orZUtHXi zzJ;aX_@l^!bI;6o;YwK26u10^NU$x-#zL04DifI`-l^TMo4CMOf|W&3r>adQwC>(h zruy4D=PPqISL+LLSGhCneX_(%c&-w^D#L|_+gWEVWN+zD?U?q(VXN)?>dSf-s(uAq zy?U=4;QJoos{gVj;c;a0k@dA2R(*5+U*Jf1p=Dmu_~ZYjxKBzul6clyigHIk@`;~l z@%{R){MO5p^P)H}?%LY-eB(5Wuicu&SE#`vu9uHK5nu8o7wM*r&A=&&aeOceA?pGeOgmLw3W^cdLeDgP@j14 z(#P@@tTx~ED$A6XZk^L^ZPoGSpPWE={YxoDd#Br}xk796WuvVPJ^J3Rm>{!T_Zn{! z%Tc!rRZ7R#aB_-#D++y^d-Ux8(;kV5i?6t-u&X}mHWhtzSiwnqzWnEd$2E29_ZohA z>w4(Ej>(5Z@;md^?yQa96Pm#oz3u)kz3fe$#)S_wRa=@2bR^6d%rnq={B5(f$bmiA zUoP&>yu3`hakE=c3CGIHgtQRkC=3HZhm&ru(8jwsFtnJ-O%XTWGl4`hM}{= zxMs(MR!4Kb=2~01-G^7vNB95Ht$&23Nj-TMvh;4p#kpOvpYuMR=z8Ls-M2*Z`WpW@ zga7&)55*;aKfdgo#na21KEI!CW%6|9hUn#2u5GZs%<1SLU#Z)X_2|JOnG%7o)edY2 zK3p!fcyTIg(ileXHTTFr zYm!$)%UDgBRrTb^OX19w?Je71zI^#ie1|jR7f$6~n;V{|=Sy|+NgKB@?bx`L>2BBE z2b>RXf9~7)i-AEeWfAW&PpJpzH*OI9xSwI#)r0e1f7Of$zy0}$m$>kxj`fM!?lpC` z(n=4%|9e(2qxWi%*YzafnKc{?^VT#de>Pfb_Ik=Toe(M6yVv(5yqx>0<$!y9?0fDL zYM(@8T|z%=tuu7E=I+h5$l7(!204SbT}O4lWj~W&Ex;Z1wVmrguIL-v-P@Rc#pkq6 zO?G#Avg_9EjQy2gPnf>HwlYz4>HXat2Lw2}dN;TF&Pfej<87tdaOLXOS0M}@>!Phr zJmh~@e7(1YA@I~ErxGFFIgh;=pVSrH>3C-%!=U$2?AX?bT_Oiwmu7^%s$$5#xBpF6yis45 zKjFgxAK9Sz^AFc4vNwsYKEh;Op2)ynA$eL+;f00N-yPPh0soF}|N6o$u;5w5(HlEI zGrst~$n?&A#uGB@H80!d|5VxjNR^@Nb?}zZw{L}mMJw*znO>hOv;B+dq-Q&Q=5C0% zx#Kfq{3b(R$&dTHJ9X+-7fwoh`Si8jK9!)&KW@vqYJEC)EB)`W2_{cxc$~fT`S90$ zr*`fBXQyB@^^_6Af}Q`ey7;8;?+A8h_}loX-OcEQo+X38TZuK5r%pPG9=P6pq3e(` zgPZlYS8OUv7|OJr{~2q$zUZ*{vOPAu`nE~byGgfd{+`J@&3}5Y55uP-295R=cXplB zDPyg2=3Y_a+{Wi{INbdL=b10N3H$ZF`Ns(BhYNZ= zpPRe;qHdY{gw1^&ud|k(zVnlZ?H0)`_pjkB-=sOceix8-&vMGweq}J zy@L6{?<)aad`bq&kEefKED&=ysL{`Av67NvT|`b#=^O5j(_MDbEDy4Ggxui~|5bhH zV%KyTcbAxOZPpFHEBRBTEgOtjWL>&TSeT5eUOLQ3t&8k^C-8QSzQN^9EV>PEOJC^a z-zamb+7qSO(D!HOtg41Zi#y)ERu)MsIq^x6rRjA1wMD_lvR2N_6l2)qcI;)|TkeQF z-NQBh?{Z6eU-zBJUA2whu-NwE)vqP#fzii1*H<|+)RsJDtw}U?bQcwWyVEUU+2>e0 z&zhP4n&n~-zFXGL@Sx}V&*)!Fn-#rgT%G%u@p#{fw=1{REHVbh)(K?a}GpIv@;9*E`0ea= zefi-7bzlB_aOHHi@~zp+nzGwwt5Dy+mln(yOSj%Vv4i=<_gk0TE8~6y9=CNm@mg1U zo1Ic!;|H^Z-il}k9YKlTTc+qSnsuI!S}JsL+cfdpExo6#9$IV++2B_3u;ICnTl2b| z`xZTk`nI(A_~jt}H@7$cwy95#auABkkT7Wr*us4BpVKqR`3?*pa$B8O%y#aOF8Q4tgiyzZ8}Eh-_*UjvfB9E z?>$9&W~Ik|PVAm6o_lLXnKIkuPe~K@yRQDTB$fY3_fk8Cl;=r_N0*9be6M(4tEU{+| z*igJude@Tu%a47Vt@PP`!|Z$4e_d&Kd1&M9*H`@9XQXb9-uXgbmL+iZ)v~!ijw*a^ zeUKA2yZ*t6b8{JfZQs*(^c6p!Y`^0__1{x<^>tNmK9b*MX?CAm&@^ImTwBs(1)u+m z`NY!a-TcMVcRa%D?AxceH*PpMe-BsE-gfWGc|mehQlA=0KCW%oI~ede#<$x3-%qP6OXU-#T~ z_vxuzt?mvUPd@BudD(w1dPmC3O;5G!ey$GP|MTsv`}z|fp4-yU`1jN$^{tY2PSPn_ zeXJY@oFuKxk{B3zX8kU`5fgEsI?>mT;qbY-zZXT1p5691^Vrr+;fdK_*`==kW&Czk z#3FlTUns-1&lY8^;Y`1N3Tz3@{rff4)sgiAqJc;jS`E`6;!ItW_rR&N673Q^{4<4U?IU^+3=TK>&%R(P^b(cMvzN=h`0PqUOXRsIt#|MFco#gLF=xfvtlj+IBNSY& zd7eFR{eb+TXI6gYZBtX9Z7z(|y}D9&^0JBhg)hgQVtY^(w`KbL{^h%a6w_z!es1Oy z*wyypRrGt-*?iCazsgMvl3&`=ae!%+;0Du*Yhkht-p?OjtUR_=ApPp?-s4dlxr16( zGIqzn^xs*9?0cc| zai!9ZwYwx`x6eFhC6#|BbeGg^50T$n^yXhFxhp8RB{ui^)Z2SC+TWHuo8EUzykq~` z-8O}r8&*fMKJkx8ooRBH!`svSnd|+Z-Wzf{x>mPo_lkci9g2Ylr8Ey`EFC)!BXV)tiBr&AOVi^}@Xvisn2OTV%F+Z+6Z<7xp***X|0x z-y!$sGy9*E(t<{re}_&?+L8Wy@!mVnzcDuihhOmfV|n|h#xXmi->JrC%MY}smR>vV z@TUD_m<*Fo^0u3ND{ejKmwfcEZ>MYZs<)0m4QCy_eZlWEb4bgS^>4R&f9cZr9{+RW z{*(tIp8I}u@moDRbllfzL-?xOyG?p7&rZ4aG%6wX$&ps%C+@zn&raqE< z>z8o%e%|IC(X+35WxJFs%f-L_{wGzOoJd`HwgT~om@r_av|_9RbfX$_yUHvBfT;1s5c#KqsPD(W%)NPfpX?}ftK_LmGy zb&_S_HR(Gq__|n2KCfSVBv+O7gOqgC5BpuI!7Dabab7Hb$08g0!iyh!F77o}jdx>E>)9X2!yPoQrgcHPr--qlN5El6 zmtE>4vYa#mR_12 zO$=P+MaP*dp8OIL3+OqQwT5G{zyyi-atlSx#G4K!+poC17WNd();TY;v&KO4+||!+ zKMJpJI3CG$`MS8~dG<9Qo;iNYf3A7n&h#8h(2jk5pS~=d|JmX5{i2Ff?tdbEIoURs zTQ!(0Z%k+8`MAGU{G!KV0wu<`)hB(@JulOF|L}&RC!0OHV{G$V&g9F* zDY5s={LOiSVW+tNjb%JcO6rU+U#8?{6(Np&UN39Gl4Q^Ha$MVJ~^lCz_O_)Kc_QP2yL&(o9A_uq3im8 zc?Su9w;RtnnV8Jm7(0LQDI`2)sJYMTz@fb-{apn^=j=I#;g`&n-&d*R=I$&yx#goy zRm0oVhk2QOJCwiNA|*Bttj_s)^eYLi6{tlM4J+8((=e*Q^D2@5aQiA(Hs zfB!JOvm}2DXTrOcQ@Vo~cE9_;C(Fq2%AVzBDoe?wWE~fwbvE$}w{j}PH_rB{l$t7X zVBKOXSr_G}zoZ#Bn7KR8ALeXkE0l2w@$^nSIGgS4Ou4S#hySO)ztg

b$|fy7ccl zyBEc%=5$kGY6UVL$zrBT9L3w;8KG^S4?xr@PY2&p&-#elBiPbVTE(3e7+NEZ%>&xnCUd zuOhwt{$UwDId8uF?1z?u_rA=YZZD)QC*1OPrPv>bq|9wrEZ3**Xgk{U#5C%|j>e{q zDF+uRv>!Owy8{C-?I z67Kc0&%WODgmidk*>1W1@E@l>pVfKy=7rgQp}^aT$G*+Fyp>7LKA=4N*hQ8Np%-o@ zv^5@4!i##pj(p$Wse zN&EP$#GW2^iR7$t^UCTz{prMuAg)r)%%063Br|zzR;&=8ajRr6PwShnx~@JIDI7i1 zI@ykx{x;gAlj~me?V7^$?HX6YjI$5>%&=3){l)U=X@h;v%~_KI!Z`~beVig<+bkq} zWZ&Ii5)G$|(hIIRX8gBmJU@Z!-zUaA`-56%`tn|HIw|0A*Ys$~S)Ox;{O?S&)ROS~ zs@gZ(DO%e4jC*MDhLlGaH0M6`Kku@Df&T~3Mw=~hdeZ`=mN{B~Fx|{B?$RP`+-_qVUqu)pD`rRp@ z^Nf4{-|nf!2jpuj|9*AUzi*nc^(Nn9m({k;F>g~nO+B`z?7o+~(>SGHW{J_7*A1MP zf^|Q9*tlAe_pwjW10kj}M=q!eeUGshoTSUu7IOSthQO|_|J#y7|NMSq<*_3o@LOqf zs)_T5BO*CJ-M1BaI$TgunxK_(a`ow_4^kL6)$ukPs?S`vddj}*cO#s+qCDlj6>s(4 zE=qOtd&k8#pZ)%ZlMl`=jH*t%-Wzl$R`IlV!Jq#r4D;ryaGg|J6uRcpIgg#Q(wQFU z%9^VG`*Q2n)z_2b&;R^$*5}{O^7#>4ekAtxZ2kKEhG9a(vBgz8qB+cFD{$r<=Kmz1 z)c@muA{4HfLty;V%^( z`b_evhXtz?YupWE3zcnSYTk*&2JSI?)8Daf$CtjaPxcDg-y~O8#4GRp_3S1~g_*p{ zJquIG1u3&sRAzIm^s-hy@+C97?*CKe{QUbeYudc1XWVr1P?0T*9WAd$4OwRT5H=aK+ZCjJQ;WYERd)Y+(#r}V?SSz3Z zCU-UiW7si{V@!MvjgE$~oqZEn*0nyr7ht-jp=G*iYFoe7iN`lXPc!PNK0cfO?7$`8 zKgXurdNP~cU-a3s>sxXTw(?#(@^%rwXNvE0ZATTWFfrbVS^}DNHZ!K&b5~Wo>EaQ3 zM^dxPp?HH<*2!7-_ilavV54y5&!0al|DTFpXSe&)Qt|kTe?O;R+qdiU=2PDe+>~l{ zdbjyx39I<2gY&1bNpr8}tNC{DVbRaj59+aR)!&w{Wqi$V&TsN_6|?x7C95>oI54bg zY?KOc7GGUcxkhQ_HhC441z*;!l#ke>*^dV@Iy{CBNSa&K}44-o{On3D%9-Ww!hCx4K1MC-|oO zy?OjTYio6`-|4xT(hmia;tx(pT&U5`aW`*|=7JjYd+~J#dI|%j?q%tnx~OXPCw`LY zG>s_LUh&zd4YyyiY>82t%F$kXGSc0)rC{#$(_Q*cjvO$H*=hXPqjYQI@xu!~9Z*l- zX#X?a_2A0S?!2s}sp)!2^+8{^@%?7oV83(LJ%bhZt5^Ez$nV|raa;TD_}7m%74PwV z>#KhE+t0#Hla3`9u6B{>nLEw8xbaPg?Msa%>g_*@iu zk}rp&vctu@+uPQz+K^Cx>F%jdX1fo3$}w*K+xafZ_Pw>is%ocO`y(vanTtH{Xg^n1 z{&4oH?U!Be5AI#8V%Bp1d-jg?2O8E~=Vf#Cd-p#0b)K@a&RjPg38{VI2jbYSpQ@VM z{?u6HLF-?$=W_!xQsVl%UA|rW?{z`S^3KoDDVI|paBkYhb>eiu*~$GETQmMGOFPDS z^N?h)%%?EFL&^o}`<7;E@0`vX@t5Vq=?DLq7}w{UpZW34mFI@s8mHWN-SC428>06X zuP}be_EbcoyM4p{<6I9ION-~qDYE3X%*Zii(`n>dB# zyZns{8S~_C+d20YM1&tn`T6#Be}Dncjw;JE1q7sm&^lS;Xclol;1;WlvN?O;;);O=j#6dAZO zEwAR~SyRD+^t7V4*QXu$-{5fYzqakyYdbvH#I!aDH2v&wm}+Oj)#muU%}u|T<=Eak zf0W$0M4$at)OD#lbJBm`vF6#!u3j^a z_e}b^{2Wd5ca)aD<2DPuHlse{egERNQ+u!eeBy7@%b9aEC&!@V^7gPlHQK$u4js6c z5+T6joa}K$Isduz&bN|rIu#F|?|l=a^u9{pZN__VM=x zQQJ)v1)kLin0+%}_PKU)hQzGBp|93DK1gF)=-IJ^G2+%9qq=Ub1M3u~Z8Yy{Q+#+< zhE0A&&HVF*cITZMMKqiH9n7SEG==fL3sy+zX7E_BJm|sf|69V@*ZQ&)%Cxn+awqXG zV&`vC544|9o*lHcgMH!ju%&V?EFXhDK5*@5+PY=SLUzZcazO%1%(iG2ZeMcnR?6+P z;LACe)n?y4`OZIrt2p9@0$;M?|L5OkUpCld5Rg^)!cC4T^D1-8kF68c`efu5X`370 zZwOf?8u!BeUX#epe%USJ&sKFDT@dm#&!|=4VfNG1=ra#5PLrs1-P;s#^U4+V6Hol! zExGUKD=iSRb(>uesL|inf3j%ldhB`i*GzL{c+`qCwnxP zZ~i><pRyewa!t9$fvdnVrFkao4K?k3$P?96696 z7`Wu$yv6K>8r#gKcqn)6*yYctbzxWiXF+$Cm3wY4UTyCo{uGGAg;*z4 R<7KPijLT){vA zMBCVKDJUo?=sV}cc`VLlYA{0|Nt)qYV@cEud@^Dmj-+ zALIm=e6B!{;ZVJ)c_|>{K@J1aP><+4WtJ2Nq!u~n=N9DWfxK-2cB;N}eolT-a6w{n zs)D{tYFTD-s(VplB}C38H90>eH6XDl7d1d|m<0*}hwBbTre~BWm_iKEhX$LWv7tFc zKDacg1RUO;xu5_CTMdZ-X#7Exg=eOKLfgOqDYz@*-dt4VVo(rpJ-F7ad;5nS_rez^ z7)pf9xW26>LhV=ZOJM;8CM5+27M6|%20<9xW3{IK^}ydth1}<#>AkJpEW4S1?WVlP zKhpEhe|&Rv=VtNpG^^}=HaD&;nmPN-bH`;#)6z^F*B<5Ey7F@R;qn6Cc~a-)*9jP2 zHc*EB>>Iciy;%jEZmA3i(YfA{%6<#TN8OV7)j*X(?E z|M@Py>z2>ktmkQ1Moyb*ljtiS_a^3l(buybDV^QQTs10UOpYWZa+Yqy?c^fVQ&{rAVUFS$N)+53rlpG`bM zGEx$Dr-+w^*?3zmsW|d)?RWA-TZ8;+<(A@qj#pj3G*Msx4|M%GI zr=FR%<+t7EgP-30UsRu1v)j6r@#TG!e`RGN=h|K^@mhAeQBZXHda=#twOl?O3on@6 zbkOhZ;fkFB(_0Yo!-jN)0 zmWj9-s|a;=^}JS#QI}=$1V(`G+~n7I`w<*p+?GW6$5XmM~e}ymzPac7CZX=xYDO#@LbUaYR`3@(+P_ zkzWtRH#IEP-LqAW{ZhaT``D94>r)dlvpE^Q78hIp-pEm>c6n(*b#Oc-qw?85pc-xz+9a| z>xZ`Im@ga%`){ga{=nEPhHLxEj>ilSjyt?~_W#%Oq}O8B=`6~7{)*jhR1J@dVGQoy zvhHh;-!kcay(gWt(#`grTN--c+Kyf8tIJicGpRn#Hh=IzbgSpp^c}?pPyG1G6Y|yb z4fq>O?Dj>@&-;HNyM9;D>8eE6?6-u~kT|FoTuv^78AUk0AlxaQhikaOQ$r zIah!A%HC$2_Sw@^=gqAC=t{|mv-Q>ct22)X+;e5`QJQ^Ll+1j#d}f zdkfvmi>`nE|Lx5T1_lO(#0h`Bv$^%d8jEe!xW$o6I4FA&<9jLHjhI!8A*gmju4CXe zu%RKmTGU6Yf{hG7)v&Ri5jAUJLnF9-pmHBtmq01()iBtJ$c~3rD5l62Lm`)ufdRDA z1vejz^^6Qal`>QeLgBB7!EQsg2&49d)RmB$6H*gLE9iUXrKDD*robxYAa^GPLj@ZJ zP@BNOKp_nb6zpJu2&pKEsHq`#fdUC)AFi4jQpIU52*oDbgaTpDp}b9DH+&=1!e6f}06oFmq@_Qbt&9%w zIIM>#xWfM>8MOuWF#a?qTXOwc8bLwD4V#=-@^NXQR z(spqegruabxG3s>u$u3E?fbjm=NW&CNdEp&?}x^ue&+|q&+o|Zd15TAwD$-*lZo$z z+rN*xGrhRz|01i)%_Xstx%((PdH&aU13Ektm_!T7lRPEob zNE45BY=1Y&wC}Uz{o8agv?IrtuOk;U)t`HbnIWpWBWum(69T8R@TLpf4U!ex>fbZF^0FqDau}+ zuzJ~oV9}ckl|{a&DDU-uFB%XdT4lsn8+GBDV6SWbhYAK;tG^5ppEtIx_L|DODPZf2 zt~{Np%@TR_J`HkzWp+HhxL)UyXX}gSDgPKsYEr()&9&IIZkM0>T;Gh#{(559Rs6;F zy04#a_UnpE@o~3b!jA7R{;eu5eG%%EsJYy2*SvgnRBi zcpv?x!pYl2hgIZZccw-v5`XXE*=LPjmS{+L zD|i_6HcmdPS7q4|CqCcpOHGe`;I0V*D{};FpD!tkd$y!;i~sxiEsN&f|7ou*eZR_I z;z9kL&+Aq9IUU<&&b4W~+N%8>9WQoEUe&z1^s1h?nC_SF6EYYI`~S6cSIWHEP_^zD z%Oa^4o91RQ-dI|`p(yvV+YZLX(#`onds!~r%)7NHe5tK|MLqKo)rJ-B{HG6=ESxMn zspz(OU!7CXtMtq&%lU2p0$N|N{bZ<;!(2JJ52h*(I&(>$(^FOb|Y-gaRow94?7x#5udZGtj?0I*= zwBZrc%UjVa0^E#@*(2p&+U+`|9wYKe!C!mY4@1$CD6L6Cmp{JvTfrQ2x>DD?;~mT0 z^`$A3rKTQg``B!(w@lcLU7=ngLAdip+_au^uCCjTu}_Hlvc!e=%M0D3wr>~z<2iFF z@7ClM_M%^&pPF;-l;6|*Mb%3;d$Dj#_#8HqaS?-Thq~*o4G$G^PAqL&CwOw{lv4|@ zcbGiVyQFKyS#@(!vVg5(Z=FQh0>zMI%f))>uk>Yu_8gxU<#;0KmhM&C>JLf`2QM9Z zeerC(($&X{ruEM;`OJC1DPZO6CMCJe`CbkSB^=j(;K}8xic!>?IFTuALz|YGQOD%( zZ#?(MF$GQk@*?VjwvLL_CwIxWciOsGuX;%+^_GDaloZ9|foiCPn>HL!WIGx3IcH);iPh$$N zDNWrZC3JG>bg6PB!<$E9MSBSX?*fK9!a#^{!H@I?b(xRf1^Eee&rCD6+b*@)-cwSnv zbH}|NlTYi)Od8zNMdyU}#I1jJw(QPAb*c3{zhl`q8NS%++`sxi$Ayoqw+df;v0FSp zST;DZBVyyjX)W(Aoc?)7VVkZ?ubAzti_!CsI~+GRxx}|utYrJ_hUwbNR!w`Nwt{WL zgjb)AK6IFW{8Fhb`wJu1Tfgq`YSr$!w5a%3XXk{H_^RHoCMOmvFWWwGl2e{QWyv`| zRi$~k*S}xBs!?_5buq&MvF9nul}D|#`TkqvJ-c{!v9ZrmVaN54w@)l(e0EVV{W+g) zPs!|Et2J`=a_>5;zO?4Qec*|IEh`ESn@CEgEP3|%)}rl~R?Z1o^*Qj1++LaXdu?7? zmWpC0U$xd}PK(mvzGqN1^TJZs<%iAgIepQX-}f))w|mDrt1r)Osx)8l+Enq1Y6 zK&t*u)aESwM>k*6+c43UOH3*E zqEWf95C7uX3X@WIJL^Z?OwdbhzOpEL&x7-kMsm+LhJyltJ*PB%w$=YRZak`+74^@l+-jXwijBDeO@yKiXVP)^9N@y-(HF4I8~9i#=BxV-M`P<*7Qp@$gN&|nbf6Y-Lh4p z_9~0y_UabuzDVN?RD99d`uW8JE9>NvLZ^cvk1n3M-4^ud;%UJzNpJ3KyS2D~rgi93 zk;x0h<@Of6u-tB~^3tQ~`-_tdyY8%d`?}Kgy?#d~$V7Tdfh6WStWt$H(0%oRGyF1GjC0^PRDyO&72?N2D#^5W>yyR&xaXSUy4 zd_UP|(XoZaPc+)kacikB^tOt~{FpMs%w@Xr@&)caa#QZs{$+e&-w->0dgi86i#~Jq z-;!;Tiwm}YFRQUj<#v(sqyQ-=^8(=sy$g)DvJ_g%{taw6mlx#p#mMl))U_F(nCFX? znEX(fCpAN3t%Yi?_?5fAa}T^In7PkgeZlW1JCeVApSVMrchkAeg)d^ZxU=u9xRiXM zx75i@bT8M1$6S(&*)tC=(%-bILilZw*Q%2XXEW?|&u?h#Sa1Es>w@z4gq_dYeir0! zZ*dpr6@Q;4VJp`kS1YjVlz00&_Y2uJ=@(Bfd@nfnh|BhMPgFEoDJkQpeU8ykiT$@7ykzYPyNk{GUvh4m2mi%bqp$-` zPn^y#Di%MVv}mN9*4nq}Mh#e3C*3*0|Ht$qI@?QoS)7vEeSu9q|3 zFBABZQ+DyNn-SYZPS*An$F5sUQQ7P5w!*rk<;73ct37dQdw3eQFJEMST(QbeO|Zl{ zwS8UelKVgU_L=5a^2StcfBh}`d+4TMwp2^#q*A?>SetcZCZnopq`CPxcc#O|fRGB}%WUX{vO8(Tx%u?Vy*nGvS_d}d zUF{Am5eaeY@#ET=v-Nhm_+Ad_qK><{y<*XaSa&Qx@S@ZGxXgrN0>s}THNB`Kq74bH~4X*yXJ+9wa5gxx|qH*Yj36F#S%5Lh6HDjo3h!ehe zPcpqF^h#C9o7!_q%jD*+KWXvhNe55NC3D+f9@E!K+z9q_(qDV@q$P{(OCGN~H6@>a z&K7h0e68YyKjX#q>h}u_J3=kK)fGC3ypM>Fxj#wJ(z5i~g?_h-(jwcqe)PU||F!U; z++2;~%-BsE<&Oy&6nMtW-FNIowo&0Wk+?iwk$sm%w%ub?sWNsczr57WUGdlZ4bhBy z?5iKRKT2BUu4}bvMW5%0XqSJeu-KGE8eC@|o zi_A)1Ecqs4B!7A1{lj+@tC%m??VVfeQ6l-Vn0Lowwvz_C-Z{)JV-FNH+%;!Qk4}~S zw|Ts*TSKaJRHR>jnd4W>P&xIt>SA}JpL1tb2-wQBP3>GQyxvt&zFjWR+}Ze4=d2U8 zo)4(+^O17CA1#M#c01Zz zI`9X)vR4n%n{83jTd(@zOm9I>f;Q!Cz!SNEK$o}Rttna(cTP4zbx zZt|XLQL?_NDPH^Yx%BpS?_N14aW7H+>bEzZT70QUx_0qo#Fy!OoCUu8!WRXq&Pu*~ zXHeypHcy{-U*WCA?uR7q_)nK{n8*;iGJJEjpZkT$dYm&D*S}wOP*(RO=S7#2E%Wxv zU7nDwQ+4>ohZmidKD_u8sr0^OR))mhQ*SrL ziYUKW<=XFd@v1E7S=WktNtP+|0>BZTIlH`lp?eiu@ZsN79`^L2I^ToxJweI{_ za$=&K)AzHR7X}#C^th%^@UeOF&FEm$Px$YzoE>c#^Yi@n( z`{mM8{MIf(Pp3C0zpq`wU1eaA991IxGkk|^VE}j4SzDuiJ${j$KEJEP?5>rlOVww8 z4lz@dZI5$bqh0ynsoIlJs{}bzb~=Kz_P{v(&}ElJ3u!N;y4WQ20W1vG}BOtg_B8 zg8UUezo>pBx=393!ctE|zSzo={|{!TyqdpY$D0eWVxE$+f0h>8Tt78EBTYtj<>tjj z1^0G;JHN-b40#zDqK2{c*o&k>aG}OxifjrdZi2UI`;ImMBm!oc5!d|tBXt1 zRsL=LeS2Yh;Mz6UUmdx*L5s6q;(~3}8ztLUX6mtiS})F+y)hZN+(9-I^ z+=@ymM_sKoV*IHy7BFS#SG?p&iszOMK6NdSbGOm|R6qW&ybOQtZh!s6-H_*5XG8wP zs6%b8^_{D96=vp^*!HZGoXb=)EjR38)cX^Xv1~U?BTOGFKMna>e2`h}tW=jk?P80k zw^C)QLV0@j?suFesg$f{(h6|T3wEv)R#v8})4Wl`hB z+QWD3cb|DNe-jvffycS73##kLUkeOEHx&s3D&Iq7F*PK0Ui`w8a)jtXBW zt+}W3beD8d?*VqE0)3WwkL_MOEDDPL(sb&@hbQwn%!6j_|9VGtZhXd_`Kcw_stgR2 zwx3yFZncPQfxzC97p4yF5)S>IZ5amn7x|-0rm~fPdsV?PH#=#~gU#1Io}bmR!LsS! zZ@#@wxfd2nCy4x&+Hj!uf^dJ{qQi4*e_izQ+Wxxf#h*1%vlB~{w%t;`bY5WgRt3>% z+*unOzbtshkbP0SW7Zz8@11e%lf-N_S@-tcX-RIgW zvfF*h=4W$Gtl9SBteIGm(TyGc--!f(Gf&=kdyb#O@~<0T?AKW#VI(Sv0({$%!rh>fikJ+lm<%SN_?<5Mp0_bKc9X zw0_;Io+h;<(iA-!}_q3MUr6Wpma`vP~`Vjy_Sm^6a%|995Ur9pCS1Q>9{c zUA$OrZ?((z+0WhI2+ZAbx|?{jn!RZ_WR;_O#Hq5W$)A^ z7KgVoY%h4triHDSbgZAJt3I{2X{~+C%ZrPj{ML?^lbz_?e|+Kdo_PU#MNcZ6ylDIC z!ra9>^)4K*{<8Y$b+^x_&*d4kcRe%PWIq4xGV84`|9?wqXtNWl>X*}BpVKH~wZe1N z)@^?M;ud?|o6hZ=cd_2%>-MBw$Jd4Sg+x#arnOuvvdb6oaRogv{<3YwPe*ITZ;x4l= zEW3KQ+KR!ps`Qh?((Bt73zozN?5&+95c+lX@pS9ryTY%gvJ^DfZJd#!bRc-w$*YU! z3(viA%(O@2?A1 z;(We$zW$vvuWDYd-(tzu5dC12NOs)G@BJq(2*&Qav$$R-Rexc%eC~sPpWd{Kzx8}^sh6uzvM%+pBjy{?$*XSbapftO*E z+~3C<@||;DMJ&JSec;ZaNl&`Hl{S9~E<9P4>Q8DR1%T2!;)Ufy_$|&C=0bwqxT? z-50{%w-*0j_|z`_lH!Hk^6$j5^QIrI^Gm!kGuriQ{hPbzvu~%#9RGM;IkPH6U{d*I zGsUI0Z!=yb+Eq+kSbyMO^TG>`H_tAdF0$(@#-+zPxbn4x4PP-{PX}XG>*Q zT+j>r^74MZeY?;u=Lh#1t~|N*!da^r9zdG>ehx z#7>5F-e)dMI4&m0Z91!adyl*HVos&(uwBU*eH{5YbZm#^^kY0^^3W zx-Wbl+%Gs?mp1d~2K@rZE+%7n8JqiSV$aUhE8yf2G3cK;tJ8RKw3@Nco-N0=Z5NUK z^yQ9$q4U-^b4{wQ%#@LuRuILxzS;Rad$ag#d)>wRD|sr)YvvoibK4(vVS>>EhC&X} zhA6XyLo<8NmD_9=SnqkDB*ysN_E$G|9pBP=@oTxtG9G4akHYi!IQsbb*S*-9SXgLz zCWuvRa#3-x>Vvg~3k=(jUb+;*FiUc+lw(j|bijeft`|=h&n+xj|Kc0}_S;IcuNCJR z^s9?99{9ev*71?H!}dRc20V)SAJ|ky7cu(kE`EMqGS-Oka_`QFKWtr|H}8abisUjp zE78{Xh}>kt+K{U5C?p^&VDjPk`}tk(3P0JD?~rpY-urvC%WV~gCpYG487{8K7737w z^^&f8?EflovdZHS&-u%=zP>8CWVCD1i^@6NufI)7_;T~%oiiuyX9|j2MQpzO^4qE_ z(yR^d9!*~0I5$zVW7eUt643=bduGRo_lme>mmvCsbP zTHs9b?4RBX)VVpN`9><*0-j@>{pZ5e1^+YQQ-c>+YgZ_3Um$<#gx-szxg5I|oHux( zX~bRie2MdUr5Bo;PAi1BFF0@eBDU{^->=FO?_)2di+@R+cz+(>-XE@CoPCsEXx2si z{2KjVr*=bm{r-u`>2)9WyI<+bb-8@ykYiZeZC}aiiKnj~aJ<7;Vss2_XK-&fo~vkTq!TZZcR!U$LYe(H`;qdGQ;%*Fo7gCEjivfA!|cp5lZ`w_ zHmsQXc$tow|EoXilp<9#uf(pAKhNA;8j=#q{O`=?IfoBBU(Lv?PBK65aGMZ4L=>?3!m0T zUEuti#&9%U{#%LzT_=_jupTM>w4>y%<5%J*&Jrp6@OVh{d@7%8}^U)yvjSsy72Ra&p9raWDDcg z{g^&)xp}hm#?v{nKdqCZnG%G80`?(WOUox{;P2MXsp+2Sd1#8g{wp!*FJPjBAUYmM4{gfuV z(mW6QrAIGSL@Zlwv{Q6T`V!$2AD%F-F?dn9^Jd~Q;i~hRdu8r!TGSY~Wx;PY+qdrD z&kM(gGR3XC`>{-oyWxVdj%CTq^h;*DxT-F6$rW_QF@4OB`+wlGHIu>@rAbnHYZ4Y* zIDBZq)C|}20?}VQrnevPNN4!x63)evHP`PaM|ocHy~tYiFMsDnTyoGCN{H+~xuM4W zr-*=Hd{-Pt$+vRO$uG+1eQ*h7un8^S5%!+xfz3;UD)9dp?6CV9_Qdr%8TI<5iPOF9Lit8VkUzj=Y&^@Zp3t*w8j=$;YCD%mhW zT`Km;7Tb5xjkoSIpWmhRzJ+h0B(Ljw2DJ$yzj7H48hp_?b=L2|?aNPX#g5I}%QCx# z%l3u9+zXQ8M!a7dp0nACOpSd0JdCUU#+~2qD^gEe*TnAozVqytkeuSsdxwwPe$o3~ za_WmoPUWJ6g%yn_7P8xZNszEAe{uU!%>siL|GjI~zV58K{6#&JeWu@wYnt4*kEL-2 zUuxpo^>dF*YuqfOf(^NLDhZ6Cmy?_04!#$ed*Nq5iR#&q?YUC~%r0DX@c+hZ)uN}= z&Nug^?6;RK@2+0d&F3=Z*u2hD!+i0zz3ozW&9?_ut`@s1-7EKX;s0kwCH0FKIOb}WSWn>A`BliQ zu>HlIEkfsgzFB_Zn&*1{(aAj@Ctld;?bY75dG&(DZR_G2Ll*D9scE&)e(gf-yxqPr z-)A!&kneXWe;Vs|Snr6?gBKo)|FhPIF-}~`;&4ZhVd4M&X_+@vie5C%tv-K%p~vO> z@8|bKUwv*)?$x_wU86JWg~;3s|8rLNw_1Ej%=5xE<)JslJ=K7j(LsNRsKF{rMwHF;u`Ng^ZhTZe0 z^S|0gZumUaxaD8yywrv==@+wCtNl87P4r@|xj=-#>;=-cdp7*H6MWGUr?4S6dY*6HvWXuxzn=9xT)%d!$%~W8zdtid7OU4k zd%zX<{Kb@)iOHwj7I~Qr{4=tq9aZmPD-y72_ zE|-YR^r?@%r(XPSu92;~&R(5gr)Ix>k@4kSr&IY;MU&+jOK*fsHk+{c;%?{ZzSWzZ z4WF<*-qX1*;Nza{wqY-h?G{*DKO>^Vi9hF3nv-$(i+dVhJoa)eeUR{mC=i) zcKuD)d|bGkq2*r58IfK37oD5u@ziP@Wc}Yf@7!OBxD)4vYUX~9pa1?-fc_zCYx{}L zdxbtI&0~D;7RxA5b+t*Zc;2-O+!ekhvCGO?>pn7^vTqgV_)@=Wo}KU>{?58x|147X zDsy+vW2*nDbwQJFz3qhJ%P%?3Ed6M6aBfHV+864(_b0A0dh(s2?10Fw^Zj+5zULU< z_vbm+hVtFFvoaBQF+-umD)n*4)vJ>q#VqokzsmiV!}@7WstgmZFnGIa3%>Y$db?RA ztHHAijQ97YMaT~V<>ih2& z7FW#sum65^@XqeJcb#wk%%4-cyM2GndhZga@6s>08H8VGbJa#JZs%Mi_+n*Nu6>=% zhqbTX8J~NY|Net@00aB(Pfx$hkpFh!@G9pLr}I)X|EV+WIqerbxt;Dxp(GBN?`wuxG!eU`$K-&x9`6?qvH7Q_jU75@2V}? z^S_>NS&EC_3(NdP`pQ>6=a})i@ZT-d{db1pjsIEy6~a})Ej2H=JzM>Gtqz1 z@=mFXGvVI_&V;awnjCYRF5o9P395kUCtwgv8Q4OI+J4T2X%`=xHSE>Btaa!3sUK+! z(ZWm)FRqG5G0bbWEb*RuU%+;z0E3f$d_I%S;}1U8&M6lc&va1#(l56r#)QFWy{PSM z-V&$y81_eORx(xh+U3r<%=dfYz+J_)LF~(B=|#(h8694TZn)&)|LekLyD!I$jRGBB zy#7(eZ54BkeOYCd@fWU-DGUO??lRre_;O@n`az!s!d9;%Hw8Ymbv228(ef{LJ+E!5 zr*xInnL{t;UU*pCcdXBOUe#C0owFB*=L9iu?9Eza|5S7Hi_;&yt^C&;7&1(GIa^m^ z-Ja)tTKgrd3@vM|7K-bF=F&45FU@pNFL@ZNBs1}4>;Fv`ynT2W7K(GsdNI$n+NZ;P8Rxp&%a{k@20W()Z_ViM&S9QxD`PQ9QVv)=AKuh{qVj5hzT?>h7(Yj12_ z_QLO78?V(meQ!%CIF$4G!Y0i*&qA;5{Ve#x`q$~t*8ksyr7ru~&ED}Pi?{#3RJ^P@ z!@~UHd$TeNUW8e-pZ58ZaS z_WLe8?yPe>a>a$l1N)XaTwznVZ^LlKy@=s;LvFRy{HY9!3>gJ>3ClI^V_oLTu*U90 zP4heHFHD!Nx{A*4Jf?qS`#-q@y4<|I{|v7@Z#Cyy(EKyfDPL{YwTIT7f-4o8SNb!| z(>!1L<!AAxA7fyTpESJj=ABx1&#r z%-Jkh73BV1-(5fa`iDi2Hfer}uD$24od27-BAHF$3sVituh=8JB|p@6ws@IZJ=QxX z{FSql_sw61HQQvAnV%nZ(+EGil;OYJ4Hrg%UyCQc;Hf)y%F(2Ep2DN6T*WselWt$w z{LkHn@yB*`uDKV4*BiVjXS6@s^@ZQhtMi|>ec)wn!}5uT*R9ZgX!&JpUCBE3e1^yJ zxjlRR)l1A|{+`e+`gM})k4>%nqV_$L-2Z%2Vf-Ui_3#3B#kxs3Q6;qh(pE2c3Nr@vwRZ&zWx>kUiY%8kaimsjUW8N_xT@ArFg)$2yvJ)M85 zHJ1yxSJgdM6+C}aq3X#6*WT+drY>}SF)cGi;8U{R`<8#!EY}^)<5+JCznx!gZR@Z; z^=h`*%inXnuY{db5cfKjW>A%GKxz9h@Ss!P2)qOcK+o8lMz1Mq#X43Uy<-A4iG7}5JUo3U}&VDhu zQ?ATlf^rGpk;U%{Hm=n&ebHq!VUqP8%L&~rsY!m~ar_>;(Y|Vxg>S|al?~;KtxL`?V1E}s_4Z{$W`U~TxxQ(7W=Y?*Vb@m z-}>k;En92mAAdOWP*FqGGI1%}Df8zSlsNt8ZvOYo?!IKJ-oNE#bG|IB=KI^|cz)43 z#thDq&3mSw{WP!1_U8Ai@tNGST%PZ9x_^A_G0oe~YnZ>;?47JovVZ2w=?m0vpW$v> zs6Jopg|*S^{kwWoC%>F;`r>2TiIRH{9r`Xj?5)ypOorKqE5MYe`C4c7u;y(+O%U$geXT{t6bz4rLY(2+fJ1<`@M)P_P^v({#XBB%jd<% z+nwb@KD^eb=IETqocGRF{B1yd*47snmoN$JdRyP0wY(zIyYl?v^gW`s&msd}AHOJS zwea}i^N+N4Z#H_7Y4FQ1B(mgGaPPnWJEt0JD)sJ*n$0xtjl!3;6JO|EzIi#rR| zJ1Vp;UbM7YxLo(eJZE{0qn!@l*Y})L$w@JmHT{y*bV2iS`EAGy2F#EEdzoP4a z?)HM`FK;*g_-)LWu<0V_UY3eD_xNnb^92VQH}uV`w*3HnJF-#1D=pRTiIvT^5H9ac5rxq3UcEKa{6db>d5*RO-!i@g^8 zo@8G1;$p~x;AQN!Yfq(CX=z>-`|`=#Hm`l2^8v3vQdQ-~MJ@Aso%U~daaEy8h~IU; z_{T4IFD&@a7|309%1U;xaFwv$*IDQ1ttzPupUw3({@f&b~^P( z3OhXcY<#`&jmpD~Rku$s_;Rpb<>LSBwih`<3>k?}UjAgL%88!0uV&t(fZq43`J?2Y z?H2pk&0K5xb;pOx2EAvTPwvQ?GU3I2rl!B13=32*EYFvf{8AgIbnohK;VB#O zRn8umrenop#rH7t%laROFLJkNihWM-R{ycfG5)>S+zZ`eUs95vM>jf5UKp8p!FYrE zBGvDk*{V`6o`0y`QLOP==zECf`HhS6yPW2&b)C%U@Zv^A{>AkR!^LcmFW)9`l8s@# z=-iWP4awH3$IbuU`O~efJ<&;jljl!Hai{wyf>!wWZ+MUw*k<*`g!%TgU)`%u&rj!V zc>l#^{tW8{RWG!qY_IS8Hdp=kob<^JelIL1*!^a|IhomilIjk-yZ06zDSP5R{k_ir zsn;1aOn24!W$avck^6DOizGn?wJwz2yW-Rg2e2;NT6?u}OEJ?7lP}qEFN$8A=S=XrxW0b>!DOZx zvL#$v?>Eb}%sX?}ip@!Xqm=N0ISZeE4!_H3JKMg*>G++6!d0z1rH>2#V*erjxz+E* z^Yi=JnPwP&d2{lR5aSuiDlRdh1A7*3Ub^%3Mb6rP*W`t(#Lk#mvG!i3L2DLn3l_TNWh1F-*$F#`_9_U#pejuZvM(iTz-=7{0JdBKIB&s-N-ky`T`a9>i z*C{u~CC2T3FUmhY*K}Ll&SDV#LiOE|#m2U?O-r2gt6p^8G?d?e!Q+MHyALA8MiZ1v zt|zB2Hm?14LGyBQoWeZy5+{9$=c0>tzPhY=`R~0nl`6yR%Ntt$9oAlO9AwrGk-bXA zPV*-!l+?30SQfu%k?ZO;XIh}p5b3g><(CZWLiYJ0e>)hKNzQ$|IhduLc0zm&ttY@JfpWudSCj?oU{*>Z7XFt1b!Lj&)YQj!`yvQ z^X!dYsG2EAaRx9jF#1UD>P}nsUgL`~=Z5Kun_gUak#p?==ilE-pSjl@ofFO=FzYbG zkGYi=A?u>?1Fe~HTwjz@#cb!k@=^bi!BVyG zB6sIJgS_+Cn3-nW`*g1_?$Grc3wKw4tz^0=^6AgXFN=h$uDZDET)6w!hEB#uJ>naRfwlqbFJ0F&HpyNyKs1Wh~SCk z@^dc)n~B`zeNkl7;dga`e!Lh%v?AjfnO&Btn_s*>7%dh1&U2a6+zWxu?%tK>8FQFc z?fJZN-vg%?WeLxeWL@}mf8E-%K)7m2+~3qY(Tr&&ZhaRbrGF*y*UD~Kt#_$|A&qx$ z6ffI_&qtp|f0tYHb>VRqgQG7j-C_zfbLY#2ewR<1U;e_MGhy3>yFV*aFII}~pS}L& zNy~GaGv8h2-0M)nTB63^d)MKddPAhE{!zc`%(r)J^~}CTyE}#&t_k?m^G|cxTkw(*Y}cshP2FgcZFxWRfZ>5ne%!xT<)p6(l(FB{k+I8 zJw_kHsxbHTLp!!fO;ewD?`XY-wq4V47K7O@+F2!@Gg-F2yD|S%tmU6b$K_Wy%rcJ- zc)@Nd@PDs^K11kd`w1^3ZRRtiaoYZqt?8VnCModWVZn3<&7JiOHEs(wXS%7EtdDzl z-=k34ag7{PPc%o!GW!WH?(ggkJ-2<{23e*Vs$ZV6c<;!)VAwHFCt3Ab^l$%V5_2z1 zUSa2~?!;q(J)0)_wn-E}Wd+E=`--hp9r^E~E;>rH$oB5!PXwea)i5-HT6n%rplX{@{fid6dSkto zd1s`p7S8{e@@|#&LizL3Rtu-k*7?YhaPETnjWQR8&#m*4yBVJG&9zl*sC2fs-L242 z>AZhNWwF!zz6A;;Tu#g3ggOcjA&$vH(xjivG{gFfl1jr*XMT=^cNdDC)W1}6IL$w7 zH}`@v(+vABGZ>!n?Nw-ybb}bP(Qcu8aHro3clG%)3wZXKslI5LS1g{;VQ1*DafgUM zyVb(w$G82AW;m$s&i-iSj8lqVbW*p?y~AP4xaj4H5~t(dF7u~-@_SL?`QoL)mow@O zoo@RNI_O2qHqG;suv!@2kbGF^(d;>!dH*R8Jep8{!g#AU!yu+q{pGq%O&e4Cb z$mpZ6t9Z9(rT#LuxfgDpb~-QUt7x?_+}u%L{O6Pf!n^Dn>X;W?nlD!3)c@{^j`*cSw?}Xk24v+oG^=(dlx5ayQXVH)Yp* zclgcRx9Mb`;Fg8eyLPb`&6(S6C;R7HfBW~B(v0jnyR7m~E_~kM^&won995nt{l&R=pHyhMfM|}!uj3Ql^a~E0wyJt~p_1>&pX(jF7!|&_ z$G^zkd+Sa$)553TCG*2Suvj$-y?x=gz3Il>u3kO)P+r?BFY3>2UUz}>Z=Fg*68n=2 z%*oRn7PX6tx)|HIi7EDzq?TQmu#n8=7I%%aa=P+HoC?4x$`_>QnZ$`bzg8H z@}{Cy?>*<~FPA2I*erbNJuf1EUctHZmhz47_*1H0U3fcv_QB|deQ&uME_}arp;gDG zTd$&TT~nOtg`KY7lQ;ffnR8#*dha?0j@X>-a;w*DdBMchll@*MCe^Cy?ZjHE^E35C zY>q2bDH_{<6sw4A$jLGb#;{cG5Z7GK=y!J(`wFO#pFDBpp7`EZ*U#C5N z4$n8p9~FPY(=hvkr}zg0Mv-52(-~UM>Fi?GQYcq6@tn=^W%5O4%Xt!CycV->jkIH6 zyZc(8;X=26G)wuZY0jq6T%MrDwS)d5=9K_euexGKaO z8crw08JZryB$vHW>y?zPvw=ZXJ6~rXA0x+HcIS)5t^fKN4}`zCHz6}(`Rs@A`R8h;1S8v^nRZofdQ)En-d|zBebyl_^Xa!Xt@o}bnG7xSvcKF| zxVoL&!0?l_?f1TjMZ3@0Tz!%LFX+p=ol-{A?k>8#t9<2|p6JDUj1xQ;pZ?UJ_(Epm zlILcZYk$w(^zh@1&&tIgm+n^h60=ad#fdrflWFktm(knv*XO+G5q)vbfZ_a<_C=z5 zKV*Gf`gnoQcXlINSB32pU(D!l-hJQZ>gEhje-k;j15-Xquk)98IKRj9Pi&ggdI8Cj z*;@{NeCN)8zwAY-&8OGhR#mDO^s`=6roU)emvYwO+`{x#)n?cI<~?Kj70LhJ{+rI> zh0K;)L*pL@$t=`jO0PE#Jl~-HrAEN^`inGDdqJH#rPck zzea_}Nqm{w>8xI2Co}i^;&ev+nX;c%bzkIrB)7|1ez{ds{arlf#Z2Qo1@UXy`<%}^ zTYjI%{_Iq>?mx$%F1dKhP}BE z>udNw>$&5LeM}c#$SkZh{`kv(bJ>}Ns#!U|4xcigs(a(T)=b-_zhgCw|8ACL5~w=L zcENdb)3UyGTedIVE8=>^q2KqRU-Q1>ZFLeFU%1}QivIWV9V5rxHRaz;cYR3{`~d@0#ukNxtO&HvC03n{BqEf6Zpk zNO>!sT`!HV6fAtb;H0GPjKk(H)J}T0->YA%lDjB8!SMM9-(BKXvxR%(l(sM*R(d1S zaDjQFl&#c>61|HLqf0ja_q4CN)Yf-mK{eCNouA&V^qD?eFZ`UqF7J2Zo^Qia1Fv*T zRY`q3)&I@NKfUULLp+CTrTOwF-)oFtT{o3I=st1YM$LEAWj703U4J2(#Cz0fue+w> zg8j!a@pdu{of{;+qP#>=B<{tik{@qG3wzY6E)qY;#Va)pB z?$-PsrWG<()n68V?p*z(<(Pvd@VVWOp}2jXA*kut z!0?REcD{Xy({q2$grw#RoV9JuNHbt4lVV_T41yEK1l;C=cMKe|#qEAC?mI4!^0ZNB zXxX=W&4ttZ->E$2v`ucG&#~@98^fF!z5;PB+vv9(b1!VS6Ri4m@}hRPUfgl{mU$bN z9Q=MX{P$(stBa2LSv#tK5pv+&yX)!xTldRWFoa$cE(yHHxi@3c^bdOM8u<-zQY9r# zelIKo^xt#dc*<~;_X0P=KW(SwJ9fy}GWSh#6z<`*jjVe4LPURE)4M{Jtv<(B|DCw5 zi19XuY5nBG>lYujUDwwr1n9aXvnzaI>UqAk@pkoksk!fGdCuhexPy@+ zwn^Od%iHk2bKKqcSQ!?w&$-_Ae)r1ntLmFKzSy)!xJsya-lki#-*7SLv+wP0lv%iZ zukSa{3zKIrcpke**h=`tipfX99lqx=1eA67z0iE3w$H9g>JH1@gz|R37Z1N)od02Z zlU%=W!-d%Mb808~*jHG-kYx;57O${s;r^@!p}O+tFI&zXDf4-u{y$gf0;la~)m`;} zKb_NG7w~-k>SL9E55+Vt*ljoW0`tWD`5d-ad*2E&q@OaGsxK*Uby;F+@5+6aVGK^^ zS<5Y}zWh^tcj%plPwKzcXy4!UuNn6L%Q|A;)y*~0?Z1!ouenn{KC@ofe)z)aAjXjI z@#dMQ%e5x2O3k~l|M}Y+=jXn?_UeSwfqFSHhZPt1TJ3reC-eT7q?7)mX_}LmY_GEO zx3pv!wEjQwIC`bydGiGq#l>eRvI)H7ohPkR_=EL9cFDCizK!)gjxT})19@$qJ^huo z;DWhmXr(->fJBKziOzwA$%d=e9{pvsCupHu=U=|Neuu*t_82k=R7tLu{GeuKmZ+)v z%Ql7SE@Sjx=gHR@4oUq|n&Ec-NL?(4SGSx;iJ2RNQ+PqIUF-p7!RhO}e*Jdfx>Gch z@2YXa$qA~wUqTin2QOAW@$u%ZrJ)TN#TNm*}h@47JG{GxD2M6Isi3(ZHD&#X?b%Kz-W z^f3Q}dnfaiIt9e6^e$qAGb;FBmhN-XX0v&*ix|+uoR0&iWBj zeew+#W{2POV%T_V`pK|g3~y4_UugZ&E_V06@e6Ii|Eg?18#cHEs2wunyt}4uXTtKt zqkZC&*&JT%%t^JBxUX&ct3TX#zTAiXp_gX<2oVPLD(39>P&&um%-1Yex_Ls=1>vyw zk6F3r-Fmon!AUO1bfdjJMJ_?N^cHy*9@J(r{c>0S@bgcTJC3U5_#6(K-0b&a-&N0+ z{J;j!64}2mK3zDH@7!-M@WMw%a_fTEuMEB^lw7$`95F3MW7)5szuoVmR=7RkoEPelp(4u9W=Q@r%oX**@WtakJ-iYWwh5E&N?O{Tl1VhFPn>?X!K#EZsS8`(pRG z3xur%OfPmH2&rm(7yaeRqS-~cr|j)i`D9)EYvVd6?q2@DvfbQZg0^7on^Go+7kk>y zZJGB(XbYdM@v2njzr~L(Y-oJPS#tMx{T+w?-45zs_H@4cX!ES}%bb0ana4LcW&=B3wE>aWtkp+)2QaBhME_8iUE|5{5$#Y}bu_P#29FB)uY%empE!}HrVE;e}# zPX0GTFT^gB{^GOn@FMZt7g=B4U3hr?!u)r=m2Q0(ZqJtuwkWB{QwVqOm#NaYc%8NW z^OUnb=ho!i`n%Lp@28ab*^O)q#iyNgeP<-r=z3n@_S(#lgp>75Gg@LI^Y1M9vP6B| z#qE>3#IHuZjc?WeXMRuNi%g;PuSv^%XT7+yW2M)JH!lKWqtt+)PvlQ&=4rR?gcO`Y#FOYX<4Tv?I(&i29gzStf1WlQI|Ul+SQ7XKF8D`jrA@b}vnHI@M{RxF;K^oPAOkGnOF z=VK!8-N`Rz=7+^UaeUi&VDGm_EDMCI6faCalbgq8o6h_qRQEFX$r*gE=Vv;X*II?_ z{;cq2Q~DQXm)aMNPxbyBeEssF$CoRU&(A!)Z`pAMKHGOZb1!V(AbK|_C5XK-S~)!W zyPhEb8><^@+TL-Ogx(X~#ZuO6zUrT3j* z6jdp2Z?$mwMbQh|JC^wWI<@<6vA5N}4ZCHntT(6fgXY zaV~#-PkHvz34Qmc)SDaK4Svz;;w~Gzet}%FfYrVkoEzsx&DXoQ)~|6#+dHGVh1Jfl zvOcrjxHA3z+@F_j)_6P&|7A1xd9s`O7yT5^^$+9Q-tGGC_(g>8Z$Im<>oO;{E^2?e z$x;89O^y1O%ZpRoHNLwmp5DE_dxxX`HPxyY7kqy$PYOP@J8b`bi4(C3Yp%8G?m4w) z?ssnkomu{3J8IK=_F65hK7Ve~+g~%R^WEcr*2u=L)I8<2=>Lkgb1J{MQv3r#v!eL8 z8={s^>{$O>a)XsyiBtSJiw55Hjd71mqwfi8q<%TSJIv~`a&Zx#O^!{Qf#Vz{o$IVs zeeL|$lzz@N5>uMT6c@l0A*pP&aD9pm;{mA`GTmK8mKXSyPEF)8I6c{Anf)h!( z#0SpgU&HJuTYLMyZn*hTkNi~Lf)@@agf~u=m(jLbXrH}czGn%;KBszrPqDgRytQj} zMC--A&D?)x!TvtCWuM>9y<6nE=zcWwlZj%stTSBBAM5mc@g_1k_4dE_wu*}{aBrBO z^+m(_8o#VZ((XL1*Y5v=npX7HC0vxB)s=M7_`vPQ5yzfcT;#l4f33#rw_3tQ@9O5L zpJw;%|9IwHRAh{k_#$*5x-nQvZ};EKWw5FLw?!vp{@zZ1&64-x<(EZWV*DBP|3BSp zxJdX{)SVg;SGAejr-z2J7cB5U^Y($v#Mv>-AH=y9L_9wq&ES{w`cK`GdzKFr^}V_O zR;-r+1^(x-l4lq8H(wWx*1N#WDr@nD=YiDUgQ1M`$~YWeRNnX`v(x58%HwS!>*W4c z2Q&QG*yd-`x-Mxx!!oYD0WX?!^nYy3k$IJs(D!!XYxToUi~W-YesS9#+2=K-K6Uno z<;6R#7Vfs#C-p0e*H*0|(_w$0NWqJTe{7e{z0myAS-oWb=Ni5XoPY0V-M*hP^Q%J1 zJe~#5iw!0ye-Y}7PqSh?6FPsMQDtW7v*ruM?ejRlG%!5lvORuhVQ}U82OnE5aMrRl zU)UUK_MxTI(!y8$ONr>;7yZu}=NifW_^<20wauNOc*4sA`&fU39xCHZ(7P}{qTx9w zH)n#}uP^N9=k4BaApW9d-uzELoz%b7Y=7f?{$C%%GlsnnQr|K7fi^67GtKb(a+3!< zlI;W;WQGrE)Z?O zX;%-Lt_V8yrSRy{)h_jiU$f2q_2uHnH*GJryD$9zv^4Ag_nTW52>)6!ui~fG3wxz6 z(^kZ}UiG@!>sI*PaFIs6{)hFBQ}@_>DJZ+p{bA~2`8eUf?~HD&xKuwkPin^E#ycNF zYmLg@-a4!2-kvRA{C?I8ZH7e)tHr1P&Aaf|(|vsm=bk&yXQ?^VSnje``K^E5^>{`} z4D;Ta$`?nEEdGBxd%^MRb*G*zwLP@`!L60szj9u9ypX~B!@5?v^WMw;r~Ezjf^$)} zeZhxI_k0&vmc$6xW~x*b$bQMHJN5f9%fZ#o`)w|F*KJW3eSE!psldI{v+_NkvHjIk zrT(FyYa_cTFtij(r@tb z@&?W7R`Gtl^ItwQGL+>uhzOVOd>#Fv((>Vpd)j-pbe*<0>U?qhc=-CKXZ7CS{;-A4$uXzW1uiWc!P& zOJ8iY+EJSOg~fbs^?!@NyFJnYF>k7;FPL7=Yo&1UefuNpkUHi!VrnvX_dBZY7WsSW z(T7EC|ALJF{@++`zPB{_x$J@e?pHeeR?mtZlx`S~I(Tb{I(h zy1DaL%Im+`^;N-_b|3inb-vE>VB|p~M zzUzM5uB!fhlgsjE8oD#I{P*vedy?Vm>yy?IdndZT(7mv*`9fm+`Qjzp!@KYGe`!9w z(7asy$&+JsM;(Hd{~ny)lexb3k*zjoLCO1ogv)mvE-`p`=r7Rbi9NeeH_NU1*8V(> z*zXt8_xQ=ziB6ZBv-0-+_y4a1Z+;*C|MIq#-~Jz-@k*?#ZfgCvA7}Rb6?NKa@=G{5 z%Wvye=lg-3m0r@`4p+jg-SQh$u;(qfSEaz?Ijs(>To8{;TZD z-5_XoNjPU_QRl_i*B5g{pA|W%;Iw|()_e7=(hai~o-v!&^XbJ|jjC3az{%d5xL)vz zafx1d>t@RQBFpO6H;1z|3Ag^F8D6Opcb&X>C7ackw|y3>`(`)vB#B+x(r}~DVcAXV zzUBYc#&>=Q|EjQS!Q8*{7Y&6UnAuc$-_j3twF}te?U+Pt-nu^(mYDOMcO3 z`8r+CrE9b1#+~1LYq$1}ubNf$$G1qbUEH`~j`U{%jU3-94Xg9=&Hw5pJA}7RTE=*U z@c`$GEj2$ER_k9~c-n5)jHY$AR_*f{xGuQ0^L6F1eA%ydn{O}gi%souOgp$4gf9kvZgiin%eu|^ysp*js}+Y|I44R+t&X4Ga*wHg z=|{`%y2TapGn=-&n9+4k?!EeNGi8-!DpS^qS#zHMHSNHkb?+97%RYKgYS`n)?Pawf zr!mJ&$##w+!*^GaWo4HRT`5lbtWa_i4C5+7}!l!gZHX}n-uryOp+KkG}c zZ_$z^2VUG?GhxU3U9N05v^F0~UH8V})%r~rl4dJ^Z1E~gKG^VD?Cy-Ni`pX1shqCc z`HTAXIBcc5o91YKap_wW{!;CU#;%?VlNZ+QI^>jg=9=B7plc%e6VfIt?fn>QcItYS z{etz26-t5|&7_XLFx!_^sQE?3>3ijN#-gb0?M1llarVWlN_TUQ zW))8j2cunq+FH(86XG4#PuL>Rv#-xIfg|?u1G~Q$nY#2Yo#uPjxHewgw#wD4OU$8g zo$fA!wEK%Lul#Gc&|5x4B>nKqzi+RsQ;XsF5j%U*g_VDe*!H^E_1yb8FS_#Jcbzs5 z|8&+^=JM+{Rfk{5UX}Q=CAM+Pi=^{wc(&b}n%}-SgERIA$KJ>uSAUgO`Q?famF{Y+ z6uR7C-F*4tY3HWzFJ@iryt?eWe$^K)!75W;+tRloi+{b~-IlJp_x7CWJ1-uQ{iSnW zi$OD2?56BP4RLN~(S>3vD-_vZ+1njhztO_~ZKia7@8L8P-)|GHO;nZo{kQtdr7JW4 zo)$ZqdZAd)XiDeb&b&az6P&R|R%_em)!RzOdit@HRPFhH{zZvQm6`TLK9Ac=){0p_ zW_B{)>i)l4o{Lw^GHA82Q@HaB-F-)1-_>Y1XH)fVecD&Ux}1x`eNydu%vG~zitP?u z&MZ+UmoVEURk~+iJ>%ZQ67iQ!T!$2PISDTQY&n1NgFj~0FIrVPpY^Y(O}eOgNz>=w zY^MW@;y+(#iAzdwId^{L`>S&k+Vw2F+%B9?a9Se1Wy8ahT(2&?xy;V_B8wx|-{Jnt z3z9kuxZV8QcgKDxe_(pTw(8D=*^*zxiYImd{~wsYN~KpyX{WexIRaR`Vg1^#M zJX7Sn`?W!8zKgV@i(D+&dU0i+?laUi4ORKJV|AC%3eYZ~Dd1Lmh^@wkn@rtlJ*vZ1u%#SJT_1 zCF0F5|K687{^CUKueBM>ydR5RJAIhfG|l@FB#~RI8YxAv3W;xG2ci+cjRjR{!A@$&95u0Cu6)xL1xj4&a72UDk_*`bOc7L4U zucLJg3+<(S-shg&m3(owq=Wdyt@pILzO+2Pyg0jT6{{xem#j}6G1ngOmOpksSX}PF zVP#*u>7@7co%1IJrN!)+z00`jsQF(Jy~+1n&+fijkuXmpLR9tV-=4b6FAp2Dett3W zRbQMh9sHlwe@=AOOYS1OhWqp!Ky2}_+Q+b zD$p8s?fHRJsUlBS@t22eGLyf24tVnG#dZcCH~yphTi4Hgvt{#bopjdNz{zDSvXb>P zIBdNRsjg6y*wHxA#Ci73h5uPxWmZm2Dw)>tu6B(??52rs+OzM+@!76bl>PPWTdBhP z9apFCjWcSx$26lYj;)06&8%5}ye+3UMb3<>HSfPDIkh?Fq{ZIsgWm0N>CDORqvhr* z&yDt4vrs}-_+sVxph**5>!YSlcRDZQ&>L>H{khJsm0DFV3eT<7Z(Mih!t6aAeKl@u z>4%oTzWOp#Laz3~M&~2zn;+Ny*i|0?-0bt*q_?yCW_DhDZCm~4g7j={y#?D}Phtyl)zeRHS==4klc+f%2E%rkD9j_wVcQvd}Z__}44^Ph$C3 zbrFWsjF!B$g%Y)T&tsRm>suU2=ikfZ`I*DQn5Bf}%Ut<5zQ1CB_{1;Hp2C09^D8&w z--CxO-2ZhQ-dt3>R;qjcT7e7Qj3pZ*xyxDSaeZGpqws~*s>`<>zqD-*P_CJ~uX6sy zxqt7SxpQE}Vh?vIi5u*bF6QfrKR>m|-}C)xk39`M?mVTt(n>t`>+Ba$ZOdYgc`bft z#f2STZwe)>%QCpsSgyOY%~fA*al7HZ>phhUms?yeWyjikxn2Bv@Rt^Rq(U--*bC0z_Yt!QhtR!{%# z)Vw!$7hXU4biRw?ndWut9vmsm*>*B3`3?Vvmnm774fHZh?!LcRA9^8O>B}}}abrV^ zabY>B&pA%kG zv)la2mf9m#cl43Yf_}3nJ+oeIogi^J1Il=8GB<_m6{Wi^c{z8xE+UZ(yZM`d*c$OQfYqm)jMYiNk{XN~gbA^Lx0;l{- z%lS^smp$Cv&zF9AnX*xHlAYD_Id#`A{Cr^++`s7aT`{@6El=Wl_{w*Lob*|6{chJe zwu`YIF;ca@vlhRdc&+*vvvlEHKFP(KFRfQ(uFAGcXy<%UC1U$j_D@X5)2V-7^i|KV zlKS;?+3XajQ|S*b6!+_;G5uZ3=eoSMMD&8RA=lb=?=Q1VmZv)OOBbDtD6!h)UjMtz zx|MCO*3PqT7rZU`_Ey5hK^VDngjy#LE8#mumDs+rdNrq*Zs-46$`#%@cKxiG1X#b+ElF0mbh7Y`u<~hrBd`2vZ<(ru z&z8JiZ)CcdJC5bc#?0j{5`S~}?kaGz);m?5U;I5^r;62qZ=t-!s_MsEuseR*p7v@ zXFWNxX!Xk%F&-uQmkvv2FIwjDd~!~BLv458bJ^IX_u3P)L#Y-IdCzw~|7y(p2j?%@#* zX3V8kJH)3b?|ro6pvo7k_onr;?X1*ud+U~{ADd(SWopKwrHWrR%a^|RknAeGrM)6} zo6OYrkDmpmUX%}i{bSM1nHP4iQryM*rYg}s`15mVhV$D}Rrh_nu;{VXf*h_D3gJYr`m7gw_T9_hR`g~5D!VG) zj*DO3MZIpXsFRp$ozpnGk`2Tbj5P!Jw}fCdVzX{=)uyS&qcrXTM)HT9`lkaeupEy3Q}%OsV1` zr%COyXBWBNx4jbd`SNLwErr_Rp&>F>czo{dsx7 z4DCNBp7ksUXAOP4d+w9xnsat6UiCLT^25A(WvxFp|I_!K5OGLUJu>b5zj;?*E9Lht zi#=t!XX9yWy}QQMY$a1fx_A8M;uI-ze${(s%H4mw_Ay&5?*2(oZZye?yi&fe_MOH$ z(TDD064f%lex>(bYmuAh+`Tk-sjQV_(#?hUD5849_lKcvZRN zOmT1By@mIeo6T{Zens|T*w;IYU-M=>NSn@b=6t{%h9$1ijy7#r$e9_;yYCD?dwtIH-(NV*SL=7)*L6Ruh+F;BPQ$nA`|BrvT7PW& zO97T=7wQ`hEuNpFx2sEJll;{G**$eZ`}QSoeRg80=9!<9KLm>&f6Q=BeX;Z%q4fBu zgDXC+Sx~(6#Ebu|v$u*qGyi1Nm>C{cc>Jz|T>2$f0jI0Z;(L`+*FA5&SZ(!CnRjvh z_pNM8pNQ1`pD@QRP~F7;no{?L+u?V9nx_{EPM*$~Vb8QYjdS^O#GdPd0nqU0ixPwumSKQm$d@c4X0Lt#>G8h5|*_4WSWR<3`P zzLBl&OzO0YqKoIRO6g9MwMe_gw>N6<_Qd07W`F(ayt)4kQ`O$0vQJMQihP+>bw7QR z-oE4Oq*(3mCH-G<)A0LpMw#E7)4ghry!iiPqs#xOo#E#`>%3f7|047H9Pi!lqs+?9 zxZcff*>h{3_KNehb#;bX>+)tbvUi%WU0AQQce{su*^6Cf{om#4&6i7_?(4ajFYWyC zrbO^wt8)sn(M^+9yRC1#s2_ZBpM$-X>GA4k4MJ}Y`*wxh%w86fd|3VWiHnlooEPnP z*#GGIz0gJKVqrWBxrk>L0#K8DWl~*NZaLbT%;sk0>0%)4+{x$ zV`%-qU$^*ACNnJ7(_3!dv97g5)=k=RzVrRG>6wfEeCIe|a^%AKT{G=F7o7KFd#km1 z|APEw^<83La%2}oEizBsBlxAglKqSO1@qvkHS6}9-L(2O`*`ZNhm-$3IB4+gnxNH! z`)3#AFNvI?(jsU2(8Ork>KBKz*F_#bzF6be64}3T4f`f`8TfIu+Ej04`|#Jqrf1>) zS4;u&_vLF|EU~|r`{id!`~_!;Wf$Yuy}zLS;24i`3W z_~%q#ZQjCqgsbWv=Y#3(f?s|uX@0)vmg|?QJr3u;vDZ$ia0>jQv}Tb$$Ll$msE@to^(n&)u-Tj8bC2D|L4B$d+F zW?c}p^v;(JyE-GUcY;Nauiuy7|7UVXI!~_lR^L4R3+(KDa=#y$ z7k;kaxqJS7kG|J@5AG)gzF~FBzkK0fcEx7j-;1JMoY~hd+&a5UZ{JUO4kaz42i^?< zg$(NxDtBix-IU>3-_NNqgXt5y%g&zr(PVjEvxV0>Eh84@rsKST2v-ZXY z?yn0q4ZdnC#p$ek&h^&Z#B%@h?eDLyj^DeyozbEF--Yupek9&#sV?4itB~3K)}rat zyB?mrt$uf>6oZEz!`|kQ8Qgn+9zOl9|GDS;LeKrvjU5)o|Gs?s&#$@V`z||Y-rTwD zqkuQ(oz#;1skc{N_?_oiHet4hHiOIBMx7UD-p}0jg750_FW-(Xy#D|Fh1qrs*;qg9 zdGS5iEkFE=nBGE#Yo`VF%C%%V_lVUdi1o~~E(x8qsx|Mz!p%#WCZ+8P^<4kA(vLgq zn`CSpL)ER!B(`+Ty8kcEUV7s_clAl18OOI)yZT+-XZS3A<{ab6pS#=n^VSBxdr|-Ql zZ>6Rnf6R{mV^{o@;I}tF0r`N|c^&sLdAodr{|$`OOlM zSpCW`UzJ?q-(9%gvZLTjPS}!ooev9-7gZfd|CKF%Y>uYkwjDd7T;u9xBez@&`0B@X ziJ{}->BtxC+Bf`SUD?jv%$n&tT=t;#d?X4SV& z2<+XE+PF`n_H|uDmegMN z*)ICbC7*0xUySzn%ox<~w@TFKq~iP+YwkOL)z0*}+xFtc1#^a7XQfLnxGwnr+NXqP zzs&EKcki9M^Ko^D{krMZ-?TGVGU%L)o4@#Vtb#t{k(>9Io-xrYy_Ngu;iPPXFfU8y zyh`u~Op{hiodz+$G%;QpXULm*jl6(Jj1s?g&P;vOni={okPrpyt`QOaU z_x8{16!GSN2bVjt>vPwCu*mrP_4mrmYZK2%sjlVf{wJOIs%f$K47YHHFr|4?E1171 zX>izB*>V0#>)E$TrmAT9TsQy28_VU6XDR4bJ)zbORbBuGixI3=4tPpU$)qlo#Eq~*@rK)Gg>XVsCd=H`)$y=e`@buDt=jSWaW0D zxWCV);(M^$u8xbBS6%43*xQ)$MSLPt<_}xD&Wm@qW<8PHJEP}I!Di?Dz{fK#o<6!b z_{7!;HT|29-J5^Kc5z(T(e*PIwcV^_uQL{WExAbIF7v$u3s;9vu|DIFy`_OV71173_DB5-Sp!^2gr8?>$@V+IR9_ zXExZ~X`jm$`>f=8wgHoR=dRT+ySkmsix+0s+zSotx@dZ??A?tWKkO#$S)S!~@qgW3 zg%a+-eN{(ZEPb3kF|O#v6c6SvzDcTAb|1Jp^M&=E_|&>DURw9fG%K_CcyDJi*(80* z;gZ!$+}Av3{RI#6#K6W!$B(xaIO?`tFqN1aD0V+JThU^&d~e^0t+%tE_=spoZarik zTe-$kcbbFr!t1jY1dcKB{g^S`@8@gn=`8bCI<9&VCFN~t_Q}ZQxX@;&J=0y*#yxJ{ z-&Bz>>#B;|{Jo!=E><#q+9+$eqVshOqh|zr<_^KyRog}X{!yt)l%DYS^q1;CikYZgPS!S>f*`(Irnz={YzVlUM%{)R(F?B#)N-LEAnFV&EIR(9kGmk92z&_ zvchHwsmN>Wp{m|bcvIE{7cO#4y2W9;{>G{%gYO;b>z)RwxrJLUZhFeeW7)~towBM% zXzv4YVSC5B9KYSGR;PWbFjFxs^(wruGDf4S`RB#$lekt&&q=G_>1p#T^Nx$0N~~?G z-Z{y=pI*$ptma;IOfig6gYClZXG+Sgj9<#Kx$bT$UKB2IgwV(GI>v90?hoUVH|#4%34?%U6|BEHZlv|=6WV#U3` z51O>TT~f-xbx1ELOd(JD+ooJtFn@Yu--f^ijVUXxUVgJiGiCZS`Re|Q!fxXK zHmX!H%v~03d8zg9(Y2}v3ylvR|Kai6^QuXdPb~Lk6`SoBVi>CC{=6l$-HvZ=d5E%O z|2?O7vo9@_UXV~yRd4jC;Ef%NLeAACD?(mgyj!)AU+8vs&UA;H8hw|=dgk%@PRSCx zxI{fU-cWe&#}`Xo>rbUK+_l`oSoP?WNsFzk?%!_NMSK@n1x1`+T<_eq;l=Vj-|`G! zoMMyt>;6!h+4$X>Z>>^d0tTi1^Df4B+x?69B{<=j>DKtrE8@kR@$;VC+^r}tVAVQp zNsyYy{n`afmuyICyP)0f$8|I&`Tf;^CCwK@tvctqehr^HORZeKzt~6jj*#)c9W%eK z4=|__T~PTH&%# zev!DG?7Q1K%q52!zPxA;Id}BM+hySn4eyHcV~eu>PI+N>>9pn-8#jBy(sN!#La|0w zZY*+hs|*W+o!?(MlygG!%jBTs$81lMF1!^HSiZDKWNx>xTm8WqrxuwrxHB%k?qG9i zaW$LMbjGSI=}%!S6Z=%USnC24*{Xu07O}#A=+`DkOq1D!ha&w>5R9|iQ!+tR~ah=eA zn{z#vcZ(-CiDmZOGpNeDUb&KEZ_;zAn#Z3PKIgOTym-DWKc-#6d3}rHxfgkStEVoj zns%OdP4h+L@V)$Ro3e$2PtJ+Cm1?DbPbzbDPLCju1}Quz_|1D zzo)Zy^DeRf7xiqle(#w@*BPxkv-<6(+$)|vds+3B_s@Fl-Ul!6zP@aFgR4B(4&~2j zb`#a2Y?6KU)^Rq)f0@|VV!BE$mgU82-3gPt_ulGl)|HD*)w?uVu*8tr_F*&g7e0Q$ zD%D?o6;BTbKRkVoQ#<>mO4YljD82>Pzx?BJ1eK=&{O9Bj~u@8S$xS7`1|CP z$*;{{?t~v$l%K+zo+yG-3rf?slkY9HOm@7@0Lt3cIE z--V|Qzi_dazH(~1y&-Bn{=T7;H)sl<J%rU7zmwEzMX~u6*y|o$H&Az4@{9?Jc&mZA!*Y@y`@a zlrj2*yytM)cr2(=eh1) z(MiIwyJZ)4U#YW5k+Yra#vdb&3*Oy)usj_3fYf$A@sUw^M)2jK`;#n5U)^e6jzFHf6rR9{S*&N@C zR|0k<9++5XwRrXJ&$>M_T2H(#-rsi4PVLgt=m@ucy|Su#8=e;(ztS?7mrvPQU4_@{ z9qY%)*RPX4Mo&%T*>G*)^3RKkmxRmxt@+=fQR*+aK>2pyrG?_(-oBp|{@_AbYk!w< zxvFjNI@>B<e z9C#6GByi04-J{GE6O_gBq(e*ms{~?OE}yIs4>9_-USNTwQ~gQb7>yk{?Y3^uZ?COT zyQH`1MEk|hZ2Jp0pOzNi#4_vgUY}Rj|E;|soz)*F{A-^5&Y~Bm7I$CB z+?p{h|7ZN1*2TrV>x}gFTye-2UBLcSv}WO(kS7<)9_+IC!Rj4&ZSv*9QYR0a94o1p z4|k|OeD}V0VYXpU+}XWu>wF|?o1SfNzWST_vd&T!?c9cpD7nqApXW~8xuAUJe%_0x zzbJpXGH>xtjW29sd!(VLb{FU0{J_# z^P`^r`1R$*q+2rQ;&w?$9|<+!ixrsRTX`d~q&lv&+H6+h8mF8XKW+U|?pa)UIR(%Amb#ghn5AxipE+++j(GRE?cepD@RaoE zEK)G#x^4FTMRK=B&$&G767`9@^dgVGSXXs_`^5ZJh9!J+jU=m8X8Lu#{8DEmzw4Bp zW|eAXuEueQ3ve{+uZg;S7_gdyJw*CJso#b;bhg->=X^-<=@Z9gf!r+q@egEF_30{8D z>%Kmbzh*(u@06yXzD3=hU-q&tUVX){VR}*D7rBMmHH{%LKTWNa+O;!PIbwHwSjYN* z$<(WI>&GYeM<#7WgY1)AVhV z{ujTH{JPUZqc`DHh3T({anGd}=wE)i->_xgIk)>tUHy*vPZ+AC7Zu+Nk5E0MzD&>f zmgpU8EvNY3Hu+C2wYR&p{M&FyHu=GYm2<5WOIS6lzNk+zIeKnu|Aq3S=j6W5y0=5` z{&}Y_*A_V^GQZ|_f9`Zn>6OOR6&!beyxPmk<@EeR{;`YSCvJZ#)*JlOCE)qGbrbUL zJk_l^%*QhAsKYsiI})iOwi~p&?!1*-*<4rU{Ni9*tMKOK4brmjv!bVT?##U?9NoFL zNh2%g!81>W52X*bI8UEcb9UmX-SMAqf4{T;VT@w_uAUz$&mY_>eip$nGgIMx`m+b; zFDdX^mUL`yWxK@fe0@nkuiuV>j$f5aS{CcMe9x_lnbP(mzufG7RY~Rvaef4IJsWBM}L_iq_@{Hod>^}s9eRJ(^{Y}6O;4I($V)~{%M z%=h=V%$3w|_dovA{;W0R`>Q>55!07@TMx-FFIv6$2)}W#-tvo|l4l!jP+xcHtBUWAw z$=j;;-ig`Fwd3D3s~2jV)%VZ(7VX=n7C2**`*nss&c6xgXC%vfw-n6$tbAgHuYM(e zm&F+=h20&m_I^2kdex@L7p@i-QPp2MV;{d}{Xbbk@v1@9+}m9n;$E)*9CY96ioWqL z>4#bD^N!tr_FVkW?~)nb4AVE?xwB94)*{V`vOd4d3OOYgcsEoil`P@2P2cP5(i*e% z|3stTTCRQ96{d3se|?e8u;4j&v_gDgha+d@eVd&UCGmX0{H12Q z-g~H;+4Y{1*|lNsTjxCj>#f(sZEiRB$aTJ$_f@3x?b)q2e!ZG)WbPK9-1_=qg|s@` zt&as-WdgSDe5a?|JDGb1n;e)tH#2_0AqA^YjlGuMCgBgoQodN2USq%D@82)8D{#Z^ z7_Q`5^;2w~HoXsA{_Tn#+w9^uZ;v~c)z19B;6umTYVD(}OTM#Rnl!I(v2dsR`i+x{ zKc0JWWzNp$?nhR?eIOS)V`q}QhR@cQV#zfVCNQkB+hg*U_n+j4k}d03etHq5am0Jc z?*8(ON%O_tcBfX$EXqH5^Tlkjy?YtAZxXv{vCFd7qSNpAmRjMWJNXY+Zgi~(RJp_N zyNhdU`VQaKd1Bq?KHg#|`{K=STlllI_)_ez-d?{22G8`ZH@HWfT#>lpg=&}ZuA9G2 zSrplx{uW~0=DL64$|=)cpAee%s^j=I5%yVe%toJ?A6)v9%d8ogv2c3tN~SMj5gT__ ztT%m@ao>igYhAByr;5}5R~H%ouQ(yvU3lPUUh?l3^OGAsuX7c)?y-APV$)!Le24e! zncNQfwpF4F=6`>2T=wd>Sq9Id5^p{Ad%OKbOT!uet55f<*_78U&S7Kch2RMeTiRP7q#<${Tb6!_ieXW#a>I5k^@_0tqQgp zNS!`;ciZ7W-SDkD7kOANxw11}ueCeE_E-IkHOJHzHVfxYp2DEMNx30z&3cEsKRIf1 zud=M_*|+hoVppxGp3&}-ll^&ebJ<>8-@ZN2Ml{5~g6;aY@QV{;51qYk%CmU=GsjoA z>wa&YY}hc_xz6iX(_69rI^NFQ_)`pjJ^r3F%L~(&x~#XJc}L%&KSAqa&i49+O(@R@ zUr_Ds5Xk?_;N`0P!f3V^c@qmb9~kd)HfKmV+TZ!d(rl^d1HF%5-_|DUtzWk^aI$my zs{Fd|3k}nMAO9W9u>I-7`)l>$rL#FpcZa4weBiTps;tD>_C0sb{JOrT{{NzG|Myo< zu8O_=wQ9@B)l6m^Ha@L3YME!Y^J0opI0hqS$#3}S-Aa^-(<<&(vz*5+|n*iUtv-C`Ob{L?8iS| z+O}$gZ&_5c71vjz>Fdv~c6%2U)4GOV>n>x=?$AH_)#Yn$nF+8b{E3;W=ImjVmJukg zxOkz!*T6q~v0SDTo!R><8=HQsg)O{bm>;nE>;uDauMe!JzvnB@nN#D{q;>LO?*4ph zqlzCN_jfcut}%bVCn@Zol@#aQ=5?Q%RIBtu6FK)rJKdjkBI3*Bf4sX2{f(bwEiY0n zsx*}C`af^tK39I@sZVO$&7I5j4u6uJq!WM4Zr9zurM(X4ue@F&{`(tLv;PcJ6tc9l!Ldj*o%B#GgB_=R#+;f3rw@H|X)63VDeR|prjpUL=M#VQZnZ1AzTe#T z!rx0#%l8<67U#Ek@yqAmdWUks7c5qdehY*#C=3B)Ran&f9$A`P82fMai!YzgRNud& zyU5*9UAKQuSw8cx{N~l?sxGeld!1`hmYMd8qph*(R;jY>a-QBdZC6eR)zVng#eG@G zDpvQMQ+T$6`P2)Wr+S{h*64T3W93bE(ckj^XT1*}d*LYk;)|`M#z$+(9TTPQEJ|N! zqIL84=l?Ia)t3C4$T#IvdgZDq&gJ!{ycaKbz3^w=_I*|DYSx8y&y(x47D$_yt7UY^ zc}{IxeS0_4-+#+8b4%)({HFw!=d9eN-0QV!q4&gVIU-eBy}L4M&+WgW^z-Oh*)MKI zF_ldJ)tl3b6n1T9XS%kwaJJ`K#`khOI=z-Fc(+fH;fRnoy`PZzVgV~#gz$&N6o(?0 z-K<*~tGBbBNvxS6c`m{3fZVeUcE0M{b1xpq@_CzCVi0!mnZp%F2lcB>GuQeZiw-$* zdc~Tao}`Iaudbch#Fr&zti8IfMn7&(#~R0ytB%)Yzm(XOOg-P*x4b*<)Xtu~J5ov( zJ1lylr5f9{a~%pv^gQKeevNp5nBPA_l2w`5{v+-89i)9iEW%nU3X?k@cs z*;4kM`Ag;J^p%Bz9fFxx-S2yAo!C&d{!*)W(ZknQWQCeb7+=f~+-b%qQ`n(J7@4o3O?k~0X%_XxduII%X?Oxx1q&9(f(fMg98#jr#{F<|{ zK9$z-7zoG?#oZFPh8X6FM6M8%Gm~1#M=GWf%xAMtczVd1IXU%j1d15c*yy$qNzU#wdi=8jlg)iRo?7*eUvm(0! zH3g%uRL^~wC@gvO8qe#mIoBB-nDh11d+sqwc$NhJZI}CXGCV)!-OXR0e%#p~<>S6? zN#(`ak{2i1{QIJ(-=J41tK1j2^@X+8(*4Pfyq~H}+jR48?>lc47F!a3uPxvIhi+7V zmM5D>^^2(i8(!RBV|9LIsYBnd?_z3L(;v6B z=BjDRd<7Tw#n}tm_2z5s>RNDK^ySZ!3TLnUdiPMFs_68?e9;hPOXk?l2*vAbuHUYS zN{EiltQ2wS*8BZpX?KnH%EvD*y-c6q`of`0;I0mbY*E>ll66v1ybG22Z=ds_6Yf{1SJm)rPn6sncel-bJHMH}?i8g;knkzCzLq(T19yeUB<;iS0f3 zOiuP;y}#>o@5sx_U#`63bi3aCLUqD^cEgr=I!tq)xrsl2dqVbZLc{L=>@U|!A3FMd z_o5AQ(GyG!YA)S9q-yIHueyp=TGip)i>v1%UzDaCY`s|fHjQ~hf&JAJZzgU@pQPNp zX#PFP+BvdkLV0E0$yt1Pqc5`j{Ti-KWt*o6YI=0;LbF*`J-D`QVRK9lRMQ`cGv)?&3ajFRnaD_qNMIs3k>*E+GFa)E5sGFvOh5N?P3bA^7gi|n-mZ4&0QAAdagS&2oB zh5VFK_OnY|&#!5B7I&`J_kKUyYWeHBli`U+z04gJ#_tYYws~QDZiK1Z`t6GE6eh=( zRD~p3zx?{XE}koI>+Yp;JDmFSZZS&~m$8eK*>L@B8A~8LM2PN-ii$*534NYN~u=_jb3^sq0p?B?(8{qKjG2 zPg@_Psn^wK`0Kf~N2ca;U9n!KSiO8# zeZ7D0gzR~%KR$aG&38HGY|{g)>{DS2UCpKanwncncHMTbKgL+ou$wDVKK60y^!uWd z3eFU}hUC^tGjv^C>i_;gqw9A6|B{-8*Mg#RPA!|bT1VpIDbYpu7bkqNS7@%`xO-Rg zavTD6dj~kY&knFZ<0xU)2=&zAowV=1pVY$y)uSMWS~0_x9|R9zFjl7Vo|Gq-z|@ z|8nKC^4GJ??XUK;te#xOH22sG*XMf=JKlb4(SAcosFGe)UVj;S^L|ws)SQz;wxDX`&$WLRF|keh9y=Q5whypS@gSS(ZBvrgCQc5~8;neOt}f81Iqeyzl`cVRm7iwKsgM;94I zY$mU{Dl}>H#eVI|HTLsu=DcY1&|mL72hK^`-SX zww;Vo}BIVE5EPy4*bG4A(vyW#1i$A?Y_CH%Vkmy z%BA+?|GaeV`r_gr3u;-LwU5qxAf{bJ6^-z0c<%e%|FwLOcv z_eI@c;JYB5l4!rA>d(Y|3mxTZ&xQxN$}TEqV2NDY z@9uRnu|^L6*UY{*QMbfr`D3n4p2h8PJN{fsxyliHz9FvaOO?=Fh3mB?cNe7>mSk%F zH@Gw5_{TRFz5De}EDqPZWVm(W(Y)u*>(wtjHTe>*cHy$v--H%D87tNoS6stiCSBaQ z;K7nR4YBLA1jTl+y?nrD-dEztZ<-=4XSKfR%RGZ!^FRI4t$Gt{Ty-u`Q0HLYts|#i z=*5(z=S4RMe~)|_dsgh_nfLQ`g#S;R$v4Y5M<}ahy>PLZ_O2Imf+iOvKm6}o`QY&j z4!vy07aq0eSFkL4k+}QWef2xnZJs&@M3fkc+9rS5BbA~Y!oL{77n&Mv)%>*Ydkx8J;?H zOuduH9q-kCcBW&=+#jz_FT8y-#I8l|p7?Su`ImCFp+dENTl)KYvZP#lb>*{pXuew%V{*w{cjZ`}n#`@{=|Ux9 zRVtUXyXUMbp_Y)X$lC zZsoVny_2q+Ub!w|`}0FeZM9p8TIaoMEk_nFxBYVCOytci#!g3BstgBhNwwPe@l3j@?F`#RR;o`|wLfCi?B1t& zI$iRvtwe3r#b~}o*H0dh4*i+)f^AFuGu>bIe|SF?7Tu9A%!@yK_0Rv4D_^vjxaf;F z#kpK~z4@Z4P3>eOEBO_hSH8%0T+;k*?^ikXmyLgaeDA1Z;_TLouP(lXes=l7{k%&YHB`AC*hT+HC!@;Kjn%XKSHoo_!U6nrt$=u-CXjP^CI z>m8V1v^p*0Za-(R6e;~e!b+NH+x>fNvwJxD6j`>T)S4_mZ(zgG%;K1HKC$S;BVT=7cUhrI;QUqIi}j< z?w|ak!svxfS&?AkHOotvZz{Q{i&?*1RGg-_`G-_l@eZDp$;sA>InRkZy8A8va!Km$ z0>fP@MxteRIDOK+wq{?PJafBOYdM3@x=otZmid=7W#3z!bo{c|>U!PfGnXXx3hw>- z`_me;HGWM=m5nb}zvf$(G|zbTch4`2Ha}=Gk$bsZiqVRHmgZuXhY_hR_S4PR++6zY z{NX)a*>Zo?y|cs3bgMqzxX|nR+~mT9WfE3Xj|QeH-&J({WwFv#-*cKpvd{I14_dPu z-?ZE^nxq-8A+%_5aMl~M>y>K)Bqfim{+_kd=GNC2Gu{33rrv3r`mpQ!N!>;9Px5Tv zPhu0Q-jx5K^od`3{|V9KQ{-Ca=lozizvICqVbz$u5|#RO!2v&+a@Q`%I4siYzTRcM zlOxYv1@2(Sod-O=zj)qNa^hoqhmiG&pOdo$^q-jU^{?oUd(@l!E9b7cz0ZG^rj)tG z?9byrgns<4Joo>uGs%9#FXi?YyfEsM z`;z(9>GJN{Bgel~xc7z6wANj&)Y0~?nsMK2^|xWNUv>wUY}LGfe#`l4olxcS7JC(AK<%$2g|A^Y3s<7v6KR0QwS&p#Y+80mG z8B_`BsY|K8xLg=pxk+69vg*}L%Tv|LOAdc4npUj$MXyo_RRO>9go9}#T{1(|Gzq&o$vT$m;4FE=RMxc`|2c6Tk_*d2(!Ib z%pT2;d7t{`G|emjrl|Sw!s$KcU5f7)PmU9|TVeZa<)pnwF1TKQWA6MSWFxD1yI!R3 zqTgpXHkPv_b0VA?qT!y&GS=jC`Tya6S18SAEKo#FEML ze10xV{_dSzl6r{WAEn6%1ThGPM-`4E158YAe;5)C+ z{m$Nrm(}Fo=2w4u@y|MN5<{&l!^MvW)CKnis^mVZyxDwhLi39yHQF1(>e{z!c7L0; zA=O&<=kc|QB|FbME{)sGwufbXpsvVX=Ge~S#nay~Q=HFI-73aX&C}w+ekCWmqhQ<2&x>9#7O2U-q9>2)> z^1`;+d5bO_o#GL$F#nL})5$NEsJO5CUvA+4JnqiNWho`my;pzNIpsZQ>hk;f?@Jbk zYy|TK=l$Qms~%BTo?5>v)vVB6bn<=4C{^|2-{z#H3+ul)sJ5=?1!t{axuVT3d%;`t zA9y>8O$nWJ=gH}4Mul6Mfy!bX>ujrPeEi$9=I1S1v24-y;BD__v2}@ADrebL?eP7% zEI-et#l1bpz4_E4^|`+65WK1aDEAzhnc zMf4Y`AB>eH@gY})U&QlWcv+&a=(hjKMb*H)3(Rsf)139IU#tn}n?CWLW`~bRtii4~ zc^cENbO_k~xz*sTRJ`>t?4HN7VEtFgzRq{QZXdaRecHo&Cl6f+UVZRm zSlVXyUDKPyT?OZEvQ&@%bpLKH?~kzDCw-SsaILGZ-M#wK-QcdCi{krfKJNDh_)%e~}94JIWy8pnmpv$c~!Y>fIa{I$zxFxvCPHT+%ni^jZFfUAum6 zWh^Or@k>E<*OYtDsRjyqy~)zVGZR-aet9DZ74_ruCP+c;;KZW#5|% z=5LpBOpd%Vb?X!_hMBAmg3}f+ip4+hu5fBKTx@;F zX1YOV<&je=zkbWjy>iMszotLx%WDRi+wUd+X1Tq$HZyHlx6i-ir_m<6U9()$S9>+u zR3*-?xxP~C&;!rdO`i_SF23Ba!Ys4liAI(GV;}x1)!^2n3#>Dj>{*okiV3+pRaZ4ve{a#maqPD=B0O*hRgSt zbERHBGn{yQ?#wL`wT9BloR5;-#P?WVzi{mUckHQkjMv|_c8i?#3%XPG;&^CW|Alb|GoP@j57Hl9&t*Txs$cAP){!t!Euw17!Z zUBboBa=*S6mwd5ragL9^^=5&6!k1LH{5D3Fi@S?IxNKj(x^)3_?jqmy&DS*p@4Syc zz_pHHcE!^0FL~=0u?5Yp@(ok%T48f@|Ni^Ri-n{-lN7ykW;kzVvCj~zD?9um+f{bK z_4U2#%~IhV+pl=;3cl^JyljTvzvZqLj}~$L4B&}<%lb&!x8#Y@2kApmmoL{}(GY7{ zckG4tY7dD;7OxiTCyO#{lKaIby$DVKRzAxWSU0rnVPE*J8 zy-PWb|H{ta-W=h$DPa21#r3;7bT6nf9Dj82dPDx*js@S-KFHq=V?W=t?Rw(H%e9M) z9Ij_}*F8NNXk%Da>}zyk>goRuF23I${#RPW{3V%zv#>TPSyq4 zCD+&{?3~1Uvgd{AgiwdHi{^1jUnb^s_*oTh%<*0*Y-Y+|?Pqi1(#1TP1@q_p*Io3P zWmk;AuJ_Ibe|PT@bFW`}|AKSbs(FQnk94y*OwweF+WD*h{l$m=SDXKzKPUUz&F@H| z%zsJu=h_!`F0fQ-nW&Mm-A1t`{E_P+=?jj9lb$nZh8z$UxTkV>DTi$T20lTr3{#1X&wvu^zr_^8;euzcP@7SvXv`_w=wQcd-qhZW2usr2g}bKE8i{k>8ja@ zwaX7r`M2!R!VCF*H?vv8W@~Sclf3-+Q$->ZU!&hUja?x&Y)<+Y-X^BXZ%z*4jK5GB zx{&?NjFi8Q|Cc+lyH8$lJ2QT=o@V?7?vNLPyVWEWK6BYFy>eW-bp8chYxSwOgiEq3 zX8YZ=Jb8(~MC|WE?|lnqt8VwQ*u-*I>FEujFUw_Oo8}d`a8`X>!!R#JCa@{4?mO>? zjF1;i=Y$-N6so?6y1-idU%2GcoWvXZ+}{^iXq zkuLr5epNNAljHp*GYc7Je`}b>XZZ4Bz>7~`S6Lo7mg8|fL}&W@_?C!{$n8W(lC^7b> zs+Io30!dq!P2DffTwc52^^8(hp{g0B4*EXjB~#y=U48Yv;Kp-5I^&dX|FO!td(n{P z7caZBeOh$=!`v@3*WVR6uIANU zdN1tl%3S%sOW5YQ@$XnZnJ@f{!-elp=ly&1JwvwY%7=zqt0(N3kScpoS7`l(Q0=Nt zj)z>fOVW4tFUWG_FX#BX!~F~QksNNZ#_#e^tma7GlHBX^t1a~BBhmNKH;k>mu6nsZ za(d|7MLCP-# z=c*-Tiv=D|j`)A^bjjV`|5HN64w_z1d;K_~`(*X}kl=;!^Va@Ls`}e-t2*u4&EMN^ zxK8t**8QJZY2O?1*xd`G=U>>|re3tUO?dCe^4Zt@D*GOI793gjOkJh2Dq#I(hJc^b zeoZ{(5#=KoZ4wn(clNege99Ut`-huaXR5T+`5L+Jy^&CJ$C9I$r-W+RrSbkz zGz`^TdEZT4;7j+rZQL(TbU*wzfA8HB5S3c7V zW{cJwkY3C>WLKmN|F?603^cEEd1b8Uq1d-?x=e$;+GTi}uFl(qSyN|o27 zS#yeKyS|;*;g`b}|IEpaeY<3B@n*?{_!~PdziDh)b&=b)D-+SfxQ&#s2U#IjW=IlBaH1+;R zVYmJBSKc;bzUs1`VV}witrE-Kopw?8Jqi|NZFg(ej(rivyx4tNfrRki&>ZZ*5Rd7rS|?%y7skHsl(Q4`5VqJH+u436mZFk zEOg%?`loiW!%L1-N!vBwTbg%Q6`9}NTCVt`cVYW}&6n4Ar!RbCd$;@FzYTY;nHM?o z>&l;ulDvXPZ9igK`g82 z&3YLrR3Epr$f661GF`tCVhS3EM6MfE9d!#?edFh z7joGWeM?(|-3k{d>Xj$GS{^7Hcu+{k$;kFV@Q#uRH)IxdT{K-$&%QvlLpzjl*<$gQ zXX4JO=iTKEe>FFIN8QQ_Tl!P~!G**xn({fGj2=rBE!;YnF4tj@`MQ{S(PaHs0X&S0 zXJ5_V_tkpKk4mJH^y1s*S8Xm^?+p97c;WN6^D~1NNLqJz23xB@PeU}jrXC+FRKHq z1&n^(-V(Az?N0U6Eta329b3Zq$R)4i(job(OsHd6E9ne3j~h zPrA0}6&D816HV2fnzuA(=}PfU@yu16(YYtePF4iXKebEy<-4F`dx}rn8St+B6cu!2 z+l`$+nV2dw`~R68D3E@{_Dk)|iI$-=5+$Kam}NMe7aWzE`;D3D*HWJ8JS>-| zNH6fJI=doz+J)1*Z0+97iI{r*^&*+Gvzq4B23=TeZCZ6NMp*M$M_#hjQR69c99#l( z`}eWj3O_FP*?E8ZXOBs#mm1|v+GLEGUuHABoqxOiK}Vh1U-Qcq`y-e-UT?ow4A$ABghbPaY+PcuIdZ^J2Q5N z*@oZacHuYpBx#%a<#^3}_V4Q&{kBLha{p3j_Dr^Q;UmZY)?pVmvVS=JbK!l_5?kTD z?{2FVwJ-nnB2{bpACr$yDk>N+eQW*M6Q^f%6PRIXI#wRSx-FL)o_vb#MKY&4vjxrY@H7sI%L(7DlF?XQ;KEb zkzDD!isFK4;*tkaYch3gt8&s8s$YJgGDrXFi&eez&T{`<*mk9*>D$Y5PJSMWS&PyQ zCb)^ZG8gUPoBKj@@oakzql!sO)-Sbxz{1phN%J1(xewPGAIWw+oGsQ=_r>J4Q;F5L zCliBHbsWPRw;W7YX#JBudyebgBs;;ZuAoJJU%!-o>iEMl@wS*`vOx}8r_S$B^6ovp zMzSvr7kB3h`Yt>iqx{!eD(UTotBV&#KaP%VUsvgP;kRHGC8P_|)e%W+6IJ~7eedJEs(7bO`x`*mLgp5L*B^Ji4?MW^E#6U_GZ#+h?+ zE=uoMH|vE|jrpz6_n%$#6NQtMog*)Nl)UQ{-#k6R;XqZ=3CWrz;sO%=dhzu(x&nKd zc4#;l{yH{mfxGAdvnAUaR_};fvf^T>epSb(g#j-QlrA*(Q1#&kJ_ilTp#V4 z;bK1bMP%0I3E8{ad-SeWZ=D*z_vqEet1dc6N);^M%LG{lo0v~NwRFa*h4$?cpB?ob z99>?h-ic{05VP9uX1#$Y@M7mWf!JQg&1O|y9XoQocbs}*zIDO-&<&H6e=X;?6wr4( z&tW@_d6H)PJo5`@MXGKs(NKKjmM<^h@WtuE+RO!qMXGL{3K8I}lG}1IHP}*v#YOqr zoC{kwMb1z()p+ClWqG+P=b9BSlN`f^uS@cox;)u^byiyO?5=s)C5t@IX+PZcSKy-H zi6fTFzKi#3T)w1`Dt4yp3zJ4jQ?ldr37Z)E%YJ?mchorHtGqXUxw%W?>oaxHZe znHkjWgxRVl1g_9bzVD{RyWy_W_DG+=M!6Tto;Me#3ocxKyd^Kb_f6^xONRQpi*M(Z z+|=OJ?P6kUnzxXtrFP;x#i~B9l_?3DFIa5D|Gkd9)v<_Q;Ydw@o1SxuNSU1N?7Yj0 zGv%uy=f~U%SiHi(@Y}-Kysyi+nU)!Rl? zU$$9uUwi85Blq1(Or>Kjzcl@ss>39I)on@kRVRJhU5#Jv+!Ow?f;+bT-?5LE7XQ8( zS?+x`l)X#YWy<=(_X?R``aVbGg^QMG7C!6XvAJ`oPmpg`^$cm1kbKs_$x? zqHL@DCB$yM;!HRGJ@Tfv_P)5Nva6+O;+jZC-J^P!ByC@Ge3>njwaab0@dfAAq6@=q z93CjoIUHYTFTd-8@@|o!UpsZ(o$m>j^nbgdvTGKzmHdS0JDWYNvX`k&nwC(Kd7`4k zb5*(S;?>tY&;4{zFUq^V>xP{7vJ08LaZOvFGVS{+v&^riUaZQ|MqrnRp2fb%cZ`2+ zl!SN`s_yLTjMWZrPW}7Rwqix&>Frne&j}ctnMY|~?w{vg)-X};;DtTXe$`*NBvjlR zGWcH9Oey1xRiEp~>Zot!uGc0=jDMb;mWmXSWlI3cph@*#U{%yH@2Qh6TkSh zTa@FMq`J}ElCUj7bC~|RGt7M<2iFnaFS4`u2i-|pz3p&xqU`DF9UahjW9=Cvztm~Gf4 z>Zz9eCFPrQ(|!IN(I}ZzGll@B+ShDfYd37a;j}!lvB>X5qq~2_>2&v9Dfg0$H*~}a z?)tGrXlCx0hzk?vDev{leRq+0U9Vrr(XW1c}L)H=AV7;`NuqZ<$gXo!J-jg$9G$Vt1GAfLcD>#zVNR%i4iAt z55=wvJl=g#v%_oMz5BOIis$@Ry8Ge5nZKJ4H?I({eN*7oI;ZXPj+PVV0u%SIyBB>T z_0?goQ2z%nh3DQdn()y}{$*}S{Ka$fw)0!#VhdkWelR?MG(>?}oo$;`rj~GR7Tu`}L0Oa@${C@MQgCQ{}yG z=63>mFDAFgDeT(I^Ve~ym(6VPMZxE8N?r7{?zXzH_^0pmrVAHeXJ5Tp$M^7Q?yk_u z{>eN4*Eziqcz4TurSdhYtRK;O z_Wri@Qt%5C+37rQN|tC>*KJz={>xuoZ`;OuJNlMSX8U#aO3Z%O7hj*5U7Wpm#aqFh zrxr##PTzd>zyB$9`M*c+TsX=;>5ufYwmOwBSB}aHe%`x%xBYG&+gT3mZt=_$el8a9 zoW0NEA^-G+;s!}yjx0=fkA0Ec|M_zByyNW~bxy2UbR zY#->p^X&EBn^v2goBsW?xme7zs`srfv$3+QY|U|x3&A^C%m1f-$hC5LYi8T6exdo+ z;rnkgnZ(z9SSo+B=-^`Z>Hj0=bzOAsky^NpcjqOMu6tklUPSD*x0Bx6Qq;5W{rXEP zmr_@MV$hy<)11jJWV`nbr!NyOKAqn5E-^x}D*P;K>~Xo;Exfy!@9sUMCZu_3*>t|U z2bdO?_no-s_u;qRJE;p!6Epe5_yT$q07=!%F#9zyQza_3%cT)XRw&Pq|1y)7py zU;Qxpa_+Z>&y76Zdn@{HYxBkK){}O#djG_(-7~#vdtzg|jlh@SsM_NyTxE94Ccd{^ zl)liNKk7xH_KW2gMHjF3(QVkJ`{M1HZ3kY)3R}6gGX1`A!FBgj?GK!JuFEcNjw^|p z?(oR#_6yc(v(K~N?wO}I#ZEJX_sgrbU(U4bt}~mf|3vM=S)VVi0wwcg%sS_lim>XR z7mRJ+SM}cUdSNzqiB;hHJ$@>dEbmuOSM1nzv9rcs7Z=Lk zU9)@}S8b*A@0UeNkrUD_xaX!Set6;DlsEgugc=6sf)|mGE^x2;Cof!LnEij{_WXEG z$+_F>HgTR>Bv!?G@y~^Wi!YsufaS*NE@ zx_4UdOVHcM56x#n@3>rlxL)$^)|X-Wx8D@sx8B%q^i9rx!Tz2JGB+p6D_zU|67t#6 z`IdI@)p?zL3{l4J0)ZFw-<3tgE@g@I^mC7oyi%xq>Aqjp?GpQs-Db{OFZsTlnAI!c zZTD@$FMICTDJ(5=r3=_kF)1y|__;%$Q`zn1&MmWCKd-r`++KbCoK&^{?C>Q|9u`fC z?9W`EqkSp!c*(0kUyIO}axUM5|KD|5ZmQXq+14BLJ9Cjc-`%G*pElU{>B(1bZ&)Dw zf_F>njb}agX7F~qM84U+RalnEu{=12TOiW$1!OJjxTn) ze;1njLUoBj;EER|Qh&GY&At5WK}*?-yEoT1bX=~szI1oee!DLRT#PU8K6omN%QnaT zbbs=>Y|9hJeOV|NLcV(#MOJ?otN{r>%{x?TJz zvsC|=!AsZczF#`>1q>z>^(oZc?3ggwQ#y9m_MFBG^EZmg{A$0j;$C=txq+L$pzSRA zUC*2q7uB9zV9TSua6@GNr0>>ZCAS4%GP^CGdTiYV!;SxQ9iPoU+V_J?Bx#HJ$GK5E z16uTgjFg^V2>xi_zGQuscu9E!^Y*p7I>m0y5jarwg)3%>_U`4vIX_FpciozrcWm;t zdx<4H2h#--x-YIi%F1x#SBMI~jMkOAmi>EQ+kM%PUeH!_HbLrdpY!#%dbb^X7o1X2 z)13JDnMiE;@o%$R;z|}|pAr9ekK1i|@01zx3RS!%(;GJJC_d6;?_V?vVYW$DKQ;*Y4{m*z?kEF*5_t)!gX2OP<-^nE!cc z=;h{hQ{&$_#FxEz`1FONv~6O^Q$`Jc#{5!ceb?>)wx5%ezhv=-ZCAheFf%&$ zfbg|@^M!W_mUK?inb~?S-{}#z<(n+Gb-wljU%u^NDbKrk{8PgAKkKr5+7=cp(YZDG z-ph`Ux$7nGJgu#0u+O^Va{9gU0p=<$hQ;wFiglh`jN(1tOBHSxU2xgXe!QdQcTEUG zi>0Ylb86+6g?o1^@b7Qj!7Hd>wRXiWK5fQ%84JpqcrH%gw(?4$^5dgkCO)}kn|?2@ zT)g`9#m8Za;fA}Oq<66HP0q<@y(ls9ALotzP2Lx~^GfV8i*_7-!QOQ4XF2N$myciH z&t?An#PfShPt(QAzoX7NJ$G8G^hIU*^W`j;PFwU!)NbXrePX5DtpD2HE5UW4Jx}=+ ziM}^SoR@}uVaXV9zd!5X$vpH(J+*=iHbFu9Gd(&ppp0LI3l8eKe;toCg+_Y@hh2Ut` zcbfL+jec<);!#lhGT}(V#0#slJrmddZ2NoNa)xz<`f0L*hzh?iIjkO-~S1x|aFyVUI9~N_i{j+q#Q z%J#)O9%)DSOt*O7mGb+t-(Hl~`EXbD6;~8*cljEbkjt(6=1jSh&tP$(dxq>c(?$Hj zF5A~1P*5sq@X(l->F~l)=BCXBMy7_H8;?7^|GGfE<=;2A2p-P6yHRAc;l66#DdTFCrHKdhUAu5itPMVS+iTJ zQN*J>nAMrEPtwJ9gU0zxL ztj9C;!;SKr1wu(WFHFxT9%DYa_eErHpTkp$y$r5P7)tLvF8X9|w6X2NWVsTSoJH)v z*}pkQv{&hR&A)YhdiEE7h6G(#u76r_~H_CK9$`ZY;Ro}&M zYKP#Yss;C?Z|RkITOO==Hs|>4+|vF3o^RQ>Y>7{OGum&-LqY(iYZNep7F}x+$pgt(r|?;pAhB zm&?`4uG@Ao*zzjZ+Y9S^FMNu95}ou#km;k#{ao<{#(k}1hIw!w(%|&1Fs*hhwdiGgHF>};S zyHTXs&e5#1OzFtt$1mI&3T!18H^0pAo>9fkvA5{&^6wY^e4J9@=5V*+u3^lqA;|NpzTC|%}dfpNw;GrQg4 z6T9Xw^6T93DAl!B=GV%@Tq0gNYF9bhmg(>Ho*uo)>e-%M3wNBHnq|4makagz&9-OH zGOQi9nC+6%S-WFReZ?KYiNY_R{d=Iy@I%*SJ;&YuwyW5F)$VQDHTlJ{#htyiZUF7(SWK1eQ+YqvIC z&FsBqRb@zQ53wO_pVaY7=j%8Z5=-Se{=U>)*UyQ;~(cw=qxgr zxKLc2%PZe@*YaHx+;)DwVdwOTtu?NOLw2d&*`*VrUoPBI*|Ntmg{Rc-EAy8G(doOE zvTL(E%Cxz<-k?JKz;f*r|i)CYu5@DQ-#*++-2X(a5uNqZTf|hYinOGdp@Oi z?o|KP>7Qootv0c}b7rR72}PljPx*XTf6BbLSGvIa0aN&cL#E3%%NQP?zCb)cyIJo= zPUAZFHoX^}wI+(;8JAe^Hf80vUSO5oSIT-fH9#iR=;gmNp8oEgdM}Q}FTGvjxZSu; zeZl6MoEJ^&Cf9DtC^{=p5W{Kp$5@JGb*tVBo3pdjY~`hQEs(BIUw+bSA-C1Tw{w|3 zYQ>)5tlG4_{DZIXYB$h&fcBgJ-aS?etF?Z%L#X80@!jk>C+-BwL^l2VbX9M*(PLxr zl1~lV(-+Jy{J|0{RN{F-)p0s*hW!$i7mxif^heA`tSRtr(vw{yw(HBBJN|x=N7lTz z;kH~YzkPqWcZ=Tl&l?p#G#qQ4xAg0aMi%V_x}GOeuItS zKR^4gddo$-N*O}<_EjHedf{*(z4Vaw+jE`o%H8&J&Ss9yzq%-W`bG1Mi-$#>?9Xm%ZHpEh z?p&9#xNVW!jx6z_Wj+7&Z>$sX7yRyiJlgU4_PH-UvixOJe^IsaV)xt)hnLOHoq9o7 zW1AMEKwRaf=4nYw>wdrXXV`Y<_U?ZNI^*ueb{Q^sEjxed9?2I^J6V;saWSluc$m9b z_<%UCc#$>Vm*1i#y%|3*em?t2M)cvDnZ2ha3ZqRA>~~suYw>aBh3+S-+(zH^jnW_3+^*MNuvdNlN9%^JBWoZrnSWQ}bf`qZw~6UXFWV zr(1R3{73i^N42|)p1*sm*&D|CH@$7{h30qd1qTql&XAi#|KGt%&$Iq7C>UK){;WLeO zCL&gk>Rtr?v{LGR+dW-2cI*D%rU6ekrms%8Xv#*}&?MZd`SN?QTirJq5M;#we2#;gTEox=(fQ)q8PoL%u=j0`abO z5By)A?Ek;=uXW|M4>S7;c6sbux5*@J=kBhQSM2Ri6`S1EI>h1(_yID3ChqpHE|fYKM~j)!$``?*M;RloVGnPuwIqCNj7`8=F@vA9B~A>-nO zPQAm@o3G?Wf0?z{_GVOGuvlza=UA$_6v!p`I4$t;i)!%>5SlhJe|HXU9i?aMq_hlO2{?K<` z|Jr%`EgCP=PTc1GpTNAo>D|Uzjz>y9<#hEf=!#3oea!Yp$VzYV{muRJFO}^%@nV;~ z)$PT`y^$~LKfPOS-uJ|c;fJxZaNuHLImW=iZk-y|l!GvAQ^@gQ4tB7E~?W$&dJj?(lQdG^vtZ_}kssDSh){>xJ1KGMqctw5@5n5L+P8zxE!_m%V&*Wp}J`czD5aPEFSZ z|2}~x$*Hfnm}_qLy$&*t70qd8+41(Gt42v-?IBsKmmv}Mx>lUO%&|_K|0L zR4Vj(e)#P#Pu9L~{_vu3?Z+2V3~mO@bF00?7UsvSP4E{}tV+Fj|KIKG4dK3v@2_I) z*H2sY`O@O+hF1G7glBd0#jBLe3Y?IlzH_IC>f37*>|LsYdAA+XZNIn0|NZLYoyW_{ zUr7Dm!~b4tTYcO&k<=@8_cy!y+5U`J=hXSa%I$ja3x=OEB@Cj+zh^C7)9Bc?^jOAf z@6@GrAz?Rv-b{~S-J7>mr38$d~#*s*%I%Ec8$C}4k|_QfAd&pNj!~+R+Id73ZAUu1v2@Y=D0{T_qLwFlV=M#j8PY^@A}4|6feg#J`$UHj=pyoqVY&ljeL z!+-reudSg}^1X!n^|G}y-1U>bXtl;AB)_~Mopy0&=N_L?k@Yz^ey^EVGUF|kD&9Y8><<*HdZ$^mvu@a|%$+CyJEpfNM?L6zX zS^j8Bq!epCx+m?m2w}G-Vn=>1m-W_19wHHAN!MM zSe1D(`-Jf03a9moF8ORPZt|3vOg#Vl%y~Am=`P_{gf0abyfEp{_qo>P==*=W(CN*S zWW${odvE2e)@%N~@TTjRgmaArvQ^zL9-iALmD0D=q?K!ryv#=P&+($0#dYo<5BE#G z5V~mFi=KZ+0tKqL6VBxZyj|XcNZ^bdUUv3uDUv1SW1|U7ud;T)ykc^$#C*T|IQq znnSun>H+5$u9o7w>Mut3jCFt7eBomKIA=kO(Vpwv7mF8P_ujqJNSwDrU@j-?8Bdez zf_DpirexQ-Ey_?IE@hrnH%1OIO@T?wpsAM@k#m692qajh@z2)Ir?{vyZG zSN%)Kx@Bz>UtXN(&|+D1^U)lw?pBNLudSMX!Bf~%sl;vSqw`lJnoo0h+XBH-FoSra0Cg^(5XSu(QJ@3>NnIfWJ&Utcq0ecAV z;h@#rPURfD3;8`7f>mb7-m`wy{VtwWt##hl7dID}{om#uez7O!i)T2itkSOo%twV@ zoXMDfO!a2so}){&mwfN)b8NaWZBuG(YwnvIulj$wn=`h&SS0i!e!*o{W1-rp9oQ`6W*e`^3kXW=fE+sV5P_A`s=_=VKG z8p41jc3~X?M2Tv)vp_uooauvY5hyCPcvU=x88mm=e+W?m5!a~i-2v**c84j zP`tkRf~L05PVU3^r#qfsd8zch)ZNW1kIC)%w@O?1WvZ>!@nyaj`270ctyT42WX~|q zGcGcbY%T`x?VcRRdc(ugdITx2P=fuh{m|j>>pH*^&XT_PaYZ0tNroPjwxg-gI1fS zDQeEr3%>p5G`Mq*uD`+feah4TeeVy)Gd0AQJ6wNrwsf2M?(clsudi>n-KF&9r)jM3 z!upOESAOX<-(!&QzwR{8<#_t9i}{&8+rKI~oUi4+r{l0*VDEw|$wkZy+ntx}%>5N^ zW99R0@{7`!{Hi4dFJ@g}yy$wrT3>8oa@sMi134u-H|^uT-cWs7!Rq2|_sXLsvi)qf zax-_FzGx$r;JIQ<)`#?cXoW^_$~v z$iBDe4DTZ6hIJOKuB%>d;qc$OjxC{U;eDnrZ5LJ>oa|K!N z`2Y3is{c_hqc%RhnEk`2e;sql6xWheTYvm|}B8^hH5i3Z|-nfNbR`1$K9WoyV(-CpQlE>9&HVg zrw88g9IDpJdOl_X!B>R2-Kr~RdtbqC*kPZITCteh!glX{ciYn$_z z+u`%YYNxooPdTReYqjZg#a~=C!3P)Iw-UZ7W>Q6%!YdzfUw)e|lq0=dO71s}VZ{HT? ztY)4l-SlXwqV$Ywb8L5-%gcVcpq^iI|9x@YLDrW~jb7}$*nQD>er0ry!qKwu7Tw|( z;$0V_+xc$Fd{I9Z;v%)!T&Hne@5S;aYKJ#Ei9w1qGBknm)8^P_fl zhcdHW^zYx?F!x>dqDK8K=gVKLPTjpfeR4wA{rCSiyOg9U$ll}3m_6t80^jeCzu4?y z|L{wd_34`rc2gg%`qJP0oAuqBe{*j+b2*i3*M9#MQQH;7blbFC=Iq}87FVl}8_)h4 z>)jk`Cb3BUcFD9IdM9ts*>qZWm$a&Wklq%RdPAsU-c!2ud0g8_g&oiSHbbG%Nzc zYVd6Dll%3)_29Bp)#Bi&vkC2{(&zM?h34`f%CoS%T(fTR?HO+22l#rMSTAolyXH+O z?_^nq)Tq@*g}&#p<|+I7?lAo`GmR>ok6hb*#+r#y*JvKiC)&vA6&S~AS+NC=i9#X#qGuY1lm~(mI zX15H-I>*x-XTlUrUG#r{EzxyUxv2fXuPpORg??E4WpDG~XM9ZiM5>-%bo?~)MaQn! z7iSvwUNEa)(0Nh%L_j}Ngx2B<*6c4#l+~L~*E_iW=6C(%x_dE)?T4E|w{K2~yPk33H==Znj%v<5Y6ra#b zUOjt`bBSa8Mu`R7S>3_&mt>{Q{ujk`*TbxL9s7&z$zO8he#Hv6FF&)Wn;};9g6RSC zw8h$D3l1AQpMRkDL1OO$;k_TS7yEX;5K86vJI_^q&Xh-*B{%PDZJO)0T;;xba>?bz z=U<&TxyXI};eDKoF5j_UsLlQ$YnS#fof&g7pDi-)_;*_%_Q%!zQhdFZt4}e=a@_rP zFS$gkwtHP%o!EL%=Fu0&7M->%|GoC-hR-_eFW4?>{+{M`HL(7;xw#ORjmE@sTfM51Bh~#` zReRQIw%)5KX?wx?FDhf0wo?k<7c9D6$_ zYs|X&!eMiTXNiW!Y}*|I{DNH9|K^Mg)y?g{ih?hAUjEECcOHY7bhe%f ztGYFLgYtCyqiO897md4oGQZ4{Tva5+_Ttcm?jO0P(mS2&^-Aj}u&wZ&a5B{FmYVZJ zTe;=heZ_8qzY-z?IJ@_Kv5kFI4bqB5~I*E6_wR+Yt>!U#I z0!N!)6JMImy~T54Z@4x8-mfnX=eABb{-Pmnrt|rr7pE4cAG;f(Ce;{}A8_g47m2wq z|9S*52VC2~&rh**<*ap#TsjI|CoZntp)Av>eeLoKsZWZ#nv~YPbV(4H%cbUaJ;3Ef z)#1hI+!r2yn5TYGaOSt=*QeNq+~ki;d#CxJMXvR`x~aix;rCIwD|)_oEsN!~e|1Mn za8Y~4O>AIIJEb$_jS?{CTHG}W^7h4cph7e5xb);~El z`&auLjjm%AQyyx5S+qRDcftE+#V;wd{bmOkSXWZH^E&N>Cu!)f~$MXwYM%Jg* zj*ol-Cf_XySh%n>l5L+9qfW@IzSmzhtSXs!=i#F3kIyRo;$3j~QtqyT zdv8sjOD&)K?>qBn?~11z_WwVyb~c~vyyf{GhZFM4A1PNwT-H3gZd1th2d9_K3(MZ7 z=ze2efZK+x%>lmK&3}C6s^#Amy6)eX?LsdmO`CCRz4?s5TMPRWB};OH=l+`4|E&1J zUq>VBOVgz%XZf}Y66x`^Q|wHdJX#ftS(>nOAuaP z#-5>~ws(F~(UqS!7F}&#`uNGSIt`hseaxv6e?RW4DDys_Asu*CswYwUtV#jroIXF6 z4O)xO`DLDwT=e|x)TI}$M!lPt^&lwUfx}tT>u8D_Gs}x_Uv@bES4|O_?{MVe(HA}U zl;%u4y!VWD64zYYBh}Kf)?bXaK6sRV;AdjV?~Y}A>XPMm9XayrS^l*LJ2&n(k27^C z=a2d#HutvakpROgUvnSlD}_CFr*E9#2s9CnI zUy3|S<~X~Ew!V04V#Pmgr&KrJ(ze{w{+b`pv{$DJ@BJNcukD3&z1RJhR}Pe)vAoD7 z<>6>DpNZjVZt^)kh2O`O8u_+5r?+d}D(2mt|7K_CMp2$qwpGvi4{R^lcgM(}|DMx@ z9=6Q$bB`Tcs?fD=P09{lrv)F{{Cy<$2R=^T`Q*(rhPmy|X{Y*g&AW=%OnPjieX1{k zvt*a>(k6X9xBQ8dE@V5o|C+^loIjxIi-5!7?79~}UxXY?{p{aat-B@j%UYRy^;7i% z?oM0>M04IbTv5KYH%|S+c5Q{j411Fb5^kN{e@teB#NNUe_juF}vgfw0Pgz#<;_=p# z+X`QVi!?ZXxjcLO3!a@boA1Sbh(2RubI5+j{|7&k7r$n-(%O2mPo`9JhO@a??B|1N zKNl%~d%^Wf^SIS1H@91IdutQ3PJBCO_on#EoyvR_>Zr(-*TsCwBhJRGDpy^WGi$Yr28caw+Qp z_U62At#jwrgv=BxGcP|5KO>XO>k@^MCw$ z)@Pfr-^Om~tm@QXdTp<+>b*_$-jZ7MF4lPN%71h7++J9n-=&!s`gXy$6k*1%Pb#mO z?bg3wa$?W!b3Z0uwqDb`>$lFn?5j6DqfaM&*?ee=IhTF+qW^4O@8f?y+w81iW7WpI z;l=AiUVGx)U$k1Sq~S-^#6 z<@3#a)ipQWT4NJB_sp~wySprU+Tp$!*2%ex6}j0S-`M_7^hw0?uuIWD7Jhm0@sQlD zkXX<8vmOh77S3K8tyNLG^!fe$^H<1mKij>ZefnAXx9vf>)7`@vTE%*u&ud=oxKOXK zUB>Fm3-)z>$u|Gr>AhzEG}o2+jYRLSE17Z&?f2C$Ki6X^-BWV@@1)tAUR27?@oLvo zbidU6^F>jq%!2IlUH1?CD!nf8l;1-Ai?rN=+Aj*N=brC=a;V0rPVBx~$@=I=M} zeyz!%Dpu>5@O8?IXTg1a_SYUgP3GV~6(IcjPSkN{?q_@4<-1o_9%+-6pIvq}nb{#@ z^Xhlkw<&8pUAud-sY`#*wZndb=RT~P7JKf$?_~Kp69(2TQ+|G|Q*rpX(kXq~r==#l zUJCD@>v>v3dtni$YRkQSB|YrtUGn4B+|}B$v-tUgCvSbb`95FNtp2yo^3RF85vEaM zv5X!5qE_WA0$EnGy-3Ry)!6>%q{74l3*ReOeZS(oc>N)%;)6CiFTS+~2cNFr_;vf@ zJK`OFr_;~>a~DlJ`C(n-Y=2&>!0R8C)}CQ={d(?(eeDD%b1t`5*X_lE(_Mb~)&F`E z_)LC=y0Cs{66dd_9eW$ z3)r2DPBdCDSuI{4A8qY^o%coI)7ehXzVV)^V%?*8ar>&(Ik!_j-H=;b$`LIZTjkbY z@KR%!(hIGbclxHcqpCT8% zDivuDgsgINkD3w>)0>oJFMW!rMcJsb8+0 zQGdKpdy?Em)1w7elMSp`7#4WAIfZjKoO|(*=da|S>GKyJ)~C+j}RD zGF|9-cVS!XlV9%aM*ZeR=EAxq4c8=WCoI;z>2O>xZNY*SLISiQs5rRUwHM3#hEajYdtU-=`wrxq4R?x}eZ z6Tc`rTx@^K?|N~EacuDK zj;rTG_E&zivs#+bdEx8sAKiJHrH;oLMEzcrmHE$a_rIv>%gB1VXP?ytrBvznG{ zIPdelzwi0N2&i0?*%g|Do;0%!O@{hIYTseiJlaWg-5)ROZ|N+wbM(-f@vL zUGzV1|BhPM3#E^m^t4}Gw_YS~^P$4m|G}h1>t)YZF>Nh7b<$O&!Xez{wen)-$nEXM zN0)rw@bALm8-ljGOm=N#KY89KKH$rOEaER8s#n~^d-R}$eDV2N1wA9V|?bQEZ#;^MhFF0d1y(kL2 zAehE{=F7It5kZTR8NLYcE_x(*cXZ)CW@e@e^36?3Gl85TS#Ke)echFaIe zi#&@LY}u3Y&qw8IzeV=#MY_CJriT=#c@SBkMw_Tf78uChC=-;fQ!b+dTK=X1ukiO;A2Ay1U^IpQp2nkd}8t zi9XYUl}+nBe9Tq278xx4Ih$kAr@ierRVH;YDDc0jo1N+|u3F`LB(r;iW&I>;Q;|oS zr?RH^KJxFna6^#cf&9C756mY$-X;HI`*IF5-ryWIrX;R}D|V^56G0fX7g9-9g{U%AMxoYU2K{eZtf z^=9te;_Wv@zgMsS&AWPb^HwLR#^(ks=@yrbzrOtbDm1kG^MCCs$y2kZZ9aDOvuz2t z%9giEbxo!+Uyr|hAFv@Ud5L7hlbX2~5^KJEpBVE;+BWT<-$lmF=U*^PGL%ehbgg*& z!nNe{3%-|MGH)E(xhj3C&z*FKBXf(U*8?Vwmj}gcbEhwidFAEzUeYOBqDpLW#5R-u=f}QW zZ&0hPPJVHB=R}EVGY%5Z`uDoN>)T8}45ncar%3>fdSA*4WD~c_`+>-TKaaJ8OO&#c|@0fJ1_O~5Vi{5je z4_@X6lYRHLL_aX!GQ&Hs&h1@&bKIA-1%<9Od<{3_CLg}i z_3uu|mmHzLYhur@b%=leJ~tpqKvuWPrePlAjqHp0zh7L7apM;h`n6cBcH)Pnr&n|A zU1Y(wSY6i2XKPEU8~;1zkMFtuPR@s<05@iZ<7SMAVp2w?bdf!mjB%KX}^e-xr!JYF;& zjrC>m3QDwFGUH#+<2(-A=WoL!n1o#aFY^_#UE1{Tg;3Qck0%^+TuU~;xD~bY{P!>S z&Z)BKR>d!|{kZ(~#A>FB1EP7Eo9!$t1HTlNnij6>Svs-xQ>xay>63&Ny7qX;GjEst z>wG7+$<3#UFvzSw)F|E5oFuwSvLe-wv zcjl%hq{uYY^mbJqS~Y$8b@hvvZJy=tIsC#hu2=O@mdiS~OCAvx)Bm2k@HuW_(Z9Cta5pLX44{zUPGPt+M5^`GbB4>|LjiQoQ`*X%!S!nTJi zpG)#ARJWE~xY=-};jdi_oL?_F-V&E2a_!VrlI zD6Zz(wfcm%_#3{t|FY&PwOZvlZ#M8fE?;u^j?O7%xAjq9`14t8e@E%OcrI4wKJ&0e zc>9-F;~1y-P6=CSr~TXtIjje=`O28)25;eLz4GF;1p98eUBCT8a*w%2aqQac!?b{J z`+7Hd%kTmrleSJ!`1woI)XO`|RW#E% zioTRhl`4Of%K5)`&0)Vq?st~&&QZ%a{>CinuiY>8&)u^nmb2#CUs`-(`#WFD2Md}q zFLIY;{$FCVvUf3Wi=VRX6X)dJA7%Gkds)TXUBO^m#wT0l^K;%s?}O(SE>8Msag>|) z>k7>$&RK``F6D-_^u#Ie$t=0Ou)24iuvOKIZ$eeqoa6nLtloX&7-y?!{ye4F3y<54 zRI7Oozj$_|(mK{mTqHu|OUT9YvSkadPnkEN=yG;^{0;pgmbsNL61EBk+3%gcK;_Jf zmkL$9Z$m1Ad*qIEo$I(e?+3e0)aMJ|=WZ@;$7z|SIyf)m!)uZlRVdpyW*LA`kL z%6r_oFK;*PyVz3nLe@z>oV#(gQs##Fic{bEJC=7?eQC&)ka#U;b@@ax>!W)QYb)F1 zKKs2Bys_f)+7GKzKK5K_X8e-XrhIIxpQ-I6Euzq6#Ed19PtepXrU9Ix1iOHT{uOS~yC<;bk_4f`^e2yWSSj>+k&$M=iOpA<_zytn)H zyM6MDJ%`+GMD5Lf@V->~s&Is{)B2U2lD0EXq#p2Hu>aYeb*8z~KC{0vjY=s$m#Fk2 z_E%f6|JwQYdM7ifFJ7+fw<~szZbSVj{d3_2_%$Y1TIbDp=9WF5{|y>NQ&<<#Eq$D4Pe@xoE<=0)mKUQ-uMm)oV7-dA#N;rWI- zj~9juvsGg@-ZiL_jxIG2d3kZ_3rEB1dZD|r-q~7JXWp)Pap^g?T1`oNzTmHaz6-rq zD}&g}=wT;bOS_SN*sT>!xHlzbO1T>D$Y$d$T7_j-PLp%jdTC zEZ5E{2lp)ykBF|%ZoiPfsOtJnwa)F0#?yN~UpVT2>ZZ_3fhr&ORUO->IY(YO=5MF` zB1%sABVWez`0RqL^Jmsw%c?Zd-7BoPCB`bH#LEB26ius=8SRpb)0tl!__dwOpvoj< zuliB#y_<`+CI@~wwZeJXWOIiMud4< zw#mhH#&J|B9Zlz3^Ca!1+C{||K_zK*7mC`GMS`o3^WFV=;r30vL!w{g)%=_GmX*w2 ze#>Es5FD-mwgY}cwtd~ezfCx_KOqq1$HfolmBb> zKwd&ab$+|ur3!tC+WAb!bJ9JQ`mIuY5h2a^u1(oixTLqFevVzkJk4GIm2^Y7-v&>7 z(iC^B>z&BlOVu}=r+b_)zc%gSBJnhfOJ8qId)+iIOVPYz-`3RY3G1xv&oRB8;nLo| zk7bE%NpFg4`<>q}4*JR0rrH`Cn0Lr=?^^Pm>){29yBfRd%9dQd=oy%uJA2E=NVO=T zd)9U(%kQjek^A|=bjR`y4eNMsynJyjO5u}OjLO5RR+)tBoL6lfZn?+D#@}XGl=qHx z<#(H~s`RI?g8Ki2d^!58ZgKTKrfd5&OZW9YF*Utr&bK}N2;&aTFaE5tyPvO|Y-}oG zJHc+>f2G)^G0HqgDmA2gwHQ}?*n9Z!;+_ltUvhnE*j2`FJEKhRdt|~!rrtA|FSa>8 zm)xoN<@%?syg9FR4Cl1Rt>IrR?xDYM^Bdi@_bgLB+~BNQ`QrYMZ<+0R`3!fxmYn8! zw6axFYSYY<`_^e z97=aM&v@ri#cQ`YrxqO85~VTGuBUIu3A25>imxclZ2U4$f9Ipm|0LxcR=u!Ha$V2m zD9+ox;;nW6DT8&wm*Q8?wT|**)!32vC3yP5_s)6?omVh@|M21FMQx8I(XQtu{xapu zwOH?-b?83}$lbn?2UoCbVl(clM-!L`B zLnWOryngB;VZmpLA(0X5I2rDpa@yRy!gc-MIgaxeaB%$H$$fiK`{fHSqf=VsG^$$v zZJoLH=e@9RYaNUo?bTL&`LIv;%RdvHcVdn6?^WtuaQ9zgx-`P}vGb%|&IiiuXY%)d zSGQ`Bc1d>=u=??geQn;(v^&QXw(B!xO51f*MPG9C-nmWg#ZE4tusxBU%yVx9e&}=C z6PuNyDc7k(2`Dz+4(f|y%EpX<%z<% zzL%Gn1WbO>VY`hx|HYHL3{`Vu(yQk8&Er{|{<+)v#jU+B13e^!+ZKfzxOSw zp4GG4_r;x%-Amu`@UPc;;??r|lcd)U$1h0?cYnV4b+n}RkbD2`$M>@9Pb|B-+wE7{ z_rffX{pv5w3OCMnx&PIE;oQA?Uvw^IMmWwl^Ya&}ZM(p}JW)ILz{=7&sp5%!syqD_ zJm>v&`;z*v7p)7NFZ8Ql`rK?+bz_lrc=|2&^ojHD6@RHdbBg)vK8ZyyUrb%(e$OJ7 zd*{_H(hFHG`Axc~TdnbH+y0`A!(XH>?_Q7|vG=lbxs+B^zc@$n1Fe#|Cj`Ez=1t4_ z`bK@>grl)P|135B?tAm=gqGoC(ww#TK$*|YAQ zUiso{K-K#83)#!Acue0PQsrdMTYkmu*WnAV1Ls;qxSC&Y<#S~ZyL_jr?oN=|rwfa7 zQn_}0sb9jO@G#1{{w!zJy#?yG1m|V0{j5WM)=JtYJuF!;VRoHm z!D_Ma#jh)-)y@=}w3_XL=03?-{wYBvW`=ve*m0dMeR(A>*WpE7f)^f#FD&=C`oyrvqv}PI z{}!%S{t?&ib_87%{dGJo!_Dvb%eT!(=K9HTs+S%=A>I7xMc+pKoC}@``#8T?zdY>U z{J3tuUe=`?rM=;M*9D&l)VJlRYRj85r*%G4)4VCk{Bx)NYWEK053!Tp#cQ(TmE5nz zo+XQ45-e?6gsbMU0mmuMia)YS(lyOWy}{+TAb|@uOPYNUg)NB zjuJ*!{hq!>(O;Otb)BXesw&kVbQL>em&D`hZs8c>x9Iw=^2uyUwhsyO;#kie z+jQalMK8n7yy_Q~8|)kS9G6Spm}0Qv<-y!&YIMnTh>v1dGO&A&rW?j^6mVVeZ18Z>kiKK;0+Q0W%K6M zn>5?y)fc^-C@1(H$zUSyMqbYCi<(*3^?bE)WK!;w<9>%5_cpU{@iRdG%61#{WY**;`=b>a0s!CM~nbM1G?zwGSS zn|WfH^XH$Rq#PyeSav`$=zC{TFa zEwWVH)VzbaWsOYCv@c2DA9-X!MDC!ad$m8u`m zSaq%OgHl?_AN41X8aH!yYh1tgXx;)j=F^&s=RaBe^G4(q{b;-6smE2<#Z8>5SYs_6 z829-0Y@wG2OBWut-uc2-zx_qdMz<>7o#I|XZ=4%|D>~SHVxgq;P`t%K40#!kmuaxqhQPs0I{5|Et~M{gS4><#!p> zob{!{*FO{6od3q?*IbMKQ!|PLF8x{h?CD*HDrrZyGwfaWKJ;Bbb8uU_-ud8DQV#2% zGn%|eC~0Z>r?r7&RoF%jj$c_Fp>FYBG5@aSw!C}Eq1`m^=*C6K%u{OCu86m;s^a{+ zaeDmG!x~mUYdoj1bTsez^C_UnuEgU@+#43%>mQc)UTevdnj9bNR&UZcFL6(Lsay3~ z_j*ls1OKN@ySp{`@8!-iyyJdr#k|GZPjn_Q7VG{xY$Nbx^4i8Hh0lLZ4Y+71uU%zV z&M7PVYd({=!9?xGkBb(pUGp!oS?>Cb$f~1HZ@3+qq`J`gqO3t%p`6t^+1T@oTiWkc z*s-nDTz5|Lb1c8?Mr~1B?p>$*KO9K^w8(4yH}~n&D^I@_G+w;_z={3hyy9&glB{VQcVnch z);wP>T^hQ7$5Pw!b81_8zTD#06g#`>&VD9#E(eu`r&;#SW}Diub9T&$XL>6PZe+Qq zh^>y}o4j+@4beoe7q2@+?)H}~-7>@Z$cwTvy`rNhr&(|34dlNY8R2%k;_$77Y>#sF z&Y0(i%$e4t_bNV9Drw@kj~|=gbxrN|Sbat!{M^2|Ax&ZG+lAek_pK;N=|TDVqOQ=0z_VqqazeHlPm|eybhkwj-{BX9 z-FAH(KUw4U_ww&BmlQv5?y#-y<^2Dx9){PQ%hguC-q04}%9!M89(=*^g-M3fo320R zFMdckOurW%^+jfahx~f}h0~kf?VGH!u7JPZ@TQ&XO$9kU^B0eKzYDIuQrvC4E=y@| z%#Q8x>lt|p!uQ>jyK8TkwJ80H&9=L?I~7Gg@4B+Oczc1;*VMAJFMcbW@>%!gsraMR z49?28>>S4)+AaK_TfNcMk$=w|39nyjKcfY=>m9u7$yXsCt+?$>*o{}ej^9&t-@f>j zUvr+s-DP*h{;S`Ylf3tuHQ3@Gie^antVB?zobj&xFtHQ~tp8 z!oMQoaB*9S@o%FGzwMN`W?cx?-4%58hu$sDd%v!4KV+m8(2!K1w{fvSnQG|1Ia`<~ zcE~bSYhHSv-}~w8&D}EBenxFrD|za^b@$W<3d}qUl{;6L8e~1;+Es8@{p+O}A#w@I z+V@*B@2^cW>NL_UBbIRvtBy& zFHm<=uDosKi*tPYFD>nAJ!rG9UpRi^E}_4EI$(LZ=zd@NmP-zeGolVfF^z>VmMH+f^W%1q{K z@+oP3m2P2Qd{v%*!D(g5XM*A0uRZ%(?y6imtofpG zw_1r*O1q_LP3ug%1FjpcFUs!|y=xTtHSKX(HN#(rU1x5#{xrYy$}33E`0aV|ONTFA z&^va6!H)aD(f89UZFysVocX$Ny2B*p1*#`z$~DY+SgKvNZiU(7w{vuM=3M`8xhH)7 zmpC=;u&+6%Y^(oFE`ha#qBTNDyP}TSENraxf421 z`_NSLy|!`MRZY=?p{s>|$sN{W3s#Mg4R;f*YFZk*R!}yTE#$QL*JBITuNPVUc#30u zam!TC9J_No`Mw*OOXBYf7b~$ch`*>iHuJqvuBcV|zS$dnU-aa?^WLKSVory#!}1$S zC3^lw47Xd)seZBhk-GcB&RV5KYn`5-`RabV_1p``J+Uw5a4lZF`mNLR5*fyxbGiIh z$6v@SF77#J2xb_ANjMh=1EC(OAf{_q=kpga=Ak^dpZw|Dqx4yPZgsbt_>ARmFMf{w zY(M2k_c>PIuY%833QY55x%%Q%a>=_elM<1?7yhWa?&j8ctG1l2>dXsg>4TZ9V!yvg z*tV>E9>d2rlMdqq4*2veK`) z9J6zLSMA}B4Z6`RC;4*HMfT~zk3{yGZscB+)4BPD8-LZY->RygXFjz4#oA`pmaFxN zJ$CB_)xcdTi=Kab@r!fea}x=-W!s)99r)1YyVzguZD8P?d0&@JE#D>o-*uxS{~o9O z-xmu?1P`y{e z0W1s3olY0es4@$yGQAg{wnB^d*ToO%ySBNhzrT3-n#=py7mOVIKYKnDt+I3epAx1r zt0O7@!s6b==>~s23%_5!Cs$kZh<}&mG2@Fa{V@x(7afo<+4^YaI=RKiGs3?tR%iQr zO8v`~J=fHZ9le{qtIXiu-G$k|l0>WQL}cX_Uccg^qX|8tvc~Pu1%kI=lZjQ zQ!_WY>FMJIE~ZawpMJ)T{Q4jKsM zU2Nplzqv*Cm*grt24=Hm^O!Ztr_Z}_g?+h$!9AI(oc_L-lilZkTf0du;_6MFzf;(B zZ+iXWsn|Yy;VZe)=f+>cHAO7l^vlkwEa&KzpSyHU&Tl4-;2Ul%QzElZe%rYsz)#^~ zP9ghuR)cv9-v_e=zt*|?F#MtC-sI2f*YFEUZ* zVw~cl^7tilm8JE`r7u#u?z=pYy8EL`pmZneF7d~Ize!l!aE|wsp4*|fcyEc|UWKX$ zwcdfjLGM*`zeu0p@(`}wD-*kSL05cWt6a#7^BOxWSN+@IRDa{^^mP}X#$9Uko2dVS z>u+te?9tU6Kc-%qw|ZYm>4N&lKLmeOFLM2|!j9KaDQnUDEn=G6C3lQ6gqw=9X{W^V9_G>?o*2`Sf9_CshIxklGmlx4EJvEe~+6a z{(1NFIX~TB2|ZpN^H{b@?8zUqe6E_W%TMcPEq~1SGVT6?>;!&U4%r3&m$eA&tqYx6 zz2f)67m^FrWvwz^m@@RKueGXk%8%gGKQHokjiwTJ+WwDEo)?wv&OQF#leu=~lwX1R zFXXZtvv#)E$;5A7{I+ug!-Mk02Gh2jvUzsCs4tTxI7D$l%bh)G_qTA~h!4%Z61+40 zUTY3dY}$`QDm&M4ams#p#avp}r!AKx%W&?c%^QvzJ{S96`A5FIZ*uAT#Sp`9PCu}ZUz=fPyWPeYSMSVeoA!$#@$>7Lj(@*o z81z3p;oIA_@cs(pnHP2!ewlM2TrhU6V~}=e))$@d>0XPsf0!gPLHf(`1L<9Srvq9p z*>0ZQe4FpGzfjQ`DF!pU?{Ts7vc+$&so!3_#xF7Ob9dgarqU&=_0p#??AY=4cv0UJ z;fQau=PJ#t6Z;W(@uI0glb+lsVVl)D zh^>=Z@4i|PvUkZv@tD=z(q)AuFO+O$9iKDIbbG!^?TgFe>M2RTlhvCpJa2np<1+of z(~Fzs*QPC7bbYmS!iB`m7u;7(a=#!QF@L_L)oNY;e7UNlKiUm%-;2LutG;H6=jN@6xfL%&g*Q07Xj*BmxVS3F`in>{`wPyy2e%YoiOeZc06zhLPW8)$m%lJByejSdC1kPlzO7C-b>v?1 z&7HdBzjDKxmg)&gReNtaTy#FJ{J@ZQ|*ZvVUDos?0W(WJ@i1`0T6E@zsJ>&%TIl;C&~zSJqLV z(dcZTg30%+xWCs68bEX<+lx=%Xh?R436(&7`0m#7CF`C|3asq7dpY&iqgCyFn+_VZ*sZJQmp1%z=70EMt}_fB z`;@ulZD*S9dXy_-TY5xpKFbLMMxKkRDhGVsza6ytu=2$P$pqhafBUcr7yLaByq4m4 zz&)XK;zfovmS28Xvd--P*S=~|z{?AN*IWHxDkd$c?!EF)Hk0s6b;Vu0woezU$DO;p zu|NhD?Zx8yx6x56+IUaM`_8|& zlfl;gqiK)qtiOL1T4M@S*+h7m1;R@(t25~(2GkXQ*L;ybrqk`{yTuhxNe6*Sd~`famJlz=ik_3TO!$`!^z-& zKd$O~K(>qj+vhCuj_lL#z5Hra{G#W=%1hHXuYK8hI$=V^Rqd`g|B}E-9rL`Mu3MSB zh&X?RC=C}u@Z#`F~e0#A{t5{w3kIv1~4Ypr?PnaQ*S#0W(J}YTb zZ|)48gK6ivwzxY6r!7xWOzBeJu+Z(^fxSLwUf55GO6`bSJmmm?>auRN<%_vPTozgE zdhBc!s%+J>p!>{K8K?7W^;EtcK9JUTFX~I3j47Y(ojccstV{!0VuSk@UB0NgxF$VK z^U;fS_cm_$bB)35mCaP6!Wa6F%T?0b=e54DeA+BvD?4wV;-a%(1OiNxUxY5aBDwr@ z1cPDq7pqVsGt0@5UkbjgmR0#|wX{=u-<}I6{oT`5yuO((d~M)TVzGe9XKh=VqsGj0 zcC9%b^KSiOh?AGT@x|l(vW4uACA{OCyMnF?mdr5==;%9p{m>)tR}#T5WJ{jB;`tn0 zGso|~Lg$xwFZsRMM_mi-f-kO?vnqb8ZEn1|-tWLmkrJ_$FE2TNzg(U5EYvT*ov&kF zBI}&0cZH0zz|KjD!6W?7>-N3ot zxum@2^#$Lsj==pl{Ls#}|%DR9TAJhN~U6m3*hj7MkFG z|8`^9;#G20?-qx%{S{X0$nJY_vHIy<&j;Qmt^BV30SX^#dqf~>9SVt>-}e(3E0T* zDBdwINN=ZURgQg*b!ySGLyNTKW3#@@Tvk3YIJ4;BOBoIBs{#vlf70rW&oX9F+3u2` z5vaR$vW#z9^sl4|#aa=kUa$Fbp>+Nk-<>zSL~R6DwX#g!>Y5&W@vi$y#yJT>3oe^_ z&uDjT*r+&r_LCzbRy>jwEVW;_-pSqNy}9T`tMhhu=Y@W}Rd!LAS3545FZD%BdfKbk zj_((}x$v2zsyf0k@4{+-Un*u>d#dr$lFdBFZKk@duPM24 z=K$;fs$kcPPZMHemejqkt*Sa7uylryIS1dWS?dg{n@_yfT zb?@n%`S{$0%8qkaT=zb`5bNwAZ)vfA=C6D)f`5`@l_o8Ua!l1){_O{=& zX0&2hc>2&B<-U2!ySASSJ9=@YMF09tfp4a@?3@43l_fT!M9^`47IOHrt|y;XL02hmlg;|Jz6Zj@pqHhboUF|9?ffYLrnh8 z;>(n`{mpwJxc#2zw*I`{xNjGBdTZoOElqndWtD7|kmGvo`sdZcv#awTzP9&fNH8^j zkyBMw|7kNv)zbqDI4AglS`Qwx}+Yz*mCOFV+K=} z^8en?vdTJN=qcVgX|Y$7)Vf!Fed5oBFU~UTZ925f^V!1sv}Z5Y zIQOsFq}sA?<6gcy?&Zz0?R+t+_v3s*)`{3lygzaB^t78E{&hURPt0n4vBI_L?XR*K z#p|x{e^N7Ft1{`vW_HVIDvRs0Lgqkfm$aeGrarZ_{=8yyF*JUAO=J8~(4-aq;=5+gu;-^gMzFbsln&+alS7z5m%PKYo1>_*d(>+^_4uxb@~-xVbaveRk%l`HNp%ujyd zwRQcS=tFJ#@dFPkw>kCU=>)F#f?lG{t^xI8Ypth5v$~n%6 z^W78y<>L1uqW2#+Yges|Y;iVo+TU@$-1qf~t>3LSA6cNCxGj3e7dacx)z@C!?{(c} zz9L`1w)DmBiQ7J1-u>o+xB2RuM%(Yrn)OAmK7HAXdb5o$l=iAtr3WoP%qm`)UBcSx zIpxu{7oj?rrk-efamjstiu0ESmQ{V`<5{~vB1j;6_rzmzpY#`$A9a)dpZ@O8;XWBfnPX6sV$m(04iDE!9lDF>^nCQI~+$?UQdGFq&V9(}q0 zedvV9{kIoS7rUz+=c342tJ^eB*eZ3S+wN21$1lt_EU{8^`M>n7rl-mgiLw`=?dOzV z98i*s{ZqeSyKTv`>|O2i&Q@`k@a{N&;QbuF-)FRpQdZ9pe|giu{#46`Nep*=|9UN3 z`QpCvUy`l{FX^=%EkCMi`5Okq-gze{Qj=-uile{{1w^1Ui38eRTXb` z(U)9Qt-4xr(ejD*+cq3)XZO5yH|^De*(*O^T;Ce>Hp}hBx_i6kt`VrT5$@e-^F=>P zLd#`|(39l1jLT}Xts|Bnc9C0Tzg0PK$4nzX_xwEuL2Ru`@0k)bkLGnhvHc|_ zC~EAQzihF#{N0%i)9)>vJ-zYF!f=6Qd-(o-4{FyFu$8{=a&WKA z1^=l_Ivk~2mhm2HTWqY#W>wn}w?g~v_bqD}UtQC_Aor^%k?qgc>72KGt{Uv``2Dl^ z;OftMzZlzd>Wi0XZGXFk^}zCoFou(zR#OUQp4Z!JEq^^lWR2KH-zHA>_C5(U#l$^d z-?dI+a9yH&!zo|hn(2f|-@Mtg65WB-`#wT@!G+K%jK+WGPWsHGM|ok zwfH*shK-#swC7ruq|QC7zH*fpr;PGnk2i_dbN?vsRxUAgDBq#9TX&a)+b&y=h%Yue znj$+&8A3r-FvaX_1$Cj~P@xuUy4&QQqd(qmF-e*7xp)O4;5GE}DFn zXKsy3%Zp0=N2?`%r9BE`imkloTG_6r^uj@2{nDfhn+u+w;h)mLa(!J3UjT>uoL9Q* zZgm(T)-s7r_;v2FfRJ$ zwHLYS`#N7#7qjh++mWw-pDUr|T>sQRn>^Bb^mwX-+^U+^9lVvyZTRIBuho-~FW+wn z#J0SM-?EVDKz6TE@(W!X`_SweqV?am#C~P)-nRVK9=DRRKfNWWWQgz z-{rpLk%Ik){^`g4)0wO@qk6rS(6=(D z+lsf^4>X1Sw{U3R$K$iLv)xbnb&$*b$tTo*yxVTnfe>tA)#bw|+IZ`sqAzk2j@ zvSi=o_HtR_cc1kNzL{37EDmP*`(AM)d+2kixq;u`*j{jzo~6`xZppVls~7Y2UQXrT z*>e5-g;xpgUoWZ}zj&xD$@+GI-gMRAw|7{Clf%qp78{-E z+pAyHeKFttu*Aly#%v#rgUb z{P}WX;XlNsx^FJLoOFJ{viOi+tKS5F-_>a1!GBi$gW2mPO;4Vj6g0bE*t=PD%7tyk zF8+Te@ujc2%kts^>uTYN*|x%QuiiZPu=0f9-IZ4lY`cDVBEuQxZ`^uiq1@dQ{M_%) zRd_GVslUS1>A1=J#J>&Sl`*S*%w12GPJHfviANGE-7qG1i zPmfxwqItx%LGNotDf2~d#sfBA)PL_*O|h%ut_!%-s~@$#CQ?@O>)WMMSH2LvAN%Nq zLgPx^L+g@Wtop^pWTUMWnwwd|aWQlKnX|h895QCl-*(jazpJgrjg9vG)=tw8ylwup zn*G4}^Xmj00$Q%UQMhxRD?Y3#e8YuS{+y4NN0=U1thwYR&1cJd!Fy?#N5F)Xmls$I znpWvrO}(!Y!#?Bs@$mI4jhBA=|07OhZos##x6?$##qDNGgSfwG zJYcl?e4&uLN@VV#{nzUA-Y%)A4fxU@A?qwYV}iiuTYE%(mnz%NW<3+Tn7g}gEni!2 zE&GWo+cWi(WKa6^hdg5lk@n5I@Y-6x7+TeH?JOZ7{Y z6)RhqKl@$^czpMY*!&lDSH9)T3;w!S_hX7e>DqI7|EE`^{tI+qXXsfj({4U3ce#=m zt6|B?<7Z|H*?uXy9l{bSQ?jvq@uC<0`mc65*MDIxa@~^iJm>qB1^P2)TZvDL@H1SQ zWiz#ZR_%&#G4K7?&06mZ?fucQV0+XSrc+lGF0OAmJeQmC;>qHqlYy&QB5vAE?sW95 zmf#3j6msFt{l7WAie(!gZ!-M*{=-bA<!*02$Ew?l&42FxfBc^8 znq!47i4uEVxy!|l>7G!E(#TAmnwq95^w6lT(Z=bC`GrfKJfUp~`Ak(Td!5Q6R~;3Y zGW9P5DJsK_2!+*I{C%U z)1@@#zm@m6vdKng^5zuR=f!JMgub8H`@5-9LaCEu)uNl1C&tWST>4jhZ&lD-pJN{~ z-_#zhlI7gh`Q*#HP5h!xp_80gA_e)YeysAi;Caw!SNvDAlNJ|>C6Y?kXLWJL8tw{L zb`lBf>$+%bym5i|l%|V+wUiy}CteKXS}(D;gXOM}te4yOEVmYmODG+F{$6r{n9)EH1_=}^m zF6m!*%5~k=7v;qDoHHaew~2l7^G?p>3p+WP(#jJ>0>bAXGG5KrckacreI{JrUS77X z_!e+a#VR~KBU46Xu3V4GD_bSj`Ff^L%{$KRT`Rn$L{98)^IjR#pNnQEE|PzIZIwHh zte{}nuiYMp773>s+Yf+tJr=y=akTo)`}V54 zb?l36qI-2TZYzAr>Ui;en)iy6hZoF#l)WXr@0+=>?YF0MrAl^hUdp`Jx#`=hy>ivR zOmwPzUH(5;2!DBS;r07MOHY?}T{4)zXh&7%l({eZh1aCiaqe1v`>kct)9$_^ixWxdionXYuI$t5iYUi6We~l}au{Z18;hSdlrDpF7m+HH9;!L_P7rdDE z&htn7$F|c4eReH%Q?an>aJe5HzHIFa*S?0Ue72iB&9@vVf6*r7C9q(T$Hh3yTj4Ks z9KENo&nsb-sPA>+WsB9l(CRF%a_zV8l4ZAEw$4*ZnZjtmv~Oxr!iyV=+;%NH9Pq;P zzVzJetZ>f%va>?j?te+RFx&K{O7@c6AQq?VvM)9rzxysqDdm~-_wHAh6ZneWU3S~4 zqUpTw`@An*R%RCiLYX$GJaQ|)G-b`tiG43O?rC>dZ|9S~_;PQ_i&;+dC#1c+x?Ziu_T3UGwv^Yr8K0cacB$XDs#w@-bi7Ir^)vFN?yx-X8~=WDDfX}kKb>2)M~)uRiU zA2X8HsD6l9J@vZS>F9zbbLZum{n~QodV5^ig41#@q8{GxHOaZRi`C+#H^YJKJrg3X zeN?zVA?Lz<(-%?88`O6Gy7g@Bn}3bx%%%5g7^%xj?e(x#|69R2H|k>XcAMTttN4f- zOXh~E^611ndlrZ}d`&UjlGImU#8_8eKQdn;V-#jK7=NBMs^UVc}-I4|?d+0u)`YgV7^)qg*EK10aa(qen$yW!rl1$99Z&$L$>(l=7(QdANO_k z%2%0Qr*zhwe$!z5_vh^IuR=>p^%q~S){@S8%YXmJ>N{<Yxrxh_+s5|O|G%LDCj(;lZ{Fc{i0dPt_u9sCSLw@Z9+)p@ zS|j>JtMAwAucqx$H-8@ZIb}z7-#3ehx$(UjEPEqQRxqqMcu1AQim9Ykyd>vhJ)^?U zt&8tVI^UkyCH2wb*U5dBUnK9xxe5Alt!+%Tue&uzd9KN$M#DvB&-QfuD}H;yOGNB) z_ZMZ?__gLqB6>{+S4iz(+U~Qvtn7-zO^)kN+WBndoZXM{mUimDjdRPoc%N5!Z)k<* zXLR>Td7xd!c*irpPYm z^aaR~jE?Bu>XWFUI6;r0|+C61Le2jA7;uo#X z=ikLN#%fgYwmQ9AyghXZclEnjuP!gB|5fe#&UfnmE1f^%=JYRcS8cmsYymEQ8z0XN zdVYBEzfZUGIrpZNeCmHc%Pm!|D$&ocBu_jme1aX*my-JB-){=dzfz#`r(0Gd@1Kcr z_45Vc&M*ENhO!t6?-eVV^yJIB!i(m7-$ zl@pU&HKTZ+*gD*aDq$_~(T(}M(x%&c=>l#|q0W?N!HibwNhi#$FS{iQdUmV4_;+Q` zo@1L&{ofI_&@xlpXx+!PS0n@OEPeGprN1oElwYfHS#;=bL08>NLXrzFGsrJ=U-ro3 zyf5!8qXog{Q{tHxzeqDR&b`CW!PM$(e&Ueaay8*BJ&|XP_ik7)iE+OyIdT6d&r!9q zr}>8jrk{5_uhsEo=ZlH$`?%Iry!dj|WW|5)C+U0k9AaG9KCy1)>vr=F*{WVX&BF9h zRbRm{rj~+j;(JB!ubg<(CD7PPDXM+%o1cn4Q!iLAThoxFvNu>!;lyrb|ULuv0$qste28rupOA?|CBXBjd_7U(2_M?FM570_;NwBtM1U%!v24= zT-`g?S-epCthGq-%M;aUdII9_?!27wE828<7))=RxtR-T5pFU#fwASGY4VEwR z$h|%BruwckCb!u+&)>VSAmEhc3g!~miGKB01S6jmJFj=UIDe&`Q@^}mRr84{3sYkL zXfMkAQoW_)(?#(!Gq=6yI9t#7@XXAxL%)AnvDnV)oyS>qeVKq2>zt%7Qyf@qpLYI> zzOvhE1$+9j3$Hy(jJxXAeEPXq{VmtR=&toYrcTj1bM0gH+ykCVI##h>aB=Hhbow77 z^NX9htNr3HYuJ2IdG4|5f;y{$(y9kZ*Ei}u`A{mgZMNI;KQpaLf{m?mr>y>N-Wj6(aPm<6`UQ_0Oi? zyFI(sUu4Id3v)%e7H`YUR=-{TXod6sj_$AXE-rpy+I6$p_^DHxNZ_RHZtoWF_P7*r zpe>Yn62so!U5bnL=GrMLTGf2HlN)lO*_hLD{bG&uFIT?2_!D-aIF7sal1|`~{>8%S zlEporvzs$ZQu;Y!-_6!tJpE_-eJ1s#Q!@hYIEN^1Pi^(nWpzCvQ6;>{TXDHw>8754 z1}pr%mh^=uzTA@Z#bS?G)Z{bgwoU$Hc`4!9ncov@t}fEvAYQs=da!ZJzdRp@i+sFn z1`Zv58orLD42!RGFTP%_nR!LibGh5D97QX08?9o~SnKIu;us74WKFDkV)Y8<3hpV;94=~D!!^?FCH4`EH1t=-TB4V1(Uaz?LR3nx9rb>MbVn?KOeIA z(zis3ll+n%*<^_v@vSj= zw&7^xYDdMu&>pv{XN!#$FK$r3UT|B{>hr59J8$m)bV2t;nS4sz&uD$IcXgt=^PBJO zGPC6in6caclkOe6>)Ac$7RDY|d3x^gK^HX#qcZvJ-CzFykz;VG_X%3qKkuCM-X71e zMOunq?k!fC-?y~$*V$>^&%el@y0b#C3*QfaIk#Hg{keu+hlHt%`$1Ob^(;%i zs4m(%r~RcwmG_JDyI!qrYo8~v^0oX7W-ovH<#UYM^DraDQ`m=gj+~-Txvv&wBXtCV8f4TDX&$A6nL}F)L{P*p} z$AH=MzO~Anf9LhQQ7_oo^4}AdX|`3(kJgoFNz`tsx^B6&r|?U~GgX#bZ9Ii5TlTtL zY~Vx^mUb ztJO8X?lfn={_GOI^Rc;oU+j+k`4=@qC12iMR@CHfHuZYJ3$F{D4o<>BT;(y|b^YlZ~FHL?01+8al?e6QkAYcAM%Kg$}KmXds2aA_4b64*QUAO1O)kW!wySCf9#Yd=> z^!VeqRe#75wF`#QdmZD%Z0_wj?0hoS{ORT|W$_ngTP|K6vS`D{TH{^FO~>MwVFL zcYW=0Uz?3NE=rot?kKs-Zz?_ST~|Q6D^m%Bf7OfaY8TErZ|geuGRRz@#Ju6UjeW7p zxkb+5KK(D?NAeTeUsp!ej;h&(l&&+7@n_d-}A#o$rg8&ins) z#ZB7zS>TJ+Z;KZ)ibk*Od*1yyyXV5haDgwY+^>H#eS5paW`DU&eC6_$fvf_z-%elqx1W z>tCMOe|fK?UEe*ck{^2&)~^&Sxpg}!?giTkwu{mcUyeN0u4#$e=k{WK|I+ffseR5R zOLD3*S*0VEoZP1B^d+3(!z zG~HYG=zGG}oA*3^$*jchSQclNjNi7!7W~}1%vS;1@*SmY& zC4tzlS0lY&9A~Vum+RT#K0WcqGs$&&&R@h>-J4Rl9Jil3`{=n{LP_sSkNOwSEbhrq z_#vlup>fW1{)@ux_vSR;6PwQw`+wp?&gg_MDOcTIFx?`;$J`yX?YGiC>dk>lfw!^Lx)#CGm3Iy8LHT?nw)9FaE6J z@Ss08V1C$+#DEXyH`O@m*M=Uu=YHWbXUY2}?)%!;YRaD_S2ngdFK2!CF=^60=Mt_N z%h@lio*Mgk`@{v&RyCXZwM@S}zB=>D)(KtFHFuAk(E8~8?FyeS=P&IOZ5NE!38yUi zzW==A_L~;-jjN8naBY{{pXsja_dooufk98)$GvBI^6WHsc|FnZZB~otw@p1MHe+8y zSlf#~JFAK`F6~b|!?*Z&&$&9?g|8lL6O=4Cvz+hJqUlD`VsT4?lO7etT>AYeAo))> zx7himJlik4448ed>B8ili`?g&%M$r>>rqnPnR~Xo)-JzyMn|;h|C$dE+g^w+6yGez z5#xUEdiw=`t0hqvI=iN(o;?0EarMQXj}OdF?cgma>A3sQ?hg0GpDdr}UoE-v_T!|@ z7gk1VUbVjeU2L0G&%9!;q+j1Bw7lR=;fu38?s0x^oz>e9^S=I&-g@B0%KES+Z+^DC z`0y`S{XF*(fr&ofqvkcQoEPP4^N(-)`*zi6*XXnv8u^32*V8I^r6ehOH5@Bi}lg@QxB`o&~Nfji6i6?FZq zzFsU9dl#v^c(N#q?Jni1Jqx72C4PzC?O2sBv#apM(gn)q@@+3{_^n;?6*ljkH&@T^ z%5{&6&i@%Y+%6bbC#O0cf3VQFtp zGD~c}Fj>x4$|0XM*7cHq{i$sClwBg93O?+bxhHhy#ha+uH+|7;EaqHvG9M!I^Bic-9IPbq3Y^sUW0yD0s$>vqd+i6xIOWIImp3-w-J z$Fx_<;r)tNrS0?h{CjM=RWF@)>}O|BUsA5(s&}ct$kZ-=$9d@|e6jYAFL=w>{?}XY zxhtoLeR52}%)A^E+sz*94q1LNP*YJ5sd(3Z@lH)V$G+=HqJGO=4Ao1HzOboR+xQ|T zW#t{|;wpXBJ^KzAs5Z^H>}In$c!}zZ)N_efwtrg1aq)h$&*5n=4^F)=biw zFWaMIGS(ehmzn=^qrAd=X)BqkH+~DYy>RU~H*sF?g5te8FE)NW?bI#j?Q7d}?skrO z(z`RlRfes8v%4ktUKQH1>AXOR_64giMTZxyuSvJ!i3%uF`11RuN|jsRiH?2F2e%#K zlP%#i_|>#vdh?5^?eh$FT|4D`d@JAFUD+LaE@^QuR=N9g%x%8Vr2Q}>r6jQ5W!Igj z8Lj@_YJ2miPu7zUb=!PFews#7^rx$H{g-}>PTPL>$g_~-O;ZKf&*u4k7d0qYe(!a~ zoZHi0%Oo~@z0AV=pyE`@de$7zZGp;>MeXvk`<$+th)tACy(svfok9D=i>O$SyzSx= zwI-LcAGjZ7RxnjH^Y6K7b*bET_SDz(4$ZZmYHnnup|- z8-#naSqgkaZ)z?)9LnFYKRt#yban;j|Ge5W zUE@z%gfN9!ePyBynger#`D+3lsg z+~L-~XysMj^HMG@7T&rroi!pQeh`$e?UR+#K^X0&d-{1Rf=EVMXsPbG{?6iKu3Au?k&#$B-u7NP26*XQ!n-fTOQA{@4{FA9^KY$ zdQH^qBvSd)q~T)h6*3)O;&7q$1atZ4Pq_I1sBuxj0&9rfDx%zNZYe8PNpcT5fa zT@vY7pV(~{^7j7g3(hl~i*owD`SldOxV`SPqwMGQV;sMH7<6y{Dbc^qefgB&dHKC_ zo${}}Q2QEOT(se6@P)y?QO>mBd^K7 z`MslO-6X>=PcB!TtUDtgt5&yq$fg8LUV6HT!|5Prrv0j0f8Jb~FVqw9!H*AJ-Euw7@?VLPZ%y2i>Hp(PKCig&`_I<5r&LQ5<#{X7ta^SsG z_~mQg>SF@Gv=&M)*pR>Kxao(Y4es0<4W;7M>)q^)M9%%JmOHT3X;&0`^U4zQY>vBo zj6WFOuJ`)##O8~XhyCmqt%a}F`D9&Vcp@|T@0WPq50|+^v%Jp5Fic{w-4qk~eOHsI zQ++Py^+M5V=4i6u!%X(ET zFL=nLYF?oI*C%t6B<5cFebb)lZbyuGcRFYm{mn7omO_UA)_qwmN^|BtxRM*HV)2b5 z=*r3?Gs`(1v3}#|I>OWCl0PNc|A}ZtvqhBVOV+BVEP9qL!iHTe!bdAEyjx?eUH$e= zhHdusfB(d_wbcv1+_itgcOL z|IazW3m;88&v&Z3v##dr8)ZE`jNHaMMABM z|2voI9PNF4c3Zr{n|NhxUrX=$@LuY2s7XnQYFqoivdz4s|Y%fBo3t>5$SU)XC^a%x_Em+V#ZlD(hZSgoEueQbQu`3rOK`=tR- z9?w7e{IB!;s&1xLWc5-$mawZ6Z$7 z1wVN3YPGE6n0tX;(5b$O*>sJ-udOMo*cD5-76x<7b$TJ+UT6&z#0QiK*qCS^K+-{|}rHnk&3(a?3A) zW^4EUw_1N~8w7V9t__>Sw*%l7{Zb%ko3ZB7;ZW_^+NVK`{08p*9Jc4%l}y4JroUhh1l?#L@P}>Hm!xyz za~j!>fB6*d)pW`xg zeYivR%MY=+JKgthaQS}QvA(#h`<-vg^@Lhc;h$^6s#*fY&0lPdUa_IRc;P<&^nL#h zoa~J``z){Txaq^0?Hao#>8y1;?-pFLzU+6_9^)*wSk}Ia&0!ZBW$vC!?R^(#ykq}` zb)WA?s7$+YvV5nPo!)dOF5cQ-TPGd(TY9|2nYI0e=Y>;`4GC4dxFDv{K^Y^e1mf;!=Xoxjh)d4T1? z`c}KBC0Xy>*7eAlch}`?)GxWq6_J$nEK1*H&%H#g%56)gTw<>@uX~Ys?{P@#cc1=! z9Xss=@4bF>SSIJq%91ou=G{`dvlsPm6g?N%yz;M%ZA6pYi}P~5*VmSWNj#XlusiJ{ z^K41x=5t+NCcb0{n!ju7*<}W`lx@~5Osxn*QBbOs* zJNA^6#!GJ4BCtv?|Jm8u@$-t6{MFu#RQH8wi@2jAwgFcPSFGgS^WjRMNz1bz0=s^n`^Bel z>f`h?YzzMI6innR*A6qR>Rz)fWygw(%Q7#ZoYlYM~*Lb+b8o&JpS@~ z%?_2zWlsFAe=HK?O%8a47clPdUjKIA?)aB^`u)m()>Ka}+cJIW>Gno-hBO_H2lw5Y zFN+;BbarGns0E_@vwf~TDA@TLg}S-?FGNHH+yaO zYCWd@`c7H!|NL${)2~-!j<1w{|1@Wx z_Q%Oh+ba^UdYsZ&7RGfVdH%b9N1WC%D>|?&cDLpJih`; zx9l}~arZ5NlXe^z?(=NCvsoB zM8R6S^_M^AuiG|0q4|$?UE4n0X?@9xV#h7c?=F_#Tevn+@BWLIQOwLMe;lnp#^zYk z(r06InlJ47fuE^dvK7mx{kg^0u;b6Q#EQM`{cbOg@T{r&!m~o@%bk7N2P?v&OV)_V zI*3$U)abY0<6M{U`+jcN4!8TUCPDSPbR{3m=<}R;_P5Ty9rsq6yY?TFnWLw9DYLxx z^Dfu=(`V=1nrkE#Y|Oj-=BW!Q+owDJU)^2mQ6IH8K%u()_v~#ePO>eU?{NO=GS(j) zt5YYmbYxf6sL={3y0tau7e7y#o7wl|q{n^^S8?MNE3CHPthG%jUeK%H%iB@)-Z;5TRa5e6t7AG_ zd)$#1ZJrWe?kw27S>)t}`WlOfVFMiIQsPR_G>X)bQixQt}il4K&Z`m)Od*lVH zzjgar(?wBgm&)VsubJ4Q^sVF9sp4MnGptEZ=}X3$(B#XD z%rlJdE_-1cr&iKx^CeLF;RS84<(#|n5(4uJ{w#S>^JC4eMLaJawFLis`a;~!T6u5t0a|sE5{4b^YcpPF4(^`_vgD7_i*lu?IL@1g}%gSvy|McllmJvJ&3Kk z!h7NW_*a6eFIrs}H@rG?b;Y%XulKI5wFx$w!6muyHUFD=MaB25WETd93hk-xe-!NQ z`+UB=i2eoc#mZk_=q~Jcez`p|)1}IO(>sZ^d=uy0`z-lp9?R3~0$!D_T!OpOF5D~l z$0nLDf8pnc=ud~{AL%zMYwgdjO-x-H{p;M+_6ylF?YLJKytr^xYponJNhWO9F3??})(|&yD#ri%=(M^U{ zn+ke@U&f}|*#%UUOkFfPV)w*#v-9sMd{J^+e&L`>)w}8sTGlh&_?4CY-nX;uomC;V zwA%4?_oRS5MO}KAJ@U88an&AtA-gy|MBB+X{mYVNpZ(dm9bb66Rwb&^Rk`{J+W9^_{E%7%g;*kS4&L4C-Dd5dhahw-u`?=*Ttp1$IJe$ z*~<7g-N}EE`OL}TbxWr3{0cFP3OqjdMKJ!M)!be^26YKC$eA zuSn-sI?PLE(Kbj;K3)lU% zzL4z{+Yl#S{6gGx_r4!CFZNXDa?kuT_l4~FV5e067i%jfEhuk!=l?}5M*7;~bUVkg zry{>TD#w;g5Ha_5vsy0R!sqC~FP<~|rb33RPhfWKyw5%=k5?VGo)fp`n#i3U94~sb z4!!-GCp1?hU{ZY3x=jWWdHtrja`#iS)i(M6JTdJ(qsIBK{~JDRPCb&pR>J>#>VC$& z8^QudETk{?Hi*n+FS`3X&*{$NdG7HO+x^=c>|CN467;7&Tot!%+4rJzE210j9hld} z#%Wb9(RCp>%4EW89jot#!e8$7N;@xcKm6HffpAvWyy7MMOA+ul=f9}PVR=MlnH>GT5?&nkd$o%A8^>R4nS)={-eNWWQ z#G{K6k`%(3J;E+OQ~siTYT)8|rsc(L7TzRP|8i_#%yce)?9Qvlh)thnl}*XnQU zcZGk>k+OBEnkFN+{JYreR@=v3E3)NkS(Qr!{~uFvnl2Zs6MZ>!Zhuhj*}5gKE`)3S z<(BGlSzP{g(lz%>`6}WIuiM#4uh3dj9+GHNcX^j_@MUX;t-_5azewTB+1s2}%#=lSwdUb~i){pIcZ z`%f(hKfUbxIr-9#F9#QYzSL;zq|f`S!Qzs2qDJTE#&@#2Vp`_?RE#xqdjCOj{{JQG z&%JQCop5oA@AVcx<@djuO-&rm$-lmEf8|Dr+5@MqnK|UH@chEax0l_>&GXiVW8CZK z*jiL>FnlMpKH~YVBIiF%moF|eHVyf~=ucz?r0=7-C^Y2C@q zd454P^cKh539AcsXZ}==iL#8+Yc*Q%cx$%Y-svY_>{a;kyRvJa#op^0>Pogfm7RKS z?+dxBc>{LNl3P4m_~47xw-3mr^xRwMb)mHSf8OJ~S9W`S9VWkwefmx)H;}F5(K`1R zcSU~9y14E|O=9ce(E7GZXJ<$my1oC*@t2h?`Ae1LmoT*#b7Cu3>ajfEf9GDvn7e>~?*zCE(lwWw-w|(RY689S>&UUA240)|PuZ4*%y}S*7`MCxc;($;DmYGjtpN ze@MSryv=7i-xHat-VJ(lcduOB81p@f>%w|5=iO16U0?2;+%|!k!&af_mh6*7_YW?8 zaw+#lUfsNVJr|3e;u*IEW=;)W!p~VXe^IfZ$0cid9S+-h%bj0a^Y7FC+SC}6e!+i# zys;IlOUfmw65*F$rd&Oo!D({jOVSMm*}Sv&vzMzyRI)FI zHfuTMbemkV?3z;f-fFRa_2E<-|K}Ymba&mjG3Tt-TF0cneo`;R7*2ec9O{<8nsx8@ zX-a#QS|$FzIe0R;oAB|Y+kvryPaC7PZ6g|+F6Z5mHpFb?n8uB@Kp5zfL}ipZ4DGobmpSr4v8i)9jvN7ufl{lO93voC{l%U-p5uNSBNZ+>chVJaK+)#2lu zWk%b$GNq>}|6Nu0gZroXx;Uy9Uz*n0wdXD4y=_=!y|YMd!Qu$1 zyWzJQ&S}cD#AR(=dj2e9*plkQoUKM8oTJpFrHlHsoY5)0OI$@@CoQw6!hrYb}Y*>~4_lb(t`HR0Yzr_8THP7`0#|!q;d{468 zOK+N6^g<@2WwCGQsxANSar`w~Yo`8XmqcoVYqy5WmFo9_y}BI{tfpC| zq|A={>*?teSF0QTGUqv2xT2Wx?&71x1qxrLyn3-w?O?xa#Z-~Gefw@}-#x&f&+_G* ze*JC7sp|}6J{(>0{rCTL#j3fx6+<2T-))k8C8xQ|Oeg&#@5>i$-p3^P_SVRK`MmS_ zzN9blUR6^s%DcTd)N;A7CvM4$JJtRdxZCxZtkf=c<|$gmP8Yj7;Y{2-XZ<^VX2K=^ z6x@y*z3@GDif``Dom*BOE0=mVt1}{H^@mw`cX_9C)OPT_-oo)hG+dSMwyl%C*!3?B zG2#C<#!0F-egEw2^t{iZr*5K|zU5Mh`SZD*^#9uYjr$emHhrOi$)p2|uD<;FMXLW$ zio>ssOxN^&bY897U2^?O>zD88_uc1A-goYW?E0(mY_iiA@&DjfFI=ut(sR-Ke)a!O ziN)UqtI{tn=@#nPDR@JoYKB92Ph8cPqL|CYe*^Ol&dTAk-G9M3hLc^+rTvxgh5k$1 zr7Yb3-&s^+^lI|z>`y{pwk??ZrKFp|pgUQ0p*Y{(%`YT(F6n%CQoPO2WP;4MkFjQQ zmkWC?HqAG3n{KvhR=As!pjDtjRrcnPXBVSia(BjDX}kL4vE>)zMUo4}cRg&KXuq}2 zvslfuYmU}!n==~T=exgbG_O0hhvUNAsXvdpWX0|h_?7yLLAm7}r>*jrzfR>7?2bNU zoBzFzKZ<|*0{&y2&qV(&{=2oy|D3VNTcOWiJWKRuEk4zGDRgO}OF7@))t_IO2bEYl zy_Z}dy-<6jL~UxzKhH_`IPXT3h}^cSBLzcy8>UQ{cgN?+EteIcD4UzM}Dru4_~K!{#_hv z+i;Qfbian=-CDJPUAEhQl!e3vWV#-!2sJ-(WtpdJ%F~OAC)%sF77AB7)ZOTImy-?M z&MtZ3^Y%A)`*PoexXHPOwEt(eee-wAO2u~BHTsQmuaX_)1)}0i7cO2THl-_jmzh=h zikQII@7$A@U+`R5_QJWv&))OG<&#U}8t!G>HQu%D#h#lId%G^g7pSgluSmOS{iOZZ zqRtCv&sP*3coETZZ@~AS@JH3jb+;nA=b*md=pNRd{W_wI@W%PsIE$VK&)4w$Kr9alLy0NyV z>ipdWLLZen=3H1mZ&vWX`T5gq1Iz^2qxi3{dmE5g{n}c(CEKy3jTyo*t^ZJ$1DPNwy&wo4Zu-?++U2gBiYS%V( zv^ScBo|l{T*4!&O+BC9k!PO0$J&hi{y|swDrn}>_=a*ZH!@U<8Gwf~&;7YdEOP_l| z%Sm7IuD>c*d0W+y49msZ4Lr{r9JE)s55a&hxU> z`k1W=SW<566tDl);e}IHY4-77clQS7M}9fB=yZwY{4GC{BKY<;u3zGF=tthqN8ZQm zH-D&=)AwCnIj=6qMsjZVZ9lR4AP4uQoYNLd@$Fq)P*pD*EzVihzwcGgDV}Y687-aG zpOLd$8Y|oR>(m*A8?4(-NW}X7ofWV=`mxy6|LF=}q&6Gwx-uuInj!s{-`wbu1s~3u zvD{519W;jo3jZ%Tie88lz{^4uM>v-cjZvyX93JazBQuhxb8h2AqdlqGG8b6+{l z>%EKj492<5)1qar7hJalx69Qn)CmiC@tgBk*4A?`L}u~7d>q}Z$7##| z!auA~ov9+4|GH%Dfn`_3pIQHTIX5QWqGW5=qYGy)aBkMG3-kSR*tW-yefJTYzvlw~ zyq{>~!TevS)x9j^!sFS|OWzqu-{+dDFT!Q=zvDuy`}so)0u{3pZU!h7U)&zO^u_E@ zcE_T~U)gc~?_c#7{}tSASatS==-wpr-mKh;lMCw=)`|Q|46ZNk;cH)K9dq@UqTrPx z30EumOLwi0a~@c3zCQcziyVKlJiu?!Ww)&QVcj^MCE! z_W88e>qW!ShzQ0GDMAGvvG0J@YRJHK7MC{cIyj;JN zexHl3vUcH3X_=wRomtpsY7q1!2CU-5}@OL>c*Q7I(KHYh({MqzI z?*jK1_B*b6+panCGWM)f;g_BF75?pi_x6R1e8%7VD$^&`8{MnXUw7`=FNUE1iGR0C z%)R`Z|K8nWYmbXL-FbiXZ9|+vmE1ypgBPmysaBI^YrIul>vK2`d3^EcKX1B_TfBC5 zuz9%a2@jjzw0);ziBx4rhw?>m7;4o&MS-u#?; z;n;+z|5he8Y4Ji|vS&Kqzs2-TUT$OL+vrU{^=3M)V2QnWWUkBl$~zhP2y#zl-b%=U2=rU-tN0?9BcPtJ#0OnD~!jr>Mmv-<0(Y>vwi;`ElE( z|NE}jJ?*F2kNnqGiSoO#`06KKjlK1j7q&>}$wC%Mm7GW{PFE(#o)qU~) z?UQ`+YA<+K`z_XPwv(G`R8?qcwg2*x#r#E)XKsh`e_=iwY*zd0Md6Nxm+~Wm*M7au z?;_N_o6jM4@vrHBn7vTEkb){eO3NR@u^5>-*11Eht~; zuNhjx7TXr)l=AJsibACyyDh)))h#~WFY_+V`f9EG+s_}kyIRxFvoBCv&+$b0i@xWQ z*?fTx^HpEmkxnXEc~xR>yI%h5_Oh$%9Mc=zXZ-lLN{pS)>Id(>t!1m`TYbqhYrMy> zzds>@E4EJZqJQSGmltMp%>C;f=dmqb<;DFg_BNOJXA3vqV=t*?vD}%Y?0WrtSa;zW z#uv9=@$=Nqc$Xyp`Fpd-lj#g|S-sr$KfBQ0U}y1#dDr}Par$mA-oNNx{_A7yE7us4 z`!+=k=^APb^D6dV^k2L}@0ZX2hkwmqO0EiUfAMrW>)!M)r8$CD>{B^Dh8q6b?6vy4 zo6~cV1$l0_=h=T_SoG)ro4aST?z+jeKb)^*e!N@Ft6ZT=`;_Tvh0Wm8U2AeY#?Eq|Wl-SGU?3nqFV7-*)ir?fbn& zyD9GZ6=|vJ-xJjL&f4^9=`&WV2Rn+&zuxiOzIwyn`Sr}xWNvi7JCHj6%R|npTix#* zZ_P+{nr?F=XxnkE&y`IlS6ooN&oxVR%_+teh6}y#t=)Xn_Q)EoV{LjnpTE|WzRMbs zdb~|omrHxBnxB=3q}-xRpe``a})?q9#&&i*Pr zr+xR685dNmb3cc1|C%fP*me4xs(DLz8@`@hyne|tDE%x9&X*VM$4^5@5S%uiw!<@E%4qg z_&n9bx=8MVPVs@R1>W^Xx);AiQ%L*Dh}DxNuNf!XCHuSe3fxsz?Jjs>msD-0P!+Uz_IJkG zLy0^8^V;%u$X$K$V9zG*KhFHqxnheWwu#?Wc6s$+4t`0`7z_q*NcyZ|NoFI-+6rU zyt#KRdiMEzd7`ATuDaS^+$}x(-Q@|B<9^Tlsco<{t&jiLg}Zwv-Dk8pv#DbV`#QS= zU!3QasLsl~l%M!XiFwKb^UwT;7FwrHdbUJ9+U&FGF2^f>3yxjNJ)&}QKZEV*hD-Il zJ7Tk*c5IsR<;0@xHYd6$5+es*`S!opCN~fvvxtHGQkaWS77Acz$~CI{l`12Njo>AOCjG?TB6^exvD~ z&B>?*ZuJ}~=@(5s!xvp3e^_T+54LXRud3 zwTS0_hn9><(c&(ypewN=TW^W1;L+~Ce&Wzl!}S(7F}^zyg4SKsyF1ow3Auwn_fK1^IDt36dRxA6?sABSn^%pR=6v@aY8e^G9=U zF@#P?zxh3GQh8l*+POc9^6RI?pJ$ADpBnd8Vxj(`RMGWzQ-7uNz6x5+kmfACMZfUp z{HH&DJzl)|ns@G(_m(BBM_+CJTU)hR@;;NIV(y~P@0%`^a?buxZ$0s`>-*ogre6tl z@V)(`=SXztxuC+SWSH_QLe)T}=%aTc27_lKb`gpwp@( zx39u?&$y>Am=V`&r1B!(@#o6^bqU#<|1K0Vd$d%lHcb2C>=Qd{H;M}>cyDAqk(m2= zn^FnqhOVifqNIMy{1ac6sP7-0n;-vuHN*XN{jJxE3nSg`P4K;{_v%G7+o~fCyRF5y zTOH?82-~u6tJ~Ch%q%uv_9m#FU2L_YwcpPD#T$MW`hFR~xqdTzAe@`}~OmVMP(v%kCVUK7Zsyi6x|_WL_a!^Mp+ zxPMhS<`$!OK56R7bIP;#DqC$|@7P;s)UcwpPWr=h!wZ`YsunlO{dn=P;iB=3n<`&| z_UvrTKkK+`^TQ=Gx5>ut&ANDS(fyN*CDMOmtiP^q0M4yc0iHY+C8V7gH`6 z3q%xLc4$9fp~qz7?N+Mc6m@Oqg>#+fBpx;{nLO**@-Ms2n%}s%hlPKm7sHEIL94Yc z-8qSBD=%`p=?lzNu4-)=fKiVzi!WQQ!x5sp~fb4_H{R@|6t4;NY zcXIfXxsfa4pU|Sq^L=jUDOCh@+8^7`a&I5wl(a8*Tz2g@h&8O5`s$A9`<2V==5{H& zwLb`Y{$f72F}HoIhx=Ux4^z=S&hhGBmiN@}C|+}OS5=^00-q%ayp9p zz@l|LD+5Cv6_~ZUPBDC49RA=!yRwec^Z6VX(-hYz?>!^4%Ru1shnI=cd!P6l|9WGt zy43x+kX+om-@hlXauvP4I#A&E>Wwl}Dw%85E=Gq;5mNemoaIYCXZ}IgM`s>SkN(^9 zQsJY|OoRQbcCJ^>@rhjFHvD;j9&P)JL0;Y zS zF1Rh9tgwE?{|nv|UC(pazI9)3XZS0tWnHFy66h<%A*hy@Zc(Gh zC=qMCOXSxdpJf^wrUjjxa{Bh77{BPq)hr9|H>c#JdvhAyW!sUu#;D_TrH92OZPgdN z@fV~{oW7C%w(E)-90zAjpQ z#eKcPGds!k`|KwQpS9}V$-lMhmYdb0=OVc$zg*xJ;1HNTyFo}n%gtPBRR!CR%=;bl+I!I%rzy_$8etI^7IxRx@;&@v)~RqMbdG^lSgKyWnI_{zj=M5D zrmAy9v5LE#G}@Id_*hoY^wH;4{QrL!GT$u_w9@kQnOk+yJF)za#c?5*`@1x+I!;Uy zzTSQ=en!C0{nI1Dde-GCsV{4<)8e!}^jSXa^^(w*$ddaDk^^gx{9NIz-=s9JCGOe_ zm%kq24kjf6|8v8F(hF-Rg?d=e6Z#-OZCZI z@hhF5Zx>|up3!f}ZGE@biC-W{GUz z*?3CJfxmm5gson3?AG63guC-rtF(OgZa-J_RN-Jmsi?)%zYBIrTt0HS-g=RhN^<+U zmVUX7GVyuYWyi$!I`?NUm~&yG+qXsR$uE9=yfA%b@5hUO3#)E9v`gLfDfjlUG(j;Fb6>ai=kfD*+I^ST zJRW#N^a88=d5O1=uV^fuzi%h&1S3CF(=RP;#(P#xyZR#DF?CZTbG2J{9?RDLds+96 zDozVK_#)%3%Xt|q|AX5vo=jmd6d|mhDxq*Z2 zm;9L0#li14E^9YvsW@b@x7APU^Mr^mrtNWQ!4A*8bIK-m1!P`QdXznB+sQB^m7XKs z+9?SaW9Mc2IQQ6g>5CR!NDY*Z?eZ(>ebMQ<+--+`h0*o2iWdgl?kD^w3$@wx^JVHX zOtoNsTmR@G|6;LSYz;ad(J!{m*GLe3qxJ5I;%uhdOLlopZ0tFs7^Jr%+tT-o`YT^S z(S^@-OM2L6UQu>eZuk_pBK6S1;s+v%T>_uK>r`EGaF_4+bzx$YQux;e#^$nN$9^$P z4zAC9u%@+d@yqXuTkrgR5r6&Hymmdtwh!&Hl8es;TylOdyU8)@;@hp-isA{q7df-F zUPxUGG%|3%Q}m+MnLqo*oarGxVO^x#NhBp$Yn78Ka!^=5}y(Br(_TUK3MzI{`9 zyWji=UCUg|t>&kEX`RTNwW?We<`;hER3zokVPy=`zR~oRo54?|`d5it{LbEwPZ!=k$P|84 zM(p0bsHmda8^`w^oz-h}+wn@Af4E)Mx%Pg++QTLveamU@1*p*fiY^!=+*GF3R z#}htVRS`G-JHMtmeTjW5UfcUlY-jg`3;8c{L|$^~{HmO)U={R4_Qk?~mW!Bm^1dya zp8R6FSxpJ&!)sb57vjwx?>Z>G`s=O#r#s}nEv)~$D5@)_#?R&X`rvbX??ld@xO?0o zocBx7jf;EOFH3K_cyZH=MVHk?S|#K;j`GQFdCTHwTk`HxtPoGh?yV^eS@YD=w{-n1 zZI-+3@%{KqkwX)TJ{m0*vAyc6^y|;5UHd=BcHWgyD17m3SAyWxJ1H*pOAcQy7T9vJ z_|9UP{>8U9#v5jTIOo5o`>T5TC)K=rmQN)9dP^=WzHhHv^1|u-@r&IOjJIC=>)-UE zv-gYIuiqh?w3k-xYTS3|#ogLp3T|I|E|p9BBD z6YlyZhR_Hb(^am0aJq?A@VkT~S4c8q=WJ zJ?~84TckV=v%Na$LcKM6So7k|m(7)Jy1xinzIt@Ouf+3(#5B+I@j0qFR^bN^t?hN; zi(>uqqH)$+hyS$=mG3UA9}7}D;EK%Zy6|;BUNF0?Pt)Diq@KS%*KOk2d#mfdDA-Ijt$KE$ z+nmwvbLYjyeqW?H-Uu4GU4MDy;>OJ@vc60Xc;4MoccXFN##kQph3{XsNA}K(__9at zwaEHdgI_`sU*;}OmbCa1v*p6YK&4*^f6e?tO75q#diKU0TAsi3{#0(cs*OI&=NgNM zmn1vJsIfZ#o8(bn_TS`n$CpRP>z3Tsmi)5dw{a2c3$|U1{*rSQkC+HsvdT;sWtp^@ zO}xNpW9OWvscY9(y-vQ6xR3Rb%eG%7o;R&mOzHy?we*euOST;(zPjpp<7 zFFx-5x^`{;i*u5ed9%Ho{C~Yt)t6$MQ*0#Qd_-^J>Y)2CCtdPsleNycWNIn=Re7rI zi<)mW2OC}+mbkundF3Kw)|VQYza95>7&ys)%DAWOek%B+end&kJf$y5pN(`lYBz>4 zD>j!O)>)@{?}PA88>eGDld@`e%g;3y3cu1W`=Zr#`JCXb7foI6o=;nMMDp~Oef!-{ z=UUY+_ZN9t!~0b0Sj1+>#EC**wp<84b=%u5|BU?RY4>~u5;kmI+NOWY`)mGD@g)Zf zk3V0t!(X3KkDqJLw}aeL^OV0lc%;7l(c)t3luy4(V&dbDdR({KTj(09Qnw^~<~P%e zu2ym{_cx@}u=n}=@Q;2OJ!LvyEMs;2x5dlF{<{4CT&+v z{KvPbeO_zuE#A|1ZLAmP-`J_6_~1kNa4S~onp!&4r(HtBLyIHor`KR@SvcFC4g zX-`!BI}^8j{B|&HiRj}OT>Dn3n!SJgV&;cQmNTyug_>7oZJZSyed(Fj!r44+D_%eJ zc)Fl3z^^)w_-VWS#a;jk7-$ zcQxnn;?I|OH-38Ve5RPgc3F1*BJc3HyOWLerAzJ{cIHdp`=U^Pm-o5#9DgmX?py3q z`7+~zbG_UG=2%y|s-rJXF7fD(dLiSMaPhqa<5eg9Wvy|JUN2fLHdJQwot<>Sa?#@F zrSFv^GLsjZryRE0Gx^h5WvwTB8Tz~JE-eZ*bKAd>S>k48x8o)A)ysswKR7xeB*4Za-=F?sjjwnyXYOvhy-aVFVxzxQM7=OHzsWLn zS5^A@D8;J7i$D0~z2E$Kd7}G+YLU4%6Lv{mykI%k|No^`k9VG*)((SagnJ)frfLY1wGjkqIc;t-ij^)5=DWC)b=N21ore~MHv~vds<~>jUZJG=>*SRS zR~P6cFKI)quUMn?L)|`khrPhlTqhG9=x_AGgt&0v929?}j`1<6F zO|5(`2lx1X$#%JKX)5z*PzPoW-Y_`r?#s1e?%yz?zI=+e4FL@?)zF<$=QCyl?=FPZJzMv>%(X`!Dz1f0Y z&o8pMxVBQTy2NEYN36+(!zW+l-LLLC=gxj-v#6kW0$-rC?YHotE^)&?yA9^nFYa$? zeam$~^$P34=5U9yj=c5%*bi0JHn13NcyY=9$-To}CbMnSsu(8}sFhv5w?1~+v!wGI zOoSw#MDuT)mHTovzXH!$_sdZVoB0%|piOv%ezar-I>2JUA`GC9o^ERCjn$R@=qNU{a6I1uPI&Qnbx^|P|uSJ2gH#WtosB72?&RxDRM}1{zMzIZp%ln_& zVw+d|O!;Wzk$2;|ywbA?7GF-(KRa`KC0oT;#WoKW#rrj_7xUj8jFGn8^kU^Yv*qG1 z9o?7ozZaYEJ@-r4uIzVnRqy8Ac$=bg>HB0ctM_~lzfad(94-{~v|Z&#@%M{UVb;d7je7SB5?<}icV=4!LvV&S#r*V1?#CgdKR{K;}n%JMnvKVDpE zZ`rxu`EzSI!#|$eM|VmxG@KPYwCaRAlYzKINpQoI+6IdglCs+abC(1=wJUO&dVg#R z+&QHzsoQUDQ`2#`<94;@MVfutRz00)TxeZ#PK=q~Hoe5M?VW99SSQOx&3yt_^Z$w7 zSgGv!;+{X^IwgQ>lqj~JLEAPPII%D;o^@o7w_&4*8P%x_fe|F#>(@z%hw*`yQ_BX zs?y!9x%?`3?o>Zn$S=&}>s0S=u&eO9_>(Rl;a}g%*nidjUFUi{-}$ocdZDUmZ$2w+ znil&a{?Oi;#kpUu1~hFyrn6mm>-P2+4;C^%5IlDA`aJh9yBoulzx=Eec*3wjO(3?$ zRs3}JWvB3#a|*wnuJbrvY4V`ba?4BaCySo-$Yj+^EPKXadB-!;#WFSdi^@P zKDF;DsQWQ(Z?_k#&K{-*>jTAuf|j%&Wve(_U2v!}iS_f9fC6v32foZtYv*NtUiyuv zYAer;LW58zjpDU8tvC7f9hUudwfd&PdZv~RyM|DnM_b=M(`*IVIDK1u70 zSxOC^^rtOvna6CBZzGYvZ)*0GuPG(-nz%Tluf6!lA?vt8u5oU1#d5#-C80{JC5z%S zIrtkd*sr%gm-VG_epag$^UQx|7TAWa*?v)U&!=DR4)XCV(NYPgzrXm>Y(I6sA!qD* zh9k)@EE)E&II>r1r87JI-@AahR{G-UnF)zv7f;*NuK2p&{r#2;qFWDNeRloR#8X}y z?Nq-Au`KScUt>A(;7eget%+%hH6=|C`JeSFMz_p0GyQ+xT*TgN-}}j3PiD_MW&Pvn zCPyp1(#*KM#{-vs)=B+RsekF|9dX~xdozB_xF7eW%wfHjccjzT0{7d~Y@T^dJ{uG` z#qL+E!1S1OS*wrp9A2~wh8S@eEN;F1wxzl*^T#dUzW%aVy!WGT#N6Gmc#+3@=`YdD z`!u9h-&D^Fe_Cdx`(=UKed)}|67z)r{88Og0*g;v>8`VA^SiyYjQ{0ib_8-cT`WSC(C;PG&f(zM&sxJDo?Ax@z zpU=0X@ZD6`ja-dtC9!&UGp%*jAO0H3xFz~Uj%s~CRQK=2|6?V;9P~-L7+ueMrQ~9< z#fEb84&mFuvrh2*47(b&`-O!Vhg1LB1uwhfb_XonS7*B<^RoH2=Gi8@yB&G08Gl`1 z6*06iUX;(SvE*vZ3U&eG)-_lEQ*XX7x8e zS=+QPf0>$&O_?KO7kYkeTF<{N|9v@Yk8OFDpXXe(&{QdIMpS;p)w3&p<*=t3Io03W zadD6O;^*25J~2JXvD&e&jr&fo_Gg&6*(++5ef!a+CW}AkttnmHUUu-sRJY(YlCg{~ zwl8*d+OddU+PLa`nf9I>%BBRA|e)bDT>AlMH z9d4xVuyvAObMTUVvGYvFf{qtAtA0JK7hBY^dFRvfrWfzKd}&pD@H_mHT9us>|JS=m zXUx8;+^zSAyWy+JW3jZW?*7s*uAg3i?N4`h)5XJDtD5h{7i8XeHv4Av#ba46?w9@_ zwm<(WbMa=e*WI;oFYitjuuCmpJojBn)APjjj29DszvO;@+bb|uvxciC@txK^FV0$y zgG>zue(|Rd8%Zp@&ae1PmhF3s*^BM-_g#6K@G?yJ?=z?WH3F~m*B(8);EAip`{(y{ zO1AEM;=ZdWvn$E&_j zC{_QRP}M=(oAbNf`Tuy%tN9z~`XYP7wo}zF%@u!`EzXZ^nd+SGwDNh6YhAXo6lK&^2=xpIvojQ}ow^3&fpN{H0>|Pmkz$@kgvyOh3NXYIS$D z`S0&H&lDX@{NI>o7IkOI%jjpimzT=FNLkcAlmCUv*W)$nTlrs@Tw#mP(!bzYKke5U z?vhnKxnX-9mV5nZ_d8`@64aUgBBUnl<;?gOAv{+$roRZ;S8H$im3wZN^cU{+7goNQ z-_sgzJh$uK{K5y~R-*3m#}r>Y;k0V>TOj;`1(RyVAr4h6zTmgui+`qu+d6z4zc1YU z^ucZs#TVA=&y?k^doO1`MTy~V;Pqb#_x^>r?cR0yvC{R6?hAHno3nY&KKS~`v5=#B zjBQrCo_u+yH}zHXhPP7>^~vpInZUMd_g)jn>5X>Vo5D(@Uv97aAaTJ_{MYYOXD_td zUVm|HrybK>->Hpyw=U`~Sz4iZ=iM&RjOKE_>m?U@52n31r1@*brTR@P54*76-RP}% zN%hxNxeJcl*S}mhJ@)hWCYHOQuNNE_TjJB&FChCRNBE0Xf6V-Q{Bp7-PsR3{%t(FE zX!PRhw{)u)UmNG%TP?fhWlWps!L;l5x^oxtJC;wetWS}S+E`tZ_UqpE1z{^i?{4hg zcD{>tHk|uX^Y>)MMW%q6FSPqtJf45{xX0{dzLMor z=U!aDYw>*bwJ)}r)@q&0W8c5e?d)3yKjyus*1YIFo%;Q|P1o%UTb(mhf4vWIDSx4~ z=6R6Dh3ok-`B&<{y?bunr@Wi_$M#5JG4W+L`MxFS9SSuzm?RVK&=3)!k)>pn<9BlY zhL|4@w3}_$^juvPt(fp^!J4KyH3Dm%*ncSGce}yDtj(TW`c-j5RG`qW1edo<6Sl1h z3cd2aH1uo8R~}V|S}Ey;>+d#h zb-qIC z)+NGcluLM37^{j)J~!O&P~5e0I_KZ7ynS|R>^^xhOiGW+&aZquc?YA$&(wb|Mz3BR z`NeudZZAvNvg!M5^A*w-IkUa^T4(97-}{9J$KHszFV4#s7cJIri~Q_yl1t;!qw4=1 zPTb}UOB7$d>r>4%Q!F`gkK@VKlq(tKVnS^{L_Ch?`R!sW<=kX^dXeBxiP*<`7aM(j z(R8tOVc2Gs7nMEpZicMUUtYK>uR*kL)$(UsWM2G9neXiVBEmQJcBzV}!mWUXu(lnEL1TTbHmS)pk|g-{nHKeO+>#FLw7~IX$aiC$rLS?PWNi>;3l& zYn8OxUM>dr&{|1>!>Y%7{ooyEl`tI)|Hdwo-x+^*LR*~=Fz+ATS}En=#>hr*$@ zcNc0s1DtIhSRZBkv~HpA*58&5Z&MxrulLN#3R+wL^8{0bap}t>y$oxZ;C?ReL0<@UeFC)y|M=t0g&Ki=(iNn-j^9&zi+38Uc3FMCrx-(O|1 zEWUE`W38KJuFuvRr+j3&rts}OqHt-e`*AjfM;5+!%P#yCI@#$Jofw#0l3?`kSvX_% zi_e=Y-10+j-*|aJP-e=7X3qywmG=L`iYL9eFu#uXf%w;=1=&72?y*r{5|z%bV4o-Y zlF@2mP9pD0H=)I+Enf(@rRF}mu-I$w678qDpEu@CxfG-H#h>BhfhihJCA>e34ea=O zB^P|2IYmSLsL>+zm)fd(;}ZlpZJ>=X`n5a8d5= zg`@WDEu9zmUmQ7jqk3mvg}Ph5PSxJTTT30%d8VyqtWfBZWLVUBXx*>J$)62Pr&u}M ze|N#|5KH8a7(f0C*?%6^z5KjgJ+MYeMeX0$>;*fx6Q6rH+UUk;hL!&NFS|gNe|p}O zP44I4n3w2ZZn{|5cyCY2i`1Qm1KwJ`c)Gj)%v`zHsFFijlkS=GU*z5J`eJd{Gr^LF z7dH#kE>?)ok~&{qlC`E&x+R?P*2N89r@VYnziS#>TEjd&-m2aZ!>Y-5Y}!&pVje8o z#d3*lH)Ddh->eI>yDt|j>n?xZ%V*26t0epS+1>}t#o7+n=QiEjI8o-~#f{yFh0!{eKL@0U5! zqWPNgFXznH?s2!;B(QgpnUm$?-^XU$Us&)mR^RpHGKYzALc1gvWLuY63eT-!-Xqqs z�|##j*?9k_$f{yb;W5c>H*<%uUvhH@;`P`JZzYx2<}3;s2ZqZ|~%^$LR=G$qR9{ z%v+*!_{ILN3zaOzEU^n8w4C%bx0~+Hf9ziL!+*IBHAW8VA=CSRJ`a6$@u)=AvWe|4 zN*}U6YOk&Go48TrWo+491wg~=}H>|Q{zRP%Q;ytHZkF?#=jg@0w zIG*YH(zZUNNVahQ`pv5ZOImAK{=U5O@W1SzzV$m^>hM-gUVUYY+}>GE{+e5ly?FV? zF80zh%NOr`7OVShdsHI3Ncn1O;S07)WpB>Pd~y2ieOzQ0k6lyO>{~a578KW*JAM85 zx&F?b%ll?+Wn9hlrSWCP+*1$I@hsm#1!U-)9ylGk@MXWUxk-aX+=PKl-b-YFk#i#nV3biSW(PX29e zMRf;X_2#>KbarvHu}0kg_VQ%W!uTmwQhOy9Tt4k-@noSw+T!WY7b@HD2yS_CZ1MeN z;!0oMY_nY8{pm$`zrRcTTfr#XUcSpci{qbp*hl-7XzkB=`1n=G{J2JouLozO$;2l5 zEz$0dd(ZGXv*DfhHm*OREH9*QzMb9rc&)|#17`(=GAFaW`10(7*Lf3(y%X7QxYqw! z>Uf>+u1d>1cH0Ek7e+M^E55!kl5tEdv1ONKc%rfEjhXvJQL_n3^Ly`Uv%QdT3{Ssr zxJysbHC6nf$m!;|8_RqBmoE`n{8^%FTH^Et6Wu;O-gUc5X0NN6w5-aft2QMGO;>sq zOcbk(iq6-Y^u@`nZ`dq&>95@0m}Ot)-QGPvJvit(W4GMf$r7KV`}YYgDqgZW+&O(> z){J}Y3}O*;n)_57>YHq;vbLT*TPV8d`swC%6M0|HF@9pV?wPHjpUUP}B^>wWOuonZ zVY0;~OY89b$tevBez@MRHu$xjQ!AxM?D%WDrJEKg2i@LjINx~Jhg&a&7Cv|WqAbEU zH+fm5>6asm`@g)rDGz+=! zdm{3S|3z=vEn<7+T(A2G9SwWzy)Dgb7gJb6gZqb@9Hy4jp5BsYWvDiL9$xU`OUWg3 z`Dg#$#ky_IsjOIP`a-JKR%&|ANvFlqT&Luc#l>W2Ys3rPJwA@QqbJe#vxLiP|ZpMs_F_mTv^x$ms>4^DNN{%hq9#;SrV?9VpKx4n3p zby0qaPu1H8RoY9qd*ha_+fmhIcWHiomv^h|MDO&oC(h~KJa+M^fbELE?JthanXw~F z?+jaP{LEY2m&J~!&XKy<@G-4L@2tSy&WqkBSl!EyPTln4t=hV|?epfg$GN{awK$*I z%`EvYO-O>9kjar@g|KhJ}|4ndcxz{rJz+u%3mqZ0?Hm_@c5yf-y z>7vEj-E}!~@5`g_3;x<;x7W&uFV^ja!mbr5mxIpB-_f)8zJwDd&pp z{CQ|$xbqkF7QH{Qvv=(>d7(aag8XuqP0aHgjGxV~ko=QT_5EtUm-~gA${z0@-u?ZM zqsrE$-eCLFFpW3Sv(GU8@hdQCpSW&<|J+6Oj^6v%O}Z#=&9L{ui{rSx9<5OZt+yPyWaPo&fHlcac$A@iTBjbwFZWo9X-DM@`<-o z-)@{0f1+>V;zRvT1>K*npD6wy_jgGNgMUNH#*61!9-9i}d0M+3jQLWc_v>LziKWnF zU5UFNO;754`R4ya=+4RijTa^hercJo{9>r{i|@KsM_;5DFnm>7T)Zk-Caz?OW6?_e zx>X!^`y#I^ztAeQ%N1IDTCP6j*Sf}!qWgN%_O@x;hO{qZdvU4ai;mNOai4zKRhy5S ztWLgoy5MOKTLY`xe=%0mN+&%TtN)kM7hm6z?_=7uuJtsZuG_`C{!=8J#r2!_ecqQn zQ@=5;{e}AXI}dt3yr{~&AYI|{y`{sc%&+P{4}VgLe&qkgP#&WluR|VxxvP}!@~!9M z>nmShPT-SWw>0!lM)H?uN#Ab;xBpvH)6U-fW9oUv&llXSqc^WK z{A*3aJbm_E(aOg^C(kgpD0YdAIQQ!E>YvH)bvNx;5-boRdc$|K@PgmVgQPv~&Hek+ zrzcL?EnhJ%W80kL{kf65{5n}@G8L#AE?4KZ)8n(P+9mVSc4Pi3&f4^I205CxRjO$& zzjnSTY(Ia8t!dx<)_=0=F1-`m`)%SqpN00;ObhGZuQWWnmB~PzFV?uKYUahWp4=ik z|9@GNbMk`arPViI%yQ5>r6GF8J$!wSoVw~EhevG13U8-|d2TUSzjojE1Iz1b3*Ul3tFs~7-FA1<1>4Bq4Lhx<68qVXG-&~;a>Immh8q|e7BeD z|LBNGGTC(H>&sIM!|&X#SMu?-K=icUvMnp z&%7}wlIeQzXJ77*Yd1@Wm8XXoZ5N+E{l<+`%jD}A&3gDw%RP|V&?$G<#Z|D=xQh3q zVar>pyIc5{^}jS>1J^i#Zl;n)o%I!O^~?=KKu(^|9}&K1NAi-gA9n zcFAIc{JM~m7e>n`IfU-Wt;_b`dgBFGzrD!QBYq7Uk*Al-TAh43<=O0-MZL+J4)O)~ zop`^-Jv`#Ile)uohI@;yXBbPniodl=e8>FyC&#X%J1YH`Uc2g)o%zqirT&Qh&jp5V z^_O{1&n>(xQ(|Ruvn+#g8q41&KVsvGukFj^Nzv`=X;Wyjbz*)e$xw)^TiiJ zSz``li7ifl`>U$Gj>+fP1@(m2?cb-|maf0Oc=@pw-j`F%?S5R7sC~s4@*zCkkGUc1 zpm^Yw@9(6IS8A?I_n#1w-@p|=b<2&@Raah2I~B-(@%5b5jJ<2UcLwIKRg}-y^YNU{$rO_$Qrzmz~tV zJ^!?|)+b=^ix-z9in(_>PuIUNb;s6uv!!f*m9@-!bE0d-?QZ+kv*h;v35?&A8@n$= zV(Rl;LzCAh^b?wuYTvW|`V--u=l;U{T9+Zi?s*?qf7!~orMvH5Wi8*L6InF@>w`6z zSVJG5b$t`t>G5RodV8y#kYO;uDtnz$=qnK zC#O~R>CYGV<7EH-z1|t8V^lTuCo4~OZ!KSm{A$^%#4Agm>;G*(ecHsSe|EQgXMJz{ z`oBLv38gK(7@KC6ReN6IbH7OX(l^Uv*Uy~pe%|WK|NQrTfwwshG4yz8JkYzm?6P1{ z(*vX9lg@}~m9oSu#?EpOVJQ<{bjQIfWQT`fBUkQ;s{abRIdn58mA$Xr~%C?wat z7vHWQ6a1b|t zz8g=Fog@f2{LDb^q6z9lK{azW;IP>56+i_U>)3 zCcU`-{?eM)tN8x@4YcE~3Kpu}+w(i`#Rc)^z&Gs8CtBp5-rBx?_oCTfQghf38$0~J zy)E|Q!u-X%Rw{Y7o}aL8w{Fmd*Dv26-W26F?cNfL#s&8G{#faxf1kLtzmUtZ|Hr3* zvyV45{d*j+_R*#$xs83Fd8^u3VlS|wqJ`YM7GOor;w6qa(2FKL>P#(+%q=n08Cw{kiJ2J~7^1t+ z%)rP1-8?e`V?#rP-jbrk%$(FBE*l#z1qB5Keb2nKd<8=V1BD<6KPW%HM8OEb56-Mg e1%;iR9anKlVo?b=G!4y6O^i*sR8?L5-M9c;I4I2k literal 0 HcmV?d00001 diff --git a/tests/test_main.py b/tests/test_main.py index a5af5b10..a08a0dde 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -21,6 +21,7 @@ import shutil from math import isclose from pathlib import Path from subprocess import PIPE, run +from unittest.mock import patch import PIL import pytest @@ -572,7 +573,12 @@ language_model_penalty_non_freq_dict_word 0 ) check_ocrmypdf( - resources / 'ccitt.pdf', outdir / 'out.pdf', '--tesseract-config', cfg_file + resources / '3small.pdf', + outdir / 'out.pdf', + '--tesseract-config', + cfg_file, + '--pages', + '1', ) @@ -634,7 +640,7 @@ def test_pagesize_consistency(renderer, resources, outpdf): first_page_dimensions = pytest.helpers.first_page_dimensions - infile = resources / 'linn.pdf' + infile = resources / '3small.pdf' before_dims = first_page_dimensions(infile) @@ -647,12 +653,14 @@ def test_pagesize_consistency(renderer, resources, outpdf): '--deskew', '--remove-background', '--clean-final' if pytest.helpers.have_unpaper() else None, + '--pages', + '1', ) after_dims = first_page_dimensions(outpdf) - assert isclose(before_dims[0], after_dims[0]) - assert isclose(before_dims[1], after_dims[1]) + assert isclose(before_dims[0], after_dims[0], rel_tol=1e-4) + assert isclose(before_dims[1], after_dims[1], rel_tol=1e-4) def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): @@ -794,7 +802,7 @@ def test_compression_changed( def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( - resources / 'multipage.pdf', + resources / '3small.pdf', outpdf, '--skip-text', '--sidecar', @@ -802,7 +810,7 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): env=spoof_tesseract_cache, ) - pdfinfo = PdfInfo(resources / 'multipage.pdf') + pdfinfo = PdfInfo(resources / '3small.pdf') num_pages = len(pdfinfo) with open(sidecar, 'r', encoding='utf-8') as f: @@ -858,17 +866,18 @@ def test_decompression_bomb(resources, outpdf): def test_text_curves(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop) + with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100): + check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop) - info = PdfInfo(outpdf) - assert len(info.pages[0].images) == 0, "added images to the vector PDF" + info = PdfInfo(outpdf) + assert len(info.pages[0].images) == 0, "added images to the vector PDF" - check_ocrmypdf( - resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop - ) + check_ocrmypdf( + resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop + ) - info = PdfInfo(outpdf) - assert len(info.pages[0].images) != 0, "force did not rasterize" + info = PdfInfo(outpdf) + assert len(info.pages[0].images) != 0, "force did not rasterize" def test_output_is_dir(spoof_tesseract_noop, resources, outdir): diff --git a/tests/test_tess4.py b/tests/test_tess4.py index bb9cc49a..0b835ea7 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -37,30 +37,6 @@ def test_tesseract_v4(): assert tesseract.v4() -def test_pagesize_consistency_tess4(resources, outpdf): - from math import isclose - - infile = resources / 'linn.pdf' - - before_dims = pytest.helpers.first_page_dimensions(infile) - - check_ocrmypdf( - infile, - outpdf, - '--pdf-renderer', - 'sandwich', - '--clean' if pytest.helpers.have_unpaper() else None, - '--deskew', - '--remove-background', - '--clean-final' if pytest.helpers.have_unpaper() else None, - ) - - after_dims = pytest.helpers.first_page_dimensions(outpdf) - - assert isclose(before_dims[0], after_dims[0]) - assert isclose(before_dims[1], after_dims[1]) - - @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) def test_skip_pages_does_not_replicate(resources, basename, outdir): infile = resources / basename From 607eee198d81a614f67e50d8f11fdde35b39e2e3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 15:52:38 -0800 Subject: [PATCH 271/880] tests: split out preprocessing tests --- tests/test_main.py | 152 +---------------------------- tests/test_preprocessing.py | 185 ++++++++++++++++++++++++++++++++++++ 2 files changed, 186 insertions(+), 151 deletions(-) create mode 100644 tests/test_preprocessing.py diff --git a/tests/test_main.py b/tests/test_main.py index a08a0dde..4ef299fb 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import logging import os import shutil from math import isclose @@ -23,15 +22,14 @@ from pathlib import Path from subprocess import PIPE, run from unittest.mock import patch +import pikepdf import PIL import pytest from PIL import Image import ocrmypdf -import pikepdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import ghostscript, qpdf, tesseract -from ocrmypdf.leptonica import Pix from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo @@ -61,96 +59,6 @@ def test_quick(spoof_tesseract_cache, resources, outpdf): check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) -def test_deskew(spoof_tesseract_noop, resources, outdir): - # Run with deskew - deskewed_pdf = check_ocrmypdf( - resources / 'skew.pdf', outdir / 'skew.pdf', '-d', env=spoof_tesseract_noop - ) - - # Now render as an image again and use Leptonica to find the skew angle - # to confirm that it was deskewed - log = logging.getLogger() - - deskewed_png = outdir / 'deskewed.png' - - ghostscript.rasterize_pdf( - deskewed_pdf, - deskewed_png, - xres=150, - yres=150, - raster_device='pngmono', - log=log, - pageno=1, - ) - - pix = Pix.open(deskewed_png) - skew_angle, _skew_confidence = pix.find_skew() - - print(skew_angle) - assert -0.5 < skew_angle < 0.5, "Deskewing failed" - - -def test_remove_background(spoof_tesseract_noop, resources, outdir): - # Ensure the input image does not contain pure white/black - with Image.open(resources / 'congress.jpg') as im: - assert im.getextrema() != ((0, 255), (0, 255), (0, 255)) - - output_pdf = check_ocrmypdf( - resources / 'congress.jpg', - outdir / 'test_remove_bg.pdf', - '--remove-background', - '--image-dpi', - '150', - env=spoof_tesseract_noop, - ) - - log = logging.getLogger() - - output_png = outdir / 'remove_bg.png' - - ghostscript.rasterize_pdf( - output_pdf, - output_png, - xres=100, - yres=100, - raster_device='png16m', - log=log, - pageno=1, - ) - - # The output image should contain pure white and black - with Image.open(output_png) as im: - assert im.getextrema() == ((0, 255), (0, 255), (0, 255)) - - -# This will run 5 * 2 * 2 = 20 test cases -@pytest.mark.parametrize( - "pdf", ['palette.pdf', 'cmyk.pdf', 'ccitt.pdf', 'jbig2.pdf', 'lichtenstein.pdf'] -) -@pytest.mark.parametrize("renderer", ['sandwich', 'hocr']) -@pytest.mark.parametrize("output_type", ['pdf', 'pdfa']) -def test_exotic_image( - spoof_tesseract_cache, pdf, renderer, output_type, resources, outdir -): - outfile = outdir / f'test_{pdf}_{renderer}.pdf' - check_ocrmypdf( - resources / pdf, - outfile, - '-dc' if pytest.helpers.have_unpaper() else '-d', - '-v', - '1', - '--output-type', - output_type, - '--sidecar', - '--skip-text', - '--pdf-renderer', - renderer, - env=spoof_tesseract_cache, - ) - - assert outfile.with_suffix('.pdf.txt').exists() - - @pytest.mark.parametrize('renderer', RENDERERS) def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): oversampled_pdf = check_ocrmypdf( @@ -442,64 +350,6 @@ def test_algo4(resources, spoof_tesseract_noop, outpdf): assert p.returncode == ExitCode.encrypted_pdf -@pytest.mark.parametrize('renderer', RENDERERS) -def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf): - # Confirm input image is non-square resolution - in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xres != in_pageinfo[0].yres - - check_ocrmypdf( - resources / 'aspect.pdf', - outpdf, - '--pdf-renderer', - renderer, - env=spoof_tesseract_cache, - ) - - out_pageinfo = PdfInfo(outpdf) - - # Confirm resolution was kept the same - assert in_pageinfo[0].xres == out_pageinfo[0].xres - assert in_pageinfo[0].yres == out_pageinfo[0].yres - - -@pytest.mark.parametrize('renderer', RENDERERS) -def test_convert_to_square_resolution( - renderer, spoof_tesseract_cache, resources, outpdf -): - # Confirm input image is non-square resolution - in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xres != in_pageinfo[0].yres - - # --force-ocr requires means forced conversion to square resolution - check_ocrmypdf( - resources / 'aspect.pdf', - outpdf, - '--force-ocr', - '--pdf-renderer', - renderer, - env=spoof_tesseract_cache, - ) - - out_pageinfo = PdfInfo(outpdf) - - in_p0, out_p0 = in_pageinfo[0], out_pageinfo[0] - - # Resolution show now be equal - assert out_p0.xres == out_p0.yres - - # Page size should match input page size - assert isclose(in_p0.width_inches, out_p0.width_inches) - assert isclose(in_p0.height_inches, out_p0.height_inches) - - # Because we rasterized the page to produce a new image, it should occupy - # the entire page - out_im_w = out_p0.images[0].width / out_p0.images[0].xres - out_im_h = out_p0.images[0].height / out_p0.images[0].yres - assert isclose(out_p0.width_inches, out_im_w) - assert isclose(out_p0.height_inches, out_im_h) - - def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): check_ocrmypdf( resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py new file mode 100644 index 00000000..4054e4e4 --- /dev/null +++ b/tests/test_preprocessing.py @@ -0,0 +1,185 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +from math import isclose + +import pytest +from PIL import Image + +from ocrmypdf.exec import ghostscript +from ocrmypdf.leptonica import Pix +from ocrmypdf.pdfinfo import PdfInfo + +# pytest.helpers is dynamic +# pylint: disable=no-member,redefined-outer-name + +check_ocrmypdf = pytest.helpers.check_ocrmypdf +run_ocrmypdf = pytest.helpers.run_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +spoof = pytest.helpers.spoof + + +RENDERERS = ['hocr', 'sandwich'] + + +def test_deskew(spoof_tesseract_noop, resources, outdir): + # Run with deskew + deskewed_pdf = check_ocrmypdf( + resources / 'skew.pdf', outdir / 'skew.pdf', '-d', env=spoof_tesseract_noop + ) + + # Now render as an image again and use Leptonica to find the skew angle + # to confirm that it was deskewed + log = logging.getLogger() + + deskewed_png = outdir / 'deskewed.png' + + ghostscript.rasterize_pdf( + deskewed_pdf, + deskewed_png, + xres=150, + yres=150, + raster_device='pngmono', + log=log, + pageno=1, + ) + + pix = Pix.open(deskewed_png) + skew_angle, _skew_confidence = pix.find_skew() + + print(skew_angle) + assert -0.5 < skew_angle < 0.5, "Deskewing failed" + + +def test_remove_background(spoof_tesseract_noop, resources, outdir): + # Ensure the input image does not contain pure white/black + with Image.open(resources / 'congress.jpg') as im: + assert im.getextrema() != ((0, 255), (0, 255), (0, 255)) + + output_pdf = check_ocrmypdf( + resources / 'congress.jpg', + outdir / 'test_remove_bg.pdf', + '--remove-background', + '--image-dpi', + '150', + env=spoof_tesseract_noop, + ) + + log = logging.getLogger() + + output_png = outdir / 'remove_bg.png' + + ghostscript.rasterize_pdf( + output_pdf, + output_png, + xres=100, + yres=100, + raster_device='png16m', + log=log, + pageno=1, + ) + + # The output image should contain pure white and black + with Image.open(output_png) as im: + assert im.getextrema() == ((0, 255), (0, 255), (0, 255)) + + +# This will run 5 * 2 * 2 = 20 test cases +@pytest.mark.parametrize( + "pdf", ['palette.pdf', 'cmyk.pdf', 'ccitt.pdf', 'jbig2.pdf', 'lichtenstein.pdf'] +) +@pytest.mark.parametrize("renderer", ['sandwich', 'hocr']) +@pytest.mark.parametrize("output_type", ['pdf', 'pdfa']) +def test_exotic_image( + spoof_tesseract_cache, pdf, renderer, output_type, resources, outdir +): + outfile = outdir / f'test_{pdf}_{renderer}.pdf' + check_ocrmypdf( + resources / pdf, + outfile, + '-dc' if pytest.helpers.have_unpaper() else '-d', + '-v', + '1', + '--output-type', + output_type, + '--sidecar', + '--skip-text', + '--pdf-renderer', + renderer, + env=spoof_tesseract_cache, + ) + + assert outfile.with_suffix('.pdf.txt').exists() + + +@pytest.mark.parametrize('renderer', RENDERERS) +def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf): + # Confirm input image is non-square resolution + in_pageinfo = PdfInfo(resources / 'aspect.pdf') + assert in_pageinfo[0].xres != in_pageinfo[0].yres + + check_ocrmypdf( + resources / 'aspect.pdf', + outpdf, + '--pdf-renderer', + renderer, + env=spoof_tesseract_cache, + ) + + out_pageinfo = PdfInfo(outpdf) + + # Confirm resolution was kept the same + assert in_pageinfo[0].xres == out_pageinfo[0].xres + assert in_pageinfo[0].yres == out_pageinfo[0].yres + + +@pytest.mark.parametrize('renderer', RENDERERS) +def test_convert_to_square_resolution( + renderer, spoof_tesseract_cache, resources, outpdf +): + # Confirm input image is non-square resolution + in_pageinfo = PdfInfo(resources / 'aspect.pdf') + assert in_pageinfo[0].xres != in_pageinfo[0].yres + + # --force-ocr requires means forced conversion to square resolution + check_ocrmypdf( + resources / 'aspect.pdf', + outpdf, + '--force-ocr', + '--pdf-renderer', + renderer, + env=spoof_tesseract_cache, + ) + + out_pageinfo = PdfInfo(outpdf) + + in_p0, out_p0 = in_pageinfo[0], out_pageinfo[0] + + # Resolution show now be equal + assert out_p0.xres == out_p0.yres + + # Page size should match input page size + assert isclose(in_p0.width_inches, out_p0.width_inches) + assert isclose(in_p0.height_inches, out_p0.height_inches) + + # Because we rasterized the page to produce a new image, it should occupy + # the entire page + out_im_w = out_p0.images[0].width / out_p0.images[0].xres + out_im_h = out_p0.images[0].height / out_p0.images[0].yres + assert isclose(out_p0.width_inches, out_im_w) + assert isclose(out_p0.height_inches, out_im_h) From c434b97f55a4fb39d80f098920c7aef9dd78391b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 13:08:17 -0800 Subject: [PATCH 272/880] docs: more install notes --- docs/installation.rst | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4e37ea6e..3398ef94 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -442,7 +442,7 @@ Installing on Windows production-ready solution, use Windows Subsystem for Linux or a Docker image. -You must install the following for Windows using their installers: +You must install the following for Windows: * Python 3.7 (64-bit recommended) * Tesseract 4.0 or later @@ -452,7 +452,7 @@ You must install the following for Windows using their installers: You can install these with the Chocolatey package manager: * ``choco install python3`` -* ``choco install tesseract`` +* ``choco install --pre tesseract`` * ``choco install ghostscript`` * ``choco install qpdf`` @@ -460,6 +460,9 @@ Also consider adding: * ``choco install pngquant`` +Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier +versions of Windows and 32-bit versions of these programs are not tested. + Modify your ``PATH`` environment variable so that Tesseract, Ghostscript and QPDF executables on the ``PATH``. From 55ae838cb75b442963cc7ee649d28ee1d2393b23 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 21:11:50 -0800 Subject: [PATCH 273/880] azure: fix extra build step --- azure-pipelines.yml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index fe2d6bd4..d1b486ad 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -249,8 +249,7 @@ stages: versionSpec: "3.8" architecture: x64 - script: | - pip3 install --upgrade pip wheel twine - python setup.py sdist bdist_wheel + pip install --upgrade twine displayName: "Generate artifacts" - script: | cat <.pypirc From 9af59c0d6d47e6c5716f25278dc60564c4dfe836 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 9 Dec 2019 21:39:01 -0800 Subject: [PATCH 274/880] docs: improvements for Windows --- docs/batch.rst | 22 ++++++++++++++-------- docs/index.rst | 4 +++- docs/languages.rst | 5 +++++ docs/optimizer.rst | 4 ++++ setup.py | 1 + 5 files changed, 27 insertions(+), 9 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index e3fe9920..73d512bf 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -57,6 +57,12 @@ of ``ocrmypdf``, again updating files in place. find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}' +In a Windows batch file, use + +.. code-block:: bat + + for /r %%f in (*.pdf) do ocrmypdf %%f %%f + Sample script ------------- @@ -67,13 +73,15 @@ processing. #!/usr/bin/env python3 # Walk through directory tree, replacing all files with OCR'd version - # Contributed by DeliciousPickle@github + # Original version by DeliciousPickle@github; modified import logging import os import subprocess import sys + import ocrmypdf + script_dir = os.path.dirname(os.path.realpath(__file__)) print(script_dir + '/ocr-tree.py: Start') @@ -91,6 +99,8 @@ processing. level=logging.INFO, format='%(asctime)s %(message)s', filename=log_file, filemode='w') + ocrmypdf.configure_logging(ocrmypdf.Verbosity.default) + for dir_name, subdirs, file_list in os.walk(start_dir): logging.info('\n') logging.info(dir_name + '\n') @@ -100,14 +110,10 @@ processing. if file_ext == '.pdf': full_path = dir_name + '/' + filename print(full_path) - cmd = ["ocrmypdf", "--deskew", filename, filename] - logging.info(cmd) - proc = subprocess.run( - cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) - result = proc.stdout - if proc.returncode == 6: + result = ocrmypdf.ocr(filename, filename, deskew=True) + if result == ocrmypdf.ExitCode.already_done_ocr: print("Skipped document because it already contained text") - elif proc.returncode == 0: + elif result == ocrmypdf.ExitCode.ok: print("OCR complete") logging.info(result) diff --git a/docs/index.rst b/docs/index.rst index c3759efb..3f611f9d 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -4,7 +4,9 @@ OCRmyPDF documentation OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched. -PDF is the best format for storing and exchanging scanned documents. Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply image processing and OCR to existing PDFs. +PDF is the best format for storing and exchanging scanned documents. +Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply +image processing and OCR to existing PDFs. .. toctree:: :maxdepth: 1 diff --git a/docs/languages.rst b/docs/languages.rst index 2b26b6de..811dd06f 100644 --- a/docs/languages.rst +++ b/docs/languages.rst @@ -57,3 +57,8 @@ Docker users Users of the OCRmyPDF Docker image should install language packs into a derived Docker image as :ref:`described in that section `. + +Windows users +============= + +The Tesseract installer provided by Chocolatey already includes 100 languages. diff --git a/docs/optimizer.rst b/docs/optimizer.rst index f6336f77..02f0be21 100644 --- a/docs/optimizer.rst +++ b/docs/optimizer.rst @@ -14,6 +14,10 @@ optimization and ``3`` implements all options. ``1``, the default, performs only safe and lossless optimizations. (This is similar to GCC's optimization parameter.) The exact type of optimizations performed will vary over time. +PDF optimization requires third-party, optional tools for certain optimizations. +If these are not installed or cannot be found by OCRmyPDF, optimization will not +be as good. + Optimizations that always occurs ================================ diff --git a/setup.py b/setup.py index 7e0302eb..214bc62f 100644 --- a/setup.py +++ b/setup.py @@ -76,6 +76,7 @@ setup( "Intended Audience :: System Administrators", "License :: OSI Approved :: GNU General Public License v3 (GPLv3)", "Operating System :: MacOS :: MacOS X", + "Operating System :: Microsoft :: Windows :: Windows 10", "Operating System :: POSIX", "Operating System :: POSIX :: BSD", "Operating System :: POSIX :: Linux", From c5571388e2d0976bce142b098af33a2ad20d6e24 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 10 Dec 2019 01:06:27 -0800 Subject: [PATCH 275/880] Improve test coverage of _sync.py --- src/ocrmypdf/_pipeline.py | 6 +-- tests/resources/baiona_cmyk.jpg | Bin 0 -> 148726 bytes tests/test_acroform.py | 34 ++++++++++++ tests/test_image_input.py | 92 ++++++++++++++++++++++++++++++++ tests/test_main.py | 6 --- 5 files changed, 129 insertions(+), 9 deletions(-) create mode 100644 tests/resources/baiona_cmyk.jpg create mode 100644 tests/test_acroform.py create mode 100644 tests/test_image_input.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 445d0faa..6435804e 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -47,8 +47,8 @@ VECTOR_PAGE_DPI = 400 def triage_image_file(input_file, output_file, options, log): + log.info("Input file is not a PDF, checking if it is an image...") try: - log.info("Input file is not a PDF, checking if it is an image...") im = Image.open(input_file) except EnvironmentError as e: # Recover the original filename @@ -85,9 +85,9 @@ def triage_image_file(input_file, output_file, options, log): if 'iccprofile' not in im.info: if im.mode == 'RGB': - log.info('Input image has no ICC profile, assuming sRGB') + log.info("Input image has no ICC profile, assuming sRGB") elif im.mode == 'CMYK': - log.info('Input CMYK image has no ICC profile, not usable') + log.error("Input CMYK image has no ICC profile, not usable") raise UnsupportedImageFormatError() try: diff --git a/tests/resources/baiona_cmyk.jpg b/tests/resources/baiona_cmyk.jpg new file mode 100644 index 0000000000000000000000000000000000000000..01d6badfa2fc3e008d3600593e653472deb3ee5d GIT binary patch literal 148726 zcmex=fklv2NYT)dO*k--U8zvSsBz*#4rQl}2StM}eo!$^Dr(~75)+q@ zlu}hw*U;25F*P%{u(Wb^admU|@bn4}2@MO6h>S{3Nli=7$jmA(DJ?6nsH|#kX>Duo z=Qq5 zQ@H0Z%UE?d=8{B%X|#|(19!{wud80JeW~}Vb5*3V;eo({mdQV_2)>RA*`!nWsx0rF z)2cPEJk=L4n}q(_AHHIQ#PKpuo}w)|FC>oLzA*oxg|hub`?-b>d!O{VImFpaQU6h7 zQ1~NQO5$G1)CrOT0&A+%CI5wliM8{T3uSNgYJ8miooD&}Q(Idf-bkLun`9YaB^y-a z_$|Txbwr$Go8_0?ty7s9>$MxI1=Sq?aEOFN?b}L$fzz4zPNc9bVz&Bi@z&Ty#o&z5-I*Svh?%K9x^XZOYjY}nRzly^h##3ktpZRc5f zirlx$eFzc%-4#-7<4OEv`j{JA9Tr+MzH<@4hVUt}?_&1U0`a%P>A?w{mw+M+_U;{e}-)4?d)6Y-`LsbHv#VCHGdxLnFn_k(oWQW_5!1eTOING* zS*}~_dfQ~dhE9v)482Dt_btADYyI4??A@oUtp#O@JahDxOiI-g^gJN#D5>&%$1&?? zTb^9W)7vV);fBQGl-9IqkrQ?>Zf@n`c;55eruuo)BB{4$cV!jC9_TGkYI3;9yeY); zFyl!HSvUWUtM1?ad@y_E(kmMJCZBenNtlw7Hz|1?!%1b^U03ygcr5Go-}l=`Q}*tZ zl_F;vcCjYeS2muzBUE_3Cw;G|O4ON&FSnb2^3ATy=~$+z+Zg-KQ`+lkhkg0ym%+SA z5hgd+r%bx_NkG6^b%lb*o7-iNb=)}PX8vnBo%QMV%k`zox9@6uOgyN&LN!&V@Pe3) z%m?Gk_ZgOKyqdrC>a<(g<&*AtoaVfCF2W_pzIFRomJ(%|Z%Zz3^?te4ZR$9T!_AixupT2$D4x`Th443YySO-6S zae%MzvgEq-s0`gdp3A$tU%q|2yf0Y)SMrgT6|s9I7=I+2gcQWIew1BeW&36Ov`Naf zYkl5YxiDD!ee%2HS8?9rrR&^tHoiPkp>MWN;}@1*ds*_5a{rWel?Rh=+qde?EjoT= z8L!!==(8q=qLMDnoTUDkF-2w0{C7{j|KYl}7Bo^sQrXK!xzoc&g?JYDkmnZxR` zZT`=+{T@eksqYYsx^q``6gfQ?U2M0T$F5p+>)9h>Z^9}UhH6X7 zx~mBH{a}}$D>{3%Z}tjV_JFNZPD|fk;Kp;%<8b=Zmvj0aD=$rdWfUEk6(_%c`-Bsd z{K^c9x2!v6U%tS;tvXctTGR6*b#i-~oL=kC-P$~9Y0aVvS@x&1v|YEZSnPc=b=%Iq zTh|LJZ)mzHJmzjw`4`(^p)6Ya^+wXR-MeO8d!}8T$y2TnBXYOVG{alZ+Hf-461-5^{G$0 z@urlWxz2dIt^J|R{~5OFrex1}JOBK&J8w$k7;@4jU;Yyb%Df(KUSJ%1WABnxN4<>O z*zX8l!7BDLs%85Jq1ZMcV)JHA^G0ne3izcE=_4gzYE;S z4SsBy>mrmz80WKRBsci)Jb3)6?}Qz_Jzi@S}p-`NAyoBz1fH*U-&kC$8l_y)gSm zKR37itg!p@?J2H*@2rXRYQC}J_53*77c2H} zE;TG%>-jV)>90di<=amWor*tIrTne^>`Sz1jEYF{OdH&`5 zDtGs*(cV|z`jjQUf0ilI-KJ>Hz+ta$as2VTDqa-S=g8ZrT<5ci%$3r8&BL zp0zj^=Y86CocZ0ReRrM|{;}0qoxSW{<@EA`K2;J~TEo_!U67>@aGSp6$T>O^U6)T66=w%%HuCBd#PZ~T0}ECci8x4pW0vN4v^ z!{*)CwC+gX6@%c*IyYZcygl&zdehd|VOw6{%60=KzOFiU%72EOOq~s}9@{<}pSpP4 z>x-OP36!L!r5w!GYTzU*T3 zajjDyZ=HC>&DbNb+`d(mZBhT~ie|4_ht54R-!{SKdeMOkjq)Ww8+7MY9Ir19xa+RA z{86pht3&4=DdtZuEY;B!NU;@u#<0EaPTHc1C0A-ppnO=)AVtsls%))$+kR1)jdb zv!c_BmwD$d$=o*Cxpe3Hvu9T4N^flC`_J∓Eft|FG$Fy@{&ZrY?H9`FQEnsrS;h z_1-?kC*N?2-|0a8*VSf^-6mPba(m5*mDjU$`5b-Zjq<%U>}M9s`We2w-s~C^eK+W` z!|n}Dx9Ydw)Su3>I_I{ks{&h0@6OwQY)?EkE!@f*YGZG^By79(il~X=Gi5g?xr8z7 zsg(9q-v0Dx-1Ok9Di@+ex9sZrCMl*lOS?>ZF%Pp%rG%wyyYhUWi$0ex>edA4C2ig> zb}8a#T4v&e%_rnlHY!^lU%!k!%j9X_`6JtnkEhmb+jMK)%{WfZvLJ8aiM#FJ7Zn)J z>DlfZviaIq&*Y7J`)*Yz8}4>g)N{x^@_#XZF{rZw(e-1 z;5B>3gFKlhGXELoglbn?FZ)p2@b6sZ(~UbTQ#bEuSwCaReg=al4;y9c?X>2YO_%@B zooOU%_%7<0+i%x7A0|Y4SRUH&v*U!aNZ7Rwg=yjQZae=w+oN|Vb$i5;;t2%|{{-J( z`Kw*;BH(e!KX*~DnbN}gN~WXAbLKqZfB5qHI{(KPGP1eXY`*=H*|+tBXq;%5yRxMY zd)|!m`?rd2j}~3;DiVl}LZqfH&T{vUY>w5+&$vE!@%>Ayc)c!eRbIa4;JT^iTe8nx zeE%{euI-_LlLLNuKJW^YaT=>0Y@MbJyv^cEO`E5g_48lhE$$2%U$Yau<3m zK4~40kn+hlV%yfHbMk7pM7nGyZ)IciSGnYv&%32AMsQ8$q=LmV^E?+VzM?bh*X>i@ ztGmP43nsB0X8!S2?yB6vl{^@(!a}LMq zz2ORwoQsn1d7&`N z@!yIaDsqo)U%OrVny1upg2mY4nOcH}?SZdCF(PF!OAIzftumHGYGdo5p)IU&oFW6CZYArKf{&CT0816UE<5L z-;mnp!NB->va86p>4ovpx<^=M+?c>A@uKjrR;+NumXB<`*{g0$H{0&H&sa%RrIuGD zykj4SKeD}iSXnJiA%oSodq8nWxEPR=@|rf07-p3P{KN!^!r@;KAV zyRWv186V}ndMD6ZlgZ-dzO=Uo7R{_J-r<>?S+LkrFpEbN{i z?DTTO$z{jWYyZtzwIgjuz|r$7*E~*}9I3kX$G+Z~5_7<-q< zT?sSq)GPH4zw8redq#8Tx*%5RB?%01{JBTuwjQYooFDw})a8;LsyzAgPfiW2oVnux zUtx^<(X-BhJ3suJzsRwmQ*F8W%_m%zGd_4Q&nf@&Vy$OMb@mS(*HfoA-C+|DJ2dUU zu?JQYe}>2G5}otSDo1F^YSvkLr)zHnIY``L4>J6+Zts@Uo?HH!T|Pa@I$Gmm^7gx} z77V-xN{?x*wYt9g3L>At3eQmO84osP_Xg%hc~9h%J|%Cx;;hfR9o z-1e#c+$+oDdtLX*T19?U-Il5LY1{s3cIAQR*w>0&ySAons@M|Qd$%08W;{-}`OV;K zx^=bUQrDI2<=ZAOEwi|J-tK^K`25%nzFNB8XVbP#VOlD%@%as&i(zlqDevCa_eklO zC0lsWREuA7uF>0e^u4=f-O(-aL^AeN$m184qD9wUN#9P{(b(5AWOV2W*bGm8g@!(lD{kkOY2PT zKGXaF-V(uWKd$a|Fum$_`*GOI3A`e&O05W&LtRf!ZQCrtmd0_EeRXpGwM9FP)~-3m zIJ3#n@zssTUstU2S)FO@V65S$VH+(|wzRCU^+OOJlBL_g;Gxjf_4HcR=rp_hXq zb9D9e7r5>-SU;<*XLa(#pV#eE)Mj0}_vn|Z_%g+3LW>o?%g+@x`DV7=*xBTQ+owaf z7Z+Iuzsbsclbe^;mbsw!Y3p{kCq>I|T$vmdeaBm5)1iJ9-(6@}10GOjMpG<634sO(+GdfpujbksTWfWi7=jXZxzvyZ8)m5!m*Xn z;SHO{s#$Gw#g=j%H<^4)bUQQ8S;_NPq*kuTyqFcRrr6UaF}lFu%*8o>9are}zSg*4t(N|7_qopCCj#+(yZLedc=+mh^ znu`xJ3QXYPKalc;edE$e*{``hg1!61JUZCeEqLa2RZD4ppT50p^X-l=KlKzRJaX?n z_M-p7SGh00Pv71h`SwUv&2)hih04bsm$<*E>aW^rx4Dqz1LN{N7r=ss^NzExShP+u zIh-RU;Q6A-Cs(&cTs>^o9`-m;?>XnA6>#0$gcWu zsBWgw{ue9kr9Se@G(XDkzbA2_;V$2t?m~_4Cnx;t+7cLBT3w4&*ucu-Kvz-UuM*Qz z`b;y;&F-GQYW|-kF}G*?^hx`!i=E!%{O}G&`R! z%H&qtV(qsRb%m>L_eV-EVCvc#D_V*(VL_4?g=DB~_&?k5zmNYjEZACTFIxY^WAeX! ziT@cctUgy~{-5Eb>HL5D-u`F!lCWs);eU(lzq@~0cTWDR;D3f)_EW1w{w(6+g2#2*@GFCJrkqLJ`dMxtTaa)6);y*?$Y=rZr8u$EX*mAe< zZK&R@PtOCoGTtv~x@A|>ai!+@-({da)7QA=7a=ho@>M-_D`Q*H*61ip03>|{c54a>NCo%1qGhz%<|JRR^~hwym@5D!CTzv zE7d2@Keqi#wnWKYsdMKzrm1wP#;&q{JpbCRSYiKWuZucsau-G{Y7-6UGGIJzzgE;X zZuJ+;aK$4MYX4?6^NRlr8PXoT&;NYWo+gvre7kG;->C_u@^6_l1^4`Cn0K`3a8E&X zVd(zNZ|&}Wdz>%1ONP_qXS>9MZ;NK>t`80~=dRs#_Bwyg3%lSxt_?5W&$(Q8zEW;g zp6j=qnRV5b0yRtcRvlixuQ=FM_l?xd9o2WQxz<(4vY&3e8vXORN$BkA)zurG-3ytZ zJf(KIbwnxK&hy-7W^LPSw_p-G=QHc%A5m_H`d7Hf&#mk~v_kEp_)&S$ihYmF1fQSU zrh7NygZm#=_UEp?u~Vy4(Q=(A&v(Pyje8~?f4At9|HH^#*Y+>_QFrEUR=6iuJ=fQC zyDqC)CqD9beyQiqJoY_eZ{I!rhe1W3iuIc`<`p1tR*6wBe&tSQZ{oKdtM=p1;+>`vU`#;0!Aj9&KeTRR5q!z7Z*?Kwm z$vii%6+0(|e~~{?&Gl-_T7I_=^JYD|B+{f3d+E%d&B?pXgJZX~K3vOu(8~Mx^4sp& zT&wbW6;)0Y)83sa)4x_UYi;N%>dvEUKR+*8lYf1Sz3i)LRaH0dyxexV(o^Q* z=cx>45(0JKvd@04mwP29r?9)}hO_P({|(yfm$OfNKkLc*J-(&ctNdl0%v+^1cxt5WAtqpnB)g1btp+EFLL&blF7rHU|N8^7Asr+YHKcoJO*601b{~12{rvGPH zzq|gggvRQE{|qbo|6csFdUN|ko-7g>gd>g;8F z?G58^&VAm0Br4|iqIuGpvJ)jN-6k^rJ}lEv`A6IAUd12RrDu0(KHN6(reO1wOab{l zh9}tPAKKuzXxF#7MmojLxyn<*c|w;2)O|RvApg|WQ*GMDmrlq1X0BmMK3&mZT~K_& z*71Fi=h8J*0UIwb%VwDI>7D}L9hp!30o7OPlQT>nrZ4yV=2*CKF~ehao964bFP2XF zlld~4_w)N{50BVosR`^eK5vsQ6Lhl9r2&vt-nS0 z{ycMeL-$$NL*mB^Z(A%{XYbxTjGYDiHSMNIpdMN;kI{n&8Diy7G`E8ii;aMrCIc!$=%(*;H+J2&@r~o zwB2pSnnx0E@E*UA)|Iw!)y>R8mFb(;cs=JnsM^1Pd9u$H(`a2il}X=9J7XSJsBHYg zP?Wi=ZEf;WmrZeZbiAH(Pw-5B!>-nq`}fbaZ=I%dd#?AZnJp67apHu^yr$Xp^`?6l z`KBkEK0evIqD9Zhh*N#R3RB+XH``X~E?u&Hz4~2`6EP|2D)XAUX6in!40`3EEFKpT zbmL^;m4;g5VAVbOTCtuiNv}_HMN4W~P2gXawWTpnPwZ7ha&*J&B1A%AB60@H&vF7tnTJ@)u;cEx9Xy>*k)(TAz9rRMcJ8c7k|ytT6j74$n>Hk%OuxL(b;sY;5b7= z-JY*&-t)e^FxT$eBioC%clsqNZnG$*iu4`$IOq3u>kr9(+45~4z4u;<{`)32;%Rsx zcfU{Z&x^j>E-P=_FzNF7oikPL?tC_7_b0>aTeG7xC*F?m%zS?5%%r?KpS@y#O1!?c zdv|qmezNNNTY0IT#ygiz*<56Jt*dT&#y6?dy-Oxb&J*ECQD)tEY5vk%wrw->GgXeI zw=zj`9LynnHntx0usrkb%yV2hDpQ=x10>e4Du2Pu2kQE87iq9UoCFIdeyg*0w? zN@!{5oT*~a+~hIQ?UKjAK*qGB3WoAU8^dF(f2ibI-3qh$ndW*cMQ#I+xhv77x+*UtO(f`aQtJgx%&KK^HT{bIo^}fZnoz_K%mpyx4@crcr&3*D8CZ834z3WPGmeqIbhdTwD_$Ze7pTegv5>bZjK1-;!zdJeO!P)WB~G}jMlB$3#}jMk)p zaYIb^df7@l@4K>8L3Itk=J^e3tMA{{?VDd%^89I=S@939u4jq)**a%0ifn)CT4P;Y zsCxI#b)z{ajkomGdRm%3C^~W3=hM#9)?(^+uFtfdsgjsu?#J^r2i@I~>u@cG|2?d6(39j)?fO#UXTXK+}^^Yi}47wSU7 zqgVZBIKJvX!=CytVr%Ai?f<0W`JW-);6KBQ&|`n{|1(U=w*Omu`#-~rgRW5>Z?`#B zE7&hB3zXeue>o+ruGh(cC@$nqoaIFl74%tOzo&y!TGTEgY~s-)AWya$$A|9 zHpx?4;AepFq2CY67yJKCzbSIRGO5T@Ti#XjLG|NnT|2|Iu8PWRIBL~8%VdH!8)I96 zyYtr-Pun!F=v-XV7-7I;++-!c*2rV=6ha|AD342XGICca(pd}%t*rK8_v zF9_}Zu|lt8H}~z1=estyWhVXZd%+p^CnVfVZ2IKmrf0VlotD2-9y9g*QIX%>(%vhD zyt^aPD*iq8XzcoH?7w5YwCsTGuE4;UTet5yCGKppTx=WW%3~kZ_3YZcPr|%s1kXyo+N6>A z>q_u#$5O$rMgwvkQYbWm zi^_b~-J`3_RCs}J$v37|SyfZFODLtwZ33)}Nv%U8Xcwo7kn@yfUp%G|T)Y!=?q*YaKY(}8btUr+uJ_S>;{+Uujc7713bl)t0+pJAFo zJ%@apJx|qj=@0LHcI}__`Pi;qqE}4we|+!V|Fm_(?^Skt>$SAqc2~V8Yw6>r8u$5c z`mAaf@%**9x=#Pmx%WT9qR!oar&c%nT9@WNvEzS~r|tO9kZ&LRcGkuZr}O`4=W^%! zZ?4I^6CHPY&O8Fum2+7-uAJ?A93M)Dnh zEUA52Y4yQ$*?r6MA7u)KNtE)H*(Vt@U2T24`$*=z%9DDRydsh>@HKE@6^zaf z-MVkb9#1u6KhL61LWVEc9au#p7yH@=%zFIxd$Dno@8%Ocne+bI1^)O~+PXG-R(i|3 zd(Ym7?44Kn`Dx~?5n=n}KkE0)xl_^iE}K`&?ary!d65^0ud!7Swnuypic^zPm=FJr{J^rZHf9Iklt4oVOwI8IP zu{f9aP1MrI-WHS21+3+ozgM~>@ym;PD;&aglk|lHc1NZQMOlQkdj4E&D^Y7DrS&&{ zEy6&Ev1F6lXZ~4bt9$<_Hr-@*vy^kv9lqanpX<8&kL}Fw5Wk=D;q;NU?K=zQA71&- zpmp@4>-mX`H!-(OZK{*?|Fmdd=z7I(2k+503rwYwwxzi_UCw zF5TW{e*W32xBnR$zM6!_mianVa%N}qOb}yyQn_ua`qKvq&({{s7Zv;D&v8zvk#~)& z>ZD*bNsq(^ahboxG497Am+$@{`pBK(w~3nNlj-GWxOOP8zMp4#R^)8d$YFq^~%qwT1DXs8plm9-FqE1XO@WZi*E%! zm!jVPebega^@PFjtf>E?Z&h1n<+azZ@^gI@X%y$;#_e@PsKDYubLbxZ^wplrUX=e@ z@+3}mvg#9;ge_|v)tUP^Ox>NjLv~y)@!lI07JcK*O})oCyq7kz%;B?8xEO0&@^vS) z$U-Z@Fu7^5({}CmPVfkQ<(}L#`KJv3`mF_VT9G#Xk8SM_Et#^R&9J( zxAwd~#;M+IQ@ZSj#iUs|g1m138Q5or$Y|9X|M|OOt!hA*r~?-pgY4=(E)f?4MYL8N zU|t~lqN7_%OQ|(ff}w<|D=I3AoOHNmzu>wO?rpNkf%<xo6@p!N_>_ynoj}ocB2&5#_O8-Ttww=*HN) zA}3WnlJ%8a1Mgfh4-fkmcD}IG$eAZuSEaXs{o2A^ySH!LI_cTFiU&=GJD*H!TGd~a zvF4}V^Ia2;=JYHmwdH?Q-~QF^(2neWv1~cbxo>LZ^MdZkZhEzyyCwcRM=M!#gZx3YOof`aNgX%@5z3O8)W% zR~y^@Tk+>)W#MPD`kP<=mF|2vpD}Zu+|2dQDothM{8IPtvYs>lTm3GL)l)Brhoi@B z(A8Y^y_c7?-Fmq8okhlzAU6q_o?rb@OIO|UuU_uSyY+DNoq~%emh>1{N`95w8ugF! zoz7j|byD{VoZXe?Bs5Q)pC6_5DDTGZ6Zac;%NmCZmmR(5_J{Yz%ULfHi#{p~aSIst z%>=dh$m+phB)NEte;1?9-?DLCmhs|dvk-k7_hy1S3dp7A%D-Wwqnc&94JK-Ght zT_&V>^&TOe9MMw~)>KPrS04;NcK?xFXW0V%KR(J9vJ7d*zpmBi_sad_xZJ|NtVHrp zVOYtXgT-0qLZMSqzQ1*?T^jc~cByFGZ~LUFP4BdJANZ;~|7E0{=9b^(t)|flyJx;l zxn+B3l6_#rJ-to;E`BK6oFQK9ovso1e#^vP7bCwL-En@H{cLm0or5fgM1S8*<=y`J z*2Wsk6)|^eG@D*s$lKlRU)qyg7R$)o_op#ppTL!OdH;kW;-YH47rs2Ru_y0L2Unl< zr$t@U4{wf*`H}BC@0+pgooO#FFO1!tqWnza?YZn~oH9!;JlnP7!Wy-GJ;%#cxDZ-0 z)>>U&P3$mcSV?qvJ0xGcFq4=_3+J!_?}tQXMWoJa^lZ~{|o{r&S^JnQmnJe zYVw~t+r9e~`;UgiIe}58cf#w@6D3BxV^<)``gX_ejXSpGO*(d>Ua)KJim0{UW|w9> z*0#}?lz-1;zg9%*z=zg@CA)f>PaQS?`=!zURokBZ4@@uf^S`sng8LM75;qw@2-`WAUVtNDp%&5zibZ@jo%Ol;!5#j7oEO)LA)V95S=kyX9bYa8Fk zf2|#NNafB6%9?d$j(cMEX9N4U6TiE^c6F7Vb6rnuS6RH{#f|?NHvNbeG zjQfg;Q_l^Xcj@Y}9m)LnQP%(G+2rU7_3)3WXR4<6t!s8!USnmuS&*mx{lxG6KhKIT zl*trg>2+H1f??e=?uy(558X!|iyT{BVi;^&t3>ziF`M=>KVQh1CppVqsx?%uflKqM zswehf3XIh=xyH9xWYgjkvsHWKC;619lqug@v2MClHhW-KWZ)CuNyqq4s#ytH3%(7w z9kq7by8KI#;b%air~b9{2flUnO|8ve;cz+HGBrHownCGnABVz~YvI>^weXs5o)%ix zyNN?_%7Z4}oQV6kv%;QLhwC-oGtog|3boNUDu7c653sv)TiA#$@a|0x6b{$ ziumfavYEUdJZ7KOmhq%gLF0F*d~wfL37bIgAG!KD3y)q0A-3%Bg;>gJ7TU;CEv0SkcVt`fkrg2mEb@aE|7wmgKdk>#bp4mRJF2~JE;w9vUis(s z`8BH6HOeb8BM#rl5*PdJpI2l*S2XzGuD*qSTb^$*Re1d$u#N zADX6hT!o39eYyIFqm?}q?34yKL{ysYM4JKQpc0uV>Ho^S*D}k3TUV z`P;tm=9av!bt?ay?s_`0=xFf8or&|aG{0Rdj85pE`LXv%iCjYo^UK)z?RzwrR%9`4 z3wSs8Kf|>-#rO{V^%}_qRg% zzZ~I{YFuL~7R-+j_|IVV@K2cGn&;O{|7{7M9^x^G+T_4U(E#Rv;Z86(EMD+cvlzC-S z)76T1iP-;mvF_U4?a?mxR`9=c&Ahs5QrFpyOxnLYJ{;^ToP5L3{|-6PfC8?wwh@R=cyO<(nxq{ip1{%b&hp=T zx4)e0;vb34Dz#^M#VPeHLL|ohO~CV?X*=~hXL)D+W4Q9TBdtOp=4ju-#nwCiFfLlU zxopxUqyG#wylcP8$UlwRCwrxBa~=B_w{4&NhpX=v`M(!5J7Sw+ zqw(i8|1qgw73qhj|5`2h%vDlg{r4p{8h_4;+_f^6T{L;({$~u|;tqd0S{RkyzRt#a zQRa4bwNI~wPx;4hVcyIBX!qog>xDnEx4yUGTv`!+bcN*l#k0>`k=|KueZamkbdS7wTyQ8zr*c>P883vTWgrn+|beqB9$uu$rxG~a`si>|(51XB2dhQf)Lni!9{ zd$F^~J)Yt+--)2h(v&fgKWeX0X)l6bb#cKdJ=J?Pk+sTH#4 zzA^3D{K&@jk7v#bqw`D+wtkOe;=2X5trrNKbdkZ{*tIfj@$x&F;h%2Cw0us=F+0w; z{DnK)PU+Q|x_WY)W$&Gvw>HYpSfjbHLN(*Q^ppn=o&<({<(2#FzP11DHXqhA0q1!S zsHa_bc);8v+$Jfz(SP^lyfuHePb?O8>)+`X{ygA-@Qnvwmd-o&%u;!Zg+U9?LAHmV z8QlDn-@2xTX7zRbR(@D}Vb72B1zvoGS<3dIs`X2&^IM*M{-eFzeu-kep8e9&TiNd~ zC)mGv{AtmyTQ^sigk_$oy76?wzdJ?y|E~YKPV{r)$C-Q)YKxamd=tt4L;rPL&CkpG ztlcUFU(NEAJNh{PgGeZEaOlpZT=f4=$(lruHr_oKg_l_>Na}{!+_8 z9lg3z>wdreCI43WKaTM~94U3_bm#@uoYPD1?P93kaa{S=b>n#_3L1d#8MOj$t{a>-aqMJXhXYW6{--rK)Xn~!w{f9|EetUoT zH1}GW{LAf&WrY7|)%*W`wdeU!mLIDh+4Vo1qPnK!;@`z}>`sUOM$P|xuj}%<%YmDG z-3~>kS$p#=?ptwqYT_3Q(_7iUbhmeG(LSs8*0JlE(ahs_ZD&+BH@k{9KV1F1?U}FD z0~WnAC+6RoP`}21SzY3rdsm;`=P2Dh*XVP=2fcM=_cA{jN9X-hIe&(I`$zs{SJ5ls zYYF$#Lh@3sC|`MN({agL{m$gRp7s^_g}3Ftx;^uCvh}-sqE}^8?MaIlC#u@M%Wiz{ z+xP63Y|u4}i7J~X7oD*(k|{WRrLd>^V)(1Sd4#RP-kODlZ@k&Riy7g^q+;fpM?C&i z_MbuO(|?A}?%u46dvpIYd`phW?>=VA!M`T{%Eh?Am1m|KI~S~3`lQk3fVF|NU+uX& z|4fhHwU~Oz!BJQ0-SH%ge>4ANv!(H!deAdf^4FJtQ^HGASNteH9OZnoyEiy;&)%+c zzoM(t1>GgR%hWVgBy|>_%~>JC!Z43FKl<*=U8xJVZ}(pPa+$|E*&u&i<<#D8rqiDV z%(qn~%T}1K6q)wz+Hc*p%XXH9Or6=uwPr%_zMbq8Nn-_v&J)*(v| z?c5xWJ_81sl*zZ%@64O4&GzZ}+wvm+C(dl2=j(~y>OP(rS$rl>Ju@=UyUg`~e5>eJ z{sXrc7Jt}x()ysuM)}{b%m1B?+xS3o&wb7V`#+y^_rGp`ZC9+|_RJ3ZT}SG+gfG7C zA0-l&x8Rxf3Hdsa$I5a&?4Ocj^m}gH_;LA=)wH*(^5kxwv$kOVHS6))gN+wx1TwF66S7ga72^0>!-@0jH|qk69UA4~sUoBVM0 zUB6RTV*NAbWwcB(Z~VTouefLap ztX!2fjo9Mh?4$b;FACN7e{cV>{GV-%;(vxt{i7d`-7sBkf8f;ozlp!D&y)W#;ZGjB zbne&RJk^qm|1-RCf9;wXULMzuKM7)ybvIeWAs58SW!!lG*6REB4ljH8U3@$5HVNfS zncs}F)~>jJaOunK=HGRyy5wS1XNq00NyhT8LwKeO*n|G>0Uzuhd@d-=VJ zwvQgWuRb#_zM9O^D)0YqYvFwEH|eu|_I^%eSxbv5$IUlVpEst^aTK zpBLklV`Pg%!Z%e^#~HfkX4zyX=5BV+|Ds{C-^gNPn!i!KaQ~jeHaimguHMevHJPPP zas6ElZ{-;YThC;6aK7HX;#I{fe!s03ZpFxU|C_raG51`}lE1ee*)>0VWk2KXwXRRx zi+6oW`>ZtC$nd^Uj@$E9>)6;A95}vUol)9dlivMx5p!=Gt9;#cpm>*?^~_D4ckcS{ zT(qzE@7%mI`vk7M^{!vJ_YV7l6!{0)m+K_Y?swhQZeLXM_!aw)aL0$=&o1fyd;Q;( z@Qv4VqrQalhWbzFTf@xcaNfyl0{io>M|Bzt{kad;?XNrbP4f8j{?{@5N3VR^BhR*E zf4$!o-FNN}Zv5L4dQ{tnp?T`W;)yR@%e+n(=B-(-sgxr4!nHi(wT;%Rx=0ZL_k|*1 z;w@iSZtb>yvUA&S+q1oIwrz4hDX(3#hVzXh+lD{tYkk*5-d8=l{pP!z{M4^y!( zKTYsY;xL`~aBo&@(!Jkr&F-h|*s^EZEsswgKmB{oS#q8$YBOi4WwT%SkHp|f_5PRk{C*UBWY(Hp z?#;TAEw}bOT*vjFfpwS6M*En$DWN|eM=bKV{B8w9cy>NZ*|v+(hZZJvGAZ>P;Ng0& z`}Wb6ZC$rmI7K`JMalvUSVdkC-js*OqJ3B>O_NZoEEUwR*+NZ@<>fE?$0j zL3MSR0!$uC%-> z(yKe$+)dWcZgtzA7mIf7`lbc6w2XP1-;0$&)wg3i5?>0>fYZ&BB!v zi4u>q{q680cdI=2t=AJ)o9rrSO=135X~$^a$ozSo_9MG#c_lN}KW5pxIdS9PYoJvc zJKu{%&Ri4p)KQo7;D3gai}9x6)2>UpR);>_w`?E3F5CJw(JxOvd@;{kJ6EB@?{x_K z_rDUxW)Y7S#J-#@FBgb2>wKnrZo{^f%N80si{0RVH>LK`ot`<{lw|(ZW z{|u+2O|3%@^uJ?#eR=2OYh53s9%|2CwD5~LE{_MK*;dODrz61XmthDd1zL;IswYF~eT8WRJMC;ravi}TY?_J}!_`%+&o8iT? z%TguIYk%f{8uPq=%eSjGtZSo`^2@fSIk0)A{%o_?(%L)sk9N+)9jQ0WrZE>iD4cs( z`SY51o#yT)YOHn#JD+b2HW8GI>3;jQfkM|IQng=@SMA7}IGu5H|W zCHG0iyO1waIjsG21U%1{Y?nH+txjjdrCT?oy<+d-B&zc0p$mV^>L ze-yA=@`{38v;2=w{~1~f<#e7OzsD#42%xBX7Yrez&$Ad-dujj}9 zg*E-K)Vo-uT3T5qH(07on6N5h0>hKZw=l7E)A*%9g4JN0H;$g=O0sq^mt z{o*?NgR6E5AKlhJ+UaQU{?DL?LPCb z{I?}+Zu!<1u$9v>+dJQ?w0+`!UBf=1<-GsB^=BpjR0jY1uwL=U$z^AkWJ%fgetCMk zFxBX|rA(gWspntgk9BD;+PusAwuau~-;)*;zMXh!Z?NzF_K-_&^WUCJ+}WAx{KhR~ z(el*}q>=&H6T^_|iCh<4WneIJ*!s+yViKFdujZ_>g&6yMI6(vrFA^t$It#z!t}wf z);Y7v7TV5P%r88}?%&sE(*sPGUfE!O&rRF4YU$K{HTJdoH}-_pKbm=dZn63=hW`w% zeh$Z#e=nXRBco@oAIUoZkM}w~`%POF=jz*MPRlv^Z)GJjyYhvU#}^By-s!uYyJXYf z#>-~)uW!XPf8y&tAAkAw#yGnRHm=iBGv+7lTD_9BIidfIOn=s6mEK1Qo0K&b%p{VV zjQB4!@js6F()H15wIa)dzl@UXKkG`u*VLUiuk-I)ue<;G-tM}rQ`sASeAv|MJ#?U2@yk zR3v53XM5$SXDXqXesNuzitZ!EBPstGDhhuFM%l?<_1nJS<}2NN)h&~cNO`CRt(7ik zk*|7JW4~%{zT@UAQTMzv)jLckcrA@Gd&gn-_5O>hiC4dPdw<_P>7M6eo)4#%MoRBc z_vGtSp9 zdv5P4b4_pOqr2>fCr-Os>RnyNtYma^!e5bM*MRwf(XED}w}mxT_-D@GTgCCyt*X>E zyYZv>5%0VHK4wv7z2!1ijlWI8%bC+2-AQDxUzdIMPxMFYfU70_8#1JLzZ7QWezVAb zH~lov`Qs<{{y1;DVy$$9kJzi;@Bcp6@2NC>Q73;X`N9`&T{8pri2`BA%U&J-e6}jT z?MucU+lRY(wNx3RomSj`Xk#i8hf7v71#q%_v9dQV{Fb4A_)@vzC!(^7>N`kpFqb2HMCUcq{G$+f35T~-7a%@Lf~ z)7AAzurOCgb!N>G?VWxcCthhR+0a&9Yj;n2X3J%FiSP+Lc?zXigx1CyruiD()KeAq znb=y%aO@GAv8$h~wd|Nfi zu|O&^dH0l!PC{y&8|03^dv4!frM=;YP|h+@J=HB%%ffQ%y-u= z@5+@567iC_v* z$p_T-zr41gCj8-y8tw(&mSVC#PIvtW={4` zE7|*OqSKCK#WCztsF$7cTH{|x(7N@U3L-fZ=TDvTZ|3=ze*RriDtW>ITd%BM&jH zdnARuw*5bY+|j&iw> z#S=Tufp%9!e>l&3h`lZ2sgv=$+Kl;zm zVU(@m|1GRX>fX(E&r}=3U&Vh`pZn5QFJ0q3J=Upy;}?N%+xo104JFFHGt_%&2km9r zs{}2pkji3ok-q4T1q`Q^)>Sc8+B@D`bwA52Duu23^@`j@@<+8><=2K@F}0S}x^d@~ z_=KoKR(sBg)NKj1G_~2;CoywQSwiuDhMJ!F=S5H4sc~<5m$vQ0z3G)}Vwig`F7#-> z+W#Y5wD60&+Z7GeYV{7U(dRf$jevEn7lT7!i%!pYnOebmmFU>;fuYqYecv2 z_8#5`C;Q)*{AbwbF26JPG z+543{ciZK({Ce`ip3&8@dj9rkrDGfnJik(}|I%EuUm~quy29$Hh=5*P@f#8Te|u9Z z?#!MqUm@)mdP2PD&cqjjuA)n0&sOmwD%8+eAGS;R%q0eP_3tN#Y;6s@Xv&(uD|ZR| zzn>-n*~*7f{?2;;Z zVvElK%8t9gr}XvqM=hF_(H}8AAbRHUS&dS>_jjw5Ev2@GjlG(G@W?ngTR{s4RA!nYi%2%o` z3f;&w-)^;MXUDFjN=6mIc}oJLGG4t{*J8Bcgz`esWqs?qUN~#>G+7D83R&&aUdXe? zh~tGT_HByjzQEQm2seHFo_BrMAMq70s~u-9{;Z(BrTf%>hW5vibxN0xtzQ^(E7?f& zuKZ>`*PyF8NB_+>k@(qhs(z1EV7!D!uSl=-%J87jDPLL2*7K%n9E}h-@SyqDo+nL9 znhe}o0)3Y(S+wq9(79XhZg|Rkw39tnAZ;1Mzy3#%@}A-pGmgWWvaaEMkwL#a?q1Ji zd~`hU^}pH|>(|yv|9Ex&*glyyM}vQtxAz8p68_$tEZx4sHv85e!$%yZcMT;heurd< zrPvm~{$hIT<@J@7d3^O1i}`H-W~@86UF#_%S%@Cb-nDP$L+^(x?%guH(+Y*ug$6EDQR@JK5mqxzNBhMSxeNxTyGA>aN zKKRA-`il+sf0TKDx$-q+_obBC=Qhtv`}@S}&Xlwl4(iP>vbSG+`+ket75govA-gU| z#ams=oGs+G_GDOI3sc26lW_l7+Y*Q$pk%1_?uT+o?a!uLoRGX#Yk22(UB$ZaBmWFy zwNfr^*->|Y?fD~NCr|i)n}4DFXpx-m$6XoM1z-Gf-T!TRhh9b7B|De;wuMKm=zq81?bj85)E~56UN0Ya z{TKeTS3Czht(q)uqebW%Ny|XI?DcI>qdRwUAqV)U&Pg+oDXYV(N~)ipcmvd*P7r`CNk zunt)7+<37)C(OR-kEH5-v5yP&k&SdXV39t+npqN8NdGwUvJ$o|E#tv z^h12F?lIkeiZL6$%sVUpbCxEH)TA^1bEkay%>LE?wZkPlB}cX7k7{!Z_O@KRT|enR zLzLs9KYrVDjy_te^|Drd^X>U30-fSwqP?HqRZKYE#Bly$pY+e)Qhoa3ukEDwMCVIJ z9^3Y5o6?nBkr|C&5^w+c7D>EOjmI%5Qb|h+T*JKX$A2Fr!QQA(HLJ^0L z9h~w-;d*g?A@9{){`p$oJk3Q`X3qS{agF(fym$AMi(OiB;e@%$#BHJNE0nrQJ%dnM zlOE?*ES=l5;+5wCk?SR$+$UajELNTIfGc~KTkDflm%vG93fLdJ22T%eW%zV(vMvAJ z;2-n$w`EVc@=xi7nQ`r`Ymuhbn-3jKnU|=3*;c#nYMuIzV*ee#Rp(rKb^Unjw#%2# zs{BeR&{*Mo|E@rY#N_=S{w>L~pRh_k^W*!DkKd}Z>-leP|9JNI$&O!Te_GRPS3ffE z{P-<9I-V!e{Lhy8->pT8<)HO+0S6nS!@A$eq>*|)jmp^iKwtU;Ipfh?~XLrv{ zd%tqhJEw5F)AOIt+M4)by<6n9&#^mpUthm@vg((X1$HOrB)^`e^`y{U=xwhk`=Xxk z*!*19B}&p2$!(@DCawsuXN+CDcZ=ABHD(vwdAu(wJFK#2bhTT$64Z=bbMyQkKc&xn z0y#Io9`1kU{xZgF!@)Ye7iv%5biXO;koc0xS7fgfKL2%|_~Y=le^);_f8#v5S*};} zocmMP%<$s4MnrKeI(Pj>fBq~L8Tri?@-JxGOdk@Lj~g{|uA z3`5qwdz15C>0w9+lUpP}qBYC+8J-ng7=`tK zGHA(0sD`SU|8f4I{)hKv?!+GE)@@r=Ub8f@@w?r|(#_=bgp9NHLS3bH^F@p6#6EAZg2*w0lj) zxhJfFll&81y>&M0IGsJgDLi>gef7d$mmOH%%X%zxm+@8H8Q|@)AmvxtufT8{`$xR) zhpmoCEt_uH^+@?Y17qRch;p`dFK_jpn_4YDG8N4is*dm|06s!xmNa# z_4Ic6%36(eo3?3P6$x||^$xnc=!))5l`cz82~CAlJN~_tn}6bS%<~=_)$B)WP23(T z?`?m+ZO*!~?J@G9c2mz(TtB#Np5CRa(m7VG8$av3+xFzx;|ct$<`ikHON%~sr8Ij! zPj;30?8R{oHM1McH56)ID*R)zQ0TMnda@;6YUw z?$h({9Q?}i^{>VfvF+QHzkhlo*Ycmi^}wH#3o43_S-m*7{gGfIcTw(^rw4zD|7ZAm z!AAdazrea0$48}?b8oL*Hfgq>El>OW=dMxPHg(;27qj`XbO*D%gN5jYiuyxQ{%uG4 zlZCe)+5cKkE%%fCd=&boBOM7;N ze(t<`uS8}^vPoHrEA%gQF3rrZ%~sudNANsLbM|qiwlKK|7b7o4`=>mME=za*-ey!U z_$NN_{Iwm{{k*T^c;~*~qw%(M`Q%->YH8=M9ILI5Ut9U~eCyWx%znSxn^m&U8ch29 zV)9Ofsu?rB*xf(ZwKcn1_5}B#6Yh7IG<7<%+z#pj)=br`eYl_xxvQ z_`YVnKt!&TAu+=)09|T`6dOo@TKSQX<*UKqWo9~?Q{337o=e+*`uDN#dceBo^ zee~O1xe-{U5vf zMu{#))Gi`{t^rpb=}LJm(bl}Z`+`iuKOT|mk{{Jh1ut}5bBq7WS+&xg&iC``694o+ zjL|=8&Dr0sa6PB}S6OGl_K6d(EU>rul|27Y*LIdg6+!`qJ5u_c?}Udgk9S@9a@+L_ z0?9i%Q(~%nt|u;wKfLN?*{;@|Y|=BPg(zR>xfm2W3w9=*X#b-~!=+1X>i$(cw*2?* z!P#G$>%<@ZTl{zahuc91564W|9#_BTu>As)K>rURCa4~ zId@I_)bn+`Cdb-bCoG!o^ESy+-q|(8%6H{ewna8icf8*g#&suo>N))l4h*o|?bw@Tz%7LbG3M zlyZkPB_4cHz4EI__=$b$ANH+$&|4Pmz1&g7zhjBp?^kZlj9={MR!rNt^z@Rf$(i3J z()-z!7i+%AKD$*rbo&+c%j%3bMU0advM>M7u&V31)tswVxu7~~<-_*^S#}cF60Vx; zJQFS`|AV{l)!`kc`My`ym8^d8?yZWZlKjE;h2K`@zEjyYGqrQW;jfGKZTleKZ}MI? z~?SQRsZC;;}72p+^%Rkezu#_s8Gk{)AB#st2XZI`zCw(aQ~it zQ^LceMYBNRw0K2C_Y}S6@6D1754Kq)d{f@|uWe_%#1Eg>6J1wiUaFlPwdLp9=fS-< z{;_>q=YH%Se{Ge8sjpW4#ve_GpKsY2XP*2+`ihkH^owitOK#VC-H*89w|dEh)Gq58 z<(^Vcrm_5I5RgeoR`~cb=6>T=>y=;r#+{GbW4gys51p zen-s@y!^G=x9Y>WZ)<#y%`X>AZFn;6`H#uJ_Y_{f)Kxb9QFr&*q-(J{(XZ;HcKa@R zc6WburUuJIrFqAfSvB9Db?effz{1KkQKy%7*ZLn0&)gDod;9xbRfb6_!8g7f52!AV zWt;oRyVpHp>vSLQT6c%t{TVy9=XmC7u*^5~Prl4N=i9Tcm!;Yb3Le*=eYSk?m06}v zrSZq*e@a=CoFD$!vRm!!yR{ly+cTeM{yzTc@yAv5!e2Aj{xOxCm~j82!L?=47EE8c z?$_>{@%ZDSwb6&4L@%j4$^6W2$tVBkv1${S9bJ+sHmii8vg{*^#QVPGA8oD&FU~B; zn58>!W)4HwHPznQ50>_w&$_0qyNZ1cfwt*JwhesuRNkNe$Ww0;e{P58{zriy+c*t7 zwpJ@k+HF$*&k*VUJg(aQk7u1l^vT{0uV)qZ%#TdA{O1rD6?*sjLPRSu+XQ~(8 z6cyF-@4tShIOcrEAJOD>XQp59_*J)B-tb=ZhnxQyGRoHlAC+Ui{KwTQOTb%g_L{?I z!oS_Iuz$<{NF>Dm=6CzgKlT6q$vXR=p~e5`{x3Fn>K`cmXXs7+&u~cgasDrja`_)j z^B=p)!&S!raQPGcpTYU;@qbMEKQ8`f;3)6i|6<{t`UhO~y&^BRTvEMnvn;^cP-VW) z>;KnQJ?KQE>TRP#vyAxkkntwIbOFg!e@>j9STe`^O_OYoIA5>qFl3uh-Fx|p4b1lK_H%=F zY{>2nJN5aT*0e+ZcOSnOV@Ut-{PU7#|LoqlYoE_*O*{3UVfXQSI)?KP&(iuDxAqhs zD{x6^EnB|AdFjhf_xvQ*B)7(`@n>}3X%p(}AJFc7`RSfHh7rfP_D0NO>MyETHEZ5t zzNyPkSK4{GS%khi@`2&IdYAUBkNX?G+?UBPayH@2t#gAIetAF>Cxo6By&STuSO)?kKdO(zqQL( z?%ws#50_#OXPA7unh^O+?18rSuKpP^weRNq4vbmWm|FSJGJomSuRl#?l`LI0o-Hv{ zahS$loX(Q>T)uy6eRZ^)?@l z_Gmt~^$)+Y_4iu0(}yqb5%jEJ;J8%#?fx7YnVve6@P*piD^e3q9adphg&{7Pf#(xq5h71&E5jYSK0O{9(tLKUyqWa9K(4)_dmnbDFAGJg957?@g|;xpd$6 zL+h5w{a;pJ%3G$sQvbyHhh2w$q%QGXTA_5wd*Z*Dby43Q%v-v|Qu2?XYwWsJySYHQ zbk+Ck6EnXB-SwS2vFAZ)@(uo?id9y})@Cn!DIPWVb<6f`J-g-O9c)iN>e8P3;q8*` zRXw+Q&+AH?TRu2*zgq5;LG#D;dO!S^-L2?8Vzs`1tBc~hl#Df3_qUX&R5LL3AM4Uy zwLYGN5P^i$rkTlGKe9|OH~*+~mdja$U(NE4r-g0t1Z~Y}*6{+K<;C?vIv180I($l= zdt`F+bxWp2n zH=?`v$!$?y?UMA@4YoU;b?py-nX#r{;77dsN8iTH*L0G1^#6$Hv62tXTX0p^+jG;^ zw`asR=&ZgmZ}k`3RodMdd5dovvB~ar75#j^ua5oIFZP+$3&pQ|dH7j``TNEG0=p02 z<)>C|+y6lOu650?y4R}{PX^dm>~F6>wDb2}u?V9NX&!&AyR=VnXdR6Za6HJqweX36 zhcZjqnq8V2i@LTh&nmy=`lWnV>6LBU4d<+xXm-i|*S)KzYp))idgy$v@lmZWDHbL# zo>a`S$q8+D;uU=_$M)kp@A2u0kFE%uJhb+Y_b2n;udHg_<-NYhGCz{?+;%@Ap!Y(I z!h`<|d#r0+^CGhsZj`-n@ZXP@KQ6}I%`MHZ&Xqm8=HgkF<|0pNwlKLP;gupA&kLOv zUNdpSsZDd5qrFbOlJrQmjYZJ@1m>uymwkzA0i2ZBYalUK26F)~k(Hq-;)SK>{n*2NUNZ9Yr2*9ZztFmN+G@V4*X)XO(3%sn=L*mF-l!{=K$*P#ce ziY{#~O{%S*TKX%#=}WEtp-Vq16EojD=l#}fcn2yaI{9JtT7B>QA1(@4Zaw?#+gdDh zcBV^BvJFq4`vX@;e3I6OKfjDh)^AfZ(-F%QzWY46rZr-k{``8om zidT~>0~8*sZa)88Vg3tyt4IGdVsbtdM{2M|zwUT@K|bi_zp006Vt3qr5UC@=Y__!{o|NF}&$#Y~(l7Ca}X@2jny4Ddn{|hB@%eu5zZQ0s& z>XLSbcuwKEi?6@heSK4vH$QjD%Wc1E=QvqsR~#u^@rG3-$$RAEp13$W&xR+=FWidk zrBdQtWzGw68=ZF7z z?ntR)TX8#Es`lCP%H@tl{gvt4&$s`WTyWwKZ{W6TCcn@6efDd5D);Gp{(SzAY7Zy= zkkYvNUVY=`dD+>;Ha|o94ER+hSRPNDb8+=O|K#O1*RDOQ{IpUgLC#a5xbLv#l56?P zE^j%hr~AP84A&}k3y|8pV%HeS1LutIix#9aTb!>6bvpB8bM51ki>6JR1Wj!ZZ1O)$ zclg**J=HOB<;0^`v=7UFyS})#>bmv%3TwY>hrPt#e6L;FQ~IBwqI{M3t`PCcoXVp5 z%Id!v>*v(OKj5uDSng_S9dzx;BkLWTF4{l7@muP%Z1&+@eXV-8S8Nuv*ExC0{PC@w zuXl64xhO6^z2sbg~^|t{;d4Z!2D%(*!0ggx4*lseV5nk>-~6@ZNBT+va(n@fH4$>+EAYYQjEd zzp&4&_Eq-k*s`ug=(~;q!}*x{UeSxU&#KLH-Lp|%|3}%Udgedz=U-pjw&iYhzz5A{ z_m8@(>eJRKZ@aQ~@uB1m1y=IcUtil&a$fLNJ%8Q#MHQPa+HaaJc&LI%I?OJ0aO2cd z$KFnJIdCHIaDm71t)f>mb{(C!fAQ_CX+o9Lu5<46FMP%E#_>Vdt1Yof=i@KF-RjM0 zb2_`_&hi~!S;`bXxRzMCt<7AQ8TM;x&+NC)eNR>Q2e>Ekvw)V2$6_hRHs|xtU2&ML zi0w{8t>616^Vb&a)69Qs-w`KqiT%t3<9nx!|K8vItNG{k`4xNjZ{4R>;qD@DQubDD zvB8&H2K!f-giS3EeL=#`k+9mor-i>o6vghyk*oYC_`-jE;`yb8`;0#F-ty;f`=UQT za<7DrRrIl{onP$c&tF>b&+?<`n{t7Q=)*fE$6aAF@0+%y{C> zuW0P~&7XK{+aJ@1ul;#5KHj*^d-nUXDSuzad`*tNC;H*+{8kf-Ew3GSwx2gS`+eDj zugR|dQD3)$#sEc6y`Hr@-tzLAW>0DJY0r$$oV{mi+wf2Ajm7IY&yW1azR%jX_&)=O ze@n6L--Dmtty8jbvtD7l_k7-zz-TGEDZ5g4>V>{bJvjABLgB@bvcL;mZ)^A;oYrT3 z;hD-H6zkcxbH{Nb`yYIVy0pD^kx-K$m$KT1snKtfsw3BwyQV)qDe}hRGl#A59lpoc zHj9PsyQaQ%*71_Cx|oT6hx-khFWVk?XPdc=Rln5p!MwvoQ6iJH+hy_W|V6 z+Nxa!EkBSgKx1<;Q~7>ZGCW|a{`fw0{*igMGoN44wUc$vHE4O}xjwl1?Ma3Pqw0_EUEMR+y|P+T zl{4+z=VQ}f#0#zHUC3DE%p}n~iQP2nbmWibb*{-O@i%`}#V@Ql*x2pbIC+m*LFFpJ zY7cp{~d?Y)CDL-p?D6D<7~JK4Ydx-5O@ zpWek8cQW>=RzIq&u60SV{FZlr`L!AMgg&loIriA`W6rEjo4xDS?4EjlGym>ifnKvF zA3La*sJQ>8g!(_Gs(&n(AN^=mo2hmC$7;7-m!6!vSkRTy+P^~iKf`+e?nf!Jd{*Z^ zmgRcyHev3fry{5Gj$bi&`RihM(vGg0pG9oSE8f{H-dcLAZe`-C+nsOp9`5Y^ebFMP zwpM%V<_rY;StU*7Qu`S?4iHm4H)q3ofU0 zSuQp_)%D_TXq)4b)Ucp6jO-gl!e>-0f8gJ4y7u!+om;*ap9}V0H2m&-`Ge9<+xUmy zX4dZVHp+Qs^JmY(Kh|}kGfNlVK0B@3(0se{r$u}GkIHlH34Iv*=}v_7Zqw&)&1RM| zU+G)@qwAq*KF6zHZqFy(Im_{%VcLd2hvohrO^$h49=20w_T=9uPb}J4qq@Mnv&KDl z`Ls`Oe@=T9d`TmGdh!qcb31#TUcZ^v^Z5PzsqTA=<20`wYhG}^Y;oz~x>tAE!zS*K zfBYjb=*Rve_P773&H8plB3JP?Yfo0;Q>Qr!a_!9a!EUqja^6m#q?)|9zvi-y>Y{C} z3!QnEn@^f-dir~&^yHrkiky-Mwq~Fde9V{=VMowel6A>h#T57I^QDWV`MuV&=ZXHe-GEMVsKo*~xFE z*w*u2d$@Dq-QO4geUkIPzW&;p+tD!@McWoWsQ$8VWAeUSsqmle>&|@Q|M_fZ^Mb5< z0y}Em4_s&Z?wh=?map8VYDVW1`TXaukyra><>lJAmdIJpHL@%-deXFp?StX>9kN!x z!x}#Nx0TLb?OPT(?|SH^NddV_Pn}rUaBp{g|ML7(TXO!SKD4cSbvu04!#7`ZjVv9v z%-7UP;eEaR$FH(Qn;(8OYhU>Cn^fP_+-EPOrXH5m%N5tmSDy3qN|DiV_RD)C+mGCm zYhQjN%PMsH>;+ljJSjW3#4O#@S3EuN#Az9Mzpo3$J579*eK+j8^tR@#)W0&Lm|J>5 z;>B}z^|?hn_*K4WcWP=W&k}*^uD4UCUNM^B_)dEri@^y?q(#lXlU%yYCb@^WN~R?@ zuK?pfEPcmEU*R~CJ<6TZCsBkQBN zYgO1&x48e^=`Y`Kwy?Zj;NhN~)$*J#FaK!z=yz=r^QK$M|L!dBKajT5b5Ftc@A|0^ zW36h--uhjeB)sXC@xLn|sjk%JM>OBpSZ5ceFO0eO+Q{qM<;)KTO!rbO@(ug%#yeGA z*0odguYPzty4#lP+4smv;oD~WH#UfURyWs1#awI6wJ&_?@qdw$_;2ZLKY|`So<_+Bv7r1Zk)mwZ@60Zm|3ry?kx@54GIj z=pT*ivvwU_BDOj5*tTpDqsp#1PkL`i+Q^^V@JCQ@;krM*AH_DViaMHa+GF_gwTS;S zEtz);=EcYDqcnfzp6^<8B`^7!&X=1yPMe(z)0N-bd|1~NrmCWsSTHS#=`y)1A zPT9zHmiOqK{ym=$Z8CV>_uzW+vh$BpuH-3SU0d>O6T=pBu}OCC_Gl~0%{{ni{=y%- zeXlRuxxLN$boA_;$_sJPCO$h~R7xm3_+s*x|LCo}y{2dP7~YGzb36H5)2Ew@QqMb| zILOYvaQTaU>Km`C>}-=*_x7Fc#^=tZc^8eIFMN{F+#jV~wS6@aJ(Kf0_v_vBU4Q26 zu@A;yZ@>Qf!rtq#>3hjj*WWIzND{1Hy`AszjJ1Dnd^z#uKf~8>?nnBvKDQ=UR38d^ zy)rw!Q>K`A`ES3wMz;2ULqevN)w!iiXx99G?#|ceJHMJ%ADSon@1XF#xI44&#=UvB zb64VjhQ9R+*2=U0u&F4`%U{2)Z{gZ|bFY{8S=zr@zo<*=Y01_r7hPQo3wf9D+a6gd z;$d{|IYWqe`Lt_dlVaoiWyC)Ao>%GJ8d5z=V_kf}wRKw~>y9Rhtm?CS@iTR5t^c3T z{@?o5H>_8@{!b}e&U9ZrUj^5#p#D$iR|szBzA&M_rFvei#-goXaz#y!d{)u;?z7ul z>tS;IJB2s)!B?*5KCoxF@k8A&VU3=t=CM4z8AbIs<6Zxi1V&xGj2eUHEVsPxsms%i2K&o-|seXDk1qTS-09;;h$Vh-nS$A zsHv|`l-QBoEVdCYL1w~7f`dhI-^qg-8B1^D|G)P zt~kjU$YL<1EPj`tgN$wcqt3AAH?`MZ? zpA_$R&R`RY_>$R=i`pb#L9&V^KfSJulj6z5G^Ima*;aa+}%l&wn14 zpZ)B-P58WsANEckt{<1xI_!6OQQy@sl`1#>GuWl@f8TWTIrCc4RcvhRUSAFv7+8gG zWNd7_1ZpNee*aZA@AC1o3lULkewj~MUA0Hs?Oep_mG9W@2)>W0{kM4St%`TIxVb~; zyM3CPAKsdA^R#HG?Sb1(wy)c-gLN!Z{BuS}j99~vb_{G-!LH30GKMI$gEXX&hd}oq< z=lPi}MIwL0mp|=#v{kPp`(|8@1Vu&uxH zzhBJ#N9s)de@7qNCt-0lM>u9<@P7t*v*erwRa4*ptTXlgoqT+ski|8V8-JXaYyM~G z-_5wKRBo+&{o&7x)*N2pXm)XTvbn=u{tI91Tk~JknOxZVM|JM2wa1 zuK(~qs}J8h{ii3^br)G=NB(Ch*P8FUeEq{&T37S0ySvI(Eb7wkys}4p@$3Fo&e?gh zs^guX+=-1+TGG?-*YeKYTBqOFr`^pcv$Op8^_HvD;bku~H+d!o*xWq-qOkw+y*>N9 zEo>@3nuUhS-}omlk>+@0yZg_#$6vm8t(e|b$67Jj#mIkY?$x#86QA;KY6Kn8#gxSx z|M1_+#9iJ#aZZ*$EkK8MzW)&(TDmCa@U-slE8kxKv}n(<>dLk06?%%3BquyvJke{@ z{DSN$lg!)P`dm^?YSLKJ6{#>awavZH)iT^oq4|v~$6M)kH?6}9 zgPSy1R-RB1yb~C^ulZy4Q7O%kX-~2Q_*NF57X0_&$v?~0!m&#~Fb8dI`%+)$*lGWM zWAo2vo4>~LdtP_{<6H7VFaBTEvhK~UG@H2S{PG{gfnA5sB)ybL<9qY?IgfB< zo2J5`m>-jm%&J`?we;)TxBC|6f7%s!Xi~&IwKos8`CU6!mv?XK{Q5`t1zwt6cxClu z*3xN9b3g66b$iV;^UU45pS;aUfA;pJYe~_M=|>FXqB{PVe(SCjt@*G$W3$7N2Y;5o z*N}Mt{@Z|Kzv*(6+M{dbAll3~~ z8gFyBT7EXO@VVd}%=3=l>e|1$;=IR&b!*ptb>FewoLS>PLwdR2j)SQz%wN~8jXqZ8 z)>@jIzsmIJ&K;8OzpZzb9^=+fkYAho>Ymo6E!wNIYrQ-dnll^jXuWIon6=@-Zz-*N ziyy@bdkEHNW=-2xm!W1@sL->|>alHA*Y*`Zrn_8XiN6)KbnX0EoINof%KCjf&c`iU z^UCVm9UWO(uv9(<~jdZK#y zt?BxX@3Zo|-ltFZF3>m`zvRv%0*n=Vct{CHt*${8MHJC~a>3`?Bu+(^AO97*BzNf#eJ`mc z6KZyDWrAvoTr~YqRO+|R>ea7qO})G4`m8L7Z-nYRkqZgZta)1Y*ZI@9yrcZHTll(l6Cq%AM55n+#4UZd)NN{@M~Km#lkO} zc=${__8__KoqzG$OI@!^R?M=HyjyU_;UtR?!|U_5-txzG-EW_j`nIpVxOips7V+)h zi(*dn*(LBh)Ss_Y45*JZT1@W4SzFMf4v@_>FTZ{ zcxGvARZmKR`-Lk-2W3u$USDy{ou!u7^l>S+}=OEd?oW|%Xb_9GtB?De|G$9tLyz+qxT9Q4*uKjx^42w zy;r9^U1EDl|KooKmDSh2ZM>}~KFv1btNjDp)oU))*em8*^}bJ-cJ@+tjPs~>mrO7%^B zPrj_HKKjq_vW?`D*>@ekd)yR=UitHBzT?|h)q$?X+pbl$cPx)y`Rq1>Wn|f()&Ch{ zCVc(RP`jToE;;$(ebJhTo&CWdWsl6tYusHTHt|DD*#Y}sf`3>=!u^w%-_E=+3t8|t%zP(BJnUUE z^{Swox8VW#US*?2)5^FOuT*CU5@I;1z#r^dX!ZW+dA1DeNxqSK&*VRyI#ap)vkd>a z4W;~&8C7-8ah!ANS(q#T{0g=F8f{@=)Na>#R`i+G`?sRCWyQx>JH2(#^54quN z?%-zZIkD&2>}P7%au2k=a<^EtWXG23$S}*9j5c%ENXy3rX|qX6u6Ve^yW{2~9*37Z zy`I0ErB%7$^5T{m>r`?8&+AM^{>iUh zt9{GAULzbCA*ltsmNEDa(4QJl zz8;q0KeTC&efr9b;QNuUq8fH@s%}>8*|hOlceZ~oTKS)$qV!M3swVE!`d@XQxmh#L z|5Gy0p7}?(ok_psj~M19uQ=D*UZI9u*WLdN_J?*JT=e7pAqQFU;Qk|0MfIoW zEhsy)_+rZYAOAuEeogkwd--(J$?eZlpD4U5ydsmnV%4toM>W=d@!pYo`Q4YlIchiF zKKC=1GiUoETX{6-TAlvwTN&mXe3#vOIZr|GP206+HcZcZ5AG7M`u5T_@^W&>mhhk@ zCqK)`G`=u?W1zkKLs++1`@`-GA=g=v?sIKc9o&B6&b;bhb)wgP@H@00-Y2wWt7ll{ zQn}hAuWs)-89mMN{mUPLVXqgio7VOGkMBcyzKUrVr^jsAK7sMb^Cu<0ES`kLZC(0O zeA{}pl_E?gGH0GB`u`BmT3IhU{pP&v$0CIar@fU~xqp0Hv?u?_{LOb>tu207>#L`1 zvv%*rgU1&X9{l65PCMvF(V2O5{S}kl?yT~t>Q9rAIyKw*UQ@K?G?&DwCtuliymqY= z+xl>C`r+wnyz#y}w$wb36n>h&xWvA-aQm;=Z!`6$S7)o}-q`e^u6^N}ook7iO$TJL7Ber09zsMhLRb);=lk?~hq{>Q=5cke`*y{?>RRQ1+Ie#Lvg@Dvtp z0sg=ne_cgi%E|wDygFi&``XCkE1c~6Y}_8-m(cQmpcU{d^y-1>ftRlAugtk~;Q1@| zQ}tR$_I{{u+p4{O%bk}=n*EVEOAHKuwLbV4627`3`}jTKkHON_wgFLb-j{BDG87hm z!8Jkp^SW-I3twVgW^7q6mTgsY{*vUf6TkQx{+0Z&laDkD_xf12KjHY2{Nla!%GW~|X9_z;{k&Hcw z`PG}>=`CKaGP|yQvTN$=d;HrzD&1bUK4rRgrs5<~6~Pk++4Igczo^yLkNwczU3xHQ z*U`S_)$W;!lLWkvUP)-~Ns)Ixn)W(Q^y9Pj-%_ux*<-%OJ?rTs)+H@-dISm^9KYBE zhAVCEy5h{RE!xA8G0EdS5ry-K4$)@ z{p)I7rQ+6R7M2$;S$lqFZNdt5_umf;en$LyekFK~skY^MJ(C^lYEJJ~bu_ZS&(O#D z^;eg6*w$<(Q0EhEQbJ=<*ZC`XoJZFr7uzv2-nSM1_?N--*pKJYy8`nc?p1gmdgY%_ z|IcLckCtEF|GLimkv(pYWA?*b-py;XZaN(P7yIy^CjWBDpQWcy z=IeJtMW?S?+fZ_CEVv_Q9^A z*BLd=ADm|hhRg*F|c}zHV$?H|e zVx2Pu3`^3umd~N*r)-AZCc!@|)=ONg310W3DOhdos=Kl2KjNIX)QS|?Ydw6w zHP<<)==$`@zHiPRe|l--vt9EZJd2Fn+xe*LbeH%)m(RVxY;<*g>9C3d9^{7Jb>8bKQ9|PAs%Pv;oo>cc@g6LV1zH2W_q(mwe zV`?^ih<@G0_Gr%p{;gU%@AJaKLZ&>Q@w=`hQ1@hY>nw}K4eQK0jxij+*03W+$p2c$ z15VD)Cza0Ws^`_ZQoakvoP6f5ed4dtag$XG-9#8PFFK2GZqm3ae$h>1#mh|4as1&I z>NgkuXXq^3Xa91I`j@OP{jJacN(ainUH-}R+7I);zBm3eWY(`0eOG$_xRdTxWk|5d3I+Ys#C8KJ$OP{Ldiu*gv7lr56KZ{HATYbjS0%lGEES%%-A8US{X&9+_XHqIciH7$kWOE@=`HCeB^C{PtZP zH$@{(Gr7geHdio2uBmQ4v3G0N%VfjOE1ny^Nrc~dEaYmd$%o5X1(M7Bh}%?FS+}zqP|~_x!un< zcuLmxgYVe#NBQ_`)p2eabUw~++ptgd!-Ffjj?aF3T5quRNuU4J)zD?voZs7X8~ItIEq)1J-TS-j*(!J5 z_eV~Dw7uk;`LMeD?;Z80)!SsxFZj=3TNT*F{k~hO>3pTgw>W`q^>=S?3-W$0d;6mM z&u?08m!CEEOxrZ)S9+%Xg@4K)SNA`N^>?1Q{KK^Ex|_cK7QOgcF+V^5^Vzuk_F3!O zmt_7CeZ=|Z`ucCRFO3e|J8ZL70^G%be<` ziawvG_Ji}`-RPQIHk&fd)#tAMe8m6v@`*pBV#6Okn_0Q@&D=FlZ64H|{8_fH>ABPP z>1s+z`Hm+R?QuUA&%4L*LGPwl2FG=0?>n-qal4{j`>y$G3##t7e${2?FY5YWc<8j` zspt1EZOk`$-L03xu+V~Muf{s3jaf@Ym>gMT1YJeKm;5k(Y~N9%|Mb?qjaN;cK0ZBj ze?{}(>HpeIuVmC&T-h>r!#Ur(8y;uRpXt82zBl9MYjLp~;eX%M^;F6`yVf#3?C;no z|1o#6_p+GDMa}w|3b(cX$<2G*_4fYpTRc@JySneZ{dC;y?UB~q4AxioMj!p<*1hQL z%6tBsERGBQXOP~W8hB@xwuAS+mufr0OA_bg9(z2`tnc-GSh{ik3Yp^DzVS3}*(6*ew$nzOdG1|H zo8Kh|qYZs!UhXsJvE3cMF7n>3S7IlN(+g!j>^QY2M?&gbxAt#euZwzX)+O!OUO07< z8qe~NBEhe!HiC*8$XI=GokFa(!_LXa!~=Gn&d-11YmzYesTlhNtA^@t>++9qFU=4) zsyOsYJTmv~pU!1HLN$vHd}0saue;j3R&Vmcv*umv9+pnyu2=J1%ralUz^3s^`u;5K zMSU*QOedM~^f7BQPjGyZjm|HWP3=u_+%oWGQGU@c<0yB{|ueWmVVZn_-~p1!@!rv?7g1;lg}v6 zxgMu`UHqTA)b=wg>Yv`@f9Uw~80R02wOnttcYzncJYKIh`H0me$Gj&FyQi7UA8PJY z{WX8b{?7G6A4NXbg#73gZ`XOM?09K*?UfZ3MSK!}>s#v7y0q6eNxWUq#}S}XxFR(0 z0%P38++5zdYr?L5ToZ7lQBWj7Lx#~+RMi_17mygiN~Tpb$6x+qyy*O@SH+cK7y7EF z*-qhma@E&6aNDIdDSLL;PE`2g`G$S3yUoYjPM1y;EdIJ^U(m<>$<4GL5R1X(*S0Rq(@Bg$}oww`@Jtzb;y|bfw1H zTJuA<{%tsA@>c5C6LYsdHtnB}Ma!;xT*>zkyB03=d||{nd#jDSy>}1F`R-Xi@1dW@ zy6ZkagdcnVU0cYywoCZrju}GstY`CX)!1FnGf3&vF17!B$e9&~(i%Rlu zxBm!?T{iQ#g~RUX(Cbg?L$)u z`OWqCtBiG*eSfGwsI{MdIb~1&`r(sSUpDfuzMKxV|9V3HdARbWH%h;!74-jL zShOuSCs!wA?b0=3ukN}8wy=L%eKGoQ9!HzRWA+uU%s)e0x-8n1T^;Io8M;QS)C9FM z@Q)va<*ne!=RI&v-16XYCyDwO>Yvw%Ka%*(`@FsM;(<;3Lnbu;;lKW~ZtJuAqV3;O z|1-4g@v)F9TQTv}f&UDrEi3Y${<;_ydE&&n*Oro^FD*};ShO;0`iIPGb-9Pc-Ag+Z z7~bhPHJ-nQRd7Yt_YW^`Rr^|rho^#7mNlNg*0oJUt2KaunSt+Wjz%cg0tPk)*;Nx5 z9Re7b8(2Zd$Anw1x%JX2Y%6=$x@W&lHoAT@pR!H#>Wu{R^|rRgG0!$^GN=ir=^wbR#@GvEkmHFJI0kW=-4WH+%K*N6{zuoz*G$J9VjUYiXJU@9XpaJ{MzU zOGd?j7?^K0Fi0>nb+x^FnVZbxpi}p1a{tO-jFEDxA9>feZJD-j zN7%$}mM#%{lXX>$5%aV@nl5iO3BRH@b!pNeiv<7uChd#XiaNjf9zIEOiN~>U1`h+n zIbUSvuT*{+C?qN2FKhI)_%r z^DW&LbybngyY>9{8G`-G>bW$pT{nt6`K@}MNJ-x5_y691dQmNP;ezSy{|r2@CjLl2 za_N28qr-vuyN$ku+3kOR;o9V5rlE1rP@5KJJU;h~z7p-|! zl{#@V&i49+xt1S&A5KILOxZ zQ~Qzs;n(Y#vXcysBnL&S}97R(wx> zh3ZC!ZN6CYGJk!l;9-eB5#Jcf)wI+0y=$-$~d}zDZ_r%64r5VSKt>*kJ{N={~ z_iM&Ft&&-$Ic8ty(tAJpL%NU5dXi?b*m}!vu_tcU)_l*pzWKM^ z7h7YK%u?mH^=b60AGeqj+>77s>-~5BY23w!=f!URXuJR0I@qK2H1ARwrpf#Z-@lF9 zm|35{^1!v}6OSD|cFjE?%<+r;sje@QZ&X_M+zzkle|Y}v58>-C>UlzK-4A?^GC9Ah z!tCXncaC?y>D+9~&yj2Y^D(Ax`GSi<+y-N*-iTeD~d`qgsxpgI+pBBRw>{V9 zB}XdE#XFn&R#5)=1kjrNoT^5l4Txp zN(LgLP#NvA?b&g#p7&yrATj2Z%paGi@GaqTa(LyiXt8%q~hX)*LRq<=-b3ar0q<)k1=XK)8gdNvctUqieqc3{I z?ZMp`R>ig}vV8VZpYO}~=0B2`tnfa3;iTvhwU_JjLL-dc^c3*fRxa9V#l!W5ubFvi z6*%*U>35p)Z;Gg$tNQ854i<6t8IPIu{xdW@sGb;Of7D-WK6iziW6qB&`JGL7_!iq& zwqK}Q66k7ucvt1-jh~MbN=vCoy-2M_nG>xya)5T zfAC*735*uKg>R+;>=0>pb(h4}zyVa3>TPyIQ?^5+lONQm}|eYr*qZ{y~|dWzSV6)n@8GFnFh{ed1k_ zxBL_9z}WOwJFy>&k6v7RbncApn{D;-I$iAFeVq1(H9u2M=<3#~LXHf5%>Nm5Q>`6; zm78u%TzEf=!KUy{|L^w)?ERy(uYB>pdFhbQrtGRS*V*@ktx%5t{o={Lrd_Z3+sa-z z$KTtt{P#q?i$CXFd%C^(QH=ELw)$@Y33Vmmi(c&K{iD1f*DOkQ<%Igvk=M`GpWWPl zXjis8PsXh>9x=f@`Qr-r>W!ntZkf%`?AN+!{P&r3|7+J$t%V_44Gb?>vl|!=NQhSO z@Nkj7RcTSz$`y;Y&PeVpxqau!X?MNr z_ptH5;?1hC50iS{yrtJeW_EYP&HeX|Nq=SQYJ2zYLYCMxVfSagW-kMBPsGl7XE$eJ zf7QnF@(fdMZ_dS^)n=EhFngkRu3X;F|R)p1d4k8W*Q`OMeSNa6N`-IjN1t-7?{ z#;yR3qOZ9xkZCsSS##yG``V^kK2CV($1n4tT29;ih?MWLoeS<|Za;N4*CxdxxcRt^ z-0`l3YhO;Ab*{H`IY{-%PNq3Bd=<9+A|bPEpRWBkF}FCqV7u6x8J?lr*_eBt@AsPQ zzGv(HOS|h2 zB)F&e(f8|VEH8E?&sZbaE_Cvq|L3UyX3-isFQ+~z&w+jbti*U1b2Gw^4fzWzm8Uhrj6M7;Zyyujo4dUqSmk2?8L zzP;o>gGk1wm%DnR7emG89SICsyDsjM|K^wvix-~1`{h4F!|!h|s>Agk{;m3b_|~F| z_#>B6`^CiHeov1(<^O&E-?*7iYD^#LIv-w%6kEn8|nQAA<$={sc%d_TV|+OQ{dQI--|xF=FaVnm*-4mlu&N&pDk-I^?9$d=bn$fXYACkK0Y?7 zr+)KWBgqQ;+K=~dT#WM%%ig+ZPT$89o7~Qm>|Iwry0^Hk)v4My+5X1X-GwvO z2{)?$)_-ODTKnidE~zsgUq_i9|5o3T+m^V?kEQwdq^f3D-KF1W)Xdx$S=``p?D5OK z8y|mM!`(es-|J)V z8aIYnZ5yQ(rUw5_=qzrtJMi(xHJ$HPakC#)XGAf~>RYvDQK0g)K{dPX{udAv*QcF(RU3}ax0=Piw8b7f=xeJgr=;mG`l-v1f6&YkoB@#Ff| zlgAg|t^d06{)bgM@&D{lDL^M>^pubL7N-wy&Abkx^B;{K($iEw@vXx!q%wm5ZKiGIueICc6t=gAfUj>zli?{ZWgLLaR#kuDZ-P1#rY z$^B;it*nJBvpiGft{(Pr=YIJvWBqjXdGF(|^Ny%(QvIll&9xCx|`~pI#gr zv-jxXEt@@Vrm0WLp15asX=z#5w8(e{zW)plG**3^c74{;yL+duDZ1mYD0o=tlc#ON zqI>teXI_#}*4OE64V9T3beM5c?u6qrRr0zllU-}&Km2~sH+PXM-{MCmeZPYaB<5AN z34i&g{40F*#*3a?{_y)Mr7y}b0Z!q$1N3OkZ|qKh?;`-s_8C2jm;o}9S!rMv&-WuFzc zPd&(bSR|?3Ximw&zBXy7t_7?hH-3Mp2X*v$y@8@|E*e`T4bm8x=bp>;M_d<%l^0C*F4dD zesc0g!#dI_u$}rPP?7ki@{8$}`#-e)GdRtQ{?EX1YLTVo ze6Jlp7fva^J+J-Gv+eqOcvt?&dKVjf;5xU|&bxN~1;s4)b{Bv5|9MvQv7ONm-OLp; z^=k}HW|nxW9drA8dA>{Xsrlc0i?!32{by-Zt(M#{j?zI1MX8u#R?nkob zM}qfGpYkhgySQ-X_IS)DZSo@eMK zRV)o|T4&%IKgqoFl*iJDQkKBbEUjG8`#;Jbi+9^tE^}P&Sjc;Hiw;-!lzRv3cSzgX zPc4Z`@7>yKbjdp6NngZ-kI&XUVqaFP^>^R;(tM#~Yt}6IxFhB9rLJGnhxTb++NY4w zz2fB?rQHr5+RU-jz58P>h{h~yk*kb?fRFOpPsk1aoy>q$Jb?WSIu5O@8QwT zTb^OJr@y@X^nBE@t4}XIzWOC~uS5R4vS@^kEn$E2ubma0p8Id+3e$rO$L6emCI01f z{ejj3D~|sREsGzqEnL4%>bMWvxijDN|M=J&|LpoapRfOi_deCEI^`U@`Ae9W-zeu* z|G4AN#~)qVLAzglfo|#3++@+EKCw?UWTX16nI@}QgnXW`^m-fGHb^e>)pBH+?BU>X zWw-DR#>uXMmlmgXxff6Jh}-#uum0nst4IGBbUwY1Z(6x?dhnh}Ma8!* zT|-(IZ*d%dfBCJn#Lqf~?(>WGCD*Nw^Z%g5T%5lCb&>x2_4;P-zR4T@jq1`~xgOhQ zP+0#TJWLLrg5PuaQT@^Gcn+suaZ^qfuHE(dZtlT%_m*F%D?L@Q{+M9KifhwaQ}!o* z6|A0LpLO=i*3d0%8XGRlpUhZ4o}J#nFny`&3O{+wCBRM=uS{L;xdBHHt^3dL zEztdQvd8y&A+;FyWiK|!hfSR=zgGGFpV%ESu8S**uFU%sWBvG(NocoX1+(-bK8ITt zC%(S?5%!;tLxO-f-f|`u$uVRRV=t$ zL08RDNS(X!P=9^k`=6g9>tsKwUhAxtZ>#a^WIgaRS6?Kz>uGg08)#nh(>$qE( zxJ_k1Ol zzxzCvIPik+&gTd@xl3zJ=K9A(S1;C?Uf%X8J67v=xY&2M zNPWHDI~OO}mz+IjV8y)M=3?~M{n6-G{e+kY|DDJ5Ahn;1-{<%5@-^!PUr(qgciwSj zIcrNYNAv2P55qsNsTZkex_#)M+^(zZ98)_BCiZQzd>B&eD;n*)U90Wkr`TV4uba=+ zEnok{I?TMWM7mqs^XuAh-8EV1XJ>4WxbkU*t^N6>iGOTAyuI2w_t@?=e{MZnbFcJZ z-bTfz$#OH~pLc2fExmdR+Cgy@35@YSx;N~)*N^PgvnNR0Ir=SQ!u&mN5B%9_-j?6N97_N}QuSI@P4)v1k{m)`wT+`?b%YW>i3$Mzrb3)Wp&xkCauMrOeXUzO_U2nC^)?Ex;m$PW+$={mU;Q{J)0 zUa-R7`=gHX+d`3@{$IlbYc{!6Cw%`U_56?jgS*N{3pDn%EWJ{5xSaF0j{8R5zr3O+ ze=I&^72oleb>8k2RfZ(#7K=&7{|^6Wc(eA`b^evT&dxoN75lDkwbiKE@ao&^-#6TI z|NdvNzqaG%R^RG_)wwqD+^=qFSSMY}y=>%UUAg*i{an#c^E*tm@AGGU7c-yWEk9jZ zC+~&BM8^kRPf}gOb%i?2!Vz6k_E2~LeH$I&2 z+-lGHIxwp$=9y){!wY|JhVR<&PV3#&Rd+MKpDfQ*yB@!1?~E#`S+|~fN49J<<9KB) z=vo{9P+u(5UZBKu@1{i2JIOg8%(z{SZ{OhmlOV6_PeiV1&R`~QT`Dvv8d)@n2 zzk z##3$n^Vihnu}`aHMbAC=xNZ}B(CB`K`kXv&VXyfURqOSS?Y#3+>UsALwdl7jpL-7` zWd?szS^xCMH3=)X&5zH7f9UCBD*r`YTuW z$9r?X`1AN(d-8YNi|*`HS?}$V@{W^3>J)!8`Sbtqdid7Us`aMWwk)fe@_lyAYF(4g z3A1i$W3a9{+aG-U!-G1}@t*Gd(|z`avTdE-}2(T+(Ge_D>m z>^%NSZKr)_)yzx#b$8WndR6}E;r7mBF%KqBzQDldY1;WH zHt(9OTy}Q3_v+nSG8Ed>gSop3s08z8s*{(^PtN8$w(qRX~loU`Zb!8z-g`g6XT zPOM+Mpl9jA)AQE%v0v)?`S;)Dvd^)e&(0iqKP~LE%;YPxK3A$_s^}|EE4Nhb^E5qg zb4rEriS(ThoKIIQKdt2}BH)-OvOdT%J^5dS$W=GJN*&JwOpDtj#WH1|r0W%|veJ0N z@O9DN^$+JCDioi(>h-ln(bIn}*izSFzvte6hSPVxT-kK_-Mg)fOEV|x+*;jpTDe$u zQn~KG>)Z3!``rHh*E!(o{)dtu4z0O$n&-}4xunaNcP3xhzu@iHMSGu}JFBbAz!1+M z#Ks`kP$v4?PV&Ri>1`(OH{WJfdH>nJ;6FoM*jM$-xAtlu{Kt2BkMxqilg>!ysaU^T z_vD%9{oRKD=B`+^Dcde?jmiBDu_kBsC8Qd@)tn>t^mEUApWmVVZzulU_;=4j`&W2;S?!9oR+raWX->9w z=eBaWv_X=ih3Vd%3HDKKuhO<}otDzNEG4GnN<;KV6|S}QCD+1)W0vq*yS47wn!tH{ z=YIzKD6Mr9SJY>{4EugBeCl((x3jk=&nZ2yV$13KZ)J^ceLnr^&x^xv?#@ZymYJis z@AmTO*f%$yADNVUuIwGdW7}UB!?W&2fRYjP-acQ`yUA|l39A-N@^{dj>@6d3(qxU0 z#g)y5FBqQ8(%k4Z`OsvS+lwbUYgaX&SY#fV94YWLf{p!(RDnW+yR)m^TyXCi>pBNa z!SsW*tlyTFW^H(9=CgB3gF}&V<2%N`*>>D7^tWoS4s9>olzL_f6Z6D5{0024Bg&e#XU^el_hb&`3vVRX zZFlK7xG=)SjqCOTP%viQh-kV`PI@ZwWizw#Rqn5C6#SkW0@E&&)BJ zzgy{hnC#KmOJAec%Kg-Pt8(l06}cW;{;j^b#j0x#Mf7pJvX;_*&CmYhu;-7)g}PT) zSB3{Z@Mo*5kpI>E`oi^<$JKft|7YOey7!9Low`p(brsQ9{lBiQb@#ngle)+=dBuTE zNA8CISe$oe`Nr>y*1oEG5+`zb`K~MXbLVK!X`G+9rTod?dyBWZ`(;#2`u%X*Yk8q9 z+oLSl-&A$2Oq1()eErXtmq(+2^auaQH2)UfC6#?wph?Fx)alH=i6_|WKY!7B*Sah+ zBKmxBy4bNRs^-OpD$~=LC(lo?cpJN|d0E7b+viJD#V)V#%6nSmsdt85-R?l)t-Z`1 zm*40-n)mFO?xKJ3TRo0Uik-pdSKIJ-tH}Ewf3N(K*R#*#VE?t@Kf`hDY4#sOYkZlf zUEad_{P)^|`w#y!Y`QMu|3^(;%;D*l{k&@a-{W5H|1t5;inUhnS6_rpkEieIx|4V= zqNhdH{|A5Z(bVjDN|*jvXMXGTdYQ!j&dsviu4}JvN=5ULU+%4M_FY{ar=nH&q&Cj^ z#p){;L)dSbit#utiFq)AL-N66`-%q>=gs-PPOnr^ox7W!_D=Jllbn@)$?X2YU zIR7)ezRq5}yrTKhVh2-c=fxFCmM*77=70QL-1Fc+!zRoB3|}vXvCU+NT`<9RBKw!M z_CG@Fq>o9zJ^OW6*0ui(t+o%HzaIk)smJXVmAtt2bo!a;9~A1n*bnt{KG)4Gxc@%J z{>KG7-XGeQ+1%~_8LD16|IJwS{g<#;?mC^kDCyJT`ogEo7qTQyZI=;zdE(LH{Xgm# zzxWoE6{TXIvVV))RN0wxEVzn#IA6FgSMFNA?vKugyKSxIm5XmlTi?!@l{&%N?!hY0 zAkG&V%QJt)-g;G(HEYU;lO;)?l>1nfEGApp`gyAG%ufvaBb#cny}WAemyLTbFx@g~ z5$Kc?{r$;PrTC53`h(dO>$h~T7qAyMI+5$d*U{blW0L<%*X--@am!sn*%#4OfpVQo zD|3^X{mTnQ_h?^zTCmDq*k@M8+Q&w|T5ip&tr;{WuZkCM2z=$IvEJ$<|8aTFjD?Rk z&+-j=SaCbY&gb{e$Dhs?TTSoZk}Z3hdF!3UTb!T%XV}-9ekVI@!V^WK?Tfzrd9hyO zBmeRFBCl>gy6hqq!rp(E_f!3S_vH^*3x7>Nq8s;FGfR1M&wqx3y!;fxB}x9TBQJS;+V)a* z#;=WMMQ$i<&q$qAc}i5mCG=$SUsi{sn;2q#&G%YwGFfNhl0Y_=Kj)4!y8pQ7BYf_H z?yB80PN%q8&wuSDAbuidBJ?Ecw9Gud-d*eBc^(8;TKkT?Q!*p%Z{nE#Ou3bC( zpFyYn`|Jz<8T9UaxpG-YKmV)XvX@#?fBrM9-)XaS*01Nk|1+F^eOc+*@1Sd?%@5ae zCvN4NzNXbG)3#cS-EHzf)}A&v)`O)$86I|MRG8W3=dA*F^9C49Zvc|82hi zpTSwQ#r{cc{%_X)zf##;_y4(PzyI~Q|38D1N$A<4zgzyQJ*$}Ma%#Swzt4o-yFVDu zo0=WFznWJsQ?Vp+Qx~^^G_ytQ@#Y4j%7ZU2T%B9?^UWKF#tS~5j><)Td&76*q=m&h z_RZvJc^;{E7PGH&usEKl)gx9lb?Ke+IvbbHnf8W< zw^H?RzWkA;R|Jm-&N}y7ZE1!nzg@(&OWx+%%&~jk$?)&tUSVBODYs+M#*)pZ^BNz% zUb1U;$@EN4hbL;sbU3sx{Ce_v{(-Q(Icb|N*O*$}_3`)FHa9ab!1xOtB8 z?JRB6w0+Z^KioOJ;?Cl;R_=;-=l)4UGWWcTAI{y?dP?^Ui`T*(`FB$$-N=LT#ja6p_by#N92OpKZr+(MyKA>r%n5E|tp-Nkc{jK$o^4TU zx^&!S+pc-l_q8l4x~x|^d%Y275V)z{!+3dev{Z}PuDba$pU&32S|y~RynBktCc_6Q zW}dd}%S}Q{Vx4_1DNl-A`a5i4QRmaMMNb@-TR5EE(cH&*tf;m%Oz@Rd-1Kab%_qJ` zwO$Apx@)HtwB%>j^Fmp(C&ee9FBG*|{nzZc|E5U~pY7T^bI)Vl%PZoi%-4;7rhS2@ z&yU~dLs;L`>=n1NjJszh|JGR{Wx;#-+>65u6V5Go`=sz3|Ftb!x^vwweR`_5tmpO) zei65)&nwP5e>Q4v{&8(uZ~ls?^X0NnE^bj-eOUgMRCB`J2A>jhLM_eTyd<~Ng`E`{y+kMOREi=EozH*mmb_D|myZCSlz zy~CoNj~7q<(fD`U=EQhEv3{|8`z;f0{^2-UeB538BlGXvv_t#m=*sEttt&dX`G;WF zk$=av9|x~pb6-26+r2wKe$IxRe=JE|FR~Y|-KBvPj-WApGyVIo`5rcM&SU%{?fxqM z(>kHU|J437h;NmidwbEPHP=`UzIcD~Uv6CG^7@q1Ac&9cqkQDx5j%Sv6B z*1g-y2aOhp(LWzQ{&MU;!*MCLNlWH98QDIWEdN`#a{ZC;-){StNA{ht_|H(e@vk;> ztX=9G!RmTr*IITKCl2M-x;I%OwG2nATr!k<14;rI);=&>A=fMtGUXNYrc^EIojzuh ztp2`9bYCsU{wmu(_R4hk3;A2wZK4mIvpr`aKM%BIXHmvHxhtlY{HJr7PuQN2;Xf}D zZeP9paP@NAy>GOO6N@EVj)kwOW%^rS&is0QV4U}om*4G#E`52M_r3Pq#5Hch!L4mI zi;hgGKhamWZ|mcEa+~*kV4J<<$E=r2=X+dBW)AOctC?2h<@rF?}XbaiUzi?#hc72P2pj_qVyyu|Bt(enIhMlK8fGjPftT)iu9dFhh)rB`Q9 zo#@B)E%AkW)wFxhyggesZsU0UB4ho^8p|~|YWy$RT`yefG0p7rx`UQ7XBG%K$n>r$ zoc!{AT>H{{FXxMx%$)V!t9YraMZ~|02c}*rI%U-(r?3AL{ixA?AotZuQqbI8&U?>qk}OtNo%&$lqK`9Fiu z@6FTt`ybd}?mxCc^~3S*8ey6KhurVq+L|!cZ+bslS6=?p+5OAqdqsBTZ2QROdm}AJ zTdswV;ru#lxguZn!@A*~+b+-do>kT!Jt=M4EA6Q_-aYwtKF;vH&kx@<8Q*RNzxaH| zMow{3`mAr#TR%Oil=yc3p2Shnm2sVMb6u5p9c#S{xzz5aN}t3m&)6VS*zsa21xGHO zdNpyuX=51%ThB$Cm+qgG{%LLCruuG~$Jf~xe7A2fJse+g?{nntTH}M;=I6|i@vK!i zB3S)=|IgQP_4SW`zb*UbtA1dc)4x*d)GMhH+nPWAQ~p=EXny{(RdYYjH|#tax7hCI z@8hq3Fou`dGet+n)n{d!^RLc2dvaD?ApdXf+kd{Sz9;viGCQ&M@5Xhpc3anOS9|s# zRoSj%FZ<7zuJP-3?FQYJ4$3bpp7;2&OqAin{&25v0S0PXZyDQ4AELEzr6n$ z7+1LF-$g0aZNVOw-mQ=R{aKK!KKNI-aqUN~Lpp1xF1bAaz2y!2m1|z^apd~Q1(bbiy>22Pc4@BLLYR3=xpRnB^BpdKl^43V zb=yQq7j~ih7C)0@zx~?2{HN);+xN5fneCtM{$YmU?Ugsn*BJc0qkYHy@3a35c3;Bx z-Huk>QV|#N_u29I4{}%ETCumksj*z9I<=>Qd8SJKUgh#M)7SgNKWv-%P*%#;+^0K> zf0k$T^qIdmgO9<`&H|OJr{=qU-s=CBam77X3umb}f)7LHJZ9yukGU2yOR+ZetLT;W z$~!XtEZpMS>htavqx1cp&+l!DcDBD2bbs0MYMD>#4sUU7Snz~_OK$%By`oAl1*h(c zUb%FyxQ24Q&5ECH6^tQ_S1upx7d~7ndS}wT?g?VfmJAO{m@_1pCm%1KRh_ZHJF_&m z?9)BnlWcBt8*CUDq#YP|()*rV42Y;G;&A;#>5k~g=w7D}s*~H>|He7W@cfzc zcvj%U7l&@>-+phH*vRqseqvvpU&Z794C~K|X2ptW>ZkAWo;{=Y(5iD2H?>aBP!K-m zc$)ab9bjm!>MB?z{%2w1%Sde-`% z;0q~TC(U*1QoY}$@^46Zz;Kw$z>R&YsBlG@+4jvXuf%uf+%R#OabAI8C#yjLyG-@h zWhWngoe=pvaPQrWVo5dQa|WEOh7&Ags=q#)7R%wQyJqc*dq1Zgc$0s3p1-8RLdoN@ z_UpW+<#KVK&hF4(r@lM#YRJqorrDkf>#j;zs?9H4aQogZ|NPp*!qR7t`4s~$PMjlp zYXMKnw?&T~w(s3M?cS$Hj~>o*RG8Vd`q+U4kW{V4N|Uonx(1fw;^9Wy*`imLZwS2T z_(rSN+`HR1{K$86p?%E0#cJOb(wXg=)%%OXO5PgjnyH1(s(eyBebyu2clR3RJKhM4 z60KTCcwKl%zW(E;U*`@kGTM7HvG3F>BhPOy?4=f0*2{=rE8`WM`tCGOS!m!S|Lu-H zuEnigaXi;_ZD+@c@2w58Q$tcF|KdN^wL3fuI<~L1^y~6l!d3kDRy>}QCpSOt!8g;h zKaL-^|1cr*rS9UNQfp^dZY%ENNIZ9`mbd7h`G@&`RCeVQN1psvn|5^1Yw3vx!)4y8 zU%2Snd;ga5jps%hPaaFRum7UFap{Np&i0Oy=iXC3nR>f@TKB+4H1Q6z8ys`?ziT>ZJ!Fbl0Y}c~t&9JyqdX;g)};y6-01 z+%voMwPXi3>((_FPV(DZ7A=wYGn;>I{oPFCvyaZ&hkG|{h>kwZGvV{+=?B>__b&gu zX#47_ZJ;%N@TTCh6R%}*m1VS=j{N3I*`#A=v(CXK=!}1Y<|dB`M?Rf^a4bzic{TY? zrsl12`g+cK{-4U;!g!v==2Brx*Pfhj9K3H*famiJ3+A8KnLo@kE?t~8>(5N1tY1Bw zo2uThbJ@=oT^xI~iWO2vK}gXDa*A8*RR1nr^XAaIuG1&ek6T1sKV$jF&tB^DS});M zH9<`>sR}ouUo0;FY}#5dKl7uMdcchH4~$)_|5oYdA<{y~R)hPgVL?o*n~wB6UcvrB zMgGuE`OazE!|Pc0M#t4`n&j0`baCZL=AZvWkJf~5-o5un`HH0rbdqoCtYFud+MvV! zbpD~P8cor2b0)sgYGr!+BBcKn$6xKZlO^v3D}q=BwO@WOJGC$J&5mPsi+_bZop9-o z|3mK5h-tA#-!6u@xl3~#vs?VLOS|*q@sblaE>wRFJ;!LNTc8f)#SsFU)&b?vRs$!3GuaQa#%9gV4RAto-(n!9w zcA|2=LXov*+}xk54DGh(KZ-0@-C`!-zSYmAUwKZ=VHv%9bAK+{^}hMOfc)V&wQO0h zhkdhGdUi(Mwu@Q5_@6ERe}<@C{VYE^URo``xlZQqq^BE>%-{8Q_Mhx>_r1SW{sqr* zReXNq*TwkYuj^yhE$Uidn96(1Q#z(cEbc!;;Dvvto$G)2dwf)h*uPD#=<3a@n?HZ=vd85r-l+gL=l>uZrrXu9R>4`d;GH_BgG%YredB*YCAGC#CGO z^q+sCYvo&OTpxE%dwPXcSwSk%|7 z)}^st>c`}d>+ZV?y|j~@WIfk;>ZCU{7Z25n|33SRzv_`q;Nu-@w%>f|v+2&4%w4N$ zZGBU+tWF)Cv9=@AO8kw+`nh@U-sQi0x_kFL&v&Z1mpqkIXZ)V^*>c&mN9nUNr95Sx z&IC;#T5su#ys=qs$;E)nnjl9nKPv32b!>LI;ek5^_Et^>oF+F_1l`WvJk&FXXO>pd z##?bo{(4V^SGZaGKlYsT@>t0Hs1D;hw^PjzoV~EAr~gn_S7y!2S^m1pJaw;l7`@oo zjQn@5RO&y~8(eK?-`VH6XnjW5_3yK`gzd^*9DB!iih$Ph3ftlrw!cHdjgIoDKCNA4 zCHM2&s;IM9r%s#oUHrS!J&6;EYLz@z$ya^{#(n*;`(bI{obxNb6OMY=v{m} z;fbO0+~bY6w*Aq4aC^P@9m#i3%x?L(q=t8@o;YXQW_^`c`Eb?xBdhm*H{Lh@VaK=0 zGU>v1pD;K7z54t=!*$nc6N4q3n@kV9@(=P~X?wQXZT|A8%&@?lHx5jQK4!3@`RjMP zuWx?EO-)|(^1{nsxogU|y+;ey?lN+E8*XE&~NLG@=oSMXPn`pYZG3C#txc z6>;>m%S>Was5N2@*s{Ot*SAk99na1=3Ql-D$w7sGnZp`QQ;GHK-!1Ss|4aI(oAccc zmd7s-fNrTQg%rRdfv%#ntC<~*?jAmP{_Gon$zO8qA8c2*e4IADc+o4)$GVdBmvoQm zFOXGl-}$Tm^QAqXk4O35YhJ0|w?*YYgV)yBUA;RV#2mac|69ZRe>!4u9Qo%V%WpS$@2Z>Sy8OfH!s180)xpXNJ_*iWUjJ!LeR=X(mic!1 zw{JwkS6@$zngTZ`$V#W<#9^(&EOL)#BVYGz>uTn4DFU>?)I#f2l|WUOwz+>=)1&C{ z^FfZbCtm;Tkw3R1p6SFdxvi@!+#bw{_Q=cWt{2g%;$JR5bG_}4YULl*vDaz@3!-zE zw7oj8@%OcNRs0L>S1wvzu{h>Z39o71R!>Q9v(qJ=-rg$hix%~H+Re4i4nH1lC)a%V z%|4}gBgyC#LATarmDU^wo|peL-TQdihnF8-AAV%<=Z0GDgd<&1;gdJ?@RiqUthfDW ze>gf{)IWF4>bozcx*ep9CblO05s2U`vP!o6^0Mo*ZT8`fH+>hyyf(_Z7|F5Sy<8#i zsh5X_YUs~>fydQfUKV}6t9MpKSw7#ky=q(4CdyP#`c%ZwmQ>WOwf^jDk?{d+DnT01MXk=9GGPldTrK;RIU#v%w28LON#gduHLk9P4*0( z>iepGn>> zoy^YC_Ake(mfh~rnv;)Y>yPza`zO)4<*%`7e8`R%WBy-h_U9kQ+?l8PQN8c(+C9qL zFK4-xom>;ECVBj?QT*#aD^`bH6O4Y?e_r8*$AS0-Z~wer@Zo*)mi>IqCLY(t&!5>o zdG004_viBu?TY8AkoWkoBXfb<&MPXvyuAX$I{z6T( z%;APl&YI6Rr>XsCxT$^ICh$Ll5P$zIz9oD&wn|LXv`(3Gto~b+#=32L`#?p|^BV8R z^Z9=GDc#z(RHW#!5{vcI{6&5BR%fr8{xx=Y%GaA1za{(h`~#uYvu{@i{=Ugze^+My zsjd(9@_(|g*!VKP+@(|W@>X$*$J71Q247TaFTSz2`n+E7$GhjDy=!wq-{$u76o20F z$BOIF{%a{agJl$sOTH>TxoFR#om{6uCE=W|*sDcsk9#&ASTKjnEq9e4=YezVYF)D; z9&O#}FzI-aIO_(!OSQHy7fcJD5VolI$&SWuw+~_`B$W%~_FW0q%aeMRtgNLk!POAhFIz8zqBa4~B2+{ItFEptxSWI*>twJ+S)X1hGYh`!Qz)YmyRRW_ z!Jb`x$4*tR>{D*#-hP4Siuv@w8#nz+4UZHoEc?YcOLJS$+M{#AwfUz$Uol}Jv%e~rcT+hs zsp3v*gs8o9;CYsEvrmpjZC#-nud162vMe9;y}B1&ImISup?mNh!6uD$-tk+n=9_SL z$7(%)^Q$M=-Nq;7hQ#XLS4YZ}RsJ*Vjf-67xVoP|TTbNiyWV}P7soBxwkG9)9giN|?q9{gk^p%c;pj@-Tz{caIbHfv%++xBa+!;$Q-& z+|*YH=4&l)1(k$}C%%{jue`D)BZ_YUUlIR8F| zsjhi%;FYa&jon+$u8hAOr676c66^U3a$7<@O&#~A7>eb8+hPCx#Ob<(a83Iov!hqm z^z&zy-#T*h*o(!Q62CUeU)u5VpH}2OuiQU|OR}uq{5+7Mq$yzc&i%`xnNMGdTk;)K z`StcMoBt~<`%M-98RE|0%GWiKPW#8fK2y(b|KGyEZu?F){6m_NVHve_gkKEYEV|#WpeVOLt@M-#cB;6Lhvz?d>Vg z6Gw~wS^cPA_M)n6<_h2P+LiZDue^O!=lf>yJ3W4G_hy)ct}MA-*0LmUiSLb7|Jtuq zhnKGYq3(ROWcwyDgQ)p?Wv+c}skl+1(El{9ckK`E!?(^&R&D=Qx_9BZ7t6PG20kjh z*vQn?cJ1ZcC%;Y4OB~?wD}2HEjz9T_T-B3`;v?m^!j{<>J}SFiu_r1{es{^~Gu4&m)G9KV^0&eSrz z$~KPq$Ts!iX|_FA^Od)?l-#LHx9hT+{AJOKeH(;II*y3WHD8g*=J87}yRZMz)z-Hc zw_o~_>-EjSxW!eb>ap66j5?FdS-qEUXWSB7ztUr|MVF~e)nm&=JMRm{SM==Ncwx(p zy35*cvS*&=cyQcJp&??$;D5^N;%1CqIgo=RUv8J9);_T{~+}Kiw{GRNx?OpME*# zV$ew;UrA39qj?LiOb^=-ciS_!l*^Z;Is4dS#a7Sq%>3GH-L-cN&*y8st#}f%n!nJs zI8OP)+jXm##$?{=ae8#D@UxhO*4^3mhYtU~KYwlIx{V9se%NugZVAe&6QA;K)`dBz zGymO^cl^)r`s;e5eV0t*+jUibr4(n)6P~!;KY8w@o4>ZC@Aw8~_zao15s{ypFED`9zyAMciw zvd-so`*rlmD=Cv+j}Z3VlP}926?OW4?Aou;C9~3-b#D7?E!>vwW zUd!7?H_seteiibcfpgvd!}m1yefYh~rRKEXw>K6485({`y*K&&)8xa(W9$zrYlq>4LciW>-#r;mX+~cmVx#IB~uIK(a?O9h~S24TI|M0Kt zzM0j$PpTK?Rj+)dv2M~d;x{|x;Ke_o&b zXLZBsMOD~Nh9kFXZ%FTDF0%jlR3u#X(fQ7M5+B-r;OGA=Ua>D(Ke4Zo}HbT6NJ`ox)c$5m%MnRQlWqiUzhSGBHGvq@crZd_|Z zTJu`^cqNzISu&MHOxD4sODntM?SqOV!atuBJx>$2Yqrf?71?-acdg}$#LA`NpA$bd zK6HCCr}|f*XKA_VoyMigPbES_88p5y1gsJ%^xM9z?`>&g-UHqe%X3BlkV%FRJkMG^SJA3E1e6`j@{yUTW!&P*3Y;M^V#!Nz;ZNfK!?x%3(d%aWq(9!9zRu*t zgxF)(%1=o@a)Jo-CGs@kJNe}neXse`(8Pl*MaS4(_^h4G9O>Rbxr;e z^f-Lug-M5l)LwnkROL2IdaF`>J5hme$Ll$v`TRe2M{T;GtD|)3-Pha4Kg&sP%~3v* zlF;*}%8Xs%(PWR0%eJO3ym9kZ#`S%QyXSc7hHTE+_ho;0qpjSJ#0xV2UApGx3$Zno z?f(8oF;(KW_ROxnVl%m9b{oq*s9ErjdHv&AJD1m(fA~A+(4%*jTea6Mi@wP-=dV>= z|gjD+FzHUa_Alh?f7LIH&)x z>FHZp`z$W#9-aEG`R|4aPlMiZ9>^3aQIVN*T|IU6AIA?*H*DDd;r_Q*j#Io}XHN~E z%I)oUkZ zR*Xmg;}CT+Jj0aGnyD@?+#n!dhz<)X(KQEx*AK2O_~ zl2ZkCZwO^pncTptY_v$M4019_VC1J)yviMGMJ8358TntRs-DBYwoqz%-?IICYqbxV z&A%1CxP7OMBmYj@{|x7&9+dib1iv!>lyQ%1)4!|x=*pr(d zsjhXY=P2iwUaR_fOq5f^y1sMB^C-go6REf&YxEJCv>FwQ(yDSb#=#d z!+t-yd!zmEG_x>+Tib5GotZPab%Nlh%|iUAZ=|bjUZbo3^vCu}9XrX(Zyt1AJ(R^{ zvq9xy3QNtJBTa@k(r0O}n-Y8bz(pUP4HJ7N&p&SK{)qqD`g{Akw`gm=HO&tBDAune zQkfu~9%@^(_di4YTC;!tQa`GWJv$$@ssE^z%8H98O*g&S8Dd*B|LDKUMU&3`e7OB! z#GP#stThk!cNFj!&;KEsRH1)-a;8;uM!UzA^-s5_nLX0jmndkHC+i-6ZRdN&KW?+# zE=3<+?sUym^Q~;}hAtla3I3<|$hk*pm0rn{(|vSt^9#}Qp95p9ew;sYoqKjZ^Sz!g zpO8G|B|{gyS{yYI-gU%Ae{N5q{=@Fxnm1w`+c0> zE&b-#%6{j|hQCj|U2p9#ewmLKeivRI^U9Z_KH)rnf>nF zEvXXKzv@hnI{atI=$skS-t#EpM&e?-M@uf$IH#|_ZS1^Ddq&^w?npP4&fQJ6wlCI2 z&5gPe@~wP(RqcTtn|98+mHJXeFR=ZQmPh)mTaGG1dxH3SdK~{8wUk|)d*W#H6`8=; zWs^U)AN%EZH8$q%G5e;sEc4gLU)!$pCwqP51MbU4{SRk4*X_-E61DeD{HN;$UkYbw zZ@aWVWMv0k2@2B-z^sj z&t3Yb;zR1ct2?K?J9s|OMrFmHaMQh+AK9IE9GLoSd-8tT z_8);!-|NNqaA(Y(9p4t*^R~^r_!rks^PWHUTu)wU{5!F<|I^Cou#JxY_Q)?#>pHaO z?*bdyJo}4}ujKXr7GRbC@lkZX{gGTD`>)x>n*SNnZ0d~z?Dl5tFY&nVdQ|lM-G7(F z78h*(@Z{`^w4L0)7_zPMJ3sRLv`U`u^r`VrtHV8|_y^n7UtI~eul()uqc2X^WUtko z`mOt%{~S#_|4~1<{;S!vjem>oUt3Tw5F@`+B>U^hMJr#*OW6e1KYFXU)%wTd%SSJ9 zE^JchI_1Y#J#Xjhh+ortw$A?QZ1PIE%XNyo6f4Wh6aPZb{{5TWztUJ0FH@5LW`DlWMhroGyrx@~=_OvR^}e{OvIdtJKy z^SbPZeT%=nXFJSoma?U+rQh*NUHg|sd+x0YPX!G+M9j?%(>TJfe6nX}VyoQYKbfm{ zOrF5-HsDB}l)KR?r4k1|&+jI|d3w)xX+QUfP&m|g%C=Q&)q2OCONaA$_r2F-nr4}N z=dZ6QuP%6qE%!(Ep{o5|@AsWu7A?;FH|6Y+!v3(lrp41L7Q|lK5m3dqzN9w&0RXrNltA>{hP&qJ_o$46a2{Ef9mY2&M*3W^ISJr?A#^)w!YAR`RCWJRW%0g zd9#ycD|IS$YCYeB?l~~CGVO|d`RybN=N~Pr`9an5tG@1)cDY<*zwG$(mtlobx7{o} zjv3W@zN!rnJ$r53^3J7~@>|%93|T%Z z`koPuSjpO-_E+29@7j9lBj#4VzYc!tzvC+;A|$|=*mQi3fqmnmH9sbYPj|Z%_voJK zq?$in2c|Obv{3G;TM}|l?ONK!E%&XinM`^(Bf&y7w1_de$>JpQt3%5^%5AipoX@bO z&wO4a)4vZwt9CL>ZnEAW-z%DT?PIa*+Otnr-xs};-2cSl^`U9%OEVlxC!ldg9rwM~qo4 z@^ugA+O^ASe)~Rc`@}8Nt{a@|Du`0&Vf@uqBR*}rfyuS+2B(Fdi1!}WI3DLJGVS`c z>9hZxE@k;`k+0=&yKcuz*R5~YzSms_9r4#BHaxwbS2oH1G~`O4k#Ufcf7o;AyMH>TWEzLmo8 z-d0@l^;`GbZ;mX|-+J(w_@_fZo2~yd?C$ycq|lY=-R+rID0x(6Pe)oLL|6AX8@d9c6i`J2_c_3Pem zOUs#Dyl)yUoYL5P#>m0mC;fFqah5>bt8lTCg1RXOqny~x4NfxrHJrazv}%12biF`; zcO{F<%E`t34;**ySB$J$-)WL7$em@?yN9Kov1aFghK9au<%o}Ntqa_uL$2!Pc1Yg1 z#Ps~d;>kY)L!*N#rBnA<+8Wj6U#zNA2{*bgu+G%nup<5N_g$SfoKNDl^w0cf__}&( zerLJV<{5j`AI^0TRbP5$^%KYc47KwAwu;=}d_Sfkb#lUar@~*F|8g?-->5Uc?qIcT zr`S|xnc20rML)jW{m*dS2%r(g>x)MZLe0p zaD4t@(Jrf(+DI2$FI_3)0BgQ4KHjDqeL0}v-?AO@-Fqs$&hFp6^0xe+ii_dyjCA(#R6C>X zXDpg=%|Ir>>hZFZlI~AO!NAWf=veQvai>C_};#CuD^%(UE^c3dV>C*RGL${FrwDq|In748o~9oQ+$_J z{0{HjC9;btf&Z~pKqY3nw}LvB0dgJ0F26+QlX+otU4 zVz=(jeRVe~^W#a?rE~l{-su%MZ8mttZWFG*?$*h5G11v&<*Q3GmmgP|Hm%T3dB@I! zTxSlikog!9b+l*K)uYQcCRgbtZ0hpNU8_{%Zt_Dha%NS7@lnyO57n=zeb_d)c(q2X zTZNeC6}1a%+zR_{{Ht8Mp3`ojdFz(fb0eKz86Vi>x6rIL$+)NN_~)}(KeA$-9v_}n zdu-K0{ils_DWSInJ6`tPe>_V&YuoCuu5)4c*K7zplwSD#*8V@C1s@s&A({&?%14vQ@pr^@s_ov-|5(U$BT>;8e}v@{lVox4)fUHv1)z0WL? zP0n1>{a4@d`#-O1KI&(_m3{u(Yxm!LYnq?z>?!zn&bsS~P51+C-@3w;H+*iT?G*TX z^^LOh-Z}LGi}olVmuKytsr$TL{B+^8NynTU|1&V!8vd%a3ZEH!<4#oke}?p|3A$|y z>K>{K#8fXXVZZV5%gdOh^Y)h}u2|RkXm#o3t;R>fC6DgCzH7Uy-SICkT`SH$oqKM6 zsp{INd!~ICoMTg@87AMjYO|y6p0y$uw-o7zCN^m=N=ZptvqbR5#ek5KDZz_-pG{Tj zm3&ep8tUe$zJO_J=-flArlnah@Ods}*vuN>X~UqgR=S@*+w3|&%dXTL6Zx(&iCNgq zWLk4_*}bF6Zr>*71vj^-Ev@!$DO$@iQ$~KuMehZf!moba2`h}e6SL{vBo9Tm^iqe& zDU9blB;=k1e>rWbuYb*S+qB0CdLP}EY-W9)FVpkoe9o0=X3we@*mSCe{G4aPGKoFU z@^zzY>GgR`+v~doGg6;i`ttjV_ym;~|B{?e#h*SHPgXQ`NCgpU+=+#jfA6vmCz_7My!`IBVL?p4^iP&2yM+&$$I;3eEU) zxiD|z&uqoCBm;>Dg*{)tEn0i_!>`XiTldZK{>%1t(dB)&pQ@f&^GNL2&EO1%Ip0+J zo+PX^@_%^qO~Rf^^$l|uHmamVMr>L=_15Vg=IeLrvb@eOJ(NvY|s)O!pS9fp! zMN7ADoA)lY?G~rn)IL$=pd<4aEJ+J-D^t+=2wTjIng_b#yG`XB$-F>kl6 zTvu`I>6$;%sk<0%r(VqqS!4L??fH8q;rm|vlfC>UTVri*?!qpKe{TZy1mt7N?VhXc z4L&T-Z?czR>DA-E>K0j=3-h$vzgeHZKI#iU=O6t`HMUl-qFVY_zjmLtcbD`w`EMd0 z_MhvzzU~m@eRJV1x9UO>fUL>_Lx`Z75~axOZYEbJ9?A*v8>5F^IKFD`(;Ef}Dpg35Exr^n56cSZliW+RR7Y74H{z z&MPapy}v!H=i`le#_uEIYS&(G-{UM(zCqFay~61pqiHeef2Y}`FEA`TMNhfRd<%&30IwU z(ZySRj?9AvnmZ=>$9#YCbyZd7hJBZFCcO{cDL%oEa|@H|#Pqx8*v=lGq$aDRjPhka4gfC*xpT zfZ^liD$HL+7cb77CX}`7iGOO|McqS;(r#RhTjbas&&FLa+0B`DEosuJyN{#4g}QZ! z28bjxFL;%)VoS8fbpPv8w{Gd%X}@k{6R2r*9>`7=H_008+f|- ziO~X`B`Li7=H2+Zs;YF6exqL6_U-;#R_wU(YQNQ=Gklk?)GU?>49u**S$^@Q(7LsS zx@UOyg|7PP_}Jnw6BFC*hT`)I<&Wkst51zPyX21Zs`EAklkiJu5(Qc{N+0b3xpam7s3J@}B zh?`tIuVU?j%H;WyeVM#zT){j?FhNdClQYvr7s-Kk2ic@w0x_!>MkF}Pdy7fPC1w&tjf@x$G_x4tbd{=MSgp09t77CnDtwtCIR{zG*l`}W0MFR$Ox{`!3UnLy8P z-lr%2{#nNFRQ!vF|I94=hteP0b}(N1pkiUk z3+eY!nVMT3UpP_uuO#H-%WW@qM@X{AuKiyBe&X?$QM2L?zvq4F|7>l;&iR(cvGYwk zzdrB3v~^2=+dk2YPd-ofjb1(H$(=ZP%ab7Su5GJUEDT{-ynvZ$S8Av$SC^2QT~Sd{+QtEC=GQ(h9*m;B*$NPp(h%Q05Jv>wmD zaD9Ho+urAWe*~8u{p_JH_wq`a3-jb(7eoFermnh|k+>!D&s?W_70dedx2AAC*kt%c z*>wA3Zr!lY&b*7xKY1ng`KU={?P`T^)`|TKCD+YAYBp=>y*~Sm>uhGfSk@PE`fWJl z!Ib0|uA;l8e@dG!`2i@o*sW4#(u ztW$qvf8domyRD~wJa@3#Hcc=0vU%8o4GcxE@6<<~y{6N9b^DBFBli<>i!XJ3(r=D4 zmlOHue|e>iajh^-uhMY~86CZ!5Z6{$-?GIjefL!&CE|#Grx{~g>aX;*Nv|*8x~u+; zId1FTdy9FxB|1N<9WXev-{|2{ugx#6=V>w&e&nFK5kf`@$>pBhl>Ui;bQt(^effZdjl+ zRc2fCI9~s(rEEVZ+3wm)9;yaJKO{ zBG@CuTE*JZt2Z}SnMv}In?g^*1nupcCobK%ed0l$kJJ#zV#Ji&IS+@U?sLqAOK zFj?Wd$H0Bv%g%qxtKaNutz|A}zC3@ydi&WHiEi?~*kR zt-Mun-(%L3sJNfMVxK#+en~Upzw#kto&2-j59Qh`e+guz6dW)zP~OP4OXG>BfT#3o z%_l~4ntxpM-?3}ntFUR0(x-^lvNCm*$=8;YXG(0|m{c_>V|7{8 zqMI+vuf^Z|HfigvR&%4uc_~>EX5xI}tC~;Enq|FFw&&WRTSAv6&g^^pijkdpPIBLa zf6e@nO}B0Af`raUpLt$zJa5PGhpxGiKVH2%rYqvQ^yRW#<)VWHkCNW>?k+NaZxAv~ zy}|Id%+(iPeo#*{$Fqf}h7tOcVgz1Xz+Kpm| zTpY?j_3ti{pFQ!je0^0`R+0MD-!o>P?p^e}{7k!zTnZ(jIO@78rA z$~&#G?vj_ga(e>jj3@6TPo7`yrxj`Au6)yU#pN}Vj2XAH)|^?DcmG)YioD7{#r~7S zGIejsZ@Rr|OOl=W6AerGqpw|^{@Ls-)cAM&()BZ!j!p6S{G%nArSALt0?ThNugx^? zP@VnOa?{rMEzzo*bMGHoXW$mSp8G#TO@Ga{+`wnzl1+loSFKJim##{lpZGH{uJM6F zjsFMpBfpq?FWs0nkMYWvdxrlRn8N;G{}V!KyA#^&4D8Zl4T7 z+|q2BdM0~`Y&^g+-$7%&^F=$Mio-{yU2FVu_3}ao_gnhXiK*(!Pk!IqySm=zgW1-) z+h^p{I)C1^y#1?m^{o81_j!5CN{?h}@^)Grmw3Ea^o5oEf%kkbmqrKgo*y>f;?~SA zfm(~x<`?AGX=R_*Wp~{vCMfy!Z(^U=rRz^!?cWN{sh>A>?%LP4)Y_{|R&zZ6$hYr~ zZtMKJ&i`)nycG#wnlVrK@|P6{wrp%VX}_C`HTg&1{F!!Nw=S)*nIAM!?~l(#m4tWw zK3k-d{J(enKK|#~(#v*Iuk4iP**YH&*SM&8U#vU*)9s@#_w>d52d(@ODO{oUjGB{YVC{(2?g==;0vKu!yo35R@ zh37C&QfJ7-7X=ziXZvs3(Wx}=+`WhkPXix?an&};1e(oGE|e)epPH*v$m6zNXoKS$ zjz^teZgWrRRo}F2YLawvq*54%Et|V6t4Qy%2-8)!w)Sq^?sjLR#PkH4Pi%(s{GU6& zoF){#boY(hCW3R8_zF!8eq-3^GAp&^!f~E$X6HrXUtL@jwDOC@qpr5^zWj;BqD|4! zXN_9Z3=19h4kPOsmxSQ*Ix=U31QVS(zUD^LbO_&U<*gSDL$Jui2SH-r7-r&KzknoZH>l zR9ozt^|$N#+xqO^Cia=ka^=&^A4~kcTNO5A)7I$CWx5N+bqvx3d!HoBSFiWFn{<;Yl*_h~WboT>-G zYjbjSnz_9r6!?4Zg(Mn(J!*IB)U}h5xw#7)Bc8SdwzB*a2|Avd&3pKcYc7|+`u&N? zjI&==&5FoboBM21M()DCNWYYl9=77^ldLYXPF=fZlI`9t27B7|E*=xv`2Bd?vaWf{ zSAs4HC!q!1WyLc~Tk6)QWZl>Cd-A326q^)ewD^q9URGakt37Gg$GW*oJgaWKTbi}- zevVkdL~RzCl*c+6-@dKS=!zGJJmj)|OW`DOmC5ffb%`Z>G~GLQ zinpM;)(Xjb0fYT(Q`NgoeZ$nYec0b}=k`X6c^_Q@(n}u)xa6h=& zbaTnh{Y)JG`|f=SIdtk(8E>H;`*+W~zj*l1o0|Voiqp7uB440l;)9m2TQ5&f_?7qm z*Om5z^YpHt$QSyt@$CcKoG(*9O@9&dDP!fM%WEG8Nczlp^QZAkT=S>3y~+pX$z0s< zA@kyC^U8Jh6;FPBzWibBUh#*|TfYd&UH9sWJg!zf|K17zbD`NwtqF8>)iuOHRT4puL*O3_Z` zU-tO>{($>)E1xz0XApkh{p;TDESDeHZl4$btN&QnB?SY8p5}w=-ZUI&GU8|3 z%hTY!t3@yy6wepx;A*qhGe(mQ>2dp^F6sFK@bDw`a2No9-K<(HFJ zJuY+Exa*y}d|bu1HM#OJ|1Mtr+H2&UoNqVRXdQR@ncibh3%@@9v}k*F8EjwWs#%>% zH$5c(zT!XI@NetM8Xu?s4E$ehc&2jPpS-rPUM$JzkI^*;Nv)3m43k$bnxFK{*Y4(x zge_kFPqY_Yd}c1C+H!o&w2K@6gh;*oQn~Nq!k3@_Go1I%xc@mIW3^}*#y;41%RlKq z5=vHg*7z?vPp||DCVcB+KsMSGi+5-zNKh znsMjQe#PIftmUJ$H~%pHC~D)ls8={-4zJfC`-yk=Uw5+p>t26uyN%$Zk}EoPhbtAt z%}!1@Q}>JSKSSWXb(vO^PUj`fY^a!S-T7+qUeQ}G3No$kEvt!7dux3s`tK~sXls|m zPHV@PldpU~8hhcvrP94VKlYuyRlZ~UZxf?!WgUlw%9<~$uYBLNc8TMXSHBuQhECq{ z-gf`Z$5WnXI2@U~;LG!s@4Gavo9qpGebh?!Yfnp*!*;>D$5L-`r2OQwo^bfXS*d5! zSH)H=ciy*gLbr$5BkLV!c02OSS3F)MQ75@L{PI`JJ-&pK7adY1YZq0e`D{yX&J))`7v6&{Icfo`pmPdyq7wbiri`m z>{QwAc3Nq2gP&EcR@ZAWp;tvOH{3m}`o%+Kl4#B*O$*86pBAmixLC31S@q1VFLe|e z&nax(eCJHFZR9y4k2mb1&TidD#f|TrUGJI5!dJ+;cHXlT$yetJC%GqBtg_+)v>XStmZ%$Yxk>@sgI zxyx6%PGQ!s%4ZLs?A}$`AM-)F@dsC|^r34~GM{6jPgj~;E_h#BaJl?E<0kfpuFlJU z=-HYF&HJRA96C4c{5!pM=1WrMJmoXmH{+aE(;w^0NADilE`I&{2jBT$vR~F*>6tud znOpefb=ubdME_0IJ$v}RfBDMj^B_qZsN|)t=Xrm`E7!a-FQ}OAd~tnFU(~%%x`%So zTay@a9`Ey5r|?JfYIw-*fD4hgQuglowrO+U9___E&t(rDEi|anyr%Otd;goaE-%e? z>gTF%d9Ni`5#HM#Q+`fkz3Gl0@egy0e!aSUc-Nh8)32R5f zcC0R_aldf8Cfu_+yIk_{ZQIS_?3cT=%YA;xiXM5SyMlzm@2+nyn-d^*$pfRB8$Q_U*3Hy%Fu`m*xYsP4n_46dgCwp;jU zuDa)hdAVyECdDsTU#GF|*7bw`L^d(q)F~;+f5)Bi`9DK{%5_KnLuZ+c6n9)-uQKCn zpp~Y5yXk`YTwnJ73w_>Qx-#8n>hBDn{|pB*RFZ!3H=Dz zpkG+2DLd<7{h_7u?0>Z0{(gNosQ5pFum0vgt)k0$%*-4bRlL@iO*rac-*D78;6q&F z!ykcPzJBVx_S&=iT(s6JhgbD`x9zy!HJ@(_KYxX%sjQf|cgH)y6#as8US{r2wS}(5 z*UPQ=CMt@bU;gqVpYR#mUA%$Ka*;v-=ktP>U0Tz&KKd(sDEPK-*`!AEU;FiLepYzj zP<*-h_S&qw6$|$;K9ZLF${QVc<;&?6bED61P02khp~3Nd`}0_yU4OqH{?8zM!*^wN z)w;Ed?;d*fpCNl&#?6>yRz~K+Cv_!3uYdI%s+J8kIU5|Lx1;UWiM)tS#ygL>tH=}; z*uMOyGyll1YYWq>wn@#2%I4j8aM^}wIf9d-Sy)ayo_u_7^v94oqw8Cxqo;1X@@7l9 zbdav$we00ADbo%z+4Nc31V(K)&MwFlIKSw{M&Ek13m@^v=sDo_)d}?r1*b=egZ|((C7;U$y-w zPU`(o?XYc^n-cQ9?4H`bi^_f6zio|!a+i4q>D!9;&(O$8zcK&R)^%_9_|`6Rml6L~ zmA2%z&!i=BRr>M<0XNQXw4a(6d%I%RZapiboL$1gjq{W0wY2@c zx9-|}`BB@qpX>8PYCdmb(W@=%V}BS`6}$H8y!j&L+s{@PEPg(Bf@1OOEd};mmoL4| zTl3||!oB;lZ@0?ZylLc~AG^5mM_{&sOXQd+eq}RF7D>c~@i;Mg1WpdP8>6~( z$#o;`O)jlBl%^#yoM$LCWqo=nGSa_P>BPmu(jKR!#6Y4&{#!-o<*!@*=$;IJOy1v$ z_$${%YGNP7W@|tE&%kq9B)ze8#eChm-PQln_PqJ={+OI1tBv-9n+rSUaNkYbGv)XS zzvZ77t@K}ZH*3=^?{6Q!RXn|`wPg182g&&oaeLjgB>t#B;w_yNn}6hA>FPR#g%f#_TY(quA-~u=I1r!`5W}P&nQoR zVa0!Fm#q51I=PDlIbTlHTs%KpafRlp}v}z?laZu@8!(W zx|)C86*AU_Lb?`ew%zh%I2o$KAAF;(G(32puesOmeUslsAGsu7!*kU)?4PZ+=bkI7 z7JawGr|y5CY!aHQr7-QxUvD$xSK|HGS^k-x`D3!NI-lcsrO~6ufm7ctvD@7*zu^4i znE8k1+Ih#b9IY+bvFobZq;JW0-rLkQ)Rlz3dZWF4(WcV3apw<5#U9&Sny>6CchLH< zgr2y?o!aS^-z3+}tq$M(;o9`VHFu*=Ty)>y_u&|K>p64z)N{3M*~ZaVb!}{y-O2iH zVC^n<(b%bBL(F6K8|n5w7o!6Fd9Us%zI|fSVWl>A=LrYCC|sK&e7H1U=G!Ng4JQ^( zyzs=pmcz78a?#Ev%Lfxvx?Fks+7%mLI4)YUQp-zYQJ2=b1-EtP9yl4a{6B-7==mSd z?2QlHI@LJ&TbRYK;+XqKB=)>--}Q@e@;6P6KWz0&MSk1V_UV)`%TE_gKeUOnMs84wb62c&_9NeKEt|Xa zXyKAu-7Ak?(N?K0s#)~FQsyheS_0?U5WM6TNtT)d-79p#uLw1u*|Y? zE?;>w^P=z4sP38nE(`6cIB6)xc;Y+rERA@cmluD`|0g6N=Aj>_EOIaSt>^Ki*MD5f zY%}mk$;n>6e5C&N2Av7_Y|bsc%cNy!m z3RU-X1%CYdl>NQ_jz?Y7gTp+P=8Egs&-mH?De=d3^N)hjzPm5fxJx}BOTW&w?%PBSVJShMCTGu6ZW#zr)2~SeO7d}^3UMd>&JY>V> zoo%-sy_0ySuHwpV!S_SbLe@8A=T#k@cZu6}>~6E}JSpgOj(H!OAA4}y^lk*-9ES52PPBdEi%Iq+slizW)qIf}Bi#=zQ@A>EGh1_cBjfMkMfiNcs7w6J76* zl>bwjB)I%w`XZ+#wTnE=Dm%{V z@M7`4s*6!kX`Rb5o|RpGv#{}NGp>%*DlE1^Q@I;$L(jiKR3Rd zw^caW;d13!F3G~}XKsH^zHFO0&nK>Y?$Z_1rE3lt-1u@nN?Ymn!FzI>uP(cqyJbb< z{@LX-tx}BNSz12dxu!dxy`t)T`0TK4mlm}B%e|L+D8)F>($ex>w%+YSw^XOCHCvs# z+?Dsw?LDVkR3I`LD|P>9FSzyU{HnuCqtzBXnYN4b_u+Z`XRj^U{$X8n_G-J;d{c8~ zwX`6_n-A}@-L~e}8mkh{6+2beHU0bWy!mtKqIosO3+F~;E%)5~?*hAh)~EKFbCRuh zeR+gSd`=0k!mGofR=L~1by)rmpJsFBvE<$B%SA%M;{-qaeYa6=ec{sRlPiundQal} z*f;0-R@c&p^8EK+3%>d;6}o3uPm0J+h4&IK&sB$n+eu!`H|_XZ-D@RjDse)ysCZBE za=R-b<@H9KpLvzE>2V>}+!_x<2J`75Rwrrv99Q}#CJPuHLE zVBO`32P|Iu?ESE=dqviTtm$D+-Z#vC7R{6py;WqvE}OCRYK`>6zeS(8*ZTYRy{ceS zN>BS)c|Q4dh)wcCUEz=owO-#2s_)v=Ut?r!eoC}i@Zj;R!#^fJT+JJP=vUif6^2)4 zVQW>l{O(KE{#H|RuKc+D@Ls)d|F~nFgF2tI+=rWTD-Cui5G6K|R9^;VR`6Qw0$~}$CC1-|-Xh#8ss7Tp>_*QI^$c|`BmYaYRhl$rxt$Xc zZoG0a^4eRy&CAYgzZP`MT~ll0OoOMkes(XvolTG4vg_Pzqx+HGnKDaO9ZNYp<*UW- z<+n|n^p{;TdAWG!=1G?{ix&FKD4MrizT|gcxNdI3rfuHY8Iv>u5(C{BR!`fuAu?KZ z>lRND4TBYn83OZjd6#gq?#Y(%Fj%#iAuu@PD)x=#qRyccIu`eO+=(vZuPOP@aJ=if z=tr|>&JSPLs<9pUwe(8aDf^q}{J;Ep-SlH~spf~h`?_;(?7Wt*v^l2qmS4ob=Ff}P zIZL#O^vJvk*!gwczDqZC)-2fCkz|n^as36G_V&$N`;MLB_GnXXGVOKWAR&LMtJXI- z5LDfEf~wns(1~r%%vS_41bio%%u?$L3GY1^7ADCecO)=UYR~jpYz^&)f0=~Wr)jZH z$y54!;gxVt3SZME(@K34UPk`*FB$9poeAOAxa0bJ;T3g>6SAx+ulmauzQ`)kRkL2d z`7P5qGr@mvN^jm)-mqwk$?2z8zMRf}(pC{x!lyKs%PM^{^Mz||Gi!Fuo0(fP;XK#5 z#v{QC-F?0#R7ve$9=C4k^5}ai#!j0R3S~|NUcb?N;bPdTtC5Qq>|wa+JFk&*&3%Dm zsvqvHl$5-isqvrTxNqu@X^Q5pTPr%wy|!%sXOXci^Yy07dByGZe>twDAAV_bo-yOO z;$X~FnxI19~MCD(BVXK#}TZNv~&RqNBxU{0|VqR(VjkSx*IA)*M zJHA2IH!ZtP;nE)KjO|uy--KP<&Eorfdh!jvWt+d&7W-Vto%T3*PuDH;gw8#m&Qvqr z>sq{I-Q-8sQ$DH~s=R$_yTzPmIj`LN*?CnfmDgE5-x_p@ZSo`O6@R>0JgqzK-QsRq z&Srbx?9SI9_n6}IS!kaj^E}w&$ajMqn)=3g3Hftw+#nt6@45pS|{oIaP|5h z-Zffs9Ln`Z6^RdSn}5G7fB2Pkt@gB!^{w&V(|@OX+?C0?y~N{(q5OOPtoD%0zME^; zUDsOhwBet8p=%X?i%nwnrre_hxa+xG&g z&o#_f&08S1W2M}&OWrr$$K4X!)#hr4`|ZVl`|Dw#tD*hcFlWLcB}XPO}^PH-7}?5NYxxt`Mmhw3rp#`t@lAP4>nt+ zEtg&5tx&m{WAeHF-wU4Po?o%v^@_s2U1q(@vjlIuY+|2n@N-^C{atqDyjoXL5jlyA z53Vh_=Gvma@&rTn^4N0>5(m!vA6nFWSjtA~%B&L`TEAK{@Rq#Zd4O3$^5u_^!XMS1 z{gIO+e)H(AIKg1fvonF=7+Y1HNZ8)&YjaVCthG*#?V!!azU%79P2tgj*0$}AEgj~(CCffon&pLZ->)w z@82rAT5r>ZySH7w8S5BNWhyJ-l5FR`uJTuV&*aFAb#E6}OHPySU6?0%K_g8@KW6fm z_1=>s4P$S+eKXb(mv*{iyddC=g=~@krLL=Mqpni33l1q%R@~S9xcBnel}wD9=C+o; z%jMT;^?tf9eb?sJ1O=hV#h0htD*W|mQ|ZF0l#Ra)k1!>F>zlLga{qOa&|A9$u|^|Q zFmh7;lB`!PH5(s)dGc?mz1Bm0_HXx(?$*vpa`%X7|rj;pHH&1BoW z%zycvEvI=-w`HUl?s#s=W68EEtExn5+g#J$!(JPG49<$Q%9Ni-JYf==EA2Jg=@Re! z+wZ=b^=x+e-p4cJP3y+UInv-!t<9o#V8ZT{il8`7B9#MyS7g#Ht$+&&~3xLJ14Lk$kTdbGnB~GnJ#>! zFOeB1F^S{-(Z?q()J(Q*t?WGhd%@$6YkjMaR;9PgalCK6>}jE9vVB{n=kec*yRr&jEJGluQC%NT}C%!-GuD14(c5fU9twhGin^R%_?Fw zid?#Cw}<0%y-J4PT=?WBefGQdu5^W-?$?dk%hfKcF1o#XON(65^9lSL7EKPixb@~& z?drMKNBYy5azE!r-As*SE2>LAp;CBMZRz6d&9B0~Ta|aponyKE?AF#zr#A6C`Ez`N z$Kx(vZ?9{5rrOoL_Ho(PCAUxx&PwB5#xUMfcueciVgE;dbWlr^~r_ zo7*1>X|VXZJX@e*?Yjq7rthZn?vURXy(9MK&VNEn7>}>Dj_5v86zfouz1B{;z`kaC z;pUCce|SCM(^z`lE% z8o6-lp6fm4Hy<67oNdc2!*%)^+eH1YExmHD4w~wQWu)etKIZ*av9T@fyu)#(63~D+)Hl3Y!ra)uerc2kjr9EEGW zbxYbhV(r?;5eqte+-ODDSLv?<7+xIU-GVruKVmHCoCrF2h!Woi8L zn%Oz|S*+XgrkV0l+P7*XOIO8Jgq^F+66Rav-8nh!p7R`;vJ(f_Z9h0YYU%UVoafgT z8oW~3G1DyfiRHoLIXzMvA4abY`h0lW<<|i-O2zb_tDW8R&cfpD#Aet0u&+561J82# z%N>^~Whrp{%D!>ctZB2;SKshXtz|l!uW?wL|4Nm7)$&!DWl`7Wnzk*>n|#}6cdSg^ z;tRhoMxMCz@-oMLUT!%{hC12CZ!fC)y&gQ=edV9@owakn&tFkdk5+sC z6Pw z6mMO(;l8)nqQ$vks{eTYGu(Fh&mfgu6Lj(T-OFFgenrWDTsFV^mi&j8vmV##`2V=* z(;p%9?k1nde}-iZ{~2CbE?a+O%Y6Hk59fB;StZ?P__yWik5!YOzljt2D0_UFt^xmR zSDmPyPfT^E+^iWD|7BW|G;kGIc#HvVTY zboI}y(vw)c?zl|jdFB0aE#>l$qxMOBOv=7=H!?QLZ2qdKZK_))ZgTpTP~COo!}OlX zM~d!6#6{Z8Tz%6{edE;c261Qj3tgkvP1-dJOQ&r`-VN0`Z>l(M{F}=AxZ_Qp%$JB6 z|Ju6T6FSaRX`EbFen>@i>*IMH8=8MOE&p|WY7Mh=jpk~MgUQQLatxeJ9)F zQTx}v*cPX7QE$ua7l&`h97|5@b6+U;;FH~yFxE5SLAA3i-Y%Z?=u)ytp{J(ngM@~H zsyvqTjo;RBAHFU%$)m=!oGUh~O6E|6%}Fnv3lCZ?4?fA0IKSe{dh-vt?^g3%`e7c_ zoA#DEM-(A|4Yu+VKa^c;h_Tu0%k2|)uA7gK2 zrN6xxc{aURW|BhgLvuGiBe#by-*v`u8Dyi~Kq~}z} zxhJfFll&81(=)Rom(JbbEmQcaF8R-aKiX|S3f9*D-TFw=GS}#n-LeM+rT|I_%_KLeve?}jggwsyb`*ucQ^MhCe)Zd$xx52|JiHMMoS z?QuCz|Gf$)3rB4K7rEz4gRhoa`$~K7ypq2}@OJdV^5hrR{Kc2L7AJE1Ysa2Gss8kf z`ZD(oFYJZy?c`^Dc|B_1V*8%bic@!Dd8Y+z|Ff}u*#7giYxHl~mAk~%Zpm-qv$?O6|L52B^l#R|C3?PF>@#F4_eI?Q z`E}jF70GiXGm{+ z9a^gE#H;PI^NQY#HDVvPtaCVdK=MJ=)m*1vvc`Tp_U%w{&X@N1A!)&E!=-I~IM{Pl zX4$Q~&*tg8*L%p5oFsXX^90M|%j~kN@yM;3*Ie*x+05+KF*mraS9dsS95G%sc>?G0 z3y-&oZjUZq?OJqSaO=VaHTKu^n^`BHFmyd1UO%PY=AY=~N7)aTR}_7FY>}_CsD95K z{^y}vIrnsP&bN7c;MXkeQ}yY*mS@lHTq*c_!^GPwzgp_lFD;Vi(VczvieAUhvNZ|6 zzuI5g^v7EAu1)(gri*dAADnd$yz(XU*_F3{x?A?lQrOMa)O90_;a3A+_Tmp|i&wm= zGWncU_RsZDoUcG^tD|d# zSiNZNDNq_!klP|SZ%gT-$yvYnPfcdN5R_2GrE%bk$d)Yu1#JEd0e)yND?Tgg^zX{^w+2r<_H2k-E_-~(Uw6^tKaL-sZZ35A5iifr zs=FPh8ZSL$56O?uKec1NOhw$gR~a_m4?ih=WSeKyP@tFmUb^acVDQbm+$%O8DC|8T z)1PG9n_ZE7q->wsC81A@PfsmR`5?bnxNpI=^DTRV(;wXx)PLL97it%Kba(BG6xooA zVcJ>SQI1mwW&F)sU;0F`{?~!`oQp~QxjXC z*6-?yxO#f`j5TYX?V7voD3_${UIt0iwbkXktv%nT>j`lvXR0LMbU0}eDr=;s7Tsx? zrdr(h^K{*w`U9?o`}x)`-P0cU(e~E-Em@nV`SkA=nxi5Ae8YOXAHJD(3v*Xjocrmq z`+A=D5s9K{FFYJ<_a6W)_QPHaPAttyi z2=(=H?K`u&WMY&2!q>mjcCMFymG1SyZ>d%KnMwAl9B*RY{#&*)zEf&*#(n;;*)u*r z+Y{%KQkHIA{Oh9sM&D_LPo8YJ=fCsGe+H}l4|`2b`gvYn)9k%s_2KW65}D?cJ#{|n|Jhn`Kx_H! zTU{4JxfX^nEJ$FywfktOONcAiDgy=yhVVsNs}x)VRxM^=Y+x1HvSkZec!XQr>RY&? z;^dbdYqq@%+xE!2dnKb>N>5!#|HcFC>i2dYyHd*Qe9V98(bVl;Tel0v#CmXiR!RP} zq_&U2((Y@UsjSoMqwANurf&D$zU{v0m4@C=hCjpXjx)&2uhL%pGFIN}&Pz~QsCoO_ z;m6+mQtf=6Ti!F?^3N^Z@*zd}Y5alWAJ=EqxLtGk9+(-hFHC-3j{GBA_6PQJi+_E+ z_fzbqKj-P%jP-#(JRffRaQ*OWiKR~4&FfzVGg-2JZ2X}idac6T+bX_Y_u?_t+27Cc zCspyE7G1tH`tPiRj1p--`yabz{|FA-c{MyddUhh?<}0aN`x$y)?6`I3*X0_`)iJMD zuU>n0L&MEiMz@YPxP6Jq&8yAoZ#T`J89^J_c!g?Iby=qNQ_=9EYvrJKnNcX-l(L zzr1k&rkCpWndRF2Pil{rSB0%iUz<7YS8vN{(=9I~W$Kk*xQePquSOh_4N9yLa$370 z)+)5jlk&H6uzx-x zH|lNFr%z=Oht@efe;U|$T;la>*T^eR?RPiY{T8^t9(1N*@&~pJ`;_CnZRhs4y}8l7 zH?C0f_=h8W5%crruPu%jTfc1A^?a(P+bN_7{w>eCRKC;nm@un3A7R?X73KC*vJ`H&#XMnJ)0+dr_2| z>JE>Yi+$1ynU`s-e3Y!B@n)&FhKk&skdKCwM9PAeP7&hI)3!6y7cvZ5vO-1fO7P1s zuRx6oP^F6Ea;;@c7Tj8?y@OM@e>LNu-lF+Jx&JOc;0?*W6Da0j*TY+QUsQMB)eo1W z0(N(~Y~ugH&Ft!#cIsMy=*tcKm$sXDx9qx5^7CS|&a{ggl>c6g4!}PoKzDDN?rCwNjt=2EbogA|KfjX}mycJ>> z-jou zZ$DS|pJCg*hPtiie=I)o?HB9J(A4Vd^1Hij?Qhv#^$xrubV`YlEy*tQ-?FX`^=)yY zyFdPC=<|KIC@$MW<+tybbM@B@ExEs{e_`A6`_L|zdrH^Oe7Bw_ar^At^B-lN@E^G* zH#_V4_uWrUx?0bf_wP#hH3!g2@Ur!ZFJpa08yCFxUy;4*->eTm(|vb%T-`k1FxI^J z?RELqgbb_8ucY4XU$)0C@^YS#Y0mRA;l2-kmmYr@vCD4DufEM|Ge0>$ooklT8*TYn zbb5u|-NGM%(O>sRhk!~9SCLTL;J|*b%TM;slwH)*``fC)L-H5TRjh*9zi)hbZN6)2 z#IcsV+c$)J{z)#H@N(MaR})|5omZK->_)R$pz+m)uXbD7eYVXH=?^ZeROvN4BDtw^ z-Wv9mU)M@(iFRIbe8p~?%}Tl5#WkJtmb$O}3c88mGALOu*t2Q-72UX7uMAcEdscq4 z6lHxnb>G6QZ`T7mm1@~d~Rm6+P2`|lS9Sv+Qcc~<1M_07C@>vi*VRJIFj;5z55 zv!c4m;_ao`>D9VVuRdLSwq*iG1gm~qWWe(Y?y|c=3eMVWHxJt$X5^r35%eQ&KKy^$zsj%p6)s@@$Kst zuD)}y&}#q8-3hKiAkiz^!jx`Zvws`ip{FOh=j>A_t$8=(?rjNkkyF^P_jS!F1p&b~dS zFs_(y|FZ1;a%=Ju_PTG`b~)}ICx>FnoQ z=^OLB%S$dMXZEE9xD_`s9u#$Z@40Qy-nO^5N@nTg8lE}Dajw9@LTU4Q6L0I=C)O?b z9F(z3VrdFXx`l#_l%{^dm3glub=D_!sXp>p(q8s)-f@NIF7F$r>bu&!wQnryteB*! zcy~|w#uEyU7Vf+BUS(&OxAuvLAYr9@5+LD=-nFIGxjIg_{jyiwe!uC|s^AZ{nfdG1 z=PvDt&OBQxUHdIE)T@qtt#iHT-z^)tqlL@iLL#%;Hl{A!qg}1peO5&~N3~vvja6;o zn#jz2u6w7FK6`TA=8Qkm*ckB2d_$(c(7nwrswcv3dHquV0t)pAsr&%!6W^tc*`*hPh>%H^SDsHCUD=?dQE{prb z+UJ|*NpEPo<#98W|H4WM%Rd^cta5IvR7f`7ofc#zeCMAf?~hrJ_AnOx@_o1Pq4rAi zt+Oky<G zz@4=DYwe-CD)A5QczpdS-G5&6NKGJ{JzwNK{+}gR6sLbx{&k)CV?kw|*v5VT8RqNy z-OK+|Eq|?RAKz2uKEAw&ec52miR=Uw&;zzF>_*+a23=6Fbvhq$@v-U%vj+ zn(V*1$!E%Tq`ipW@T=-#oc!|4U)Ov8_)NQKvVB+Y?ArBP-&So@ ztbCW2Kl8@(SDs(iL_D7L?TNyIK{t;;-*pU=xlIIZ~V_^;~l z%^6W9QgSn8m)LB0=6+o7?%C3^vd;9pjq2*F<%3;IOqYgftl{R(lH8NII-@%;C_1|$ zPb%g=!y5jHmv864ju4Z&Vq9@K?4sD6CEKojxMQ;OKSPAP^73u>uWelOb;Yueaf=t} z?)1z}=zF;I{C)n3J=M8?GuHmPzZNvEtYvI8z2b36=G5|>yXUvnvbv=Oy_d=BHCp#} z_ReqT7c8`|RX!@{HZ{D^_pYt+OV`j1O<&*KElq2RR7vtFTz;ba@4dS3SFT2L=*OMC z9+sgx$xY?BR87bJyU*X(`uZ=u5wmu7Z^%KDONOd~b}sSnK3{+B8V}m&rS*e>hp{oL zfx!UzAmzpo!-NHfuL`eC&}!2WVYBtKy2KiAw5j3v%HvU5%a&a~ne=|SN>R*87Pmj_ zl26h%D(?-sa_gr~?3X8f-cuDi?l(NP>G`HozRK!UoY|ab2bCv1O;t?Tb~@icPmHyOhH?N6(6>nm_sW<%ut2rnhXhUdtSo&z>D5R&ngD z)ze+R;;Fy-|7|U_D~RMM{K9t-;=-8b!puv_%+GeQ?8q; zUs>|+{APbMdE?VHGMx8+&;0k`pYq$TkM}$78SebZw^qL4kKQBi+0#G0dmeUxd8I=w z!_)q48Y}ejbJv`4dX(C8>XF64S7)jiqc~Nkx0QSq4}QO6?x!1)l}l`=);qpnEvrgh z*ZN`q!IvzTTl-XvyndWM_v1<9{o5A;0}d|Jv08g>!DmmMO)Fe@zb85VQK+6*9P^6T zKkVANcWa-_n90z8y31iT6aT}^s&tzP4fCrW=$77>`{Qy=@9Br$J-gE8>M5CDp1V|_ zlyUEQ6%F-!U8~oen*J;I+Fa&suV>7YH>|wnq1+#2{N=0C&ijHtg?`==oBQNd)R`9- zXS~h*-m%GW+Qd#v`RAg|>no&gn)1Kt&KG!9r~1*S{dwE7C%LIJ|M}>e6mxP_ zY-B{H`7DRc93S>vx63lg+PG=XF~)O#>;VRxue<6t+-lvsZ{p(DQU#Z%JeX ze+G_wt0n&4y;65C_RPn+*)zfu?1U=*9iMCcNcAA|yEXk9^1lQZb!}Z1^?KT`o}_1y zw{Fk9xx-U@xom)G?hkjjAFG#DOuJh8H|yGOx1W1HEsfmw@|e+ruSZ3~-Q$*mMl?|R z50F+THAt59wGbrgA8;wB)Wvo0HqUJ`3oTjwH_CJ^Uy*5MwMgmSZI#=60dty`-{^A< zx?(b$H&ie`cT#S-MPFrtiqr?kmK3DUAP4k{{T)Orz zTCL}aN!A*U<4zJqTD?weaWv{&9l*0=p@Ubj1xaoWTo$Q&04nnV&wIcY>N#qPY;-P@1cm- zpZ#NfA6+fGp_d7b@ zah_FR(4CiI*S8r)-mRSM(&~Bdx&HRw_x63&JvTMCI$3n?^_eCfA#>jq{jT5p`h8UJ zi?Z;r!l*l|&q|4^SI^mJzxTCk>Z{;8x3ed>O>1%Z*evtHf^Yw!do$0m|Gszi;E%uS zOX>wO=P7L7lsiqkQ?(<-B=WFwRkBU3|I*6-qaOMt@&cK0I+x$Xr_C|6naI=LFvrS# zneoK0c2mOM1g*KK()m3iC46C2taEiH@73MyZ7F;`)=fs*`?lA6?Cu{cUf6IK3fk|TmN7|GHu5`j;}7N-`D5b! z|3t4oQTg{Pb73E=Y?SYV276`K$gO<3ibAy?rP_Wyp7lD%W%csgI*)}prda%E$Y!=* zTei(9bi+2`9v`KE#}_vk|GF4-<*O~|+P9$3AsaW$;>cU~uq(UmCQsf2llF`5RVyo3 zMcnXZF%K^|YI-=Q&AdR$`NiVBVLPwryj5@F?w+xD*28Vi(mQw+zc74Vw2Ak6r+$s= z6X)dI>_sm7ZO?2Hn^xq`C{lZYKd9>4n+-uDG71#aAqdNEP|e6ddJ(m4|kxXPC> zUzxXewZWa&-{q7(C2!i=?ABRV-Knap^2VK!;RWOCd3#r^wY7G+T(UARZPVs9$#rv1 z+N@|W>3!v!`SO<9?4{ntrc>t_oGp3g`P7ARNq$h)-)R$VpWtOLGQu)zPWKeDr0f3pRcG>g zuh+dZiMNZtxxRL-`TS^pN1gVM&WfA_&q){0pIm-(wb{de*DUjo_h)_1jXtt#e`mVc z$JHI4KXVtbJb&G{_59r|@u=oE0Zwugmw)}bXkA7BJrWZ}z>yrKI_^-fthYI-vQtpiQdy=Irl=m zXX;7h`re&d{C(e|j<1}mSC?$|w%oH~x>1b5t=YS^p1-ek_0GHOyL#{Zn`#0_j8zZw z=vA}+`+gc zzBS`UW>{6W&$kP2C#Bt2X_=O0_vG8RMe9OO?S|(!1S4qoWu;yFHt)Un@qE^i)#@KL zCA4b$<)`8hPS3LP3IA@kJ^GgA+fJ?1f4tnmszp3y_bt!cbXdzkp>?9cfxZX7dipPI zF1^qHvQF>vqR%SUi&fXYTaZ7?>?`{#tLkv=NA^4~vqmEm@ebjTJmS|7f|Mf!s{9`*K zYJ4BbGoG1Nr{DXPS7q*o*Jdv#HD3yS{o_A_mA0E^@=WGq2dCfKU16#%J=Z4wa5?ua z(Yw#tr3=5V`z0IhzvD97Qqed3Nx}8ESen!B$6WMZmmPZj$^OL5^nDg7^7n44oPIs| z-&*@yHuJ7Nsnhx~z5dbcoi?@))~w0>bSJCGeA=qLCz8Lr|8}46rL}Hx)itqPqv+Fx zYEDZ6d8Q>_sOGO)S}^UKnbtMO_ok(#5&zB#YRWIttI~M=uIj3-y~BkqZ}sjSKJ-0z z)sIcmE8g*b-NLJq^{O)HE>pt9p3nEXQWo7X-Fx>Ahu%elhMF@=zHs0e*VT*lXS|W}pSq!fML+(#_{T*Hbc{PCZm`h~@J(>S${?7fR4On%IF{O~~9x%IZm>lX`3u_#jbJNw|3pAY1VV&p11rz!`ip8eXS4g8XulkpLTieG=sW7 z?b9dx+W0=M@!@*T7x#oNADuTbZNtiEp8Q5$wUN6fyXs!uv7=#G!QrWPZF3*-3f(?p zrF(34iCtv))4Ugl7xOP%Ua|P_v^wn%Z}0Wj?B6}NE}VJBo!V=14=#q5ygbLdrJ>z_ z7yp7+_I!2=x(EWlWDHOo22@uYw8a@zP|ls%=yFd0{8uS@3)qIa@)OF zP^PE;=^p!o3SS>f{;|zi%k?IEA4)F~-d&_FW7Z3m-U-inI4oZKmI%3>C~jksDP6XU zQ#r-*FcY7~nicDGOCoi)v^G6d)DV&^6LYHHr8_#q$*HFhVuYCQrUUAp+?un|l(QQ^eIMHdruTr(HdEDC7 z)!p44i@0?UvrYXqO(7xVLuB;rJ25e?0+D$VChMI9?b;*1*xCk{%i`u?pjORDtjloRVk zKAPmMG2&1b3G-_%EPnEL-cOctnQ4KS3+&Bo`+4l|eOo90@a)@j^Sx#+&kascIDFtd2Y4^{;xFF&U(N4!U}k7t~uTR zWv#E`^Q?k;xBS0T>eP1V=Ks^u@2rZOz?ki;X?L^0I^VuP{_$_|8z=l{Sk3;$zH`x**L&Byf(+eYL?MrkL)3yK~Ls*&#QS#rZ#r1l}yZygo4a_`|sx z+4Z-JBafS{+~B4s|4d`?rsvD!7A=>PzjSWv-_5FX{8zjYTyy@jNZ`wPbE~>EYaAc- zN+0^>_C8{N*Dd>pVV31~U5h7&eU*(m?%8~9?+%aY8`P)$;`>@vJtKE=%_Ua(m5cU> zADu5rwnNyY~;z@2z~d@$p{gkm+x#Z2Y+_Z%MkP)-PY0 z7@KW;XZzHwrHqf2r`bPrjr!1jAp5WHvBi&GuhhNfyyw=j+so>|-=0&q{AJLK{rq`) zGh$vFZE@`L@hW=z)pq5#S<)wJPH)qmSJr&;*DUQjCGtEk?n!PDyJa_@cUAH;js3>; zdyY^3alLl=%P^}DyU?jy3f31bp8VC;H`g(jcT;Yg!Q$I17j0ej6nwV;q#sed?YfwF zo#8c|2c3(qC~WQUFId5N>%kPp-H&-H*OlFRbm;vttD|!SO`=-FHgdibWw4GBYA~-! z>^JSqFHJ0d)a!SSqo=Ax?Bmbxj0~#r?-=fWY_l*4TQ*NG=gm}8>lG<$`8*Bty~ zDSz)54&iX;CH^L6+G0zKxzDZL;Kg>|pq{V()~~?uSAVl$W3~{-qLLvdx3jg4OL-^v zR33fOW9E`~K#1Y(>;9^R%Wg+T&R&r-C;6PdO5!HvGKLB4`7zbk?Pb?aoqP0(nUCSK z>5~j@a=bBev#xr4uWQH0H7f$I7|r8QR=6UyZ*_96$h6IFdP_?zw=u}{f7^U${mymy z$KuRBd~aQ&;u{t5e42}5P6&G?`$F0For@Or-ON3_MOp9eoo`>x7D~CdrWTj#^~)5v z8~^M|)w#CO;N{bWzn}kIe0*uex$X!00zaPIJD)E3;d}eKl+VfA(u$?!ZU@`^t&p1% zTKQ;Sy58oW$-h@U;J?bNT_Jt6yHxkW^)KHl?f(7FVg42Nbozm_(Y{3=U;PR_V1Kwc z>&*U%7K>I!IInjVS@bMq;+b_JGaY^!fBMhBzBF#@r8>o^V<*{`E?$*q5#BEOT_SGk zz07^`Kc+5^;@Gm+|M0bMj~j)f1nd@PJgDN*dSg|XW)!>aTi)t9&0j4f{DPeHMy%xN#>>(o)Q3x@&r&&9LB<;8HGgQQ6Jc zMLD)?i$1nHx}x>;0=F3o_d<#oZ42MJ+AUp)D3(A)33clh+@`Iu4Tz1-x^UNh%WaeN zatkHblkx%K1?#Op{))+%r*v^oXr72Wb9Z%5f!D?twQiihlJ+cqv_|-ao#KU>%p4Kj zEvqw2+@%+n9phi|%29svi1CvsEoF>#!*WM3)*SrB2A872`I{(HsJ zUgy$c(Wy(W2XDKv&{=uD1DmgNd8z2sIp2#nL?reuDYKozGfS&)y5y0~`|l-h|9CF? zKZBe~9n+5YZQnk{+{}_w-WjB4nmuh|@gKt`r~c=^!g)VzyA{2g*Z-)wm+QIhC6_hw zR&z}fsJFk=b#dLj-M_TOV|HH2D|tQJpFbw5*F#W#rs16p&m~p(9>_G*uEr~=9Wgy% z=jFVF3+r^l16bVTf_EmVgEc<*9T;vhoBeXh>FUgNOLot*ICQdD&)ma3s1PE28(lbT z%WY4clGoLl@0RSIrE$pS6N8kK)#NViQ~#9zhi{*6!Le>K)G z`lI|np8to`>A!5O=F;`m)pk6+ug`zzPmG(rxT4(mg2I0W_Z~^Z&*#LvM%l2 zJE^Aob9vco*DUAS6?b1<*t*4VuEOHHsaM+5WVt?{njfTf$-mh2k;KHhBOy<32psu# ze5;q7`;=VfQ^!@lp5H3EK0GT5d45Hk-EGHZy+5w&PNsF`J)CA;z0)jV8u!$g?*|fp zUgPU6To(JIJ9g4pk-LX=%Dxxn9Mlol+WEfW@aHwTdXIf}UHe+T+4Xmfk&5^vu$1wA zhW>M1SFY&l=I5w3eHDFs>gnd&GEb@mqh_@(UHi6WL)@ahJ7cz0C(r43(suiH>)yL_ z9(fZ?R8lwJmOZ{uBqYl8N%*cRPnL2f)}P9i$D~}D-xr_bU%6sS_KEc+UnX(9W8wbf z_PNKTi~Zj7Ir70#QoJwrU3xrGd2UktB=Z?Hmo((7=T&}Rv_;!#3$`_|3wP|RouSTEsI_}?bvbr)qry*s_B^Kt-CV`PfKjt0LmQJ1ZTK=e1qW$}AhyHCn zoLw*VM|d$?$d%qdhnJLP#H;^(C;oNO#^>3&T#?Ts^46*CpJ6^-C||Xy@e4Ea&UIn2 zVIR(xd^3#MxNnBBd&<(22@8_XDrfcZ6;6I1ms|MQUi!zrXD@zuJMFx@HvcKhU5}+R z<}RpN+r!76^EYF4RO{New_LVOe|DmDLPzt)Z;}tfmj6-QZoI_J*TKO&f0D!_zBSb; z8&&&vr@xH#Yy6-l?fK}+*156kx4Pwi+A@(-L_YVf|IYK5x=dxYbp0eA&&rg%;d_v=fPx`nfA2=hj_kH@(SkC7C;Z@>H%fe$>15 z?)4_UuxqvtBU7d~9@6u)U$Bn(vFY5a-w(&SS!aBIb7T`wk!W4czl6Z>HM2{nSmqwl z2&(mX@Sj2J{)e~EzqPk*IlcFG&$f#j?UrYpe?4pe!`biO+H+@y@7^x?R=KG5ThG6p zi`InA1qVZjsi)VHLiKyeyFV#2e@xoHwy<7EU)9QVU1!FT*jL}V^LD(MDcyg0!Jq7w zAIocvGtB16Z%)!LJw0`cPnl2C^Orq!C80Z%1RN$Z`CVmJ-@mtXnpj}wt=hd)m<1iB zj_r3ajGZ<=nQ!K@+Y!<0ZsabCm{6^8RCH?P`pEwbdFt&8XP#TC!)q9EtY*o}X|^JB zKF;?#x99qr{|wuO4_`R*JZKu*j5W!2K_#ci#r;8`~Lo{$9rNQCl|!~T}n~X->vfc>fh9YUY&BX28&m=exGyc2z zPq44>=c6?*?@3Hpb>|h|M(5r$Ee{*Me_6Emw9{1^>r#e|7fgALD%MWmS36cxDD_1~ z(?axxRPD9dCUJgq3$s-U8O)+jtPv=Cd1+U*aGH{4x{lxK>!%Y6Uta3EGE3rl|LJL_ z6OH(;NlLOSt+mqW-g?F8>p_v%SHrf}9*{}+{B_Y{^J%LY=iE7@7;ZR|yLVx8uJ-)c z`+Bw&%PK!~9BB)=5uuTLm}PQ<@#}eZTc_?VYrA7mCG&OBvXz?9V?ZJKBi#OB@RAt! zm0402ta|TLGViecd>TKs-|n$Y;nCuMqE@a)s-`Vm+z|Bco7>z2`SU`5Mtpu}cw8n- zzuj-5P5Jd(7OZ-CbIv_5KX3GBP33oq$9)=WbG<6xZaF>kZIaM=nbb?he_YD230a@+ ze5>e{6OUMB`y=jITGf>`zcZHJ3a?_j<;HR1W$^9O&u_cxUD|!Y?2>_-yVt$# z)6Z|~%3WH%Aa_YZPv4Y~sj8lcJT5x7c5P(D$tRn)Y?nAcnd8oXhW-TGbJ{aL`XAxv zt+2I!l)GZl`JS~~;%(OcXJC}ITP_k9k^d;x>F1x_9D@8yvhrkVJr3zVI=1Igd{3PA zN1HZ&mLE=)A|LefLN@$0So|^BRb>9nU4Q>GoHdzrj6ZW-8UOERihpiRsBnKQ-dXf( zd*`D|O8GyNnIANM>8XEnB{2AQg=65Klj|nNd^P)^HUIa^?dGq;-t14`tLN$R>)YOy zSwa69#5f%4g?Ieh5+ZKCZJXEH(qbO2<6Nl|G}J$?F3POpo0VMsy=CX`{#v_79$)Si z{tS;@USt1(zvYaN$z3~5|2xyZ9^R0vpk=zaH?fIXDxqf*S0GqFLW2pT%Vh~y-vAUt2KY&IbM-tSHC>W z(G&Zmsc6K}eB}1A&lhTTeigZPb<0N)qwXC``ZA}nRGj+|^F=b|l5E%k)!l2a$Z=Eh$k;^pm zG!qls?>%2EzpXNQEmgzx=5)~tT>~{8j$0llcH0?}Uq{z+9x?}8K$WrWbd+|i)mB)-YPAposGV=>0IS(uumbUA9 z(@~{O9X6A_Hty7(&{6kUe!-ggf@`nb3f#J-!{fT2!es3U8#gG<_$S)_=*W)wPLs~v zQ&ilY$TBmg)s4l_zExChy2hld9XWq)Z|s}hkW>HoYU_u6PtHE5&WPOfZPQ6+ejh`H zI+t+%4}DLU{ix2&S5EV`Ox>3Fa_c$vC$6GZyS3hd*50q#v9WyBmYJ-|FOM&C_**W2 zAe=R!{-N@Z<(1WQzg4Nkr*~gCb-&`D%s;0+ul?*lOsR?AxBuL4dH#dnB5HZwe?Nbv z$X-iJUZkSUp7*8F>o=cbW6o+F%*~a*H!IA7%htB|ZLEBkjkA2)mx-m{D)rW-O-yq(|H?LJcVrB41b>&YhXnNKGsZr|Sj-dkCQ!Pag|sIPb6hyM)Cx_PB{ zBNuF*_dewY_*TWraelnas5mK=4$I5HmyjGE4#hNtq z@|=F2W6=BU&M@!CqCM_M;>c~TRGs?E@Sh?5=XKGQ*|*O3 zsAtQn-E!G{PJ)k{|J&^aa#v>;R+q2Z*>=lJu>5%a%0+v*AKpK>*Zxqy*z(mz@ma5& z6`8*m$bWNub?^Cyb(0UracsMIQ*S-jow&a6-y7%W2fvTnrPnLE_ezm(rz*ShMUO9w z_B!5Jl5JvC$hnZZ;WD;N;qIx|-)>tj>WMDqKrvXCCT%F83vTPDJKB z)qf}2j(O`A2`#%B=e}!8@1ou`XO%(||GBN^LZ1AeG)|{8Pv5lZs@S1hmm=~nXS-}V zk=7}pz#tmbZq2T4WA3+gvPgK6#x<8+q2a;G(;aR*P3)iY_d?~@=X*tMX4!XKy22uL zWV6lnG)ul^jBU>Pzij#zIjj}%NoA|O*!d`A(^WBx;&0a`cFVpku`m1mBb{->i4guN z=?fpLTy@Ky)}^9URJJbi*^8S$xAQz`GJJh`TgIZxZr62|>=m0fYv~@(a_yg)m14JB zvkxj&YVKK+rF2hPQR4maGgYcf`xfTRPHKCn5*b`n`O3LbuP$&m*Fqx>3H3f+M;7B_ zE0|8L;%ZuY@YnYT$^RLyx>oU)Y+Z;byj_cpqNC2*tP`@hICsW3hLl-d&r-H+omCm5 zQn=XK@D0OBQ<=}Id-rVic&OaVy;<%6&t%usA3>LWH-EUh>Lu^4L$Q~S{8Xy+Ul@u!|E#+EWY>h<+uKgvQx=xJ?5|z0dR_yc zr~T9no2s7aWiK;X;v93%+Vmc3sA8~G`4P9Oypm42Q!ZVie)|0C@(I7UKaHvr_@Tb!WuK|O zu>H4d>lN=S&Y0F2{i5#Ucl)U#;m$IdB1sk(14Rd%_E?nDCll-4S>CSz<_xFw-U;kdL zYkhTXL5+36{F~-o`%@D4Y_VWYk5f5k&y}{jw&0%p6@i!L6IX68h@2&(ajVi|!u%5< za;p1wOg=pQ#jVgMUP%hOw?xFRlbL_1t0T)iD!FdmRUZ{N5&d{*z(?o7d>GHTp2g~Zd;3r?svUfv$_ zPbB2K)wDIGKTCE;*Q-2 zGjS?EJ7)EG!OK@!hAXqprH=k>-D9Sa+mh3(Y|SKXTiECE?eV>?NtXln&JMq#qqk_g zYm;b+p+eXd)(@96#+-xTI`#bj6ZIzkTd-dt^ zjsoS8;Y$6c zlI$IC*UnwP%wq}b%z|he#@owd=2S_o`}J<|w^*TwwQpy?{p~#IjKNXQ2W4_U?Y(k# zU0bAbF-SwVRPFq13hLN3U)l* zDPj4-{>W;n*soX5zFVAIlBjkq)HP$mI>&%JP5x{B7stoOZv7ITCiLw(o4`fIfWsM< z2i31%Ta&Tb+Im+40OW!0Io@+1Rd-c_Ooncgi|zer2B+rFH7Xx+S8C6Wj}3 zz5SO$hZr|)J?r+|x9OgD#lzA`C7owlp0Fttu*d!V@qNwO^uV2$mt6kS@i60^-s2{> zq9SIw19djfKbTHkon>xjKC7hAo!hy%K(Q>ToTsTmzCGrjNZ8t}YirT`E!tjRU8%S6 z8;5fIW{$RR2TxntXxO%~`EOjuek|(S>f*IWANq2i78eh!`H@gL=n z%-VWxGken?i;vy6Z)`3!u*}XlA8zsT6KIAu?AG1fg&da2r_Br}$A&d7yj4-|-t4zI zev;v|;;T2lCcn#G@Zov;n>~gLZ@(9|dv|K*-I?G% z<0j|4d(Ur_u89}<(ZtXDA~@cwSW{Y|r*x9@1;&Uty^p%vTWf6X7Mq{dv)NtIuB)^B z-3s|AEn72wt=q*+l0ToXV>odB;rT1GmOro+%N0Lt$bIza*UMYap0q9BZej1(weZ7h z#~l}HRJrc!nM!ZZ-o2^nXJ2vi7YUJY`SQn&$;&I+$2U9JH`y zOg{MK`PPu_A6+K%7KiNJ{AtT??#cNn5dxn#^@tsOVJ^r2IBd(^vg)6v)!p6pYNsx} zKCv~O#pbBHZu5mab-69!Gk-WAnzj5GZ&tiW#kxmaDtv0bsg_HxN?Ua4c&x>lF8 zqce_)SxjfB@BF^rW$Tvho3=;jCVDK2>uP4&p4_SEFmXYsY^!MAvXyU7Mt7X}_dz7+ z(zTrmOr8fsf`Yi$Ze+_?wp zr!)MT^Jvo^uW!nSuTRVh*Urzhsuc2w&SU7q$)w^_z^)~MoO(A}V1&n`I z9h;vnHh0;rn-05^ws$5uzF;`8Xwjuh@2*{(9<;+#EpyK5w#S)CA;oWxY3b_fz0H*@ z)oz)THuKbuJhN%4?sj*Q*1hnx-P_pp@1b12<%!%oqPkO=KRgx*Yt0DT3-Vh?xwycs z+j~BP@Ao=Wz_3DJKX=P(%k}!|9=Z#IK6xr0S|NK@BwV?)y!w&#{!NZzH`w^6JZ`YI z*?-CY+#dC?kK!HsjI)hi*hzlYu`v}+diAwYF5~;JZ}I24w3l2iSz&VHc4DUbWVxL8 z-{=2fUm(NvbwB$LefJN3lXOZ}7@Tc3J^Wea?5pFyN<;#~S1(%Be7TF?Ps7#%n&mf} zzP4WL>^bk$)x^6-Q`4q7Ep%VcRs1W=zvGSGS6`)QY1d|B2Fa%r*4+`PVDzi14)gEG z(KofVelx9`{miQ~mrjIBM+i2EJh!l$@qJPE4E^U9=3Q~l(NpO-yMU#l&&K+Nip)@$2G_Xh30Z0g->?cJEPSt#ufwx77g=_m+32Uy;@QcV0(ZTxerq&z#43*1xWND>L8lqwUu( z*X_S;Pc2$^)3b21g-Y&ma4n~B`MvNF^X;2-nC?AN?JYau-MjOka-QY+_0I!uzntD2 ze)XN2fR6Zy6|E{dr?-f*^e!mktNQr<+QMefJ=t?J&OcF!YI+!D$-Sv&cO|R)_4V8L ziiTZx-uv(TmP_yM-G5VdPU57n&}Y`16UQd+W)?amPS8ea(=zS7@&jw(I37waq#+`=O_xA<6bbC1yG$xc@vUox5PbS-Fa@2PWs-yK`e ziCg;>&)XsK#WrhE#;!eOvrJsqEX`);@%!fB5WBGM<@xLVOWk(4t?FGQW>z%Uw&01i z+M9KE7`HrUp7U|Oqic3nn0auPRPEp8lP=ww$?U2^C}S8ZH)LWS|lJa$>j zUzbaKt$zDgzu$U3U&+7Vv;LG8nQc~2K3-mK@zo^kXbvQ6Q+v2Tw72Sdc+?^HTo&Qu z8M~K6rfS$6myi|R+^uyt^d>>oPIy#U2#uM4aQx$q-lKTs^rp-8C{$TfM=}nIwomAaEJ$tHU z;Bn&Tn+5*z{ZY%)b7i+peb{~ClU~Y> zKC>x*ANBN3J^#Jp<&Uq=7H!$qcg>zV=IBJ5Q!7>4S8H}#dw!9SkX^AlY7VdIy-k~< zeG6;6xFr|%&OT+q8e;Lti2qEHI}g9*(kr@8XMQa)`|P`N>!H70&qR`+u0OrirMczy zIY;I7HCN2bHv5^1M5s*;%fA05d&k5LD<>YGVG$s%u+F_o>*yK1TTA;Un_PQRJLQDX zN2!OYlgrOuJ?8OtQUT}Dm_5yJrJhvzOu8)+?Wj_C&+T%1(%Bt7JFFX+Cr@H4?92Lk zcf-Exsh8$OC(DGLNbLG#HfMwNb_3OecQ2$*e%Y6?>if2~!-qFS{@oDEQ7SP(V5t-D zfj5%s7U$P*&uV&g=jKZ*)0KOE$4_4EDls`z;;`WMNmV7UZ!MZ9oS*veh18dO5?dMO zRmS8U``EX9nc6$c` zTLwRM=k)!2A8ztGI{$}f=?|NZJu8KnKE9QXR{T8izU_-g$!&?xc_|nGA{SwkCY$n zkM{rYTA!S_@a4JWOA88*tm8Kp{Imbpl@L3pK_^b$%Xngo=<#xt?Ro#|4V@vcJdo8t-YdCwqZ?8xk6FB z@Z`Am<&o~oUvB$0u|Z7Z(8t$z{*~@!@n+!^>R}DgSgW-)2w&k9bH6)x%iX||w|f7| z!m3xiQ*M;Vn{n*GWA;^_C%@62{V{jj>{WVmm)<{j>U^o}kpgAj%Mv$a628V2)wU-7 z=sWFy^=RMBYHRJiXNr$<8Qe8?1L^E}@|#OLE?Ro}l}j&8H7_ytUOOnVxX$AI>!?pR z?(ADO-E*bi1fINY980A?9;=UfA?-SK{iO$4!E2nQ*B)AT>`&n=t#un0-^n_+(e{yU zKkEa7rX$T)<1K^V-&*xr%J$fr*TVOdZhVMi=Fsu54gEaT?(eF7s~?^;)fD7?;cdr# zD2cU`eMQa_Da~%%YmI5g17=0v4lP(|Y<*(2KQ4;WvE z=YOr{H9Gyayqr_UCABvC2H&zZ*>Z}Pw#Zt){nh)(neq3`r*pn>?e+X75@L31?$k#| z5_b0fxq7eJLTBeO2b)JbHMY8c%v&fGeaCHL|6G9yF3!J?)XU6kUUa=9{%-bv2B|** zB~#Ka*%)X2oEX{mv+%zBk$ZQ=_nU;Sdb1`c;quxr?ovf}og+`@_J^tOG<_!}*1EhR z-=o?4(%Wm?5pGOgB!sn=Oma=APe|UNi6cGxhm(UxSYOZM!`2+Rc>EvO5Q#l|Nr; zZ)kC@^6R1n{|;>szZ|#9Z@V#*oa@dBLi~Bs%a<>gco}-Eak=i9?_87DTxS;Av2%*J z+$jI&sh%G3oVUo(P(0}Azy0PvPs+Yn#?SY1(Ix48t<*O0&q}VlrgPox z__=qJEnnf4OZA6rmrGYD7@kw!o_kSa*Yqd09Y6Ql>&yyVa3g1%)8xkNe@nPm*7^I? z+?!Kt@zOOV=7rdG^Y6h~a`LlYg-?Dm?cM{{lmpxoV*Ls%AAFvdEtSuIJLc{k-xU>A z-rb$5>gSrC=;T-z#CkklFI!Z(#_Yq}wMQgdoSnansOJ~PgtO-cQZcW zFiZdQ#^`W4xy$e0tWCJ0AkNFg$`{#OVEDa}<)2B&rx#T<*8(E9@ND}aklbe}$)LRO zr^m_nlkY9taWO|PVO!(F)UF$Ct;$m$bGSQynt0+(b;xxq)81RrOD0;a-DWejE4)X| zNATLFrIRgh z-B2$0X1P2j(=0qZK4X@ek?4)1DXDz>?!V9Hhx!|o4D-e%Yqp>Yp;o@7bmMse&?^&wKMG1U6DXn(YeJd zqddNsXbeuVB|f#iy-3!c@3*TtZ|H$8E zvR2*yO7Gc+-ETI$n|XOn$io+3&i*>kOmQ9kGp=AH* z!t)=u*VVpQd+STqms1*xy0rD)h40xFo#B6K`Q$4prS^`$=2S_2{?8y1d;ZbcS{ub> z9%0vvRIcg?{|NtA+5D>hymMUsqt~bVC4cnV2rl(Hoi3+xEid?o@QM!k;D6N@qYPap z9ND{TsmfV{_$fs?OvaCJ#5y0hW0PFtMg6spHJ{rx9Ai!7ydDa zFIzw2>bjh*XNCQ@Z~GkfD^e_zuTV%UW&X;SXGI^$srkP>e=F+Rl_yKW!(P;NEIeCe zwc>aEQH>v7a4a+BzoRIF?B`7vvXgU zCzme0y=#eL8~Y5lh`zlH48LbRU%p|}6`SJBs>>U9U3z;rNZ>H@8D<^#xCVyWvuOvH z-Ms(f+O@yWlS`L5yqeG^A(tSTWcB3$`<;uiqWOC+x)`f;H2+=3v}>lGVoO@&;-0F_ zjIN?iwbkC+bK5>ys@N*GfA6VhJ-LQA`S?AN3(2}GvbP0p_|GscQ(@xyKdeO=i#ERd z<6CUAKYinNL(4a>pLx|yT+jUKNcikm_0n5;%N}mI&+wz}*}sST0*g)3vwpvoKijcr z^Zt!GXP+Ko5K1 z!9U6$>aAZByKMdAx9)Gn?GmTvF31s8Uy&~H_-4J8){$SObI(jTX}ZSm!-V9d#TxED za#t_+O8au=Ejo6GJBi1_!DdfY_$v2>F9V&kMXG8ysjjnW*x2~=@#RZh%m2==PnSI% zao^^XhqJrv%R8Ot6$H=wuhZgx^ZU2OyQpJ7pHFg_ZoY`#{s@QSu1jmREm3}x%R7_?dg7?>IOu7FN}2+`MX z{rlT)dc0!&QE`L2#~;3Bboq7n#6jh$+v4AE->2R2vA=uWg1!BR?W7zFA8~KIzUr2G zrTpc-=0B5`ZQdfb@tfX_JMNBs4b@@mTlbjeKb&iDd-2S5Y57La9`5_Hu4?}BrP)86 z4^6u#e2I5Uk;(U9``P6s{~4+>`GacrYOFTh8OthqUT>?)KikPJeLRPAleQcC&+9qk zw9NLH|HGJ1uY>#_-!$=faK`MKcCpFN7iW%WCErQ;C)zRV%A_CXVudV>MZ~fTb$-4$ z+7WX6vigTbd(s^f%YO&N?GKnP{GZ|R#;NO~{gm>%g|~lCnfxUtq(JtcYVnJ+e^#tt z`1;c7I`kx>Wx2fVma@*LC4qfw-Z03|buBKhD>|Ps>=e`9H%$#jE*6 zISZZUe>ucH`#*!6b@T_WVd|&@cEJH zr~Yj}J}&+$Xfpq@NVx9e=v#3Wx5KY1%u&ob~Sg)YYW|kxQT})9DfD6`iH&1SrVYf-@udvOVYTO&S+SWGWpiT8Cuhlm>MVf zXSpS;itONG^R!*DO21n5+4OI7We!?f9?4Y}XrCi|;Qab)EB3ybJ}drKNioZt=Ym^$ zoa74pneFyRot_{5+imvqm%O}3dI}|roJtQ`E?OcK6P z`-A@s8Ly3{Z=QU`Tzl{b|8v*4b%%CmY4v`6=={WSvgU)SPu5lQ*FU{-UG>p@QQt(9 zD79{ho!9e@DLbBM_5ZkiY2iHMk1fA)KD>}(ae99we6gULWrX;@D*FW@!Ee?a-SjR; z(&M>It6cN1dC%7V@ISQSN72Q5k-X=Y7nR?QHP?Cj!gkgB<{EkS)-_8iQ_ZW^-u$ee zA3ts4`GjhXFN@Y*ca1d_6+L>SY}=(fK4nLR#EW|#FisAPiE+=))je`NCs486;v9pd zsmy2h+A-Zq{7aIRNvw)6I@dOBxU$26T3mf1di$6K(* zE8*=GvH#Bg_PMXq&id}%l=Qn1U5~d+`?OEXPrCQAm%TyO){VCxxt}jj{1q7Q9=<%> z8PPQo6+L=JY=X&mp(TMDr>*BO2E@hPdiiZqY+Q_ce!j%%lgg|Cc3T2to<8r)Q=gjq zO$fBLbnuD@OKL)>d};7*&`r5E1FJ*>68&D5o3UC(*@`4-ZaWGv2Utp0dx z!m*9Vx;hqvj*LX*EphQ!9MbEwsE3uAflJ#yAZ^|=>yWQcnlH@x&!D*DU+AG9PjAOw zFX`tvr(Kh(Y9sFGj#Z%SC#szq8F zhgKY5-mvbaU~+G=D_0jofE9!Gg^6MoR|TCGHLO~^fO)H^-?#54|6YkXRXB&|?Rh`z z$v5h3wP$?H@4D^&Y0Ft7{jIOd>vaV5t&RWWzR;fhaehy<^QV%zYVzBDG<|=%ZChMf z`j7h;T}8Hq)j51tc^3SB?#}P;YW=GwABq!=Ws}ryJ!-@dtabIe}r_xkda{J)1^bQLYy74-_V`DI;s$W{ION2SWTlT+gEUVOfHSAmmD zb8DzA_cY~xuji${L4W@QF3;Fz=KA)oT=nhcEGixrp}`k+CRu!0v zpX7f#6@Tx4zg>N6fta1p$M7S+cYKW(k2$^k#m~E}J@uN2Z=|Fwq9PO znJW-`T<`g`;#WsL`Bq))UXWQQw=vpQ>y&nedu;PGyERplzb@+9iaHb&F8|n8@4~jJ zRed?@7F~HXCswC=dYRdTKYfyQF@LrDKhB$;5$(0t-7GvXtI$txn(gV`(jIm0Jp1+P zrk<;qb?C~r=|!8Qt!I>4tP^?}T-0{#lKjNO`lshF6}|4PwBp`OhHs^u!v?^^130efKuZXnl@=wfv(yMjESwtd2^1Jm*?4sg^Zz zT1b=ltT1kwtjHN(7>ayVHXUctH(4iSab@$y7Yt8kY1{ktE}a#~A*nX*KSS8Z+(Un^Uz)MD_>aMQ=FbQItUmuw#_89+#cO_Cd&>9LEtz*-ecN3= zHoGNtQ-a@Z6>-&Myz}Jk9X1F1E5G?Kt@|{ZpA{n^*v{mbsWJ$L3m)^98KNIrOW;m_OXu=#KIDeV2ZZe8@;=HHz!7XJzi z&tAE17pQzcef@p@n=jpMmvY122j1*ZOy^qK&oHT)Z+~m*>+jpYRk{6( z&8;}32Ay1M&!uG%wSy{mcohw7fG6|9Q*z^T5y{!rI+w=KK3Tw441N}kxW z->2m>W(sGdzi+&ew)Em!y|*T-E51s%?%en3_$@V;Tav#Ylu52M*`fFL%jt}-oR_MP z_HZtBy|w1<2U}y;_|;#Fx<1^$rGEIVY1`~a5|izmr87^+AG)*BzR&$}4F8e6CVTg; zThcvWXT}n{;}d^9|GXw%Ao^XOb%WjeQ}-`*z1hcKv7%!3foUgo8rPTniS{!-$xze# zWBn)7?GM(*WbKphjNNRz{oRtUj~D*gKQ;Q2-oDF83+B(WpSpjk=nK({UsnHR^A!m< zURBX?qVDW6p>nhHw&I0p?>6#3vir|a5cGGQUt%k*_ofZuI z$M*7l`=-sEZrVJGyd*hM`e>N@#|tihm9pMRv8VWSgaIe7xY0lf$RPOvA6Y zES6p|^)R_8InTOHNQiwhw`c#li${(AJ;_{TY$MJiGr?v>luvEhmqpWlhie@N}g`tBOU?sY=^wF8%chmyWCCzs~uS7K$? zP8JrHiWuxO+tVW=^U3P+^D8=U*G_u$=#@rdO>U)xhUJserSsn1yLDH6$6Te(k6WKu zJDyiwE2^-w;lSq3#h)UWHswis*jlb+J;TO2$5(#R0ny0?w;W%dweS~gP5T^tDuHX? z!{#3sT{>B>u&PEpZ9T5opp`LEVOPVF)iN7)JTYMP<(qJxGg4Hgb64BJVArC5eA}B3 zuP%J}xU67$(AFhF6C{(=3qt>WYN-15q-}@L;V8p&7XRy*&eD2M#uaftp59y;koTj*%j+{?RxbFQh`C5e1n%< zWmWoi+X6lH?L8|ZXNeUpsBEn`zvGoX*W%@Ix3icYRacy9ow+bwi+x6T&+op)AJ%ZZ zU;9;Zo!a)E6`npiPZ*zxKA2y4`@^EytGryCRB!HRR%W+g(NC&9$NoH~{b+9Pmiye@ z9GvD>-IWgdvCZ$3zsUDiem~|fU^e-|e}*@&i;lexso%V6!Ree=KjeM*k9Doi&o3=z zayy=)P&~mQKe%u0j|bIK3zud5Q~$^|Jv-^lJ-G{udXpm-|7Ym$oWi%{+xdsr7q0oM z?ffynv!vVCcIF=ahc}WUE^+k#3OT|4QU2liuP-vzYFB%#{ScB@lU!YTe8CsFL%V&s z4_o=S>;745bv3Z#=hM8D`qkm(5ASon*eAC2lgVn8&W@kY@@IMSuQq&g5q5WR9BL~wntw-Ej_+dmi@Q#!^2pH*W};eY6OG5iA8mwvV%{9YgO=3UIZAo-pvEqfaCnaMY*_#ItCF28@W zugOzo!>+Ab1wy3^+b^>%#}wKn&9-ZA>yv8k?xf|?=?r4Oie25)Ew4^IdvChwE>{J^ z?g=~7WZp3d_k@%c&yAG1`?5HCNf5KdPVQ-m=N>Rx9Q9nPoMruF@8#6SqzMw&9`J1D zD`zn735*I|8a)Lv6d&MLbpebn`b?8p?mOw&B>nHPjv^-$yGm~AWGekZ?8D8$voRnD>^z#1Aw>k4uji%D=sN{Wo&&t})?)unl$X3pR(FTpL@rOWT!y2CwXYkxvhd6(<$ z?ayENKc0nE@cE-1+h$cS+2-3MdCOe7@bg#ouZtG$kSkj&l*^(R2dZulSSk;ps|-Zf?nUH{`>9x_=T0n_6!A0y^&##^15W zU#uxtckz}T`>yD#e0uvf_qyudh{=ZTyr|#tb=PLL?nJ)zX6tmPzx{j+ zqTvF+e@$T2*40t48!aNEmhH}8F4k&jdLj0h!3_@kJ#R0%zkZ=T@#FF%w$cZuJy?Wp8HQSv5=6=UMX? zpIo%^kIavTf4Uc*ZoH~z{_2~{OxX=Q8z&#GljVOHD}VTJlJs7g8{4kz*mg4S_Lct( zf?w^H?X5rbGCg&v+WbtGvX8fWv=j~;Uw{2x*ZVhLemvU0>q^~)%O{H)mdu|jP!f7P zZ@Jyqi;*us9<483@ps|!NzWNQ?X?tMtnRz(SNnS5rIH&D<~)+>2TvzvHpj$ArbD%j(@k7t86%OQzK1#8u!iOp2vHO%>pz2 z30`$=DKe429bV+k<2$J>X=lRT=@#d#ujA7wdO6%T))ZtunstMTD;gfKYu=^XXSQ#i z^e*V2OW}t2Ba<(L#yY*K_RcrDcg}mKH>=>WgY`oGjv-&IZ|3OQ+{jhgHDN=bl!1Jw z`hrE%E{XOjtKYoDnP!)=V!6E~v(5LH*RTq14p6*nusuR$#?MKM=KOh5e7;I+2}gXy zG^ZKNWqkE_1ON0E?$Vn5zB^7~xAv!-w_S6V+n(e6=eC2Fr9Pu3{iEqi$!KX0U6Eg6 zhvlDStW|Pq@?_v!a=pkSz*B+o%B(L+4lI)xWY3CDUGrnI&xI|IYb%Xsl(WVwnN6M9 z@4(bNDO)@8quR`cna6&8N;i1Nwa+D2+gPrFm0jgp-r99HGIka2+davd?ZM)cy-ReC zFJN>$xoE|`vul4eJ2if>^qcTq+WT~6p3g?Ns)w8pe)JK2B3Z{8(_o;K5i-5R4F#pD;?nNv79YFpUa>`ay*zwbsa zjXswcclH+*9`h9~S{DjB+*f18y1azVjmvjTRqCwT3>~MvHYqml4$3&~*>8Kx>kYPC zUOdgN0Tdu@&X)jz>2%vwMEJLMVkXH)#92O3{vJ}kcTkEu)R z+VqDj9&+pc@RTw8+URMxx@uBo*?0Q}(X9{ew~9rX9=N*oW&PID<+*QFmTYpnx7fbo z_bjb-OP5Ez`jvZZeyzZ?TMv1P6yNdY-iELw zKdr{`p=|!c={=9lcFsCibUN2CX|<8u#*Gq(lbfus_r3n5d%d+(s8~75bHXG|<*M=^ z_w8&u-X>lCZ5neX<+tIh6CQVVK9Q;05`L+U|0CbrNAbM3?q{x2>d@$Y`%6NmAfP|% zYc6x(<||*SE8jRRQFN+w{uepH?rjy{l2G_Kh1q zyj!?pWgcZItiUQ*SKEo@;dJ{+hF;!XAICG}vX0ugzZL{^)e) zgIKBOK3lfGHjZ_wzZy*B3PuUbf9xVdJ`} zqB~&L$u0W@&L#@&(-YIUFfI6w82j>N)>q_|F50-R@J+cD-;uVVGhUoqB*R=3{R<-G=FR&Ox;EYQs;NYRx4hup%w9?BdlD1R z%qhRaXV0~D&5K>jmhx~WKeAqH_*lLAu#n~CzTb_<9|pX(@>{&(C`(z?(_=F}PTu>l zGv?&++MCB8x&}>m+j;HF*GAL1?tT-^-tX9|m@-{q@|-62d*>O7W2U#8+_VW!UUfdd zzH-Tl{4+(dH|-yPduC_h<$1wM_w`TZ+#-5Ea!tY~apSP5}4R>$$xjOIJKZ$Km zmrhxI!r@rNI?nIwPJKRXCbg`eZRRS;^@`njQJ!3TT6j`t7WnTKUAg>aSybp8vuS20 zc?2{(E>_-RPn>n-@|WGZTIZy_rBm8CL{u{Nyk*{SHY4KYw^_M!%%)Ay32<1b?0D{i zHP6McS9c@9O*Qc3B14gorxNjk*WbQ_17(a zsbAf6^;}ei_|g3W?-pO#lUu7gZ64o?EkE{!9sj*o{8drLqFvdtcGvw{j^_$E_(gV} zw)~n@skS%S^}?3({1Li+Q`2R)aNlItsr7v2f2~XFYz8O~M#d?9_}kw3#q-NvC)*3B z?84-$%w#3~lX!!#nw-v!_n&s|C9~V(^`*8Q-&d^5UA((-;Z(;kiADQrKgca!@yF~( zx82N@{`*cndvh~3isAeV|z*!e8!%Bdd<49JG*> zuK%JFZd&-!{m_I2)i2X|{Fn|i3;$}2DLsE|KC?sF(N!l)ejbR<&Xk#dAuzB$d$n;( z%eMP&cl;mhT=n>oSIlN!`@>aN?^zjaKVW}uHGlWYvLl&s{2z5cyg8k@(f^UJ=v2iW zm-E(b)qT$*-1yurWS4dcpLy;<8=f%R^Bp#k%TL#?j{cG6?HwsxbGp2QOPl{Oa}$3~ z`5IaGBlFC+RsT)8eb6Z4)|S*~ze_*m2C?5~NNBz-`(o`{OHqZQH4%~$DgPNZfBMhx zX6uC-!|et?#Ll_>{?4y*rq(n0pXlljlkP73QM~-rEO^^h zX!Y*b(Hwhj@9uc$AD)x`@^z>wZ_r`SZP|03K5<95htE9Je|@b;(uVgnVJlm=Y`3i6 zsd_BL{PzmGvOn$D_S}y8aeLX)<2Pzdm*2@2?e$awhU$Z5BZq;sC`+c5j@>hlMh0&Iiu8G@5GTKiS zty&X;z8??85%j4f$+9I(_1@yt>ecScUCZ-TP9HwBNi%WYPT9H_`RDe0Kl)wlp2ElJ z?SITgE}ZK$j*UGgVaCM2>+t8t_x5Z*zB#&L`SE)qALax(7qXP=*49cS9N;+4b@=`~ z{JCIu^&9zA=lFSBc{j$L%R9T|Oy-BHGC$DdYF z-H6pDVR?IHgfsO;Pq%bF=@E*~|Je8SOpWu@#1C?-&Ym=ESbS&0(IR<~ytR@zv1=k-S5M;nR!PTMb{*DKAyYS)g`o`@U!!z}`TUblUu-z_Kd;bTVc``hcL zJU+Sk=ZD(c3;w)Vw5fjU*10iLH@ii29xz*d;Mb|g^50L*dcxncPx;4{DIczNnCup} z5HIE{e7$qWpTVJdZJAG`U&1A>Nf53f5IAl`5)J|g$4RGiR)vt!d0Fe zm!7cS_xSu1TpDXn>*}5MKe=stNZId8@-uJ0IzBmW;^K;Yw+~h`=99aN78ql{V(skynKz2)rMe^iPQ8V8VsCyO23~Kc`F?#wQefw zy}Nyfh19>x6de96n(TE&IONjX_X5dhV>bO~DBUIaL+0vTyI;EQeof!>l&3O3wlBXT zzwq&gMJrz3+WTzj`g^w(+N`y6cj!v$D_Bahmoc2M-4#|garNC-yN>=zaF(6BS?%D{ zNuRVe6*uo@kmL7j-D#Z<+Y)zeyYgtq1S|HP5mSCH*ddefR<`y(1DAH_`nPLy6Q_N8C1Jkeu-8LI zi|=bsuYT>Vu)8b^nRe=Bo!y?aMM{M>yrhndHh`HBWdU3+lZWOBL8 z1>+OqCwLz7PhxYoH(+?)chx=m+QQ2^LZR~*RnDvydoH0cft#h>39~%audQ9TCe`WNyEl#h;`g6wSi-mS?fs>yy%)Dv z&UzoQb?4TJF4HaAIn-Y5>E|sx;#q){bDpyT@_#{RD4l!C$@!e-})x4JC6Q*Upeup%Z zyS!hgOmg7T_FAyPmP7KnNyw+W(FY!;&C|HCGC}r$3ZLim4ex_am*+Bz&U&PAL$SuX zp{#HDyu;V#>poq>chdBX;ily?4&GrBx42V$+a|5cs=_g{Z{l@KC^^U%a>ijrFpt_SMrp8+ugmF_i?k_&X-*p+jec=({nPr zP~>BC??sEm`959I)id-Sea>CpmvQpgk2?nEuLK<|W<8^nq1G4CSMYH|s>OGaV1+9= z>y~BYzSn&=)lp~SsWk_fn7I9sbvAmfarl04#oMji+LsPp`gE^o z&xWpm8^0c049hwTX^vvaz>5ZEz%zSg#ZAY93JryGLdpdt*DmdhSy-YaQySB^B{bK` z^4aCQMO!DI-cb>oRR3-xJNxH#R<94uS`&78-P}r+qCGi+6Q8#8_VF_eX3k zRI-`)HYA}^_53Nv=ld-e?Oyfj?DI|~NrSudlXHdZU+8JC`*-KZ^UO25SX&SLXGjhB z&!E1jEC1X6j$O5r>i;nOXL!4%{!seIIW_w~wBCAt{j~j`qOQ}|-{_j^>fE}wd5x~p zoaHgi6YRZ?d@Q~C<(}-!uPgG8Hd##Ir_e%wXfT$2?y zeRhWy&v{TUXmL^`tjCqdaKlsPmE=fdla~uKbhvq zx6L#9=p$g>`6t44wuaKZ8`Go=YIzw7hdgLU2C-5+Rg z|M32>K6gcS%CuWmcVgE6*%Mp$aq}z22X#zc+R+P+mdUbN<7{jMp0_Ex_Q~J4rAt-s zO%(&1=dzPZN2NWUXldvi`NjlVf2q#EGugG~;e%Se*T?=dNUk%@-m06LYn;XEc_Q|> zq5k!E^MCMG#<%7EXLzXc@yM^qdI^1Db34D;Yy69!T6|BoOE)=g(pTZ{8BgbSyb5^` z81MN+xWTn*o1+Cwc;y5?Mv+k7?0=5J{}~R{e^CrEo&S@a|JQr=f4yD{SN~@?zD54) zSN4BBQQBgTW|P;)MntzoY)yRTaQZ)k&eI1T_3!7;{;;9=C0l@eBkgRm;0~7Z!cv2`ygZGm)jY_M|SO9KCO*0 zH(#Qsr)jsxFSej9={{|e>;Jg2Og#2+<`jvW zjuJNvkBZI|T2*6xWvW9czX1aegZZpu2jsa_m>WOWzxW)P{?Bvn^4}NhUtDy}Jg;T9 zIXz!~I_seuGv!|Sl^vbHwQ6$jopf~t#tYB3nR+Hqj@i*>uXiuw%csjRJ5PSvUVAn8 zz>}1X>~d49Bc@rF>IP}-*t8NdG4tvk{fFyk2_ zmCRkWwD@)Nr7ikz-Y)sLZq9j@l)ZbFo(%jf*kJs+ziMk?YPNdcqWk&6`5Vh>7W_Q- zz^Jz3@hq*rHM$3{=e~K|b~blS)42o18X8_o{JR#<5$>7uAmMZ5tb@&O{wa1{RNUNk zPvh}y=4Vsh{uE(8q;bwd(spB{)nlh$dHc4l7F}Ynr_f{Zw3!DO3MxXG8yKF;x_fC} zxE<}g@<&{-OwQYkqqk)?MqF4|TR5lZ>W#N+mwwcnXg*ciI?~vf@%zlOg!w_%>9@8{ zx^>H6+S#DX=C&oDzx>`b);pqPTcYf8fi)XX8glVH`^kCxAoulOAs2m@pD$dOE##W3G&{kj zWtRDr6+se{THc;3Joxsk$E25M&TFB*6f_GTjh6?iNa6Uw`K7Yi0kdw0SIe56 z@^#AWe#*>oRr1@$Cx6aGtis{~2^H#%?&k|0r=%k=wn7wZ7p;ZiV&Q zPb@b%Yter;W9{119Tk^(t+fQ7}2tbc1C?Ui?(<&m#GF(3MsVC%{*mFxIu4722mdp0E^lvJa*cnbN-)R0 z(9GnDM-zjkPk9_rer)qrUBa$iaCvWywd$K`^KW@=TH==Y%;m+VWre%UwZt4Jv7hi2 zRbAg*QhjyrZ=3D@x)Z|Y-MQ(f9&o>;qbon5T9hR)|FNvttLp*tB-hP&d8WIZ{YJX* z(H%@n4h5e%$ZmPNhpAddr_Si2+U526i|0LEw6S)2;NPx#E8&K^jBS+-3D2zyd*)pU zO@4fzul|s}d}r9V@QY897Cp1JoTJCdvZ+O9u|m%TcK^VDI~AV~PrE1Cd}LC*_uqc| zJG;C+=enQW{)~|`aCgrJNmJSDeKoeJ3nR0O?|GS=zEf$gooT%CBtzaI*_kSgyCoMb z-ScsK+Zt1DYwo>Cx1?9^m{qa3$W57h&*nKvhTGNkdmcPqxbnSC`oq(_Nw#XQ!?$() ze3V%;E%}o|?9Jkf7Ekh&Z8rzqtvG*pTGZti-=+5M%Hout^&w?a_wjoQ3Bo%ImQSA8 zQ0(fP{AjOmK=fzP_f!AQV~$;W$3O4FO6g{U37TB)io0awvb^O*U(9OV`sL$&j-Z20 zce_vAySd;55BHjdUQHYd6PtfXmG190k=|?c?f2W4qFX2bEKhqY#mrrJcuiwu@x-3K z;-jL^>-4vM*tYmXc--y}XC_HA>P(kjI4dQ5_8I9<61}JA9T9!uo*b9>_;o%{>>uX6 zdq3>P!%zJO{ zpP83Z=$dS^V42gqI~^em9w*g%dLDI65Bu2Gxyawfe7muv?xWA|SobcuF6UCAUNSFg z+rb9^vwe(TR+Q#TWmu_v54d;Nci%;p>2CKfWnR1MapJR-Ny1E-haHdSO^puuxWD~N z=DMUkduG&5&%c$@b}4glvn@+sM<~B@o5F+UJApCjM^=`+e)2(WcW(K!1OJpZ-88ym zJ}FuC#Cz1+5a`P+#{ z)IO!Pd3Yc5WOBK;xT>Ho^~B5bO1pKfkIqtCSzDZ%UlsQ1mc3O(o??ZJ`+o*ohKV7Y z>bDqK%6!t^czI9sqO78O&b%+XvQ<9oCGKwX&Q$VAFIHK#M9;ok;`^eNKcYS#n*ROP z-Tw^jdW!QR^TdBURm{C9<*-M<|CF2cdQtyF)Axz&^34wTaQBFj%lC8FKQFdBHivO% zV;;vbzVN5Eu0bEB-fuZKebp;F>3Q-SJZnvlzWIH}E&YIzU3|k1hNpEgC5v|CchxCg zFul)vOWOPNO4Yu9@|!(6t)l&ZU*+HWL1{*hkU>O`ko9%`kIHcuw%)L@`BomDZ;E)D-)h7v2f`{PfpYA6jrj>+pq!N}1w-qmR~d z)$?uJFSnfW($%8tEE7bYOMBR~on+24p2TJxxV&qNrH!S0hw_z$wR)~Rlh^sZ`5>Yp z-Tl0G#@%^_^E6hJ{AUn*F>BwFTR!6Eg7+nA9_ANko381VRLKieTB)q?Y2Q@Q_AZmy zqF;|anw~wfe$sxXI=j1at=gyC%0-=`(UR}>TD&~!p8UvG{7CV=muA<^HA`jU_ivx3 zxmMjJvEhJ)yZn27CwGNMU621T>rQW%K0eu3_4tjqrk%2P_nhuzY6|6AD3z0uLavAkv%BTw_3czvOY|H#F-sUNN%e11qytWSFW?BHET zu1}KldsDFUOW^V6vH2}uL@$)2v`*pfw5#m+`sI)I>I)@HE|pwq4Gg{gtfuN$VC>Yq zYfD90&rV(YYSrCnXRDQ`I+J-A|77jG+Pm~cV0kvnL*wmdd+h9HuKFG(R=(TrdM4|! zGjBg9^KZ8;oHv(M5OOzCM@zl(i}2R!mN$$!*g# zR?i3#A&!Hx*3F^8ZkHb4efn}+=1#9|H$+n87JR(a6}TjGHM4`sYe#N_WQo?YN!E{L zGwU~*+~53n{%@<*bOis=5k@)1_xZu;WhEyt!G%i!*bjj+NXtmYNjm!`SM`tqff&o;9|Aq>hL;YAGw28GAiMM5Ush&*pGFWW<9hTcWx zW891{Cs`HMTb1l8o$`6+)6Lr_B`xLY7n-K!#&OOnVd9TVCzstgYkXFB>gH3gx6WM@ zbjGkpxX(>v#bo!($`yA_UYk~&&^W}(>Q{VRrueMeF1s0h>t>ZM+Esf^Ktub>qZzyG zW^7`QwYp#-!J`oKSmL<9pT?@`yHxDcZZ40!;XYGzS?5Bo%_}T#dB{{fXJ76k9I9X9 z(ZV&WcG|p-l(1RsEn0r;4j$|h=iJ$6Ew5hWCZxJylQ$L^_{%(byv#CMR>|UB&G*YCzkOGQ=&rcC-?m(K z_3>@IR>wWMEbnES+_k%IxW&ynafj8M^uk*^W%vwTV|IEF;DhBi(IQ)Y|JKDBYjgvSc=oK(G84YiXR&Bq z=&9{EvNNu14i>N2@=s{{$NHoE@*Md~HeJn`?>F0jcbgNB5eJ7#L!~5x+@n2#HOlhc zHo1?h=h_Em%U1hdQJ8kF)lq+3U*P zq^r7LE8pc!3VXVEf~KeCKcNN5H@LL>_ip`ECsyORqnp`l*HyJ?$Cn+KzF@4LyYwKl zOySW&@oD?hf2cnCDWP%co#|tI7EW9OqK0p|OrXC7Z~#Q$N{uA1t93V*!%OxZkj z))cUZ*|{|T(O9Q+=^3~G&CMZAm4T{v%h?XyH=p_3@^*hralcW!v=v6Y=*l8>g09 z^w|iwFJyn=xbA$ww8KF^c0RP;U8%cnkyYxQ31_9XS3YG(P5hcT@m0p!Y$o^Pys2&s zxk7jQWj!Xg#$52b-!o4yutDn7$$nW6KFxhn$^RL+-5>eX&S0-4~v+(?Mpe*EDUA(Ac1^-od?_Mz5Q}txT)Z1CDToK>2d__-IUR(aq{4nc0kD4Xd-}D^3 zk)ZanQFq=Q^LXdTxZH&w>JKH_Oqk;RI``p?ME_|w&ea~5`y#nk>iUECvsF|DlbrXq zvCL2PUHtdW)Z;Z1UCZxjwPeW){gD=(z+0X4T|As``3&7N)phv}Z(HSd*tPBEEfcQ) zSbb>OabB~z`K7wQSJ$uZyr7eH|Gk+9`|l-vQ?JAy{`Q}tSIT9ZMcU<(n{A6;6mLyY zKGnOIQ*j2rgRkhR`0i&H{^|0XSnf2BV6^)fuUY>1AM>sqy#ev_+v`Q^H#g6$->!Cl z^ZKbX_MAVQ_U3om4(($*{afytRwz&Mn119tZ^^`-jSW7n$1ls)Zk}iC8uRhp+Qs*3 zq%OYK)yr+2Gx=uijA_i$vE>?Hv_uGiq=^0`X0e)iT}C@?*(Ddvu3wW8 z_$OpfZhhF!?Uzeq?e2bA{kzsHJ3Bq%`}<%Ozg3k*2PQYZiAY;AYg?=Qk-us7f?40g zrp@XRyKd$CBx=pIZI_*tezyI%8}jF-!ldHKJt@p?y1|a^kIXyP{PACXFG6}-+h_kB zQd_=F`+R%GRJTi$v}8nW=IwdBR8()HdE=M$tl7a45q2w=-Hmt9Q>%3fYpQNK2LIT9+otvd;2J zw^YwOp-VM}%MUJkIX`pjp6eHL^7QhaW*3)VKmVCqEQK|?q2Q>Wh4qS+dQs1=Kelf> zw|-63oqN1iFPHX7T$6YZyI3M9@_XYQNF z2r$OHx|3Dq-|;7K-KOsQd^dhH&JVhLO(|YlT)0Pmj+)~QH;cPhnD(seuUwH)wY;PB z_f+wH5?ep!zSy@pqO(F?Bj|PW=N84fAiMZIj27BweiR4n|L|th>sxKN_j9|iyL-5p z>qvRTsnnl&e*EW*94zJaS=~HO#@$V?%v}-bT$ygo?%(iYa%Qv6(;q7VR z^S*ojX=Qi)NY!>;mNm=!!>`&u6CYb1nmqB>vpw4%wj7h@&2}rj@hfemiK6iwO-uQ8 z8Ea>~)837J0&u6jz#q2{@7w1-nDqW_`6}aY-agj--}_zvc&xEaK6=ZpYmJ_s%Vhqa zsp4nu|IIpo#dI~VnbCFO6}C)$o+jaq6~`YvTmI<#tdoyto%cO=Wy`kuS;6yvX8iTP zP#9PENM154>is)~{|s(h)c;1wO#3zOKZDovufAcw{-~6#nQ-LzvlA+hy580C{?VT9 zKh0h^``d5zW3RO%*$tonxKU=;yglmW9>Is~+fuHUN_;!nb5NKs{O{uTFJsl)P1IV} zzP*+B@%I+x;ElD4G&tga7vGWo1@50<-Ux0nt^fVL2G)9 zB;*$KJz6}EBX8}~?K`?G7fj{ylX#z zF1DF`{^4Ay$R$aiJiZAUTMJL})IO=c*QrMQVb~6H_QO_&bM}2!+|XV2=DFqVInVcs zW(R)w*S$tZd*w?p@zo4Ug2g$LI1U~wd&e$$;N=Ugd&?jG6_3ie9ePPe|CWMF+N+zY z8vHSN?6O+-7C-ze9zM_VtF6+ZZy9aMlb+pFNs&Cqm{53D~u6}Lh=?DKA zc>EHLeiVC5IMig(edAcoR$0-yIEM3%q49_8q$Xvg)$?cS>S(PN-Z~|9G52?;xqBEJ zdrB59?>-#GBQ^iq94ociatn9tzcuM1%kGYY)(6h{`PY3l-QV^_+t+iecCNTz+uxZp z?z(4nYwCXQ`(QkI^1RBkzUdG5a{KSkeyuC+w{1_(yB%#;g`DT@OfWuQr73^die-p zx#IJtk{_10bKTXvV6sT>gt_vY!hb)Sck3TNd*0^zvo7!b+!a;-GM4SRP_lbwqHCY| ztaIimt}hl$JRsYud%dlM`}T?VdS7hK9=$SHs5b4Ac`5hzf+H!(+n3rsK3pYwZ^^Ey z=en2s)~#8iw`8l^6w%+yDw=zn{yQ!|HoTXyqR z-#Bo*i@V9w!R{c@(-ZQ!qS|H0mHk_mU;Oej|HScI^KMGTBE@_ zWKXg2Gw{`FCVu$Woxk+-GUp``?#oYkPP(c)F{!BVq=P{f-`$m`AHEeenRY4M{qnI{ z&vnyuC!b87P!ud()lgL`60$3A;^5pxN#%u)-9Q59Eu@5&)1(~vs|?B z@*m5EFU|c+m%Z3kI`!~M_i4ATDwhR0DokaVID!9Rz{`EI7fmwbVzu2aze$-`w)@tt zC9drYwb)#pmsQleUW{_Pw56MyZ3?5})@>f$jhC$*&PQqK|8Uu3zOVd0Lrb;i{ep6z z`$_#qvbQ(gDgMk*`Tnf4e9InlVQcq?Z&jt-g)E~r4yLgy_RCDDWLtgvhx_5M8Nbcj zOJ)>m`Pl5%xGtntww?QX+vDadDZeg;o9e3jUdz39KjiXT{RQEC8}B-8T9G3-K|*kj zkny{$drKF;tU9B~w{qFd`!2mZgI=*t4f%3{hmSY<$`@blD38mbmvz=(GV;>By6cq4 zyd`<;BEhqLHq;vbHeHk3+dfU{w`OO}+U7-Xjv2n~s`-`LdT#FDxwgGqwplaP&hnVL z=$2vQmY#%!gygNB<;x=CZ%A{Wmd>7S#9;Q>WxX7`@>6A*gyfg50dj&LE*;O8`0#C# z(B0Dq7M-$jInFXwxg57XNYIW!-><#)U0r}j6Ek79m)_bg66d9*^F<4494 zjT+k}Q`*eBRLue(nIw5@{n^75xrsgY__Z$GKgthRyUf+yW4q+gbitp&I%R}|csx~V%>tQ*19}|n^D?>**-15BtB%-Xz2ta) z;r4T#I*U2=E>FE<_Ta%M<4avTm%aEE*UNo)>C0_phSqs0TN{(YY>k{6Cx@5ZRrkHP z_SSvx%WriO()BLhIBAq9#;W-m~l;vj)56cb8X1*_scx zj_+AintXxrt4gSEa6q@$<@9|sgeI(e#2)-YOS|*ht-H@^CM>p)eX*Xs|5%*WkHu@| zY8_(TvZ;7$X!*yoq+PFvI#nal(RAzoTdD4DqPDG8-R@bmA z`YBU?uW`=!bKZ1o_Jb3m0rz4|*6qK`Q2ce#?z7Af+gohR3uVvtO_Yu~{ylAOPY?f! z^9+AkORCCSO=9(2r`3^#pOS5wce+2Gq_KMbt%Pj(n`bkNmaI&fv;5s{W|j17yY=51nD)PU zwq<8f$R@ww@41bWj9FF+M`OhHq_~W8|l|Ovf z$5yx>-!4_RVV{p({EcM{ znEW#q%TJrM+0SHmrMbi%HM6&}#q;*AKJ-t%V(}yMz8cs1#U~#dF5f%HSJ}g4cV)A{ zLyu!uzrK!B{&4-Myud%D`1`l(4`u(I|IK5X@BN!$pWdH5X4raE^j=n+)Q^?fna?xp z?6+Of6WshQyH`s>^uTlJhCF}f?Qx5C?+reDU-XTx>$$0qXUB$B%&v~S^RnP}X9!bS z8^b3L=1Gg^@6q0VwM4#KDuuIhX~m_>k(0#keG9irF?$p$^Ok4lxvJ8*{S|>Bdl)}9 z)s`QbR1tqvDyJ%M%es#XyOqpK&m}8N3f*8~sqp#21F8QEe0KWVyG%Q6tRHbpe0f{Q z-75IRU0c(7a<))tgDj`z!IzgNKb(K`=*OvVW~psV*(UYGvG2$U&uxDnFc+<{Ua7v& z@(tJ5Szn{Rf_A}ZA1z(q<)8lX*>Ml=MJ5^SH}pabZ(Q~~_lCJ=bxzF0WAfMbg!y;n z+1a}<;okhWOZ>OtDQxEZpYNON1R@7+q_!e){?F3=YNJ{c6uNG zGqnFxb9@+WFSh$&xehnW<%rwU92L3DlMFc7Z4KvV{rt}m|1tlFuJNN?Kg@0)eR9|@cbONuKi0zpZ_!ZF|%5{^U;U& z=G>%x*PmIZ&I_KATXFK7)uulG#N+oYna}=-{HPze>tnzN|Hg#2*HLL1@AT82N&DF) zzYAZft{}KQTe$IAozjoVt6t0L+HrpnDezvgGO=`}pSf!e!_Th1mX*qrC(bh{c8%_S zxNdFL{>*Ft6n@O}{%Me6m3v`JLjnWSF_Fg0D~?q?R@)N%aNg3AeRbrPAL&+P9`m zK9#BFus!T<;q-q>MRsW?tP4M-NIlycFIaK@q3YAsuc!Au%I7|Fnqz0ip_to`ZG`{$ zxcQ&)d%HJymiFZ>bJI>AohQ>Dc=7#`=;|Hkr`u%hOYyz%;Xv=HY5_O*!nnnWE2Apr zAHEp>QmUmpw^Hs!-U9BumIn`%Dl0T`C_HFBTD8W!?4Qzwq}`WFEtR+L?oDZWcQa%| z@HvNx5zIH*;w&FXX`kGul{H_WhEa6i%EU*{c3)aP$3f`PsXN}%Z*>^8C3yI*Jz?yd z_w|Xqi0S^0dVW1!{kJyZ-`;9HitU;%7WT*`@!YXn%>@ebGm`f+H`T;HO7GaGRAYHn zXWgQYPOq#&&$JwyeA8ha*94#Zt!$1gHK*JvA6A@DJ-j4kvQf);yTq zm3w|_&pX+&FK^>wbMBcX&3tN=5q?L2NzU0Z*+ZVhrj*DNi zz4zTe&DU}=J3mf6|1GqIyI|7s)U3@GH90{(E=)#6h3D=y+lamjEAY(DlrfGyTdvfX zK1ZSa`3~36Zx6V>YDc7O-`Xdcm%6hg#^f?9-@oS>q317fef6%GC!MiqQl*}(h_K%6aud8yl``k)>p}lOeq1H0R{wXEW z*Nf!~w2#$E)cDUnDlcT?m{@E*+xk-4<>aD*Zu!$&_>NBIabJ;quj|{YZ?nFkRNA

TyZP%_JS?(DeyQNjPCZzh`)5vprA2&BxrtGPW@uL%8 z`rE6nf6Lr@HTh0|+ zlTL++Cx&G#wpU)gYJ+9Z!k0bICw6JY3swA%*tTJZH?P$#$!Pzr`o}cOTFB}X zaU#h*h~s(lQU6E%ve{v|#i`3*Ufbv||E=cYx$i0Ox+=E`gd8a-T3Y?U>_L&)>^jwt z>vm?Yj>)vDo1ZCj?N;Ahftl}43cD6+^9UqabKVGfY{5Hml7ZK)T_4Mqa$Mn&(%pOh zo0I01L-7KEU7V7MXX^JVa@d~~Z+@n{&gQU8wOxttU2je6Xzt6Z z#lOln7biZ_m&lrBH>XtEEIRt7T=u2*(+2rUwg)?NWLm95_>8Q=t(Uznw_;oRxYz5j zkNL0My^rEzPOH8Q|J1w1?_Na0(GxLq*90VIuy1(2B{%E&q1TVKZC4ZA7&`7awnpMaNNPt~oPo&X1KLj4kA=zK+O=d?-9*LxyXRfvusPkZKPOX3ucz)q zuPCERgCFD1;>M_Lb-J5<)5H6HFV6n$_c^5Rn(DOIOmAB^+0QzXC%B-~@quuIk+~4t zwz-dH1bq+Lda1;A*V=1aRd+O1?_F)JB4~W>4RcPwY&Q{e>k8IBt`nEO{_f|kXg)eQ z{^k0=3(M~Q{Y$LW?^YYLQqIvkK-+E}755L1!&pW&!{=4F*yuh&r# zkGFPJzPnz2bK*%Oom{oMiLP0{&YFZBN@#NB)$ zRyFa0&CF$OnR1tYl{dv+oFm$2DD17~7H=)W@^f~+g6O%(igULfAH7w#RVG^BbS{(p zrU>66hss-NKbaF6ES@&?v3Zn$M>*F$4qAjU)R}kGAAc5FVehevh|-Exm}&pEF?}Gc-rRmm0Ntz zq;qEbn>h6P%8#=g3yJeHI2tm~@Q?E4d0StJ^_G0wxb=y_;Uf3<%5A|)n%|OE`TzZ- zeK~j5!?%mGS1js!u;R0jCQsM$9Xz3)azEs~w2t})U34w_K2L4;mOZnCSyY4ELUkrH zIoK2lH2(b9L2Gk!)wtNsB z{rvCC{+CZf^?$_8digW_;kSkRFBku3__#}BdHtc#$}RhEzCF|*GNW#{CjYVXqAU6C z^Xt3V?ca0$U!$w&+J&z!XEhb;WHg$+y|K=C&i7Qwm$!D;x9>B(x+nSXqFeLZ(=T4S zed<%q-TQm*?k_W1yzJu7i>}$%^NC%qkenX5@%O3&bMG%L{k3n2($)|2ZFWIrCg*K7W*Zc)GPCfObN`{gfRmz(PC{xM~n`Pt=hni&_EcG?@r-S_v>ez8_R zWYe{h8>YN=5A}4@V^1o{ar|q)wb$>`JMPyJ?{b$!Do#j^2rp&%c<0b&r-v8Y97RHR z&02!q$=+A=!~Q>m>g^-*-^JCx__}_s9drFjzmMvFt55%DSUhcC(cfA7pIEQ2zrLmZ zm*Ap(ZF}}^xx7|o_r6_i*KQe3pH<1E^tmtMgQe*g(SM6mmqzY=wy>Oa>6(iBCZAYb zPaZgFQ&h2j!B*RF|BGwlz6bAZI(OCX%jO-f=a>1{=Y0u3^DD``|35=U+0`!=ORP%w z{t4H&VA1GPSX}t!UU1{}XK&Bh<$P!fs(AdZVSd~tg$bDx;!Lkh&Q)MxXjQ+kQ`YKJ zEpN@k5BDzBi`SIy{-`%M=i|$QyXQ+)b|^4BJkP26F ztcX9fYuZ)=)?7usRaM6GMC-*A>P#CS=L_<4{jd(Y#&FlfFp1?@x!3V5p7Ng^=eK3g z{L%P`)5i3{?%%HqqP|{n{@D0?9ElT@A}R4`DxT?{CdZJyHnO&+IQ^aDkfB{Kd>uq)j6q$iv;f! zUkdpd{c@7=7p1-L57sI9i&ohCTnkzEyTg6fs=7DL?egC*n0`@j{d%Lu_hGMfRC9*+ z4l$=yMS{DGp5JVgZxy|K?Q3>myxAY${|u8mF8pVBy|w>ekLc&}?)^_*l)V4-XVZU% z3qkvG_y1>@yw?BMpYHz*p4Ym**Gs(ieR!We>(`!&&$+p=k58Pq-=kOmz$JWJjl=bn z57!&E&W#egX=C;@DJMOr&q?u*WNWqhk+*gEJb$dOewB`tvR)|BB_6lx-9cu4CnNud zt`*1MJYH72KJfb2-)k@3;hmszuG~=1+r!iLobdeOxbWZRnI0dqcYQgq_wb2rtF?;? zd*7emso{SieBQsayIcxiNbTEn@!zlNi8<@sKPa_X{F?rkeW~ad(dx6+t)g?+zchHg zPwtAz$_pmtW$y*qH!3%OWoQ4e#$Ga0;PpM;t6xrDFgl<0u=wVIQ}Yit*tZtj3w-I= zy1)I_+bJDu>W-^ye7>c_T=Ux#F6{&L>GA?+*>2XNqf(a_38O5;(q9z zdnYpMR&9~S)81pwnqe$bum7pP{5ETQ=O4l4_YX`Ln=by@UC008l;i&@-?~J{@Hn|3LkSz^k;~AHM!PXjAt~p-Qg#tk3Us+h!^47ropum;ZP! zx}9|^z(C0Y_Iekk>gmuxTH4jc-$r3t-fZZ^UggrKd$usjpG-ECXP3bqF?6C zU2!YxR_VTli3QWergNxtg}gXbf8Os(xTOA{IX?E^wCDfTG(CFrKf}qoEBoKX_y1>b z59eQ0f3o1C|KH|Z`?aDkroIS$VG^$TQ9W+HPG5$~*~e%5X6T6tJ<>CbSv2PywV zkNyd``mW?X``*5}lUHR6>&;GMJjk!l@}%U@yXPB^7Tg!vDm~+mDPQYyL#b(d3wG70 z)H7}Wb=;@B5pWt&XrWag`M8SoxSIU)0(VUPEJhWnqY(; z|7YO3ePo{4^)I1H8p>T}%1`g>SUm0jo!#KRalPV?sf*Sh`p?i2C;gGFUs`q1aRH7! zc?PZSLG`zM(y!}etY1`O8+SkM!kW_T;NP=%h^0!NyVEnF_TYoR+ArQ32W-E(Wwo!{ z)x)>%x@>lHd~;Z7-^9(koBo~RKAkD%UEOglVy>UFHcyj1vuj4^s##sQdKz%?Q+qGk z-Ch@wmAm4e`!|K#x|2_TnzKP+bDq=9gU6hCcpk4j7kaMq+kC;u*X)6NoU?_GZ0mkH zBf30HVPWnaeZ@A0<4!lvOm`Fd(RV-6{JPhL=&d`p%0;%cS?Zpe#^Vv!WB20w$&h>- z`P}M9(re><-G0_x&-|v{?eeL`$hFC!&iCeNA-*>~kC%(iu71#M{Wm^r|Lnm1H&;j1 z_Px7Uvrzb$=#=`?wzu29DQ-`nQ}Srz`x@5N6>-^{&#%l<)6L^odUw~y$%ggTn*F=) z?sRwZd|^C+VbNykhf8}lTE)DsI(qE2$Obm|c5Nw(l4DBDT8G!y{ah7PW4?Ju<&&M) zO*@}X<52u8yWrGMqfVs+7KczPgr8r*|{iwyH@XllO6xUZhv~$pLo2E zP0#;9w&(L*Wmmqc_Wd*I4{DS$l>K(~R+QuBr#gGgrv)pm^?r~oxlZro{{3Rp-)xG% zsqOyc?hXd=!p%j~xOO**^7gsMOnJSZCF0KAyC+4z{$0BEP2l0m)SQcJf*0u>;M9n9 zyFcYk>VvA!e!D)n_MUzEXMXO!uJeJZ*>~;CdYlWlIwj1VeL!m!M;Jrn<;l~%AI#QV zpP!Qb`eSl!k>0v34=(-`_7HXRy!kBQO;CtMaF1Y@yyk)Hs15DE9H0>xQL%+%|Z$FcOOD0hI;KtY_NWE(dR!y zYkl5#^YU`3ck6AR9@*%pCvjjs&&J*rFE^<&yjAs3Xp(5+%iQ?KZhPdry!o+LpX{6U z%KGSTHP%UzQ*HL1+@`Q2<)>tXsKPwsvu5`n{@uI#yw&Sn2lr0<{C0Ku43>#r0m@UE z6qXpZvPKAgYGSat820=_!@OswK5l&*lO;AUL%i{!-fOkHr$UY;Gjq%@4PRC@$#?a9 zwxw6^u6>+6Yge2z%eGH3+DD=)o@hlzNSs>nll=~N*T(7xGxF0sZ%1FfRGjgxTu5i4 zaIvu2Gae};@kwbPOMGYUYj01lXs3;p?C(f!$z|gGbNK)tR@xQRG(LT%=pod#O23%Ui`YY zUiqZxmcmoVSS#$#6?40t7C5Y-v-7!PU$JYMRq2v#W}f+`f6sni{%8A6&b|9~i@Lx3 zdbX_0?9;8t<1E)V-#fAM%;y%qpDLdOLn~e$khQH`cj^+S?%L$G=#}f1I5yiHWZu7% zW7Yl%BB84`27}l9A`jsc#|~ZpNoPp`qv0bz_wRo7OA~*bI`K-5@9NqqOvi4{EV5pI z#(rvXy@1K#)jybHHwAF-oxO>tUa0r*SCLR&`Jn3E-K*_YI^JwMlKMh&=L3ER!}EL7 zE4bro^BuPZa&qS8i{xtl(_~O+u-#z!eT_b2#ne5{4|$CRWhX=)n&kcWCBtFHF9)9A z>$<*f)2=Ld`KGn?*uwNoox+&A&ODR;9-Vk-+OM?LFwOFW+M>{6fnPJ0LIzA2&YcQOu5Si0k}-Ih>WP5pDC`G1DE zuFv)IHFn!R_P$%GHdSEj)wh%E4ccx^Z)_{uzwmL}y5J*qvfDq(Ev(x%p`(=BwCCr{ zpJ{hE4}MXW%{F@N)gHTRZh^62+3_{1z)xQKD6S6)zxFY6*^m)L_4h?%iYP7 zf9P6konOd%cT2e0C!Gza7GAva#Au$Q=Ktsk5uW|?n(S}^UDsH{DYB_FZf9 z3!U6Nb&`+QR^40sLHNM;Z*`$lm!6wxp|a*e%lLarqsT*$qN{3OZlc}KHspj@b}!q%6S(*GtVpjDXQOL zQ_kK|W4j_#+>ZI2g-I$a!2Kc|7N0O!}LT>&`BJ z_3L=c`sTyR!i6$ME5kg4pHB3!X>Qmf|4!<;eR{lrC(o??E$_{b>-c_Kku*Wzlq09s z=L^l7`@ipfZTzj|qp;`lS9RjMk~ZJ6zm?G2#^F@(Snh29?{{6l#P+`1d#dZ__DA`D z=D4;0XNX->|0QAl>^-sn8PuJZ*I$1+{a@46&$<5@{>;s)zy2itm*1A~^?xF^O?v$> zy+gT2=+Nym_O7YPznJ5ee_VFUdwt8TTSu18WC;?m-?eF>{e+W$v@h+mx_#3A&|1fd zo&slidi#AJ)JJJI{GGMj?V{=H0=F+aJ5QXqNL8qRp|&&r=C;&)_N)s`=1c#5vhB%Y z|JRhQ{;gzg%8%+wfdd&^H5nz}B=nN1^#2U57i0T7Ol0kk z+>*UEaploxQFrvN9Xzc0O42`xk7M2HqjFN0|3p80y;eHNDPc~*>+g#TPwko?P;ck> zVAe~U_=mps$LeI~yI%=^Qa?}A;pf5$%v-PDN%`xjyjS$|{kE-h!ylCQrC%_;bm@h8 zpZqx&3*`kf*=@e1KKXc`*^cwUeSuf?Yy|4v?yyk!F8WGLXZ^>|fvE|nYU~ABWSVSVeR(VV&%3>={*83z-;DL{>z&s- zF51(4be_6!P1L6D-@;K>zx-$5JCZWdWVNxSc#+2K-4hr>o)lm2DP1pn{ZEkUkA2hG zovxc!^Yb0JaLi=4vPJYxhIB@+Sf z6Wa~^3zv%aay>D7Br(Bk+MmLguI|6BeoTJ5Pjc^vKbO*C?c&Wu?LVJ9=OOQYr~ZM{ z`hVx^RR3N1kn&6CQr?bBPY*F)KUuTsZ^7;38;)-+v1hdL{=4;o{+pYrcT1O?d@`Zx zr|)FR{|pzV|6_Ko_O1F_wG?XuLG)CO!@k{@zHsEfEuC4iQTFMcV6WAj@p_&UWj0Qf z;L#T1mv}$%Xi=Q)zFn7fOkSTby}6~s_e(&`sYSf=VtSa3B?B9788;tz^4unQ;>TwT z3zr^W>^JMrEZZ&HU+0$^2;|;4bzF6+8<(SF^MRjz#mBf-CO=v$?zLs_RsEU*Q(ZmF zpxfu>zj+##WS%I+5_>aDQo`bvKZDfz#7BF@JvZ;ZlK&@bSKF*SDI0DN?WE+ZaB!Jv2ePMQP~H_lSf6D z^Y`tsE?)VcA?p3pdx{ur|A{}&ga`~$;Ut9f3 z&f-Gq#jnn6dv?`KjM#c8&Nt_GsPxmw#e55V7Ir;Od*B)P$Jz92z0`}|MOpjYJH)+a z-MizV_k2t7#)zq&$?gGaL5>Gw5AK%BSo5Dj_{S+(<%8eZE{Ex_S~o{8=xNEz^Z>3W zo=OTwqBj)oY2Z&1>00`DzT~pbhqms&t%FjpF4!!rb9(v7f)x=pOTYC_oc4Cl<9!^f zH$@;xK zdgR*&lS^xNZ^<|(&gLF^Q_k~oVcgc8{~0*_W@hB-zR#PRJ~1hLg2OYFi3e`2S6X)@ zc}IdF%R9M6yQ=$zGtFi#sVe!-xA~;fA?Xtnj;Q4Cmi7`)+WD9(LU?Dzla#fx$w#dG z8n(S%b>B7n+f0{$_LD}IZy28ZXZW=@>)7T;zVe6PXUeUtmEE&_lEmA^3Y$+INNEja znRj@3(Bjn$zVSzH*|ipD=B~K6##}tLbq`n1C$A78r&Ud!kD2>C8B!*2h8*uTG27Ji z%DQ`ISY=A9V%Mqi`86q>jI9ADc6&V6GMxCzIB-wyg0)k(Ub^eA71PO+$vWrK>5g+Z zx!qR8I?dyIVBo>F%KLFGleTpB`K{-5%mwzm{8nwbEKzEXvC$$i2Y&Vf37fJ-i+6pz z->z%&I@j_2l1bN|UAd<+A@)p9d)PayxTb*P7pcw_2~x7 z*Vnr07C+jnCb5-+Z?Wp^+eg)2pVU<`a5-gAs2Hz%)Bi)8xfxUVGm3=$lznpYDlI**NjjMxTJ1jjKfi%z8HdnpC+apR;1;vJdZOmF|4D z^U8ZYy^FqUd-}MJ7F~SOakWWjW zJ9nnBw48;dr%|V|$X+-!Z>zSn7dLm>Rbai4&9M z)Gl*xWMH(gGzr~xf89L7!(fp9;?mW3_RgDFUMQlLsodFh%=oiR$)lB7-}y5a-}9GI zEBBbO#@XKRxS!UvjRn_cnLc^RId_tplH#A3l0~au1f5i!ry3T1tH<$;SA+YrWjpy> zgFWWS@4c)RH0eg1O)CF~=5zKo*QftcnC0=ua>YwwMbDD6ZV_ej4_V$Ze>DkPT3-4h zB+ykPr0A>&qt7NSwuJ5-k9Df}*ShOmbPzeMv>=URcc0Pm1-~xGx%B6K550F#TwPLq z!LN%Uvp&7^jjGmA;Fc--{pTm+?PtQDR(zF^H1%As_vGsGO-&3NkLT|{KS6e1qPDr@bcUYa>+cSA}&Ay;%0rn4KUtOEK-0#|xS08SAo4udnwsZ>r5AiC^>s2A)+Y9b^ z7MAP_V2 z$%K4HGisuL zRR3q-6r1zw+;qu*mm3XDJPN<9Xjk?B?Y#YIOndJitzCN;?%&c=yG!&O!+(Z|ee(lv zZK&~Iptm&t+X}(Dn;fcs#`FK}6+QPy=;MaIvJZ3btjj%_emP}I!pGCD=L>ItT(iEt zPARX%`}kp=8AerA>%YqP27Jid^ltvm?LCW1Bm?*g-~RY2623Eh(e*^6InChMYZf=} zG_2xO`6d#6_CEVNMSEG($Iq6jd2bZ+GoNo6&3wJV|AN%d{SQ>n@P9OqQwsU}b;~mS z+3)*J{pI!)$6Ne6TXbxn(zQpK0v_L1gmOM@$n%i>IPqtf_KV%SPe*e{>aEvH+_tSJ z&2w2nNAs1*2i7mrU;pOIX-j?mcMrB~?U#Nw&rwRxgIT9SQicE6zP2l-+OleA-!NWOl1JTP9b z;?l!^`hQ}&E+5{wW!q_~Xx$UejY1psjhTBi^r}7;um`c$>|OW%$MKIn>$;h@&Es;p zJVm=h;LoM)%Zdv4__Zud!Y^~=mdNtfU9>u$nViFX>N#Ve_@sKqc>lP2kuRc5u9pXO ztasiNsKd%*SQwN6ERWvgY}8h)`de>MqUc)d3Etmxj z-f_F+MLTQ9Pu2JzLjM_Fxb4gQc)P#tzFh9s-=#{|-mHDHapCV%PnH~h<&pMlV!Uxh z-$Og`e;1TP`n8we-LZ3~!2GYlDW1n`4APiy1jf7Gh>y;ys@?l_&T8GQN3vCdG#^gs zX^!`}9doccllSVDmAZRQY;skSS-nz{Mc#?GR$pe~Ss_qI=apovpEL7?<~xD$mVf-F z8UA*Dab#Dc%r)zCb?@IZKEK>N>*wL4pN`pQ{$?{$;I2+tQ_jbC<-zr{+dm!u&%hhf zySMD`F4c1t=O-+dkNNm@(Z2qj+uJ@}%g*jNGT}_GOh=W`#3p;KU-|(Xw_i26y_$Q= z{oG<kPjIc3bglf)aI{}G>PP(HaH-rqO|j)c z9fvcs<)QzK(rqpIDQ*v*@V&8MzHhr?)qb`*`Lhr9^PMeS z=_Vd_g=@>oD~BcIwQ4I>9$(?Rn*5NxVaW&k)*1H~@kae@*{KT=kgN$7v({yel!;k-zl^^cl;A%nUs={m z|IiiNnEEm4xyQ9XS9f0evNBi9veIzDR0)RabrBP)5_$@|KJ9O`zjfh{(^ZqV>F1)? zTKoJ~=*!`y<Tf64Ifk4!em~{oQ`&-#}%P_vnzHJlz$U2|%N7C=eUE16383l44pQ-pT zq}EUS+MnV_9X9e0x&J1G#T%L2%xn42ux9EG*FQh)4$RV?_akzdcjki?rq^5BrLK$!ET4div+|Uu~t_Co4{$)j#8TT#}jl`1dS(h79$? z_qZ;vd6N4*Nd1=Hl5^!})z3V){b_N2fk=3EY-!1wRj)#)7)@;AIHA<_b6WL^ePVSN zioNISd+(|_!T2z7Hp}yy@pGzFYfdhWbNsOS@zb|+`V=qesTkUyZSB)6c;UZklGQi1 zeQ6&`|HSjyS-z5zx%9It_RO&hll~~}lM}P|S)uZU=UCUT)W}$8ZMGmI4ojBs%cWng zVp)$b66jhy&*-vgY`pijt(O*UD)&C27jrkp#DQa}Cxdiy&GS6#WXtDEi|6TG-rDP@ zYicq>C*`;8oA4s(-OJnmMze959s7CSW=?Y7&c|~;#?4)mCG_I`l7q?mH`cq}X?pka z;km1+TxQ4qme2DuY`ktG+PnTmo%Z#Ty-rsmUX-Q1K9#h(UT0qN>cx3!pV$mGwlXl! zc`}bR>Gdx0uCKebJEk3X-eMGc>&}d8|H6e#^46R=|AdL}#$pGi?@9g)GO@`{ALCX{ z+;lUe>|fl{%o>I0ml4J;zvoQnX4n6v-(Beqy;q|^P0kC{A~M2nOxc8gE2Mx1iKH}@m&>r z?ET@6z|EJ}1=ZGdh`lN`+@z4W(P`(M1>3&eag?shpWc1VMsJ?zkK%Q@_ohBv?Z5eD z@vUXca~k#ZJ}nh4(AX`zFQ(!BGz(eR!Z`I0uYP1dT(%?Mq@U;IC%t*sR#zuAs_jm4 zGvd-aRQ6=u^ONt7tNLdy%8t|iux@7QTm>B|n`zg4S#R$;_vwgoTG^+64hw`g zcCRFVkM-Ig#}B`*;OpBYD-|s*oL})?=1@W7lk!hL7y|?v%slPQt8RW|Z~b~+GIRU% zSGSCOOP5MZ+i$+-sbb`=y?D!6_30-L>0RY=58m-4VYT27|D$47X)80cX89g^bxrV~ z!K_aqzfD#%v`k49&UX?qH#^ z%;l`lZ|Qxp=gBg;*38npeCcBAw8gSd?q|$gd}@U}Tc3;?ysiDW>?mUl|eLq=?1Aq7*jmv3zrmOVmP_EOfceS?bp9!8) zy%l09SaYzhW}$R+nZH8v)aHZxw0>J`-P7-}bHk+So2eVKf3`#~pWx9~U;6v(>^V7! zTjrR@?>_1NAYWu(x7_raU9MM`PcK=U^ZjzAx5}n1-J0>*LT`8*dKg%lnK&P;%gQx+ zefY!6U3^Pdy!aLr{n9>ZcHOGnuzzA@Kb>uNKVSIJY*YU4;|0f9kG5Xx`BtU2ymE2$ z9xKzWy;nD9ExVJYu~y-R`EA1qf{h*DHce05@g!O8ZC=UjE0@==Gnw&D@$y>TcO?&3 zo2UmE=G_pCx;ZcA1N-zXPo`LJbj`p2Bl$?w*}j#gy>nZaMBOn9HZk8Wq%v*2K+Lhw zWJ?we$rH+R)@Qxg%Fq2{dPslJymPnx+jOJe9S=5)i`nJPd_+-~o3*z+;@=1L7gg7t ziy!I>o-Z!AwlVW`X7Re{?BuuNdYd&J#a?)=$yxBs_%oZmUlG^N`iHV*kG_Y`Iwos( zQ2*__#j`B7O7?{3Yqf+=QWm>$*MVi%lWiQz&rY3O^+P{ycXxchf4BdJ{-EtUS{L{_ zr8PZ#qQE)JSWK~Cr+V7kBNM9L3B20J`jPeB?6o&hh^`%;yOl z{XK~xXVp8N_(>cmwao)R=y#P$Gv*fUay{6Yb9S+3wb^ZFzC))vp9)Uby1(b)h9?OP ztd1)yjE`DPUNhY=I1*Y)H>^WR_Ry*?;1i+SZ4%{71|K@4TAk zRGIACbuU?ut+$>72cFZ5Hcdbv_eQnE;-Sag5Y)PAVoqNvXChN(rrY1jx7tZ*b zUi>#}N8o&9r`X4Ij(|2`T1I&uGTa6eN; zMEx1A(;xdy6YC^@s9IGd`~FE}>&kk$Eu*5PPG#0CefA$kY`k)vNlRp!pR5CUAjG%7wgbArl6>M`2JdRzw8+gKhLpJ@Of-KL;ve4tG&Np z?kbKs&OhhjMW%GgI|u96roS$KzwBkuo#Xto9;{+HXS}oNKSPw3cKw~KO~;!=KYuy- zv+-~Ir*+;g`ETZTRlS?{U`YIkvcp7gUjq0e_2IC-_~!_$&LHOsXJHI^GI}~)~O;!<*rrV_TLsWTVKgC?cAg# zk8}@AO|@j0c+^{L`rqa0*}89nZl9R61|+Oi#Ap&WH8bokT9Al7mNWYiy6XI+-?_nu z&TVkDxt}qEN&L-ceZ}jxvv+H8C zU->fQ`dgMi2lidCJeRRPxxs7ON6wp32@M{n47iggu`kQs*JIJi(jtD~j>7@zC}bolFO7uGa2-gb%jwp;sO?*ZkZWQapK>W5$Kz+`>w_bup0{l|FL)^N%F#O$rWkGsjQ8iVnzwM8 zdBFV+lW@iV3|!t1pa12)|I@ngKf`hD$^RKzRX=?G*ZTg?+#UZJ9#1`8e`wN=%Kr>b z-~ajDssDIN15LpcGWg`j7j4Y*pS-vgx^&W%khC@f zIfuRG+wUcxeA;O<>-Cm={)~Te+YK^)r7rDB(a2X)So*Dp=}6J>J5N4kZ@g4utM0Vj z?3dr{%w*w&O}7uN+|k&xja~iDXIIxp`dk_Rw6}W4rg6U0y*^(xZ)vCWgt}BW#tja~ ziuVx}TR@nUyLyQP5*>UVms)py#6FZi)D$+zfQ_KrmgOX9O< zSsau+oa4?F{Y=^79G|gk_=oR}d+Z;&+25FU-K-;GV@tf>MKJ}YV`fE59`JdbXPWu& zJ?o1&(Vry|dTW!Of49$@`{?8`|7qvKZAy=2-jn-K*ZnZ{)0d3AwGVHFI9$JcQs_yO zNAH~#ma-1#ufi1h&>zTVkiDV|{I;Gza`I}6wPwMM`nfdPB zl4C!YU&%`;JHU|IGt+{B|4`S(i+^|*{HQxrEvqJd+B@<}TbHfdV`f{gf1jV^*?oHO z``Wsx4|=(sFRNWH*|9!1=c$CD`^uPxJWFt+M5FlQ7%s2hTIb$9k5QFZM1h)Xtj}S;{@V+-}KCBi|(d z59tTO!-ej|WK8n4Sr_N2vVUhaI6PipvF^Cvtzs8!PcKk;_UM`C{rTUXq~%}hGVPW2z5eB9$=k3~ z(z`t!d1Y45DeP&!8j;>);GP`tx{7b=BU{rgf0K?$pANIUsrC1Aley-rwM~^Qa=u~L zYpfUlnC)*Sb!|TH)18*w3Qzr4ZFuUxL*ZgfNIi$iZe^7!`%atS#Wy$SM_cpyEb%Bj z#~4(3yePZ3`yt=l6e(TpBmZ>JQY)(b|!}D!5s+PMizwz1C%|Gwg%QKrTpR234s;gT&oUgj`Vf*3pycHHd zcZJQfyKzFy`{a$2#=qtsfA@SP7lYlD@YvYck~c`HQE_#R!H0Q!jZUwQjElEQPuzLs z-IAO0JYr5B_$GSnRXo=p#WklUz1k)|{gT`6z;|1{S9@to3o9(ZLP(c4!~JaxLkFRaM2rtv?># zTw%L$>%{AgN_k<@?L}rEYB^;-_M9l;wYq;K-0F=%W%>629pPOcRJrPc+>6*HzZ`gD z5Xvw3L;mnT-HO{WTjsq@K6c{llao^ODsQY4x0zQs>)c2AhCh~%ZeEHlI#+#I{*H6C z)uP`m;avRtjI#BQ#Pin}KRUf-=XIU5JrfxP+xN~5vT5ME63(^rz&VCqignEs(K@X9a^DY6H*UKu_R6|;)oPLJll@kRTX3kPaajJ7N$+=b&GD3&rrz7l zTB_x>F}V4KyTb3T%eXEQ$cVYU^zN2vpG;Ow(hM{7y10S;;*_@FcN`|+f|VcY58P_K zGrfG5&2@7lnLU@Y#P!}(6&>@L^(b#2Uxm=GSKh1lN=0{fMwd)Fy&`7!&aWybCuMic ze{jBG-8$7BJ1(lss*Kq)$+N!b%o{&Xm6MY^pPPhxE&#dyYX6NbE1q{rR}BQ! zc23G|U|?D~amK0!Hjx5bjkQsi)`E6~YoPA?(9-Vi(A8bKbLT9Jgii01I%`A&-Bl)4 zFjQSJ4==lBwkbCDj%$9t(7Q>IbCeXms(fH|tgOyl9_3kHTgtn3OR~PnCl=S|G8-Qh zWvrQ6vwQzR-)$cf1ZNcWJ+vIcHu2!%ekKZ*Afq&i=jPZO$L}B{6qS70LwM zJNEAQ1%{iZvr`}4=Y17kcq`&ZyXUT*6Xs4hSS`IfA?2a7PzB4y=+dpxCZJi9_^j=} zBe#5h^xLEV?5FbH-^RVSs~+2=EsWlsw!7MON5`E-oN@w>t5&M8XLE9r*YjwOH>|#52>F^46cFEVxuJW19Z{t%xFaGUz-uTo#^Ec5emlbA(cF)BRx zE#fRE4kI$PJYWh+;#nD>lhc^9Wzui9+8kemm` z4^keKA9&r>`7ZbI&aRu!JM&b-G9KwMTg2oklq%%K{7R|1;`Z!L?AnF%6pK!zM0oc; zWn|b8w}Z9e{En}Smafc-LW^?pxLVVeuHdzadSb#O)En}Tq#q0B>&wu92_ZG?7$=|fv*Z3$! z)phBKrU!i=?SIR}_pj2LHZj?>bz*&Eg7geg1q1%plelG`G1D& zox9uLd_IzUKkcPB*YQ8ZSqmN{N^s# zLnp6KwLLBpdTM7lc57Cw(|xpi`=om%Las`0OKds5fCP4RpPI{jEHYx{3lD|s?-5PtsuX*4DPr-|Q_3C_7OO=tIow%& zj?b3g>+zQvH^0hD?FsvE-X>~I(V@v6f!=G^uofKEnO<~~drz9{%qb5Y!*bppP~8{Uzqjke zhp+!uudP`1IO1aRr8Rk9<@qDyVr-B2y-3m3ILH+;Z^pf71&xQh(=w+2ntQ*)#(8aq z;mff8^2IN2wH$h;o?4#qEMdQ9kny~jC%ucG=6ne4_@{WMa{evjjBnc7F9U7-B)9Kb zU;Lz1FZ^kDy)D0l-PNS)k7SRYJsrQUJC*1Ath_n7lAT7zcf4MRg&#PQ^|XW8QquBt zLd=$rPp_W+`exfN*WRsn-$+G8+B9}g+bJ~h`38omK89RYpO|_7h{nD99`s>-e`#*e zFIl^9-(uBn2LHCZ{pMME^+c}s_mxwcLQ*#TbdX6}kp0&C&^_yqyk|Z{+9qqa+tYroID+NdLTPjX3G;g*lj7<)Dz z2-2?j(S5WavujrG)TM0;bsx0dILSWiI_HM%C-X(le`eyaoT3pBGl5NeLB3$cCRr_( zeH*uKyqdPVVximi&-b1g%$~Bu(8p=c+$DPshVUM3uwX5kZf7*LRG0aP)Y%5_*P5l# zmix};mZ2GDzVWHAJrD8Ow3|C9k_YIan2iSrVAcht0wWL z*DtcONF?u|O3x?n$C_;W7ysycc1gli`z_a_%Vz{8^A!DEXL0eF;y#b&BG2Ro50*(r z;j1$q2mL!AepO#5Z+VT;$y8=nxl8QK#XIwgKc}P!9(ezHS62MeS-oDbu04A&=jXHy zCYpVRv#sYJw{NZb+7$1ev3kXw=sPh_-c7p4*2?j=HND9G{VJ_kw?~gI`_HTPHrq2- zTr;mv;6yE>4gay}fjf3=x%_r;I~9+w-&%F&l~hlZS&+Q?tYo7-+`eevB^wkR3)?B;r!^v|kOQY-k+lvFjcjwytle-%9 zJpQ9e)aGdChn#9#gjmJv^LicQ-fsE)qdL@kZvByqS!Z+kTU2IU@DPdpyUUcddfVmm z58t}RT`SeiJ9bs&vWekJulQ^iy)fOyvPFmGTS=D?*%h{UZCu39>%WX{#N;WTFlUOh zmOXvkgf!u>yt(Us?0Oy)kzb=Qt!%mFj(-eR?~uOq0IPp*nZvxjg$VLOxYB3({w4FK>KWq89PZtSa5~@_|pg zgmtQ=ZH;HWd{7z7InB0r-psS2y52#ctG%IPE@)#w7~GJpx=y-w*5$K|9z|XfJ)U%& z;i|58K(~M2-Df>Jrj@i7-q?7+boSc&u8HZwUac*gmuUAL?Pu$n^ppFR-PJoMKd<)XyL+P^|1tmcC6 ziXUa!){%c^di~5)^Ji9OSDqIxKR2Z!@73cY_xW_MNuDuJli$2$row8)gEqw%>b6dQ zaQbz>XpQ*>L!YEQ8LeCTJXvG;F2BG1bu25C%xZ@nSFZpoEmf6+Z@o=-g{2$);5 zO4{)Giac;;aAc8l6jk+f)x`)!A_E(uEOg0^xq8=Bx9SQWOTA-YTe0d=jNkNZpumUzar`S=*Y$Z} zU22BMpLwu_893-heLi*3<(S2vn+&Ipaj>qAtBlf^aa?}3f@H)&hV9Q?#XfoJaC|!Y z)Wgjy-MU^%_sOK_lFFn{OO#cO=kD(n`{b#+%VuL`kl-|*=lPFay@Rf7LD`7{_H~%+ z^0qCvrGNjvU0AVQ{O50H(=TtN?=U^lWNrvyVrORBxYkB@>5Y9$zP+wkyE6T)&-Kq6 zi`?!W)+^h_z_5;kA>o6z(B(r>u2Z)y&S$#U@l|H_f?utNr>i7Ph&&nOE*T@ozF^U! z?t{N%?Brv4bycjYJXHG3cTSyU&b!0NGU=7$lZ2kKW%|uuTt58W*S2j_H+OoZ?)A4p z?^~xl_43?ZHSso&fyK+C?1!v0FTB}OzxnJpYcbyJ?v*YZ1n1;TFZ+5xg~yz^Z%$y~ ze}+Tii)t6$dVYHEZ`U0Y+|9rB9-sNS$1v@)(QyWzjVi1G99)LN73$8}EAo26_h+<; zH>yhY9>`t){gcbgG@F{^x}VnDYdsSG;MVk=>A}G=HL>&AdrP?+eDaDOO!za+W}3|; zMs=mGHhqSQBJK)>wp;I1n#?+RqIwS*_RcPJOH_+2p1ie_S)F-upz8yE-h=rp0=Ht0 zix~W!qZNOL_hZUL>4*OrDw>NEqZOFk43E0aWBl->&{Zh%+(vh%Xo;h5eyetJ-kP-T zkJsWt2D^vNeRBfC)IY?xugg=Jm1|*8ZJ3n!ImSsrR=`C%fqBJ&2cIlg=KbM+c>Cx3 z)LobV+~0OuVfxrJztTF+ZQ{A> zQoTox9x{7kS;;5quXUKUsKCuvRQrdv)3mDz)1-HDd2Lqt6o1F$#VL;1&ElU$7|u3G z%NrM-Jaf51-(?<0@-*$^6>6U!@33k;eoOS*)JM7h(hQCyms!3H_I|KeDC6zbM|+H- zPh4^@GUaY8ker@*=HQ=0Tn`eq88E!jSh2Q-`QdHtAMX>k?hq8&*^ZdMPzGOdpOWBuXiYRj%; z543Ne&PvXHdCOic{%eQ8&VVIz9`Nn3by&3Ji!ZzXhqp~PUUu4)t?r$8x>)BRg9KyE zvE0Qs62kWNF1nc^RTPnV+A3-plgf1?-wg~5p+?4{0St%w*q7$Mj0#=Xn=mb`uA=HAQetEFX+m`LxLBe%Av@pa_)TVMCC?GZQMRr>1K@j_m<_v-AclNGA2 z#jRZDn7{P6?X$U6EB-SW^0K|x21{Iw7CbT8Ws=*$iQxTk_qHidRO?jjZB}kLb$B8B z!sTwSRkD=~QW7LRUbu=j7Uq{7^VYWfGm~dU`Mcw%)^N&NO;M;5eJ-WD_iB6R9_NRv zJb@&EFa2Zkp=p;TYEJLiYMnf%?&S_61zMl0lb??F*m(u6mR8{*Zcz*KUnFh!67*u+`En0DZHpf<9 zos4hK4!wQOIOjla=%W*NmKel7nY4ph{bi^g@06;4Q+Bzi9ZH-2{q{n$(<-sovsv%? zRUEKbbtdQXkzZMvg3@9+vs*KIcb%GHAuYlC-0}%?|C6w5tJnH3uU&fSWR!G9OTW%# ziK!>giSk|E@$qt>$J*HN*xA13LA5Q?)u}<3&gxHpCiYhTl(XPhnQQY_KjJJozva|7 zwxJXKUshNqThW%-Ux{ zr`(K|mh}|XX7q>5-MaqG^Q|Fwv%j@WKC3d%;g;JyOS@%Vy``5XX62UVY?Nl>x~+Oo zT0x+=v2W4B{&?-=75=3;6SmIz=gd02Xj0jOcef0T4y(&>t(WgGTz2=)o!iH`Z{L*; z+E*4_=JZZ?OJY%_#RL9BT{E)6J#!09wz2+MG0o%OX~ikNb3QE3c;0w>ufN*Ta@M71 zHoZLa$))SWv~wv!(^L}u*!_%;%gr@T^L6{g?fGP~Z`nMfRT~dIz2;QiyhVoRQ{gdw z_MHK@UQar`;!cEb+WxFA*4e3h9{DsCguLV4x~rgCs!6uYZRftNlP*d6S+h z2fm$VJt)_<>gXf({I?dj&c;1D=xh{yPBucIzN7V^gr&rv7po@xnDwq!CaZI?$F+Mt zlMKbVr)tkTE*WRlS7B3fHt1p)UYN-o zY-Q4R<-n$Paqfe*9X5~eFwe6P3A+1wI=joXZ?!z9(=MnqMx`h=zLJn@sO2lNkZH(@ zx^wZxcJcBm@!6T1%t9*zR=3%_V@XNwJHF!OIr){Zg%2lux|dOxx_zdE@27<0!;nPdA!pbimyB+&B zJ<@5uQ^*{tcHgUC&Ze==_r8*=ztHW*rKoe8R^+6!th9O1XZvzi*~;Zv-;O=^*}rvf zgQ2tTmeWe_WZgn0&T|#5T6c99*qM-q=ERwS-o~r}MxGW7fsu!gNNGoS*6rHcxVg(; z%e%2?UX-1~t!N8&6`s;3VI_467t=0Z4(JZwP^~_}Wjl9Kn^{qTh*^PO#nFKNxQkCW zZQt7;wz1l~)5XAVJ4a9?r-IV9f@p;vzKTU_Ro||eKI!JtGrFZht4}hW;9fMZWqkl2 zAOF^rPBe^uc1D&9^6PJ~oLh zqf?Tx;favtv#$89w<~qcZz+A6Iy3Tn?}C+zjdlx~_?PkWR%c8*TXa2O4fpPnWey^_ z=ibbllWY0q;Fo15S429e<*V-5?j5kS=dA*V2$SdCISCB@4_zJqZd_I=Hu*n8#;r>8 zo|W0^^LD4FJz=ptEZoO%*8T75hm#gbgVgDG2j z*XOoOZ`eJw{gDLeKalo)1%&}T$k*Ow>sL)DQ>(qE!OE2_X^##J=^;pm1*f{<9*`zp;Ab9M99dnw{FV=gy@q1di8rXIWXdh^XcG1qh7J{DSgZhoeL!k-1L zul$#v{S_EDdFe41xspjUmrtHPq0g@QM);DKg%jE5sw~TV^7M&v|AKF#FRfxy9G`n? z&Z@MmrMV&({boO$s^b4%naSR(B-qs|zb#~<{hl6xm(SL@p3pwPs~zQQ!6!I6cDKOpNQLlXy!+=7cC`PTz`-_G|FFv++vP0}v* z^sBWu`u)DT{VkQack|0bX1CtX#K}+Z>|1rWb8W+CTTCZvgVz_T&2bcljC`jP2boz+={mMdSa6)d8sD;;d=pXi3_K6-pN+{jChiM z`+Ye3GP{0V?8_HAa_6KouRUBo^TPJif3yT= zuPdJ1HznBA*FIp!{wpUp{|@raej0fqC1Bf=h2an8G}Ss-J_wOoHhaaLJK4-{;xmnp z2%Sq-YV9ohbGoT$$$ZC?$|`(W^Hwj-DtC_Tx$$>VbPsoSvlgqz0rmuzr+)6sIA2v? z^i-Q~du?iZcKu?TlcKv8aOxyQEMb-m-oaD&`sMKemAxBw?Az3&y;jQ8EV@zH)g)U- zsHf+7fyD3SO;vpS*;Z?N&n*?I+~v~d+F`oqqEEN(p2fD&1&&RDO+pBzUAInc^y18KIr<=}V3N?od89v#r zIvn*%Y}VS|kFD#yZc1nVSr%~n)AE8l4hI!3dgkV?UV3k%a^{DeT`3vYJRA8xwQ9NX zOPs6sxe_*Q*B0;9KWCNhUVG$GXzw!~750qOV}`v8K2?8P^(}1H8nIckVluZ~>Rd0C zC^*GnYDCX^Wm(397SC6Rgyz5L_2Tvo$TT>cD05MNYR+dpm)V<+awLg}K3*U1z#3*{ zx0G-0O7~^XwrhlD{F&{kF5SkLa_eA6z~TGrstQB*Z`>+jvipQi;yi8lS;nH1&zW}| zJkup%$Y6f``n9fEmoB}#9P53;B)siH#7}N6Pr1I7Vpcxw*VZ#1 zUc47wvn@&Dv8NI%hkBEFLb`3IOkY)_R%HC)%2j)!oZVUPu5#tcR7p+~ZR5Dxczv6j z^|wXc+|oa6xbE$=TjJ7xTCZUphsxyL5|209y6@A?{`P(PKY0(`H%l|` zK2|e&!Ihif=l7=Ujj!nT%>8c)eoowyC&1L>sPK=a|JwG)Ngr2l$(0tI)Z4Ru<3Z(5 zrrUZVMY}{)m_j#Bd{cN-)YNLOtHffFq^`3cS%S|$JnJPLUA|*pbf$8jN-FEjDGf$P z7_aXXo%m>l-nu`kt8d+mK3P~=n5kpvF0HgNJ!tWC#fo#4w!6Qq$rn9XUi`57hfc!0 z>X`yMnapnTDt(Rp&aJ*ik$Guxdh@DtzZ!9dW>UpW9u9sh&@ArDwEDK6`t0e&=?NS)t4J=-!f?XFdH& zvVNzSx@+h(x6`ZHg7;_yL}!<lx20fRD9>OUng_!hj*Rg{-(0!rO8@J!%0WB9%`EZT7)gq@Fttsl3kZ< z0%eb`Df3_Bo}H_E%B}9wW{;4~xoPD)r5l45makb__QmgG{7X6gyAm5GhxbV=*xe^8 z&Z#m}*L+%(`woVuQL;-Xxv^SmpHxjhyRc%Juid3H(OG9cz4+uzN7hE#c*gLB&B z47K+f635b)MsnNtEsse(EaweL{t{)2Gu%0^>Yo`(~~_n{~)N z*lg_SB~4MP7ghBq{%Fx!B2Ls zgvG&(V@AvkZYldWel&`?b2}>Y)|zjRw5NWaAEIFQ^Kr$V4eH@K-Ou=1S#e%jL{@dt7zKgNDeC$BRFFF<-ghu-wYjvb<{- zlV_^WOc6@hal2W&_r@!4nYNhC36ED?6#8V(=df#P@15zZQ>9{DbyoZDrz&*wv<INMNQeC~AO>id!B%l^sClV1I>G~af~ zq;ll+=1-MPPL(H4 z9N!!F@60vnfGfHSjG`-N#jf2_HK#Gft>Tu`e+I1+n^HP0Y>MYqNM7A}`D^xB|H!7h zQ76;3ZR@iBc`>ywc4<$8(y4Vu@_BxjEEPWRimJDK$#r+XEOzOy?y1{%?)q}wI{H+T zUwi5SBlEQ6lM;Ljo~$%gZ~d~?+5M98w!KYaW^H9{+7>My&K^fYLv}KAUshp|;WNm% zU~<;m>%*0>@6%rDd@Jjla5$q|Xrs!XuBE4%7y>vGENxjuXa1eK`uxHhM@lLe&x@As zW{*_uE?w>MDl91Ug!ZTBPgFOF_w6{d?UHH!8(*RJF&-se@ z?)2`o3^`|cRAS58@He+&e3yFf^0wAks;g7~A~#>{+uio(Wo~z8?+o0{JR$UNvDX#7 z*)?LiCeycF7d!VasW~gf@J-oiKGA^mG{2B!OO?1QUv~d|mt65CDs69f$|Ongzr1^n z)*jcL+`#a&s95UTlc@cdwz6MQ{`972(Z}2~lV^(OE8O;>vPc!DqM%GRZ| zJJNGb{5vd@ER&GHv}@KY?G-4C5LRYYWxXmmDjnm?KV4z6p&94#bjh~_e%Me@SRh3*V-jxS8Nh4{qU@P^!^VQuKuxK?YYo8&Ntt5jbUf| zRLMgO=coO2mzh(m-TUF${I&OgWMuu5zq;hY-1TbPXI+yxa`;rwLx%LzHH-WF?4}<3 zQS5*1{T~z4dj40JGW@Pjjy{{sla%<>t)Rj9>At|@>hfON7fbB9vo76sw|(DRn!Vvp zziwi$|33?lKd(uac(Ja}S2Sfy-`(~3(kUzTzE3X}T5e%?Y?oz8=(KazWqbGJPFk}2 zUhEmKIS+nIy?A;tEZXAjt#2QS#GFR-25hEt=E;cUXTvFw!V+~hP@}QsPoS> zma{zY{6c|MVPsX}-qv?+3oU9+&iHta-LARua@Xn0yl?c2OgRrf^EuvEVEILcZ%$y8 zTuuGGzq}chAOAB{e=%M7{P^GZ7cNFtP4D_+zxa-h)U=K1>Bf3rJ)d%Td_DN~XzHst z?GOL@3pQj$9i2Pp%i=h83jdQe|BsG5`+tTDXP5qGc;NM)p*Q$HLu>Pc{|vt*x9tDJ8UJz7 z`j_)mFPbh^xb)RK`mmV!-+2zdi!WQeyhN&a*Tvs1@gUFrU8MiMu5C-*Me*j zY<;Wk-k5%6D|fX=C_hMIkF>?Ng z4K?!r8Jx|g|7U2_{%1P#KLa=W!T$_jj5F(h2YYsH7HiF~o+VT1FAYT=4Zw}Zt?b+L}6>rxZ(0TK`CE(}X6AFnIf3xTA zk!szt-8_Hw@!T}V^mH4g_H*wXnK_TYj(Pjwitf^#yKCdDvkkbdEgy>5l^1bH_}NEk zoBMt|*39yA@5jE$RzE9LViufEx3&2G=V;o?eHIrI=9}*`yp$fV&(b|bGUxenx&I6^ z*4#BeF2y7K?E5h*pVv=11(!CmAZHd#ik4zZL!`W9`R6L5B7RzoqVr3kq=5@o$6+Xx9Ar z&#?EGRJ@)2`?|L2=lg%(U$|()^N-hCb@NoV@A$|!Tjb(K_uh9RMiZ|H&f`ry|9CxD z)E&?K!bh@Znv0YArSE8@Dz8wskvcs8c$~=A9h2@oK4O&<>~>Vnd`G}3_W-vFlaP9} zn3YFQukL73?V1yrwN0vbQS8lF*TQ`DJJXbRMsH^cJ34K3;)+|Ucc)$n7yDFuGRb=C z?gH#e1HN=H6U%jhOdnARXHkD3bS8sCoy@&bh^GnMv zemyN)9=2n$_f_5Zx{95l(>Sj=$jswpe)oa@`g||#s_m<*@QuuHoR5MaZ8Kl41BXq@ zK5m^kwR!Wc?bD9*wRfGJ?6kq)tn8lLwA6$*-_}0esJ!se=9CL-c#Pfir6XrXSlh0y z%<*_pv1rNGOAEH``W6G4?Z}2bTPxj;=Qy0z`sIkAg<}u5;iIpFcS4^}^>_08v zE3!?fa$4pjbG5*YuPn|#zW#BwoY2+j(I2+Wuj9~Haq#RrXOVwJYUhL6QX9_)Yqdjz zxX&{E+;@pj`YT)4=IGE1=o493#_=ea-4A`Tw@-V?HmzswFPuK!-~TXl35;5yeib`dEtUPbCf6v@w-18sGHD3r8_U}Gd`SkCTcLp*4 z892VkHO-h+6IXm*|Gj!))AJwh3%-biTR-U4GY~kFVL&9sAB4Y=4olt}~(APw2Fqi~6&S z1LycxfccTB{Yt{0O%l@CuY|j7a9ysm>G(v>pR){%n_p?H7C5!}(4-X!YmAtAq&Bjg zw$apT<2tNhaB#JN&*npu7Cpcy_J4Ez_i^=q zhLrgCQegI?RasJOb5}cT$TFJ{dt~hnm5=NUzaCumb#}PCHA9To?GpEnsoPoRDLjym zky^PTL#k{o)4>(D9Lt5aJ(zq%hWT}~YvtARZ+(I2l&+_t zZ50n%7j5~^@Gv;K{=&^i^Z!W9|FG8nN65ANU(70&|7T!R|D)~yPtAKE?mTTE++bTq{{B>b+Tt z_-_jT(RQd{KD5eFPX5wCndWaU{~5TnlWHy1w@uE(al$;}a^hbYX^YtwFFCz5;>L_ceZhrH?!pb8HZnE)OP=4j7`{)o zLfrjB-F(wyldh*^arM0}RN7}aC!xVM>-*vLOfUZNUw*dZQq-GX$H2RJD&fbBdj#0m z#{DRL5G#N1SIu=P@7Reew5-=2>H2Z3u!-U8qTN|fBF%_WyE`s?nVH`n6Jpw%=5}T+yKfgWr-uv@E z1OK@_mmjvh<&OR6-~7}fTKfFfsOEc;l@h;X>Po_3rMuHwVW<~Z!Ta>Y@l zen0QtcK#`W6Mu%4%>oHUpOk-pI6v{oOjk+yscA2laKH6s-81puAG2>Vo*-p`QKpNd zKnH=rT^Yp6J%_=lsL(@1uYlp?$)gGZI}b3ctBV9(x^&S^2hK2V%;m3Yqs2p&Ny9a60oAfGc;6v z9eb3hAg9)W|kROCg17OJhd={m4WdBSJ8@4MhAuxra7xb9Re6+85ZGNCJByqEU9DdH^*-u zi{xdBcD(N?{j*;&V)@~T^@kGnDfX3l)}6_H(xK0P&QbmLAMJVhoCj>G=4p35d_4b+ j+ocGnKbv}Q^7yY. + +import pytest + +import ocrmypdf + + +check_ocrmypdf = pytest.helpers.check_ocrmypdf + + +@pytest.fixture +def acroform(resources): + return resources / 'acroform.pdf' + + +def test_acroform_and_redo(acroform, caplog, no_outpdf): + with pytest.raises(ocrmypdf.exceptions.InputFileError): + check_ocrmypdf(acroform, no_outpdf, '--redo-ocr') + assert '--redo-ocr is not currently possible' in caplog.text diff --git a/tests/test_image_input.py b/tests/test_image_input.py new file mode 100644 index 00000000..b5deedb4 --- /dev/null +++ b/tests/test_image_input.py @@ -0,0 +1,92 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from unittest.mock import patch + +import pytest +from PIL import Image +import img2pdf +import pikepdf + +import ocrmypdf + +check_ocrmypdf = pytest.helpers.check_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api + + +@pytest.fixture +def baiona(resources): + return Image.open(resources / 'baiona_gray.png') + + +def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): + check_ocrmypdf( + resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop + ) + + +def test_no_dpi_info(caplog, baiona, outdir, no_outpdf): + im = baiona + assert 'dpi' not in im.info + input_image = outdir / 'baiona_no_dpi.png' + im.save(input_image) + + rc = run_ocrmypdf_api(input_image, no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "--image-dpi" in caplog.text + + +def test_dpi_not_credible(caplog, baiona, outdir, no_outpdf): + im = baiona + assert 'dpi' not in im.info + input_image = outdir / 'baiona_no_dpi.png' + im.save(input_image, dpi=(30, 30)) + + rc = run_ocrmypdf_api(input_image, no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "not credible" in caplog.text + + +def test_cmyk_no_icc(caplog, resources, no_outpdf): + rc = run_ocrmypdf_api(resources / 'baiona_cmyk.jpg', no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "no ICC profile" in caplog.text + + +def test_img2pdf_fails(resources, no_outpdf): + with patch( + 'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError() + ): + rc = run_ocrmypdf_api( + resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200' + ) + assert rc == ocrmypdf.ExitCode.input_file + + +def test_jpeg_in_jpeg_out(resources, outpdf, spoof_tesseract_noop): + check_ocrmypdf( + resources / 'congress.jpg', + outpdf, + '--image-dpi', + '100', + '--output-type', + 'pdf', # specifically check pdf because Ghostscript may convert to JPEG + '--remove-background', + env=spoof_tesseract_noop, + ) + with pikepdf.open(outpdf) as pdf: + assert next(pdf.pages[0].images.values()).Filter == pikepdf.Name.DCTDecode diff --git a/tests/test_main.py b/tests/test_main.py index 4ef299fb..0380508b 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -350,12 +350,6 @@ def test_algo4(resources, spoof_tesseract_noop, outpdf): assert p.returncode == ExitCode.encrypted_pdf -def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf( - resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop - ) - - def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): out = check_ocrmypdf( resources / 'jbig2.pdf', From f34130d193f00f53c2b0d1a1dc7d214f4cb887b0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 10 Dec 2019 01:07:59 -0800 Subject: [PATCH 276/880] Fixed case where page image was not converted to JPEG If a preprocessing option was used, and all original images on the page were JPEGs, and --output-type=pdf, then images would saved as Flate instead of converted to JPEG. --- src/ocrmypdf/_pipeline.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6435804e..e2ea0ee5 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -41,7 +41,7 @@ from .helpers import safe_symlink from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps -from .pdfinfo import Colorspace, PdfInfo +from .pdfinfo import Colorspace, PdfInfo, Encoding VECTOR_PAGE_DPI = 400 @@ -557,7 +557,7 @@ def ocr_tesseract_hocr(input_file, page_context): def should_visible_page_image_use_jpg(pageinfo): # If all images were JPEGs originally, produce a JPEG as output - return pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images) + return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images) def create_visible_page_jpg(image, page_context): From a2d89f67c4846136b06b185f424cda1480e87156 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 10 Dec 2019 01:44:00 -0800 Subject: [PATCH 277/880] Improve help messages for Windows --- src/ocrmypdf/exec/ghostscript.py | 13 ++++++++++++- src/ocrmypdf/leptonica.py | 11 ++++++++++- 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 0f2499e2..7c36eb30 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -42,7 +42,18 @@ if os.name == 'nt': if not GS: GS = which('gswin32c') if not GS: - raise MissingDependencyError("Ghostscript (gswin64c or gswin32c)") + raise MissingDependencyError( + """ + --------------------------------------------------------------------- + This error normally occurs when ocrmypdf can't Ghostscript. Please + ensure Ghostscript is installed and its location is added to the + system PATH environment variable. + + For details see: + https://ocrmypdf.readthedocs.io/en/latest/installation.html + --------------------------------------------------------------------- + """ + ) GS = Path(GS).stem diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 6fbf2d99..a4dec30f 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -47,7 +47,16 @@ else: _libpath = find_library(libname) if not _libpath and os.name == 'nt': raise MissingDependencyError( - "Please ensure that 'tesseract' is on your PATH environment variable. " + """ + --------------------------------------------------------------------- + This error normally occurs when ocrmypdf can't find a file named + liblept-5.dll (Leptonica). Please ensure Tesseract-OCR is installed + and its location is added to the system PATH environment variable. + + For details see: + https://ocrmypdf.readthedocs.io/en/latest/installation.html + --------------------------------------------------------------------- + """ ) lept = ffi.dlopen(_libpath) lept.setMsgSeverity(lept.L_SEVERITY_WARNING) From 91456e19a4b58a517b4fbcc7b86de0a956833eca Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Dec 2019 01:05:47 -0800 Subject: [PATCH 278/880] pdfa.py: Fix misleading comment --- src/ocrmypdf/pdfa.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 617da444..3aeb7612 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -82,8 +82,6 @@ def generate_pdfa_ps(target_filename, icc='sRGB'): :param target_filename: filename to save :param icc: ICC identifier such as 'sRGB' - - :returns: a string containing the entire pdfmark """ if icc == 'sRGB': icc_profile = SRGB_ICC_PROFILE @@ -101,6 +99,7 @@ def generate_pdfa_ps(target_filename, icc='sRGB'): # We should have encoded everything to pure ASCII by this point, and # to be safe, only allow ASCII in PostScript Path(target_filename).write_text(ps, encoding='ascii') + return target_filename def file_claims_pdfa(filename): From 9559b0b18660202111600505f42884b2ad38f38b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Dec 2019 01:21:15 -0800 Subject: [PATCH 279/880] Use pikepdf to perform qpdf.check() --- requirements/main.txt | 2 +- setup.py | 2 +- src/ocrmypdf/exec/qpdf.py | 56 +++++++++++++++++++++++---------------- 3 files changed, 35 insertions(+), 25 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 2b6f1fca..d6c25b7b 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -4,7 +4,7 @@ cffi == 1.13.2 img2pdf == 0.3.3 pdfminer.six == 20191110 -pikepdf == 1.7.0 +pikepdf == 1.8.1 Pillow >= 6.2.0 reportlab == 3.5.32 tqdm == 4.37.0 diff --git a/setup.py b/setup.py index 214bc62f..a79337c6 100644 --- a/setup.py +++ b/setup.py @@ -98,7 +98,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six >= 20181108, <= 20191110', - 'pikepdf >= 1.7.0, < 2', + 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling 'tqdm >= 4', diff --git a/src/ocrmypdf/exec/qpdf.py b/src/ocrmypdf/exec/qpdf.py index 9be8692b..d46baf1c 100644 --- a/src/ocrmypdf/exec/qpdf.py +++ b/src/ocrmypdf/exec/qpdf.py @@ -17,35 +17,45 @@ """Interface to qpdf executable""" -from functools import lru_cache -from os import fspath -from subprocess import PIPE, STDOUT, CalledProcessError +from io import StringIO -from . import get_version, run +import pikepdf -@lru_cache(maxsize=1) def version(): - return get_version('qpdf', regex=r'qpdf version (.+)') + return pikepdf.__libqpdf_version__ def check(input_file, log=None): - args_qpdf = ['qpdf', '--check', fspath(input_file)] - - if log is None: - import logging as log - + pdf = None try: - run(args_qpdf, stderr=STDOUT, stdout=PIPE, universal_newlines=True, check=True) - except CalledProcessError as e: - if e.returncode == 2: - log.error("%s: not a valid PDF, and could not repair it.", input_file) - log.error("Details:") - log.error(e.output) - elif e.returncode == 3: - log.info("qpdf --check returned warnings:") - log.info(e.output) - else: - log.warning(e.output) + pdf = pikepdf.open(input_file) + except pikepdf.PdfError as e: + if log: + log.error(e) return False - return True + else: + messages = pdf.check() + for msg in messages: + if 'error' in msg.lower(): + log.error(msg) + else: + log.warning(msg) + + sio = StringIO() + linearize = None + try: + pdf.check_linearization(sio) + except RuntimeError: + pass + else: + linearize = sio.getvalue() + if linearize: + log.warning(linearize) + + if not messages and not linearize: + return True + return False + finally: + if pdf: + pdf.close() From 437c2357385ce70ac6cf62aae5095142796d1a3c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Dec 2019 13:13:51 -0800 Subject: [PATCH 280/880] v9.2.0 release notes and docs --- .travis.yml | 33 ++++++++++++++++----------------- README.md | 8 +++++--- docs/installation.rst | 16 ++++++++++++---- docs/release_notes.rst | 20 ++++++++++++++++++++ 4 files changed, 53 insertions(+), 24 deletions(-) diff --git a/.travis.yml b/.travis.yml index 4aa53def..1f05bc7b 100644 --- a/.travis.yml +++ b/.travis.yml @@ -146,20 +146,19 @@ script: - tesseract --version - qpdf --version - pytest -n auto - -deploy: - # release for main pypi - # 3.7 is considered the build leader and does the deploy, otherwise there is - # a race and all versions will try to deploy - # OTOH if we ever need separate binary wheels then each version needs its - # own deploy - - provider: pypi - user: ocrmypdf-travis - password: - secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo=" - distributions: "sdist bdist_wheel" - on: - branch: master - tags: true - condition: $TRAVIS_PYTHON_VERSION == "3.7" && $TRAVIS_OS_NAME == "linux" - skip_upload_docs: true +# deploy: +# # release for main pypi +# # 3.7 is considered the build leader and does the deploy, otherwise there is +# # a race and all versions will try to deploy +# # OTOH if we ever need separate binary wheels then each version needs its +# # own deploy +# - provider: pypi +# user: ocrmypdf-travis +# password: +# secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo=" +# distributions: "sdist bdist_wheel" +# on: +# branch: master +# tags: true +# condition: $TRAVIS_PYTHON_VERSION == "3.7" && $TRAVIS_OS_NAME == "linux" +# skip_upload_docs: true diff --git a/README.md b/README.md index e5f1d319..c60bc683 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,8 @@ OCRmyPDF -[![Travis build status][travis]](https://travis-ci.org/jbarlow83/OCRmyPDF) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] +[![Build Status][azure]](https://dev.azure.com/jim0585/ocrmypdf/_build/latest?definitionId=2&branchName=master) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] + +[azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master [travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status" @@ -48,7 +50,7 @@ For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/ Motivation ---------- -I searched the web for a free command line tool to OCR PDF files on Linux/UNIX: I found many, but none of them were really satisfying. +I searched the web for a free command line tool to OCR PDF files: I found many, but none of them were really satisfying. - Either they produced PDF files with misplaced text under the image (making copy/paste impossible) - Or they did not handle accents and multilingual characters @@ -63,7 +65,7 @@ I searched the web for a free command line tool to OCR PDF files on Linux/UNIX: Installation ------------ -Linux, UNIX, and macOS are supported. Windows is not directly supported but there is a Docker image available that runs on Windows. +Linux, Windows, macOS and FreeBSD are supported. Docker images are also available. Users of Debian 9 or later or Ubuntu 16.10 or later may simply diff --git a/docs/installation.rst b/docs/installation.rst index 3398ef94..8c2f95eb 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -442,19 +442,21 @@ Installing on Windows production-ready solution, use Windows Subsystem for Linux or a Docker image. +.. note:: + + Administrator privileges will be required for some of these steps. + You must install the following for Windows: * Python 3.7 (64-bit recommended) * Tesseract 4.0 or later * Ghostscript 9.50 or later -* QPDF 9.0.2 or later You can install these with the Chocolatey package manager: * ``choco install python3`` * ``choco install --pre tesseract`` * ``choco install ghostscript`` -* ``choco install qpdf`` Also consider adding: @@ -463,8 +465,14 @@ Also consider adding: Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier versions of Windows and 32-bit versions of these programs are not tested. -Modify your ``PATH`` environment variable so that Tesseract, Ghostscript and QPDF -executables on the ``PATH``. +Modify your ``PATH`` environment variable so that Tesseract and Ghostscript, and +any optional executables can be found. You can enter it in the command line +or `follow these directions `_ +to make the change persistent and system-wide. + +You may then use pip to install ocrmypdf: + +* ``pip install ocrmypdf`` Installing on Windows Subsystem for Linux ========================================= diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 9f1a23b5..949a6cc3 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,26 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.2.0 +====== + +- Native Windows is now supported. +- Continuous integration moved to Azure Pipelines. +- Improved test coverage and speed of tests. +- Fixed an issue where a page that was originally a JPEG would be saved as a + PNG, increasing file size. This occurred only when a preprocessing option + was selected along with ``--output-type=pdf`` and all images on the original + page were JPEGs. Regression since v7.0.0. +- OCRmyPDF no longer depends on the QPDF executable ``qpdf`` or ``libqpdf``. + It uses pikepdf (which in turn depends on ``libqpdf``). Package maintainers + should adjust dependencies so that OCRmyPDF no longer calls for libqpdf on + its own. For users of Python binary wheels, this change means a separate + installation of QPDF is no longer necessary. This change is mainly to + simplify installation on Windows. +- Fixed a rare case where log messages from Tesseract would be discarded. +- Fixed incorrect function signature for pixFindPageForeground, causing + exceptions on certain platforms/Leptonica versions. + v9.1.1 ====== From facc4750bc68ca583203a87d91dd13cecbfb5ec7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 12 Dec 2019 00:07:01 -0800 Subject: [PATCH 281/880] Remove command line qpdf from azure and travis --- .travis.yml | 5 ----- azure-pipelines.yml | 8 -------- 2 files changed, 13 deletions(-) diff --git a/.travis.yml b/.travis.yml index 1f05bc7b..9fdb4dde 100644 --- a/.travis.yml +++ b/.travis.yml @@ -26,7 +26,6 @@ matrix: packages: - ghostscript - libffi-dev - - qpdf - tesseract-ocr - tesseract-ocr-deu - tesseract-ocr-eng @@ -54,7 +53,6 @@ matrix: - libavformat56 - libavutil54 - libffi-dev - - qpdf - tesseract-ocr - tesseract-ocr-deu - tesseract-ocr-eng @@ -84,7 +82,6 @@ matrix: - libffi-dev - pngquant - poppler-utils - - qpdf - tesseract-ocr - tesseract-ocr-deu - tesseract-ocr-eng @@ -108,7 +105,6 @@ matrix: - libffi-dev - pngquant - poppler-utils - - qpdf - tesseract-ocr - tesseract-ocr-deu - tesseract-ocr-eng @@ -144,7 +140,6 @@ install: script: - tesseract --version - - qpdf --version - pytest -n auto # deploy: # # release for main pypi diff --git a/azure-pipelines.yml b/azure-pipelines.yml index d1b486ad..c982d2df 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -30,7 +30,6 @@ stages: choco install --yes --no-progress --pre tesseract choco install --yes --no-progress python3 choco install --yes --no-progress ghostscript - choco install --yes --no-progress qpdf choco install --yes --no-progress pngquant displayName: "Install system packages" - pwsh: | @@ -39,7 +38,6 @@ stages: pip install --upgrade pip wheel pip install -r requirements/main.txt -r requirements/test.txt . tesseract --version - qpdf --version displayName: "Install Python packages" - pwsh: | refreshenv @@ -85,7 +83,6 @@ stages: libsm6 libxext6 libxrender-dev \ pngquant \ poppler-utils \ - qpdf \ tesseract-ocr \ tesseract-ocr-deu \ tesseract-ocr-eng \ @@ -98,7 +95,6 @@ stages: displayName: "Install Python packages" - bash: | tesseract --version - qpdf --version displayName: "Record versions" - bash: | # -n auto is slower on Linux and breaks on Python 3.8 @@ -139,7 +135,6 @@ stages: libsm6 libxext6 libxrender-dev \ pngquant \ poppler-utils \ - qpdf \ tesseract-ocr \ tesseract-ocr-deu \ tesseract-ocr-eng \ @@ -152,7 +147,6 @@ stages: displayName: "Install Python packages" - bash: | tesseract --version - qpdf --version displayName: "Record versions" - bash: | # -n auto is slower on Linux and breaks on Python 3.8 @@ -189,7 +183,6 @@ stages: leptonica \ openjpeg \ pngquant \ - qpdf \ tesseract \ unpaper displayName: "Install system packages" @@ -199,7 +192,6 @@ stages: displayName: "Install Python packages" - bash: | tesseract --version - qpdf --version displayName: "Record versions" - bash: pytest -nauto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml displayName: "Test" From 9fe354359b8885b25bf811c9f7979cdd5f3990d8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 19 Dec 2019 00:27:37 -0800 Subject: [PATCH 282/880] Generally update documentation about available platforms --- README.md | 12 +++-------- docs/contributing.rst | 21 ++++++++++++++++++- docs/installation.rst | 42 ++++++++++++++++++++++++++++--------- docs/introduction.rst | 2 +- src/ocrmypdf/_validation.py | 10 +++++++++ 5 files changed, 66 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index c60bc683..d3632418 100644 --- a/README.md +++ b/README.md @@ -79,7 +79,7 @@ and users of Fedora 29 or later may simply dnf install ocrmypdf ``` -and macOS users with Homebrew may simply +and Homebrew users (macOS, Linux, Windows Subsystem for Linux) may simply ```bash brew install ocrmypdf @@ -113,18 +113,12 @@ ocrmypdf --help Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/en/latest/index.html). -If you detect an issue, please: - -- Check whether your issue is already known -- If no problem report exists on github, please create one here: -- Describe your problem thoroughly -- Append the console output of the script when running the debug mode (`-v 1` option) -- If possible provide your input PDF file as well as the content of the temporary folder (using a file sharing service like Dropbox) +Please report issues on our [GitHub issues](https://github.com/jbarlow83/OCRmyPDF/issues) page, and follow the issue template for quick response. Requirements ------------ -In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. +In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD. Press & Media ------------- diff --git a/docs/contributing.rst b/docs/contributing.rst index 2a7c81ff..d240bc8b 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -15,7 +15,10 @@ Code style ========== We use PEP8, ``black`` for code formatting and ``isort`` for import sorting. The -settings for programs are in ``pyproject.toml`` and ``setup.cfg``. +settings for these programs are in ``pyproject.toml`` and ``setup.cfg``. Pull +requests should follow the style guide. One difference we use from "black" style +is that strings shown to the user are always in double quotes (``"``) and strings +for internal uses are in single quotes (``'``). Tests ===== @@ -36,3 +39,19 @@ New non-Python dependencies OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for its functionality. In general we prefer to avoid adding new external programs. + +Style guide: Is it OCRmyPDF or ocrmypdf? +======================================== + +The program/project is OCRmyPDF and the name of the executable or library is ocrmypdf. + +Known ports/packagers +===================== + +OCRmyPDF has been ported to many platforms already. If you are interesting in +porting to a new platform, check with +`Repology `__ to see the status +of that platform. + +Packager maintainers, please ensure that the command line completion scripts in +``misc/`` are installed. diff --git a/docs/installation.rst b/docs/installation.rst index 8c2f95eb..c1186aa8 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -8,11 +8,20 @@ Installing OCRmyPDF |latest| The easiest way to install OCRmyPDF is to follow the steps for your operating -system/platform, although sometimes this version may be out of date. +system/platform, although sometimes this version may be out of date. This +installation guide provides information allowing you to compare the current +version to the one provided by your platform. -If you want to use the latest version of OCRmyPDF, your best bet is to install -the most recent version your platform provides, and then upgrade that version by -installing the Python binary wheels. +If you want to use the latest version of OCRmyPDF and all of its optional +dependencies, the easiest way to get that is install the Homebrew package. Homebrew +is best known as a macOS package manger, but also works for +`Linux and Windows Subsystem for Linux `__. +After Homebrew is installed, simply run ``brew install ocrmypdf``. + +You can also use the more detailed procedures here to manually install OCRmyPDF +from source or with the ``pip`` package manager for binary wheels. The reason +for these varied steps is that OCRmyPDF requires third-party executables that are +not part of Python. .. contents:: Platform-specific steps :depth: 2 @@ -55,8 +64,8 @@ Debian and Ubuntu 18.04 or newer | |ubu-1804| |ubu-1810| |ubu-1904| |ubu-1910| | +-----------------------------------------------+ -Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later may -simply +Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users +of Windows Subsystem for Linux, may simply .. code-block:: bash @@ -303,6 +312,19 @@ the following command. If you have any difficulties with installation, check the repository package page. +Alpine Linux +------------ + +.. image:: https://repology.org/badge/version-for-repo/alpine_edge/ocrmypdf.svg + :alt: Alpine Linux + :target: https://repology.org/metapackage/ocrmypdf + +To install OCRmyPDF for Alpine Linux: + +.. code-block:: bash + + apk add ocrmypdf + Other Linux packages -------------------- @@ -427,8 +449,7 @@ Installing the Docker image =========================== For some users, installing the Docker image will be easier than -installing all of OCRmyPDF's dependencies. For Windows, it is the only -option. +installing all of OCRmyPDF's dependencies. See `OCRmyPDF Docker Image `__ for more information. @@ -448,7 +469,7 @@ Installing on Windows You must install the following for Windows: -* Python 3.7 (64-bit recommended) +* Python 3.7 (64-bit) * Tesseract 4.0 or later * Ghostscript 9.50 or later @@ -463,7 +484,8 @@ Also consider adding: * ``choco install pngquant`` Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier -versions of Windows and 32-bit versions of these programs are not tested. +versions of Windows and 32-bit versions of these programs are not tested, and not +supported at this time. Modify your ``PATH`` environment variable so that Tesseract and Ghostscript, and any optional executables can be found. You can enter it in the command line diff --git a/docs/introduction.rst b/docs/introduction.rst index efe1de91..b878439b 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -2,7 +2,7 @@ Introduction ============ -OCRmyPDF is a Python 3 package that adds OCR layers to PDFs. +OCRmyPDF is a Python 3 application and library that adds OCR layers to PDFs. About OCR ========= diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index af518b2b..b53f0fc4 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -58,6 +58,15 @@ log = logging.getLogger(__name__) verify_python3_env() +def check_platform(): + if os.name == 'nt' and sys.maxsize <= 2 ** 32: # pragma: no cover + # 32-bit interpreter on Windows + log.error( + "You are running OCRmyPDF in a 32-bit (x86) Python interpreter." + "Please use a 64-bit (x86-64) version of Python." + ) + + def check_options_languages(options): if not options.language: options.language = [DEFAULT_LANGUAGE] @@ -292,6 +301,7 @@ def check_options_pillow(options): def check_options(options): + check_platform() check_options_languages(options) check_options_metadata(options) check_options_output(options) From 39da931a565c7f6e6c46b2d81ba320095c3191b9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 19 Dec 2019 12:11:32 -0800 Subject: [PATCH 283/880] Look in Program Files for executables and liblept5.dll --- docs/installation.rst | 8 +++-- src/ocrmypdf/exec/__init__.py | 59 ++++++++++++++++++++++++++++++++++- src/ocrmypdf/leptonica.py | 2 ++ 3 files changed, 65 insertions(+), 4 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index c1186aa8..526cf0eb 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -487,9 +487,11 @@ Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier versions of Windows and 32-bit versions of these programs are not tested, and not supported at this time. -Modify your ``PATH`` environment variable so that Tesseract and Ghostscript, and -any optional executables can be found. You can enter it in the command line -or `follow these directions `_ +OCRmyPDF will check for Tesseract-OCR and Ghostscript in your Program Files folder. +If they are in some other location, you may need to modify the ``PATH`` +environment variable so Tesseract, Ghostscript, and other any optional executables can +be found. You can enter it in the command line or +`follow these directions `_ to make the change persistent and system-wide. You may then use pip to install ocrmypdf: diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index d5a7de44..b571c333 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -23,6 +23,7 @@ import re import sys import shutil from collections.abc import Mapping +from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError, run as subprocess_run from ..exceptions import ExitCode, MissingDependencyError @@ -39,13 +40,40 @@ def _get_program(args, env=None): def run(args, *, env=None, **kwargs): + """Wrapper around subprocess.run() + + The main purpose of this wrapper is to allow us to substitute the main program + for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces + the main PATH as a location to check for programs to run. + + Secondly we have to account for behavioral differences in Windows in particular. + Creating symbolic links in Windows requires administrator privileges and + may not work if for some reason we're using a FAT file system or the temporary + folder is on a different drive from the working folder. The test suite + works around this by creating shim Python scripts that perform the same function + as a symbolic link, but those shims require support on this side, to ensure + we call them with Python. + + """ if not env: env = os.environ + + # Search in spoof path if necessary program = _get_program(args, env) + + # If we are running a .py on Windows, ensure we call it with this Python + # (to support test suite shims) if os.name == 'nt' and program.lower().endswith('.py'): args = [sys.executable, program] + args[1:] else: args = [program] + args[1:] + + if os.name == 'nt' and not shutil.which(args[0], path=os.get_exec_path(env)): + shimmed_path = shim_paths_with_program_files(env) + new_args0 = shutil.which(args[0], path=shimmed_path) + if new_args0: + args[0] = new_args0 + log.debug(args) if sys.version_info < (3, 7) and os.name == 'nt': # Can't use close_fds=True on Windows with Python 3.6 or older @@ -55,7 +83,7 @@ def run(args, *, env=None, **kwargs): def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): - "Get the version of the specified program" + """Get the version of the specified program""" args_prog = [program, version_arg] try: proc = run( @@ -91,6 +119,35 @@ def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env return version +@lru_cache(maxsize=1) +def shim_paths_with_program_files(env=None): + if not env: + env = os.environ + program_files = env.get('PROGRAMFILES', '') + if not program_files: + return env.get('PATH', '') + paths = [] + try: + for dirname in os.listdir(program_files): + if dirname.lower() == 'tesseract-ocr': + paths.append(os.path.join(program_files, dirname)) + if dirname.lower() == 'gs': + try: + latest_gs = max( + os.listdir(os.path.join(program_files, dirname)), + key=lambda d: float(d[2:]), + ) + except (FileNotFoundError, NotADirectoryError): + continue + paths.append(os.path.join(program_files, dirname, latest_gs, 'bin')) + except EnvironmentError: + pass + paths.extend( + path for path in os.environ['PATH'].split(os.pathsep) if path not in set(paths) + ) + return os.pathsep.join(paths) + + missing_program = ''' The program '{program}' could not be executed or was not found on your system PATH. diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index a4dec30f..344c2d9e 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -35,6 +35,7 @@ from tempfile import TemporaryFile from .lib._leptonica import ffi from .exceptions import MissingDependencyError +from .exec import shim_paths_with_program_files # pylint: disable=protected-access @@ -42,6 +43,7 @@ logger = logging.getLogger(__name__) if os.name == 'nt': libname = 'liblept-5' + os.environ['PATH'] = shim_paths_with_program_files() else: libname = 'lept' _libpath = find_library(libname) From 8c5f8b8ddd81ef34f972866336dc52eaf959d78e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 19 Dec 2019 15:16:48 -0800 Subject: [PATCH 284/880] Add isort to precommit --- .pre-commit-config.yaml | 22 ++++++++++++++-------- pyproject.toml | 3 ++- setup.cfg | 2 ++ 3 files changed, 18 insertions(+), 9 deletions(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index c9826a71..544da6f1 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,10 +1,4 @@ repos: - - repo: https://github.com/psf/black - rev: stable - hooks: - - id: black - language_version: python3.7 - exclude: ^src/ocrmypdf/lib/_leptonica.py - repo: https://github.com/pre-commit/pre-commit-hooks rev: v2.4.0 hooks: @@ -13,5 +7,17 @@ repos: - id: check-toml - id: check-yaml - id: debug-statements - - id: name-tests-test - args: ["--django"] + - repo: https://github.com/asottile/seed-isort-config + rev: v1.9.3 + hooks: + - id: seed-isort-config + - repo: https://github.com/pre-commit/mirrors-isort + rev: v4.3.21 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases + hooks: + - id: isort + - repo: https://github.com/psf/black + rev: stable + hooks: + - id: black + language_version: python3.7 + exclude: ^src/ocrmypdf/lib/_leptonica.py diff --git a/pyproject.toml b/pyproject.toml index a28f55c0..9cfd37c6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,8 @@ build-backend = "setuptools.build_meta" [tool.black] line-length = 88 -target-version = ["py36", "py37", "py38"] +target-version = ["py36", +"py37", "py38"] skip-string-normalization = true include = '\.pyi?$' exclude = ''' diff --git a/setup.cfg b/setup.cfg index b1b0a740..17fb9043 100644 --- a/setup.cfg +++ b/setup.cfg @@ -22,6 +22,8 @@ include_trailing_comma=True force_grid_wrap=0 use_parentheses=True line_length=88 +known_first_party = ocrmypdf +known_third_party = PIL,PyPDF2,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,werkzeug [metadata] license_file = LICENSE From c5edff2c2ff2a4e8082be567e0881dc1c4d58296 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 19 Dec 2019 15:29:56 -0800 Subject: [PATCH 285/880] Sort imports --- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_sync.py | 2 +- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/exec/__init__.py | 5 +++-- src/ocrmypdf/exec/ghostscript.py | 6 +++--- src/ocrmypdf/exec/tesseract.py | 2 +- src/ocrmypdf/exec/unpaper.py | 3 ++- src/ocrmypdf/leptonica.py | 2 +- tests/conftest.py | 3 ++- tests/spoof/gs_feature_elision.py | 4 +--- tests/spoof/gs_pdfa_failure.py | 4 ++-- tests/spoof/gs_raster_failure.py | 5 ++--- tests/spoof/gs_render_failure.py | 1 - tests/spoof/tesseract_badutf8.py | 1 - tests/spoof/tesseract_cache.py | 1 - tests/test_acroform.py | 1 - tests/test_completion.py | 2 +- tests/test_ghostscript.py | 1 - tests/test_graft.py | 2 +- tests/test_image_input.py | 4 ++-- tests/test_lept.py | 2 +- tests/test_metadata.py | 10 +++++----- tests/test_optimize.py | 2 +- tests/test_page_numbers.py | 2 +- tests/test_pdfinfo.py | 2 +- tests/test_rotation.py | 2 +- tests/test_stdio.py | 2 +- tests/test_unpaper.py | 2 +- tests/test_validation.py | 2 +- 29 files changed, 37 insertions(+), 42 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e2ea0ee5..2705b2ab 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -41,7 +41,7 @@ from .helpers import safe_symlink from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps -from .pdfinfo import Colorspace, PdfInfo, Encoding +from .pdfinfo import Colorspace, Encoding, PdfInfo VECTOR_PAGE_DPI = 400 diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 27262135..5b27a8f0 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -25,8 +25,8 @@ import threading from collections import namedtuple from tempfile import mkdtemp -from tqdm import tqdm import PIL +from tqdm import tqdm from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files, make_logger diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 29b25a38..9a2cdb2f 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -21,7 +21,7 @@ import sys import warnings from enum import IntEnum from pathlib import Path -from typing import List, Optional, Dict +from typing import Dict, List, Optional from tqdm import tqdm diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index b571c333..0666621a 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -20,11 +20,12 @@ import logging import os import re -import sys import shutil +import sys from collections.abc import Mapping from functools import lru_cache -from subprocess import PIPE, STDOUT, CalledProcessError, run as subprocess_run +from subprocess import PIPE, STDOUT, CalledProcessError +from subprocess import run as subprocess_run from ..exceptions import ExitCode, MissingDependencyError diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 7c36eb30..5f6bc61f 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -18,20 +18,20 @@ """Interface to Ghostscript executable""" import logging -import re import os +import re import warnings from contextlib import suppress from functools import lru_cache from io import BytesIO from os import fspath from pathlib import Path -from subprocess import PIPE, CalledProcessError from shutil import which +from subprocess import PIPE, CalledProcessError from PIL import Image -from ..exceptions import SubprocessOutputError, MissingDependencyError +from ..exceptions import MissingDependencyError, SubprocessOutputError from . import get_version, run gslog = logging.getLogger() diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 26332976..16edb7cb 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -17,11 +17,11 @@ """Interface to Tesseract executable""" +import logging import os import shutil from collections import namedtuple from contextlib import suppress -import logging from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 1143c0e9..2984b455 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -29,7 +29,8 @@ from tempfile import TemporaryDirectory from PIL import Image from ..exceptions import MissingDependencyError, SubprocessOutputError -from . import get_version, run as external_run +from . import get_version +from . import run as external_run @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 344c2d9e..7ff3a265 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -33,9 +33,9 @@ from io import BytesIO from os import fspath from tempfile import TemporaryFile -from .lib._leptonica import ffi from .exceptions import MissingDependencyError from .exec import shim_paths_with_program_files +from .lib._leptonica import ffi # pylint: disable=protected-access diff --git a/tests/conftest.py b/tests/conftest.py index 1b1ece82..8521827f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -21,10 +21,11 @@ import platform import sys from pathlib import Path from subprocess import PIPE, run -from ocrmypdf import api, cli import pytest +from ocrmypdf import api, cli + pytest_plugins = ['helpers_namespace'] try: diff --git a/tests/spoof/gs_feature_elision.py b/tests/spoof/gs_feature_elision.py index f9856311..a06deaf3 100755 --- a/tests/spoof/gs_feature_elision.py +++ b/tests/spoof/gs_feature_elision.py @@ -25,14 +25,12 @@ import os import sys from subprocess import check_call +from gs import real_ghostscript """Replicate one type of Ghostscript feature elision warning during PDF/A creation.""" -from gs import real_ghostscript - - elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 not permitted in PDF/A-2, overprint mode not set""" diff --git a/tests/spoof/gs_pdfa_failure.py b/tests/spoof/gs_pdfa_failure.py index 6dd90e29..1d9fdf7d 100755 --- a/tests/spoof/gs_pdfa_failure.py +++ b/tests/spoof/gs_pdfa_failure.py @@ -23,12 +23,12 @@ import os import sys +from gs import real_ghostscript + """Replicate Ghostscript PDF/A conversion failure by suppressing some arguments""" -from gs import real_ghostscript - def main(): if '--version' in sys.argv: diff --git a/tests/spoof/gs_raster_failure.py b/tests/spoof/gs_raster_failure.py index c07b881b..067f8a6f 100755 --- a/tests/spoof/gs_raster_failure.py +++ b/tests/spoof/gs_raster_failure.py @@ -24,11 +24,10 @@ import os import sys -"""Replicate Ghostscript raster failure while allowing rendering""" - - from gs import real_ghostscript +"""Replicate Ghostscript raster failure while allowing rendering""" + def main(): if '--version' in sys.argv: diff --git a/tests/spoof/gs_render_failure.py b/tests/spoof/gs_render_failure.py index a43833a8..65509f5d 100755 --- a/tests/spoof/gs_render_failure.py +++ b/tests/spoof/gs_render_failure.py @@ -25,7 +25,6 @@ import os import sys - from gs import real_ghostscript diff --git a/tests/spoof/tesseract_badutf8.py b/tests/spoof/tesseract_badutf8.py index 472a47a9..3cbb6625 100755 --- a/tests/spoof/tesseract_badutf8.py +++ b/tests/spoof/tesseract_badutf8.py @@ -22,7 +22,6 @@ import sys - """Tesseract bad utf8 spoof In 'hocr' mode or 'pdf' mode, return error code 1 and some non-Unicode diff --git a/tests/spoof/tesseract_cache.py b/tests/spoof/tesseract_cache.py index 83c09535..adf3e257 100755 --- a/tests/spoof/tesseract_cache.py +++ b/tests/spoof/tesseract_cache.py @@ -59,7 +59,6 @@ import subprocess import sys from pathlib import Path - __version__ = subprocess.check_output( ['tesseract', '--version'], stderr=subprocess.STDOUT ).decode() diff --git a/tests/test_acroform.py b/tests/test_acroform.py index 48b9ff91..cb188e80 100644 --- a/tests/test_acroform.py +++ b/tests/test_acroform.py @@ -19,7 +19,6 @@ import pytest import ocrmypdf - check_ocrmypdf = pytest.helpers.check_ocrmypdf diff --git a/tests/test_completion.py b/tests/test_completion.py index 8837fbe2..ccb024aa 100644 --- a/tests/test_completion.py +++ b/tests/test_completion.py @@ -15,7 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from subprocess import run, PIPE +from subprocess import PIPE, run import pytest diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index c659fd68..c74b4952 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -18,7 +18,6 @@ import logging from decimal import Decimal - import pikepdf import pytest from PIL import Image diff --git a/tests/test_graft.py b/tests/test_graft.py index 52aa336c..a14e3cee 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -18,10 +18,10 @@ import os from unittest.mock import patch +import pikepdf import pytest import ocrmypdf -import pikepdf def test_no_glyphless_graft(resources, outdir): diff --git a/tests/test_image_input.py b/tests/test_image_input.py index b5deedb4..ceb94cbe 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -17,10 +17,10 @@ from unittest.mock import patch -import pytest -from PIL import Image import img2pdf import pikepdf +import pytest +from PIL import Image import ocrmypdf diff --git a/tests/test_lept.py b/tests/test_lept.py index 116060c3..8f7a12d0 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -16,8 +16,8 @@ # along with OCRmyPDF. If not, see . -from os import fspath import os +from os import fspath from pickle import dumps, loads from unittest.mock import patch diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 6d33db12..530dae10 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -17,22 +17,22 @@ import datetime -from datetime import timezone import logging import mmap -from os import fspath import os +from datetime import timezone +from os import fspath from pathlib import Path from shutil import copyfile, move from unittest.mock import MagicMock, patch -import pytest - import pikepdf +import pytest +from pikepdf.models.metadata import decode_pdf_date + from ocrmypdf._jobcontext import PDFContext from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps -from pikepdf.models.metadata import decode_pdf_date try: import fitz diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 85bcc1a5..0c78d653 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -19,10 +19,10 @@ import logging from os import fspath from pathlib import Path +import pikepdf import pytest from PIL import Image -import pikepdf from ocrmypdf import optimize as opt from ocrmypdf.exec import jbig2enc, pngquant from ocrmypdf.exec.ghostscript import rasterize_pdf diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index 46643037..1fb494c4 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -19,8 +19,8 @@ import pytest import ocrmypdf from ocrmypdf._validation import _pages_from_ranges -from ocrmypdf.pdfinfo import PdfInfo from ocrmypdf.exceptions import BadArgsError +from ocrmypdf.pdfinfo import PdfInfo @pytest.mark.parametrize( diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index a775e950..b6455aa6 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -20,11 +20,11 @@ from math import isclose from tempfile import NamedTemporaryFile import img2pdf +import pikepdf import pytest from PIL import Image from reportlab.pdfgen.canvas import Canvas -import pikepdf from ocrmypdf import pdfinfo from ocrmypdf.pdfinfo import Colorspace, Encoding diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 7aaead96..05d42300 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -21,10 +21,10 @@ from os import fspath from unittest.mock import Mock import img2pdf +import pikepdf import pytest from PIL import Image -import pikepdf from ocrmypdf import leptonica from ocrmypdf.exec import ghostscript, tesseract from ocrmypdf.pdfinfo import PdfInfo diff --git a/tests/test_stdio.py b/tests/test_stdio.py index ce2076b1..1f260cfb 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -18,7 +18,7 @@ import os import sys from pathlib import Path -from subprocess import DEVNULL, PIPE, run, Popen, CalledProcessError +from subprocess import DEVNULL, PIPE, CalledProcessError, Popen, run import pytest diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 4242743c..17dcd6c4 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -20,8 +20,8 @@ from unittest.mock import patch import pytest -from ocrmypdf.cli import parser from ocrmypdf._validation import check_options +from ocrmypdf.cli import parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import unpaper diff --git a/tests/test_validation.py b/tests/test_validation.py index 0ac16e24..fb03de3b 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -23,7 +23,7 @@ import pytest import ocrmypdf._validation as vd from ocrmypdf.api import create_options -from ocrmypdf.exceptions import MissingDependencyError, BadArgsError +from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.pdfinfo import PdfInfo From 343424b4d2e51425eb5075a52682ec2ede36ef45 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Dec 2019 10:56:10 -0800 Subject: [PATCH 286/880] azure: only publish code coverage for macOS macOS (due to Homebrew) currently has the most comprehensive code coverage. Azure's code coverage feature does not merge code coverage, so last task to finish wins. --- azure-pipelines.yml | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index c982d2df..2373fb69 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -51,10 +51,6 @@ stages: testResultsFiles: "test.xml" testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" condition: succeededOrFailed() - - task: PublishCodeCoverageResults@1 - inputs: - codeCoverageTool: Cobertura - summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" - job: "Ubuntu_1804" pool: vmImage: "ubuntu-18.04" @@ -105,10 +101,6 @@ stages: testResultsFiles: "test.xml" testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" condition: succeededOrFailed() - - task: PublishCodeCoverageResults@1 - inputs: - codeCoverageTool: Cobertura - summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" - job: "Ubuntu_1604" pool: vmImage: "ubuntu-16.04" @@ -157,10 +149,6 @@ stages: testResultsFiles: "test.xml" testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" condition: succeededOrFailed() - - task: PublishCodeCoverageResults@1 - inputs: - codeCoverageTool: Cobertura - summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" - job: "macOS_Mojave" pool: vmImage: "macos-10.14" From a53a3937c2b891e0e4e5f11383cce3a13b9f87f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 20 Dec 2019 11:25:45 -0800 Subject: [PATCH 287/880] Fix exception on parsing Ghostscript error messages --- src/ocrmypdf/exec/ghostscript.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 5f6bc61f..f60e1354 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -329,7 +329,7 @@ def generate_pdfa( if _gs_error_reported(stderr): last_part = None repcount = 0 - for part in p.stdout.split('****'): + for part in stderr.split('****'): if part != last_part: if repcount > 1: log.error(f"(previous error message repeated {repcount} times)") From e4e00de79fb07129a1c1962d181372a34748a68f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Dec 2019 15:37:08 -0800 Subject: [PATCH 288/880] Add improved example demonstrating watched folder functionality Closes #466 --- .docker/Dockerfile | 2 ++ docs/batch.rst | 56 +++++++++++++++++++++++++++++--- misc/watcher.py | 70 ++++++++++++++++++++++++++++++++++++++++ requirements/watcher.txt | 1 + setup.cfg | 2 +- 5 files changed, 126 insertions(+), 5 deletions(-) create mode 100644 misc/watcher.py create mode 100644 requirements/watcher.txt diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 29de00c6..527fed5d 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -39,6 +39,7 @@ RUN pip3 install --no-cache-dir \ -r requirements/main.txt \ -r requirements/webservice.txt \ -r requirements/test.txt \ + -r requirements/watcher.txt \ . FROM base @@ -69,6 +70,7 @@ COPY --from=builder /usr/local/lib/ /usr/local/lib/ COPY --from=builder /usr/local/bin/ /usr/local/bin/ COPY --from=builder /app/misc/webservice.py /app/ +COPY --from=builder /app/misc/watcher.py /app/ # Copy minimal project files to get the test suite. COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ diff --git a/docs/batch.rst b/docs/batch.rst index 73d512bf..9e4fe749 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -198,6 +198,54 @@ and all inquiries are appreciated. Hot (watched) folders ===================== +Watched folders with Docker +--------------------------- + +The OCRmyPDF Docker image includes a watcher service. This service can +be launched as follows: + +.. code-block:: bash + + docker run \ + -v :/input \ + -v :/output \ + -e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ + -it --entrypoint python3 \ + jbarlow83/ocrmypdf \ + watcher.py + +This service will watch for a file that matches ``/input/\*.pdf`` and will +convert it to a OCRed PDF in ``/output/``. The parameters to this image are: + +.. csv-table:: watcher.py parameters for Docker + :header: "Parameter", "Description" + :widths: 50, 50 + + "``-v :/input``", "Files placed in this location will be OCRed" + "``-v :/output``", "This is where OCRed files will be stored" + "``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "This will place files in the output in {output}/{year}/{month}/{filename}" + +This service relies on polling to check for changes to the filesystem. It +may not be suitable for some environments, such as filesystems shared on a +slow network. + +Watched folders with watcher.py +------------------------------- + +The watcher service may also be run natively. + +.. code-block:: bash + + pip3 install -r reqs/watcher.txt + + env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \ + OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \ + OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ + python3 watcher.py + +Watched folders with CLI +------------------------ + To set up a "hot folder" that will trigger OCR for every file inserted, use a program like Python `watchdog `__ (supports all major @@ -225,12 +273,12 @@ told to run ``ocrmypdf`` on any .pdf added to the current directory --command='ocrmypdf "${watch_src_path}" "out/${watch_src_path}" ' \ . # don't forget the final dot -For more complex behavior you can write a Python script around to use -the watchdog API. - On file servers, you could configure watchmedo as a system service so it will run all the time. +For more complex behavior you can write a Python script around to use +the watchdog API. You can refer to the watcher.py script as an example. + Caveats ------- @@ -250,7 +298,7 @@ Caveats Alternatives ------------ -- `systemd user services `__ +- On Linux, `systemd user services `__ can be configured to automatically perform OCR on a collection of files. - `Watchman `__ is a more diff --git a/misc/watcher.py b/misc/watcher.py new file mode 100644 index 00000000..99253001 --- /dev/null +++ b/misc/watcher.py @@ -0,0 +1,70 @@ +# Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander +# +# This program is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with this program. If not, see . + +import os +import time +from datetime import datetime +from pathlib import Path + +from watchdog.events import PatternMatchingEventHandler +from watchdog.observers import Observer + +import ocrmypdf + +INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') +OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') +OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) +PATTERNS = ['*.pdf'] + + +def execute_ocrmypdf(file_path): + filename = Path(file_path).name + if OUTPUT_DIRECTORY_YEAR_MONTH: + today = datetime.today() + output_directory_year_month = Path( + f'{OUTPUT_DIRECTORY}/{today.year}/{today.month}' + ) + if not output_directory_year_month.exists(): + output_directory_year_month.mkdir(parents=True, exist_ok=True) + output_path = Path(output_directory_year_month) / filename + else: + output_path = Path(OUTPUT_DIRECTORY) / filename + print(f'New file: {file_path}.\nAttempting to OCRmyPDF to: {output_path}') + ocrmypdf.ocr(file_path, output_path) + + +class HandleObserverEvent(PatternMatchingEventHandler): + def on_any_event(self, event): + if event.event_type in ['created', 'modified']: + execute_ocrmypdf(event.src_path) + + +if __name__ == "__main__": + print( + f"Starting OCRmyPDF watcher with config:\n" + f"Input Directory: {INPUT_DIRECTORY}\n" + f"Output Directory: {OUTPUT_DIRECTORY}\n" + f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}" + ) + handler = HandleObserverEvent(patterns=PATTERNS) + observer = Observer() + observer.schedule(handler, INPUT_DIRECTORY, recursive=True) + observer.start() + try: + while True: + time.sleep(1) + except KeyboardInterrupt: + observer.stop() + observer.join() diff --git a/requirements/watcher.txt b/requirements/watcher.txt new file mode 100644 index 00000000..e8ddcd19 --- /dev/null +++ b/requirements/watcher.txt @@ -0,0 +1 @@ +watchdog >= 0.8.2, < 1.0 diff --git a/setup.cfg b/setup.cfg index 17fb9043..28823133 100644 --- a/setup.cfg +++ b/setup.cfg @@ -23,7 +23,7 @@ force_grid_wrap=0 use_parentheses=True line_length=88 known_first_party = ocrmypdf -known_third_party = PIL,PyPDF2,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,werkzeug +known_third_party = PIL,PyPDF2,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug [metadata] license_file = LICENSE From d12b27ac1ddc571aef1f7ba6be15f6c4ec500868 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Dec 2019 15:42:24 -0800 Subject: [PATCH 289/880] v9.3.0 release notes --- docs/release_notes.rst | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 949a6cc3..8268a0e8 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,19 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.3.0 +====== + +- Improved native Windows support: we now check in the obvious places in + the "Program Files" folders installations of Tesseract and Ghostscript, + rather than relying on the user to edit ``PATH`` to specify their location. + The ``PATH`` environment variable can still be used to differentiate when + multiple installations are present or the programs are installed to non- + standard locations. +- Fixed an exception on parsing Ghostscript error messages. +- Added an improved example demonstrating how to set up a watched folder + for automated OCR processing (thanks to @ianalexander for the contribution). + v9.2.0 ====== From 045bdff95a636bc71c8521d01b989a88b332aeb2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Dec 2019 16:10:43 -0800 Subject: [PATCH 290/880] azure: homebrew broke something to do with python@2? --- azure-pipelines.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 2373fb69..118cf53f 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -164,6 +164,7 @@ stages: versionSpec: "$(python.version)" - bash: | brew update + brew unlink python@2 brew install \ exempi \ ghostscript \ From 868b3b4abde71927ccd576e552aa9fd9e77654bf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Dec 2019 16:11:08 -0800 Subject: [PATCH 291/880] exec/init: os.get_exec_path() returns list not str --- src/ocrmypdf/exec/__init__.py | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 0666621a..00136ec6 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -69,11 +69,13 @@ def run(args, *, env=None, **kwargs): else: args = [program] + args[1:] - if os.name == 'nt' and not shutil.which(args[0], path=os.get_exec_path(env)): - shimmed_path = shim_paths_with_program_files(env) - new_args0 = shutil.which(args[0], path=shimmed_path) - if new_args0: - args[0] = new_args0 + if os.name == 'nt': + paths = os.pathsep.join(os.get_exec_path(env)) + if not shutil.which(args[0], path=paths): + shimmed_path = shim_paths_with_program_files(env) + new_args0 = shutil.which(args[0], path=shimmed_path) + if new_args0: + args[0] = new_args0 log.debug(args) if sys.version_info < (3, 7) and os.name == 'nt': @@ -143,9 +145,7 @@ def shim_paths_with_program_files(env=None): paths.append(os.path.join(program_files, dirname, latest_gs, 'bin')) except EnvironmentError: pass - paths.extend( - path for path in os.environ['PATH'].split(os.pathsep) if path not in set(paths) - ) + paths.extend(path for path in os.get_exec_path(env) if path not in set(paths)) return os.pathsep.join(paths) From 95ef5410c23c38ccc3bebcf3c7827df45db01d4f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Dec 2019 16:12:53 -0800 Subject: [PATCH 292/880] azure: tweak windows script --- azure-pipelines.yml | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 118cf53f..06d9e14d 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -34,17 +34,15 @@ stages: displayName: "Install system packages" - pwsh: | refreshenv - $env:path = "C:\Program Files\Tesseract-OCR;C:\Program Files\gs\gs9.50\bin;" + $env:path pip install --upgrade pip wheel pip install -r requirements/main.txt -r requirements/test.txt . tesseract --version displayName: "Install Python packages" - pwsh: | refreshenv - $env:path = "C:\Program Files\Tesseract-OCR;C:\Program Files\gs\gs9.50\bin;" + $env:path $env:pathext += ';.py' # -n auto helps Windows - pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + python3 -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml displayName: "Test" - task: PublishTestResults@2 inputs: From 708113a514924391343a90fd52f36990fc4ac20b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Dec 2019 01:14:00 -0800 Subject: [PATCH 293/880] Windows: Remove Program Files cache from ocrmypdf.exec @lru_cache doesn't work here, so let's just remove it. --- azure-pipelines.yml | 7 +++---- src/ocrmypdf/exec/__init__.py | 3 +-- 2 files changed, 4 insertions(+), 6 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 06d9e14d..6d92fb39 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -34,15 +34,14 @@ stages: displayName: "Install system packages" - pwsh: | refreshenv - pip install --upgrade pip wheel - pip install -r requirements/main.txt -r requirements/test.txt . - tesseract --version + python -m pip install --upgrade pip wheel + python -m pip install -r requirements/main.txt -r requirements/test.txt . displayName: "Install Python packages" - pwsh: | refreshenv $env:pathext += ';.py' # -n auto helps Windows - python3 -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + python -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml displayName: "Test" - task: PublishTestResults@2 inputs: diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 00136ec6..2eb6cd88 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -122,7 +122,6 @@ def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env return version -@lru_cache(maxsize=1) def shim_paths_with_program_files(env=None): if not env: env = os.environ @@ -134,7 +133,7 @@ def shim_paths_with_program_files(env=None): for dirname in os.listdir(program_files): if dirname.lower() == 'tesseract-ocr': paths.append(os.path.join(program_files, dirname)) - if dirname.lower() == 'gs': + elif dirname.lower() == 'gs': try: latest_gs = max( os.listdir(os.path.join(program_files, dirname)), From 89aa78b724ca935e1fef9f3475eae4fbc6eeb6c0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Dec 2019 02:29:52 -0800 Subject: [PATCH 294/880] docs: fix obsolete statement to "brew install tesseract-lang" Closes #469 --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 526cf0eb..4ab720f9 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -402,7 +402,7 @@ packs. If you need other languages you can optionally install them all: .. code-block:: bash - brew install tesseract --with-all-languages # Option 2: for all language packs + brew install tesseract-lang # Option 2: for all language packs Update the homebrew pip: From 054c0773a3c8ec6fe0bab6dc586d4f0217263e57 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Dec 2019 02:21:49 -0800 Subject: [PATCH 295/880] Update completions --- misc/completion/ocrmypdf.bash | 5 +++-- misc/completion/ocrmypdf.fish | 3 +++ 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 9ea30f3d..3d00bda8 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -58,7 +58,7 @@ _ocrmypdf() COMPREPLY=( $( compgen -W '{1..13}' -- "$cur" ) ) return ;; - --sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages) + --sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages|--fast-web-view) # argument required but no completions available return ;; @@ -76,7 +76,8 @@ _ocrmypdf() --max-image-mpixels --tesseract-config --tesseract-pagesegmode --help --tesseract-oem --pdf-renderer --tesseract-timeout --rotate-pages-threshold --pdfa-image-compression --user-words - --user-patterns --keep-temporary-files --output-type' \ + --user-patterns --keep-temporary-files --output-type + --no-progress-bar --pages --fast-web-view' \ -- "$cur" ) ) return else diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 81d24e6c..ce9fc9e3 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -59,6 +59,8 @@ function __fish_ocrmypdf_verbose end complete -c ocrmypdf -x -s v -l verbose -a '(__fish_ocrmypdf_verbose)' -d "set verbosity level" +complete -c ocrmypdf -x -l no-progress-bar -d "disable the progress bar" + function __fish_ocrmypdf_pdfa_compression echo -e "auto\t"(_ "let Ghostscript decide how to compress images") echo -e "jpeg\t"(_ "convert color and grayscale images to JPEG") @@ -111,5 +113,6 @@ complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence" complete -c ocrmypdf -r -l user-words -d "specify location of user words file" complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file" +complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF" complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)" From b0e92760a24539fe9426093a7a516c0ea5a6e58c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 15:52:10 -0800 Subject: [PATCH 296/880] tests: add coverage for helpers --- src/ocrmypdf/api.py | 3 +- src/ocrmypdf/helpers.py | 50 ++++++++++++--------- tests/test_helpers.py | 97 +++++++++++++++++++++++++++++++++++++++++ 3 files changed, 128 insertions(+), 22 deletions(-) create mode 100644 tests/test_helpers.py diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 9a2cdb2f..cd8e2576 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -19,6 +19,7 @@ import logging import os import sys import warnings +from contextlib import suppress from enum import IntEnum from pathlib import Path from typing import Dict, List, Optional @@ -46,7 +47,7 @@ class TqdmConsole: tqdm.write(msg.rstrip(), end='\n', file=self.file) def flush(self): - if hasattr(self.file, "flush"): + with suppress(AttributeError): self.file.flush() diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index b719d4e7..70e884d9 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -104,28 +104,36 @@ def is_file_writable(test_file): can replace it atomically. Before doing the OCR work, make sure the location is writable. """ - p = Path(test_file) - - if p.is_symlink(): - p = p.resolve(strict=False) - - # p.is_file() throws an exception in some cases - if p.exists() and p.is_file(): - return os.access( - os.fspath(p), - os.W_OK, - effective_ids=(os.access in os.supports_effective_ids), - ) - else: - try: - fp = p.open('wb') - except OSError: - return False + try: + if not isinstance(test_file, Path): + p = Path(test_file) else: - fp.close() - with suppress(OSError): - p.unlink() - return True + p = test_file + + if p.is_symlink(): + p = p.resolve(strict=False) + + # p.is_file() throws an exception in some cases + if p.exists() and p.is_file(): + return os.access( + os.fspath(p), + os.W_OK, + effective_ids=(os.access in os.supports_effective_ids), + ) + else: + try: + fp = p.open('wb') + except OSError: + return False + else: + fp.close() + with suppress(OSError): + p.unlink() + return True + except (EnvironmentError, RuntimeError) as e: + log.debug(e) + log.error(str(e)) + return False def deprecated(func): diff --git a/tests/test_helpers.py b/tests/test_helpers.py new file mode 100644 index 00000000..534e5959 --- /dev/null +++ b/tests/test_helpers.py @@ -0,0 +1,97 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +import multiprocessing +from pathlib import Path +from unittest.mock import MagicMock + +import pytest + +import ocrmypdf.helpers as helpers + + +class TestSafeSymlink: + def test_safe_symlink_link_self(self, tmp_path, caplog): + helpers.safe_symlink(tmp_path / 'self', tmp_path / 'self') + assert caplog.record_tuples[0][1] == logging.WARNING + + def test_safe_symlink_overwrite(self, tmp_path): + (tmp_path / 'regular_file').touch() + with pytest.raises(FileExistsError): + helpers.safe_symlink(tmp_path / 'input', tmp_path / 'regular_file') + + def test_safe_symlink_relink(self, tmp_path): + (tmp_path / 'regular_file_a').touch() + (tmp_path / 'regular_file_b').write_bytes(b'ABC') + (tmp_path / 'link').symlink_to(tmp_path / 'regular_file_a') + helpers.safe_symlink(tmp_path / 'regular_file_b', tmp_path / 'link') + assert (tmp_path / 'link').samefile(tmp_path / 'regular_file_b') or ( + tmp_path / 'link' + ).read_bytes() == b'ABC' + + +def test_no_cpu_count(monkeypatch): + def cpu_count_raises(): + raise NotImplementedError() + + monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises) + with pytest.warns(expected_warning=UserWarning): + assert helpers.available_cpu_count() == 1 + + +def test_deprecated(): + @helpers.deprecated + def old_function(): + return 42 + + with pytest.warns(expected_warning=DeprecationWarning): + assert old_function() == 42 + + +class TestFileIsWritable: + @pytest.fixture + def non_existent(self, tmp_path): + return tmp_path / 'nofile' + + @pytest.fixture + def basic_file(self, tmp_path): + basic = tmp_path / 'basic' + basic.touch() + return basic + + def test_plain(self, non_existent): + assert helpers.is_file_writable(non_existent) + + def test_symlink_loop(self, tmp_path): + loop = tmp_path / 'loop' + loop.symlink_to(loop) + assert not helpers.is_file_writable(loop) + + def test_chmod(self, basic_file): + assert helpers.is_file_writable(basic_file) + basic_file.chmod(0o400) + assert not helpers.is_file_writable(basic_file) + basic_file.chmod(0o000) + assert not helpers.is_file_writable(basic_file) + + def test_permission_error(self, basic_file): + pathmock = MagicMock(spec_set=basic_file) + pathmock.is_symlink.return_value = False + pathmock.exists.return_value = True + pathmock.is_file.side_effect = PermissionError + assert not helpers.is_file_writable(pathmock) From 63de7e1677819b187923dcce7628ae5209cd6e4a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 16:14:52 -0800 Subject: [PATCH 297/880] Improve error message for unreadable input files --- src/ocrmypdf/_pipeline.py | 8 +++++--- src/ocrmypdf/_sync.py | 8 ++++++-- src/ocrmypdf/_validation.py | 4 ++-- tests/test_main.py | 9 +++++++++ 4 files changed, 22 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 2705b2ab..cad9d74b 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -19,6 +19,7 @@ import os import re import sys from datetime import datetime, timezone +from pathlib import Path from shutil import copyfileobj import img2pdf @@ -123,7 +124,7 @@ def _pdf_guess_version(input_file, search_window=1024): return '' -def triage(input_file, output_file, options, log): +def triage(original_filename, input_file, output_file, options, log): try: if _pdf_guess_version(input_file): if options.image_dpi: @@ -135,8 +136,9 @@ def triage(input_file, output_file, options, log): safe_symlink(input_file, output_file) return output_file except EnvironmentError as e: - log.error(e) - raise InputFileError() from e + log.debug(f"Temporary file was at: {input_file}") + msg = str(e).replace(input_file, original_filename) + raise InputFileError(msg) from e triage_image_file(input_file, output_file, options, log) return output_file diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 5b27a8f0..87546049 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -332,11 +332,15 @@ def run_pipeline(options, api=False): work_folder = mkdtemp(prefix="com.github.ocrmypdf.") try: check_requested_output_file(options) - start_input_file = create_input_file(options, work_folder) + start_input_file, original_filename = create_input_file(options, work_folder) # Triage image or pdf origin_pdf = triage( - start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log + original_filename, + start_input_file, + os.path.join(work_folder, 'origin.pdf'), + options, + log, ) # Gather pdfinfo and create context diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index b53f0fc4..1ebd422c 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -380,12 +380,12 @@ def create_input_file(options, work_folder): target = os.path.join(work_folder, 'stdin') with open(target, 'wb') as stream_buffer: copyfileobj(sys.stdin.buffer, stream_buffer) - return target + return target, "" else: try: target = os.path.join(work_folder, 'origin') safe_symlink(options.input_file, target) - return target + return target, os.fspath(options.input_file) except FileNotFoundError: raise InputFileError(f"File not found - {options.input_file}") diff --git a/tests/test_main.py b/tests/test_main.py index 0380508b..0e336d80 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -270,6 +270,15 @@ def test_input_file_not_found(caplog, no_outpdf): assert input_file in caplog.text +def test_input_file_not_readable(caplog, resources, outdir, no_outpdf): + input_file = outdir / 'trivial.pdf' + shutil.copy(resources / 'trivial.pdf', input_file) + input_file.chmod(0o000) + result = run_ocrmypdf_api(input_file, no_outpdf) + assert result == ExitCode.input_file + assert input_file in caplog.text + + def test_input_file_not_a_pdf(caplog, no_outpdf): input_file = __file__ # Try to OCR this file result = run_ocrmypdf_api(input_file, no_outpdf) From 0c0d53b10ff72c79c76d571ee97f9f68bd721c0a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 16:37:51 -0800 Subject: [PATCH 298/880] tests: AcroForm test case did not work correctly; fixed --- src/ocrmypdf/_pipeline.py | 2 +- tests/resources/acroform.pdf | Bin 0 -> 10733 bytes tests/test_acroform.py | 11 ++++++++++- tests/test_main.py | 2 +- 4 files changed, 12 insertions(+), 3 deletions(-) create mode 100644 tests/resources/acroform.pdf diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index cad9d74b..fd1aadb4 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -183,7 +183,7 @@ def validate_pdfinfo_options(context): ) raise InputFileError() else: - log.warn( + log.warning( "This PDF has a fillable form. " "Chances are it is a pure digital " "document that does not need OCR." diff --git a/tests/resources/acroform.pdf b/tests/resources/acroform.pdf new file mode 100644 index 0000000000000000000000000000000000000000..b80eb44a2e78cc299a960cd80caab00cdc795556 GIT binary patch literal 10733 zcmY!laBi8ykI}%)HdZqRgt)6a_<3MTpiMX4#7$tC$k3Wi2@cKU9aIVGt@`ffRiC8-cC z`kpS)Hm(*hJk^(o2jd_k)ef)iR=o6$EEir79Si>4#Jnr0PcmrKahJM;j^_DHz7u*%g-*r6%U`a%J30 zN=gw(NJvWXViV%o=g{;~qGJbxQjFsfQ-Oz!L6i7{gcwxz{ZC9tN}0j^$jCrR@V{jF ze^w3y2Kz^x??J%`H3=G^#U+VFB^5=fX}ku!s;aL3ZoKdi<~1@!#2vN(Q!q3zg!#(C zSU*_7#6Ukx!Q33hwFwCi5)u+7r7$o}nok=OdEu7+542Hr8|Q#Pk9<^gxNj z#zx;UxhUT)zbIG12oj{esU?Xii6w~&M&J+*NK8*HRdih9HG7 z!}9X-ONteYKte(K&iQ#Isd**E3WgB<&PDkJPWctl26_exAZ%g)BF&BT3}W?tQ&Tb% zaY+Uw7NzEuKwS}(TAW{6l$;7mtpP>($-${5(fR={Zu*|NiRr0MvHB7INm;4MB{nwt zp1B4JhTx=RXJ==pACzB`Sdyw>07)(&pCS3e7!h+$$nG$-&@)f~5wZH9U;>2!C@(@% zrm=;-Z(>PNW<|6i)K4HEA+cliaXB~E&Q3qLG^qrXazO?|90hedB0U%?q;r8XtDyxb z6pK>1(zp!tOcg-b&|JaLT%jnHtC$N*6q^v?AIvNbatW#z4Gks= zn|CbPqJPEvI;;QefYw>tl15U*l{5^ENb)HSk_M9@ENPT)?wr15(cIn*3O6*X*u@UL{NK#^d0FL$kY$r* z6*mTDmVT&BZ1Z}!!?$@|T}0xW+x)wBJi5G?wf(@mGJCs>$S(C`Jt|u*HfNX~6Ml9n zs_dBL>^-S_3b|rFh$k@U$VirA$ztY+aKuvByJZ#?mw+lKXf0-BZmjR8U;@eK{z+NE zCAo-fo^j75;I4nOfx!RpcZJmtGZL96FXddbsIySWEosV!1MZHGbi7x%{jXkcamV@g z?%%aB?=6_cE!dA)6!1$dKf~Uio3Pzka_xt>x%)aMm6v?o;5JjRVD;>cvL*hBTEdoQ z-i%C(Qk@cvvYcP6xpzxyU)ZK?_l)x-o2|A6JNm8+;@ar4q$}w5oO8j~8EcQ8ioe2= zQ(b7UGo$?Ix^ovVHqB`0_w3tv@y8E`ri%C-ebGrZ;RaQmGcN7f?6^!Sc*B9xJ-qDq zKgR0sU=(XBKDSf1MODr|`*U&F51TD}jb^O>mcOd9z2eWe>ocNNE>)R0-AKE-fz~t7AT>5bgkT|;i2gp;+rlzJw)^PCuE&~gL(fj1w3c z1R4GxVYtM=$iT$Jzz(toidooM*%+CaIXJnv1sE8ZnVFbbSo)!Ij7-c7EUavTLc)q7 zN``@=%8rTb9GqNY;u4ZlDynMg8k$;0#wMm_<`$M#PR=f_ZtfnQUO~Yjp<&?>kx@y> zDXD4c8JSsyMa3nhW#tu>RgF!}Ev;?s9i3ehCrzF*b=ve9GiNPaw0OzVWy@ErT(xo2 z<}F*dZQrqT*TF-Fj~qRA{KUyq7cX7Da`oEv8#iw~eDwIq(`V0LynOZHY% zE7s?E=7vJSrm$P5qca!!&FY)%w>0S4(u-H-^;H-9m@KLGeP{CO!;48_?Pb~*`O=$x zcBN~c3-*nC-Ds4=b|9G7YnyJ`NxfACD-9<xr8q=LpI_=f@;eda>l;%e`HPv>G`Q%2*wZ1rBL8Hf0|5QrtLm zqrQJ;tnL&;O_z7G-Wf1Ehq+vNnz`=lVXa=3hM5k>0==XRghd##{1S3!o(r;-(s;?$ zZPDB{QN2{WSZHk&!<7Zh8Z+NUm8dh#U`trUU0t%6|I|XAl+PWPyCz>uVhZ$p5bScy zS4x@b#3dII2F{jdH(`!muc^t0mu|ii5f~@WvwWAvzi`ux@VP(luzcAVnZhWzK*QJK z`YYxf{qFs$4({Rs6_y7MsRcSXG6kG9kZ@QfS2%AeC)=SlSJhk99v_x3tqgT9EQvk& zvs`>i#=)eiAsHDfd>En>*0V-7E))xK73*?K3beN@~omaO7bmNhG>eMy17QUMuy0A$iG^ml|wBjZK9H zQ`GN7spx4+cAnbtNARzB>B4mlFJ>%Bz8D}_5@PGH=ArN6m2PYXj5$nQw>x&&_uB68 z{AtR~?j(CpMgPfKw}@bobF6yPoB-ZOhfNA2z=4)XHyFZI1vNA($RoY-1)L`+Pdp6FT=-qfhc@1VY{LsE%X(9_@o%jc^MQjIpr z-Ye(4HRDq<@VXtDook|bIq*{El~plUc3Dlncs?()KdASwrpKDuW$hD7og_0^M8&%z zeMAzbG9Cz&;`+eB&@k&}qQzOxOL1Y#HFSb5KJz_g@*`XzAXH<;tY?NNL=UJ|OmPud zG>2_zYs=r#)a+AUr(CqTl9tSI75!jyAoOye@%m(sIc+Q_6AhcHCZ7%6*q^WI@-(Kc zmr2z%{JTceGKLw)nPy&CHPz6R)8#;5K-$V&si{$wPY*CJT%5Tx@W6>5e2aejGKAzN zC}}f=Y-IH7Nt@%xoR@rNW1ZtOyEh(*qLMbVyjLVtF4XE+F>~4+b*G{^9KsjQC?<)7 zGS#)X1-nh@IXTI3vP-P8-+6tJrB_5+1GIOBY5FCUM+7fi%*d$o#W^62Tjolg(Py?b z>Yg|HIc@hXd9m``iiKXHEnLo5Ts=?QI<68;+q_chWfrqg_hGLN#)W>ROS)E6u}oW_ z(Y4S*;EIB?PoUH`<)Ec;(M!%Bi##MX#b&WsZL#sIzGX2U-MbWw8RmL9%)tt<_$vm?xGh z=y70CiJNDj;ejih=UtRtEvB%ojas{NV^6Zg{r{K1#cqH9!3+Hd?=5)GCM5V$i_3uX z$TCJ|fy*KDl113kT3)qm5S=4j6M0s>JaNyX6Zg!lR8>T8*)S=GsHRcATZhCB#$<=D77kWo2c%8EH#FSR_h zT-EC}O}D$%Bk0g&6Pph=oImf-SvaA#%7gD)$hkxdkta)d4ZLj*m|VItbH_f96H`Jz z1v9_Ad$FW{*$WO?Sdm z=m~u$wR+*Dz?&CK`My2+ki4wRwkAMY^FWePVGq7_Kmfey0lsJ#&M&9O*7*B7L}-pCoZrE zP+KZ;HC^MZt=ED(K|8E;>@H^QR(mci&Mv%jkM$AHEK%ka>0OCz)AlkpI?u{_aYb@g zOq91vcuAWwM@idG?kSlPi&iXAFLNt#YFWalymQ5w%vW=~B9}4Dh$>0r6|J4}!Q$`& z>uVXlS<9JLX}I=Ab#O|xEOmLhZi+hR*^sP|jX5z|PSeC@E_~j0NPh0YRG#M7mmJDB zEi2|cuv}Fkvnlc!&Q5AOm>dw%;&t&^Y3MPvKjp7fdlrT}hB_=*Td+hRVC$_$ zx3FhrO-BaNpSbRgAReHO#PdbY?d%8zz z?LNH0ec^h&_XQt3jt4BaI`hkB-@roiyfKda$d$mL3Hn+}OiCLXuBVyh9-s-tV{e4e+(-n!|OB>VLx4ewKZ zo(VVlc{yBW*?flCSJ!j)s*h@mPuY1bJrVB3>GWdq?me%y*nEOL&K*d;7vZ$hPi+QE zn!v)+xNE?$xr8Wk#zMidJ;Su51(yzV1XChDYc@0=iTs9`A6@c)mBG3YfNIbysC6;0fTQa2dfSjgK^~T+1k=gM;`G9WSpL{ zz@tUel`kd4^0nouT?w;}FgGP-rsVglmcYRruXCZtOV*j#ZCbzAcGI>=m@ zfyWGc|6f720b85={{O3lmE6SAA0(ppZ3*j@nHd`D`zaVf`e2xyZ=Wf*^KKjPw8bCw zH!zEwxXeLN;kJ#=feQ@T#TR*ZFHpTDrWL)C@8+iE^15oP@^vQvv7fXkW&Ntzuh#8~ z4&&$lz{y-&n%t6~;xE$cw8H)A7D?-CcS7tR2(q6Ftnl)+k>{NEbK?pB85hpBeXcRg z^`7VE`tbTEF)4p_rq+AC3wpvcGj#v5zE#q4jalcP+-}D!7R+*e=ft>!#ep5ivR2vM zI~-j!k!iP1=y&ZGHuF|Q>0G~{lC9ZvY@Kp_WJz-48q?zmOcxdyesM^Si+XKgxJ541 z=qReZINbR}&feeY=jPZwzPNZ9 z<6rG9ZNHn|AM}%bmt8R_aBs$8=Q-a#1k2bMcq>Y5P%33{RQmrZNH@$pK>huXr(z+W zE(^}N8SznYN4S*qmz=c=Rl?p?OnWxxy6(revnPu8MmLqp%I}wrZQ=iO;Y!o;b1e6G zG9?0@wypT=Qm{qy#F|I);zImqEciD*kJuQn)Wcv&*2aGvH-uuiQ2_c3bW~uMN7<_n&cZTl?_y z)*9Uc-2&U@*G_L;HVdE4TXs{_SGKxhN>TO1_pbZqpZqTFxUel{vE?n}H%|-ro8(w* z&s=a$j8pk>$NGHu%s&Upw`^n#h?rtH=1YwiZC9mISZ~m!& zfZl?uP8`N&Q2{+NJ>r5%(GQ` zzk9!uLS~AsQn;zFfp39xYDT6 z4%lfa$@-}|skxvT41GgAL#+BLatnNY;aag;mz#@KX8}@ppy(?|Nz*sfGsdbZF(o5E zxf~IYC>lXCIZ25njvzJR`9;}jIr-%f4};?a5==G~xdm3f`6-!cl`e@Tsdh#NhL*a9 zM!LoZA%O!$QH(QqRx`;uT~?NcvNftzbMxCF@!wfy_-z zO4Lm;Gch(WPclojFtdc2i=-FH+=7%etAL{Xl+xtXB5Z-903MuFaL&&wOD!q}<*$I8 z(sbR-JR49TAhaSG2~CAo#h~dlJ3||NQ~@k1p+?&o>LICy2}0G|=z}sgQbq@-T8LEkDFjGTJes;jgNvjPAW89P z>Y`e3aUnWEsd*{3O65xSc9?xEq%}P_C+JOJLj)$4`ry@cpanUYBLtfQ3<5T6yjXrt z|EzYP!Ikgld^G#aOqjUXj=pLS5nJ+#U1iyFvD-HpStJUdr)*l1*Au*Jm0$jrdR*vNF#1=9@y<^}=RTUMOXzf|CJ z(%&nzz(?!KkI++LT*k%3d42TfGP8XX3gme-QAR)msGW?skkmK zGAq#Vf+82A<(HBD`Q|}V+i@Iw++*8NXuj()8YfXnh<>fp@}8tszF2OGB7-= z1{s*t+zPqb)Yz8&b!gIO^)UFtz. +import logging + import pytest import ocrmypdf @@ -30,4 +32,11 @@ def acroform(resources): def test_acroform_and_redo(acroform, caplog, no_outpdf): with pytest.raises(ocrmypdf.exceptions.InputFileError): check_ocrmypdf(acroform, no_outpdf, '--redo-ocr') - assert '--redo-ocr is not currently possible' in caplog.text + assert '--redo-ocr is not currently possible' in caplog.text + + +def test_acroform_message(acroform, caplog, spoof_tesseract_noop, outpdf): + caplog.set_level(logging.INFO) + check_ocrmypdf(acroform, outpdf, env=spoof_tesseract_noop) + assert 'fillable form' in caplog.text + assert '--force-ocr' in caplog.text diff --git a/tests/test_main.py b/tests/test_main.py index 0e336d80..cc2686cf 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -276,7 +276,7 @@ def test_input_file_not_readable(caplog, resources, outdir, no_outpdf): input_file.chmod(0o000) result = run_ocrmypdf_api(input_file, no_outpdf) assert result == ExitCode.input_file - assert input_file in caplog.text + assert str(input_file) in caplog.text def test_input_file_not_a_pdf(caplog, no_outpdf): From c36e9950ae2b64f1e4df3ea724803f4036a1a983 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 17:51:09 -0800 Subject: [PATCH 299/880] tests: test TqdmConsole --- src/ocrmypdf/api.py | 10 +++++++- tests/test_api.py | 61 +++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) create mode 100644 tests/test_api.py diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index cd8e2576..3270b25d 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -32,7 +32,15 @@ from .cli import parser class TqdmConsole: - """Wrapper to log messages in a way that is compatible with tqdm progress bar""" + """Wrapper to log messages in a way that is compatible with tqdm progress bar + + This routes log messages through tqdm so that it can print them above the + progress bar, and then refresh the progress bar, rather than overwriting + it which looks messy. + + For some reason Python 3.6 prints extra empty messages from time to time, + so we suppress those. + """ def __init__(self, file): self.file = file diff --git a/tests/test_api.py b/tests/test_api.py new file mode 100644 index 00000000..fd09f262 --- /dev/null +++ b/tests/test_api.py @@ -0,0 +1,61 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +from io import StringIO + +import pytest +from tqdm import tqdm + +import ocrmypdf + + +def test_raw_console(): + bio = StringIO() + tqconsole = ocrmypdf.api.TqdmConsole(file=bio) + tqconsole.write("Test") + tqconsole.flush() + assert "Test" in bio.getvalue() + + +def test_tqdm_console(): + log = logging.getLogger() + log.setLevel(logging.INFO) + + formatter = logging.Formatter('%(message)s') + + bio = StringIO() + console = logging.StreamHandler(ocrmypdf.api.TqdmConsole(file=bio)) + console.setFormatter(formatter) + + log.addHandler(console) + + def before_pbar(message): + # Ensure that log messages appear before the progress bar, even when + # printed after the progress bar updates. + v = bio.getvalue() + pbar_start_marker = '|#' + return v.index(message) < v.index(pbar_start_marker) + + with tqdm(total=2, file=bio, disable=False) as pbar: + pbar.update() + msg = "1/2 above progress bar" + log.info(msg) + assert before_pbar(msg) + + log.info("done") + assert not before_pbar("done") From c4dc5269d29442e1b4fe62303b40adc9595fba14 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 21:16:16 -0800 Subject: [PATCH 300/880] tests: remove some obscure things from coverage --- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/hocrtransform.py | 10 +++++----- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 1ebd422c..8b874ba6 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -314,7 +314,7 @@ def check_options(options): check_dependency_versions(options) -def check_closed_streams(options): +def check_closed_streams(options): # pragma: no cover """Work around Python issue with multiprocessing forking on closed streams https://bugs.python.org/issue28326 diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index e7c28e54..809f858e 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -88,7 +88,7 @@ class HocrTransform: if self.width is None or self.height is None: raise HocrTransformError("hocr file is missing page dimensions") - def __str__(self): + def __str__(self): # pragma: no cover """ Return the textual content of the HTML body """ @@ -190,7 +190,7 @@ class HocrTransform: pt = self.pt_from_pixel(pxl_coords) # draw the bbox border - if showBoundingboxes: + if showBoundingboxes: # pragma: no cover pdf.rect( pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1 ) @@ -231,7 +231,7 @@ class HocrTransform: pdf.save() @classmethod - def polyval(cls, poly, x): + def polyval(cls, poly, x): # pragma: no cover return x * poly[0] + poly[1] def _do_line( @@ -269,7 +269,7 @@ class HocrTransform: # of the line box baseline_y2 = self.height - (line_box.y2 + intercept) - if showBoundingboxes: + if showBoundingboxes: # pragma: no cover # draw the baseline in magenta, dashed pdf.setDash() pdf.setStrokeColorRGB(0.95, 0.65, 0.95) @@ -318,7 +318,7 @@ class HocrTransform: font_width = pdf.stringWidth(elemtxt, fontname, fontsize) # draw the bbox border - if showBoundingboxes: + if showBoundingboxes: # pragma: no cover pdf.rect( box.x1, self.height - line_box.y2, box_width, line_height, fill=0 ) From 16dd8b54a8d2dfdfd3eb576d8b96af62bd03651f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 22:38:38 -0800 Subject: [PATCH 301/880] ghostscript: don't delete output_file that will never exist We stream output now, so no point in deleting. --- src/ocrmypdf/exec/ghostscript.py | 4 ---- 1 file changed, 4 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index f60e1354..a72572f6 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -195,8 +195,6 @@ def rasterize_pdf( try: p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True) except CalledProcessError as e: - with suppress(OSError): - Path(output_file).unlink() # no unfinished files log.error(e.stderr.decode(errors='replace')) raise SubprocessOutputError('Ghostscript rasterizing failed') else: @@ -320,8 +318,6 @@ def generate_pdfa( except CalledProcessError as e: # Ghostscript does not change return code when it fails to create # PDF/A - check PDF/A status elsewhere - with suppress(OSError): - Path(output_file).unlink() log.error(e.stderr.decode(errors='replace')) raise SubprocessOutputError('Ghostscript PDF/A rendering failed') else: From 25d2b0cda4cb50de4cc0f13b14af555b46d4e470 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 30 Dec 2019 22:38:50 -0800 Subject: [PATCH 302/880] test: environment warnings/cleanup --- tests/conftest.py | 9 ++++++++- tests/test_helpers.py | 2 +- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 8521827f..075e004e 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -201,7 +201,10 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): @pytest.helpers.register def run_ocrmypdf_api(input_file, output_file, *args, env=None): - "Run ocrmypdf and let caller deal with results" + """Run ocrmypdf via API and let caller deal with results + + Does not currently have a way to manipulate the PATH except for Tesseract. + """ options = cli.parser.parse_args( [str(input_file), str(output_file)] @@ -211,6 +214,10 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if env: options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] + if 'spoof' in first_path: + assert 'gs' not in first_path, "use run_ocrmypdf() for gs" + assert 'tesseract' in first_path if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 534e5959..f3c964d7 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -59,7 +59,7 @@ def test_deprecated(): def old_function(): return 42 - with pytest.warns(expected_warning=DeprecationWarning): + with pytest.deprecated_call(): assert old_function() == 42 From 4b759af6ffbc5637deedc7856f3a804722a046f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 15:33:03 -0800 Subject: [PATCH 303/880] tests: fix problems with ghostscript spoofers --- tests/spoof/gs_raster_failure.py | 6 +++--- tests/spoof/gs_render_failure.py | 2 +- tests/test_ghostscript.py | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/spoof/gs_raster_failure.py b/tests/spoof/gs_raster_failure.py index 067f8a6f..7619aae2 100755 --- a/tests/spoof/gs_raster_failure.py +++ b/tests/spoof/gs_raster_failure.py @@ -35,13 +35,13 @@ def main(): print('SPOOFED: ' + os.path.basename(__file__)) sys.exit(0) - # For any rendering calls (device == pdfwrite) call real ghostscript - if '-sDEVICE=pdfwrite' in sys.argv: + # For non-image rastering calls, use real ghostscript + if '-sDEVICE=pdfwrite' in sys.argv or '-sDEVICE=txtwrite' in sys.argv: real_ghostscript(sys.argv) return # Fail - print("ERROR: Ghost story archive not found") + print("ERROR: Ghost story archive not found", file=sys.stderr) sys.exit(1) diff --git a/tests/spoof/gs_render_failure.py b/tests/spoof/gs_render_failure.py index 65509f5d..d0c1d60d 100755 --- a/tests/spoof/gs_render_failure.py +++ b/tests/spoof/gs_render_failure.py @@ -40,7 +40,7 @@ def main(): return # Fail - print("ERROR: Casper is not a friendly ghost") + print("ERROR: Casper is not a friendly ghost", file=sys.stderr) sys.exit(1) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index c74b4952..45020402 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -119,7 +119,7 @@ def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): p, out, err = run_ocrmypdf( resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail ) - print(err) + assert 'Casper is not a friendly ghost' in err assert p.returncode == ExitCode.child_process_error @@ -127,7 +127,7 @@ def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): p, out, err = run_ocrmypdf( resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_gs_raster_fail ) - print(err) + assert 'Ghost story archive not found' in err assert p.returncode == ExitCode.child_process_error From 96ee21aee92b2a6389127ea5d2cfba7b68c8ca78 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 15:39:45 -0800 Subject: [PATCH 304/880] Try to set up subprocess coverage better --- .coveragerc | 6 +++--- tests/conftest.py | 10 +++++++++- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/.coveragerc b/.coveragerc index b4e940b7..703ae66d 100644 --- a/.coveragerc +++ b/.coveragerc @@ -1,5 +1,3 @@ -# Coverage isn't really compatible with subprocesses so results are unreliable - [paths] source = src @@ -8,9 +6,11 @@ source = [run] branch = true parallel = true +concurrency = + thread + multiprocessing source = src/ocrmypdf - tests omit = tests/spoof/* diff --git a/tests/conftest.py b/tests/conftest.py index 075e004e..90e9057d 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -229,13 +229,21 @@ def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=Tr "Run ocrmypdf and let caller deal with results" if env is None: - env = os.environ + env = os.environ.copy() p_args = ( OCRMYPDF + [str(arg) for arg in args if arg is not None] + [str(input_file), str(output_file)] ) + + # Tell subprocess where to find coverage.py configuration + # This has no effect except when coverage is running + # Details: https://coverage.readthedocs.io/en/coverage-5.0/subprocess.html + coverage_rc = Path(__file__).parent.parent / '.coveragerc' + assert coverage_rc.exists() + env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) + p = run( p_args, stdout=PIPE, stderr=PIPE, universal_newlines=universal_newlines, env=env ) From 2f1c743227b062b08a983d9e235f47f1e8e3f3f6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 16:23:41 -0800 Subject: [PATCH 305/880] Rewrite main pool loop pytest-cov documentation recommends using explicit management of multiprocessing.Pool rather than the context manager. This is supposed to work better for collecting coverage data, particularly on Windows. --- src/ocrmypdf/_sync.py | 55 +++++++++++++++++++++++++++---------------- tests/conftest.py | 6 ----- 2 files changed, 35 insertions(+), 26 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 87546049..12f85738 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -267,28 +267,43 @@ def exec_concurrent(context): unit='page', unit_scale=0.5, disable=not context.options.progress_bar, - ) as pbar, Pool( - processes=max_workers, - initializer=initializer, - initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS), - ) as pool: - results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) - while True: - try: - page_result = results.next() - sidecars[page_result.pageno] = page_result.text - pbar.update() - ocrgraft.graft_page(page_result) - pbar.update() - except StopIteration: - break - except (Exception, KeyboardInterrupt): + ) as pbar: + pool = Pool( + processes=max_workers, + initializer=initializer, + initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS), + ) + try: + results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) + while True: + try: + page_result = results.next() + sidecars[page_result.pageno] = page_result.text + pbar.update() + ocrgraft.graft_page(page_result) + pbar.update() + except StopIteration: + break + except KeyboardInterrupt: + # Terminate pool so we exit instantly + pool.terminate() + # Don't try listener.join() here, will deadlock + raise + except Exception: + if not os.environ.get("PYTEST_CURRENT_TEST", ""): + # Unless inside pytest, exit immediately because no one wants + # to wait for child processes to finalize results that will be + # thrown away. Inside pytest, we want child processes to exit + # cleanly so that they output an error messages or coverage data + # we need from them. pool.terminate() - log_queue.put_nowait(None) # Terminate log listener - # Don't try listener.join() here, will deadlock - raise + raise + finally: + # Terminate log listener + log_queue.put_nowait(None) + pool.close() + pool.join() - log_queue.put_nowait(None) listener.join() # Output sidecar text diff --git a/tests/conftest.py b/tests/conftest.py index 90e9057d..4f0a3344 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -28,12 +28,6 @@ from ocrmypdf import api, cli pytest_plugins = ['helpers_namespace'] -try: - from pytest_cov.embed import cleanup_on_sigterm -except ImportError: - pass -else: - cleanup_on_sigterm() # pylint: disable=E1101 # pytest.helpers is dynamic so it confuses pylint From 422ea9777ecc321201799348005fdcbf8b3f92aa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 17:09:23 -0800 Subject: [PATCH 306/880] Remove session scope from fixtures pytest seems to prepare os.environ in complex ways, so we want to ensure these fixtures are not reused. --- tests/conftest.py | 4 ++-- tests/test_ghostscript.py | 8 ++++---- tests/test_main.py | 4 ++-- tests/test_stdio.py | 2 +- tests/test_unpaper.py | 2 +- 5 files changed, 10 insertions(+), 10 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 4f0a3344..37054626 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -131,12 +131,12 @@ def spoof(tmp_path_factory, **kwargs): return env -@pytest.fixture(scope='session') +@pytest.fixture def spoof_tesseract_noop(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_noop.py') -@pytest.fixture(scope='session') +@pytest.fixture def spoof_tesseract_cache(tmp_path_factory): if running_in_docker(): return os.environ.copy() diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 45020402..32d66455 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -31,28 +31,28 @@ run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api spoof = pytest.helpers.spoof -@pytest.fixture(scope='session') +@pytest.fixture def spoof_no_tess_gs_render_fail(tmp_path_factory): return spoof( tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' ) -@pytest.fixture(scope='session') +@pytest.fixture def spoof_no_tess_gs_raster_fail(tmp_path_factory): return spoof( tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' ) -@pytest.fixture(scope='session') +@pytest.fixture def spoof_no_tess_no_pdfa(tmp_path_factory): return spoof( tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py' ) -@pytest.fixture(scope='session') +@pytest.fixture def spoof_no_tess_pdfa_warning(tmp_path_factory): return spoof( tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' diff --git a/tests/test_main.py b/tests/test_main.py index cc2686cf..cd12bb5d 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -45,12 +45,12 @@ spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] -@pytest.fixture(scope='session') +@pytest.fixture def spoof_tesseract_crash(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_crash.py') -@pytest.fixture(scope='session') +@pytest.fixture def spoof_tesseract_big_image_error(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_big_image_error.py') diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 1f260cfb..82a32099 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -33,7 +33,7 @@ run_ocrmypdf_api = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -@pytest.fixture(scope='session') +@pytest.fixture def spoof_tess_bad_utf8(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 17dcd6c4..f0e90b80 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -43,7 +43,7 @@ def have_unpaper(): return True -@pytest.fixture(scope="session") +@pytest.fixture def spoof_unpaper_oldversion(tmp_path_factory): return spoof(tmp_path_factory, unpaper="unpaper_oldversion.py") From aeb7b142a96937be8177a7446bb88599a870b89d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 17:10:51 -0800 Subject: [PATCH 307/880] tests: skip tests not compatible with coverage For reasons not entirely clear, stdout will get some data injected when pytest-cov is running. Our tests that check for clean stdout need to ignore this. We check for an environment variable that is defined only when coverage is running. --- tests/test_stdio.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 82a32099..250ca6f2 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -56,6 +56,9 @@ def test_stdin(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): + if 'COV_CORE_DATAFILE' in spoof_tesseract_noop: + pytest.skip(msg="Coverage uses stdout") + input_file = str(resources / 'francais.pdf') output_file = str(outpdf) @@ -121,6 +124,9 @@ def test_bad_locale(): reason="Windows does not like this; not sure how to fix", ) def test_dev_null(spoof_tesseract_noop, resources): + if 'COV_CORE_DATAFILE' in spoof_tesseract_noop: + pytest.skip(msg="Coverage uses stdout") + p, out, err = run_ocrmypdf( resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop ) From 1037d73efbf7cc91856568fffb6bbf12008e83e4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 31 Dec 2019 17:20:28 -0800 Subject: [PATCH 308/880] tests: use smaller files for ghostscript --- tests/test_ghostscript.py | 34 +++++++++++++++++----------------- 1 file changed, 17 insertions(+), 17 deletions(-) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 32d66455..3aed104a 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -60,18 +60,18 @@ def spoof_no_tess_pdfa_warning(tmp_path_factory): @pytest.fixture -def linn(resources): - path = resources / 'linn.pdf' +def francais(resources): + path = resources / 'francais.pdf' return path, pikepdf.open(path) -def test_rasterize_size(linn, outdir, caplog): - path, pdf = linn +def test_rasterize_size(francais, outdir, caplog): + path, pdf = francais page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3]) assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) - target_size = Decimal('200.0'), Decimal('150.0') - target_dpi = 42.0, 4242.0 + target_size = Decimal('50.0'), Decimal('30.0') + forced_dpi = 42.0, 4242.0 log = logging.getLogger() rasterize_pdf( @@ -81,21 +81,21 @@ def test_rasterize_size(linn, outdir, caplog): target_size[1] / page_size[1], raster_device='pngmono', log=log, - page_dpi=target_dpi, + page_dpi=forced_dpi, ) with Image.open(outdir / 'out.png') as im: assert im.size == target_size - assert im.info['dpi'] == target_dpi + assert im.info['dpi'] == forced_dpi -def test_rasterize_rotated(linn, outdir, caplog): - path, pdf = linn +def test_rasterize_rotated(francais, outdir, caplog): + path, pdf = francais page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3]) assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) - target_size = Decimal('200.0'), Decimal('150.0') - target_dpi = 42.0, 4242.0 + target_size = Decimal('50.0'), Decimal('30.0') + forced_dpi = 42.0, 4242.0 log = logging.getLogger() caplog.set_level(logging.DEBUG) @@ -106,13 +106,13 @@ def test_rasterize_rotated(linn, outdir, caplog): target_size[1] / page_size[1], raster_device='pngmono', log=log, - page_dpi=target_dpi, + page_dpi=forced_dpi, rotation=90, ) with Image.open(outdir / 'out.png') as im: assert im.size == (target_size[1], target_size[0]) - assert im.info['dpi'] == (target_dpi[1], target_dpi[0]) + assert im.info['dpi'] == (forced_dpi[1], forced_dpi[0]) def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): @@ -125,7 +125,7 @@ def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_gs_raster_fail + resources / 'francais.pdf', outpdf, env=spoof_no_tess_gs_raster_fail ) assert 'Ghost story archive not found' in err assert p.returncode == ExitCode.child_process_error @@ -133,7 +133,7 @@ def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf): p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_no_pdfa + resources / 'francais.pdf', outpdf, env=spoof_no_tess_no_pdfa ) assert ( p.returncode == ExitCode.pdfa_conversion_failed @@ -141,4 +141,4 @@ def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf): def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf): - check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_pdfa_warning) + check_ocrmypdf(resources / 'francais.pdf', outpdf, env=spoof_no_tess_pdfa_warning) From e2a563cc761a6e2201b202f1a7b62d40fe12fa2d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Jan 2020 16:47:15 -0800 Subject: [PATCH 309/880] logging: create a debug log when -k parameter is issued --- src/ocrmypdf/_sync.py | 15 +++++++++++++++ src/ocrmypdf/api.py | 9 ++++++--- 2 files changed, 21 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 12f85738..119aaba2 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -23,6 +23,7 @@ import signal import sys import threading from collections import namedtuple +from pathlib import Path from tempfile import mkdtemp import PIL @@ -335,6 +336,17 @@ def samefile(f1, f2): return os.path.samefile(f1, f2) +def configure_debug_logging(log_filename, prefix=''): + log_file_handler = logging.FileHandler(log_filename, delay=True) + log_file_handler.setLevel(logging.DEBUG) + formatter = logging.Formatter( + '[%(asctime)s] - %(name)s - %(levelname)7s - %(message)s' + ) + log_file_handler.setFormatter(formatter) + logging.getLogger(prefix).addHandler(log_file_handler) + return + + def run_pipeline(options, api=False): log = make_logger(options, __name__) @@ -345,6 +357,9 @@ def run_pipeline(options, api=False): options.jobs = available_cpu_count() work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + if options.keep_temporary_files: + configure_debug_logging(Path(work_folder) / "debug.log") + try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 3270b25d..bad68732 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -89,11 +89,14 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= overwrite the progress bar manage_root_logger (bool): Configure the process's root logger, to ensure all log output is sent through + + Returns: + The toplevel logger for ocrmypdf (or the root logger, if we are managing it). """ prefix = '' if manage_root_logger else 'ocrmypdf' log = logging.getLogger(prefix) - log.setLevel(logging.INFO) + log.setLevel(logging.DEBUG) if progress_bar_friendly: console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) @@ -108,8 +111,6 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= console.setLevel(logging.INFO) formatter = logging.Formatter('%(levelname)7s - %(message)s') - if verbosity >= 1: - log.setLevel(logging.DEBUG) if verbosity >= 2: formatter = logging.Formatter('%(name)s - %(levelname)7s - %(message)s') @@ -125,6 +126,8 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= if manage_root_logger: logging.captureWarnings(True) + return log + def create_options(*, input_file, output_file, **kwargs): cmdline = [] From a4dc5e365f3741f318cea6e192c5a40929c693df Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Jan 2020 16:47:36 -0800 Subject: [PATCH 310/880] logging: fix incorrect usage: logging.Logger() --- src/ocrmypdf/exec/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 2eb6cd88..4912fe08 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -29,7 +29,7 @@ from subprocess import run as subprocess_run from ..exceptions import ExitCode, MissingDependencyError -log = logging.Logger(__name__) +log = logging.getLogger(__name__) def _get_program(args, env=None): From 6faa8f72215ff8d2ad49eed078b3b835e9156642 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Jan 2020 16:48:48 -0800 Subject: [PATCH 311/880] logging: always log process arguments and stderr when at debug Also remove ad-hoc logging of this information. --- src/ocrmypdf/exec/__init__.py | 13 +++++++++++-- src/ocrmypdf/exec/ghostscript.py | 3 --- src/ocrmypdf/exec/tesseract.py | 2 -- 3 files changed, 11 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 4912fe08..838697fc 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -77,12 +77,21 @@ def run(args, *, env=None, **kwargs): if new_args0: args[0] = new_args0 - log.debug(args) + process_log = log.getChild(os.path.basename(program)) + process_log.debug("Running: %s", args) if sys.version_info < (3, 7) and os.name == 'nt': # Can't use close_fds=True on Windows with Python 3.6 or older # https://bugs.python.org/issue19575, etc. kwargs['close_fds'] = False - return subprocess_run(args, env=env, **kwargs) + proc = subprocess_run(args, env=env, **kwargs) + if process_log.isEnabledFor(logging.DEBUG): + try: + stderr = proc.stderr.decode('utf-8', 'replace') + except AttributeError: + stderr = proc.stderr + if stderr: + process_log.debug("stderr = %s", stderr) + return proc def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index a72572f6..856bc0c1 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -311,7 +311,6 @@ def generate_pdfa( ] ) args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs - log.debug(args_gs) try: with Path(output_file).open('wb') as output: p = run(args_gs, stdout=output, stderr=PIPE, check=True) @@ -342,5 +341,3 @@ def generate_pdfa( "Ghostscript had to remove PDF 'overprinting' from the " "input file to complete PDF/A conversion. " ) - else: - log.debug(stderr) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 16edb7cb..6e020ad6 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -282,7 +282,6 @@ def generate_hocr( # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) try: - log.debug(args_tesseract) p = run( args_tesseract, stdout=PIPE, @@ -381,7 +380,6 @@ def generate_pdf( args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig) try: - log.debug(args_tesseract) p = run( args_tesseract, stdout=PIPE, From 599028bebbafdaf1c3b320fe93a806d9bcb9dc62 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 4 Jan 2020 01:17:33 -0800 Subject: [PATCH 312/880] tesseract: don't explicitly set lstm_use_matrix Apparently tesseract does this own its own as needed. --- src/ocrmypdf/exec/tesseract.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 6e020ad6..a70eed36 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -275,9 +275,6 @@ def generate_hocr( if user_patterns: args_tesseract.extend(['--user-patterns', user_patterns]) - if user_words or user_patterns: - args_tesseract.extend(['-c', 'lstm_use_matrix=1']) - # Reminder: test suite tesseract spoofers will break after any changes # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) From 32041c43e1f1751eb92b69a0757e18e38bbbf24d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 4 Jan 2020 02:35:14 -0800 Subject: [PATCH 313/880] tests: improve tesseract coverage --- src/ocrmypdf/exec/tesseract.py | 4 +- tests/test_tess4.py | 93 ++++++++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index a70eed36..abcbb2fa 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -260,8 +260,8 @@ def generate_hocr( log, ): - output_hocr = next(o for o in output_files if o.endswith('.hocr')) - output_sidecar = next(o for o in output_files if o.endswith('.txt')) + output_hocr = next(o for o in output_files if fspath(o).endswith('.hocr')) + output_sidecar = next(o for o in output_files if fspath(o).endswith('.txt')) prefix = os.path.splitext(output_hocr)[0] args_tesseract = tess_base_args(language, engine_mode) diff --git a/tests/test_tess4.py b/tests/test_tess4.py index 0b835ea7..f2b81636 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -15,7 +15,9 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import os +import subprocess from contextlib import contextmanager from os import fspath from pathlib import Path @@ -81,3 +83,94 @@ def test_no_languages(tmp_path): with pytest.raises(MissingDependencyError): tesseract.languages(tesseract_env=env) + + +def test_image_too_large_hocr(monkeypatch, resources, outdir): + log = logging.getLogger('test_image_too_large_hocr') + + def dummy_run(args, *, env=None, **kwargs): + raise subprocess.CalledProcessError(1, 'tesseract', output=b'Image too large') + + monkeypatch.setattr(tesseract, 'run', dummy_run) + tesseract.generate_hocr( + input_file=resources / 'crom.png', + output_files=[outdir / 'out.hocr', outdir / 'out.txt'], + language=['eng'], + engine_mode=None, + tessconfig=[], + timeout=180.0, + pagesegmode=None, + log=log, + user_words=None, + user_patterns=None, + tesseract_env=None, + ) + assert "name='ocr-capabilities'" in Path(outdir / 'out.hocr').read_text() + + +def test_image_too_large_pdf(monkeypatch, resources, outdir): + log = logging.getLogger('test_image_too_large_pdf') + + def dummy_run(args, *, env=None, **kwargs): + raise subprocess.CalledProcessError(1, 'tesseract', output=b'Image too large') + + monkeypatch.setattr(tesseract, 'run', dummy_run) + tesseract.generate_pdf( + input_image=resources / 'crom.png', + skip_pdf=resources / 'blank.pdf', + output_pdf=outdir / 'pdf.pdf', + output_text=outdir / 'txt.txt', + language=['eng'], + engine_mode=None, + text_only=False, + tessconfig=[], + timeout=180.0, + pagesegmode=None, + log=log, + user_words=None, + user_patterns=None, + tesseract_env=None, + ) + assert Path(outdir / 'txt.txt').read_text() == '[skipped page]' + assert Path(outdir / 'pdf.pdf').samefile(resources / 'blank.pdf') + + +def test_timeout(caplog): + log = logging.getLogger('test_timeout') + tesseract.page_timedout(log, '123456.png', 5) + assert "123456" in caplog.text + assert "took too long" in caplog.text + + +@pytest.mark.parametrize( + 'in_, logged', + [ + (b'Tesseract Open Source', ''), + (b'lots of diacritics blah blah', 'diacritics'), + (b'Warning in pixReadMem', ''), + (b'OSD: Weak margin', 'unsure about page orientation'), + (b'Error in pixScanForForeground', ''), + (b'Error in boxClipToRectangle', ''), + (b'an unexpected error', 'an unexpected error'), + (b'a dire warning', 'a dire warning'), + (b'read_params_file something', 'read_params_file'), + (b'an innocent message', 'innocent'), + (b'\x7f\x7f\x80innocent unicode failure', 'innocent'), + ], +) +def test_tesseract_log_output(caplog, in_, logged): + log = logging.getLogger('tesseract_log_output') + log.setLevel(logging.INFO) + + tesseract.tesseract_log_output(log, in_, 'dummy') + if logged == '': + assert caplog.text == '' + else: + assert logged in caplog.text + + +def test_tesseract_log_output_raises(caplog): + log = logging.getLogger('tesseract_log_output') + with pytest.raises(tesseract.TesseractConfigError): + tesseract.tesseract_log_output(log, b'parameter not found: moo', 'dummy') + assert 'not found' in caplog.text From 9c5f0d0ec60bde8f5782d64b9aa74208658e9605 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 4 Jan 2020 16:32:01 -0800 Subject: [PATCH 314/880] Eliminate last use of PyPDF2 from test suite --- requirements/test.txt | 1 - setup.cfg | 2 +- tests/spoof/tesseract_noop.py | 13 ++++++------- 3 files changed, 7 insertions(+), 9 deletions(-) diff --git a/requirements/test.txt b/requirements/test.txt index 58fd0fdd..b2f10302 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -4,5 +4,4 @@ pytest-xdist >= 1.29.0 # For DumpError fix pytest-cov >= 2.6.1 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi -PyPDF2 >= 1.26.0 #PyMuPDF == 1.13.4 # optional diff --git a/setup.cfg b/setup.cfg index 28823133..f307a2e5 100644 --- a/setup.cfg +++ b/setup.cfg @@ -23,7 +23,7 @@ force_grid_wrap=0 use_parentheses=True line_length=88 known_first_party = ocrmypdf -known_third_party = PIL,PyPDF2,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug +known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug [metadata] license_file = LICENSE diff --git a/tests/spoof/tesseract_noop.py b/tests/spoof/tesseract_noop.py index 857c3e2a..30f97209 100755 --- a/tests/spoof/tesseract_noop.py +++ b/tests/spoof/tesseract_noop.py @@ -32,9 +32,10 @@ In orientation check mode, report the orientation is upright. """ import sys +from pathlib import Path import img2pdf -import PyPDF2 as pypdf +import pikepdf from PIL import Image VERSION_STRING = '''tesseract 4.0.0 @@ -99,12 +100,10 @@ def main(): pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1] ptsize = pagesize[0] * 72, pagesize[1] * 72 - pdf_out = pypdf.PdfFileWriter() - pdf_out.addBlankPage(ptsize[0], ptsize[1]) - with open(output + '.pdf', 'wb') as f: - pdf_out.write(f) - with open(output + '.txt', 'w') as f: - f.write('') + pdf_out = pikepdf.new() + pdf_out.add_blank_page(page_size=ptsize) + pdf_out.save(Path(output).with_suffix('.pdf'), static_id=True) + Path(output).with_suffix('.txt').write_text('') else: inputf = sys.argv[-4] output = sys.argv[-3] From 8f984bf9589025c5651fc9f3cc771de12418c727 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 4 Jan 2020 16:32:47 -0800 Subject: [PATCH 315/880] docs: add note on limitations of sidecar file --- docs/cookbook.rst | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 4ec4ff4d..7b5be9d9 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -89,6 +89,18 @@ This produces a file named "output.pdf" and a companion text file named ocrmypdf --sidecar output.txt input.pdf output.pdf +.. note:: + + The sidecar file contains the **OCR text** found by OCRmyPDF. If the document + contains pages that already have text, that text will not appear in the + sidecar. If the option ``--pages`` is used, only those pages on which OCR + was performed will be included in the sidecar. If certain pages were skipped + because of options like ``--skip-big`` or ``--tesseract-timeout``, those pages + will not be in the sidecar. + + To extract all text from a PDF, whether generated from OCR or otherwise, + use a program like Poppler's ``pdftotext``. + OCR images, not PDFs -------------------- From 5b6ab1e003234f0dd8cb8552f319c5850ecd2d80 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Jan 2020 01:05:36 -0800 Subject: [PATCH 316/880] lept: improve lib not found error message Closes #471 --- src/ocrmypdf/leptonica.py | 25 ++++++++++++++++++------- 1 file changed, 18 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 7ff3a265..328b0630 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -47,21 +47,32 @@ if os.name == 'nt': else: libname = 'lept' _libpath = find_library(libname) -if not _libpath and os.name == 'nt': +if not _libpath: raise MissingDependencyError( """ --------------------------------------------------------------------- - This error normally occurs when ocrmypdf can't find a file named - liblept-5.dll (Leptonica). Please ensure Tesseract-OCR is installed - and its location is added to the system PATH environment variable. + This error normally occurs when ocrmypdf can't find the Leptonica + library, which is usually installed with Tesseract OCR. It could be that + Tesseract is not installed properly, we can't find the installation + on your system PATH environment variable. - For details see: + The library we are looking for is usually called: + liblept-5.dll (Windows) + liblept*.dylib (macOS) + liblept*.so (Linux/BSD) + + Please review our installation procedures to find a solution: https://ocrmypdf.readthedocs.io/en/latest/installation.html --------------------------------------------------------------------- """ ) -lept = ffi.dlopen(_libpath) -lept.setMsgSeverity(lept.L_SEVERITY_WARNING) +try: + lept = ffi.dlopen(_libpath) + lept.setMsgSeverity(lept.L_SEVERITY_WARNING) +except ffi.error as e: + raise MissingDependencyError( + f"Leptonica library found at {_libpath}, but we could not access it" + ) from e class _LeptonicaErrorTrap: From 5169ac633b49719a6d03d14cba7e06f9ce71c84b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Jan 2020 21:32:36 -0800 Subject: [PATCH 317/880] docs: mention pdfgrep too --- docs/cookbook.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 7b5be9d9..2ffe3be4 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -99,7 +99,7 @@ This produces a file named "output.pdf" and a companion text file named will not be in the sidecar. To extract all text from a PDF, whether generated from OCR or otherwise, - use a program like Poppler's ``pdftotext``. + use a program like Poppler's ``pdftotext`` or ``pdfgrep``. OCR images, not PDFs -------------------- From 6f5d77d930933ee58b9e9fc1347dcb3239ff3981 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Jan 2020 21:33:32 -0800 Subject: [PATCH 318/880] Also generate log file in temp folder on verbose mode --- src/ocrmypdf/_sync.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 119aaba2..9aa99edd 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -357,7 +357,7 @@ def run_pipeline(options, api=False): options.jobs = available_cpu_count() work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - if options.keep_temporary_files: + if options.keep_temporary_files or options.verbose >= 1: configure_debug_logging(Path(work_folder) / "debug.log") try: From fd991a2380f1803924b1b8192e42e67a80998dde Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Jan 2020 21:35:52 -0800 Subject: [PATCH 319/880] Allow pdfminer.six 20200104 and update recommended versions --- docs/release_notes.rst | 12 ++++++++++++ requirements/main.txt | 8 ++++---- requirements/test.txt | 4 ++-- setup.py | 5 +++-- 4 files changed, 21 insertions(+), 8 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 8268a0e8..47fc1ac7 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,18 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.4.0 +====== + +- Updated recommended dependency versions. +- Improvements to test coverage and changes to facilitate better measurement of + test coverage, such as when tests run in subprocesses. +- Improvements to error messages when Leptonica is not installed correctly. +- Fixed use of pytest "session scope" that may have caused some intermittent + CI failures. +- When the argument ``--keep-temporary-files`` or verbosity is set to ``-v1``, + a debug log file is generated in the working temporary folder. + v9.3.0 ====== diff --git a/requirements/main.txt b/requirements/main.txt index d6c25b7b..5ac09a35 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,8 +3,8 @@ # installation cffi == 1.13.2 img2pdf == 0.3.3 -pdfminer.six == 20191110 -pikepdf == 1.8.1 -Pillow >= 6.2.0 +pdfminer.six == 20200104 +pikepdf == 1.8.2 +Pillow == 7.0.0 reportlab == 3.5.32 -tqdm == 4.37.0 +tqdm == 4.41.1 diff --git a/requirements/test.txt b/requirements/test.txt index b2f10302..aeda7a7c 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,7 +1,7 @@ pytest >= 5.0.0 pytest-helpers-namespace >= 2019.1.8 -pytest-xdist >= 1.29.0 # For DumpError fix -pytest-cov >= 2.6.1 +pytest-xdist >= 1.31.0 +pytest-cov >= 2.8.0 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi #PyMuPDF == 1.13.4 # optional diff --git a/setup.py b/setup.py index a79337c6..50ce881c 100644 --- a/setup.py +++ b/setup.py @@ -21,11 +21,12 @@ from __future__ import print_function, unicode_literals import sys +from setuptools import find_packages, setup + if sys.version_info < (3, 6): print("Python 3.6 or newer is required", file=sys.stderr) sys.exit(1) -from setuptools import setup, find_packages # pylint: disable=w0613 @@ -97,7 +98,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20191110', + 'pdfminer.six >= 20181108, <= 20200104', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 123fde174d4a51833475a5639bf5f5b34a419b79 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 6 Jan 2020 01:46:19 -0800 Subject: [PATCH 320/880] Don't use debug.log in pytest pytest does not reset the state of logging if we install a file handler, which will cause FileNotFoundError after the temporary folder is removed. Semi-related: https://github.com/pytest-dev/pytest/issues/5502 --- src/ocrmypdf/_sync.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 9aa99edd..960e3220 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -344,7 +344,7 @@ def configure_debug_logging(log_filename, prefix=''): ) log_file_handler.setFormatter(formatter) logging.getLogger(prefix).addHandler(log_file_handler) - return + return log_file_handler def run_pipeline(options, api=False): @@ -357,7 +357,9 @@ def run_pipeline(options, api=False): options.jobs = available_cpu_count() work_folder = mkdtemp(prefix="com.github.ocrmypdf.") - if options.keep_temporary_files or options.verbose >= 1: + if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get( + 'PYTEST_CURRENT_TEST', '' + ): configure_debug_logging(Path(work_folder) / "debug.log") try: From 9ad8cbf1f65836e63aadbafaf7236b7740f1a422 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 6 Jan 2020 02:02:05 -0800 Subject: [PATCH 321/880] Fix assert that depends on POSIX-y file handling --- tests/test_tess4.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_tess4.py b/tests/test_tess4.py index f2b81636..a66f2186 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -132,7 +132,8 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): tesseract_env=None, ) assert Path(outdir / 'txt.txt').read_text() == '[skipped page]' - assert Path(outdir / 'pdf.pdf').samefile(resources / 'blank.pdf') + if os.name != 'nt': # different semantics + assert Path(outdir / 'pdf.pdf').samefile(resources / 'blank.pdf') def test_timeout(caplog): From 61a26743175b1579ccdb802043ecadecb5ba4b15 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 6 Jan 2020 02:35:54 -0800 Subject: [PATCH 322/880] Skip test that needs chmod when on Windows --- tests/test_main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_main.py b/tests/test_main.py index cd12bb5d..39abbe98 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -270,6 +270,7 @@ def test_input_file_not_found(caplog, no_outpdf): assert input_file in caplog.text +@pytest.mark.skipif(os.name == 'nt', reason="chmod") def test_input_file_not_readable(caplog, resources, outdir, no_outpdf): input_file = outdir / 'trivial.pdf' shutil.copy(resources / 'trivial.pdf', input_file) From 3831c4cd4dde9273e487db3598ca9b6223746d1c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Jan 2020 01:10:15 -0800 Subject: [PATCH 323/880] Refactor metadata_fixup --- src/ocrmypdf/_pipeline.py | 86 +++++++++++++++++++++------------------ 1 file changed, 46 insertions(+), 40 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index fd1aadb4..cb78af6f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -727,47 +727,53 @@ def should_linearize(working_file, context): def metadata_fixup(working_file, context): output_file = context.get_path('metafix.pdf') options = context.options - original = pikepdf.open(context.origin) - docinfo = get_docinfo(original, options) - pdf = pikepdf.open(working_file) - with pdf.open_metadata() as meta: - meta.load_from_docinfo(docinfo, delete_missing=False) - # If xmp:CreateDate is missing, set it to the modify date to - # match Ghostscript, for consistency - if 'xmp:CreateDate' not in meta: - meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') - meta_original = original.open_metadata() - not_copied = set(meta_original.keys()) - set(meta.keys()) - if not_copied: - if options.output_type.startswith('pdfa'): - context.log.warning( - "Some input metadata could not be copied because it is not " - "permitted in PDF/A. You may wish to examine the output " - "PDF's XMP metadata." - ) - context.log.debug( - "The following metadata fields were not copied: %r", not_copied - ) - else: - context.log.error( - "Some input metadata could not be copied." - "You may wish to examine the output PDF's XMP metadata." - ) - context.log.info( - "The following metadata fields were not copied: %r", not_copied - ) - pdf.save( - output_file, - compress_streams=True, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, - linearize=( # Don't linearize if optimize() will be linearizing too - should_linearize(working_file, context) if options.optimize == 0 else False - ), - ) - original.close() - pdf.close() + def report_on_metadata(missing): + if not missing: + return + if options.output_type.startswith('pdfa'): + context.log.warning( + "Some input metadata could not be copied because it is not " + "permitted in PDF/A. You may wish to examine the output " + "PDF's XMP metadata." + ) + context.log.debug( + "The following metadata fields were not copied: %r", missing + ) + else: + context.log.error( + "Some input metadata could not be copied." + "You may wish to examine the output PDF's XMP metadata." + ) + context.log.info( + "The following metadata fields were not copied: %r", missing + ) + + with pikepdf.open(context.origin) as original: + docinfo = get_docinfo(original, options) + with pikepdf.open(working_file) as pdf, pdf.open_metadata() as meta: + meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False) + # If xmp:CreateDate is missing, set it to the modify date to + # match Ghostscript, for consistency + if 'xmp:CreateDate' not in meta: + meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') + + meta_original = original.open_metadata() + missing = set(meta_original.keys()) - set(meta.keys()) + report_on_metadata(missing) + + pdf.save( + output_file, + compress_streams=True, + preserve_pdfa=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + linearize=( # Don't linearize if optimize() will be linearizing too + should_linearize(working_file, context) + if options.optimize == 0 + else False + ), + ) + return output_file From ce97af5a7971136a17d1972a920139ee50520993 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 Jan 2020 03:10:27 -0800 Subject: [PATCH 324/880] Add OCR quality measurement API --- src/ocrmypdf/__init__.py | 1 + src/ocrmypdf/_sync.py | 1 + src/ocrmypdf/api.py | 3 +- src/ocrmypdf/quality.py | 60 ++++++++++++++++++++++++++++++++++++++++ tests/test_quality.py | 35 +++++++++++++++++++++++ 5 files changed, 98 insertions(+), 2 deletions(-) create mode 100644 src/ocrmypdf/quality.py create mode 100644 tests/test_quality.py diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 8d779a95..2f76bf4f 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -23,6 +23,7 @@ from .exceptions import ( DpiError, EncryptedPdfError, ExitCode, + ExitCodeException, InputFileError, MissingDependencyError, OutputFileAccessError, diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 960e3220..493664a0 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -381,6 +381,7 @@ def run_pipeline(options, api=False): detailed_page_analysis=options.redo_ocr, progbar=options.progress_bar, ) + context = PDFContext(options, work_folder, origin_pdf, pdfinfo) # Validate options are okay for this pdf diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index bad68732..e48320b5 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -18,11 +18,10 @@ import logging import os import sys -import warnings from contextlib import suppress from enum import IntEnum from pathlib import Path -from typing import Dict, List, Optional +from typing import Dict, List from tqdm import tqdm diff --git a/src/ocrmypdf/quality.py b/src/ocrmypdf/quality.py new file mode 100644 index 00000000..293173bf --- /dev/null +++ b/src/ocrmypdf/quality.py @@ -0,0 +1,60 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import re +from typing import Iterable + +"""Utilities to measure OCR quality""" + + +class OcrQualityDictionary: + """Manages a dictionary for simple OCR quality checks.""" + + def __init__(self, *, wordlist: Iterable[str] = []): + """Construct a dictionary from a list of words. + + Words for which capitalization is important should be capitalized in the + dictionary. Words that contain spaces or other punctuation will never match. + """ + self.dictionary = set() + self.dictionary.update(w for w in wordlist) + + def measure_words_matched(self, ocr_text: str) -> float: + """Check how many unique words in the OCR text match a dictionary. + + Words with mixed capitalized are only considered a match if the test word + matches that capitalization. + + Returns: + number of words that match / number + """ + text = re.sub(r"[0-9_]+", ' ', ocr_text) + text = re.sub(r'\W+', ' ', text) + text_words_list = re.split(r'\s+', text) + text_words = {w for w in text_words_list if len(w) >= 3} + + matches = 0 + for w in text_words: + if w in self.dictionary or ( + w != w.lower() and w.lower() in self.dictionary + ): + matches += 1 + if matches > 0: + hit_ratio = matches / len(text_words) + else: + hit_ratio = 0.0 + return hit_ratio diff --git a/tests/test_quality.py b/tests/test_quality.py new file mode 100644 index 00000000..99f15653 --- /dev/null +++ b/tests/test_quality.py @@ -0,0 +1,35 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import pytest + +import ocrmypdf.quality as qual + + +def test_quality_measurement(): + oqd = qual.OcrQualityDictionary( + wordlist=["words", "words", "quick", "brown", "fox", "dog", "lazy"] + ) + assert len(oqd.dictionary) == 6 # 6 unique + + assert ( + oqd.measure_words_matched("The quick brown fox jumps quickly over the lazy dog") + == 0.5 + ) + assert oqd.measure_words_matched("12345 10% _f 7fox -brown | words") == 1.0 + + assert oqd.measure_words_matched("quick quick quick") == 1.0 From 2e15d52895f9faea3df6365cc7a8ba8a1f3ae0a4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 Jan 2020 03:11:33 -0800 Subject: [PATCH 325/880] v9.5.0 release notes --- docs/release_notes.rst | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 47fc1ac7..a47a7c14 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,11 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.5.0 +====== + +- Added API functions to measure OCR quality. + v9.4.0 ====== From e860c56b7517cac6e0c6ef0cad6e6961ccc76f5b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 17 Jan 2020 23:01:37 -0800 Subject: [PATCH 326/880] Fix regression: metadata updates not taking effect --- src/ocrmypdf/_pipeline.py | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index cb78af6f..e3b0a21d 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -749,9 +749,9 @@ def metadata_fixup(working_file, context): "The following metadata fields were not copied: %r", missing ) - with pikepdf.open(context.origin) as original: + with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf: docinfo = get_docinfo(original, options) - with pikepdf.open(working_file) as pdf, pdf.open_metadata() as meta: + with pdf.open_metadata() as meta: meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False) # If xmp:CreateDate is missing, set it to the modify date to # match Ghostscript, for consistency @@ -762,17 +762,17 @@ def metadata_fixup(working_file, context): missing = set(meta_original.keys()) - set(meta.keys()) report_on_metadata(missing) - pdf.save( - output_file, - compress_streams=True, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, - linearize=( # Don't linearize if optimize() will be linearizing too - should_linearize(working_file, context) - if options.optimize == 0 - else False - ), - ) + pdf.save( + output_file, + compress_streams=True, + preserve_pdfa=True, + object_stream_mode=pikepdf.ObjectStreamMode.generate, + linearize=( # Don't linearize if optimize() will be linearizing too + should_linearize(working_file, context) + if options.optimize == 0 + else False + ), + ) return output_file From a6567f2ae4602da734533c8ed5823c16668ba7a5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 18 Jan 2020 01:48:33 -0800 Subject: [PATCH 327/880] v9.5.0 release notes revised --- docs/release_notes.rst | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index a47a7c14..0c107aef 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -17,6 +17,7 @@ v9.5.0 ====== - Added API functions to measure OCR quality. +- Modest improvements to handling PDFs with difficult/non compliant metadata. v9.4.0 ====== From b7f38e976b5a85d9936d3b1c40c46a1663eb0de0 Mon Sep 17 00:00:00 2001 From: Ian Alexander <1693187+ianalexander@users.noreply.github.com> Date: Sun, 19 Jan 2020 19:11:54 -0800 Subject: [PATCH 328/880] Watched folder bug fixes, new flags, and docs updates. --- docs/batch.rst | 6 ++++++ misc/watcher.py | 27 +++++++++++++++++++++++---- 2 files changed, 29 insertions(+), 4 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index 9e4fe749..e9c96fa7 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -210,6 +210,9 @@ be launched as follows: -v :/input \ -v :/output \ -e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ + -e OCR_ON_SUCCESS_DELETE=1 \ + -e OCR_DESKEW=1 \ + -e PYTHONUNBUFFERED=1 \ -it --entrypoint python3 \ jbarlow83/ocrmypdf \ watcher.py @@ -224,6 +227,9 @@ convert it to a OCRed PDF in ``/output/``. The parameters to this image are: "``-v :/input``", "Files placed in this location will be OCRed" "``-v :/output``", "This is where OCRed files will be stored" "``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "This will place files in the output in {output}/{year}/{month}/{filename}" + "``-e OCR_ON_SUCCESS_DELETE=1``", "This will delete the input file if the exit code is 0 (OK)" + "``-e OCR_DESKEW=1``", "This will enable deskew for crooked PDFs" + "``-e PYTHONBUFFERED=1``", "This will force STDOUT to be unbuffered and allow you to see messages in docker logs" This service relies on polling to check for changes to the filesystem. It may not be suitable for some environments, such as filesystems shared on a diff --git a/misc/watcher.py b/misc/watcher.py index 99253001..86059cd4 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -25,12 +25,15 @@ import ocrmypdf INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') +ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) +DESKEW = bool(os.getenv('OCR_DESKEW', False)) OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) PATTERNS = ['*.pdf'] def execute_ocrmypdf(file_path): - filename = Path(file_path).name + new_file = Path(file_path) + filename = new_file.name if OUTPUT_DIRECTORY_YEAR_MONTH: today = datetime.today() output_directory_year_month = Path( @@ -41,13 +44,29 @@ def execute_ocrmypdf(file_path): output_path = Path(output_directory_year_month) / filename else: output_path = Path(OUTPUT_DIRECTORY) / filename - print(f'New file: {file_path}.\nAttempting to OCRmyPDF to: {output_path}') - ocrmypdf.ocr(file_path, output_path) + print(f'New file: {file_path}. Waiting until fully loaded...') + # This loop waits to make sure that the file is completely loaded on + # disk before attempting to read. Docker sometimes will publish the + # watchdog event before the file is actually fully on disk, causing + # pikepdf to fail. + current_size = None + while current_size != new_file.stat().st_size: + current_size = new_file.stat().st_size + time.sleep(1) + print(f'Attempting to OCRmyPDF to: {output_path}') + exit_code = ocrmypdf.ocr( + input_file=file_path, output_file=output_path, deskew=DESKEW + ) + if exit_code == 0 and ON_SUCCESS_DELETE: + print(f'Done. Deleting: {file_path}') + new_file.unlink() + else: + print('Done') class HandleObserverEvent(PatternMatchingEventHandler): def on_any_event(self, event): - if event.event_type in ['created', 'modified']: + if event.event_type in ['created']: execute_ocrmypdf(event.src_path) From 3eab161771f63aac04891f4e7a0f4d47dae96afa Mon Sep 17 00:00:00 2001 From: Ian Alexander <1693187+ianalexander@users.noreply.github.com> Date: Mon, 20 Jan 2020 10:45:28 -0800 Subject: [PATCH 329/880] Update logging and env var extensibility --- misc/watcher.py | 29 ++++++++++++++++++++++------- 1 file changed, 22 insertions(+), 7 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index 86059cd4..114365de 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -15,6 +15,7 @@ import os import time +import logging from datetime import datetime from pathlib import Path @@ -25,11 +26,15 @@ import ocrmypdf INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') +OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) DESKEW = bool(os.getenv('OCR_DESKEW', False)) -OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) +POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) +LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] +logging.basicConfig(level=LOGLEVEL) +logger = logging.getLogger('ocrmypdf-watcher') def execute_ocrmypdf(file_path): new_file = Path(file_path) @@ -44,7 +49,7 @@ def execute_ocrmypdf(file_path): output_path = Path(output_directory_year_month) / filename else: output_path = Path(OUTPUT_DIRECTORY) / filename - print(f'New file: {file_path}. Waiting until fully loaded...') + logger.info(f'New file: {file_path}. Waiting until fully loaded...') # This loop waits to make sure that the file is completely loaded on # disk before attempting to read. Docker sometimes will publish the # watchdog event before the file is actually fully on disk, causing @@ -52,16 +57,17 @@ def execute_ocrmypdf(file_path): current_size = None while current_size != new_file.stat().st_size: current_size = new_file.stat().st_size - time.sleep(1) - print(f'Attempting to OCRmyPDF to: {output_path}') + logger.debug(f'new_file current_size: {current_size}') + time.sleep(POLL_NEW_FILE_SECONDS) + logger.info(f'Attempting to OCRmyPDF to: {output_path}') exit_code = ocrmypdf.ocr( input_file=file_path, output_file=output_path, deskew=DESKEW ) if exit_code == 0 and ON_SUCCESS_DELETE: - print(f'Done. Deleting: {file_path}') + logger.info(f'Done. Deleting: {file_path}') new_file.unlink() else: - print('Done') + logger.info('Done') class HandleObserverEvent(PatternMatchingEventHandler): @@ -71,12 +77,21 @@ class HandleObserverEvent(PatternMatchingEventHandler): if __name__ == "__main__": - print( + logger.info( f"Starting OCRmyPDF watcher with config:\n" f"Input Directory: {INPUT_DIRECTORY}\n" f"Output Directory: {OUTPUT_DIRECTORY}\n" f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}" ) + logger.debug( + f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n" + f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n" + f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" + f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n" + f"DESKEW: {DESKEW}\n" + f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" + f"LOGLEVEL: {LOGLEVEL}\n" + ) handler = HandleObserverEvent(patterns=PATTERNS) observer = Observer() observer.schedule(handler, INPUT_DIRECTORY, recursive=True) From bcf77375c015060c42f0c3108b798f188ec22ca6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 Jan 2020 07:33:28 -0800 Subject: [PATCH 330/880] Fix grammar in output message --- src/ocrmypdf/_sync.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 493664a0..a4c18b21 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -231,7 +231,7 @@ def exec_concurrent(context): # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) if max_workers > 1: - context.log.info("Start processing %d pages concurrent", max_workers) + context.log.info("Start processing %d pages concurrently", max_workers) # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want # to manage how many threads it uses to avoid creating total threads than cores. From 4952af16047bef36b38c115c2b5f563b7c626b9d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 Jan 2020 12:56:19 -0800 Subject: [PATCH 331/880] watcher: some refactoring --- misc/watcher.py | 55 +++++++++++++++++++++++++++++++------------------ 1 file changed, 35 insertions(+), 20 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index 114365de..d06108e4 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -13,9 +13,9 @@ # You should have received a copy of the GNU General Public License # along with this program. If not, see . +import logging import os import time -import logging from datetime import datetime from pathlib import Path @@ -33,41 +33,52 @@ POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] -logging.basicConfig(level=LOGLEVEL) -logger = logging.getLogger('ocrmypdf-watcher') +log = logging.getLogger('ocrmypdf-watcher') -def execute_ocrmypdf(file_path): - new_file = Path(file_path) - filename = new_file.name + +def get_output_dir(root, basename): if OUTPUT_DIRECTORY_YEAR_MONTH: today = datetime.today() - output_directory_year_month = Path( - f'{OUTPUT_DIRECTORY}/{today.year}/{today.month}' + output_directory_year_month = ( + Path(root) / str(today.year) / f'{today.month:02d}' ) if not output_directory_year_month.exists(): output_directory_year_month.mkdir(parents=True, exist_ok=True) - output_path = Path(output_directory_year_month) / filename + output_path = Path(output_directory_year_month) / basename else: - output_path = Path(OUTPUT_DIRECTORY) / filename - logger.info(f'New file: {file_path}. Waiting until fully loaded...') + output_path = Path(OUTPUT_DIRECTORY) / basename + return output_path + + +def wait_for_file_ready(file_path): # This loop waits to make sure that the file is completely loaded on # disk before attempting to read. Docker sometimes will publish the # watchdog event before the file is actually fully on disk, causing # pikepdf to fail. + current_size = None - while current_size != new_file.stat().st_size: - current_size = new_file.stat().st_size - logger.debug(f'new_file current_size: {current_size}') + while current_size != file_path.stat().st_size: + current_size = file_path.stat().st_size + log.debug(f'file_path current_size: {current_size}') time.sleep(POLL_NEW_FILE_SECONDS) - logger.info(f'Attempting to OCRmyPDF to: {output_path}') + + +def execute_ocrmypdf(file_path): + file_path = Path(file_path) + output_path = get_output_dir(OUTPUT_DIRECTORY, file_path.name) + + log.info("-" * 20) + log.info(f'New file: {file_path}. Waiting until fully loaded...') + log.info(f'Attempting to OCRmyPDF to: {output_path}') + wait_for_file_ready(file_path) exit_code = ocrmypdf.ocr( input_file=file_path, output_file=output_path, deskew=DESKEW ) if exit_code == 0 and ON_SUCCESS_DELETE: - logger.info(f'Done. Deleting: {file_path}') - new_file.unlink() + log.info(f'OCR is done. Deleting: {file_path}') + file_path.unlink() else: - logger.info('Done') + log.info('OCR is done') class HandleObserverEvent(PatternMatchingEventHandler): @@ -77,13 +88,16 @@ class HandleObserverEvent(PatternMatchingEventHandler): if __name__ == "__main__": - logger.info( + ocrmypdf.configure_logging( + verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True + ) + log.info( f"Starting OCRmyPDF watcher with config:\n" f"Input Directory: {INPUT_DIRECTORY}\n" f"Output Directory: {OUTPUT_DIRECTORY}\n" f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}" ) - logger.debug( + log.debug( f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n" f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n" f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" @@ -92,6 +106,7 @@ if __name__ == "__main__": f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"LOGLEVEL: {LOGLEVEL}\n" ) + handler = HandleObserverEvent(patterns=PATTERNS) observer = Observer() observer.schedule(handler, INPUT_DIRECTORY, recursive=True) From 82f393dd096a5895c23cb2b43b9de2b23607d84d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 Jan 2020 12:40:19 -0800 Subject: [PATCH 332/880] Order of events --- misc/watcher.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/watcher.py b/misc/watcher.py index d06108e4..427a23bc 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -69,8 +69,8 @@ def execute_ocrmypdf(file_path): log.info("-" * 20) log.info(f'New file: {file_path}. Waiting until fully loaded...') - log.info(f'Attempting to OCRmyPDF to: {output_path}') wait_for_file_ready(file_path) + log.info(f'Attempting to OCRmyPDF to: {output_path}') exit_code = ocrmypdf.ocr( input_file=file_path, output_file=output_path, deskew=DESKEW ) From b8a780d684bb85260fbe9dc70ead21e8e153ad78 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 Jan 2020 12:40:48 -0800 Subject: [PATCH 333/880] Wait for file based on pikepdf --- misc/watcher.py | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index 427a23bc..2c2dc5be 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -56,11 +56,20 @@ def wait_for_file_ready(file_path): # watchdog event before the file is actually fully on disk, causing # pikepdf to fail. - current_size = None - while current_size != file_path.stat().st_size: - current_size = file_path.stat().st_size - log.debug(f'file_path current_size: {current_size}') - time.sleep(POLL_NEW_FILE_SECONDS) + retries = 5 + while retries: + try: + pdf = pikepdf.open(file_path) + except (FileNotFoundError, pikepdf.PdfError) as e: + log.info(f"File {file_path} is not ready yet") + log.debug("Exception was", exc_info=e) + time.sleep(POLL_NEW_FILE_SECONDS) + retries -= 1 + else: + pdf.close() + return True + + return False def execute_ocrmypdf(file_path): @@ -69,7 +78,9 @@ def execute_ocrmypdf(file_path): log.info("-" * 20) log.info(f'New file: {file_path}. Waiting until fully loaded...') - wait_for_file_ready(file_path) + if not wait_for_file_ready(file_path): + log.info(f"Gave up waiting for {file_path} to become ready") + return log.info(f'Attempting to OCRmyPDF to: {output_path}') exit_code = ocrmypdf.ocr( input_file=file_path, output_file=output_path, deskew=DESKEW From 6f66232d44dc75c2af5b2b09c67cb5b0b13d75f2 Mon Sep 17 00:00:00 2001 From: Matthias Braun Date: Fri, 31 Jan 2020 01:24:41 +0100 Subject: [PATCH 334/880] Fix typos, add instructions for training data (#477) --- README.md | 19 +++++++++++-------- 1 file changed, 11 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index d3632418..28824cb6 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ Main features - Keeps the exact resolution of the original embedded images - When possible, inserts OCR information as a "lossless" operation without disrupting any other content - Optimizes PDF images, often producing files smaller than the input file -- If requested deskews and/or cleans the image before performing OCR +- If requested, deskews and/or cleans the image before performing OCR - Validates input and output files - Distributes work across all available CPU cores - Uses [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) engine to recognize more than [100 languages](https://github.com/tesseract-ocr/tessdata) @@ -50,7 +50,7 @@ For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/ Motivation ---------- -I searched the web for a free command line tool to OCR PDF files: I found many, but none of them were really satisfying. +I searched the web for a free command line tool to OCR PDF files: I found many, but none of them were really satisfying: - Either they produced PDF files with misplaced text under the image (making copy/paste impossible) - Or they did not handle accents and multilingual characters @@ -97,7 +97,10 @@ OCRmyPDF uses Tesseract for OCR, and relies on its language packs. For Linux use apt-cache search tesseract-ocr # Debian/Ubuntu users -apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified language back +apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified language pack + +# Arch Linux users +pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs ``` You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested. @@ -105,7 +108,7 @@ You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what l Documentation and support ------------------------- -Once ocrmypdf is installed, the built-in help which explains the command syntax and options can be accessed via: +Once OCRmyPDF is installed, the built-in help which explains the command syntax and options can be accessed via: ```bash ocrmypdf --help @@ -118,20 +121,20 @@ Please report issues on our [GitHub issues](https://github.com/jbarlow83/OCRmyPD Requirements ------------ -In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD. +In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. OCRmyPDF is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD. Press & Media ------------- - [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a) - [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c) -- [c't 1-2014, page 59](http://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't -- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](http://heise.de/-2356670) +- [c't 1-2014, page 59](https://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't +- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670) Business enquiries ------------------ -OCRmyPDF would not be the software that it is today is without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system. +OCRmyPDF would not be the software that it is today without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system. License ------- From f6d7aa6e333642c660c7feae237b9f799b4a3a47 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 Jan 2020 17:35:20 -0800 Subject: [PATCH 335/880] Refactor page rotation and re-enable message at info level --- src/ocrmypdf/_pipeline.py | 71 +++++++++++++++++++-------------------- 1 file changed, 35 insertions(+), 36 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e3b0a21d..a4a41dcd 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -329,6 +329,35 @@ def rasterize_preview(input_file, page_context): return output_file +def describe_rotation(page_context, orient_conf, correction): + """ + Describe the page rotation we are going to perform. + """ + direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} + turns = {0: ' ', 90: '⬏', 180: '↻', 270: '⬑'} + + existing_rotation = page_context.pageinfo.rotation + action = '' + if orient_conf.confidence >= page_context.options.rotate_pages_threshold: + if correction != 0: + action = 'will rotate ' + turns[correction] + else: + action = 'rotation appears correct' + else: + if correction != 0: + action = 'confidence too low to rotate' + else: + action = 'no change' + + facing = '' + + if existing_rotation != 0: + facing = f"with existing rotation {direction.get(existing_rotation, '?')}, " + facing += f"page is facing {direction.get(orient_conf.angle, '?')}" + + return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}" + + def get_orientation_correction(preview, page_context): """ Work out orientation correct for each page. @@ -355,44 +384,14 @@ def get_orientation_correction(preview, page_context): tesseract_env=page_context.options.tesseract_env, ) - direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'} - - existing_rotation = page_context.pageinfo.rotation - correction = orient_conf.angle % 360 - - apply_correction = False - action = '' - if orient_conf.confidence >= page_context.options.rotate_pages_threshold: - if correction != 0: - apply_correction = True - action = ' - will rotate' - else: - action = ' - rotation appears correct' - else: - if correction != 0: - action = ' - confidence too low to rotate' - else: - action = ' - no change' - - facing = '' - if existing_rotation != 0: - facing = 'with existing rotation {}, '.format( - direction.get(existing_rotation, '?') - ) - facing += 'page is facing {}'.format(direction.get(orient_conf.angle, '?')) - - page_context.log.debug( - '{pagenum:4d}: {facing}, confidence {conf:.2f}{action}'.format( - pagenum=page_context.pageinfo.pageno, - facing=facing, - conf=orient_conf.confidence, - action=action, - ) - ) - - if apply_correction: + page_context.log.info(describe_rotation(page_context, orient_conf, correction)) + if ( + orient_conf.confidence >= page_context.options.rotate_pages_threshold + and correction != 0 + ): return correction + return 0 From fe2b07652bca3f7c4c1c77cc1ff433ae8ea13cd7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 9 Feb 2020 23:48:53 -0800 Subject: [PATCH 336/880] docs: simplify/fix Ubuntu 18.04 install instructions --- docs/installation.rst | 24 +++++++++++++----------- 1 file changed, 13 insertions(+), 11 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4ab720f9..c06dec78 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -141,18 +141,20 @@ first install the system version to get most of the dependencies: .. code-block:: bash - sudo apt-get update - sudo apt-get install \ - ocrmypdf - -There are a few system dependency changes since ocrmypdf 6.1.2. Let's -get these, too. - -.. code-block:: bash - - sudo apt-get install \ + sudo apt-get -y update + sudo apt-get -y install \ + ghostscript \ + icc-profiles-free \ + liblept5 \ libxml2 \ - pngquant + pngquant \ + python3-cffi \ + python3-distutils \ + python3-pkg-resources \ + python3-reportlab \ + qpdf \ + tesseract-ocr \ + zlib1g We will need a newer version of ``pip`` then was available for Ubuntu 18.04: From 4fdbf55c11bbf95b61c87670a24fd50074aae3af Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 9 Feb 2020 23:50:56 -0800 Subject: [PATCH 337/880] setup: approve pdfminer.six 20200124 --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 50ce881c..0b4a61da 100644 --- a/setup.py +++ b/setup.py @@ -98,7 +98,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20200104', + 'pdfminer.six >= 20181108, <= 20200124', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 09f15ac4c0e21fe3c4e4fbfa0041c592acd44d3d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 10 Feb 2020 01:01:49 -0800 Subject: [PATCH 338/880] v9.6.0 notes --- docs/release_notes.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 0c107aef..c61f816e 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,15 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.6.0 +====== + +- pdfminer.six is now supported up to version 2020-01-24. +- Messages are explaining page rotation decisions are now shown at the standard + verbosity level again when ``--rotate-pages``. In some previous version they + were set to debug level messages that only appeared with the parameter ``-v1``. +- Documentation improvements. + v9.5.0 ====== From bdb7f92131acd7f16ce5649e564a89924f32f1a6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 10 Feb 2020 01:10:12 -0800 Subject: [PATCH 339/880] ifmain -> main() --- misc/watcher.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/misc/watcher.py b/misc/watcher.py index 2c2dc5be..e97d4625 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -19,11 +19,14 @@ import time from datetime import datetime from pathlib import Path +import pikepdf from watchdog.events import PatternMatchingEventHandler from watchdog.observers import Observer import ocrmypdf +# pylint: disable=logging-format-interpolation + INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) @@ -98,7 +101,7 @@ class HandleObserverEvent(PatternMatchingEventHandler): execute_ocrmypdf(event.src_path) -if __name__ == "__main__": +def main(): ocrmypdf.configure_logging( verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True ) @@ -128,3 +131,7 @@ if __name__ == "__main__": except KeyboardInterrupt: observer.stop() observer.join() + + +if __name__ == "__main__": + main() From 2f2602357bf0eec88f2f8bf792ae16486ed73ee8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 10 Feb 2020 01:13:28 -0800 Subject: [PATCH 340/880] v9.6.0 notes updated --- docs/release_notes.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c61f816e..cda2f118 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,10 +16,13 @@ licensed under GPLv3. v9.6.0 ====== +- Fixed a regression with transferring metadata from the input PDF to the output + PDF in certain situations. - pdfminer.six is now supported up to version 2020-01-24. - Messages are explaining page rotation decisions are now shown at the standard verbosity level again when ``--rotate-pages``. In some previous version they were set to debug level messages that only appeared with the parameter ``-v1``. +- Improvements to ``misc/watcher.py``. Thanks to @ianalexander and @svenihoney. - Documentation improvements. v9.5.0 From 683ffb84e88322b574a7c67f8aeda508b985720c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 10 Feb 2020 01:20:33 -0800 Subject: [PATCH 341/880] Update reqs --- requirements/main.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 5ac09a35..c45fee40 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -1,10 +1,10 @@ # requirements.txt can be used to replicate the developer's build environment # setup.py lists a separate set of requirements that are looser to simplify # installation -cffi == 1.13.2 +cffi == 1.14.0 img2pdf == 0.3.3 -pdfminer.six == 20200104 -pikepdf == 1.8.2 +pdfminer.six == 20200124 +pikepdf == 1.10.1 Pillow == 7.0.0 -reportlab == 3.5.32 -tqdm == 4.41.1 +reportlab == 3.5.34 +tqdm == 4.42.1 From 4a27124eab1d87466bb5fd9818c61377f88e97c4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Feb 2020 00:07:18 -0800 Subject: [PATCH 342/880] Simplify metadata for invalid xml in output Removes possibly non-free resource enron1.pdf. --- debian/copyright | 6 ------ tests/resources/enron1.pdf | Bin 86377 -> 0 bytes tests/test_metadata.py | 16 ++++++++++++---- 3 files changed, 12 insertions(+), 10 deletions(-) delete mode 100644 tests/resources/enron1.pdf diff --git a/debian/copyright b/debian/copyright index 62a52737..046a20d7 100644 --- a/debian/copyright +++ b/debian/copyright @@ -95,12 +95,6 @@ Files: tests/resources/vector.pdf Copyright: (C) 2018 Catscratch License: Expat -Files: test/resources/enron*.pdf -Copyright: EnronData.org -License: CC-BY-3.0 - See: https://enrondata.readthedocs.io/en/latest/data/edo-enron-email-pst-dataset/ -Comment: Unprocessed. - Files: src/ocrmypdf/data/sRGB.icc Copyright: Kai-Uwe Behrmann Marti Maria diff --git a/tests/resources/enron1.pdf b/tests/resources/enron1.pdf deleted file mode 100644 index 65aa7fb60435ad25bca409c5854d9d9e006935ed..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 86377 zcmY!laB_-B3^7X+6Lc{{12YRu zF$;4HF+&4ObTI=^n1G#t$8(0_^7@DOeC8Zb{rzTmL zrkR@?nwTWo5mph)YiGw>T#{H+0uD+8UR70Be>YxmobZAy21OA#>O(3EQWf-_6H5|v z^3xS~^#c;qQ;QW0A#Ov8Pml*;2}VCy!Prbe-z_tzB(+FE-z_JxB-JG~IX@*;LEk4e zFTEr~!4P5s$bjOKqSVA(UM`Q$q=bZo^n`?>goK2IFFOx3@D?~^GAg9CCb6+4vN^UL zVbh%8wpxRct$~4=A>%!B4=61`bwg7X*jSL;O%32F%@jE$L%aj>gaWU=Z)!?rqEmi_ zLbQT`f`NjWp_!hsk%57sg1My;I7tU27NzEuzy3}uj+AiGU0;DKxb4P*-w zGh-vTC@4rk3PFy9ggh+J^n**2O2A1A95@R4eu=rMppY_D&=1c{DalYUHZd|$(Dz8q zOwTA$FfjoYANo$2CB*@$Mb7!T1^Ic9n8BXPobz+?i-HRhlT#J+T~f<3lT+P`5-UN4 z6eRh9oErp=4n(NDxxRPZQxDO$hepvGnB9_@1WvA~RpuxbcUk;QNKGZ(gIP{(B1J`R)W`;L}M>-$0;!f6xP&brv zK4{f>P=f24=NG%xYTuPUOuCjH`ZSIu?$L=?CTq^GII2r4QaQb}(5G>1aS6#`sn0?mh#G2fU6br74ii};%<=Wq zf32eR%;$erh4P5334U?tr+e7)3vD}_no7C$#Z7&&f&X?3rwPZ?~O3n>dPOLf2+n?Zwm5#r+WO%x>T{+@_Sjp3R%q59;6saaQ6e-@}- zxjb&A+g_{wiB+Xi%{RPtm3L>Ee_eHEn}AJe@cL6Z+y{e;^nZqj&&`&t=&LG~OVa1P zqJ8yA+^d!0t9X^PUro*X8@TtE|F)+O8vT0S==Dm*PR=x~IP!j}QS5D#v>jH*gFYO2 zc;ekE>-lbeN~?3vY)GAV^nsw|CjQqm64!sX>#xak`zBE3kcYF!gVop)Qk zwOeg*P|_5H+u1v;D8aFokjTWvU}Tb9*ZGIG6D;QDnB&AFy&?UlHBePRK}^G6;BgO&;J zPG1`yThl&qs!{vbRlUy_-SuL+a?y&wSN`S31u^UvsU?`MgEFL$AGDatFBI->UK9U?Cu{v>FeLcO(uLj zwKOv6WR?-n+gUQ#w70)e-+i{@g#VS*o>x}?j9HSfGHIJd-s{jat7`tMKD?n2=Xuw6 zZ&$fSL>5odmFU+~_eU<{ZYwjDyQ;nEQtDmr$r7Bqr(8X8VO__X@YfzU)-AS*)O_R; zrsn$OTKSB{2`?32T`haC^Q*4w!Js2K(&^HiZ+Bg{+7#r>$N$>HBJX*U|G}Wgg?qcE zrz|~^xL1h%zJgoTldJY?!VOnH`gVUsTKE6k%?~!jmEN{~QxZAzaC1WWbyM^ByMLSg za<9COn{~wFM&8dk?|5odS6BT%TKN6)i!)2EB<|gGyLoON-=5i5#X8o0kC*BC>puJc z!A<*G4p*tlZMTiyUzHuU_!hg}n(!h&`{=O8(VL~N*nI%S@l^h|+g~#Ky}k7Q2UFzK z_`1;erZwTQxBT{=IR4UhU;p2hn@Q8B-V-lT-!$p#DnIk<6TNpSsnoM-QdF@LsTI8!zV_T2KKC-qt=rP2nUl31ntbrT_`^AQ2LHzyp%%)27RKMc z_3xm`;{S&p$lsi?z9#IS_uQZBxaJq7p8I-Vb#qq#rcLQ*Up<=tsQU5t>ork2k+FxE zTi5(~|4;j0*u%}gmUVP)`&heRTRd-4Nrd|$$1B>!+t#>!{&;-K)#a7dN7uYc^^4lP z>QB7%|AV^P(>Ja9WOqn*ZiVSu<%RVf{V$)rd=)xl!l?^Vf4~1ZU!3xAx$d3b$eE_A zFDTuv2spOp%%w=TwUM&dv_JMq`OOXeUK#fA^wnuoCr9cUZ@f17$lLziR(E~Avu%9u zmvnB`_xu0iOI@G;|B-h3m9_9csc%=$`%1?7CtfmmR&KaD?X1@8U+*`$fA!jP_}AsD zt1VY2eYpQ`|CZ?jk(*OWdFv(r91z^375nnfoV?XjuB!iU<^L=hVe`{IMd?-Oem9j* zFSp3u`+jtdp5p42d)o?@>8>{XbG2>Fs%<5qCr&L7J<9)kja0O6VbAr9+uo;Lq&wH# ziQFvp_uQ;G^Kbl>-xj%PSI@`wimR8jAVPy0UY-Tu8iIc)Cz;I~&h*4WL|Ou0I9(dl0d%PI6LVsf$yq(^|Fthxt?@5-vmvfDI&AU9UyJr;^Vr8It@ig0 z`xk!Zl~MZDA8Y>br*{;bJoU?eTj(FVKPylDVEDd6{p+hMk+E`VSKn>kbz)7(zQuow zPM7TMll-vs6B7N7tTdk}8d+WJ1X9D+bx*hlFc%r2KuTQe!U)9`yukwrR z)zrRx=lrTmmj0bTUelkB@a(V6NW0`I}wPA;E zt)De}&7YjWzpods{AafI_o<9?tAvy134dMXWq!T(`I!D&DK;mwhoi$bGs-ZE9x-hcPZ_hmCTzcR5~ zvnJf_%IcDQk@p*Z#fE>o+*OM0|xzwlm(_Y@Q+O(c|Z=KQV#B*A2 zSLe+Sl~`00cXrm)-u~AiX{IOc=1q@j`@2^yZc9=^h~K{Te@aVdPmMkAu>bX+6Y860 z)tmf3T_^eaMWJ-zobJ*+y^3a>Qw~||b2=r`c1`mSN2Ax?CU0@};zp4=T`tW#%Q%C& z9J{6{PGCuNoH$cbZ|COf`{f&JF6vh967+jiv|rfbe!ubO|F_j5E$db-y16qY-mF@+ z$}nP1=&j9*%%YuXq1oqL!kaqiUm z?9I8M)vjxo3p#Vx}u>+o^d@y#D){1Luy_zMo)TxiYl+ zYe)p^u~!mFWBz|sC=?0|C(pk|DF2_ z{hK3h?d`vCQ?1w}BF#EK@FUCGhjITBH^!u=uA6@6W@G&R+ZSA3@rA#+I%D}wVUFUv zH8;xAjkcMVn?&R^9JKc2Ua{k(ooMsZb>G>q=X7PPH~FJ_nr%}zPp7VF`-MBfh7m7k zug*`}oGTZpdXCF&wUteF)QmeivvURf?LE95WKV8tR6c3Q8JKG2F#E4<)OSm}vL{}G zPSbd2AD^Ukekz~S?(@!e$B*+>r##(s=kn|12WvKE)!bpWbLZXsOEc8$g1>I`^3Sem zM?_y4ufCk5$Wo@aKgIGn;UQ4T&*NM|8bAQVHO@G=7a+8b^NnE z&a;ZmbN!u}7;dm;qvXMg1;0+eZjG>&Ui10T8s2_Rw&Rhlw?2o>S3RQUJnwN}a-{41 z-MOcWPs(*o>v)}J&i49m%{R_}_cxjci!{EsNL!zAvG?GnN6vPC9p&xRH_a3(|Mu!f zQ{}fwVv(E5Q#QQ&@~Zsz@w=>FWlg##L}tdU-gGi@=A}<>cD*>7;Fwmle{=O8!*>xk z`6fk0p54&1Do1Zk$*L(D7yfMIk>;&m^IzfTW&Y~0Wz9=2+}h=%c;yPuWd5l=kqf^j zZFqNO|MWMCeueDI+m|fZ+*`n`s$)ApaMc>^)suYn*SrZdvOFnwXvgZ+H=2bVmkJbb z-w^Zr>}w;rx1r{2r)8?QPJVOnePpZLq^651!SBRoANv~p^v5BskPL5mw%5%OKjmVS z=UxJ+Su5kV^-T0Ti!HZ#7AnO#$E=!i&zWt~UfWGl zlP0cwd*bm4Md^34zPfXkZgCE7&sS&T?$dt8@%{Wgt~VVMPOX)>BNOk$Heu60XRnIq zCsy6yEdItdE&p=;hwJy3e(C?rSFia=F@^8OnKV3~ZZkDHEXC{}+7?t}n8)l;qMkQagE+ zStfRB=+hwXp0qi2TZI4JaR}osEtlC@us}o3}GR5I$*cINw-%kK=B`C(lX)r&gupSAS3{NO+T+mdaYWs(0#&HOER7q!Yff zH6Ggjfpt@pb{1=0Pz^utX3;RAe+yS1&0yKP_xy97HLt2ZC>`nA#aPMXprEb2qg;4n zl)~dTV!VG6HgVZ;%uIT%aPEkgb!C{C%{9f=tjy!9bXNWPTr=zI0WAUToa-mJK0Aal zZ;<+MPf%NBQ@U#BRfk+T%VciBTcWlEO){d56!J5*?d#}bso(-Ef^{AfFg8gYn3npIm;Xa*}`O-03K)WEqY350v zGNVg>4t3v;Rn-$)2oO$M(HCM_zv!-ItZsjSP zXPr(Hw<<_o+BmgD;o#g&LN%h_9Qf;H&MaZroDuin?ci$zAWdCLu#IRb)d`_KPL*9aCp7*w!Fwqol3Pt^IPrGaG}R zj7?&_OBz$P*W@HV(A#)xiq&J`8627BTH2m#6ueig-nb!1do`=ooXM?I4^=q_9XPeZ zzR{*FlYPzEnk1!Y3pKeGn|*sT)*o7$b~7d=%sG|y-n0nM6xLb%HX=4oU(fihyOnfu z4O>cUAGh@iw=kDae;V}TE**+1@_jdH@1sc@?rN-gbBaqQH?fs>;;XEdpr|dAr;1*U zNll$FHMc8t*~SiTuRn|S7TMld{bs$<>YIy2<~Lp~v0rU@-YB7Uzqson8y6Ip4uyvcVT)*jM7wDR`ok&!!BPOb43<2Bs;tMLD(PsV$1u0He1Y;%$DHvZ~2jd{tetM061*4{HYJ^@2}hvUa+M3z-q6#SA*U-ZCc#n{b+Si_=J-s_xBhX?R=#vXmirM+>J(1|Rku!LQ+cFtbC>hzjJ|JXvmic4zW zZ{6GUdg-DaYuBV1ubXvuvYvVv^DphUDjS%(F2`-&FR;oq%%W6ytCPi~`~Hji-imJG zI^dNa78ZFdi>GPhCZS#2QzSNgvJSJ_%yo7{l&kfv)q)mh{=^dC0 zy&5U0OQhertdZI9Y35Xh1v}U2Bur+zzt%ROIBe~gKQp}D)~pe!XFsZznRr_1Yts5p zZ@VA4rG!n6jPxyQc(qVYqT%VBNXr9CXP;!nzI@8&xi~GqJya>oaPy|}h_hQ>oR53D zt}pZK5w1;}A6cIX;EAi@wKqTT>iPi}>zT*58Y%N^ojSGY`}UpdlC0N=Z*cvySw5^y~p>gSku9XnbsmXq}Lh|4MoN z(zrLGbEMu+zj&MZYx&}Mb@tV%$CNI=%>AU5$ltjC?1Vi$$xKmcUS$SrH+mH$YzkkH z_kGQ-ZQ=Hs_wrB2A3VSR-jlEQ7Z+%?zP$f({>hiqBKF3<6|Ly4J!$T?J4yE zM`HU<*IY7uckI50Yt6i={rA3Q9tir9v0hFj@Wb+Lv+Q`5h&#zx9MS%^ivQ%l1vjpT zM*GUMUFwzK`*v!&`?(9XQfn$=Ten3Vbf|J&Di>k>_PIC z=l&lKnG?~war(}^`zp4E7hR7#sFCHN)^wFg>h_Cc89m{PRrkDDMah`juHuMl(aD8 z&ypv7e(#oNT)W*Zu5PDAYNy%bDCbh~XVc#Kt+}x6i@;2A^5g(_1N)KD~DD>Uc8q;ml z*Jk{Ok~L?zMta#zlAx12qvE~~G|%hX%JExu+|)2@@gkfuuIP=U*#L>7yG{JZm`SJ zwtp^rv%C(+a!ra5l>N1z)j8zd##14`Wj1ClJa@^*`nlF&`+GccwkxilD)-#5|M~k^YmUzD zuhZAu`Qk`V>XYSsr}aTGeZk}Wme0r z<~rfBbo<3k-O-C4iRKAKoK+7iFH!I{>$Eeyb!Gp{-kFx3pT)W?Qzmq#EA2?%U9oii zhArx=HfwDW53${4FH$}AoYdXb=WdA9DAcq%gw4u4qpZ!7yhq!sD?CBeJ*;%qtG7WW zy_Yl6pWe*n$U3wpc+aJlCnb|F9BVnIUH0o7_LXy>tU2Sq?wbczC##fz%$*oyoneA9(i2 zI>x+6jBCvq)9F81mY&^Yed~#&r{udyYbsd7vbP-)-*{=tj(1Y|=fZD!M!oprnKE@_ z`KJ+AX*waUxz7D^v=v9cQbRmOE)xb(y9LLnI86T zwx!jEOD{q;v>%+ZQsCV7UtOzfy1FjyJu-XOW`X9d9EUeEd~Mmdan=gk(y+ATr@FQC zS$`eabxeEd>2pp$jO8OguB!aU=l4W&Liuuru#WH-r?(`#Pu_A>sW+5a_!iUB39_pT zo-Iz<`08llaV~qau)Z7x>BC`1H!gBMzGm5$&DDil-b{LQl~voSuKdiY+gF{RN(z^6 zG4%9o`I>8=x2Eyxez((lS;4grH42{me|0MB*{_3ZE^XSBwP+jb-_MC*&y(h*9$noS zDiKi_pBCn5&J{EHNcGiSsp&5h1Nb*T{_*fCr-Alyn-kGjcIK=};OM+MCuLKYw6-zN zti)?E|Bg-Ex@p#gwKwb=t{%F%=LZ2H7-P(Y4(?Uz7yTse(iQp z=hUS2;hmw1(^nh%%=KyvoqDQ^Y09?*Q@($>)dFcZ&BMCZi%mUb;(ESn!*#Fj z@R*wcukQWW8GWbKW9A3zd$JpYF3;cO=WXH{Up&`e_}k+}RYzudcZa8@KCQIBFV5D>7=J+YrN!5!egEZPc0~3 zRF(4f(j(E$U2Lzn%@v($>pJfP-{y;rp>Lbtt>Uhoq?E@OaZ}Fw8`q(8@%zKYro^PX zwiYOTe{Lu(JKtxP_9mUnT8Ea*oVwW8?#Lr?<-pN}#tFAHkul3zuu6r>}HHv$~CTZ>Z%3Dh!zlpEwm@Pc@ z*k@a#r^jA-biLT*`107Y_lq8jB~JU>s{Qa8%Tnp&-3h9?aeE6=zi6y4$(~&-FVY(M zO6E;j3;Gv4y2JHjrmm8?X?yp^g3sqSAA0J#`{k-dvz^79HaqUm@%PHwQ?>Au zn3&V%m%-wUujk#|@p2UxUs~+um%9^6yTjRo@93~>zO`}Z(bdf_yK{z^*lY|dGOQi6<5pFta3R$ zuXMF_*lzD<)~k&^I{uMZs?r#{=>PBh(6jy%|0&(GxxK1>SJS1O3+Xv-^V_)lIWF6V z?Y?qwP0nfK&A&|TSiU^zuxHxu`2Eu5$@5>VkSqKySf*b%Bgkuh+vFn^tYME=y|xY8 zZ>X!A9ws-Ne@0P-a^|eUkf0YUH-ypiwqd4j5mFR}U>D}gw zj@?%%R<>Tmt(S1 zH)TbLNG~yYG*M!G)UUuWle^!azTB;4968ZM&~Z_ZqqD!&%#=mE-z3HU6qlB7d&4Z( z`SF#U_9vK~RNnRP zCHID3iF20unO@KPuuNFH=*3Bc)n02}Xx>`H{BP~!m%G~|Qa7$SQTF;)!fHd`=;)xxEdK6Wx38Zq_0U=cITIy zv!>1neOAjTU%rylvnx-??ck}fw69k;MuwU`v7h*{>)*4GZ$`PyYnqI*oYwW`U%$p3 zre&xtu=rletyslb7Yl_>ORrw@zb+z7X4$C|UMYKCt$8uSbc(_HEwfhG?bvaZ`v&X( zl?Rn$*cWcz5njwaH#BIQ#gzj~ZYM^vPI%e6!9q?j^Cn~S8#7(mH6js;?QbGl>(~m9 z-gtTT>DkUumo&vNvCZshPs74)e(IQ>aWjy23g@xi)>_%Zskc~nnw`lGd$s0=rBqeP z6T@ep`XWky3)+>QNICi7?^$0t-!oc9tCv)!tZ6S2IdE!sLg&@pJ9SDhdjkE$k#uGk`|?LSZNMq}u! zbb&P=rhJ(hI#*IH;$r{ed7=A{X*>7tIre`;yqR^;oiK^i#%$3&d zwb(c?>}@h*+#=C&^~0g{uFBl{&1?4XFtePP=E$;yX-e16cd4c4J&a5amD~=^sbLQ) z-?&>!;rW%nr+xdvUu7L|cq$)rkSQs9MLheQHFnKH!9|yzsQg{Ku4<9WC*`!>%7b@q8m-jk?7MLAki@krRjr`XtDX`k_4j0~S|`x>!=>uayBYCi z9WxxCNb80ye8Fs$vuF8pgMBsM@5b_nU%K^>W%cS$H$qDDBJkEZiCrF4T2N`cd0X?Z>^A z_qID<`*>=V_Imj(7I7eaCXF6piFW;QU? z?7m;M^zl?%(N%7-vI6|y*FBNW6kl|aExGJ&VVF>Oso2CBm$syTm3@1Hdk5FzXC8~c zzyHdw(5avGdis~ryy88Rw7xvhQns7MAFKN{VXMv6SFK;dV;341s{UGf*+RzL>Ts|} z%+-&pcFnpav;JPW>1;)%>%QmYVr5lcX72H*6bip6AA5J*x?_xWM?>#C~Vcx2B@7j`e-t}eF$gY3(Dl4aW&!m-^ zEvq@g&00^cTs`~vx<%7xtZWpvD^L}+{{HIKz4vD878SmW+VMy$@eyt+%yy@!WK68PmEb+4V=RCcd1O_inSV-EThOw-xJq-Tqb? zI=X3>vNn3Io5=rkUr`Zj@n-GnzuYRNPgk4Ve^b%=g`N3Xsrl`U&EAi)41R2hR%uX3 zG&6nQsr}Mqm(USWmecPGuXBX|xOzvwx^(K*&=u?0BELoFzqIVo{hf8DX`>ft|(C0LftEqlLe-&U)*(D21CkBUw@oKRyjcgL*o zaLI+L)8Fsps<>{`B>m*i3Wn2l#@?l&$F+rL$*y0ux})>jiuDTXs;X_nxQsGL~cXCB9+_`@J$E%dB*AM-Qtcg5zJ7CG*a}RH<%XX`dyzp)6{KBunf~!q$ zs(Y-eWHEYbc=pr7I^)x~HmR4IzTdvVPWt=H_18cD+q>uOzKLlY-)!T(l2VX!b8X6@ zy<*qWA8)&N;Aw~H+um1PGcV-sp4;90rp(mG^zGG6S6@|pZnoR#_p6hQa zzH%;0d@UQpx8s*vd~e|-&i3B##hJ@CeC5;Y*d`WS8v46u^`rXe{0EcvR7bw5NS%{% zsJGmCp32J^rM-`@epP%tRdQY7o?Ui5yLKJ=`0Ce%X?I(+W!A6jzj{@4+nej{ec{if zj<0_09Q|$GhSYopkyZC+<_X59JYSGnpJ>$=uG`>LnxFHm;wpni$mEwdH<>qxtjft* zvih&_{21Qo{8pQTDO(~MH+&5{A?y(x7@NCyvhh#ts9Oud>rUtXndK3@aNRB0^^?v? z+b{6_wCRjdjmw|a+FQ04I_%c|Z+`nxYKMFH+rIFtq5YRuC%aEs&bLl3?AXHAPe1H( zcAVyJ_^72)c=cVy(sk41Ud9w3*6?z>v9WY1pW(Cz)%%~Cop+zrrLpeLG3iRXz?ieP zb5?47Vyd+~RCQ6>?0{JUo_=AId!7GXn)1Un{8;|&vaKhKEw|alrG_uk zPM)x=_SI&6VWpGnS2KoZJb$=q%WZqjPd;Y$aR>Vfp9Ta?cztquPm1A{(6-GIZZ)^J zzSCM1u+=uM@k#pUd8=DAxHfr3xYqEM8)WE>QH=q%Kw#Cn``XC-5Q%_v0qu&peG<4@p#=PA0^-IZ(NtH zT*AI$U4dbjwqw|aZ)~xT?(A8eS!$ZIcVe-_lRx*SyRVyYDLY91WUTGCo7ac#qSdg|HL^W2E+FKv{c6=}R>OLd-47c*SxQkc)Je*5Il{I@F8th&feI?Xlr`PodTHEa6I<+O& zuF@`fccOao(JQRstcGmi&XR@hmqQ~ZrSor=t`Q2qcrEgEZ|mx~*iu!mwN0y;-))F~ z|0QSLv8#7di<7-~y;bkO_hG`BwQJXX+PWiaZ|e1pUNuwoI#>V6wYS_MwYV}-zUn}B z;ni=7Pdu%6zIw=|K5NgeRhzGNbX{Az!NhhZhxhBpp{1(g?XO<(zu#H4byru(R%R9L zKTWIU7w=rVj;|nfy(+|0PdBdK%^u!9d9@yUq3PwT*H$!bJmZ*l*Y(Ysb*E0>To~WJ z+C;}TX6M@NReF!IZd3_B->_>^-1{f1PRE^!DBXI)WK{%Ir8i?e-py zorU4EZtXI;cNXU@M(U)fHif4;xg+=1h!oEno8hhmEW9f{07lPNdrzhwSkWm81V|L525Oa2rnd}{a4#3rg|?!L3f9{vwdEBw}fKOy@G^Q7uaQw@Jj z>-}s0yR=8yJ+HO!&qdW{K9=BxOb72M=d5s$c*Su}M)I2Al*oe#CmNaB_X$aJnU$pE zWaxckiaF45iXrs;`;Durce!t#!TIKy-ml-Dmt#YN84c?v9GxV8yw&FOyZ?3uKe}`Z z`nP1MO!izqaiaX`nNR-RHs;}dr+Ha8s`BqSqy1fH*6QrNkkxWL`fS*|Or^@MpS=&i zI7ZF<+{4!=`uT$0haI6`x89$y=$~=Uts}f^*)+aXtS)^d_`E`U{^UQiFJ?8UZ{f<; zK5Th{OJjYw+qdTDlCm<|BC4)sfAs$y+O(^%IB0Q%+xKS#d??j-P68ZE;~~;iZ>wc_D%7h>T74^Zu46Es_U}1Ro4^b1U}DOq>5yR{P`H zZIis#?Npk5SZ1l{_prph)89oU#_<%Ldo4?PUbynq8S(a)wDQ(=Ij>&0W$N{?ucy9+J$xND$s@Ht z%z0`=+^+S}ADPz%IjmmjwPye8wa@J5oLTJ}y|$YBUR2!%=LN6t&z|PRxjuUwcJ@oy^|0u~I9`SPy-U{bEvu`%phB@yalD$zOWA-R;-8-`nq-2W*orR$@svQph?R@s-+?$!a&!tyuD@XAbWNZzX@iu9j z>4osj(BFGjee&LG%$hqjMb_kXGtW7nDAn+okGHb2E^piZVJZ8YwKb*jPolrKE?K)o z?AcWg*L&HqQ^PfDubNigdN%cJkoMQ_tF@kHP4GLfu`FYTR8;TAe9wUCr>K+JqHCjf>umOn(!F%*M2*qwGJmIc z(V;~#Pj0q_DWBk5&w6gtn%YUZEe|98#N)eTCVYK2`Q9Y;vp1v9URrr-b8mTbJMW+6 ze;q%snY?m!U;6sV?`E=e8{Jp>-2Yr&+UBgxw&wRwpMO5kcwjN#+6UtNYwK*YWNeu# zw6^^Ex?(Xqga5br7nuiQYe2*Q##b}F@A%z18-J{qMJ`yTnD4fy>ZkVmZ@xTk|GHyl z{bfscR({Ps%htCpQ8(*8DCe*J^s9Sjhf=n&Wm&P!Yk|bzH<{X(`RzWQ-@a-aTN~pR zQIN{D-gn*IR>#jBTIb%&m8}e!4v`AiDNi z^ZaY^XD-|Ce}CU-_mWVX&k757-(OW>JLBD5)7@Wpd#=veKPNM%V3h=?z&7E?J1XM# z7f$8h-+Ogt8{79MTd)55UcV*Od3EEpu+O{xzFsRf>Hi*&iOWL2JXjI>pe3w1Wm2oI z$fv4H(V?4+_f}cH-68N_LH=Q#c4genw+GqMw`@E0%?E}PMEfJ>T{WnVw1Bq_qJwE3k$Z+`+7(B{IxK%*22a6e4f^7ygzj0cbKL` z=F-Np&ADdh!;H_}R;tYSdi~KxL7VMs)~@4Oe_ngi)Z9l(YjvLA%+x-fy;`ogF7I}6 zX6g^_O0K&tC2M>|R}LOYuqhnrUGR=Vxf|y%zm8 z<Az=lWH6XVr_4+1vGQy*hg0G z<+XOx0U8}x!+3m++yNfcOw(V(z^^gPwX^diB2M`l$lTEt?|C8kVp9bS>-71pR4Sid_GF zD_^^8?Nz@hU#T}sLMy8?MB90PIjvp0Y;joFy4=rMFV~jjKfJzssp#8iQ~R6?;#$W` zZmshD9kzLQ_S0*-mrk{`S$}7PW4uP@)6-KYD!kjkwk$)%WmfKHtDSyPSJ#I2{mw4X z+Z(zdd+p&}pQFz>HCC<8oEEn3ZS>#d`(E?&Ekn%nu1wcFzGiJkR@wga|87xxCz@`U zskC<4u6)z4VV7^M`thiZ?+=sb>HsID-T6mdF1?z%C8{H4Ugy57rP&D^pSbt^7Mglb z>wfmsEmLKF-g&q6Qu~Z;tBS5)XpCPVTrW}*xaIoU! z#wPB$7T;GF6dv5$(8#@>m3{iB1nEVLdDBYob{!)E=rmUFdTaCsh zj{MmA+z))p6gi~mXKFh`#tOq|5^Q&MMbF7MQNf3R!S<>pkv!t z6_xo@)@WZ3v^x(M{5-q+XedL3wCrEWWh#{_ zD`&rY)i^a?Qa4O@@|BSAZqxp?cTQz&%$gbJTdEp1zqaZ?(vP#3ydGyMYlP`Khh=9k zIG7Ud)*X8Kh7G za6+qTVccKQ87H+SFVIa>a5w(5>RtN9>|JxiepjAcxp-2lw!-QIS6vM!i*n!c39}7& zdNry`&LQ}HMzr70q|2$NB^-W>I)tzNvL$}wf|I*maKtvQT$dr>X5jn0W}&2;Hv7C; zU5hq(eJI=3bw1&6m|DZEwE`;+Ful0;dEb!>X)fI{UnZ`3;eRc1>bfXZ>D(IIs7kHf zE4MdH=X$vMUGo3+PyWllVmzC0W0kV9!pyt`?W`NQA5x~S_`u*S8EwDGNkVJ+lRKJQ zWD8^W9$%NibMu1#wx;-i^^JeGrfIhp8ovtonFvwR66{lA5qa^MX%ypEsTr%TnwARB zE^P1G{=|*>X~R{8nwmv!{>9E7ukYPv*4A2>H8s!bO}h5|(ADL=%WEf|oE2tr`S86{ zGEuRmdsl_6_m!2KxR&SNg_29+Qcf>_)TE}ZuH3)qT1;T~!Laq5dE0hfz3O#Iq2yj@ zvtLP-)wW01OIPR2^81-&93{GH(bB2^)~*lJa@udX;geML#w_7YOj+s~3(DEzHZhUA?W9eZDQsHAvnRIg4s>2jzN%_v^ZU74vQBm}%$^s%O}=`E zy(spgz}2F2)1^Y!F6#E<{h}^EZ+%n$x+u+_cH7+%TREfmUiipmbLYVAxuHfS`nw)f z&0TTF;DXc-{yDF*F7qap{^MCJI(L&zL_^+2wqxQCg17E^_&M~w`s&7Mr>z`{&hfnY z&Z|Fz*T%iz*`j)rBP}qzbDi2`PxcVY71}CS zQlkBSD;a+<{C-MAf$4Vh*|HngMY+?qy;`We`bFAm2Ddf0o|MZSO!~xHQ`meoUu|_G z%jShn-V2_a3f$B(%DQbCUto7-e!N_)>cOt?7rcp9w+@|HrEaYsHb48{G5Z@`pBvwN zxVL?F?~}zR_kHD2nLq1E^m&I*D+O22_y2WGctTClkyp8!qmShrWr1anhFmj9INkcdH3Y=Pj{{!zTeKL`XDm!PckYzG_zkV-KO(D5?oXMcy4&=^<9}ywN$p4yWqk88 zU1R3wg#A$wZ>zgawRcsQxE7sSb4~P<{k4NjmY>cJ`@otpWsdIausKnbtKsoWqqb?oC}S|9WnJ)YQJ6D(T)u@0*rQz9fCC=iP?) z1~1+_edH+1EM09cQ*TndxZ>K$xQa>K3w|tdF)6=tQ}KoU!RYCZ=fjRZGXMEnTh=DZ zr|p)#{k34TtFPAoct4H5z%fz#K;5QAe6l4C_ixDUx~UcY=F1a>q3PPhX6qF0T9*_IMkE#kOC!+yc0@EBF?4FD#s~aPh~oExV*M z6k=~AeL8>8q;`tB{H99}4tsCgrL28AEMIa%)YF@@o!%|^Px2UqP-b^4E$|Uo>s5l zu&(i@%I)6|PI5gy^r|~sBg-IgvW|1z*6m6j1h}OQ`vMP7Qx=kl1Pd!wZ zn7ZO~a7g&N z#+FyXnQ6;aFTd#JzKA5_NSNMW_-;%s&yvpPtHMI&^vXZ$Oy6>8 zYT9g3?i@qwnuMi>7wTLehi=_;HS0IK*_V?WL|2>EPFp^2?$v_f4*n zsw1neg)+siJ(fH(Vy^n%r2U6hyq5Ct{kb@d>wibYulx2?I>f} z+GE@6=BUrT$R&`s+fjA0sk!vbHHUISeG*saZR45|7q;4IwPJsl-PK(}tA&?;Yt-Ar z!Y^4g-6#Ks*TZ);p(R^`R&?>@JpDaAb}Q4G#V1bX8Gdql6>WcR^0^e&6xJoK&sX>9 zeNfCwzpA>eX;D|=>KBi;E;2gI7_%;H-SQ;v6xOx#pJk?R>ALDAYWHkXsO^V)XTlh+ za_!1`x~Nh5MuE9UbHlfd_gutnEOvxKlzOFzItX@DCzzfIt9rxy`_(D)l~SUS7uzLo zU(5%oSpH4S?(llUOX0peGuQsuJ%4S%HlxL&7awG}Ht)K4s=xK-R!~x@{?GXTaGmTM zk38u_20bUU88JTbOurLmT(@!A@I~3efkA$<`wbZ;od;)wc;){@nbe#RX0eQNQ?%Z}y#f&(x2Zn*W*nX1nUAbdg!Y z$G9f7K3S%dzhTvZ(_akN+?g@G{{H8~FGT_-Z9Vy0(`XLYSHs8!GImS0%{zR`{*`vt zzw)ezCc(! zDURn@T1;h)SpBbg3G<$|yVtm$6^jtvbXRNP456$yykR9}N52`~58TWqH*=X**doLI zM^DXNH!F(QE)?WydL+~8vM2t(~ewabyPj+>*as08@rAl zzmnx(xcY<=9sygm)DBDl={6etbhLhFOm71 zwZzmnbx+#W{+OLXLUyg?KM)i7FMsAf36YdW6*&uYBbNlSZ z1#5+$d2N-NkR1E)%N?%E3Tx8N^ro$~RJ#0Erz~;XimMF=PIX;%Tyuwa_0Av8+oO`U z&zjY<@c8+CtFAW8i8Qr(&AeJz`|!`m`*odDMfbn@xgab#$V^*N>+;D}H?FUdX%VV6 zvozT%HDT|%6#me&GeryHO0?I6FO06dGHZ)MTHtEMn`>r8Y_xp4K5TX8Y2!I9d7|Fi zy|PR-qU9?tuiBfJ8nrztNo}jtl)J0`e|&Bp?Q63>%y^B`>Z=VUt5vU@J+=QJSG?E0 z=!sXRM)i7pyt8&rOh;(GQNq`Q?|p8rU3BH@rL}RBu3cSmWwGeFOKGa5Pmf$YuxfpZ z#JRk!zdkG|HB}2u+pBtrOVliWQ&eO?V%V<(;%!&60#0SIvi&g^zwa0JU!%5u>-Er$ zoAy^F9`*WtK=9!G{|&LiCw5J}z4l1z_EUjtUbUp^*nY1$k!7l4Fw@erQuOq)9rpyf zrcTwGy>ZIIH%$?y%3+r@)7FP6H+{|O+Z=J#(u3*LI{^0H@ zvHcIE;_j?9kUi9YBIJ(ruS@SU6GAy>uV&!Ba-~P=)n3+h2~!idCUi})n=pC)$wMbi zwn|OX;-30GOwy10?K^={w9% z)`r#W+-`BMamMcgxpQk2!V{lM$MFTSxT{rdK(RuMhhEs>_b4lVp*!*|_X_nZ^s zqbBF#`<2zPC*OMi)crVrpGmIPx*IxSpQc3ko)ot}{^Z4{=2`P6{y*OpXm&OvI=Ad- z==I}0Q!>8Y)K_H9WqEJE`s$nCuFjX=TDAwj{->s`d>3x!F*@eeNNlXwV~_ioDy@zDeG+Cc)JylN~4iEY^-WZh1~aKl)qZ z#1jjnS?8H)tN-7AZSB7B)$AN)YZrXfl&Pp})swN^9{cM+m37g!Vh_Qm-=4CpE{#xG z(*B@1B6U*gnmUEmvm!j>SA>0^e^o23-|~a#t`CpI=f`qI%d3ag9#47Ivh4iZmLC(^ z%d}@orOs`>IwfMQoSM6~$Wd+W+CxHJhaDpF-B-#_x1ZmYs+{rrgYW7aYyX_|s{Z!d zfBUS;t+Q@)hVIR5{inaf$Jr(24OH*6}kioDox|I?M?Ex+8>c&z!c>}t=>wI`dV_}pVPh%}2> zc;(a~V$qHGknh$jKhY5CHxtLYBxbA}Rxr<)U4!B)DYI)%xmtWg&w^i4tjzvW%7y{)!i=H**`dy}|d_pW(Wx7VScyF!2cIrD1A{#l1t$bV_My!KgX zz39$a++jLla(9_Ls^8z6{pr%$4RuE^SE+_oZ1a-s`{n3wVZJ8jb=cRo&_nD!7p5hL ze~tfkTFo-=4$oov?#`L6Ml+aq^jJ1c-#kNT&g6=|2TR4Z8nyH{%~3uRy5JP!Ca<$v zsViz46rOsW2~kkeXV^6RuI|ZAX(`**eqZ@2?@_zTiRJbWoaf$FP~I(7;>(dd&$Rg3 z#cJkoGp%KB_#I1AnZHH+3sK$i>G!%fD<-a)&SKjhG%M!i&Cu!T`+^w5H}2^2xhdc9 z=#*&qyDi%5Hf)Z4)@XUI&A#Mj=!V_vLb94Q*X;_r7!uwrHDi-K!_}!DclrO#oc`Ql zeV`$0K&ffT!%XR}u2lk)irO7}SMc>SZH0vq|{C+20a3C_K~`qimr z7rt1CvBU`Pk}}x5TB_B`J<0D$a=8_w#kTXaMwW%iCH?Hz}czC|rL zxbUR9?>e{2rESkz&rReA53qDlOAENnZ62}T_U6>SODU?`#A9P_&e?pvI6=?cL+D}F zjnb(q-^C{@3+}8o)U^_MxmEe6E++@a+$GJcpI>{-YtH#d&um>zS?k)h){UzzT2H@O zB|bU!XqKDH!RfNAwD(?OuKlV|6I9%qsk3yy%Jrq+HKQ_`1ZzB-*XC`w6qC|B^XAlM zs~M8354~ZZta(Ot*RPlxpXQn_)vlT2m0$Q^*Y}m{tM44VlBVCWR?U}d^Ujqn;q#|I z%IaTzcb80i$DvOLId2PHUw!^`>AJiJUeDkE*tDj$vTRS)&(qC4oiY9q)eKM0{n*Ox z@b1jhTg(dO&+gu0R-pF$dWz_r6M3!IHDlKpiF^`#;PpOi!HwH11*}dOhm^}42;CmN zFi6N+)aI5 zg{DdUrVl>5`&6zwxRBR$OPBWNyqdM&!b~?Y9SwE4_im|V>f=+LC)XB;y2kUIxBj!l zT)Y44#^4xMk0Pbb+AN{*>2~_tc1+S!cylgX?sYQzI<_5?uAE;N7ZR?uJukL0GkEhW z56`ltd)j2R(=VM{=i*iLaQ|lK(x(q57H-u$I@S1GLa)_i{-aat3$O7^);6x09lP@1 z$_rOdg@l_)q==_iPAayK+L4u9`t=Ua`G>2jZ!e#>E^OH@r@D!;TUYKnbkSlr*HO`H zTIttiH8s~&JrU1O3qQE4W=YZN|8e5sUF-T9dbK+&Pj7tER9I@}bM>nBj!P?j@2#=$ znWPq;)aq*4Bx= z=QWq*;VIGZ4AITcEKVMpdVg2e-nXkIcUh@D(0cLa)Y)}$ug^uRYi<8E`JK{>;$2o; zY2mx($IjK-!4>p9T>Wn41>0Cz)}x`5|4m%NeKqtC+qy+h_KJP#eA87}Dsy0J`)U)V z{P{7qjjK&|ZhAZW)hf^bag|F{b{pLcom^9DY8O}_x@^a+*i!BO$s*zB<}9zcbZ*^- z?Vr9qkC;@pg?)AVcXFgNR&dHCoI=?NC zwJLaec9*g$pQil2Kbw2Ccfa-$4cEAI^_GRo$+=C?LP_K-|M%qGwksPqn{!<0@8tp8adT=OSxRUdNE;o>yO`` zil;8xyGg!i_KYJoAGjZ8y=A1$l9x|U1+U97>@>c2@5YyZpL?6`v_{9o-V0bCT=iYR zSyBF>SophNuL66&C0V^&_dY1(-}?=920wm1k8OARwxfNup5MBErIS~8pS^xhR8-94 z;_a^0x7p2K+`cU@`Mpq7Eqwa?SF6G`StI9!Fy+c|zKXAzu6=ChwT1Ovw*~Lr{QOfl zH(qc5v8zXy?-FWV?X_9+b;7pZ+R1L3Mvr1{FaCc%(b{SM7a5mNU!O(qjl7dSLwcXy zyUlTTuV0ms-{i=OeL-)VgzV~p={hC*4 zPr5hE&E(7Ci|{)(zqfjamwajHrvrzl)W7-k+ivf@s|^0^>*AK}C=eB&mtVT|a(;EY zudLtQ9~LF1aq3kE{V(%;yxF+=W81RdyLUCd(K1_U9&GvQD2qtcdbMyr|G&4lJm#|6 zUVL45-LbiK-HH1CwrlRrtNWe2;?Me>&wlauZMUBqKKZ@fgY(aPYF*Vz3%;gr`SbR{ zoa)^{I|`=f2X0F-u za@%#*r}u)_JbC5$v%D(s#|^<&g|dpJY=u9RXIE&M&W*EaFc(k#X)nK{#~o^tQT;^518d=j)fOUR}W+{x;WEjw|2wQ*X7$ ztIu5*6}pX@ODg1KRz9@+=j5?`@rplNnmf;j+_$_w&2#tB?B1VKW1n73-fp7LT>AKn z(ERPUs}BY@_4TfvyLR36jjKgx$A-Kqd+xA4IG{V&Y~hya#}2Fvf7#HgJ=d)9c4 zwEQy(% z-18`(_~4V#`AFt#%Xc{>igEj}nCLQ~7&FhA@ ztp!icI=(&gf9CpX8OtJLo&!&=_05oYma;lQ@fU}5o`FCqi)!29ss0Z*S$Ei`L^jC& z;;3ekKlmoY*|N~@p!k9W38f877v2A!99mv_KI2JBy2*ict5)Cqdbs^Qzy0I~M^4=} ze>DBqVo~PiB(IyBxNhc|9yqlq?Dv@$i$mXrWo?v_s0sR38)&A@zB;i#K8Z=VBmV0i zFUFX|`e8pf(^Ot}=^y?xhugCK=Mk=lQTw^p&MGq1G4Ar^t^D7o|MaGPDr@^M%hk>W z_qVNz-!L^i=IiEpC5AUwN|*MsWISFv>#Db}_O^5(&*;?`)^17hf4PIZQIKu(&En+Q z5)1B>hIww9ImxI~_}d=m$rp~YPAs1<^gLr~>N>I8Yq*@Ysh+Flxio9V)q}CZFfxJmad*9Y|23HuKY?@Mhe$TZTncmw^ zol4k{y25AK`RHjoPu&#@+Y`RO#Ye=>Z*OG#)!EaY$G?>Dtqxvqs8bdP2087VjGAk&JPnXpvntF^ zamS5?J9Iq7dXkQwd{cYcLHm!f?r)VlFD`wWxUMj_CZ8{?Zj$$bw6e)5(Jv2wsLc53 zmdW<_MB(Nh?@Mc3MU7h0YQCqp`p91~cwBDrE-W-_t$RlGt?;jM|5@$sey`vv75&#{ zcKQ_GquehRN&4?EC>x*AU!t_>jopLzV>!?BLf_^joenQJ^zedG*o705bCz!n;f}I3 zXPtQBmdr&{cOBm-#UES~(tF=Ahj}dPwA{*bRB8RAxmUL8O|tlSxTY~--xW`B`Kk9J zJb&z&9=iKRm*}&p(a$=aZ`UU$tLmRUyXlmGuOmkxaLy75HeT7^rvnGablsItW&B4^ZXV#DAFfBc< zGY5ta^?#*9hiu*Ipi%4C~u5qdp+kRB*)HfFW z<^>a->yBGLpF1(mQ`0Kk^Rxel|EjUi^Nx1yF{~AT z30aHy?6Zt`YZxjZsLi6J%ylrskZa{^D0i>E%S+#uLqWL zo8E6-8!ftV)uzaYm8xNr*6qI98}YL}Y0`d~wmj37m*O`^Z+v(t^6pv-P4%#~$3wS! z&57IbBQNx>Tv%hFmZYFj&_rcxjZA4r-aj7Ne-3Zb;M#J?xMcNFvESM$58nJJP`b7L zgvs`(m;61KHhRQ9wb&_DmaWz-dGf|_Z^3_R_01DkPdyao7|Z{Zt^dQ?iY@bcF7rwW zN4?ziBrs~@A){FSyIJacH`W`d|63d8A$dAGY1t$nYmTDrNnFm4thPrz_R@;H9npLA zo5kC(e}|88on>K~7$y-g?=hG4<#coFD~2|QbyuHEJI}W#y7Fa1#g3?jGyUbhzgd^P zZi{f2nwHKtNy8787lbNP?%Q{YIvcaT3;s~O?T|-BN6I~=hxbl)ib^Fs%Qo4=H}Shn zav6sok7Iq$Kd-2X?Z2bfep(sYuYKjwRjqZ#j}??a7zoCmCI9wx!%%`()G{&zl+_9gloVn0n|KS2)k_UeUM4Yg!&Bgie^6 zm@0Yr&pc};mmSG2N`WpeB}|j`HicAlY_jw|-M>00-n7kl-YMlxHp!8fOy_e&&TM>j zh}lba?W2RM))Y>a{JOCeFS6RtRqc}sI&VRMI6ZNn=a?J^iiB-!( zUxbO;HrxD2*{GKg%KW(N_>~hiY0ADwBI+Jjh{_zh#T;|%_OYN@HKw-l$+DeOBiC-4 za@Kug)6s5UzCB{vOmiblKf4|OIfME9ni(sC4t%hk9otZ@Ju@LuWHskb`LJhqxPCrQ zo*lOO>b7XtmA`gHyH0KkULCmRkKsD5wQ03mCbeB!e>~}`^uFz^@4rr#*(90Nc4d)j zn83084AqIp_deed@zZ_ttSkJP!r|A~giAin+P|^soI%gy^dg0|Zfd)nV@~pYmk8Ir zX}wm9@!d9~Xzf4SO%lo@WRES34ZiE8`O)#)7iGQe5r%t2n75zUdZ{y1WOd@!O%}Uy zP5Jv%m#U`!-Ep=t^U}kFtKmj#zSP`{E@0lhv8s7T(jw~!U!w;ZQ~lSTN(w!p61L-p zZJ}zk%+5>4R^3Zab^9dYcIx0MgSVgAl0vV%`=AxHR!m#8_m|hftn+o3)$FpidohOb zzfcjrzBX&UK6B9Da-*Y3skgF9H%H7|a=ut0y*%4gZI7hU>4U30AFM5wk6L(j`}Q?P z&+GTy&w6z`FyOGR~gqd|pvM>Xv>dsQS>Z#lJk>cq9armxEP-@kjR zaqm`f?dyDx{ny?yN$3(>`D=2RPR85E_U9jYwPdC)*}mh)T9)hC8;edqJm|GH?4#Z4 zZ!^NKp0nzjnsiv3Cwcp<1lPHFQ_nwo<$U|idMByG)-vLbe6&w23VNLMu)^L?g40o+`Bb#2UVO>2?JT<_Y}$kyDOH3T1s3>55pHH}#LAvnqV6%HI9nVVJTdg;`s^zfjfgZrAMDH-v7Bkh-?x5O4H-xx4q`XKI}1GQ1;md&RfFHSb<9SI#J7}d48?LbiZE?QE&IP?3fj>Sv_{+=^ks_%r=%e3>sl# zZyD5UHa>D|{j3?byRkUp?U9h35w07Vwmx%FH~OsYB+|X{Mf3d9)z6!kSgmwf9+ozn zd1KdJi3G#uC-%NfZ;yD*k5F9qr*rD-u+SofwP~WeM5lFgRUU83{rUP<=@Bi1JN8@h zOt}7_eRoa>wskM?-*X^9+aQ6R{bK(Ec7ZF?CC%HI~ zjd!h0yys^+e?54x^<67(N7K%n|MjMp#xiI7zU7sh>^>y2!F$uq%Q@GMulm)^7ZN6M zb-T`*OUW+xSF5gO4L^{!MoTn{dz0*$F46k#^{u|WZmUi!uQupkoj9vKeV^VMwz921 z(k4wV4Lc#iG1>fMrR_XFqs}#5Y5I4?<`=D7Bk{KVOk-GN;GYNQ6;gxm1&MVSNNZ^4 z*mZ2mni?AX=+uACby6R!tE?u59=)!n-CT8aPLzzq%RM?gQYCu?*SX82oZqp^#8l6# z+H{lDW3T7aWh7+s71kPh8>?z*`}3pzZkOJgL&^L4 z0vyUi8-h3QS@WV~en1#gQC4E=%qb0lwf!kw;=He`x zPfc4UR%v2qXlTiTO^3d?MP58D_q_A#bO#fMmoE-^W&N7>hjCtG)Yl1L4o5OFd(;k2?(CEhEtxwtwU5tiT?8Y; zop{G9L6hf&NgbcmEpX_%!`J2aU5!-PlSEUO&Q<(9!~bm7rVBAo<=!`*+c5b~aBhs# zrPyy~=2OaAPPp(F@~O{pR(rTs&ivFV4Q=)?n{+isLH96S*%>CaOpCj00zd5FEqfqv zEwc29#YU~1YRiQ>2ZB`9JojmG&fM=5n#xzB(<&O2a+BxUs>$ENSM`pU(h^|iA)x0rkd-<>ZD*xVmie9T)6828! z#@3yhYbH4^a%b4)9cFWVP5Yp5XVTV(}w6JvsPxnk)RhIJ9Ell+M z@*R&=6T29;-HNyTzjOtkNZ6HizkdDObSrlX^OpL)>nBrm zIm0dg>sH@Wxm0vS_rhyI#wLrYVg={D_|8mG^Ghx}8Rqxk!$KCJnF?MPT{cwowajET zTz4xe~p8EBPsean0{&4XSj_C@&*tY9iP3JAnS4%&}o~0ygbnZM?qKUkX z{9~;X`3fEj@Pex#Cj-y%K$|pIWtIIeTVxLaA}}!vE`+ zo&Gx)Te8eSNo@i?>`+}yR_q{blhwpCOw&T9pm8cbJpXF1(7Q8J^*c!2W$vf*S zVc&B3?b*Yex0bz>eR}eFCjb1Yd**E1v{d(p4DX>Ajk7nXM@KnqFWvGxOX=*bs}rx* zn%O6dF5Ol3>>^*TR#~5ygepnCu4Pc$G!viU;A&IzWD7n-8Ii*6V|^uwlVFtP}-aqdhZJ2-G5)OJEirq zR8o3M==)z^Ih!?)pUY!=a`D}@Kkd%NVIP^p=6dJl9y+3RcUtJKi)#$*stb3w+q&F7 zl=be@j%m!?t4-3c&MxvPYRyU#y?d+0I_c`|+%=PQk59aM;fwNWmazBXUpEV9ZeO>`BRpSZ$IWxK z<+JC?o!PcwR`i7#TKUltNB3@cRcEvMefpYhujcVCxjEZr{hG}p9|iYH`edE29ty>cWGh_B1Trls>!K~-n9fHR5>gr`z zC*EH(6`4`qN^R}={e4k?+Vp_ z`$42s;hOk|&dRyLi(mej>$hR6RZnMaO6vE+S=V25UA5S7@>O?!N^S7u!sV~m%sM+a z@5<`go1$WlXRTWPd&7j8SI&M|Iz#4(ufy(=X?qj2)2s3YHh*9<761No)k4k8+qXX@ z-TTV6r@hL4jhV@Bsns^=SCiCl$DKbL@pXIRRrAB2Dz9(ewXW}OTDBzMK9BC{_{t^sxK2a zw^?qp+TCH1Me^#4mSPk3KN0c{j=9?qy5;)bdv`Y)UH zPFz!EGHv!6p_3c-3VMfeCd+twpK-IjkZhxAuXgmn6W>p=yXXJo^!vwiCilN~nEOf2 zvmXC>o-aJN&u?bV-iym`{h6}hvawFrp6Yo#Pd{8>>0nJtJob6#Pj+d2SPIC+6nCjTJYrWNmSGRU$FN%Cne4W?xb-miDQ<>Gux{m6&SYn?yJ7)^Q%t%+j1)O z{+G~ADXF{C3_WgE6x{n6bNrh2YPD?(zy4rO_@=jif}A2}`}I>@9?cB3p-(Hcl0@RA zEyXAED6piaZc+*fYg_nybw^50)!WIJA{=+lJ1Y4ura`L3ja%GSFZfHqwjJ=m}{gW^Zs-8PfnlCQJlO>DBpM8$(bL-RTDq;Z}dF!p?^_Y z(!5za>tEMz|CFBo?QmyUMC)ZM{ayb*$9=abdXgs5;5nu2Wy0H82h2oxLhPEHB_&Ks zS=hNPjB8j@!wfRQlKCDe|Im2m(AclYC2gUaz^ui5=h43R-#6Cnx;jbrO%IEs?bSJ2 zV)AwNCB|YkPW^UM=8AqxeR8hHvCn+x^PgI6FV4pQl-uCsQC7TJ>WTE4i4m!K+q-M4 z7KWXe`}f4uYew5647W*yx6O!Ic_m1DZQG)MQ>zbTPBh8-s=e$M2gg>ctAF_Jz6tV9 zKei|0V_1#Y+C^oK6W5oo(v+9ne{z$BQrg;{H_xNhGJnrIxzlU0ond@}lBB-SMsxNp zr;qw4iM=)IkbdR4HsIppz0INVxoO=6iKVNOre0ZP5!I_$c}_SmTwv0T>D{59Iu0-X z*;2aWeOJt^stJd<4sUvUcCYInrN?Q@!}9N@B+FYLQ)Dan@BZLGi9Ho1ayS=SK6PSuMzgLFN|4iuKuJycaA3Y%hn6mYcrl1b1jIN{UDO27!&HcdHW;>;SM~}$s$fON2 zOm~mUHCno`Yxip|EDNoW+hCTVcb0pb#oTXilAfBacGD_MohlOMwoAc9_gVgxu6tp# z6=XFV&u^QNn=f|!#FN~Xs%N!Vxx&j?XWsqr<4Bm_+Jb474WT>oOt0*{nz@lne&P}T z7a{&#yR&@OH}e?(xg6PhN#^P9xHW!D#iwhB`K>AV_bQ`RFaLp`lfZ>lzm8`8 z5L{{eer?f%^|oGf-THDRulxPgc{?+LH>o8zN=Hw6Cf8DBhF)5o$a4LRayDMf{T^ zB-gY=e9a1-F7skW(C34Ost?;@EMUsZ7HEPA6XD<)`vNGo@{_L?=qpWdwqd(-f7`fj!FC$ zs)AEjUos$m|4F}9YaXkLf1EXIx#bIm<_}Z%?mA&>YieV?FS+J+ z;owS-qG?g5Up)QkwPNB_{|#I(6P}Cxj&^)E|6JC5=GxjL!RwzKYW^5upbheUn~gxz92S(|v(P{{u*S8?*E(oKvip4^!a>hi*>3SV7)weRn}J8U~8 zXMfzeMlS7phsvSJZ(m>Ee)!y~6P-_uwPm6v9`VbmI+``9S#a&AtkyhBL#^^K<~193 zS+H%;J|JtZeP_nqzB3%((`SEfUm4lk9V*!3-jf^{qAi?}nx1)KiT3Jl(R&L_wA+lA zANwtLar@>&8+)gwPmTMOldW!PnD|2@+hmn!`L!cEypl{$$!t~dpY&w@ajxu>=1ZrZ z-e3Q>DF3MQk))^FWh@q^KkA;97Fw~)a(mLzvu~zzwf;$(Ix93#+k3U+nmaS6`nj)J z6uEZJG10X?*501-5&?z2Ka$snS^3;+IK6ALlS*`yiX_ai@dE1@a z4|Sxe8lDyHiP8S|<4f4fyseuyl|($OE=~*Wncl|t-evWHdvoHY-yKYDQR{ceid6wsK@;@Hd7UUb>`KYxS!`AzxpgAv0o^BMzxsskrUU~?ukk)U$T34 zc=hTVYt$BpY22UmX=nQSNZ#dXQzIuFz8vkkagttC-@3(7JLgW_68?W};fyzo%AH8aQ)9Jn1y+&%kQ1}k%*_&5~J>I(Gk#PL} zJ)*0FH*;l84O>0AB=YSZ?Xb+vv#w|vh#y#0KX1pTYVGBbzb?3#B z-*Z}~t3O^kRd?{zD?80gHl^oNO}#D_hBS+ciP-*{7@fBD(Ltf?4!IMa@;81ZmAv*| zDXcx=5Lb4WxkSO&`FGZB{_QCB?5tP))2^$BTdy8*%h|{^J?!eqpQ}#AUeVh7)%(o+ z=)FJYy*)eg>e)U~%YaP^g4hze?489>@S!v$H1h4u%+|@bTd&$2$ckID zq;l5$lUj$mw3WkVC>oXYtD1&pbp>muHO+KiW>{DnNs0=ReY;km`>g7n|-w)bLv$t z$*P8_7fx15XqNiiNt$~0#pC>|S|Zk(^O-fpxa=>z%IIQMeiEjwa@S@_bxhmU1NUWT zhuVAIv)%rwvRgZ19H%1tb0zj=YQw_pOv?;uSImuhTUmbpPbUN=zo4d zt7qes8|ksK=0~};oy)D4?+Oji6kdHWjdO{|{FbTWVPEdNbqI64x$32*Lb3dtnW5cZ ztacamwO-W8$Xbw-7B*?KchIS`ORQEf|8wSE9pQZU3>TBL?#iQ@C9V;!nxfgVN(<+% z5pkatc~Z7TOEYAO@Ydy0>$fD$%TZ1LKf5Y4QZhbr?~#zrUX!{XT089UPB1-o$v`Q! ztff{#b)M`jl{Gq-)i+rwwW_^d_3)IJi*3hgKi*3kiK>>?3|TMkeBu@@j&hClbZA@m z;M(_-UOw^jV@jXOhOL-VFFY^VWv_1dn~&fWe(PVczf2#j*vTpt-S}aSormw%tnOO& z{n5Xc9lP{z_D1PV$~}L*JgXes?>L;NqIF6|!oIZTB^?z?b6EE9P3sI@x2IGf%tN4m)%pd+>o!h%rK&xFDRoEmEc}usu5B?uP$mUw^ExTEe7wqD!Qu;8)}7P?qPnm#o?Fimjqc&|}5d*>42aPpJG7 z@y4w9>M0S&)VrG(bWZV2b^XZIl7He<@W(5n%E{B$N1pX)ojU!kQqj6Kcg|#8QMqJwkc_L%1f z7|cEPI)?eqOnTWeWv0$!u2`N z)VFJ1&a#>MJ7sG5nXJ1T0=hS^eCTWY>f&`HevPM_rf9dU{p5dbBj?iKS8onT$8kk> z%zW*0Af=Vd`|rFp0Y5la+bB)-5i?)A$Krxu_w}Q<|D4e3nC05PdQFK~*q-MrzjV!t zbG(xuZOR+wE5p9wyXg7}M(ekqIP@m3X1C4msj>J^-TM+T)u0L=hg?O zT1z%JnJ~U|_Yy6aJrbUDJY+k=anH`<)ApGD(X};mw_`9HT@ z{`kKAYHZs%+0QdRz3h!SdjHdRBjpyOws6tl%|$5?cUdlMJITP`V87nm>cGHumY)5r;Lvh&OI*Ho3aw+C%z zzw!C&8Yg7({6D_tPEFePuYJ2W zo|+x0`8#!T<&m~C6;{R7)yL;Lz1tL}dFWK_#TdSAcQ@_2oIHEO6Xw;nw~fB<>&-pW zaJBC8qvqH6p zo*n-9@8+Q3M|tOit#0VwNDE6}Eq?Q);!3gq6BJMPect*`{=4b!=C!l(?N4i2TDm;_ zaDS@Yo>MQQzTe2LJ$_q%=7WQ2pWl66vqq*W^}~BFhV{~Ge&xNkp0&38?53BGeqA}q z^W~bL+e7=x&)+t$VGFNb#WqK#@9o*A_jOnMYg;Qu|2xLJChUmRbMb;Y|LaXZbymN4 zvf#ie?e4|emE~(bHC;U(z2R4-XrcZ71FPS1)!ZrL4%<7q`P8lzDVzGXEvem9wLjsi zKF6H(o4B;sT=}-rChPOr^hm`e2R&B3xU+=qWA=^Z{5;%Yi3eY$^j@vu3wwKQx^32O z*40wX55Cpp9XO@3SxE87w{x$~p0&$f^G8d2``6U|TN_dja)pY$T_qhI9=Y__&w^^k ze_X2$<=;?TUEZy!`{}@jij3Z?Cu`Wk%3hzA+gj-wc4}eAEARYM?f0Ba8VygCZ+h82 zQItb=W|2{ZhR@pxR}W1~G22+Ry~=Rr|A)JIRSyIkCQcP;4tQ{MX1>5l<~v?t+ib2I zYp;5FqG_tO@47V&VUvysZ#q@KjyX>E{>P(h&WY)&&(`-l^Be5K$)RpyqEpxm z|1=4HVGb+gVf?11U84W|L!0N{8yllLRkvSQ>awF;V0p4r&?HxJlQ8?!IuDZ``=tH8 zx@O(s*~|XSU-R?Aq^U})TXeKERF7=<#NZlsSlj%Ysy2Vv3AU45cXt~%o^r8Vu}=S_ zRVKj*V+s;)@p>^@O(d4>Lxon7zVFWoN6-s!Yx-d~yG@{Ox+ZOY4D z^Tn(9*HooCAGI}AOZz0dCb)8H3u*s7!nJTsltMsa>Q1G@Hw?M7*WcRbofo#`lu;Y= zk+xU9i;gbH`6bqGc+guv#QaYN@8bKGP01-ISM$`o*PJ|E=jHa_7d9__@aRU&H@#bj zibGfZerg{%G4`}`-KjMne&!a$Ma;a%`r6lldCSu&8?=nI?HfW>R`+QOY2R8W9=2n} z8aG$16MPvCPA9fD_dK0CQPz?7ql#8G$5$Tb36)_{5eqW|rde7~;hw9O`Z@oLg;--)Tz6Y&O7EcWU+65Ia_{p;@g9(xYzq$yvS?T5LWUdHF>$- z>88-rzCK}l9nG0n>_6<3{&lWwq@bL}Wz(vMEkE3}t}y9c$a6N|T$MMeL?rA)QH!W@ zI!n3&?~**Fugeqz*Y6ZP!F63I{3Ms5>D~;j)eCi>@l1Z_x|M(Ps^1$8Zru2FX}fRc zqQjR;1ZEx9m$vR?G1jha7PYD>`6?S}>YV!2M)anpY;XQmL5rhmFB3QgT;eAzeCZxG zZOseslU%kEmi-62CT?DSc}~RFlodxEH_JyZHC#6-)c9NIglDy4LZk3d+RS54oC2-lcM)?KcNQ(HnE ziWbe2(Vjo+)T=ECYky7^eBBY>ek_dX>k+Q(edks!=J{C?kr*o4>1I%WvtjDN;%T!L z0&07@PP!zzwyDj2QT6;k!`!HOtTuJ0PH^tz(&kxt&tZ{ri3ZX-=02|*UY z#~;|HOFN|KZ`gF8`HApj)q9O67PUT8aM?3sk>2u)ZmV6_O>`Q)-HP{Dn$#$XZCX~twu$TfhSJunl2T%87R6q_diBJr#K0rF z(vn)$n{dIk!viTu~b7haObnfMO>mcx#W4c5^LD*}-ITCzN-2+x6tbLZi-2R>8 zyOxYbL5_uiOy?oi>7NQL@|l;-`W`y*?qyr+#j?6(Tgvx-WxRCG=l{*ff4fd3@vi>2 zY2nv`O^dem*?An*l0T??Df^Zq>%8F2qPyn&;V5{TcX6@xrb9t9PM0-DHcUIM7wy<> z9yUYR_xyIAc+(x#pADL;mfe`LcFW_XjA8pC5;H4~)xXYgJ^%Mk*B6=16Q(9@Vg`I5lmJL$=}LLtLj9 z^nQuGxk@*zEOiq+a4Kze zc-yRL3qL)T-0pHEd7p~5jXG!8#)W%zA|tD^E{Von%Do-5^Cp+DR-J3c`V(0y$9ONX zonEk8Crmf8_1gYxTR4Bdd0csJM@qcz>sQOBo?Lz_@BFKyhgLnaoUm3TKXj#YaM#)? zt8Bhc(F~~$+u|a&Ixqcb?um0;`G-Ihew|iIA@~OV>zUMrRpM6y3hfaUR zt*cW{g{5sd#I`)5wXbYv?8?>MHBZydcU`YExtAjPbw_0h(ui5wHUwL}y z(*1j8UJs4$54(CU%Q(_sgP>C#fyFx%W{!x=YM{ag`x_Ux&1r}Wkow7fbcw;@O~ zK)Km=+Trcd^F`+ct(svFxwVpy<+#CmZIc55o2|8%>!pXSn0>f(>e{R;p6?OWBI>)XNa2KNXIJg5CHdV~um5*n?tA0rs#TW{{=M4vJ$y@?t?uXdX<^|v zy-FV*s4|VZdGXcum-eeSOC?;jj0i034O?M2_0OfQYF?YeR|->B?~4eOY$#aPbB}4g z!9BefE!T7|9$Wj=ZPMBt_o zjj!Cbr_7aJTTmaNx#jK4tlF??yAq&dGhd?dT4wuRm`&q>`gb4}FwFsqEI(;e%i zHbm??eX&cxT)A-Ds!2~{75+-vMrYcHy?)L7ak_h``m1*aFOp*u0|Tz|YO|}~TC_>j zpmIaa#!DHKO+%MF?>e%Z@3R`;l^xq2ZeH{&dfO+duv42wE&IamEzsC)#UKz~epx2N zx^!ps`kZX{M;E;|iu&@~S}jQ_T_U-lwnNc2dYg$`^QA4yUfUME;|+_xnh-O4i#dCl zTo|A8iLe6?SepY*NJg7%Qc7LrvBz7C-*MZeqh6xg&S4K!weQ_d)2@13((8O|)y}KI zqQ}2_?R9+qYtKiYGp|E~rXS_nzven)*bOI_O}A_g%$R4s!Q;KbFW+BwUp33_TQ=L@ zQSoI@e0sV{)S#v={p*sONgJxRZ8{}za-EWLoMpvf?yac{4)C5jbZUZVto>%E<3XQu zdR{O8dg`O@>3>}8Z#O%|tO+rVdRV4^;l>Qdy{p1AOO@q!P5F1#H=lX$pO3So1WLLq z{`&vvKJ`L(+up#vUM9*;kFFo{6}4QjjOn`KExRu>O}Z@fb?f?*t`ey% z*{-khADCq4=4tmEoB!zs$$^%7}a$T)3)5YG<)Sb8PU70is!;AoZPhM zuDzuaw*BbaxVXpn>kaSd2QK5?`lB=brpsmBXj|9UH&vZ^|3q6&V=K{%cwthoQu`BU(s;?;@@i5(BJQ^|MkV^-J4(YtTXIN z#l4%aS}tt<@Tq3|f6M3p)?TVy>-R^h-&&nd=v!3M*9(@p6AtZMD;K}Iu6<70)0N^D z%R@`|MXy&^kGlGDO<3F-zZ2h~xb{iDUxUY4utnf z?z222J=N;RPRaN^yPt~NzPhvc>K>P`^RAZvbYDNkuHa{Y9ZziY|8IQzmfn`#^KjJF8UVMb^zQCPZEDHin*zICU#gvQJxc*34y(2!{diARV zPpwz~S|)n;_X1J9SNT8VUY|& zewU-^jMY&x#agxU3zqFWvYV0hqPfYPCpV>~;*I_!W}WccXmoPB`Tiq!U%z^?znJ%< z?5&JVT`w+dJGsh?>8y5h@{Ju;UzjSHZ);z1nYw3vlh^xL-;Qmb`MK@>_aAY|J;~R3 zcBLy&c<+@|o7a2J?#*AFEPbf7ZH?BZL!0hfewZPb;rG(TjV&zpd>G@?HuJd96}I^w z?F|mDwpx8*60_v%s|Act9MZoUu=9j2RePcL??|2O;}a936B*PFn$GBy)=6N`zUY2| zQEZvf@>4Qi?lX=wG~HC>=2E+(`~%eTec{W~uIK8f8POyWaP;s0IJ>(3#cj|)M>VP9zVQG?%()}|}pSS+VmKar`zeXf(M&;hNuH~KP8&f`o zM66vFr+4s1;oqo@J9d5J5dSDD@%I(e{JF8uMf>M<-j-jeb)5Bj*ZbIhwoxzoHVI6#r^hvy2_=h6-gPK4B;aY=RK^L5Lk|Y;^z05zi<~ND zandSX?Cue@6*~KsD&Dn=p3`TUrg45!QS>JFm};X-UXtN8T`6Mu)t?=*%kQ(VsakaP zO?t%4tmhM&pZy8YsVlU29em(c%&V*qt6hJ7{&2r&+1y+IFF$t7c^;`|LcR zyvlmD*2*Ov6D_n{svLw~$0$9;3B0?tNLHT~UAhq}K6Q zMWx-o4n{I}yc!Z3LZ65d^w=fu9QEmzjw%AN~6NYHglm z_u$vv=Y`uIJvn2^?fo*cv8wxU)}wjiT;>Oii(}S0xjvY+_>J!B7DeAREairlU7}JC zHeC9{P@|#ObN(Rjo(|oak~^D3WEC$O-zZ#q#MwMx|z(^X`+w68iy4}7(|~?&1*4wEwlE_hFOPa`)QZV+#P1*xF&3lt7QJMRl^eVE9z)S8pJitn7p@Cy3k#THcC#$3 zFP^c+?|jAO$8R>aT|M$8fraJNhNSH0`_jZ#?}~EW$TeHE>iG2s=n;zu|nx@Zqb)-u=ZjbYgHP=qvvMvjkc&B(olKp$aj+GLv>z-;09Oc^o>2rS} z&uWvR(oLsMmD)uYv#u^XzUtnltgQ5@nWo3`ZMk<(%#2;LV?UShnhm#l^0yscwSBLs zYQfv3uP<*r)q39EVwI%7TRwwuWlsJAeYrb~;%?+=f7nh#q zik`Rk>eG8&3bRG!j&jZV%jzB@N`;?AOUR^^qAFZ)6R%hzmB4~Sdy zD{$>Ial^G?^#iwqrvf5bn(tI_WzxgD~ zZ{g~zZPq-SESh)JbNlv9?3+#<_L{q9O@yg#@SU|Y|Hi*QQ@5^h?p&`m3BQYXevYR^HlA}$rJoG+egm*9{yQk`S-Bsb*CmD_|tXOx>fdl zh;M!AyKtNR3yj~ciTU5a_w1tA^ZWb$9}fH-+NY6dDKdTa;?T;Jw$SLc&k{=?9zB)6 z^HKTiu(O+%&slRPjrGpI!_j*_@7n)LeKm*nvY<7)o~%-sxje?#W^ckHY2)8nS5uSD z&x&tUewMYi@lKG~{MD-G|3q|$u1K1y93Gvl{c85z>kctTR(;Z0GwrHYVxZ|-lfvC6 zgE%v%`i1crNiX@b+T6BY-TGeLt@xc+rr($yx%N-drc->TQ?0v|qk3aPBX+(izWm_h z&y7(j+cXYqUEu27yCds+vBF}H)n2Q&MJ6&VpL$)nW5cH{hr+_vbh&K2wCa>h#pdmk zGKG)MD1X#z5KvaTY1hh)JF@@C?u%a^_Jd3I0@rNGH&s)7jyT#r6u(+*yP>#~v z$HZNA?c=TFAPdOtN*>8E_+-gzlW&U!-4;HN}iT2j#TCJzEwkcg8#eH?l|2(<8pYOhE{N1(x z?cGF6-P?P*xW%UamQavrX1ry-QR|eW;=E#2@kd8=v*)}zsdf0a?u{eoPBC1cn)p<7 zvDa#k)u|>PC#MEU2>;-#iufpfu>Z*Ft5c(neoX)CCq8xWFTRYMVzXtUv#NCJ44 zxqO4NYvG|;$Ih|-JAM1!p+$AVn#b4j=4ekb*nIq%uQQOV-_zW}ERzAbei`SGjPGGxP zx$@4|N40-kUe4LYzxv3fQz4(UA6RU>lqGk6y8iKcE4wPYeUFp-ik?|6x!C>dzTF|F z9Tlr$ra$ZpoGQBR73bQB?D*1on-%5>T$(Mk_Q{&|XZ@YK zV!D3V|H;ZtMyHIqzWsBz`!G}R$o|jaEa#MGp4_)CW8UAlIZG-xO}-lO<@G$?wKtrn zPLrQ`^6#n3s~(-x58MA@zUKLlRrlppbb;ssMea)Lm!6J8c*ZutH|7N;< z|D7F@?eUI=6|ITOEDy9M*Rf95>zO+Hif8f?A3LWmhMR{XvK}6if4q;2S6r`IVaF|- zB|D6KCLCTl;pm^ZSLe^2Irr}Dy-Ih(XW#E+x=*o?>gvBgUH{@z_Z2&9v`Qa%y8W>*JUA>pw=Ec)MVhLOjRZ*lLz10h3}Y8J+}eQBsI!EBX4=z+>|_ffJ29H&iC- zvqV-iJV{u!mED2w+2LDMJb!4Pj+Nj?%kP%_b+6rwR`;~u@o8GkCa?EB@3q@%F8(LS z&s(V!F-xy|Q}dR&Z(q#5a|!Fu-G89ro5&;Ad~@pEg0E)N?__*0`Tu;w!li9ixrbi= z<(;78vu@GNg4AQLe>XinyKHq<(bOk=KPUakxfBr{Q!MJT%J%N}U0q7=>fAj=bV9t= zl*i88az1qLdC@($Y->-Zr>U8c>eq^ z^))TIOx5nvRlymtC#P!9{m&>DRloOlUHHz`>&sQ=&ptYJe^F?^`-khh-xss4|GRH% zZ%x{^7}m-9#a8Z#H)cQ5GFfc2_0j{M!@B>E``@1%^R<-Ix#<3qWi>9D54fstnH^8R zQLu9Li|UO(1^f8a=1Xs>%1pTO?0sR(PNSz&?wEVW?37bGvfc8Jf-5JuQjYT`~Gdxx-bXzstHzR3wNyAY|_2j_{f}Hhx)v( zgibm+RWf$tHVZwgi?xMcYf7&EQe}U)L#kz6SZblD{B#+Mov*&_@Y=m@26O-EM_Oms zdGWbuHd@u)4wZ!(}=XI!%^uD^6%f5QWE|m>`yKc26yAs>yJH-20$9lJOFqo^O-+|O%%udXXGz4~VFvu(yIJ8j;r4e9u?qkQkB z+tXL?J-}(GYqnYYZ$yk~^tv{!%U4(FtYa&^I@_=3oc$4z@Vk;RU(M>-!@v5j^RZO@ ze)f^p;n2&oziiN3=dwQ3(l#{Q;+(0eSop#6veTjd;l~p%i*oTyo)>2IKVnW%MoQsT zvGBzGt9P|1?##b*`2JsI@$j|1dNW^WsaWr5+OzB|D3GSe+z<=*y;a5gCIJaEUFOz@F;h^Yog^j;yHI^$N~jxr%a*;xQ%NtM@(eo`2@l`a+=<+UC0tWqx?u zdWw6;yN}PEVoz4GhJT9MGiT-blcKv~Iuu^SELg2v@KozooY$?-sux_e{w`jpu=LVj z(KjcBkIV+ej?7NcjoJGSuPk_~wNg7i)AQm2f$I$ieecFQy4mi^y7lN({44RaaH+by ztJ5t6pM5>`JjL*5M6=Q~O&u2{HuJ3-JRTiv>!M1(UU}emuS;A0wDO7C&o2*N-~Tm$ z&04SBN_mDX=Y~sVuHJV<^HcL~D}VkE%S(SI{}Vj- zr{Tcrs`*k!rk{LY{HZusamU%WSO4?he=*;`UPk)=*INfZzU8-nk{|ytE4$xb#fd`^ z{p7X@p+CRhVpeE3=s0cZ*X#c?UYoD=4~yrSM_OGc-u*rJ>`z&ysk@ZuP4_8U&&wZI z-gvyK{Zv)vj*pvfRUSK5S^k#uN%6(5AJMT_!`In;DYLoxa_dVikK)tom)_2sG@qer z*6o;5>2>#f*0sfk*L|-1HZ8t8Z}rcY9}0zhehZ6;Z$BDZ`gB!X%+IxU(uJlSSAM3I z#cceyhdV$prq0;xP=4Is#O==I=HYu!oJ-o~x*_`P`*$YSg4s_8Zt2$Ezq~o7{ObKT zt8NR=l4jX`wWj=o1?RJCPreAR`y&^AHf(-x$TXk&)!Lh{E?K?%LgZD`vk&UR&zgqc zJ-by-Y{}dOw-+s4J>S1*SF=&F;okdGcfWge{oS(XN7c4h&okQm_|+#qW!1u>&`Fbb zN$vUJ|2(A&lH{;;xB@#pigdE%XK_2vF+Hd;u;m^rV!cO~?H zM9lXyC6CzFK6`lTQ0S^6@%@v;`@Oz-pO5`#wl;wIlt)n`oA$- zV^uC)jn%k)gORl{jOFOmj<-itE@qdNi5^)#T}j{N>(eiXo_$Yp>V0+hnS1CN&p#`6 zhyG>mv0Hh6d)cpve2!9Ix_7QSB^0?w>F>R1dMEW3Yu~sy$9$(#z)Xd$^Agkc8x^p+ zdvX60%CQm=n)t?^zFsbONvd9YU zn10vti^8GKSM}B{Q3+YU?$9MJrCISO4`zriSSJ(4|3u3t+4jnfLeFXYE_E&3EE?mO zQ1xNUtJQJbI}Hv@6>jD%UnCe3V|!uinirg}R^@pIv2>iO^4YYPGva#JYBu|CQE#N) z|5<9ZlgoG0(a_#kdyc(|e<~KW>SQQ;>x+QZCq6BeN_BkQbi7{t)Ab6*cE&@WcZg=& zIxTDSjjv;Fdl=>Y^B&v%aDCtKy(MqgKUox_^~9lTHQ(lSS8qN##T&{OF-7lc=;}{x zQjN>ak0<%ZwyK%weoAcp%FWm8-oHr5@#emDTiW&aFTC%?x70>X%vR}&iuBx7^HL@? zOT?P~c%G*ozHP@U^O(w*3pe%Fg}ALwY`>~&Hr;aGRm*Q~v89`}Mc0~z`p=79>h?eI z$g<6Q9Zzjt{ObS4_~TYZufrr7`VX<>Fd2kil1P&~n97y>(%fTK_muCVc`uU^i#8ms zZJAOzrJz`FiO(#B(1c}PlDl?n;EFm^&$m=EGcEW2&c||y zoC#}`e{oMM+Q;MN&m2+P>sK1m&*mqI*Zts+<~2R6^~rPNy0z;J7GFKem986hGnHHW zxN&&dcK*`a@hL~N3ks;1XF z-RQE&(}EL{R-Xvldi`qU#FzE?Ps771H=f#1l&x(d_vGj+u56u%rPIa2HMeZqbt_`3 zqWAy4n_bmBqOsep?A5~-p9#-ieR0j&vQ2TfqFlEfx_30ovT~yqi}pP~KG_AUXR+VT zE6tn1-dpm3eb4&UYeO#{FD=bcEj8T~dN(>#W@m(}l-r|np^NukZHt~OX|vOFd6fxU z-rBmNkZ@L~cSp4H*UVdU++h*p^DEc(oRH|KciwJg9KHJ1u>)(by2x+YHEZpp2wscF zqR}&*V;Axn9{#wiJooDH1I3edrd%s+J@$J=`u_=61Md5Nt`U8Ct1GmwDkOZyx($z% zuD^b@DP`NPwX^R%O3Mz-yu0~*N}AH`HB)*jg2FE*$!^_VwQJ*{i1j#fC-Ev4X2O!miI+`$T9P8^@;52h5f_Gh@t>Tw}Ie6n)A3 z*!k`^amkl26P8wXI90p-N%(kS<)ykJudEPPhK<=bB3f^H`L2pwxL(Cy!z<$FizojD zw7XXwJ`v??mSK}9y6fsr#{T^eTi0ZVU0=IyLurPJNSfqX>-FEmxWeSZ^b*p#t~&=V z*jV8&zKttM^v~>HVYmJ{a)qyPv(-9Ly81KolvO);zgKIoTkUc96xZs=ja5ov5?ae| zYWU;Pam(V9f7Ezo{MvZfeP@eYdrpdK`{ejlYbJT#ef5W7X_g$<-(~BR zWTi^J$h)Zh-o3_Sjg8L6%#^oBU&n1cv|XV{`26pH9hRvoc2EW{ceKP!Q?fa=_}JrC(Y%#<{-MB+rGAaSJ!4yzPW2Z<%phqloHeK?j7Uq zo72km`E1vcf{hmroyy*Jt913|&;3#H*MF6FpDmC7`}*~(+j%xo(f^ghR_yjEJ)-s9 zQlV0md-d&ES>Ie&89B1&*qc@;ACPm^3|n>e>Xz+BcV-;uiuXLfPX5U>`SMfO-L2#P zZ{N3)HCQ72sq{9Tl2q;f#q+X7T_sPLtoCwVV{>9rZo0xFL(xYX-o^djFBm_czqL+P z+D83qbbO`i(~BEFH%RjSd-_)_&Mto8y^z1=`M27C?XjKB@^fxqdB5D_$I{jN=P!MJ z_~yy?!mD$Z+j-aKWrl4(zI(Io-B-n@f3NY`{_D-A=)ZBPSCY5?x^rpDKHGb?T1Mah zcvjg>OZ&S0@!i%9YOjCpDKh%CwPf{qgZCR}b^O^|*Eg^K`4Z+i&L!*R7e8lzvitj+ z^Lu&UyhH+(A8PjB|FyY#hyTUT7iUDBycT7%P5si>ZP%o4MjKaJKJr(8?o#b9zb<|5 z$tz3?qh#WGOs)jH4p4uyaw6~3j>i*ry$-rBznAY{aNPyDX(#@?a1Y-(>4bYo_+}-u zHM!N3okCvA`|G)iwfF6~XDWEX-Ds;?xB0%ce|TT)G-+MSW?;F;!zC}@G*Ekfu=a#q zC*|U{=khGDFnL^c{;1Z+myQ0{E|j)@{~9B7#qo)e=ekcT+|8;YM0Kw!^MAj>wPf3k z!>_m5PRa^fc~vyN?2)$**ZatrzJLF>e_k;;G4-O0OS||%HsM;u{;b2F|IW3!6RfQ< zBUpQiddQ}@HB4+*4u!E+7dmagBiC*(S`3Qr#T!Q+&oNwOwk~|9DibS4q_` z+;#P+V)HA{89!rFS6^E5<7IkuV1D`4&;yBUb}or@6}Md@GQaia`h;VfZYJflhE|yR z?A^Y@pp~a()9Iw8H@KLuCrz!{dz#C(*=p_T7vkD4Llma?Y`bx4!>T*UC9As+bt^>N zJeqY$GR$v{O!VEVh^cQqnWJZ3b=CRbXJ>ik(!H+gLr;Eaaiz1&++L-}5&bq{>J6^= znvGS@+{~jFzEk_lc6st7vn6YD1Le6y7PIHoe!qQ`%V~FWsF~5vj~=@s=89yTx0I~U zdlkK{>e(%YR!ueg$5{^y4St{LUM3Ri6yLD*ZFysM^ID;&DTYzkdP3Lp|AwkWA;+ee zInp*8Z>?R^6T!OC$v)%V|9^79_fGDaeCKxB)5=dL^=FGtXB69#W4eB7&4hik>l0kn zgnm_Rs@9f&rFSOmhxN@94jR52o8r!^&Rn}Hf^~01W!&v+TSag27gcnYH|2(zO*l5` zx=z>&me;;E5g&JOXdew*867LL@sa-C)jG`+Pc6$neb#EbtcKz&pGz|~YXnTGxY~I=nmuD&*PkgrOXjkZipDz-o+VEYwy-P}b^#_CU9~Fnf_FNFv)>Dn} zRr?ez$h~U+A+FGAv5lcK*LKOY)kdvpyQ#HJsirh#>fDH(d8L^*y2Mu|vC3^>+CEuq zb)Z;q;47}MKX=OCT9mY|(B9Tna4Y)a+_019!w&zwfA{gJlv}H|N2Xp%D%F1fF)3@w z)T#G$Z95{@?6Zte{q%F{x#eNIqH8B@2;I~j+PmS_)tg$29o2W)ZPi+Glk2i{ZS(w_ ztK6oq_t|$VYN1=YcIo}^-)HTqSiWk;n@K5>eVll?wYb)E6g>`SvzmWTL}xDuwQZcZv>y!zq!7@UuL@~ z`v>>ST%jy!O)3ps4}a*dC{gUly0qI4ePy87qN;S9T~C(3Koe7x@-DEz&ybo;I+)9vr%nQBI~ zCa-?;DLK4v&Ax~=ZX0){UtsUEdAU9A&b3=>MEJue#@nqFH$U#M@2%PDq|?gn^PRPm z@;3ewZQAu{+TLi%ZMSCb$_`ujZ1c5SvXLJXuC2WiFZYD~{Mof(KOUAwZMJF-eej=i z(r=c01NkW4h=*~JU!T6!4Lf@1RME?WCqCx?5)4ybBezRi`%A;){F`2fyWUT`qrTzM zw7(a$HpOkev^~FI1yfzwG(kG#6hv4$-^UHj0Xm$$fXr=_NA*LPq2uyt+L z##4Dq40OY0aO`X@`Cr$0PbfCEEc(xHyQL3mAKTZxc&%1+(nPzvdClvBb=q^k&%5h3 zeRt?mgI5<9aUYkBpRKLE!{>U~3q9d?iHc`8FNnBny+S+U=gqDYAHGc0;WPT%8+XyA zmZybxOPcod*`i!MZpH_*zS!(vnkTwTH0z{l``)9c^1tT)3Awa3rhYQ>`=vL#!gmSV zb%(NLZa0#hRHTC;;T%GR`ceB8SHdf39~ zhi13edib90T`Qy`nYAq`ZN8}d!b9`AuKqQ?`SMnmrNkyY}nh=AwS#K-ZJU9wdO3>ey=s}W}p1}bJdSihv#Lsa^zfhe^&XSesxs{r z329X?3q@m9c5e(VzPXvn_Xvxbu>DSTtM2RVu2@>iBg?1%WBrdQ3BkIFB{d&? z&o1okTcPkKRV3xf#QwE6itcMYs(4hq;Z(}eSIy7PuU)#IOX|q{SE}y~U(P;c^G0;T ztxcbA#{SM+G_(I_FmKkmt;LhB7E3;xYaco-L)L3EOIq#e<9{Buip^iWUi-UD<=^L1 z?o2xQ?@;S$)A>$C(-$~TtWj}%uK3d<=*~5Ng`L?|ll6Z}1Rc8OkNpn7zLZ4;r60Z> zd+NFNht8yOb-|mJmDiUn@W}HOS@Cn9>|h>c4=Q|Vea8@?lnI+cITLy*{O%ERzK@{cz0=j(W@}w_*Cua zoqJ}5wnaV*EPOU)%~Y53)1N(>w(HSD{SWcW-m-6N(%YB1?z#R+FVB7Z4zChOTXtz} zSJ%9?kDkZG&v-H^`gfwg&4#YJduxCFs@Z;OnVWgky+@Pp^3T}0NZyIZYfYFx*O{y3 z`B9C^8?UaYJn`63-^%0XLcTq_za5@e@omkcvxWlGH|f5(czZ(o+*5uh-xn0i-Z^!? z>Q!Utw>u1e)7ZC74Bz{I#n(;Ii8rHpgg;Jr9+)sS|5V(@(CjCwuUpsN@u;i%BN-X@ zrY(lS*ruqal;cA5nnNi+&*X?6$$GnG*Qu0B)l|J#4tk3_k57EOdd5_>v#QH>EK)n1 zp77M-qG@t96k#ZM9oH*GBgyZ7rI0)nQj;?1Ow}?eH{h zfvd^dB`xV^UoE`-VW+?0T7DV1)e}u_b}!mz6qzU!(jIn)-Aw!5%9oPSQ;+AJX170h zK&xt%$*T9s7moI{UEg!_>rvt8m6xuEpI!B}^X1Hko_=e>_`(A2+~U1(;Fs2I&-68m zCwcd+FB07tYIfGEeD#l052|0@$^H32ea08}Md1Zj<{O%Vub%8RV|{ZqZ>zS=)I!!n zlb^4%+Vf~vSJKuZA%$f}UopMic`oAfv1IMzq8*DK{j1$5b+@d)c2eWiLpD?9nXQ%z z2+z41$hT%wbExu?&b4JpTZ?XFrD)$?la*O&Dih5bar4qzwX18UM68vGm|45}VcXQE z(CE;YD|U)H)?ZxnCBwYtT`BQRYU5VR1 zZQT&M&4BB!R^+y>`0lH>X0C0^oBGEjdf}m8SI_LVydL!FM&Q=o_EpFB|9`mi=BI=H z205+a_t!M}<-G~3+xF-ozl*E6_V$RkI~1pU1*Nc%%oOc2;Y%Y`r^g&??deL^{vvF; z`_3A+^S11}`s{qX!nnVSmHv>8R&AbVTdFNB8?Ra7XY*yD|9Z*i27y-_>-xV=6Iy-s z@>-9xVJ>I6qOb3&%sZ24y0G_v<@qp+r;m+QlOMdw44de)`BLgOt<6`DoRym`a9!eB z@a9k@-LTcmBcG;C?RAsY+$NQ-ePG4WtS54@=5`U18>4J3L-ZusS10MI1?`qmv%dKJ zE|=(apEd@S($uLZPt|MYjRS5`H%9LJu$B}RmwfrYo9zm;aaJO zU027|3f<* zW3SJPSUWRs>GZ?ak(C>7sr=j6rbu+N1ns zZA9+e8Gk+cFSG?q%n$Nd)iSZBROHp|I`&sw?33?(J$$o^HNN}ms>n4$JDa8kFL}d# z^QGd>R3)zi#V^D~6XJ}|ru?wpCdOr%vY!3Iwft8*fAdvO;-4Pdcv0DH(}tp)sV96~ z$~0$)7)gHV=DNCM`mR$;4n8mP_-B0K)-LJ!63&;>Yfc0h?%+nXXe`^wSE>AIg^rj zbvYwf#-p>Bq7$ZnR+)8{t9t&KiD&ckFENI--#m5EZ|%7mO!6BKy*Jnyw01$`DyzD; zf?I|8C1%L@1h`#!)^JkA$K`COcDH}x;rgeu{Wm_nwrcwF@LLNfT14!NTh?=xt8}{f zYR5Hh552=Bo*P6b-fR1R%6$Hwh@HW!3+uC3r+n09+jO|+CYST0nfFd!)(g{K9`@XK z+BKQ56T`|RAhX|w3J1M#b#>sur(yxH+P?dr;_md3ha3NzN;>zZo4b<>=b z1v+6bm}fFKSX?aF^6KiVJu{b_-t=->XMO);xAyX|)ytwv&t_HaSof%IqtRk< z?JcjIW7bq{jna14?*98haCLHQ?ER?LRQ{0FSB(yxIxY9&U6gF~H~-~fP8Zjl*t0rJ ze@@iesf7Y{GNIfo}zWT21>UDf`qTcd!U#*_m z^YYf_^%JvF6Iz2WC0*UBUq8-DQ1&hF6T-sk)NzP)nFeOe>m{`A#JDX%Ao%{%6}%j0Lpl~;Ef zJ&KMjn{jE4%r7>+tFP`GDgHVjVIi^6$(K>3eCxLje;*zYHIUmBwN@a)SMSS~#jAhi zroTMSKPC0FuwknvE-8sK)#nqe1X{+b0h@3d5nD^G9*FkHK zrLR~O{$WSc)T0c1DH|3sIf%@N=u8e ztCbP@dbz>d%Wk4}US*tb%EQ*r>mn@Ih3$Pax#d^!*;8IuS6-F+`CRxXEMUV)`s<4?p?sBfBxp5?t_iTCdl~P z&3vQ&>E!c!PDZo7R|c$JuW@zN)6?@mzV?5+`ODwG>EHe&cDrWEo#~wVOxyp}_PEAp zD<#|4hA*hhdbHsymx({mtzT=WFQ`mX+xqKj$|A6Y+Gdc1Z`kU1-%WR`SZAp(zhA{D zg>Kdtto9%1Wl;SfERPR2f5>qJmq{&A7 z@SnR;r>_)jeI>V5O6`?l@&s?6Q%?-;T(7n*^Y7PGW31kKHLffnRc^aiK}W&EEYpa^9|U$ZQ{=yQguspwxb@BKT@ zgjwjBuI{;&6((44b=9{wPy0j*Gr}H;Y>WE6)b(6AyVZ;)<@YI*=I(kTwwljMV@}NX zea|->daSc|;_QQ$mxN`ri#<%Z@nqeMlWEpBUuL8xOl5m{Ezo~`Zlzm>Z~POC`&1ZPJVx_MTtKCv@I-aAQEHTih7wtTww!4A2w>ce+~!h?D)T`3M*J+*1qrlq!RvqJYr z+&y9JH7{L1<-xKWJ;(Olcv=~H!K!Mdrp*qM)huCulUl{rOFvy0@_A>}!tEz6*~A>? znrag5`P!piUEj2|>&hCT8y`BS?%leoc2|U<@0HnEqTe^nIwAbHJ#BT}?yK=(8;*bc zf92{%_WAS7R~uaXyw1a1aDBlKuL*}gYj;d_o_{fa>o@PyFWyHTy}YX`ZhpATy^F<@ zdDk2(w`fp}Iml(a_f37)*06a=qU&EsmhIV|)VirTw0*zyOa9XTlTTe&z{Hkq6NmGB- zT<0=xifP&~`R2*1=aVvS`Mua4`Sa=1jH^@Ao{Fu0aa^(BK)~mx-Kg$qZ6&T*}D{bn9 zx&19!i+(Yz4!fHpx@uw6+S!Xk*SNj7lbI1FY$MbeU|ZlhUvK^^-$3psuj23LZ25HB zds2L#ebdqBjjhWH3mScv$vJ%$+Ag&7X4aavy-~T7|Hy8B&Oc|(9&e|9_R9^wvZk*) zW$3Www4(jwLS6WyM5@~T(gbwvuZ+bZkw69%4^%I z7hMY{OtTEy6lHMKw&T&Xy-^vKc1L^CyAFQ3vV&J=o76rb=TE^AUp-a(*OqKII{9|Z zjr53*frUra=GX8TuKB%Uyu9mp)R-3$yy2bey z*B;9GnYJ;?_D^fEv|ON8u)N{kiSMsoEeqSRK00!nlYD>n$qQ$<%_JsfYbZJXL|kv365EB0+(ee3V@uI7!tx;ONw%?bBO zqKAcJ4zhJ^_c?P>-GwLVRQHZ6N^9<|o-6t-{N>)v;O&#<)jm~Gn!S7Kg!?yADW9G2ToEm`;d%6j9RP{X5VmmGHp zo$<0Q^uWq@VcTxmIev1>$zQtu+lBQeJO_KLl#SU|Wm;>W7f)EaSvN0#=_29Pt6S79 z{?u-mmAg2v^t`xM^upltce}2y&e%NF@`KjP*(;bwcFPkY(04D z{qaX{Hy8=u<<*u=G!5Jsy7?+wSg=J^drI$y(5399rfUN1OR^^&)~(9=!E6~eyE@i) zv5M;J*8LxxeqDLBF8XYa>k^@;U!5m5h2D-{7+B(8v+Rjc;_5ZG0`m>mm_)noid@67 zl-*L|kXQHB&`WD`tY6)fT@zrxGjZSIlBA=y?oFXBQ^i(qJskGpHh+;$QFF@h9G56Hs_8ZY>_r%RC+rZ`e?7-K}n^(oJGz{2S zB6?iTxa1$R^0qD)vB|01SLHlClJ%8m^{VcLh78)RwNnlz%opAKSS$O*&7T$5Zf*`0 zTm9Ae+UiBdX4`8QyowZM)(&r*s^51t>2{>(xwe1&e zU%nsRclvLK-6`uipmC`2SQ~fP#H+tt!<5Thj_-?^zGilP>*C2=%}ulKVVvf<<=-DHxaB_ovHs_lNwJrdSHJnaKYh)kck_(HCVHqHt8;jJJ`(K@E z`&Z)oi{44o&jq>KtIs~G(d!&L)qm~X|J&x-J}R=5owTQ9+OAuEYMLb+t7>)nVxu}KheM8Q>K6pmZnmub-dA(mgzLY_ouoVAXG2dEF!S%{wP~?!$T=`?`qFhu z-!_RPh??DeyQUy4ajB$9pt=&bqx|iV{c8M9lk@M@aXXk^j$VD!@P5$m%uh8f;_vt9 z)UikH44%5~t8<;A$$ zYj&|GPUF4tdDG^iltumTj^&>^x2pKT(j6NU+_qh^(auX3;k>zb>P+7II}(qdF`pFQ zccCzWch0Yqy082{^&DuJE@*M{N%9Hz(oi3_VhM5WWpiFKPsn?-ZBgHe4WUtMjG_-- z{N%W#{B+>!><1Q#8>`|HmP&oUb1~Jv^-aQ8?f3)N^SeFu7RFqds6IQ!rGLSyws^x( z(TKAfeo2SgKGvEpeRt*>r8RaJc>D5V{`3DksGgI${zr2iCmZA zV>jjeGUGKow_L8=Hj56K_AYGOs>pEZzfZPZ4O!j;ZqB5l$6HIh?YCUXI{19>*BbL@+NJDbHl2S z%75WL(Ro#0S3XiT;7CBi)|b|55s?v=Jw|T9cUQ@W>e=mY{7SVcYS>bUTE-~E<{1#iDpy?VuuPa9@!dLI_+TfW-VhNHhj`R<-tJmp7Fu=Q%Cu=8hDXxmnGmuy@2 z$-3rvCwoIl&eXdHPjy^9y7^4wp14$b`6$h8TKC_s%}5umij`e+@9i3oHG0z)f7SZ! zIn}uMK~k$_!M*68$HOwj`AetDMYwLPs=Kl3?)IZvQ;JK!_QtN+!SIv!sRMCUtKmQYO8$1r|uzO?82q+Por{9buI%)a|U{NApq@oP?$YD*tqCbsx$U+mS5S4)zP z{$71l>wf3e?kA41veBvSL5E)0tU0xDDr2+NCa2$qAFU1e->p3+vm^7Te%P8gCH_}h zwi~?U%%1$*m$-D(Q7^Oq)BhT|^(k*ZWKdPT)nl6&S4>~TAyFBv070uJ;p5)~MT}-B zRb@5ve&^W76~Se8r)bN(IefJDYSlHi<6R6!vY=jt#jc zxqRcgFH6r)xfvsRyIf4*%d%fLt%4_BqUNdc0`H!;? zw%$3uPJx?EWl!IwOUI_Z+|(uZYnI|Jt+i8QKO|liFDzw$vy*GGw#=Frs~+h8a1v~< zl6@TG(0{0YheViH;ekh9dtLNgkH6t6GKoIPx;RYh>zPK^n7dtDN~gx&>0aHXy>#cQ z4($~g>n>zF^A@C*#k_3%eeXoyGcMiBF_Atg=8wH%?)I5n6%)4*Pn~|H^C)Ne>#o)N zcdSaB6Jt4}wvqesBd(>2`VVK#{H856H%9fU@7+hQ`YOWpvmN<9f3rV&zBGYX_S&D6 zqN#CZ0;@vSIR5N@b}ICkcX4R-j#u}tINm%u^|9&|>4K}fQupsN->~8R+me5~R$Pr& z<6Y>uf9=ZE`>&VOH5JHg@vPhO)impq-V0OPMa}D;i8XmTS{>P`wC-K^+}2qAwch`J z{JwJR>yw)wCakEq6t#HCik)GZbhi5mubaQCL>~5e3WYT(aL@uB3ik^1`dVCF!v9m{ z(D4)2>$+ywsoR(DxhY%u)7^Gb0dL9G+*@V8_U_=?tvo&MVdIXKt1rer&obG)eKz}2 z-{bGM*E%FFymKM6@Z$Amk>>xivv2zDfBfu~>+hAz?KYbiZm||_?cW?7EmJ(Te^*qR z&N?&M*WZeZOkeS<CFARZ}qNzyzW@# zsRdUS_V1igpIvh;?$+AiXTDz<^7%^Z4zlde%#F#_&z*fdRyV6}Cs%t+u6>cQ=(;-# zkI$*=wW@oewR5ij9;=tv5BL9d-Bf!rdEKR}+1%CZc4WO>HhZf&`~vmTz3`7z^s*;4*5^(#N9E}wq(+xEZS^X0?2tJm4h zsW@@;o+8eGDD=U7`*x4R)wI-wT zXlSm!<)e`?pr|eVP9(%ckUjQ25rpDk_dFu`^xb?R*^F!e2OdSnldNo8a5E-d-X7 z-_q9kjViW_HYfZFzQdU~!{Ah~&SS0G;84`oBz7ZT_pUv z@XW_ky6A{A)2^6BYFzYh8q5My>R`B)~$y6 zS8GmNedSztWtG3x1<}73#bXaDEY@}6{5BpS9@c(+`JGPUh-~|pITL^ndG0% zNqbaxE#f~cb0UFjRd%9hyuj_Ip5IfGx?1dad6hj-k(KDmI2e7t#w+}e*>V|=aPf%8 z4H~tXGYSOf@Z6~@YTfO)AwOR7U-=8m|N9OXKRDg`=ho&muEP2gM7$TTj{MBc`d+Ta z%(->-M*nRyO730N+jS{sp?>y}*}cyVX5<(hyd2v9PcE+1hx&SpJT4!g2WUgqh_D%sX^$x(7wl8jmSi*;5CZ0usz%WiT6Kgf%j=~>Wy zR`dP5&wa7EJGVEhw`upw{o$*;X4v0w^Xjf+#=$SWmVe&2Zt?ust^4M!`^6Q$#cIoy zri{Z8lX>0?)|51dzv)`msaw0{uSiWB`BzN>0c)TAdX6Usux z!n*=*#V8)x@gi%L?y@ymQC)dUvvbxNotM195xuhIZGPF&kX*U6!uK<7hKApN*1LM@ zy7iaetoo{WFUID&!3(aY$BVwM-+b_t6_14+&r5{XyamSp6wE^KO2AL>b!Eu%qYLT`SAQJKV`yq{alqw^_Yt<)WT2XiLEZa zwyUdCnpqx;$3E0G zU$^M+C$sPlhaE+GUcIt9o1_+g^V?V7s2Ig^w-;#~w~mJ9tdp2SS&yLG43Zo#U^3){I)TvUqLdwAK)UDMVVuD^Da$7gl_>CpYJPlZfl z*}d-48!k(OGwRdwPK9)@mf!K9TRXVEHa9r6uytqFw8CEbGd{A3h6}#jEInGVR9j?f&!gi98N*{T-rPMa`pY2c z{?xU1*uTYSE}f;l{kfNExP;i-u5}yQD!a5TB2BYPu0AQQO%GqTbJL!aQkx@t9_7q_ z^Lp*-uTQ!}1^AgJYpXq4_pc{O)Nrcb&5L{t{Wn6RGvtb#S1YdD<@G;(|C_=m7Uv9N zo{9zX@rVjDa&@fLwg_>^Io-hj;K)qI%jOA54hJh%U+0o|#j&p`|ACN*Mo5ChxidmC z7P^dLH`->$*404TWa1M@5gH>0P~R2iK;y zuU_{SC-2&Itdf_1(X(#tJv?m71;XDL78uG-R=L$zSZEr#>(G2>>76^I1oFeI6xuUh zTgxrfK3~SPym_@cpZuyRac}p&f3>@Do!nn7ndc5)_E(3WX-q6_oi-!>?uDy=maX=U zF*FemU%At%GAia_qWHIU6;8`63Pn%9;<}!2L{w?!W3hQA;TvMk=sf(9UazaTrCc=Y z<++=pa(%I>pJywF*O+fT>1K1I=*cPBokNcpVd_cM-8UHkan)!^fK75A>M z_Ki7uedDH`GtMmDzAGss?tX!3Yu~zpOIu1+ML#^~s$YBP>equ)Csu{tD@lEk)v}r? zyyEoX-x}X9Tki6@@yBYh)pUQe&sV2+-u+O>SJQg5XL^MD>7Y%8p|7tN@|}4$T{?Wu zdbRz#q_(YF!ju=o>%LB5VS3Ed-qp_^Wf>h`H|gW#jp?tnE?;fQpR;{k(4@Qy`S2H~ zjwC#Be{pKN*xtlHuqQ8%`?wx_r6lUy!T~RSxM?qQKtUA zuDc||w=Ip_RU{qGvugD-E&X+E?6>dkZe3?^BPUjooliPkVuDyW^WhiTQ|imEUtgnT z9{cpUba?z{iP*%)!Tr4LtNCSb7u8%ia&+pHzn9nTYVr^ZZ>>GovFqE9SIe}Q-JD$* zTA-(48y{YKaLS~x@SmFd^#1H@UY+;8%hqTS_t~jOLf_5S-jX1m!LV)Jy7p&W-W%8a zQ;psEv-zrMUT^6>rlrbvr_wR^ojOg@~M|XEWs$mV9VOG09!_#8%!CL2!&!2Hg z&%Y}WWWXMtwr%D4)vGW1_N>;HIFnf7Tx@C{o>(Z_y@RV;n`Plr`(2-6%*s2<_dVx3 zBwF`grB_KyuSPVyb7tvlE`9syb{o=rrk_?0zyIOU^0PInUN3C#7LO1d7M7n}Y~9m2QufX|B6>gJ_k<+AWqiMPBnj|sy&yU*Y5%h3)ro~q zZ^ZE#Z7~b~*rx3qe!b-A*8*0?ojZ#jz1p?&kmRq_eTSazTanh&y8XbrKQW3P7kt0o zys>w+g~armsoQrXCEG-OdD_1_iQPVCL8M%4=DL_~Ir2jPcrR|*;t^i9&H2ouQ@xYi zgulVZSJfyFkCDQS^08#9kt?m8p*V&mQKZ^F~wB=oPF;INx% z<)+ylb4^<2U3{Z+`hmdZ*10#MS(e-lkd!H#^5OHNCHWI?KGHG`pJ2V~8CR(3v#X&x z@?L?dF{wKF>;BZZtxG#x^(HJ_HSOEJ>hKqhhf|IytVvSZYShxDeN5)Jh0gJhVTWF5 zeR^zBrt7_XqJDZ|s@+2^<5{vLeLTm#jW3ESi6lqJ?vTl{oL3rL{B%iN>qltb4gCz2{Mw2G4c5(?><47F*if_;=4#)$4B7o|@R7 zs=hxSS?2`D$6x7*-I%n2Pvh2|$ec>4hhqJ5=5|?9$ECKPZ{*l?;KH71OZzfSPUc;6 zWZJRw)1y^g+9d`T)Yh%ow)2sss?ozNo*x0ltaB~eqSfq;?5vmE=iYT_bHkOclGEGF z9_fGZYYaV}_#mb-)G~8jfy-0x!{6qKeXM!0Z>~}duigo!oOdmI+xIE0|SgvBCQ(CCHOfdZL!Ksr~r^*H-t~4n*ny&ls z6K~759Z7%YZC<_oa@s|)NhT+Qn$JkC)-UC|ulQ?%dGr5OMOLjUA_6m}GAbU(Fgjmu zZ)$60)8f9UR>e%rX_t_v1j`4T3!zV+DX%kISHqpLD=YTRqpZvZrNgGC;nO32*6ps+ z-tqeXwRxiF&n>LuUwmej+CI0MTOVI>mMp5Pd{cO2*)6S;r3(x%cdpK;iGft?%PDSQg95$k1t zboc(}={4sYR{#3`?aSBvr-`?|-5!)iVt2vmAe?PBHOOD{l{9I!DR&3#-QVw2Bwp0eQW(AksT5++OO+)&BN7duw|? z{dc_P_uJk1BYzF!skNF97p`~Mse9QuzG0C@7Y)#<>Vx)=*HDmuW8dFi zX6Nq;*BAdW=w5y0y5XAlwaxP*@6}DVl$lq*(N6M^^y?|R(@RCo*4fl`89qGnX}x-V z#lx(r(l|M24 z)ZIz;zy7ZOIN{F3)d%PKM7EA0$|$&mSGW>Hji)_1N61?pi(>zw2gC`yyjY z&126l^ZR!8>IGqb{cAgn41bnabe>{j)1LHZqIQSXMT0Z%UL4)DZO*FPQ?kXUre@pc zmu@{G%2aS=*cv;pA9H1^mzMIc`FhG_htuqPSO3T`npW*; zUv>3*N8KE)3X87EE&Wq=bZVED+OH}6su^?kSXlIB;~Sal+#Wxlzwr2JgP&)_oBxLg zJbT@JP(G$`?XF$yVd3Y08TK4IaFxM$YfIGs(C)&`hH`=8A50{}&ji~ie)y%_zTT95 z(}m3ZLVg`f1AA^(J0F=qUtcZWajNb` z@D|put4`d0$;=%tdqrW94a-H*9=&yQcI^r`ial^Dy^d$+dBxb$H;l7aE6%$4=1q=z zVQH$`=FdMTc0TL8x#N}mhhO*AH??^9TAR0t2uAzM#%?VzEm}&11u=7{N5`74 z{%*k?79J50x;ierUwZxL*PRBOffvN~RxT<`S8d4-+h!tgq*PVm*4lG|br(997=POo z;hRzx=5i!iAuQxbSlu?MT~{R!@mp=0x;pH|re-E#8S%!swO zesSeiZJm`88NIMVNb2k4S_OHr=fD2=Yi{s| zY!qF#%Asq%eL`2LLvF0=yv=SsZAZI`uCRw$oS6*{UZvi~fS=rvYXUU*!v2MB-E=jp zxmm0^ZtJA8UShZAwf9|r+HWiwmK>00sxohx#^P%6n`>`ytzMUu+C9~B6H~~WuzmrZ zf|DlN{qoxdb6Ran-iNIH(4-L-vi8qC2@P*vw#s+!KZ+a*OWwHYmUp+McY$xJ$%~XV ziu2ZAns6g)MuxGaG3?YaLb^T`RW;3;=aw9kYcVSW^lj*DM9%YYxrg^RZ0^}lDUwoZo zxRiJKuDKfZCYBxjmOd-&=H7pLz&$&^@$lZ%|MsVkfBW8?DQzyF|H>_2w+cU)7s9dM ze$H>_FExHlFXya(y={5=e(TrmFL#>vJ;{06apdF2CBF6jU0VJ{yAK!4dH7+@`|sOq|K6FuAzfN&*4j6k+B-K_Ntgb#mYrL=cYAK;p67WF z_chyoPj8y~Ge*0zeX8}2dApvM)!E$md%wy5*5r&{r;6xuzbM}AYcE~2*q%PMBW0hp zpZuG;`d@42O=e_WbL;Wedb`4=8EcoF-_YgaKKbUunZFXg|N0VXt1hf!zklu3PGP@D zSDx9wZ&p8XyrXw^Z$#;a>a7nZe`A^Y_ixyj%+<5+E(<^IS}w4fuWa|#Neo&m-ZH(~ zGWF+@t8oEK?!IblI;OLV^Jwa)=h3axAIDxVR*NqF{ zCbdnK|5X3Sc-J0{#qhVtZ}&~w9yKY5 z#q;>3y@jz0UtVjRqTai7(y@QW+cu@_QQoEU?DW5&=i8U)yyBfCez^483kR>W4_;PJ zHppA|#Cr8@U#6{jwUah*J^oW2fBHu8oJ}TP7Z%z?FI+K2nNz#l(>G^2KRIdF$^ZCjz^#4k&f0!BePh0NJ zDG869rMkQuCgrHAS4_wY>$<^qOe9-`o zU+Di+T7OY7{lP3DS>iYhpYJ z_Z>f6COzj=n_2$afS73;UpF{gz*tI~z3z8fCBvn^Tex-x8M zRMar?YO53Ou&l&xOsxZ3RH@3oKC`$x>YnRPTwdHU74 zVSQ~Ln{ou!6m9>t=)p>dygH#ZZ_B5nUD5q>r>-+sloI-=z@dG0?LL$F zTb#Omm#;O5U%PQ@W>|qlcc{fW?JEx%*2hSlwOxIWU5`6zYsDe1>VKQAUM)G5wPx?p z3HCm@`-&cIid@^bMyZq0XPF+?tf{ZUyqoQ2M;k?753jV2|M=_GrmLdSy}!d2MqT|H zdM_%ncd>(h?4N~GD{LS5D)YDHwB9@P=V7Tl^K>@ajBCb*E{y13t#wB?J%J*NB4-M_t^E$g{<8$ z!Sra?s?cTAt+n+Z3*6|^OV{pn6yl%VvoH6VpB7B{aNqXMrbUU@Jv>6@TW#<u^l#?80r58JspG7G9%So!oKI%IMdTz~Y4seO`itGBD4rbK)8 z;gae9waU()d)KpMbDHVu=#TF2-*5-WHrN*FeC78PcGO(MvrKQzDwW75!6obx$2|n{ zC%)l~|I$5icVXSnN%ha>H>f?hB%Nq^v3Mp+R)es>YUzOB{DUSj23v2bH^?pc$9o{c zY2VJA>zpEA8vgyA68&W1FpeyIH|BYO73eZiGX>t7oO7(Wp=Z@>R3 z$M2s`?EC9G>)vhDlH0oG_2)|~=G)(s(Ob9q*PAO5bMMRO`K_tGp1w3}{kN>(jW_3g zbKGUO!qNVFB(MLSmGVnpGiR;Q%jffV`FF(s(pSY?KK|?HUVXkiZ*|73&O@H_7r#DS z#jd^V;gbG~U%9{hl6{d*5yJygr%#19DvcR#LiYfTtk4P7c`_j?W3gm-DjzU`fK^h(dyx_~gxJoP&^(ieDzPtJd7 z<zP?p(R-U%Xx*CJ(gk2vxq zh%)(lIi1)7+~Uo4nN-T^UbiIm}RH znX*Z%EUbLPd97V3-y;>LHhD3v)>;$r&gA36O%rQwM9S)JTJc*mB$d@UAvhw^vQB(; z0N)0$zJQrReE*J5zHhgt#BHC2P9V?m_s4EU%4$k5Z8){B!=8P$XJl*B<~PwvchXy{ z()c!T334}tUU_A{FlpzTkGyL*!vEbl6@UGBmQs7}8N2Kl$t%-DP94%RIlS>a*Wc@= zVMfvykG6S?thrx8_u#;>hN0Up!+&yP0Bu-W#cY!)PbTPE9-fs^v^yYd_Qdn1?2o>!+}M?NO)2b6?B1t;FD$vL z(cUd2(xkFw&xRt4?u=PEixjuTbzYHAxOLeirHE6~B&ehNpm*Yf&w*aSVPT@Y4Ldv& zr&e(8U)LX2x+ZafDSNd3nyyV+L2(INTBm;TV%-}Qa$h$rF4A?=jECzZax3m|tX{Hp zT^q;u{)#oHw0>XSe{b)lX`u)5PZk_qeo|BW=qK~2h=W=|3NLdb|LonQv?-(V?kTNL zrC~vb54>7^?p4{6HA`l&{N&}D%&h%J=gpp}Ya|YyQe}>QKdC=1;^6B-Ib+*p>C5>1 zbE3{zH1;06XtSh;WA!hGNZl{H?Bb%joHB>pP8&P?KV}8vyk(9$reMG0< z{k*WR;C-5=%Oj(-V#E6hrhnzWxi}o%U3vJoa-!-yr^wVPi%-UygiX48M5H0~gkQC) z_|A<&OKvL096Nb!s^}r^HJ1)~o#a@3De~^5mZ@7jPh~lI-rCfZZNRNPMe1S5owlu$ zG%vk5n5B^?cXfJ9^43WcE1!Nml(l2F`5t?ru*s1Tic@N~N|%HbX&zj#KPJ-G&*tue zHLH})GM?3q_;^ZdTZ`rG*K0m~KKH6jWldLXgr{myn6%~0D+|oS5}Iq$q_thTc=XL#dMChig^Yqt5@KlMN?_1q_^+ts1rcR3)_ox1zvsl2eTXGhkTy;oVIkrOF;gL6Za^s)(=(&5%?8omf<@2ESIJ2hT%_q{Z7|a zZvS#Ko6om-ReR)4*|4nN(yKRa+*JOd>buvCgPw~#A`V_Z$gpP3gWI2zjVjm29h`Fh zFyHE{n-~Ars{e9|>nqpsH-YQxnmyAGajhVFSEf%- zxfU_=^!{h<4~yEbUYFYu7WPPexqHIbwSr;q*4_KKk7Lb|P472t3SYBNwmedI-rugN zQ%yMC;=+EZNBG89%S4BzwTrI1cl7sNbx+O0I}wad$)Zz38MJdgozkjnxT*mPkji-n zR+WTpSKH7vPdwl7l;B#KUH4DDdNX&8K)Zd@m9<3n_WtAGUm=AxJ*j$EssM@+rG>e|o}XOnCHcqbpBM`}*)TT_T56USDIJS$a8AcJ0Y?KaO9Ya&&>% z)|#zcX6H)U=9{jYbe>yVL^ZWlL8dD`+r5U z{nXrCeYkaXdV|ez^E{R|or-{a6Mjya+9dZSV$u06h8wgRZbjC5?Pc4#Jbd}28`JXp zehN9fzR)p?llj*2y$K&!w9}p*PWgCNNxVw-^{(Kv{E@5V?BZIp#kX7x+BZGR;@-Sp zCwK9AMeg;Vu3pjce`EZ2t4Vjl zzva@Otf0g9lrPwr7f)XQQ{qy2`mBDX()WxRCMeqqLA=fGc7MLE-rK&QGR-K^BKF4T zzhb`^UfOo}qRm8?C(|zfkBhpx!RFfsrOW4Sx34dd`I!GeczN3DC9F(89|~$KX@~9I zRbVCblWm{d$1h!Js}E#F_}#bNGRq}sP179yq^De#cRpopt-KUu@@GY;@X4m|Jvy;h z)*AgVvV11AO=d;g<@f*pv`n26GDUvfeD{rB54>maTmJTSU*$P5_3gwv8)tPd;}KqC z_xID*!#ib_S6%4euk5Gq^WM}Xnl0?d7n`8)n;#`swNIWB$S42LWuDTrL!N85oxdEW z>9Ax$uJ~$&t<#v6JW}xe{bIY^o2l#pmU_$ipF7y-CS6@KXYqj##V`K%y?FTdTJzpd zC9BJ!+2!?QmdUfSL+i|wZtvVOBO+RuDU((e4KkJnD?_022d4hni`~EPkS#>}6%5}*I-XE9MBUHDU z9?}Z4|75GJA6B|iYS-1kwPt*4dLm@2VrvsNX7R5rnCiUo?>CcZ-mI;K38_n$@x`r~ z;l17J9Baw8LxQg*I+)60mPJc$y=AeXC_~~?mh0WD9a|2iWhVssY+9RS(wbD7eKqUx z0j;=*rB_!u&w1t5=gOC)+G|_qxOQ9Lo5e?7xvjad`~KEfw$&e9`%1TZ#jVMk6O}v1 zZspuHw~|V2qJ1}H8L@B6Yn^O(CaiYWm7M8Uxr!5JE#JL#U6ak*u&e%;q77;vRZLl& zv*pwEFSp*By?TA-?k=q_+xP8{dUhZ$HPBw<=~a6RyGLvrO@79ozq>GPw@%o~=$kn| zi}+H`)cUSD*3d0~@%HIn$F<*1%4$#ks}h#CbtFxBchpc|992TTzUc2|y z!swrc_u@}}3fUdHziH~kX?vqZx8Abqz8bMRbk^ECC$)+u*6BaiI&H@jX5n6MvFhrx zmtj9HpSHPiebrU1FUuJ-HfZ_XYgzvC`;k|N^W*DxJ>Ie2Fv*nH=Pp-q!rL&-XBXz5 z`l22d*5kF^Y3;*?wt(;9(#xX*rH^Rcdei;ojb5vW(Y!_aOB#xt>XD%>fANe_-5~2t!l1MO9E6imtVWCxh5^?>+$`IK_0lg*3BsCt8{l;;kow+IZQu6j)JiXye-?rHkuH3k} zYC`bpxlCOgi{8)tQ~v*%rh#lypU!6GSqB)5D^IQq+xTEx)Z01yr$5?9%vbupHvD}5 z9rNfnOEXojNXTv5@%M*G|FY=6`}f?>5}EbF_<~3pUnS zUq3!MzL$N$MJxs1=g&)b?sZ>KnYSwSpZfpTHH(kzKA)LT7ya1$%zdN!<8RdevB&>w zN?RRr_|Lri>!t_qNDu8`+OH|N#Qj;Ldi!Ug`}h98z;60tXRcicAjH8N|M;-o8I zg{@CW-6p%#O)2bIPN@9GP2HhFtCxlCR^_+POg$I2B+UAmWZ2;>(dg~Fwnnb~=COU2 z-kNu|`f?8M(xG$-{&3Hn!9FI)Z3YhIcoeiCTQ+b^jl`Ms$E(4y#CszsJF9{8?UB|2(|ta z@5-EzeLrr}eDz~_rr#S?r-(0X(b{{|Jm$g_k?S9XGEAdW4#>x^ar-p&ME{0UPoySR zpLt~*F_lC6w9Ci#iwk;L-mbg7>S~wvoN*tEJ9`-90tE`Dsakc-XU) zsUIFz^1TT&Y=7i#_wKjsI@PZ^rqeFG&e|B&#Ts@vIZX9jm}uDh?|=EOPs*D8GVH~R zFzxGWrzJIt&RuiQ_ubv3uZ}M)+#`<0?mzf-3D1m2Jg4uiie9*VN8z0JS40@)XI+}ET%9>}_f-4HO1o&;gIl9aC;b!%^NjA!y)<1pr}};L+}BzK zT1t`$A($Y59r(MpBcYni&FKrQzg6Cgh%ZB_J{5IWRZ^(|4x1WK1};|b@%bL zZo8v)W^UEW+UhB!A2#Dl+G?iNB|EIy<(rEQbsI$YX3PJ$ zaLQ(jSK->#`ISA@b*-yU7H-U%=5M?E;l|i^>p!)I9%9)UVHzPCA)Bgn_0Ok94h^-- zXH0a$wmWaPx*p1PIO(ayPV2Cs@aCvo;RsRx?IjXX+gIti&sq^}S-J2>XV~g1Yp+>u z&d}0;j1V5 z)ZRH&T(*4P+_2T*w>IZb^ww4n>n-y6V*l;6ynEjrF6JM~UJ+ITXU<+b=yXT<4I@k7 z^a35(gN?iz65UrXFq+*saNLlUeOF84(uWDbY-~pwg3sSqUR~|>IbBsEEHvP~^wec1 zUx}^w_i&@!+Bc2nVSgrw&Rr|C@~W6%|FcQ+ot5sd_34~>r&rYH%uTK8pT=uWWrZfB ze*5b$qskX&*DAt&RO-ee=_B6vZT9cJGWBV>`07t9r+#@eb6M?w{&w%zM?Z4c?|9}` z$oVJnOXJiRqW>4WG&=v+3%mHpUwL#Ox0J}*7#Qdn5M-s!Wdr^~y<)m}%q4Yz0Zd|W2E zKWfUs^E>KpB~;FeZk(F?ZlN`+Xt?HrPm6dMZy7x5;$E-xj=5h(efPy6e)))*?XSJo za2<4NIq*2bEyij8^WBOIW}Z>wzFKok>u^#VcaZGqj^c#MQJu9*O?Q7; zW^3J(aCJ4u`deGRSRWPNe>I6;F5>C&wQ65%8U?g5Bz&?>+*^)UhOLDQtg`tvEO#D`BTv-x-izfs7Fgf_U5H^ zi|@=l$)j!3dE-@_#fb&1nhWpG-thVCro=Gg_Pt{D8&s5-PIAWY)!cVe=GV02j~wr2 zy$t*5wdTXIyw<;gYhG0xIC?j%e$9bXcOrDRuVsq8yQyk-)K^}k$k^sR<|W) z&7LVdYaSJC>w0;rR9pRhbk1UXf19p5=;M@Pvde9-o)~z)9k10 z_4*|X(w~D8ZdL5UmkZJ@vP-q&ofe9>b*{8uabbeU_Vbtcr$>i0oJvZKT|6h9czw!51 zr~Y4_|Nd&;rM3SURzE*{YE9VDc^~FYj>?@@9rp2AT&3BXOUuJrtIikx&Dt7uC2O`# zo+}D zvy5IiQ#$P7pR0#o9c9s05A&_!*53UvOZ96SOT=2U(??I~tq$EC@wDNdOxULDg_{j# zuKz!4&z`oak2%98e-oR%G&<65>*lHJUx(WlbGhYh`10&9m%7%EHDRYi3RAb<(h^_2 z(l;qIW@819VzJM}8(T6X*Ivk-k+^liy(8;(nLQ1-nl*dQ_XDkQxARgrYcIUdDe_$V z@&Z-oJtsS6-8qtf#w_~d%OjSy>fHfBe9fK2Ui?g+_Q^uMNivRgPtPYtgZ2JbK_5Z>o~T4rr{Tz zGKKUG8a-~4v8^~q zZEaY=ryWiQ|IK)B_uM4j`uo}dy|ZHTYq^YEL~lOY(do1GX881i{AXX9w#BY|=D3<` z;+x`xo8K1o>?qR?kNsG{qj|3IUe#Ln^R?RAVP~HP*qq)b&g~X+WR=6(mRH4RCd_(Y z8}d2y^4e2tza?!HcTLPAA-@+^-y%LVUS6Bl_5Szu^Q%KM%C@db+G=rb zwdk|iQE!DeP71TB&Dwe+`(0RT(sLj4u%5sEYs)$uJhbN}{k!7z@7RVv1=qRu`&TRt zo9(Vp@TTe8ux$OW0TK$f^?V2-T2V8(#5mn|Y(-YVhXAoA_tuwH^=toA&qB!^HeZ z-@~gcH@@Okc1xP-xz?;NLUf96b<%Mc)6`dCk!!?muWkAwbXRL#*t!XK3vagfL@b=~ zrs&_R)w$`bHOv!C_kMZ0Ei_S3QZtjb#^6c8>LW)Up-WYyeo?MIgHw!Ua~ zTeHOXz4g6WS64+Yx(>STPKAA=ZQ?Vzf8*T%6Q4C%dzH= zk?I-#)iQFw-A&_F55Lk~^CvY=Y;xk%yfrd2H(%|1z1ef^w!Vne{dc~+`7`0F$(kou zdqX*Y>-^mCGVJ00H`DF=WOjH_ z&_Nciw`U#}Y&^Iu=0Fq2?vO9lDWU&md#e`C%JlsFyX95VM}O1h_O_o2uPDz9GraN2 z%f0N=Kjz)9p6%c=wa-~66k}^QNui@>;`TLO+fqeUz0=k294Qo44_7K{S^fE93~!af zRjo~$b7r6J^olw9D$Ana>fx)F6HEiPJh3lZn^crKOFM5%=i)VgX4b^&+>9|>xWs!$ zRY_D#>fAGLy!wltnI7D@RjF>$>73lq)6Mf@ET6pT@7s|iYjta1jQ7z6?o)YxZ#a3P z_)F5dos+g2?mS{Besrq+`6=^tu1@8&+I6qH&!{AFv!Jx^mQOl`Q_j56^6u`{t61fH zP&p`8uuw}dCYIH`PNHM+s@H36iqjT0&5e1PF^Ml{_4iNjb2it_>R25Td-T~V0m*Mq zcU(HDyjyA0+0bV})3$MI{Vle-K4-qxrk5Yium2_V_RXnJPj_D0&X>GZZQYSu0k^Fz zE9d53@)s#IoiOe3+O@w$y$hDAe>5szacxQ9nvYMP8i=j5zP)6=#`~wuZ+5?)bk}!n z=NpX;?I&G7Eu9`Ds`+n6oyuA6<#v~+%s1P)nGg4L?36!f z9$L7%^wU?nSwU>o#y3{}Q``OZ0iRns%Twi|Ctkkm*p#nlJ5N zcx?)%z75&$mAAKY#>0c{p$`uJh>3Z-E2w>)o7@`tilzkTlBe&hMPJ8;fBU!N)RVYX z?DG_E2>eYv+qb&#_*LHHS0C0*-DRY(v%=(E!P8g1F`jOT&#TVq_OTVqTdiT?KPtYH z>&{mDD#i*0zw)Em4-&&airj5J!!2XH^H=(YPGc)0>vbwU@5)uHLpQZ=ebC6U@7}c~ z&05SWKbiehFg-l~#CDqr#$3Vo?sjR%Ke4%5&9}^Pdf?R3iF}{9b-#8kQHOy2W0x@|M#ut5>)!{Vv+g_R?$Z;6-}-KYYzoT^b*3TR2fa$Rq6l zV~*ys9j|_dc&mQe`uNnDbw9petJxQHv+FGPuf`J2Cv0Z?|__BKnQ*n*O;GG%HPH+vgzn;7%C&3@^Tk4MaQ zS_O$uxT1gb&8bP8^X$uGY%e#jT=!+cG0k;OrLHouDv8Hm>6u2CChc6l`lR{(X(Hc? z@7$Yge_Q_h+=Kg<3*|0i0Q(~8+X?SvR=;Kx8GB)8Q7kj_xAAY6xf>*y~ zv$o#NXIc{`@#=SlmtV90VBJ*j>v^==?z-D`@1%J#XP@8jH{X3KYr>AG&D!@KSZunJ z6%v_#J$~YKPj_jbg4*dN8yvs)-T%Jm^j&XZ)6#;iBGKV?J3~vKW-~t*j=jD9#?*+_ z>n#2k?D_X|UbFA=s4FMaf_;CdOyZm-T-6tR%1UzDib96yu9H_+?b>Bo68d=6inXi# zszlBysND6aD>BWHJR7KaZAXY#c-6{2y{H{gUAEz=*QaW(Ub{LcI{3i=s#av7=qhdbwR;$K4o->wus;4s=>FHM^0O7>cbcuc@oee3bvFvC7FT|Z zaW~JB-1}%rP)urFzt;7Wy;_@`c0_UAjA1kSW8JN7_dqK!Of6h@U0CO~b-$ivJ;~9J zlPnZfnZ@-zV_jEH;i{UBf~^Kc`%W#1jydU>c|3H+qpa(p3vN!G8KZgblBjgiroO(c zGS&aw53&Nk#Vm~tFMF-E?}_NTT}iVyYu}rf;y*ifscrbKeUj_MQqPC3SXYy~V%;q% ziASfyCu?tvj1{%}tDefmac$F5gyg{qdjW94@rlG>qi@OpId)u|=n=j3*;)wbWQ<=Clxblnfp z^7V~e*$*bY*|lrW@zBh5JC486G7WFeT=!}2I)QNRP0zjF?tZoG?<>FXvVf0UWoK(w zzwnyc@%ic{nH^ox;j?15?(mA87xVXo?X1-Yd(MhJZ<}}h;47_Nj_21M^3OXWwL|0J z)!TdGTm5&udRH`c-I)cG=JLw!loANv#maGr<>%`!oP&iiU~{$?`r?O#M$)7 zx{@_ZUz{xWR6lZc(VQ1vXV+yEnHClAdvbhTirm$hvrAw6=DznhLNP|y<4&pQx7eGr zHhUl1Pbu-5w$Cx4w4K^Rl^r)~#N+>(S=i zVo|eoRhw>pNYJ&tdu*zB_`&0^PARUF>*#OYuKNF7a;3iiWA0Cj_e#cmTCBNi&F1*c z7FF+doLR@T@$f(Xl#HG@&U#}gfr#w5g5A%+CvixA{m^1aX7013hA;ahM zSS&glLrXz4Z&_*iDZYJHOo_XHg-L8+sk6RwZ(cPUW~D=($cMX5DW~es+j16gDF=p6 zvj1*!fOCWPr;Gpl1nvF5&n|oY@;eV(e8Y}S-*=yyJ!xvcf$f^xy&k!bgShhf+h$+7 z|GTO{>(Qw@@3WGVwtTYO9hH0R{q9Tbhrd=xRK7l(FR|;g+V3M1??>H#pLN+P;-^j7 z)Az-uuWDXomHyz(eEiDevEH+N$J7t3@UpoYy6sctn^SLh2C>H%G_AJSdFuPlU-tu# zjN@)o+suJJDaxVpH4Z+~UZpMsyO=3nAA-!db8&9Qd(wO{h* zZPt4Cr0IEBS~oeRI_T)pvog=o==>P_1UQmrH{H%qlX znjCh!JBpZj@Ai7jQOhW)|1iYRhj;S3mhj_cngnGIjTc=Rq+$n|re>N>?w6@vU(0 z&1yWa{BY8``=Y^99}DbSqVt_cdrEHi-jvdBU!K2AdBC|^U25&-&EKk3n)caAJ@)_F zE*<{iRhXny;KJO{fJbXZpK4hxmkAA><}Bf1G}%+8``S#sEt^uKrmXL7x_UfIu)QcX ze)hxr$zM}qV()C8m+oF%`1o?|*RNWR&YGJyYz$R7d5R(LlvnPQqwzd95~qqb2PDf% zhgqC-67CI~*YoLw9slX<-+ERu?*&$8lwWHQ$lX~o+qNiY!udltudiMoY#^`y@6!M1 z($y=z+*)&@`QsHcp?eNO+Qw@pu_&$yvo2CtJNZ%jf+?{vclAzbq+PyxNiJ4qUF)O7 zk5d_YZj^j(Vd`zyIptLK_mal)$;sPd-&ZcntG(ECK6BESh=OlIRfo4E|CQgARKK(- z?nC~zlj+BWtp0hOmy(WJ6Q_~!B2iS8qjcV^t2$w`T2x)4)*kj+Wq3=u-Op#`U9ZV< zOS$?k#YiVky{=!dOvr8j9ut{=Y~sIU`CLV!QX*{>x9xK`-sTaoi0{|5tpTCZ+M)Y< zCT%uafBJ6X5Dv}#`M$IL^{F;}^EJlb?Rn2F(@bZk|w;QlM2d#^?; zeY4`GvV+oEk>JeDhmO2XT$;Tz^-NnLYfQw!0Sm--wM{=0 zaL?h5+-IZTm0gqEe0n!H_=X!C+*9ciw>?Ug<9c4}XWQ?~cp13$Ts)&*+6xGTo!ne? zqFl=H)|$Fy9oE(nwQ6C0YwDb|jifeuo#ndUa3Y{*_0G>OPd!vlXJ&+TJk`+@S3Y}c zjZXGO%T?K%ing3$Jndd6zd9mOwd#cM+GlyJXA@idyIuRfNIdwtP(iPfH&$TIge6yd zdXC2BxVq{eeLOdHlc@HR(4+0iTW@+F;!XPc`ibb{?{VFsdosgzMNGJ-ysj(HM#pc0 z<<$3KCmN-$FA`;yZxb~yxN%#1)2E;F!cK&2o>-z9CLdre5w-M}-0I$y{D)6@3B0=b zc8$W!lOgw#WE8Kh`Jnk`&86s;%2iKEVo6f{wUD& z={SF4{qsqhVPd{Vk33+D{3s~@_0*-qqU}zi;{Dp6*QIV)6rdHirm1Oai2KK|_W!5u z{97GA=k={MDWYXlJvV+ex!ZMhz16l;XIJ^ywzFj@T?tsQ(IL-!{_1BpvXYdquQ}C{ zopYgGF)y}X+m13lmMx_ATuf)IQG}CzC0Z-_EpkQ)5QhiWFV`lMnVx zn>AO~ck?NoZD*9j7*GFMIJH@{TKV?cfK|@IH5zxFwjMgxQkgLIu=BmhLcQ$GU8NCA z^GsjKtPD7MDtt@P{m>{!GcUWe8>fq`)@*QdlL?wRBU*Ec*|u=^@Z_WawnZ*{oA|IK zAvBurNmFfc&R3~0?cRdAXa7C7?hsmiRyiy>W%b_B2T#x5JLMIVx^dI-u-335v1hd# z0~P+&WJ19?yw~WrHqW>1;=;HvUqEr9+rCR z(Pq02?G3YTy4OYaYhQo#>R!!Xw>57Xr`m3sbWbcXfAXJYX{pNl*k6b4-j^9x4_CdQ~TEDY)+cNv--}O4RH^fer{X3PFcG+r!|20*Ns)r9sW(rn|hRM zfA_~*YnLcI@0Zh@6*eu5Q~hcrzxEVopWP9MUzJ8imZ?2t^z9D*ntV4hbz|47r^V*+MZ>^n{tX-cJdSmmn%{z?lo;vMsv7sobb@f)a?en5{9=RC#lUchYGCZ^O z-YZ7q2Rj<(M*U5a3%f5L@pA9fp!``=Ut4}qS@$e>wMy9Kds^~Gr`m6uCHQGW*AA)T z1eOyUX6+IUQ|r^~*qOW9INxv)&#PYTEsqvQpFa00F`j+V?bfUOpLpb-9C*7yOZsc* zw6CwYwYRp~Zmvr0U)ylR=%32EL#;cUj@k6AJ~G*@>t1B4>P^SF(KrA9*zzwrZtpgO zOGdfRJHsl%D!*5zFAY?G{EUAs$L_m)tE2D!T3u6oFze@^BVLKp_XYO1?WrSdY~TBL_ut>}XsXPLrOQ1tR_zKsbnr#y zx3FcVp?d;eulHM9cWdpe@|~_LE@f%kJFN*jHuYWha{CRx-uTBad^U62>L&-nX3V&} zqblN#R@jM_>seQe9=M(Uvii}|n_pBqDkB&Z(nYiHM5@}ZdAIh=N3+KU%73q}yvp7F z@6`b=4aNPOfptgh*!i@}6IzWY+)7Ao3(XBXUf4SS{G&xO(KlDG)@@q#@!Z;|HB46i zTXr-gtDlrPv4|`2NM)nw?wQpajgq!XE!(FOHY4X{PQp{ZwP8;q9Nw?pq5nAgwUFQM z)p=XxqCz)%G31<`DHnFfdQHU5&$GXUHQN7pwWev;=FNKRc08GU{LY#?FTxULhHUmq z^Pg}$n2595j6zu9Tc<6Rx{9S<^YE11-q1b?wx_h!2( z+X0c=zgP24KXJeK=U$=JsSz9XmQ8nOmi%C$b(ptPyW2LnW@WnF+4Txvqt+*^Z4JAT zK2b4z>%|Q-gp|D3rd>V$I<$4_?A&=>*XWO zMyD@}tb|tcuH~7z{&bft_cLi@`KZ}V$0kW{(NgOY{a}%I#Cy%14YM*t1rugeUH$g& zMBU9NCwgZm&S=TW5T3!f=>vo04922HnFIk#j%N&!sV^;?%0DU@T>Wt9^b5Yyx>);F^SPL-Cqt!U zkNV8nwb=Tx)&^~N+l_ptQ(L3jDk>gco6f%=W_?WJ(ezszZRGSVpS>u3>E#|i_37ue z%HdC^Xgepa(nu>U5q&yITWUo`!#`;ab~DxOpB`;~vg%(`j_p)Cv2(jt%@_AGwc52v zOEO7bB=+iFZRf9DMr${*$TYS|90u6<>n;@`zy z^mYyB-*J28ByGFq!lb@+Z<=3pU2nf3aM5U5%(lOa7wuZTZrwBamq$zW<^}!#x0F-< z#r!*)1HG30;%_>?YW<7nb)NGzPg*&@pH^S}tM>iN7hNrXnqq%O$gC{vUKi7&tu^(c zsJZ3J-iH+_%1e%`Ts#?C`pT%QkgYwn&aZ>f_}P5LnhVqA?;k0@{rtN7 zmou*)I4j?u?RSyk`swf|Pc>_^PKz&@_~`SCQ^&%VB`rOxtTn@^#y38@wd>~e)0%(x zs_iVABUx#(Q_K2YpH|+~BeAO^u0Lmw@eOKk-s-Z(V`2ZBCnY zyWQee=YQFK_tMl|S09&meUsXlWM1&}%38Vo!CS8X_`7!3N%K4BuAbfZIj?B0O!8OX zeGkjlWtFZz{$4EY=cU&&NB!pQI_d15RlEAQedNCV7lY5P-6y>6n{)2AdyEmMwnyia z*Lz0N^{9E=&yKh2ZxD5T;UW-Mb@SA=oYQW9f-H6|<`QdM zT_aF0vS_m7uS?#`Yi?@9UznS>s;=(x-9}-JuYu~0weitqDJ!(4-#)&(FtdGijY4>< zzVlxn)|XmEp<7cE!b9!%zkGK<^GTOc{D!s5Y%)$v&Ad75+p_uhi;5rc6fM1ZHMq!C z({QKO9IJEr+f5A`VkBQ(VBNWkV_{MS@X z>uK`Cq^YkScWq9b`s|^W{E1M{*s26)zgc$dY~>kx547aFE=f{Yb5VYQoI=wTfcA7?%(pC*7>8^D@)`>1^ zT74!#ym#9oo>G%5yFV37ReaLqH96p*myycNsiLeqjV?}YUtLon>9l(9iYD!qX;by3 zx{QA+aY#;DUG$o_iE;nK?cGb-*X-rq#?&T}kRR_kPjN@O&yj>$Z| zP;AF3(RjYw9c}j?XgTl6d61=kp)>2jE8YDEL)%K6UVNUI|HvRF^~8zLCEDt22ajDj zVhrKG6You38+L;&f+6N@du906$VBM`Mu@3rb_-?aFzXyR zd^K-I>D7awdOwoV9M!{bvd>^&_w8^_{=$iaRFP?d-C5nY&&mv1hby+}pRGBhl@(^^RLzSHdOsTFAs~wwzHHkW)2Ph z%C#!p<26WtSpq<^?_ZyZNVC?1QXWo7q3^CVVhiGfBJH?ENLy zb=QviZK+UC3{@*+*1o@lJ7z6+Ie)#ayw%#U(C~igUmvX2O7k`MbA_*-&vWe0#VgK< zZ(Lpr#vLo)5hZFbXtJ^MefQPNv6g#Qas0ov)wlao&60;p$2K z#k2Z9pAr9f?9|;Xt57|rZBOkuy53#8`XSFY=50&Fjx6I{Z$fSV`2KI2@qgv+?CEPy zuj32fz2MLFxRA+F0aty`KhaoynDMIrT-h?eO$qEzeA;FG7RPx??)}It^2Y9)K3n)# zyS?|yMb1uky_{Dtcl96ni>n1#YQOUP@B2Sz9^Ye@@ZAfZEe{O{;gMhWxcpZ8{=B_Q zzPImhUL6}Sccz^0-aQYpw$0MAUBP?!`lUI$y&nW_ew-Cx$`im-X!vGEQVCD^?gh(| zuil+w(eiVohc8OUymb^l zd(kuWUqSagrSR?lM0ZHP$_SQLesFg6<2{CTVoznav0S@&F~YMxBKH2iS;vdb&q>6* zZIIX;bA!=;OFDdIsT`A*^jSk z?I^H&S@U*kO8Ir+2-ok6x44E@hrG;rJ2k2Nns9`Bspq--YaV&Ovr=3)yEI(gwYl77 zTl}6I6BuO^H%9JlI(YS_;KL7`!CyZ-d8Egx!+8&?s+V<(vfvY<+V{=c4mYop^?78rN)3>dEZ_S$j@(<75#u*c? zh3orxS1o=U_v7n*yDA?;{eLyfC7y=z{U zZoZOB`qgdYuT}3?Za#0m?#=qpUuJ=_`>#K(ytVgwO=xcH%SEm0I%nK?p!MNNO>A?t z;moOt-PzlA-I}F;F!u4Hqt80Ka$@E0ey=^^+C8b>y)&&}-HUs&z>Sl!VrCLGSc z`s3Z_VbZ^@CNGi~y6PRho-g~?yhqQJi$ZI5ezGg@Js;lodhw!LTW?qQc8h(iJSxtu zRvr8Gn9a}gKH>YqSAUli*G&q4|M6$j)16y(bUdxUzc_foj;)jA`r~a@FXIj~Ix;&| z>gSK{&)cisPfY#&``7+R`?#km2ea-~|4u2a2)+J1SLDVrDf`u1omcOBssBInZ0*L) zB^*^rqO503hM z^^Lau!&9HnetEF#HlJgE+Q-S|Pm>d649s?Xs+YX|>*C$7>~1Rk^P|EYuNmauj*0yE z*ydB{pV{+&F@MN?`KE;9Ug+Y0?H8v$|795TP_k2bx3tE!>37TT*8UM&w=7z*{`j>6 zJ()_=ouB2kO$eBN|CPLs;QjM0GhM!&m@2dHtzl};leOZ1-T(E?ZM3ZVRli?Key@)H zC52}7rliToBj(;`{k;EzdcjnV%O~jt;a?Co* z#Zh~5(T_C@;U|?&M=o5)_kt z)x>3-V&26dZE$oCQ~1h7;_F;09i$VOXPwn9n5xxv9z;$$ubmXgn5l5uwe|gd)*FY4 zjyJCkiM&_yP%HGotD+U+qC31qckGf`CpPtIFZUA1Z7U9hu6oklmGrxu(JZ0Mm2;gC z>pDGl-f(f|m{`7oQp=2O8V5rct?8I}TF9eOG@VlIG8$VbkiB-U9szb(35-yevLEE!#or%=bOB zWG}MbIAr53V84WSL*vcbQvsDe@()bfR{aR{m3QDEd7Hk5 zDzH9L;kZ^X)gUXW{nPAgjgK!D&N2v*xyE7PE5l;Cv3VCmbVu`oS<9RzNce7fd-7#2 zw@O>bq1Fz5{RuK%$IiQ$uV+5K_;~Gkfpy`Y$;WII1#b$ci(C<~Ucq=S zewXDVTsj&%m0wyv+^*!(edXd^i=7*tws_1fdq275u~u!?{+~M&8Xs%bvXmxp`^-Bi zYQIakplkbFttAJz9g^1QvGs2BVix~!W6LMCKT`rH2}rC~DR7?budN^Yvd-_B#*Sb2 z`(hhS|FD^_WeHywb4$PN*5mmZ|FW*{J0P@o$KyE7ytSWZ%XMGhn09>K0ja7&5!P_M z--d@Dl@^p*+!LK0(xfe}&b7mfC-%lHU#)|o!6pS$rOt)QoX>r~>WXEIYFqKrWeaV0 ztkPxZRrD^!Jv{^@AKX_FzM|*PQ+@1bQ|DHLeRV4bpx1iJ}&f{59{K~rTF9M$x zyS;yP>TU1dbyDku4&;~k2^zY-4^GpNj`du{ax*-v*i&X*-}1TI<+CTcg3_Pt~R$4gR$>q4asz9__iEi&oBA9e+5K=|@iB>O*OhcD}hcWm3Cs zfIzX-)Zgn@J_#;gaPaEKityed&%LFgrR$b(OU}|+9D=K_i(cPU{ zH7&CiY>c))1*+<_;5?{Q&r8T)hr*@t>Zab{M1jgT{32GQ8Iftr$Efh6QZYY z?c9{Y_579BX6^bOuqS3J)m`27OYh?Yt#a$+8LPEpV=7wVL_?M z-qmZ0m)^>m_|ta7hf?p>Xy$Ov2R0{?Nh zS9Idyq#5_?Clg5>KAjy)~U+zPRJX&Y6ee>WjAi`+HZv-Ns35;!erKcO-1|a@OvR zPjq~uaW-_uEbH*2{_CvvA8I-gI7{2!w0xbB)uM%~gBR~C3NC)G_4DM_FeOXgPwk;^ zuH3mQb~e;MT-|+}>CNie#oE5%%h_Tc-j7WSXO5jKc6RE{>+a#pIb%eFe@DI9_MCsR zb|`o3T(QNg^LGn``X;NSa;{xBBX;Yizp8n^)83X&x*4j!Z?b^1Sn%W>eg~(TgtN-} z2VPd^T=D3C#N?aWZ%KSwNL1eZZF;@>+?jRj zTw{3lYIqoJUw45qd|8)Cywv<1$Jg$8vN?6l&FY;(8_!O4Xn$9*?Unw-9}^^5*G1*b z{CWBE)hwTNcIT~TO6+0?kAJ;!+YX+ySM_IUhdz1y%v0PY{`{)~-3MAX@2LmRT74^a zwPUa8@9B!Q50`$sUbMFB*nBUm)g2N%+WJ3pPCnDu_~W$6^?1YGS5ep3uKuw!Uv`z& zoC6_duIk+D_15fu^jX#WbhhHo#G*omJqgB^R}|~ss?OIvJN5nR z7hF+FF}0gZMgMkR3=Pk@+f!~5o(PruCl7y6*ckM zs;3LuXWqr!U0{6jiI!H(i+`8z zzpi^LI$u9h@9wp$SNE;f-tBlwcxTqSNj~cormYKOe!e=&+_dfH)%9=Ytrke_T0L)_ zU&C7M{}--yzxZvq&MfrBD;9~^w?z+@*5qzq$K%zRS7hq5F7CqBRd+hIjbr=Qt?TWK z^T}G*RkLg6<5L%|X1&?HYhFyP`OaS}c5?YX@C_?15;fTHuIl{ExvLr1mG~aLmRBNL zm-eW2)nskwaP$#q_V$p6tUDw42-1!MphC z&S}inJF*Oy7HxjX^*GBmEj+Tc;A-rNtuZfCwyg`9dU2|**6cNt&z5YJE6xpMxgGq% zMXPVOXKw7t@WrdwEY`a8dWmxE+T(As1akZriicVoR+e0?p0nOXZ%taW?75q8dXj*%Q1Id%MmxDKDr+thyJHewTJ-c~ zmMv4<|23wWFC>p#)miu|D!J84<>+&9F)i1-3J*+bSa(Iu6;fFKPxFf7S=;p=re9h8 z;FWz=e%!S$B37ZbJ6{={bdi_)b~g4tdx!Imty%NtFto?qopjeBQ2p@ZbwaUo1!605 zOl~`cFL#>g-oE){NvYPJb5mcvwT@)3wEbBa?pp9eX}M!qKi}o6oPl2!A8z)jFy~fW z|Fl%K@_WUdiNZ>ocvR2|6H2)6m2YnUo?mvo{_nx{TZBGYP0D|-zioPRN6Xf0PIBo* zOSexyJ2kpy((I%AKk8UMIHhCjE@-`d-|eq|)AMTrywCr*{G>s<_RF*X<$XW3Pv@@N zE1bo-@Xy;j+r0h%2|tS6@hX*T&X(s{#p!q6tP|&dHDBO+?8^Ig@ixk(#e0j+Og-?l z#`eP1!gsrkw(7j&+Gur7h<~@#i&tkWUaZo4n>B$yE&SoN3$fu&;nqTO2Okz+whd=B zFXQ&_b9ft*{iS-{x2$fZaG@7j-pfC@+)0q{I}rMOb)Q`SCcpDs>+VdNoF4nT@L6s6 zZ*PN8o7mig&*mMO!pyJqlQDG0vDchh&u@lK3J~C3e@aO;(aiMbwHJpEe&GtYjm7Z(CbE>JF`(qNVBKhu3QF zKk%kwc2f3l*NCf;e+vG5I8zh7PGHN%wJZ1+ukKlJ`g%#~q_<5n>vR`QR`5@q;D6+U!59O<_?%EvtFLYrS{*pV{diR&=whJ^SO6 z>OT)H`GVBLo3&QGSTd=1RXO(=Ip%wtPPi5?&2pKjT+k}d#I&Pt9ha7i;Or+S4>G-& zmf!rU1q;w+I*PxS>vo{T*eDeG_JlDc#OC3 z4{z>++>)v1?}UCi`ttAN=ej2syOax5uAQxIV{79TZGKWr_-@GlzaDPk9}l0t8QQMo z?o*uld55Ckx>Gwm?OeCTC7k*$@kZjhd!N_#b!Q~PD`qX*c`C0#drO_oWbNrO@+^Wq zmzJ)G&ySkqJ0t12*WYmS@U_q99PBfcDP8)&>pDmH-zTSd^U9u1FTVPlCpuVl=p^buTIQeu=CWnKJEUztxHya)D__l?}$J0Vd_o) zO>4zkw4+N8Es2dtIQ4yoc}c19&aAzYS8tfJI%4)}-e(s#uYP$vp={lqCthMRKGrU) z|GXn;`Xw!%sbXbc^UnxP{I3@NZQ*L8oxePE4IckK6aB3hdQk4|(~3->-2`gi&bxu{2?o8Fwes;~b&X6iGqbAKwX#!O9K z7r=b4^y=eA-TBN9?{AYTDEYc(=cj_!!*|xM?pycj1(#M`g++kVC;ph7M#ruet#g{~ z*|%#}R+&oNE~bn8{tF%-aHyEUy!dR^yZsFo3~%mLTA}-8?_aMdiBjEt=7-v^$(pejz5$6kOTQ zP{}G=YH%b^qD}7b`a`XJcI+PKUb*t^402$&$Z_?|`WtsQ8(U|;_qcm%jmcL}i-p$}ww9Nr-umZtf8t+pSF`4~OugT} z@lWdB$9?hrZmFA>r>$FHH(T${lh9kIweQEM-d4XIGqbzg^~o;#6_0Jh4jK2Xvo3Kx zIql@s`^Tm}wKy^N*(=uXYmSQEIWV=|FK@zGQLSgL&HLUe-+kJpvEX2PZTQrAvlG3; zXMSh=z3$N?E=Hf#6A$-l*Tk93Xg}=m>xQV1-jdtvc0G5Qz3WiK<6RTd!Yyv^Jap#r zr*)re^w!N-c0^{k(;>y}A+`mg|M#q2?bH~r#G>*^BPn6NQtt=lN?t zpF1=5W1V-*+{sOE4*q!1RU0YHZ&~nFi@oIJjUzlOTJDDjh0h7!YXALhoZ0^)ub1Dw z$$$FbYpaNVItJgwazE#4@3O!9?|u`u+vEo_Wpt_vz&8XcGW z_SLGRpH631c_lvP5NFuUwOKp--%C?-bzwfQJ{46+KbnE>tn7ic8-h9mw9$Wl(q2dyaQJacAt8ab){m1Ie)44yMm`j zuU0dKpICl!s_es49jn_<&5Zf`Z@tOY$ih;qvsZUNTUFx{F1!Ej&QGR(>W^NP%vzmT z7AhCH$MQ_$b<>IU;UTIv5!>y~yxsH3^k$jLI=|<9;g|nuO*-wbk=&)e*Y3>5)j78( z?y#DBrue~|Q&Zg2o^$QK>hpysbYe#BjD3Av>qOiSz7+W39n`)0W%Txcuf&~p7Mjlb zRPj7(PK@BqrnMJd8Q5lJorn+i=sas4zGCyM&|hxX!%M%cR*qr#G3)q-9jE$z%O-0} zTskW%&As~Y7m3TV>z-T+-WOGDn($Wtr8k$Y`iuho0;lp_L9Xd@HD;z*@3->l|Cc^n zJ7-N=_{Op&KlZG@%|CCx@|#u1o~?44a&M8 z1G+;>c8UMG16Mb6t_~EPT57gpWnNM1PuIWhsdS!BBYPWUx-$z`_yZRrV$`M%R{(teEGoO!T zJ;*%3UiY)Cb&YSh zj^@u|E?<|RyN0j$v6e~re%0O2PsOg~E8vvbv1`@Gmv6Y)3;mZwnXGyPe{CNN7 z)RmL9dBo;M%cd1>{h}TJ5fBgGy&c@=>8k6yKSE}CYx-H2nikqvj>=^Vi<@uY2=wS~f$Pv|^x_vxu= zs_V@PL&d@q%S3H=AKKSYUhwFZ2w%)h+i*4A%GB`lVzHqGu9f@aUHfjm(y2?j{!dDr zardW>yN`-`UA5UPth>2YX}=N2?n|L>uR8QUG!irJTK&Btw)JM{z2eXtp|fuauPy)T z)W2#)^PWeYY}qG^)Qhw0d%GLvUfIcZ?=6>mV(Nq+rq9~dx5v!<(e!VHP}Z6qR=hI| zN;TfC+V)eHb)`j_snF3g=cmT5ebf1FkD+4qarF7DO( zx2x=xRr|$Io7An}UNOg%rhNMMPwRDrdU#yS+9$n_zc+-(*aoU^I-xr~_Ud7y&Eh*= z9l6@9y!f2zr<>uMZ^!p)9g^S8RikAi;Pyjf)wF>4X;HF5yxqT~?ruDPK14K}f4z?+!>(H}xKZ4rV-KqJJc6Y0O$D-^nr$0VZJ|quEqXtYS2D0vtK=Y zpPTu7|JU-zuNL_hmTKL+_U75+q@M!oCe1y+{Oi|sH4hy>yuH_T`gGXB`0}^hNkvz8 zb}gUjyXmZ`*w2$rCK)+ggSnf(BLsyadPU(M3s0Qw%13T zA6A9my?83g|I@1wuI)d1WJ6WVF9c_;>+-alsJln^(68f#Ti0}~4%S?wD15c=P2=iD zo|v0m*TvQ)$$nh>;L!2;o3;IP?|VwxOgwa6`=Dssi^fM!vRYPqNUb~5ab2y_)Gp|{ z#uht!=ft@&sYOp`>wh?=eQ0k_%Eu@B7Ku+O2`b*76fUb-7HU@Ps#Ww!O>fO)ZNIzv zojP;Y{x#KOUJ}x4SGT%btBu{EZ0SR-fQ4(g|K&VfmA@|H&;J|ilke`BXx#8v{-Cu% ztDu?X1Fq-Kc!S~mj!1feih5O z%JWnqb$ZwHZ5PT)?H0M_O?_fqxb2SK2RFIK4EZZzkGrF8hp3xxx6NYbh}s0m2r=wYrFTHx|~we^+npVvvZ@Cgx(9d zr5cm}F?p{3%VWw*=k7a|z3|d2i@7(dmWQp}ZMpv3s~SNQohzsOEc}AD=ZE%w-L_3? zTU}@Qrw^)yzr)x+pW63jd067Yu&}g(UA=i0Iiauh!oPU$ne4vc?Ak}YVd=rEZ) z37vWRhO_m})YVS1Yu{;I*&Y@8J8YYa-X*VPVJ|mKR#BYD+J8 zP0P5-6TG@`Qr}v;|5KkNZP~@FeL39xdFZUD+%=(5OQv4`mb9nm_V%oh8WGPeR`(*F z#jfqL|7X{`w#zlDG+Em+!!BefA_ud zcNp{9zR6EwVxHgXR+@V4l-;oi*UTj4TX|fsLzbt0doz7)SNRsLT>l*>W>-GQk=FGU z*eRen`O8nQ(|r}$TQ}C%WqPapFz(#HO7EX8e7(j;|68G_ZG`sxXwhq{CMWF-n^u29 zJMQZFwO#fnkNvYa7WS=u(~0QH4^2}&lTJ^t*gCafPI#rT?59~8rBPe1R&3fhRd%EP zp*yRJi^6n1g~{!?cPe9Da&Xok=IvqUqK@uS>yln_{S25?)v)KR3|A*B^|GBoU-7a$D@}2QtH+{c<_kDQ8%vX2U#yts}d#CuG zU6u9j+QiFe?*9)_^bg2RE!}r>edAQWwN9d+t{ltUka{(Ht^QT9EBE-nhMeDKd_ON| z=i0@#$1*Dy$1cqeo%^c(b9jhO{7T(@dXb)Uwr&6S>0sX3Kn4UT&{{PA-Qkk!o=aCo zW*hwQ+&J@{e)QI$jQ4u;y}u=Vf7#w5Z@B%{vfW`zZf-xtziob<#}C1UGvBR`vW%L$ z;`Z7sFsT(acg5_rCYN4?oL&2@c7a;d+*Q)m-_|ah6sG>_*q&Rj999RXeqNPZGT~3{ z*R=au=eJD{UcKcL!$$GhDpS8H)h^z>^QywqWi<<@-Z8WI+rRJ9!b=hBZLS^HGuhkP zdF|BGuX$Nly%T@tOul%1?NlApAL>WfXtG%{foe*t0S9aUVYrn2}#clf)Gi$5+?;S^veKKj?wog8SWQ*XtFWZutOUYmEd zKJuw+VNUA$4;Rt2;zpXPbUp?PWbT zI@Hp0L(MJkea~h^Jq^E*6pSJd2|2ZwFeo7BAizBa~w+b`9%e(rgp88unqyQJp*tmGAWn=3C- z9e&|eO3b!R26Oor6|a67P?31?#^(857v4l?o>_Zq=Tz2LaXYU0ueFL=doJo{Sn>L@ z`(De!T=TB>p1<@;i#UK6Ncr|~Q z(tF(~b-VbCohwz=UK4lSB78dOdu=AWi^}$_BU_(ulW;S4`|Py(>ep)-@1_=McZc2j zI<@0<(Uyqq!JbiT^*VEv--s2h_BtLGnm_kQ>z!5WZ>>~W8)oj2xp!4z-r2Qxu4p~W zI2k3n#ftZu=h4XZ&x2R*k+?O@Y2L27ywGQn?yJ9+p3957`mG^(^^eGKME#`iyV=WMTx-iD zqn&U3Ve^C~k<}a)LTl^nCjHpj-k_`+SW6;$d+FjY=WB4<)iA4Jm2$srMmT7 znYDJ4N}h+kH7Qitf2;TZ%LgZA*V@hdz3rFdQF)u0(Oci!v)di{8g{(qx{tn;ihV)U z zOLp_P3jBWa|9;)wCp&Tv7fLK!&{~#Sc5|w&6Q6*?UWbeK7P!A)Wt-aPwjlk*qSXQ~ zSXo3IcLW^y_iu^4ab(KkRr4z44{Y-}-8?ZqV9_qC^>f_PbR{Nn|I)C_EG~|VmWuzP zwrc9Pf2+53HUFJx5t(^=(bYb`wK{$8wp__7_RGDJS$0iv-{ogn+Y^3mae1Em^LyOv zZE60F&0fnc&f2lH%3iL_L_az^b>_BfS*3ZGoU;qLLmu(pin)@%wa)La$^m(_D)qSvksy|(!lkJRS=MNcnYo^n3fD5uJN zOD{z1jmvy<#Vfna{N);nFQ)77+E4xT^TOlGlX=mi(a~oXZ`D%${`K>V@a3G-*4$qb zc2FpD&AOYpd{NrnW_k|~9Ig;Ow>IeS&Fs}S8qp`e)~!6UA#(N7-bj1ZPq%koOskQT-{?)Lk>-x2?a($G|y0`7FT$=w|g3qQS{7}@-TVDIW z)#NR$Ub?2Va%0%hsGB#IZ2Y+GYRSCt)Q^Q09~9y`ulkiv4~vR^;kPQjM0#n*WV=f{ z-`-iJUv@dVTx_s zI=62BU1_>4?B>nNi&ss5`>tLP=6rW0+pcYw3W|Kz?5*7zn)~QYOL^|p*HSKjZ|$+# z>Sez6SaSc9r@ObTdhW3H#@f)0?j_q^?cNe~yQ2JlmgV+KTRyYP9b5gOCikM%*VL8i zTfJ9*dc5-f)~gl9>tBT&S@-VQ)VuMSJL6t8Ua~^~>PEQ~yBWyY92rPpkDJPMg}UU(3q=9=>#m*|M;Q z-LFIcdT+UPJ3Q)XkQ#5jhWXlQ_pENtTdaEP&bKw8i>{u_i^}wji`KP!>8HiNc52?1 zu-!M$E-u?T$Lq=aDBpE|g6*!)xpDbbo!rtgUw|X@*&5b_m{4QtT&(ky3MU<}DD)q{P$Ex(3 z_SUk}+^Jia7AMH0&HuGD+N4x%>#HqG3J(5U;n4kUi)r=#J6ks$C|+>&-oA{btEU#* zSzf;#U2B&etX(QTX;JF;km#-Jue_dmS*bp3<;~ZV+*ZfNMz5V7ruBer@7C1gIp23w zZTfhxOmk~!{cWvR+Eo{_(!YJaezf^o+19Dgf-fAt7Z;^@q~v9IH1G5Xc6x->vYvc*}os=A{RJ z?!{ekdz+K9#U(0HW4)OPxBKk2+Lf0lhPBT+&R_K}xW7N&`{$jl7k_;^IWOko^WuP0 z3v=@Me~K3x^}GLP{}HyON(L(`wE1O!V0+ZE#s8V9S?3UI*1!6Hto?^O0(n#OQi@B8 zQWJAQto)=bUPA=~1rTjxqrj^lQdy9ypdXN!o?5KHtM8qeQmhcIV5(rC5Tp>Rz^m_^ zUz%5l#KgqZLf61d-M~OylULul zC^fMpGd~ZTIzt0Ab@@dK8ji_D`ALZ-3MP663c<;Vc?tnJrRlnvc?xOyMGE1Wc`5nj z#SEIf`T<4xDW%D&#OpxvL`6|*8n1zZu>!9F94MHXni`ubr15frg)B@hjSUpQOocpz zn579qtfVM0Gbgo(*T#ldKRB~0Rl%57-_s@9#?n01!YDb-(%8Vlz`)QfEh#C*$T&60 m!Zgj?+|a}%(T=c+SYA6j-r|zPq7txE4b6>tRaIU6-FN{s=$cyq diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 530dae10..97ea0171 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -313,19 +313,26 @@ def test_metadata_fixup_warning(resources, outdir, caplog): def test_prevent_gs_invalid_xml(resources, outdir): - from ocrmypdf.__main__ import parser + from ocrmypdf.cli import parser from ocrmypdf._pipeline import convert_to_pdfa from ocrmypdf.pdfa import generate_pdfa_ps from ocrmypdf.pdfinfo import PdfInfo generate_pdfa_ps(outdir / 'pdfa.ps') - copyfile(resources / 'enron1.pdf', outdir / 'layers.rendered.pdf') + copyfile(resources / 'trivial.pdf', outdir / 'layers.rendered.pdf') + + # Inject a string with a trailing nul character into the DocumentInfo + # dictionary of this PDF, as often occurs in practice. + with pikepdf.open(outdir / 'layers.rendered.pdf') as pike: + pike.Root.DocumentInfo = pikepdf.Dictionary( + Title=b'String with trailing nul\x00' + ) options = parser.parse_args( args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) - pdfinfo = PdfInfo(resources / 'enron1.pdf') - context = PDFContext(options, outdir, resources / 'enron1.pdf', pdfinfo) + pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') + context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context @@ -339,5 +346,6 @@ def test_prevent_gs_invalid_xml(resources, outdir): xmp_start = mm.find(XMP_MAGIC) xmp_end = mm.rfind(b' Date: Wed, 12 Feb 2020 00:07:24 -0800 Subject: [PATCH 343/880] docs: typo --- docs/installation.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index c06dec78..7bb5bb08 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -216,7 +216,8 @@ of ``pip`` at ``/usr/local/bin/pip``. **Install OCRmyPDF** OCRmyPDF requires the locale to be set for UTF-8. **On some minimal -Ubuntu installations systems**, it may be necessary to set the locale. +Ubuntu installations**, such as the Ubuntu 16.04 Docker images it may be +necessary to set the locale. .. code-block:: bash From 975abfde9a81ca8a2bace7e172d68131542fda82 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 18 Feb 2020 02:08:58 -0800 Subject: [PATCH 344/880] docs: archlinux install - yaourt is gone --- docs/installation.rst | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 7bb5bb08..5aaf5496 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -305,12 +305,9 @@ ArchLinux (AUR) :target: https://repology.org/metapackage/ocrmypdf There is an `ArchLinux User Repository package for -ocrmypdf `__. You can use -the following command. - -.. code-block:: bash - - yaourt -S ocrmypdf +ocrmypdf `__. If you have any +idea how to actually install the package, please feel free to contribute +appropriate instructions, as this author is completely mystified by ArchLinux. If you have any difficulties with installation, check the repository package page. From 32e2175891743c6610e041b6ecbacb39e226a930 Mon Sep 17 00:00:00 2001 From: Ivan Kuchin Date: Tue, 18 Feb 2020 11:10:01 +0100 Subject: [PATCH 345/880] Docker image includes also French, Portuguese and Spanish (#491) --- docs/docker.rst | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 392b82d9..77efe149 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -98,10 +98,10 @@ Docker volume: Adding languages to the Docker image ==================================== -By default the Docker image includes English, German and Simplified -Chinese, the most popular languages for OCRmyPDF users based on -feedback. You may add other languages by creating a new Dockerfile based -on the public one: +By default the Docker image includes English, German, Simplified Chinese, +French, Portuguese and Spanish, the most popular languages for OCRmyPDF +users based on feedback. You may add other languages by creating a new +Dockerfile based on the public one: .. code-block:: dockerfile From e3e888efdeb0cbc6e1d5a21c0ee741b65093b567 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 18 Feb 2020 02:41:22 -0800 Subject: [PATCH 346/880] Readme: Add another heise article --- README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/README.md b/README.md index 28824cb6..1c38188c 100644 --- a/README.md +++ b/README.md @@ -130,6 +130,7 @@ Press & Media - [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c) - [c't 1-2014, page 59](https://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't - [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670) +- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html) Business enquiries ------------------ From c16f79d51b9a825dd1eb85f2d77a24bd30b5c328 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 18 Feb 2020 02:50:57 -0800 Subject: [PATCH 347/880] docs: add Docker compose configuration for watchdog --- docs/batch.rst | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/docs/batch.rst b/docs/batch.rst index e9c96fa7..c5938446 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -235,6 +235,25 @@ This service relies on polling to check for changes to the filesystem. It may not be suitable for some environments, such as filesystems shared on a slow network. +A configuration manager such as Docker Compose could be used to ensure that the +service is always available. + +.. code-block:: yaml + + --- + ocrmypdf: + restart: always + container_name: ocrmypdf + image: jbarlow83/ocrmypdf + volumes: + - '/media/scan:/input' + - '/mnt/scan:/output' + environment: + - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 + entrypoint: + - python3: + command: "watcher.py" + Watched folders with watcher.py ------------------------------- From 2391fb0be0ad4913c05f8c536c9a52f6c557cfa1 Mon Sep 17 00:00:00 2001 From: knobix <43905002+knobix@users.noreply.github.com> Date: Tue, 25 Feb 2020 08:40:25 +0100 Subject: [PATCH 348/880] Update installation instructions for FreeBSD (#493) Python 3.7 is the new default version since 2020Q1 which is reflected in the new prefix (= py37-). Also update the current available FreeBSD versions: * FreeBSD 11.2-RELEASE has reached its End-of-Life in 2019Q4 * FreeBSD 12.1-RELEASE was also introduced in 2019Q4 --- docs/installation.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 5aaf5496..f80c1dce 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -435,12 +435,12 @@ Installing on FreeBSD :alt: FreeBSD :target: https://repology.org/project/python:ocrmypdf/versions -FreeBSD 11.2, 11.3, 12.0-RELEASE and 13.0-CURRENT are supported. Other +FreeBSD 11.3, 12.0, 12.1-RELEASE and 13.0-CURRENT are supported. Other versions likely work but have not been tested. .. code-block:: bash - pkg install py36-ocrmypdf + pkg install py37-ocrmypdf To install a more recent version, you could attempt to first install the system version with ``pkg``, then use ``pip install --user ocrmypdf``. From e04e4565a9f535c34ad73a2fb8553478d53a438b Mon Sep 17 00:00:00 2001 From: Pig Monkey Date: Tue, 25 Feb 2020 18:59:49 -0800 Subject: [PATCH 349/880] Demonstrate installing the AUR package without a helper This describes how to use the AUR package on a minimal install, as per the discussion in #494. There may be formatting mistakes. I don't use RST myself, so I wrote the instructions in Markdown, converted via Pandoc, and gave the output a quick comparison against the rest of the installation docs. --- docs/installation.rst | 84 ++++++++++++++++++++++++++++++++++++++----- 1 file changed, 76 insertions(+), 8 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index f80c1dce..aa683c6a 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -297,20 +297,88 @@ compiled by hand. To add JBIG2 encoding, see :ref:`jbig2`. -ArchLinux (AUR) ---------------- +Arch Linux (AUR) +---------------- .. image:: https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg :alt: ArchLinux :target: https://repology.org/metapackage/ocrmypdf -There is an `ArchLinux User Repository package for -ocrmypdf `__. If you have any -idea how to actually install the package, please feel free to contribute -appropriate instructions, as this author is completely mystified by ArchLinux. +There is an `Arch User Repository (AUR) package for OCRmyPDF +`__. -If you have any difficulties with installation, check the repository -package page. +Installing AUR packages as root is not allowed, so you must first `setup a +non-root user +`__ and +`configure sudo `__. +If you are using a VM image, such as `the official Vagrant image +`__, this may already be +completed for you. + +Next you should install the `base-devel package group +`__. This includes the +standard tooling needed to build packages, such as a compiler and binary tools. +`As noted in the Arch Wiki +`__, "packages belonging to +this group are not required to be listed as build-time dependencies +(makedepends) in PKGBUILD files". + +.. code-block:: bash + + sudo pacman -S base-devel + +The OCRmyPDF package depends on `the python-pdfminer.six AUR package +`__. Dependencies on +AUR packages are not automatically resolved, so this package must be manually +installed first. + +.. code-block:: bash + + curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/python-pdfminer.six.tar.gz + tar xvzf python-pdfminer.six.tar.gz + cd python-pdfminer.six + makepkg -sri + +With that complete you can then repeat the same series of steps for the +OCRmyPDF package. + +.. code-block:: bash + + curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/ocrmypdf.tar.gz + tar xvzf ocrmypdf.tar.gz + cd ocrmypdf + makepkg -sri + +At this point you will have a working install of OCRmyPDF, but the Tesseract +install won’t include any OCR language data. You can install `the +tesseract-data package group +`__ to add all supported +languages, or use that package listing to identify the appropriate package for +your desired language. + +.. code-block:: bash + + sudo pacman -S tesseract-data-eng + +As an alternative to this manual procedure, consider using an `AUR helper +`__. Such a tool will +automatically fetch, build and install the AUR package, resolve dependencies +(including dependencies on AUR packages), and ease the upgrade procedure. + +If you have any difficulties with installation, check the repository package +page. + +.. note:: + + The OCRmyPDF AUR package currently omits the JBIG2 encoder. OCRmyPDF works + fine without it but will produce larger output files. The encoder is + available from `the jbig2enc-git AUR package + `__ and may be installed + using the same series of steps as for the installation of the pdfminer.six + and OCRmyPDF AUR packages. Alternatively, it may be built manually from + source following the instructions in `Installing the JBIG2 encoder + `__. If JBIG2 is installed, OCRmyPDF 7.0.0 and later will + automatically detect it. Alpine Linux ------------ From 43a23e3695583344a18858387b610e5911e9f5b0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 25 Feb 2020 22:22:57 -0800 Subject: [PATCH 350/880] Disable Travis --- .travis.yml | 159 ---------------------------------------------------- 1 file changed, 159 deletions(-) delete mode 100644 .travis.yml diff --git a/.travis.yml b/.travis.yml deleted file mode 100644 index 9fdb4dde..00000000 --- a/.travis.yml +++ /dev/null @@ -1,159 +0,0 @@ -branches: - except: - - azure - -cache: - pip: true - directories: - - $HOME/Library/Caches/Homebrew - -matrix: - include: - - os: linux - dist: trusty - sudo: required - language: python - python: "3.6" - env: - - DIST=trusty - - MINIMAL=true - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - - sourceline: "ppa:vshn/ghostscript" - packages: - - ghostscript - - libffi-dev - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - before_install: | - pip3 install --upgrade pip - pip3 install --upgrade wheel - - os: linux - dist: trusty - sudo: required - language: python - python: "3.6" - env: - - DIST=trusty - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - - sourceline: "ppa:heyarje/libav-11" - - sourceline: "ppa:vshn/ghostscript" - packages: - - ghostscript - - libavcodec56 - - libavformat56 - - libavutil54 - - libffi-dev - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - libexempi3 # --- optional extras from here --- - - pngquant - - poppler-utils - before_install: | - mkdir -p bin packages - pip3 install --upgrade pip - pip3 install --upgrade wheel - - os: linux - dist: xenial - sudo: required - language: python - python: "3.7" - env: - - DIST=xenial - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - packages: - - ghostscript - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - unpaper - - os: linux - dist: bionic - sudo: required - language: python - python: "3.8" - env: - - DIST=bionic - addons: - apt: - update: true - sources: - - sourceline: "ppa:alex-p/tesseract-ocr" - packages: - - ghostscript - - libexempi3 - - libffi-dev - - pngquant - - poppler-utils - - tesseract-ocr - - tesseract-ocr-deu - - tesseract-ocr-eng - - tesseract-ocr-fra - - unpaper - - os: osx - language: generic - addons: - homebrew: - update: true - packages: - - exempi - - ghostscript - - jbig2enc - - leptonica - - openjpeg - - pngquant - - python - - qpdf - - tesseract - - unpaper - before_install: | - pip3 install --upgrade pip - pip3 install wheel - -before_cache: - - rm -f $HOME/.cache/pip/log/debug.log - -install: - - mkdir -p bin - - export PATH=$PWD/bin:$PATH - - pip3 install -r requirements/main.txt -r requirements/test.txt . - -script: - - tesseract --version - - pytest -n auto -# deploy: -# # release for main pypi -# # 3.7 is considered the build leader and does the deploy, otherwise there is -# # a race and all versions will try to deploy -# # OTOH if we ever need separate binary wheels then each version needs its -# # own deploy -# - provider: pypi -# user: ocrmypdf-travis -# password: -# secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo=" -# distributions: "sdist bdist_wheel" -# on: -# branch: master -# tags: true -# condition: $TRAVIS_PYTHON_VERSION == "3.7" && $TRAVIS_OS_NAME == "linux" -# skip_upload_docs: true From 0417610f9bdef647751a037c5320ce0d28baaf09 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 25 Feb 2020 22:23:58 -0800 Subject: [PATCH 351/880] docs: some mild improvements --- docs/index.rst | 4 ++-- docs/installation.rst | 2 +- docs/introduction.rst | 2 +- docs/languages.rst | 13 +++++++++++-- 4 files changed, 15 insertions(+), 6 deletions(-) diff --git a/docs/index.rst b/docs/index.rst index 3f611f9d..e28f3234 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -1,8 +1,8 @@ OCRmyPDF documentation ====================== -OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to -be searched. +OCRmyPDF adds an optical charcter recognition (OCR) text layer to scanned PDF +files, allowing them to be searched. PDF is the best format for storing and exchanging scanned documents. Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply diff --git a/docs/installation.rst b/docs/installation.rst index aa683c6a..6366aacf 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -110,7 +110,7 @@ Fedora 29 or newer | |fedora-29| |fedora-30| |fedora-rawhide| | +-----------------------------------------------+ -Users of Fedora 29 later may simply +Users of Fedora 29 or later may simply .. code-block:: bash diff --git a/docs/introduction.rst b/docs/introduction.rst index b878439b..5fb5c594 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -10,7 +10,7 @@ About OCR `Optical character recognition `__ is technology that converts images of typed or handwritten text, such as -in a scanned document, to computer text that can be searched and copied. +in a scanned document, to computer text that can be selected, searched and copied. OCRmyPDF uses `Tesseract `__, the best diff --git a/docs/languages.rst b/docs/languages.rst index 811dd06f..cc26fa68 100644 --- a/docs/languages.rst +++ b/docs/languages.rst @@ -4,11 +4,20 @@ Installing additional language packs ==================================== -OCRmyPDF uses Tesseract for OCR, and relies on its language packs for -languages other than English. +OCRmyPDF uses Tesseract for OCR, and relies on its language packs for all languages. +On most platforms, English is installed with Tesseract by default, but not always. Tesseract supports `most languages `__. +Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3). +Tesseract's documentation also lists the three-letter code for your language. +Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others +are not, e.g. German is ``deu``. + +After you have installed a language pack, you can use it ``ocrmypdf -l ``, +for example ``ocrmypdf -l spa``. For multilingual documents, you can specify +all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French. +English is assumed by default unless other language(s) are specified. For Linux users, you can often find packages that provide language packs: From 0b1db8fccd62421501f648d6dc5dcdd9373d2899 Mon Sep 17 00:00:00 2001 From: deisi Date: Mon, 2 Mar 2020 08:01:39 +0100 Subject: [PATCH 352/880] Fixes docker-compose.yaml file (#499) Fixes https://github.com/jbarlow83/OCRmyPDF/issues/498 --- docs/batch.rst | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index c5938446..fc348e4a 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -250,9 +250,8 @@ service is always available. - '/mnt/scan:/output' environment: - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 - entrypoint: - - python3: - command: "watcher.py" + entrypoint: python3 + command: watcher.py Watched folders with watcher.py ------------------------------- From 3960232ae054cae0fc5bcbd8b5e02f85f1fea85d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 00:59:48 -0800 Subject: [PATCH 353/880] docs: more clarifications --- docs/docker.rst | 26 ++++++++++++++------------ docs/installation.rst | 12 +++++------- 2 files changed, 19 insertions(+), 19 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 77efe149..27915b26 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -56,10 +56,10 @@ See the Docker documentation for Using the Docker image on the command line ========================================== -**Unlike typical Docker containers**, in this mode we are using the -OCRmyPDF Docker container is intended to be emphemeral – it runs for one -OCR job and then terminates, just like a command line program. We are -using Docker as a way of delivering an application, not a server. +**Unlike typical Docker containers**, in this section the OCRmyPDF Docker +container is emphemeral – it runs for one OCR job and terminates, just like a +command line program. We are using Docker to deliver an application (as opposed +to the more conventional case, where a Docker container runs as a server). To start a Docker container (instance of the image): @@ -69,22 +69,21 @@ To start a Docker container (instance of the image): docker run --rm -i ocrmypdf (... all other arguments here...) For convenience, create a shell alias to hide the Docker command. It is -easier to send the input file to file stdin and read the output from -stdout – this avoids the occasionally messy permission issues with -Docker entirely. +easier to send the input file as stdin and read the output from +stdout – **this avoids the messy permission issues with Docker entirely**. .. code-block:: bash - alias ocrmypdf='docker run --rm -i ocrmypdf' - ocrmypdf --version # runs docker version - ocrmypdf output.pdf + alias docker_ocrmypdf='docker run --rm -i ocrmypdf' + docker_ocrmypdf --version # runs docker version + docker_ocrmypdf output.pdf Or in the wonderful `fish shell `__: .. code-block:: fish - alias ocrmypdf 'docker run --rm ocrmypdf' - funcsave ocrmypdf + alias docker_ocrmypdf 'docker run --rm ocrmypdf' + funcsave docker_ocrmypdf Alternately, you could mount the local current working directory as a Docker volume: @@ -93,6 +92,9 @@ Docker volume: docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf +(However, when done this way, ``output.pdf`` may be owned by the root +user.) + .. _docker-lang-packs: Adding languages to the Docker image diff --git a/docs/installation.rst b/docs/installation.rst index 6366aacf..49cec3bc 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -311,17 +311,15 @@ Installing AUR packages as root is not allowed, so you must first `setup a non-root user `__ and `configure sudo `__. -If you are using a VM image, such as `the official Vagrant image -`__, this may already be -completed for you. +The standard Docker image, ``archlinux/base:latest``, does **not** have a +non-root user configured, so users of that image must follow these guides. If +you are using a VM image, such as `the official Vagrant image +`__, this work may already +be completed for you. Next you should install the `base-devel package group `__. This includes the standard tooling needed to build packages, such as a compiler and binary tools. -`As noted in the Arch Wiki -`__, "packages belonging to -this group are not required to be listed as build-time dependencies -(makedepends) in PKGBUILD files". .. code-block:: bash From e40c60d4d86ec847e92c3cce77a25734a514a305 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 00:59:57 -0800 Subject: [PATCH 354/880] watcher: add self to copyright --- misc/watcher.py | 1 + 1 file changed, 1 insertion(+) diff --git a/misc/watcher.py b/misc/watcher.py index e97d4625..58fe35e0 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -1,4 +1,5 @@ # Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander +# Copyright (C) 2020 James R Barlow: https://github.com/jbarlow83 # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU General Public License as published by From c3bd2f296d27b13628238f3ade250cfc56e8b03e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 02:14:50 -0800 Subject: [PATCH 355/880] docs: fix Docker syntax to use stdin/stdout properly --- docs/docker.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 27915b26..032ad090 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -66,7 +66,7 @@ To start a Docker container (instance of the image): .. code-block:: bash docker tag jbarlow83/ocrmypdf ocrmypdf - docker run --rm -i ocrmypdf (... all other arguments here...) + docker run --rm -i ocrmypdf (... all other arguments here...) - - For convenience, create a shell alias to hide the Docker command. It is easier to send the input file as stdin and read the output from @@ -76,7 +76,7 @@ stdout – **this avoids the messy permission issues with Docker entirely**. alias docker_ocrmypdf='docker run --rm -i ocrmypdf' docker_ocrmypdf --version # runs docker version - docker_ocrmypdf output.pdf + docker_ocrmypdf - - output.pdf Or in the wonderful `fish shell `__: From 7d55f6e01fa35378a82888bcbbf1856e78f17292 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 02:15:35 -0800 Subject: [PATCH 356/880] docs: extract example files from batch.rst --- docs/batch.rst | 133 +++---------------------------- misc/batch.py | 50 ++++++++++++ misc/docker-compose.example.yaml | 15 ++++ misc/synology.py | 72 +++++++++++++++++ 4 files changed, 148 insertions(+), 122 deletions(-) create mode 100644 misc/batch.py create mode 100644 misc/docker-compose.example.yaml create mode 100644 misc/synology.py diff --git a/docs/batch.rst b/docs/batch.rst index fc348e4a..0f918ac9 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -69,53 +69,8 @@ Sample script This user contributed script also provides an example of batch processing. -.. code-block:: python - - #!/usr/bin/env python3 - # Walk through directory tree, replacing all files with OCR'd version - # Original version by DeliciousPickle@github; modified - - import logging - import os - import subprocess - import sys - - import ocrmypdf - - script_dir = os.path.dirname(os.path.realpath(__file__)) - print(script_dir + '/ocr-tree.py: Start') - - if len(sys.argv) > 1: - start_dir = sys.argv[1] - else: - start_dir = '.' - - if len(sys.argv) > 2: - log_file = sys.argv[2] - else: - log_file = script_dir + '/ocr-tree.log' - - logging.basicConfig( - level=logging.INFO, format='%(asctime)s %(message)s', - filename=log_file, filemode='w') - - ocrmypdf.configure_logging(ocrmypdf.Verbosity.default) - - for dir_name, subdirs, file_list in os.walk(start_dir): - logging.info('\n') - logging.info(dir_name + '\n') - os.chdir(dir_name) - for filename in file_list: - file_ext = os.path.splitext(filename)[1] - if file_ext == '.pdf': - full_path = dir_name + '/' + filename - print(full_path) - result = ocrmypdf.ocr(filename, filename, deskew=True) - if result == ocrmypdf.ExitCode.already_done_ocr: - print("Skipped document because it already contained text") - elif result == ocrmypdf.ExitCode.ok: - print("OCR complete") - logging.info(result) +.. literalinclude:: ../misc/batch.py + :caption: misc/batch.py Synology DiskStations --------------------- @@ -131,62 +86,8 @@ products use ARM or Power processors and do not support Docker. Further adjustments might be needed to deal with the Synology's relatively limited CPU and RAM. -.. code-block:: python - - #!/bin/env python3 - # Contributed by github.com/Enantiomerie - - # script needs 2 arguments - # 1. source dir with *.pdf - default is location of script - # 2. move dir where *.pdf and *_OCR.pdf are moved to - - import logging - import os - import subprocess - import sys - import time - import shutil - - script_dir = os.path.dirname(os.path.realpath(__file__)) - timestamp = time.strftime("%Y-%m-%d-%H%M_") - log_file = script_dir + '/' + timestamp + 'ocrmypdf.log' - logging.basicConfig(level=logging.INFO, format='%(asctime)s %(message)s', filename=log_file, filemode='w') - - if len(sys.argv) > 1: - start_dir = sys.argv[1] - else: - start_dir = '.' - - for dir_name, subdirs, file_list in os.walk(start_dir): - logging.info('\n') - logging.info(dir_name + '\n') - os.chdir(dir_name) - for filename in file_list: - file_ext = os.path.splitext(filename)[1] - if file_ext == '.pdf': - full_path = dir_name + '/' + filename - file_noext = os.path.splitext(filename)[0] - timestamp_OCR = time.strftime("%Y-%m-%d-%H%M_OCR_") - filename_OCR = timestamp_OCR + file_noext + '.pdf' - docker_mount = dir_name + ':/home/docker' - # create string for pdf processing - # diskstation needs a user:group docker:docker. find uid:gid of your diskstation docker:docker with id docker. - # use this uid:gid in -u flag - # rw rights for docker:docker at source dir are also necessary - # the script is processed as root user via chron - cmd = ['docker', 'run', '--rm', '-v', docker_mount, '-u=1030:65538', 'jbarlow83/ocrmypdf', , '--deskew' , filename, filename_OCR] - logging.info(cmd) - proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) - result = proc.stdout.read() - logging.info(result) - full_path_OCR = dir_name + '/' + filename_OCR - os.chmod(full_path_OCR, 0o666) - os.chmod(full_path, 0o666) - full_path_OCR_archive = sys.argv[2] - full_path_archive = sys.argv[2] + '/no_ocr' - shutil.move(full_path_OCR,full_path_OCR_archive) - shutil.move(full_path, full_path_archive) - logging.info('Finished.\n') +.. literalinclude:: ../misc/synology.py + :caption: misc/synology.py - Sample script for Synology DiskStations Huge batch jobs --------------- @@ -238,29 +139,18 @@ slow network. A configuration manager such as Docker Compose could be used to ensure that the service is always available. -.. code-block:: yaml - - --- - ocrmypdf: - restart: always - container_name: ocrmypdf - image: jbarlow83/ocrmypdf - volumes: - - '/media/scan:/input' - - '/mnt/scan:/output' - environment: - - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 - entrypoint: python3 - command: watcher.py +.. literalinclude:: ../misc/docker-compose.example.yaml + :language: yaml + :caption: misc/docker-compose.example.yaml Watched folders with watcher.py ------------------------------- -The watcher service may also be run natively. +The watcher service may also be run natively, without Docker: .. code-block:: bash - pip3 install -r reqs/watcher.txt + pip3 install -r requirements/watcher.txt env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \ OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \ @@ -337,8 +227,7 @@ of Automator, the ``PATH`` may be set differently your Terminal's ``PATH``; you may need to explicitly set the PATH to include ``ocrmypdf``. The following example may serve as a starting point: -|Example macOS Automator script| +.. figure:: images/macos-workflow.png + :alt: Example macOS Automator workflow You may customize the command sent to ocrmypdf. - -.. |Example macOS Automator script| image:: images/macos-workflow.png diff --git a/misc/batch.py b/misc/batch.py new file mode 100644 index 00000000..2a565d8a --- /dev/null +++ b/misc/batch.py @@ -0,0 +1,50 @@ +#!/usr/bin/env python3 +# Original version by DeliciousPickle@github; modified + +# This script must be edited to meet your needs. + +import logging +import os +import sys + +import ocrmypdf + +# pylint: disable=logging-format-interpolation +# pylint: disable=logging-not-lazy + +script_dir = os.path.dirname(os.path.realpath(__file__)) +print(script_dir + '/batch.py: Start') + +if len(sys.argv) > 1: + start_dir = sys.argv[1] +else: + start_dir = '.' + +if len(sys.argv) > 2: + log_file = sys.argv[2] +else: + log_file = script_dir + '/ocr-tree.log' + +logging.basicConfig( + level=logging.INFO, + format='%(asctime)s %(message)s', + filename=log_file, + filemode='w', +) + +ocrmypdf.configure_logging(ocrmypdf.Verbosity.default) + +for dir_name, subdirs, file_list in os.walk(start_dir): + logging.info(dir_name + '\n') + os.chdir(dir_name) + for filename in file_list: + file_ext = os.path.splitext(filename)[1] + if file_ext == '.pdf': + full_path = dir_name + '/' + filename + print(full_path) + result = ocrmypdf.ocr(filename, filename, deskew=True) + if result == ocrmypdf.ExitCode.already_done_ocr: + print("Skipped document because it already contained text") + elif result == ocrmypdf.ExitCode.ok: + print("OCR complete") + logging.info(result) diff --git a/misc/docker-compose.example.yaml b/misc/docker-compose.example.yaml new file mode 100644 index 00000000..0db102b9 --- /dev/null +++ b/misc/docker-compose.example.yaml @@ -0,0 +1,15 @@ +--- +version: "3.3" +services: + ocrmypdf: + restart: always + container_name: ocrmypdf + image: jbarlow83/ocrmypdf + volumes: + - "/media/scan:/input" + - "/mnt/scan:/output" + environment: + - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 + user: ":" + entrypoint: python3 + command: watcher.py diff --git a/misc/synology.py b/misc/synology.py new file mode 100644 index 00000000..1ead8dec --- /dev/null +++ b/misc/synology.py @@ -0,0 +1,72 @@ +#!/bin/env python3 +# Contributed by github.com/Enantiomerie + +# This script must be edited to meet your needs. + +import logging +import os +import shutil +import subprocess +import sys +import time + +# pylint: disable=logging-format-interpolation +# pylint: disable=logging-not-lazy + +script_dir = os.path.dirname(os.path.realpath(__file__)) +timestamp = time.strftime("%Y-%m-%d-%H%M_") +log_file = script_dir + '/' + timestamp + 'ocrmypdf.log' +logging.basicConfig( + level=logging.INFO, + format='%(asctime)s %(message)s', + filename=log_file, + filemode='w', +) + +if len(sys.argv) > 1: + start_dir = sys.argv[1] +else: + start_dir = '.' + +for dir_name, subdirs, file_list in os.walk(start_dir): + logging.info(dir_name) + os.chdir(dir_name) + for filename in file_list: + file_stem, file_ext = os.path.splitext(filename) + if file_ext != '.pdf': + continue + full_path = os.path.join(dir_name, filename) + timestamp_ocr = time.strftime("%Y-%m-%d-%H%M_OCR_") + filename_ocr = timestamp_ocr + file_stem + '.pdf' + # create string for pdf processing + # the script is processed as root user via chron + cmd = [ + 'docker', + 'run', + '--rm', + '-i', + 'jbarlow83/ocrmypdf', + '--deskew', + '-', + '-', + ] + logging.info(cmd) + full_path_ocr = os.path.join(dir_name, filename_ocr) + with open(filename, 'rb') as input_file, open( + full_path_ocr, 'wb' + ) as output_file: + proc = subprocess.run( + cmd, + stdin=input_file, + stdout=output_file, + stderr=subprocess.PIPE, + check=False, + ) + logging.info(proc.stderr.read()) + os.chmod(full_path_ocr, 0o664) + os.chmod(full_path, 0o664) + full_path_ocr_archive = sys.argv[2] + full_path_archive = sys.argv[2] + '/no_ocr' + shutil.move(full_path_ocr, full_path_ocr_archive) + shutil.move(full_path, full_path_archive) +logging.info('Finished.\n') From 9f31774aa99ef5c1f988ecd8b8366dc86a0597f5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 02:15:48 -0800 Subject: [PATCH 357/880] docs: document --pages --- docs/cookbook.rst | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 2ffe3be4..3daab1ca 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -237,6 +237,32 @@ You can also optimize all images without performing any OCR: ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf +Perform OCR only certain pages +------------------------------ + +You can ask OCRmyPDF to only apply OCR to certain pages. + +.. code-block:: bash + + ocrmypdf --pages 2,3,13-17 input.pdf output.pdf + +Hyphens denote a range of pages and commas separate page numbers. If you prefer +to use spaces, quote all of the page numbers: ``--pages '2, 3, 5, 7'``. + +OCRmyPDF will warn if your list of page numbers contains duplicates or +overlap pages. OCRmyPDF does not currently account for document page numbers, +such as an introduction section of a book that uses Roman numerals. It simply +counts the number of virtual pieces of paper since the start. + +Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages in +the file and convert it to PDF/A, unless you disable those options. In this +example, we want to OCR only the title and otherwise change the PDF as little +as possible: + +.. code-block:: bash + + ocrmypdf --pages 1 --output-type pdf --optimize 0 input.pdf output.pdf + Redo existing OCR ================= From d56f7490172999db993f7a954713ab1eb71064c0 Mon Sep 17 00:00:00 2001 From: Alex Date: Tue, 3 Mar 2020 11:22:01 +0100 Subject: [PATCH 358/880] Improve ocrmypdf.bash completions on macOS (#504) Fixes #502 --- misc/completion/ocrmypdf.bash | 47 ++++++++++++++++++++--------------- 1 file changed, 27 insertions(+), 20 deletions(-) diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 3d00bda8..59bf0cc3 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -5,7 +5,33 @@ set -o errexit _ocrmypdf() { local cur prev cword words split - _init_completion -s || return + + # Homebrew on Macs have version 1.3 of bash-completion which doesn't include - see #502 + if declare -F _init_completions >/dev/null 2>&1; then + _init_completion -s || return + else + COMPREPLY=() + _get_comp_words_by_ref cur prev words cword + fi + + if [[ $cur == -* ]]; then + COMPREPLY=( $( compgen -W '--language --image-dpi --output-type + --sidecar --version --jobs --quiet --verbose --title --author + --subject --keywords --rotate-pages --remove-background --deskew + --clean --clean-final --unpaper-args --oversample --remove-vectors + --threshold --force-ocr --skip-text --redo-ocr + --skip-big --jpeg-quality --png-quality --jbig2-lossy + --max-image-mpixels --tesseract-config --tesseract-pagesegmode + --help --tesseract-oem --pdf-renderer --tesseract-timeout + --rotate-pages-threshold --pdfa-image-compression --user-words + --user-patterns --keep-temporary-files --output-type + --no-progress-bar --pages --fast-web-view' \ + -- "$cur" ) ) + return + else + _filedir + return + fi case $prev in --version|-h|--help) @@ -65,25 +91,6 @@ _ocrmypdf() esac $split && return - - if [[ $cur == -* ]]; then - COMPREPLY=( $( compgen -W '--language --image-dpi --output-type - --sidecar --version --jobs --quiet --verbose --title --author - --subject --keywords --rotate-pages --remove-background --deskew - --clean --clean-final --unpaper-args --oversample --remove-vectors - --threshold --force-ocr --skip-text --redo-ocr - --skip-big --jpeg-quality --png-quality --jbig2-lossy - --max-image-mpixels --tesseract-config --tesseract-pagesegmode - --help --tesseract-oem --pdf-renderer --tesseract-timeout - --rotate-pages-threshold --pdfa-image-compression --user-words - --user-patterns --keep-temporary-files --output-type - --no-progress-bar --pages --fast-web-view' \ - -- "$cur" ) ) - return - else - _filedir - return - fi } && complete -F _ocrmypdf ocrmypdf From 8b41f60b6e4f6e5cfb35b7eba5a3fbd1326b7beb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 02:25:21 -0800 Subject: [PATCH 359/880] docs: docker prefers .yml not .yaml --- docs/batch.rst | 4 ++-- ...docker-compose.example.yaml => docker-compose.example.yml} | 0 2 files changed, 2 insertions(+), 2 deletions(-) rename misc/{docker-compose.example.yaml => docker-compose.example.yml} (100%) diff --git a/docs/batch.rst b/docs/batch.rst index 0f918ac9..85ee9098 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -139,9 +139,9 @@ slow network. A configuration manager such as Docker Compose could be used to ensure that the service is always available. -.. literalinclude:: ../misc/docker-compose.example.yaml +.. literalinclude:: ../misc/docker-compose.example.yml :language: yaml - :caption: misc/docker-compose.example.yaml + :caption: misc/docker-compose.example.yml Watched folders with watcher.py ------------------------------- diff --git a/misc/docker-compose.example.yaml b/misc/docker-compose.example.yml similarity index 100% rename from misc/docker-compose.example.yaml rename to misc/docker-compose.example.yml From e429c3d7298a8a380cde32ed42f07bfce95732fe Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 03:25:43 -0800 Subject: [PATCH 360/880] docs: install cleanup --- docs/installation.rst | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 49cec3bc..30b57f0d 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -136,8 +136,9 @@ from sources <#installing-head-revision-from-sources>`__. Installing the latest version on Ubuntu 18.04 LTS ------------------------------------------------- -Ubuntu 18.04 includes ocrmypdf 6.1.2. To install a more recent version, -first install the system version to get most of the dependencies: +Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but +it is quite old now. To install a more recent version, first install several +system dependencies: .. code-block:: bash From b3b61c152cc65d74cac054bda1f3a57e11e2da90 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 03:26:46 -0800 Subject: [PATCH 361/880] Handle malformed DocumentInfo (#497) User submitted a PDF in which /Trailer /Info pointed to the XMP metadata block instead of a DocumentInfo dictionary. Fix and add test. --- src/ocrmypdf/_pipeline.py | 17 ++++++++++++----- tests/test_metadata.py | 32 +++++++++++++++++++++++++++----- 2 files changed, 39 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index a4a41dcd..58ac2495 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -694,11 +694,18 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): # stamping them out as soon as possible. modified = False with pikepdf.open(input_pdf) as pdf_file: - if pdf_file.docinfo: - for k, v in pdf_file.docinfo.items(): - if b'\x00' in bytes(v): - pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') - modified = True + try: + len(pdf_file.docinfo) + except TypeError: + context.log.error( + "File contains a malformed DocumentInfo block - continuing anyway" + ) + else: + if pdf_file.docinfo: + for k, v in pdf_file.docinfo.items(): + if b'\x00' in bytes(v): + pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') + modified = True if modified: pdf_file.save(fix_docinfo_file) else: diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 97ea0171..270a8b62 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -31,8 +31,11 @@ import pytest from pikepdf.models.metadata import decode_pdf_date from ocrmypdf._jobcontext import PDFContext +from ocrmypdf._pipeline import convert_to_pdfa +from ocrmypdf.cli import parser from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps +from ocrmypdf.pdfinfo import PdfInfo try: import fitz @@ -313,11 +316,6 @@ def test_metadata_fixup_warning(resources, outdir, caplog): def test_prevent_gs_invalid_xml(resources, outdir): - from ocrmypdf.cli import parser - from ocrmypdf._pipeline import convert_to_pdfa - from ocrmypdf.pdfa import generate_pdfa_ps - from ocrmypdf.pdfinfo import PdfInfo - generate_pdfa_ps(outdir / 'pdfa.ps') copyfile(resources / 'trivial.pdf', outdir / 'layers.rendered.pdf') @@ -349,3 +347,27 @@ def test_prevent_gs_invalid_xml(resources, outdir): # Ensure we did not carry the nul forward. assert mm.find(b'�', xmp_start, xmp_end) == -1, "found escaped nul" assert mm.find(b'\x00', xmp_start, xmp_end) == -1 + + +def test_malformed_docinfo(caplog, resources, outdir): + generate_pdfa_ps(outdir / 'pdfa.ps') + # copyfile(resources / 'trivial.pdf', outdir / 'layers.rendered.pdf') + + with pikepdf.open(resources / 'trivial.pdf') as pike: + pike.trailer.Info = pikepdf.Stream(pike, b"") + pike.save(outdir / 'layers.rendered.pdf', fix_metadata_version=False) + + options = parser.parse_args( + args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] + ) + pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') + context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo) + + convert_to_pdfa( + str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context + ) + + print(caplog.records) + assert any( + 'malformed DocumentInfo block' in record.message for record in caplog.records + ) From 1efa79cce274e8e813b33e54e4be045a77746137 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 03:38:45 -0800 Subject: [PATCH 362/880] Remove potentially non-free file logo.afdesign --- misc/media/logo.afdesign | Bin 26568 -> 0 bytes 1 file changed, 0 insertions(+), 0 deletions(-) delete mode 100644 misc/media/logo.afdesign diff --git a/misc/media/logo.afdesign b/misc/media/logo.afdesign deleted file mode 100644 index 023ff1850aa66d0fcbbd288f021bf35fb0eff694..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 26568 zcmZSh@9oIVz`>ALToj<}nV06L#Q+A!p!7dYFc(g=E2+eSI7$o*41UZE3;{*?CB_U4 z49aeqIT~B_|1x;2^=GUK*Q*Q@@L1Ww!nL45$4$zw%z0+kbiM z+dIB&F1hPT&HAi?h13Lvjn3{?|1%i{g8+E+TdB{@E)c3zGsSho4mv z^ZUBZxGFS^J-xTqH_mjs*do@-GuQ9;Vm|2N-A=;8b@;bXljr{Dv{#hh|R`~EKw z+x@uj*Vb!~0(e~(eVFi&gfIR0{gP|N7iY+k=}u{_`Kqt7)h{ z_9^@Ne_i`aKemQ`)YD*C%Au=MqrqUyExEjS@sawtm|Py^i+^LYdyKi<|YJFZ(=KmC}>@Bi`t z^CulOefjdAUhn<6QdYf}Bc^p9pZLLj^)6vqUeIA1hv z^M)y#WW!lJGG=)!W^mZ`p^?{bf&L1A$EH?+jGP0kycaI~{hpbe`Y+z(@c(y@;xqSL z{-2t`#IwL=vsbFtTtCf8FRhEiYM*HNq?J8%o z=zll$y7&(+2j&N%;evd6|JlF(mk7`_n7_ifUpx5g!ZVj%c|Q&Q(Wvn9%gO-N2`SGm zd8?b6&A7zr!}W9b?%n@)^C*6+X!JVW(5ZPuESNp1T7c58gx-}>M3-~K=4KleXsIkxlW!|KCr4&8q(Q|EbpfssDB6&EMLf9cM9FvhVEO`(Br>|M+jc>&Qm^8-ML{ zb$;*nm!AEvp1bzhSC-xT&-0xBH^24Y{U?>1Z2$Yud%9cc(T>&64lEK4JNbX1rbV5} zLUt1s3C7dUIaf$sm}a7QOuBhzOHvi8vc2mzX1&=u z%Q$&&*nP=!yLsmwz4yOwr3Rl{&t8-3Vp+4#d4K+Q{=tD`)(gHJj4iYYZ2$j1Q|qi) zmb6dv-f2%yUH^UG@W-=5Hg3JiAD-!|eY>@1XTbY)n_LdQ+s~PN=t#>&gP-@+bV{2R zpOrbox_kfGn9ASv0sr5B7CUTv_r6&w@ALn$x0bDW_`ss^|N0H*Rv5AI7kIcvWXxIk zghMXjNKV%K+>Rg2=l{n)5iE`g@(`M5e^xWuq~!l)nexBCTW6bow7a(?U{85wpU<2B z(N<>*!$17@zQfB_EF~%Pwto7v9Y#8_|KHEtWZL$i{(pYq$-e)w+cGwYs!n}X(zf(c z#jNH2Q@wicXNk7OTTWZ4AL=x7=~an;{6fjaPXo7Se0b4m<&nt9y;J*)K)8y3f@#-} zNA=ei9rT*=;=iqMvvJ#kkAzw#9)OH4U<@!CJ@()e5d*MIr1R~!*_{{M9S z**weIj&M%Ay2Q}ODmycDj=oIz$2G^aZWyN3Jl^WroxkV5eQw&?_M@NY8hZIdV3!}XO!?z z>d=b%Qm?&A!}FnsQLAQ9<*##wTYtF~i`Ff_#A5$h$5(ywsgjqwG__~8@z2$%aS4cy z|6kg&K!1@Fn?T313?o;r@I;GAMKgYdXa;a^hT@#@Rwo zMgQVETmSnN$9(>8xuMZ*<*6XMsVhHB375}ojjCP!L8R{CO0U|b#i^5y$;kCOP8JPt z+;+LXTs+h9j)kzysiR$+c^0t9Zm6ic#;u!Rs^H2Nb?0D~G8d1@%4*(6v%&djL4#`{ zU!U%``K{_wb%h^uQm=pc&mHB;mi~lg!X{`^ji0jb#j(B(hXhr=bmZ{|tvLA7RjfG5 zp@~%_fz|Pdq`0mCL)YbQmN`2Dt6%sib4&_ZA~Vk~UNbbRcIgG4{gb`6{_KY0@wx1fNNQvDQ#faHaCzry#)};sH%elKwJ*&Gnz-V!<-Cyih`uM& zeJ%f-m^Aa3-}B6^Mr-}lq@Eo!IPu7#s5kW5S8WHtTQsti_l9)p!28Se{}lSSu>n{NHeD8n^A-jRyTsAFtYc;s4|uY3Ab1 zq4_qS>$iu<{`fC@;eTX+=(I{D9?1(y?V^bXY9?gt?nrcUKk`TLhm~PO~g+QV-Gm zK$r7x7I0~+IC3WGPU=vcwVd}ruyG5Ml|su^H@=W{5>7#kZyR~TeFB^Lrt*AD<-X!N zXR3t%ffE-xH-&7yX5q+Ycr;kqTH2|FW6^*6Wz}~RnpdCJePB}TuC?+^XyiQgpryb3 zmK1$m>KFQBu3yogOrbm;dK42rf`w>QP+Ebh^iA?Y@w#klN)}HUBtF;`-Wp^Gvdb)870D zdpCoi#Vx8w`un2#^xiRSDA|+1RQH>`T*dQA#w_)#8GZ7VMr-#KPUW%>k@LK-r8Cth zMDyz+qh|T*_oUkTOxJoPbIv>C!xeacrPcBCCUYN#TOGdI&U9S(rOV7+>pJrY6fXkAKIOzqMj*JXZcU7a(k4s^&$7j39HBI8!z!rSdDv>|}E?u^ug#kQZ6 z<3BP5moOJJUlnrPBam&qN??z%{hONkEWaErb!VD1p+$O$ z!Y-@j<`o+PT4pc{EfUl^s(S6kg%)<{3r=iCf9pGad3|3_F`BD#$|hLzs{6C6ue6;* zfAp$f`MLCv&PWYLk_r)9r){*6yeoC@q3t9R@?__XDZ zSAt*M`~NF`-#4<5x^wxj?&-O2{%`-4Qt0n+@#f4$HK(+MSB5MLSXun%#2O|0Ni1#i zy;7I$6HxJ;q&2^;ZrPqA8`rc|XpqFaa`ic0K)ieZ`RaTBzW<>-$?? zsa$;T|8b8Po?rKmCRrH1`0wS@R_yqH_l?zJ%~4T0PwZDt-kZfO7kgX8<&v9pa6r&8 zot=&;UuRwywwY}5^VO0CJa@I%>wjon62Hxi^}oGa>+FVKbFWKQY2LkGIrrhe{~rG{ z(~>{@50?CDwo6bd!X++SSMTVPtWzIl9{+cD`2W9a(p9nMylv)duK%5T|I*)j{S4%k%$YvTu&HOT78# z7yI$;|C=xVzwOE|@c&-EYt!rh?h8+sq#pdQJ=e4Q=mY&<_g_YAmVB5{CbBmBm%RI| zb&oS*FaOuh+_g?waN_@1uWLDOnv%!eTs}+)JX}9*N>8lv#259``%G`XNISRn4HD;l`D16W$4V@^7{WYk?+SA|8L)GwEg-2)OS4ZUj8fFxJ^^b1RN+JgB{Pnb5_;O-9=<){6(Jr}8*GR&bc$G53SQVFBGK`F(|d z>T^mj^%YoHe6#O9BewbJ+TK(xKAB+U@=Fp$=~p_74#IH#xyhmxU@09xi?^ev7^yovLJVjE!TG+Q#bBk|#SGL;@dw zO=HRHU1M%Bi;Knc9LJqa+vj-nUY`w;mgQ7UWL&s5&3D0>6r+GkUt1R)nYMXa9*^d; z5M~z6vuaG2TDdMUG%eZsaE`MyE3m&{9klfjqU8>ckgYwPyS!;_S{IVd3opM z8^5`Ilixq*3AOz=SBXpUSrnJ*iT~*lKHn}ao(}@@>LzD;6ECRqn*4gE#@?*mo^N8t zzi(mTg#YdNn`CFyEjJhccmJTd__RwcYK8ZNYIa9Ozxi8#>fD2Q`%lkT?J+UyPPA}L z558955Vg9wx^UK+E3^MA8kcWd{Oozj~Ydynf$*@#|2~y%)a@nQ-kkP&G7r+H$eO z$x!BWn~T&u@pVTzJ56l#O}DkiZIW8;^(S7lP>1K&exbv*jgIdnOwX12@`z8jySz*P z$Nv}IlMnuxzp$rqcEdlvuKlfCnccAlAO85>dL+Ta-WU@T6%`S2U|!$EwQCw|@BZH{ z5qZ7QcE5MenuCf-F}&YCcqEI;Jo+EIMdpU`>*=f4HvZduwNt*G2Uf?PG>@M0$un_HXQkHw|Mdej~BPRCSNjc(&N_Ib$N-FW9SV3 zG_J2R!xVnbJQw<7q1TE_;?vSro?3ZTuqR+zT7}WVQGP zOY0uI;&H#cg^NibY?u6x1?(m(f-cW>`xMfu9+Vlic3-HcY2C6*l6C276ZdV>X?^m_ z z%~YAHE*dH2ec63h*yHI|{tv?^E&p^n{iMjO)R(OBI}W%^TP>0)b^6kuFy$4$&Rv$M zOVcu`U0i9k_ME|S#us}6QYXbsylLXU;g8o6udj_M5_Os)oPYe52gl9y3BJxWHPh9| zJ5i!Z?UVJ~<*$rA6MoD*G5L?jl$BGAJ582y>qT7nAAL$@+O@+=jyMVjlsK$0yS5<7 zbxKp8rgg}|$LyK~hj^z6uj*(LuwOFaz_M@m+SV-SWH@}Rbe{3bpHio%a2Rwj9PW`! zJ#*=+S0}^c{|o`9=OmB5*ifj^s&u>Y<-c=hmTvwZ?l^mA30v9X9G>d=jyEOVEY9It z`uly;rcJUl{##tQQ`wv;e5}Ruc)*sw|3$XUY~Gph-@VZ{^6<<5>KU^)wOzhd5cgH> zQ0CqL+8yRE0yLd@)RGcp89fA)Iu5e&u>II0(0FR1z!F8S#BQyKhHjZj>W(grOFErf z8lR+c39Z-MI8VtVk=eBHl#h`3i-vwR+ZhSnbrDXjYMxV-mO6cMp6GGaT`KtD^utqM zInVVlc5QoFxQ9jM!W z!86@IG0VN^lW@Gc;9zr!*91ML|L2<)fjstJtuEZ>+0jq?V-2^oRL)!KwdZ)4#kW)Q z=FgumAMSm+G49nK%^&|WSMRc|f0&^6<8+Uy$$Ql~+v0XdshX_+9qqPk!+&1a)0f2A z{=N8D{#I;q`kQ+Di<>@+G1oteC=)r_DSUjB_=biRty}>KOcuKDdN2Ri6I!9jsEiA_v%hh!uL%FlmE-Q?0->S-BuL2p8j*sg51U1_jxSJiI-_IbF6j9&EV^5zV=H;zGN@=7TL{W7o^1kw*KFL zx$)bD|M#Cd%q-Dfz?s3h&}Pp9Ym-&Xvz-6`zo`26zkBWlsVQu89aA%I73#_Tt(V&G zSVb=ooO z|F>WNx42Z%{`TcRziG!_{+GM7$>bj6{k0}_|Lsd-ErNc%XI`)=;!Ztx=9g0M4VkB= z^<7t;y({kW61~@d_v^SbC;zOUl;h22#d~two@EoC`n@+lim#=v<+fUx6yWVrlp348XuO8DqbMd>*-o^QmRk72apUs|Zwr|h> z^Oj|Yww&3ram{^JU$@wWk|L$8yY{DNocdNTy&yf~l-Qs7O?vayc$P^1wO>cyx2wO=r~C;3_Yb^c$j z7%z24c(YYj?%Ds2Mz3;ZOY2YPo%&qA|HA)xuW5TNAOHWmIMw_)@0$>|(#m}CU+d&f zr!U*I@V|O!1?x$N+C}eQu|^hcT6$B5@$dVIxw#JSL(+OmwVIE4{g1v-oRj$P-HXUm zVJwsp?U?*5vd)xNo~l z!fM{~{=S9E!XNVGV%L}&Fz6 zC3(H-8ZF$MJw*<}j~jPgkd&R-z^m|+VL_*OnuI~4=KuR&`_8SM$ob>HeZJwH&;OtE zT7&u%cK$d1=PmqPnIEopt5;DAPZ+`y#`E%mK7|AasQ+DpzR2w|w_?;PFBJ5W6@@htgMydtEn^|mv|rN` zT6okCICAP3FWAtr;jn;4#laT7HCl=)dF-oNBFugqVgWOsBvdkZ2}VWyU}!qZpy8U4 z_>h@HEnuNjE0?HNV1)aMt65v4dP5NSQn6PPETOfhQJ5pjWZ;W8o?<|GShl-}I- zzwTw8Pj0}qyg6rYT}n2X&dEB_=&^^;^@9vIFCDlLzwO)CI@N#WsnXpJckiz}z3soW z`*OY8O(ur9oS*+!w!76Pe*9Cpd}-y+ljiTgoP0EI%Dg_)W#4CA^*MNb&Bm9uvL_`b z^&UK^!sswXJt+L`zvCyupVuExdh&a}qwb`zU42*Gc9(dkf0}ygz@_KfXA?yBUivvT zZNqQt=Ucz+-+k_9%0ZcF*FHzhJz(qe@B9Ia?1}&N(_<`a78^<4`v0!)VS(X~{gY=F zKArk5KSZqi|99*8I^6%hU;JCY`+x5A7Ott1fBv7B{{P-hG{Ppr=^AgC&B8+>Rwt*l zZdjqr@=WApkVZz2A%{e=OG1gUvt)qCv57JsOomTC2{Rm9#IVUigw44nQNcM$;7Ec% zGXn<$g8;(;9tRGij(1vZX3txtANFk5U@}}?zGsE67n@kA@Uhde=gk}h)}Ab#6RESq zOG<+Ooj}#Rge;|Pv-?j<&v`oe1S?hV; z1YW%1{J-cNN5`%GUr#ZX_aC2qwdX;~iASuD_a1(CF5?n8tSFF0E z@nnn5@r~cCL#KMM&ibCQ|Ek7aPQ9G%uANLpFLl_`D?^- zKdofjNijdM83Np?^Gd{TCARU`zW8vrEhpa7)_ks~hWL`|W3g%L^M7iZzPMu+_~RNk z@26C^_oeq1zudB!iPP6J*7OsN8{$iEyhKhE2=878>H#|I5 z6ukQNnL?qMGdCEH`rj6azxurW<9mkYddJMw8;Uf4?-1la>A0?YWr;<0P5;EOlEw<- z=j%%CFSqxl_pfb}RE>3)>16X=o2sE?6wWg3QP}>XCt_RQ{qeBnT9~IQ*xGq$gK7am8HAj=| z^X=uLn`iZ!>+&yRgej!oS5=WX!Dm8{H+wyH`mXSde*=O+Jr zenE;y;_2$e>-Wwn=`_`_(~xiaQ{nmCx#e(p!dAH#XWN)IBrOZOI8D%?(zs$Cvvhy2 z?>pH8+kSrIMBIQ`+mg!aCii*C#OITvlbFyUF2@dK70?VYmmo6|BU{uXd- z=8ZjaY=P&3eX0Hqa^@a-w=A~rDmA#!FPiDwzjevR8TV$Fw%E>jsOmSJ_uI(@mt7u9 zCA2ZGt6x{n0e}61&Hh#O_atudmEY11(5A%GIAh&L>(&Pkr|YdTz8&Ki7qj!nUQ|jejXZcZwfnBzu3naL6=I`L^vh{`O~%Q8&X} zeEVKMR8d{s;r|1XFf|u5MtbV?zYu%g({U6NYmIXb35IM?U6;h@d1>SN z$bk09WS_MEo7m*arPHSWpZ?ROUS|KAx#=r*FwV43UvSHIO0&xTxq;7byX2Jitt?mP z&1&stu1oCNOl$9=OoJxX#V!o%*^w0)3VKfgGj=Wnfaf~jGG)}*aZ6*-hDj3z|J z%3sVnTD#`Kww5J^-YScC-ap^$&7j)rZlUp@|NR8dO)SrjH%(pkbdkGTjwHX}0^R#p zSz7agm)!Sto^`i_d;7>`YAI#S#EIj=CfXA)apKWvXf4zu0!SC^=AaHunUopO| zwHEg$ow4G5rfiXJcJSif3s#&8`P({}^p5$-ZMWa7t7~@YPu2r92`}Af=^VF)sV75t z!?a}|+1_6+yJN}p<2LJ)ea~MFnz`YEH^&$A#Vpg`3NT-`^Vs3OVS2M;j@hea*O@g| z&~VC+OlkWqyM?s>Z#6%;KH|RW;!~W__Vb(XJu>*ZVo|5EUt7W!Z-<>- zjTw{W8nq)oZ`5p;V6Zs7?AfH!*^S~}Ew?9hb|%~3E!eOUP2D!@0ZnFPU)4b>oTiGR<$6Y^_Q(G%%UXs`$pN)oSVe+zAWT z9{sxT*W6j}KKPu>zH_STn9q|fN0xDH?^~iiqr^jb`HV?cEEg)5PyW^Vz(u-Zd={=l4J3$e?7d(XAh_B%TebDooJRFTX9H|{c9YHkqXZjQT=879zwwQ2Sk#KKhoxu{)e9UaZ zjiN8Ff5+Nv_%V53?;mLc9)X4*7H-At?&svv<0`7&GO|zl>i;|Q#7ee37uS7?cDH=6 z=)UorlO7C=6OX!wNZJ>6HpQIGsN3=~@S9~SfKLc=H%b8{A_){_RX8xGX3uE<=y*k9aXt1@P(D_u(DRa`GeP|AM@P1pJuSD zwK?QL{k`|qEF3#ZSFY;Fd|9+}Wv+zGuO>g1>%0fTE?Aj;+g-b+C&-}ML_zP`3xk7? z5?-WH7afPQ2M8-;?p>pHNBab9>hNnv-N@x2?Dy zYV+(J8$^vcJ5!)zHmarMLE&WatnUeJ)307tzq9a z`A@>xjJ|G3=9`P(zpxdMP4v;*z;D#M{>A29n{oy zc7{lr^}gi0Z@Fg0oe0OQ7<;43558oy+<&!1`}Pa#o6?U@f4T59UG0IF%1`xYKC(Y2 zuFu$;b6Yg+P!I2)x2qFWV`uo^?ECTj(;J2p-?sJdIr%}i$+SGR-rZJvlhxEq1$UqB zT=q(WPuFA0>ex$R96mPPEAnT)oojJSOCxUAR(9j%eFj?=Mxattp$bYq*F8`tN)_W_b_AsaI5ISp0&4!Tu~KKpa<*PMkPZ6!_$&k>xcs>L~1 zy6RffIlUE+CwadwI{ zu1utiZP9(X%C(bNr{4+|Uw+hy`Tzg_|CQZBK#M^cKmfK<6xT{pMvx)~1_oOO1_nk3 zCNS3nD#r8_!ho$l4eqE*Vq#-pU|?rR$xqfxNi1PvU|?7V*2BP%q7Sk=BqKKoB=E_G zfk76co#Cks0|R4cfS)@rmlPKR0|T$8hf5Fx14uJN8wbe1)qF-`3=9mM1s;*b3=Din zK$vl=HlH*Dg93x6i(^Q|oVRz&D`KwB{PFSq-eA+k_ts>coU|q_x@ao5o$Cg&zkd?Z zmoh@!Hl>`r_S2?c?!w&Yb0;p%`981u+;6X0%Q88f^l!W^J9YZ>%{`UIWxH!P%zi8! z8oF}D3I$cw*3n^CPR`EkyESXYV8zr&Gg!Je!@*w%fBw zfkS)_2Sbb9M3u}9vkT(fdL$0^*Z)a2GBS$TRiYWaEypo%%E7#83LJ-6m6;e4Z%A1d zsg&QZHGlc?rOm$|kJrTSmuo+4c;l=Pi{lOj7X}5M?3F85-YLIdyJX3d4~Mw*H{{*5 z5@g|MGSKN@aFF;G6dYXn<)V9MSJ$KZf6wc0Y{?YXo1VR4wkk)H!2%Zs1&$r^?0+7K z?^iG|nDBgl{k%JOVqCg-90l}Jl^6tCHVX*}?fC!iHzyCzlauQ6m#kRf!NS<25VHxS z^KD96TG4si?+@O;KfnL~@BN$0-^Yc7%DzeCY*P4QEy%#(#6DTw|JlCZ@0_Ql&a3;C zS@!0JViSX-fSx7DY(Jk_CY|^HJTt$MV-{EWRP@`syV`Q>98CqDf{YCcC5pDTbJgcn zG_6{t6<71owNKXC?1rxpi{lG(kmfv)Z(l6#4+;)Gd}*n-rmn8+8#R!<+#q|+RXmd} zE%Cfl{eEvqNXUmH!u}l-LH2Skl3Jkf#bJ_4+SysG2R5I#J8gX4<}fqAjYCw_Ersnq z0(x=Xdcih=`t&dg!vY zdWM3X#3JSgO6wVwIck+o8*NTMucn~T0CISEeC^cV@AuDt^5n_Eyjcn{Q?#_TD}TLQ zuBoB1VBx}ti+CAg^u#Br2ue$zzO%D<=hth|oSdAJZ_<7)xy-QOCTITJru_b<&u?C- zH}hj(x>-F~MmMl?` ztNW2Sul`?U*qVrgvgLOg4<@`gevRwU()_=_zN)CH75#p@J^o7nWga$WwaK0H|9x5h zICH~qksNDI?qEm#WddB{ojZ0|9P5`q z|LN)J%1k+6gc*#Y`(c6LWhNgg{Lg|muo?R zf&Gtz`~?;=I?>yBLPH}rrEvcH_pkZx7QIRqhB8(&x%dAX#KfOWJ$S~2L!_UnrMu-| z!h>1a>kj6aZLa-YR`&j0?Y-w~R;^mJZCh9fp9l+MgP^io!SlK0i!`|M_x)5`wW{dz zS#$oI3dfZhR%EuhRi9^{YTo25xk8Qe(1DKr{^#Fr=cgZRV!iR!?$4s{2|qtQT@$f! zk$jXI$DyT4R#vlqzuV34d-=z4`+tpb>(ft8QgtrsYj$T4NMM@FfB4rwxBK6OTNWP6 zFcFfMufO;FqRHDuGhb?R9BLJnlRM|kZ+irkQfKGy^K@`vxE@o?TlM!sROS1Nz0tiB zWUd`%6zbt(-M{dw9gCv?OX3L*F0M!Ss^6Z#Xuc9uNv z&NZIrTCm{GTmg=AW&acn?2rECSg_`ylX>Zjj6$bU?jD}T4-3CXxC^j2N+<;d1?~9# zZg=I&rPC|k?R;KfArn*ibgHvba>8=vpOzmDj3mCzZS?;S_lM%#-;?hA7u5@##dh{=hOX|r z|L>2zm>}S$ANS_pvQ3|uOx_)upd+TTTGaQli-H2jp#~04&W9h5%RlF@|G|9T<}=T~ zfB$@D7$jcETXrO-j&bcOuCKl30p__)4`*C)n>1n5rVIbC*B=y!UZ%`(XsN2EX6O9B zZ_@Yt`E>fvzwi5>Z#*uSd})bivyqU3-CGt7J-5o=xm)(^Ihc3y!OvIPCH!_A7cVy7 z6>=2ND~ybc1eN$Uzu#=O`FbVT=Ff-2XXaRbZt7?G63uYoQi#8^uc|NaRtp*38Grw5 z{J}0HBG!8G$@Tc@2VZWu7!vR79n9?4&TF5o@XvAK4BdYolTP?uQP&gaS-4cQ*(r44 z($o#niX4Xw6g-7)-n{wenfd;Q8;{F9emXtgZTV%x+FxI|rC8>KGc3rII?nxl!|&VY zEoN;Ot@gfr^Z9Zq1w~IGwQhUK$O#c1lnThYxJx?U6&DK6S&}V958q&OM zVdgveB`TgQEjkoN=ZT- z{eCX%<@cTBTXM#yWdC2?pnwV2vu7=H^yR(v@o`2w-@@FxU1dKCwm+XNUn_F=&d%6< z>oTQGOcqR?zWD9^{$D?z7d-Ph&+F^u^!b1zPG< zJcU-ST&WYiZB6OS1P2F($H#hiC*-~eW>}!5cgkNj`Eb|%iTmHK&ih_!mGdUzvs|@# zo!hrFt7q=}5O&Dzi1N=5dz>$uyjp$w$Gg|xZ|-|m8^aep{dd?@x#zE+p8tH-eZA4g zPtrU7{E|=pSY95dH+}M@$?TSU@)w_fZ&>%GLM^^fc(IY%LwS)dKl45FrgPsRUhnl_)mv^4JbiJLc1gvY)7{B`RrsVP@i2hOkJTDw6YXYT*o z-i(YVjMEhS=e_y7Gca@sznuVMg91mho|012@_AKV_y4{txB36)^PgAY`;|;hzc$6} zzMlNdZ~5HjYO{@hj=ERB`D5F+dHaUqx9n?bUiZ(jEL?GX=CxVNie5!9%J2Cjy{+bB z*!H>mj~8al(ET^#%=76Bm;d+>XBYP^`d;ln!B5-QAI;0QTmL!s)RF}XCEM?HJ^XzB z;{TWa9*f=P{>owE7ud523_Uha8?Qpn;=BckY^iEqf zDL?M$@H-+U(R4WP@U;;Cqc4-Mgw9p8(3y4R&FKqY4QljXfBEr~edlxD45_kcW53h;flFkaj9U)&-2%#@!mj_7>U^HfxPQTHd=K%&6*mqtkF-2ax3A6@u+yuM-40-b0! zzx4C$rsj)HO-ujoTBg|PWLLY4(OoO?d}eu&j%%v{cZ%!b8yXu^A_S_ur==DZ75#YW zUw^9jyzTK|f7_$V{WeZ+ZpRwpwukSH%uUt2zqYP_mGN=gcMqBmEt*;Q^li_nIXy@1 zIrbjxIDE~=JJ^=JJpak;{^<)d-_1ERYgyrz1nJ+~*Cw92BW<;}Z|%-TTWPz;DG&2Z zLUZrR2?!nR(XMw0dM5I{v_~z(_2AQz3ztINQ(t<<@L2~cSIgOPcJ@5owY#qQU4$^5MDE!g(V3xRR^u z*1aNfeKpVLuDHE@aqj(lwbr7nsXvRRgvYb8?K9h~_y3Rn{|_d*%ROurk9%;@ zU0!wP-{;x(pDh^zLR*=4?|T^W&#eEPy@AT4qyK+#ycAAXJe`|=@BFJcMghs`J^?p4 z>?+@$yms9rH1+K7`ox%XHD|)!2R2Mv%=+g3FHybp+hmsiWZae?yR%iUj2wvN%MKsz5Sm{ZE56d@8x~ZE(b(g)-W^pI^W(c z{`D~b-TSAFP8hwK|G($m&gU0m>uX=H3o0=9G{bO3>>RE;=YL$?|LXWUlp6|5%+q}l_#S7P5EZ!a5c!}*t zx*W^zgQm>YkCix_-Q3*Rc9*|TyS**<%skuQd(NqzHoAH9=ANI=X5T43Z@VUD=cM27 z_tzgNV)-G%V3Bk7&D}4RujlUBz0oQyzTSN1Wc~M_EF!NNCSJQ55Q!0?4<3xr5o2Ov+Ew$uYte#LT2^z{ovzI> zo9+JK+P+>p%knon|7&q+aNYX&ctyJP>uc%%-IR|UZ@$VE%X>6?)^8cvoiQ2rYujHh zw?F#f)oPFAsyzOzq|d#QaCNvHy|VBwOO?{bNjcvPZrmnJomr$ z<*2w$@FW#a7xg8|Cyi!Z68jJ`ed>Z$T60+${FYDloozO8`gHf8AR%rsodZvcHrD%xp3LCwpXvRQXYQb zefQq3+-GNI_Q>1si?3T2RcLr?WpFNIgNowOmEt0jmvb{L(X49@B)m>Xm${iR_ zL`Hg8TE4sa`>botP_UGPq2*!Xr6r!AzHH8|EuE)MovQOxU6(0! z_3Bji_>uNlW|GU;n51>gw?7x7Cl1aQe-$IJnTc9n@}(+>#;4(9SQPc5_pz zjCEPhvu9~ePEH#lbX;6rE1yPbXlOi`TYfL|uPYZ9SJI;+oqwklIA!O2*>~A|_QCgm zgd8;I*EIcEEU)P5vcvqxPetFEOz-!y_ujqJ#>IT-T@gE@?YreW+__m2EheaR?ks+8 zlzfawN=oXNDnmf%RQvxw&tJK68uIMeqA_g*~Y@hZvDIe_w?`x2@C($WtiaPyzuavsAjHfzRtdxr*RqgWi>|$Egv;uwc!7M9<5$=OTS;J>e1~V z>)BLY=T$yh+n5-^)Kt5Fg5P`7s0T$>E7pEF^DQj+>D<4%8gC9gO%D%kwUrJJ+L7LT zyr%s@N5mrG%e8Vw>sPMKeEWx)nR%vZw%Fz6{@-V*Mny$=IXXIKKhNJ7F~c~WFL=4% z*Hxmip`k}_ZOvxv=3}@BExyBYkbV zgu=b={cMf`Y@Pnf3|)^FZQOP!Il3a}{iMAflUnARa%@%=h%UB$z{ikvs{5kJ*01Mw z?OJo>^^JxJ^EO}nI_>baGba!KzH^$vl(*Vjxb$kO%&pxslUZ*UTHekp{=%l5T=Yy> z%yd)ubq;Rs$9>lC9=uw;UQOY^-dhV7ENF=T|LgjcDO>jXdrtE3@ku#3NmVCq&x{2N z6ij@RpP!qn6T555qeqXf)$afID?4w`N4J#7w2O;e*Fi0DK4GwIcay$y`2v=ni)@?=g&OEQqphxtZ=$tvGJ{w zlmAYVb<@@V-VwC4gF|KSJiYU4t}m{Cq4X@@J%+pfspue<5mm%tsY&r}Xh zT(z$KN2$>nDb|&L|Ep_rIN5*JWGL9VXL>9v%h!MUG5H3G1~a3X8GM5`Za0}f*Yf?p z8SB^UeYZN7^!B(;L3-$P(a+y5^MBbf-!EUG_EL=6*`0iw=Wjj!_jdG#1jhfL^#4EF zsB3CE_4T^lebeKrR+i2_nDF4^ary5nxOF{+l8^U2?b2TN;Qzn(|D(RIUA=no(xt4c zuX=@riCw&SaV=xhs#Q6+xAo36PERtBD6pFQJ$jF;j=XUI{xb$lhtN*l#;yMmp zorU_0JpAr{9uD&inVp<%%R>8Q9RGg1eJ_5ovFv3oc9tFM??4!;J4P4hwr(uFR`Fq^{=9<=pDe=aS`s`s8qauGgxTI$D^pb> zR8x7w_Q#heO`!h!j-Mh7FTS$*)P?n*+r4bhj@in4GG}$IU3pLW?z(#kDk?_j7C+|h z?fGoAZR0wrN7bb}>jZ7JCf~e%uk+-c-PZr#c}%+Ua?QDOYTWZ~U(Ybv#>TK})uW8? zUF9Di9o_Tkl=j-me?SAuTeHQv#r2l#*fHbKp+nbdb;V8}xH@Os%>c7Jm&jcrxu>Vu zm0yzeRCcSVSsxm*pmaM+Kxp9ix6XIBo_?+R_JZR5AWrVPMz0c*be6a~dCJaN*ZwAh z!@x0_Uznv`doNGB*6+9HA5AKFw`_U%Pmh+mr@;&gCxtF{O3K)8*jyT2`qDGxXzQEg zsWzd38JVAAlolFGaPobh^FUnY;z_G-x4hQpKG!xVVYw`Jx6Lo#Ws*us74vcKvJL-t z>qmdLoVLHW>+h`cufJQv{4VR|iJe8qV>$WxpLc1mOSrl!bnR`spHGCJotx{O zn5g*lX>NZAs0cRub6<8kcjAjW=~xkpzfvxZEc%W~&^;oJ801+SUw4?KB(KmYo>{P37U*45$b^JHCb z?5nkAWo3PHbMx`!{l$~nf4c+*?EAs=_JrVfo&TGRwk%TGawlesusYkaxYz z%(c4uId;#guD&<{y)D10uDsv(`s=6r-jx9^o1!MF`8arayr_EgN#TDq4wMuYZx_sHw&+l2l?lsl#_kNeF z{{AlaubqlYOXlTeU;irUiNA7}U-<8B{_~5M;wGq+Y`_0*=Prl93tZd>71L&L9tv2z z=eORv-|vK1=iHi{TC)9a+q;?R6O@$N?r1ce-`>2~NKMf8RI$b$jePFqeX?I3X7v8O zzjoQW!?XS`*?9Q_f8U22Xql+rn9#j~W!;oXO*LPyUeMdvR`yCH z%KvBsi=%`{lw;$IKh{B^Ta42eY%ALkeMxh^JA=+Og=J^)WN1CaDxYof^&;_PNS^(w)?{`87h7r>1OscP=#XJ)D{Ecy-slhZ9m|*%tqtR9|*i`_cPN#m+yTKPqNW zIBB%lNZnYgn9H^D{!a5ODRsVg&pz+}+jw-ltxwJ+iGIGlOIJ3ozhB>Jd%y0El*t_9 z|IdC)3O-9e-zs7rvUY`(NYfMD_B|T0xuB`A*JanWQpt&K#etEUo*s z-{)R`4VsO(vf`d<-mV8cY)ek-InT0}|NHg2a%x&tNZA!HuYl?L?$))tW>4mD_Wk+t zan7$lx?E*7y^prc|ErVpzfq&{$+_)@R@Gv~`8#%BKF=C&dNq7=^4;Taetk=p%GSDY z;ar?w_TFhNEvgC~oW*$x)yJl`xc7cp;c+V@{;>VL-$!egJ{4`5tbVai*7^4{=_}{$Ue`^@Kf=+& z%4!?GXDRF2Ra|Sgh?L&7y7K*g`{U!auO@ue-_uZYJ9qDC%SR=PbYikZ+xPL zkAb1d$=yILU2b{5{DP$)W17TzCn@jT^Y8yV6&0nv9~UEk{A4eFcEr)kU*<^S=QSxy zE=zmJSpAmQ4Y$i3(Pm>+oZCZ~rtcV>#p+;y)XIrjFhc)zdvR^HNP zgObcyZneIw)n+xv%H#KXdk4psS2sW7+r$|ABbhdWpnzc_} z+>~Cl)iJSWekE`6vNNhidN=2$mh3Q=xU$9C@D`91E++uMP0m8_+2 zqsp~D_*vzwbpEta@4T!ucVy%XF+G>xU)KlP{}Fh)a`~N^Ub|;|r)=G|{^DuR%Ryn& ztPP{S9-YXuIGnLz&1FNibjIFN)11Gr=9W1&MEkmOnu^|AS##b{;_Q-?*kgNCzb6`$ z#y`|Zy<@UWe_wCDedXL`Rp;elTB4JI$y zXeo1R^TyKK^LXb^vaHlQ_F|3Q{PG1azE%|#6-8`L<6VCHuByA6n^E2!i}kyhn*F9t z{8Jqlx$x>$P&vF~Ugfd9M*<{1tvc=f-rU*SE4xH&F8glT((8It%f?<-}xAm0WPWY6_p` zk<_{K&NZD=VPKHy2CW5HupvX~@gz`Q6d0{qI`lb9n(F&y*OhMnyplJ@swR*s`?zf-8S%%vv^aR_)Ai zy-Dj{@VGtM+%Fdz>RMPTcgN&f*oKFf%`-CBq#yiV-*(Sni`#uYE-~(1+3Q`oxDNgN z``$VA|I?<4lO|uZ|EZ$ae)#XrN2i|@`4(Tdc&W!J*|2}V_3am~i~P5@rghstPLDgf zqUWTH{f^m-{=5}=d*AEx&Av}EbMJY?oU#Zy=Bu80W=fv=y^>=ICWd*hoZR}(J^OLc zR_5gIzD0L*P0Xv#t@!epUri@VWwDW(u~xCr!$8g-?Av$+U-G&oO+Kc0GB@8`#b(#> z)Y^%KC(rDCDAy;Oaav2=(`f3_jZ`rbCtBt{q)4_d`#>@OS|2J(vy!YDwJDV-qnWKKi zJ&9etw)A}4+r0VD(pIc;($#SI{fN8V@`14Wol5pyk4l_UQ*ZB#)yv3uQB|MFkeL(t zm7nijO#7w8A9tiA4&9mkPFbpkDD;xZrwdpHNQ`0%T+qf;gkMtwl*Q; znaDApnYW*c_RHToe6M%@!wr?o%D1t!7<@lAf5HSa#?7h>%@Tm4F8!J{<-C3Q^H&B^6O)oBZrXI=?99g-U$IBJ&9jxhd%mvi)9(AF_2=%yT=BR2 zct=8{#U@WC$gRzKzQ^u~Dw?XSASGcD9n+?}_e)6^+J0cj%FFxy`{vts=ht+*%s<|m z8tLxg!lNi?*fYOUbk(Ya?7Rg#_wg;i+^4F*(OoaVwwRN1pV936jYczPIsGJ zdvofFF7E1LW#8UXjjry72L^w17}M21mdrCf|49DGn$`6$YTwD-Qt=cDey*#SKJR%= z>n63%W?4DoSw3ocyIz9uGP(P~US3@OwqK|C=gu@pWO{tO|9xS+?`6C--WVwMUu$Gu$cvnNRA1X-7`)Em-J^1Q8>_~Pk;XOuVlKV>;&Mbq zX!4AE4@G-BH?L2<|17lTkNv!x>bYmXKFN!8{nZkwz`xPGpxHxL%PDr-nc_wBRTlcp z?|azHKVi}B-V-lMPMF_q*;JCbU6^_6?V!Thx|Ci{$e2Ok5qUOHSyjzsIV6^37BI^}nCp>{%H5zef17TET7odG6N}ckEfYZ`Pf)@ufvAZbw%xmN|R> zVWq8oeOk@?GEnYFeSUvc{J*_->nG0)erA+<@9aF?D<-e9Ppy2tFK)iw3?k6$}loE;jPnU$4w_y3uh##gRgtGdD`X>^YD=kt9bqPd^bV)p+M zyxzBfp4-++T~F>`xaaS-=R14dO0LH~Usqt|)4Ab`()AlRPMn)- z9l5W@^7r@m`FG#Fe}Dez^!UE3)1L=6Hogm)di3bwZBd;f^8F$*HY%4(<9M{Z{Rh) z+}zaK)Wp;$Z|~>n$yv2`o}jXuLE$5pXJ==-CnYIu(pjv*b-YhjSzo_j*}X3yB;?4- z;N=|r{OWpoeP6zmi0MW>31q#Jw>>yI+B-H@mY0|J%GIkAXUw=!b*@juHSzJWUb&M; zORQwM#dKa=7j?UJYwHSiKZWJno=%iUD=0?jIvt_=C z_^xrlK!vHH`$nIv^^Y%?{cmp1mru8{iI=ypvzZ=SHq-z4L?NLIwjWt`En4(&rgQwH zgeM-~KgoXlvB2M}>r-atr#aPv0zxNuWOCbHyHV~wNo83~p|b1m?G-u)S8Lk4=7vfN z8g{O|9pl%2{Eke)`XH}UXB5{=?pnm`JL$=})ooc9g3o`wmAyVNBElmjMTOt)heFuu zss8qVr@Xzr{ql_$k~VW}Dktr&{yu5?^yAL_wk}CYO8@^nxBvO+w7#Z}PR?2`u7Y1* zGK-3fBX^hSR#sMmRw?z!yqwoCcmDkIyWj8gc6DX_`t|FTt5-K>TvU3u|9_p1*!$zU zp`oFFzSsZX?#^(b@LX{Hg*kp2`t4dHI?VfA@_oe@f601Xp58vFJ z&cF3-(3QM9J0@Dc+p+jEL&E20XXSn~#ngViI>YDh9LdmWdp|X1%(`ZO`fH5dlzuy> z^5e2{vzG=gzGKdK_(T5w|EFGair@M2^77uc*Fix+6%Sj**>pbNKQ&c*=ciNJH}_VX zuUZus6u;>6d+FV0{gP|vUN_ovV1q}`vU|GIO(lF-+AG5vLPA@2tqSVTow`&#dUDBJ z<&#E@b3=sp3gq;D6?qeLJ#lr*{Q3pwZ!((tF5a_e&+T*nWu~7#dTy?D<@>$gcfI{{ zJ}`9Z-s6ti&Spjyeap4j%9J%9lf-j>mxQEdR)Eu;)&AhvDY7$R(l&s-IXn> z?rI7Pz50`J?~=E7UT*b0!zfVmEac!btKB+RIQfpo3fCGeS-#vj?TiE$7gt9|2dKgE z=R>=_ijvZz=gsGvUud*`i_uf>dSvT-`;mN$U;c}$J5%P$WhcxyZg%wPqFslKSxbT! zmzOzuPTKH;-{+C6@8yCUxi`+InO!vz4Gk=~l^Zg3Zpvrz{NhIoo-1=soI|0RaM;nVC2Oq!hDSGPl?Ck94hBtTY*l{DdPVm5;oyE1Y zXE^%wP%R2G-e~xF1mSqIQa95@l>|%_Xvtb{{tT_hBDo6O74(Ufu z+xdFk?i+in%{Ql=RkF3+ds>FE$sj;v`f26ACOf8|jr`Mm;cJo6yfv2-c{#q_Sf-^1 z9=Bh+HF=Y1<^8WxzG7$Um?yU!P0UWZZZ03ZBmMAAGtC>bq*t!hTKeQe-1J z{^Xi;{K=PJQH_m_fBrnTPrtXPQtt4_%ggG>#~5TD6f*;ODnRkt@+3y z#KXo6DmK6V&Pv^?v~E>F4vmO$BF!1vXx>W9aB8 zurJGaleTHu&7M5Z=PzGRJ#fI=n1RFOAaAn6)dTqw{h}EM-UV;?sv@&8uR5e;15;nf z!Fj<2=R-VXTC5Z-EiHX!o4uV>EvFyjVsbWn(d=^xF8L zp|6a8{`_t8aKm;`3v_$k%gg&V`T2vE$?Si!{BD}z8?TfUmFqFZz1P2;Ipd?|Kkv*e z(`=h<_v?PY^_g#X_wl;e_17m)oqF`e#l<(buFg$av~%aoUAwH-glVfy?p(cYS6Axk zX$+Y1YU5dlAMRARdVc+Lh6U0=RmblLi1+91E{$wbD0DtKvnryP zrEAsR=bbx3Zp3rtziWt(XMDyjqan$&S2{_w<=--MiHp)U^&jV!Nq-YN?{`~#)v78{ z{z)pFi%&i>ysv6-qtE|b-QudE2i}md;VRBjN{+D^Yf-7 zpc!rN{|96lu3Wz^T>Ne4_a;{EAHVPae>c&wsz6P%kAR!KUQXIXMVLyJL2l$pOYpZOn9-c>7jsW z&->T4%`dx{m-o_7wZ=E9a5SIFFiL(C@YC;dNRIE<6tKQn|Jbxof8 zykNb`>kTXweCt`;tNlb+zAxs8E>@n#_@bhA>I}2S!eDWp2x+V8clQcKFIZOXy_Y?0 z-?B`Nx6^*Pxwsglo)X#j_uK7jxl5NWeRF$z|H;Yf)w-@cJUl6Xetev}PUMBTne?e$ zr5gKxiAL4co&Wv4?PiaN+`EmQ*?I2@e{Q~Zsp;3pRxI zD@*(0>(Yf=wNEV0)avCmWxn)N%12BswvPE=;>8`tvrn&Bb?9ENwuZLl9V?TF9p%EA z!p|od7M^Rc?R(*$ws~gc=CqBKpPyZ;l;LY{=9M;^FlUZWT--eOyv8qIzUXDo`EllH zQTdMNMqe#u-lf**>aO_xZQ;IuRBhA1qtNwY*)s{U1SG99|l(R-I=iJPnn-{1Y^juHZift(V)V*LyYet77ms=BhYR`OpjSIxgI zHS4vlJmZW%e>!_Uzqyh+cixE|D*X%&g_fFk{^}+dKU--2rMPL9pU0#lH&%W=aew2q z#g)0`ZO8fEDAfOsTBi45-bFi+qsNX-nmYBW6a&K?tJ1E1yI&obE^Yh6otT*DGsEEF zp69YFcbDy4{X6zW`ZGtz7kPUwW^dk>e0+ls6N{4I1plBz2|4`p6O$J!)Fw>7&wXfu z=+WX4&kPUxdKaIP6+$Xl-=>5-ks+v2me4JkC^W(A!Xl`bO zMeUyN`3w5bOk;R<_lt1(#m$>Tjvh6iud{gDrY*CT+xr?C7(mVCgxyOEO-)VX4*w4f z3`{9qBTywcQC#PP6yt?crvvx@@?v3dx_)SF!QqGw{!enxrZ?;k+q(-W5>2iu~n<8o|Ij9 zS>iL->Z-EGju{@2kuTouzrXj{e%aF3Vda8b61`X$>`wKxsEZ$%nmIv;k5h4iAm{eJ z_Vjg42^(XAB%Z0A-8XFlUu%Vl@8iSM<&sZNe}5?=y!CQozOUBem0D%d#_4_{pB|px zuULNix!mznR<_3LcXmJcaBjXOd#me2qdcj$GHn5o>67oQjoubik#s$b$Sr}NgYzm-_oYDOAcY#v5%$4X{3 zC#;xo>GDPU-#NG0=JIJ82|ao|F*#|^YR+3ZE9C2YcBS=JhaNIeWooc9czLv7ze0(c zW--_1bvq~bYX7^g%u#S-w&jLmUG}nF+D~Tw`IEV7z1wS*#q(7>dt_|qonF6x_i@$Q zhZi}0VhgyhrXDrfI@=~Ad)4y;=SpR6?Br`c$-nrft>eY3i7pEHx3hwd$GdsBys6xq zzxSZWs~*@34=lPZr)E~!bNtD!b1zD&jMUCQt+uZaS-kV}(>A|qZr{?-9WrKCc9ZQx zryQU7_x{(@2DK4aJdBE;C2!ketDc{^M!o*G)T-0|rNvJ!2uMhrcsf1)-Bw#8qe+r> zQ@HkPSj^QpJ$>tB9lJLVUA;BWnph91q&HM{3M*%w!GN#f6+=Dlg( zy(T^R`1D-h{n};xbN$ZG&HalM#|I@e=t*?cX_wyH@Vdg*fZ?9s~8x;yfApSAt?UTMZ`!wYTSW=v7w z5Lfk_)Zu&TL0|mdGbW^>y^VTBgtVS2NZ}7#JUtaLKH62>qDQVhv@$oWYxycVg zH(p3EGOoNQ)Ia42+v1rG6Xwl-%+AtY{FC>4K+Kf)6GbncxBrndm+AdZw(!_eQA^9* zQ0BwdOIGFTD=RnO&fkAGIXZo=S;NVcPp59n%gMb9m46`9e)!<@_$o0QqnSQuW}B}+ z{q2j3wzhNZ)>FlwV&m@pEZxV){_50J+xO|)7hk=$Xs%W3uDZRumlW)t#eO%ixOn2e zAB(nHOuTlb<=R57$4ebp7@8D#B=%l(ae7rD6m)#?jyX4S&syZ1-7&|nC(Yg`jhk(9 z{C?Z`YtlVO)caNKDs?um|7R}qP3(>CJB0&J(vO~(x{%zNH2LPCC6?E}?Μq-Lmf zxVUmrT-%?$?`B`@%iI0dZ2o!M!bdJ@esf-2-lsOXvz1$XQ^7;0`QPsEF4uQ;b(Omf z9)ExP`wvI%@3J}9<8D8HY#kch{Q1@D-{0rRuF&7t@#|4Hb5-v`uxn!ktoAZ?*9(aK z-xogV_1mw1zOemSBmVr>SMHZbI)!!O_uX0i0koENzMbsh=3RX=&ydNPx@ArDloO#%4bDVong2JUMD?KJo zjGdM%=cB{`+RV{*ZEf`NRiUf*Y@09rYyY!Z*%w|{Uw1vRIsNyZ{K!bjNh*R;Qny~e zj>*}gxa*OF4zmc0qePpSPJ~0oiu84C%?A(6u`I6maFD&TzhC~s>;3-U2L+FG4ahG(JB+|NQUw`|JvOdVSLQdnUGWiwlW9t>%2b z&#LZ^#U`D{1r}{N?3F8p!J)yhDz{O1YeM<1&jS1Ab6b9V*Ob(3oByv(Ixpe*xw&U% zoA>W5em-f&j2EAm*gt%v&cv|L-{pv29h;S;+zy_QvwP3gZm{0G(CuQvZOWPx$du%)Vclp56}Ja zS+pwk%8I}v$Bupa_xpYF_S>8s91?HTVhd(1EBy5Zv`Oyb>e5SxJ}g}EVf);<<^P_W zSU#9*YoK_{c9s%{_$&^FmN-u@uSd&f=Oz98^mNaML)-sF^~sZzyuDwq zeb%0$6f^0R^ur1KQ`5RvEm$L%rQJ{mt;f3hrlSjVq`)<3;clNQF#_2Xj?Y!z~ zn`fRl;c@8Dp|+B_C5Z(h*GqM5?pX3##DR7o7Jzh_PCn^UQetuuv;{@PQDTx(t?jpvaeR?p=(vxS@nKkp{H)AJt%=~2&hghr|L|Mk14%Ecd^|B;NfHPE+0*^B|r33ti?KKgwCJUzrfy6+%4L>!$HUK$&uVVohS?@5A>rYh z^Y7bj+Ps;Qk56vHYjAdT-0?t6H|ofD2DTgK%1OB MUHx3vIVCg!0FwU0C;$Ke From a2deee49209a389ca2ce8eb3cc4886d979e73e13 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 03:37:39 -0800 Subject: [PATCH 363/880] v9.6.1 release notes --- docs/release_notes.rst | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index cda2f118..6db2bae3 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,30 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. + +v9.6.1 +====== + +- Documentation improvements - thanks to many users for their contributions! + + - Fixed installation instructions for ArchLinux (@pigmonkey) + - Updated installation instructions for FreeBSD and other OSes (@knobix) + - Added instructions for using Docker Compose with watchdog (@ianalexander, + @deisi) + - Other miscellany (@mb720, @toy, @caiofacchinato) + - Some scripts provided in the documentation have been migrated out so that + they can be copied out as whole files, and to ensure syntax checking + is maintained. + +- Fixed an error that caused bash completions to fail on macOS. (#502, #504; + @AlexanderWillner) +- Fixed a rare case where OCRmyPDF threw an exception while processing a PDF + with the wrong object type in its ``/Trailer /Info``. The error is now logged + and incorrect object is ignored. (#497) +- Removed potentially non-free file ``enron1.pdf`` and simplified the test that + used it. +- Removed potentially non-free file ``misc/media/logo.afdesign``. + v9.6.0 ====== From cdf5afa75311937a1b06911f58ed4f77f1b8cf97 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Mar 2020 11:56:10 -0800 Subject: [PATCH 364/880] reqs: update pikepdf version --- requirements/main.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/main.txt b/requirements/main.txt index c45fee40..5b56bd04 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -4,7 +4,7 @@ cffi == 1.14.0 img2pdf == 0.3.3 pdfminer.six == 20200124 -pikepdf == 1.10.1 +pikepdf == 1.10.2 Pillow == 7.0.0 reportlab == 3.5.34 tqdm == 4.42.1 From 378e4dae3ba77ef3f0387fbc6e6a885c1db7bf6b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Mar 2020 13:37:44 -0800 Subject: [PATCH 365/880] Expand documentation for subprocess.run() from test --- tests/conftest.py | 42 ++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 40 insertions(+), 2 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 37054626..8724ead5 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -97,10 +97,48 @@ assert ast.parse(WINDOWS_SHIM_TEMPLATE.format(spoofer=repr(r"C:\\Temp\\file.py") def spoof(tmp_path_factory, **kwargs): """Modify PATH to override subprocess executables - spoof(program1='replacement', ...) + spoof(tmp_path_factory, program1='replacement', ...) - Creates temporary directory with symlinks to targets. + For the test suite we need a way override executables, so that we can + substitute desired results such as errors or just speed up OCR. + On POSIXish platforms we create a temporary folder with overrides that + are symlinks to the executables we want to run. We do not actually override + PATH. We also set an environment variable _OCRMYPDF_TEST_PATH, which + OCRmyPDF's subprocess wrapper will check before they use regular PATH. The + output is a folder full of executables we are overriding. We can override + multiple executables. The end result is a folder we can use in a PATH-style + lookup to override some executables: + + /tmp/abcxyz/tesseract -> ocrmypdf/tests/resources/spoof/tesseract_crash.py + /tmp/abcxyz/gs -> ocrmypdf/tests/resources/spoof/gs_backflip.py + + Windows needs extra help from us because usually, only the Administrator + can create symlinks. Instead we create small Python scripts that call + the programs we want, implementing the effect of a symlink. This is cleaner + than creating Windows executables or trying to use non-Python scripts. + The temporary folder generated for Windows could like: + + %TEMP%\abcxyz\tesseract.py: + (script that runs ocrmypdf/tests/resources/spoof/tesseract_crash.py) + %TEMP%\abcxyz\gswin32c.py: + (script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py) + %TEMP%\abcxyz\gswin64c.py: + (script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py) + + We also address one quirk here, that Ghostscript may be known as gswin32c + or gswin64c, depending on what the user installed (regardless of Windows + itself). On POSIX, Ghostscript is just 'gs'. We handle the special case here + too. + + All of this is intimately dependent on the machinery in ocrmypdf.exec.run(). + In particular, for Windows, that code has to know that if there is a .py + file, it needs to run it with Python, since Windows does not like being + asked to execute files. + + We don't overload PATH directly because we have some tests where we call + ocrmypdf as a subprocess (to exercise the command line interface) and some + tests where we call it as an API. """ env = os.environ.copy() slug = '-'.join(v.replace('.py', '') for v in sorted(kwargs.values())) From 0165255bd97e4f743d59f70484b780dc906f92fe Mon Sep 17 00:00:00 2001 From: tlwhitec Date: Wed, 11 Mar 2020 10:57:37 +0100 Subject: [PATCH 366/880] fix install instructions for Ubunti 16.04 (#507) `pip3` defaults to the system's outdated version which downloads wrong qpdf package. --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 30b57f0d..08345ba2 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -232,7 +232,7 @@ environment variable contains ``$HOME/.local/bin``. .. code-block:: bash export PATH=$HOME/.local/bin:$PATH - pip3 install --user ocrmypdf + pip3.6 install --user ocrmypdf To add JBIG2 encoding, see :ref:`jbig2`. From 5442c97ed82ead9599bc1f275c9ff52a8416b11b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Mar 2020 04:03:09 -0700 Subject: [PATCH 367/880] Consult ICC profile when determining image colorspace --- src/ocrmypdf/pdfinfo/info.py | 30 +++++++++++++++++++++--------- 1 file changed, 21 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 964c3036..75768639 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -303,14 +303,22 @@ class ImageInfo: if self._enc == Encoding.jpeg2000: self._color = Colorspace.jpeg2000 - self._comp = FRIENDLY_COMP.get(self._color, '?') + if self._color == Colorspace.icc: + # Check the ICC profile to determine actual colorspace + pim_icc = pim.icc + if pim_icc.profile.xcolor_space == 'GRAY': + self._comp = 1 + elif pim_icc.profile.xcolor_space == 'CMYK': + self._comp = 4 + else: + self._comp = 3 + else: + self._comp = FRIENDLY_COMP.get(self._color, '?') - # Bit of a hack... infer grayscale if component count is uncertain - # but encoding must be monochrome. This happens if a monochrome image - # has an ICC profile attached. Better solution would be to examine - # the ICC profile. - if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2): - self._comp = FRIENDLY_COMP[Colorspace.gray] + # Bit of a hack... infer grayscale if component count is uncertain + # but encoding only supports monochrome. + if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2): + self._comp = FRIENDLY_COMP[Colorspace.gray] @property def name(self): @@ -805,10 +813,14 @@ def main(): parser = argparse.ArgumentParser() parser.add_argument('infile') args = parser.parse_args() - info = _pdf_get_all_pageinfo(args.infile) + pagesinfo, pdfinfo = _pdf_get_all_pageinfo(args.infile) from pprint import pprint - pprint(info) + pprint(pdfinfo) + for page in pagesinfo: + pprint(page) + for im in page.images: + pprint(im) if __name__ == '__main__': From 99653fcd32445a4e37a3ba30f2cb8704098ab05d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 15 Mar 2020 02:20:44 -0700 Subject: [PATCH 368/880] optimize: consider ICCBased 1 bit for optimization --- src/ocrmypdf/optimize.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 72022741..839a8ff9 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -155,6 +155,17 @@ def extract_image_generic(*, pike, root, log, image, xref, options): # generating a PNG from compressed data pim.as_pil_image().save(png_name(root, xref)) return xref, '.png' + elif ( + not pim.indexed + and pim.colorspace == Name.ICCBased + and pim.bits_per_component == 1 + and not options.jbig2_lossy + ): + # We can losslessly optimize 1-bit images to CCITT or JBIG2 without + # paying any attention to the ICC profile, provided we're not doing + # lossy JBIG2 + pim.as_pil_image().save(png_name(root, xref)) + return xref, '.png' return None From 9be533b5f4a53c9a735a218fe914bdfd0b6bc88e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 15 Mar 2020 21:45:51 -0700 Subject: [PATCH 369/880] watcher: allow all parameters to ocrmypdf.pdf to be passed by JSON --- docs/batch.rst | 94 +++++++++++++++++++-------------------------- misc/watcher.py | 13 ++++++- src/ocrmypdf/cli.py | 2 + 3 files changed, 53 insertions(+), 56 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index 85ee9098..7144246e 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -99,11 +99,45 @@ and all inquiries are appreciated. Hot (watched) folders ===================== +Watched folders with watcher.py +------------------------------- + +OCRmyPDF has a folder watcher called watcher.py, which is currently included in source +distributions but not part of the main program. It may be used natively or may run +in a Docker container. Native instances tend to give better performance. watcher.py +works on all platforms. + +Users may need to customize the script to meet their requirements. + +.. code-block:: bash + + pip3 install -r requirements/watcher.txt + + env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \ + OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \ + OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ + python3 watcher.py + +.. csv-table:: watcher.py environment variables + :header: "Environment variable", "Description" + :widths: 50, 50 + + "OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)" + "OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)" + "OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)" + "OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``" + "OCR_DESKEW", "Apply deskew to crooked input PDFs" + "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``" + "OCR_POLL_NEW_FILE_SECONDS", "Polling interval" + "OCR_LOGLEVEL", "Level of log messages to report" + +One could configure a networked scanner or scanning computer to drop files in the +watched folder. + Watched folders with Docker --------------------------- -The OCRmyPDF Docker image includes a watcher service. This service can -be launched as follows: +The watcher service is included in the OCRmyPDF Docker image. To run it: .. code-block:: bash @@ -127,9 +161,9 @@ convert it to a OCRed PDF in ``/output/``. The parameters to this image are: "``-v :/input``", "Files placed in this location will be OCRed" "``-v :/output``", "This is where OCRed files will be stored" - "``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "This will place files in the output in {output}/{year}/{month}/{filename}" - "``-e OCR_ON_SUCCESS_DELETE=1``", "This will delete the input file if the exit code is 0 (OK)" - "``-e OCR_DESKEW=1``", "This will enable deskew for crooked PDFs" + "``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1" + "``-e OCR_ON_SUCCESS_DELETE=1``", "Define environment variable" + "``-e OCR_DESKEW=1``", "Define environment variable" "``-e PYTHONBUFFERED=1``", "This will force STDOUT to be unbuffered and allow you to see messages in docker logs" This service relies on polling to check for changes to the filesystem. It @@ -143,56 +177,6 @@ service is always available. :language: yaml :caption: misc/docker-compose.example.yml -Watched folders with watcher.py -------------------------------- - -The watcher service may also be run natively, without Docker: - -.. code-block:: bash - - pip3 install -r requirements/watcher.txt - - env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \ - OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \ - OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ - python3 watcher.py - -Watched folders with CLI ------------------------- - -To set up a "hot folder" that will trigger OCR for every file inserted, -use a program like Python -`watchdog `__ (supports all major -OS). - -One could then configure a scanner to automatically place scanned files -in a hot folder, so that they will be queued for OCR and copied to the -destination. - -.. code-block:: bash - - pip install watchdog - -watchdog installs the command line program ``watchmedo``, which can be -told to run ``ocrmypdf`` on any .pdf added to the current directory -(``.``) and place the result in the previously created ``out/`` folder. - -.. code-block:: bash - - cd hot-folder - mkdir out - watchmedo shell-command \ - --patterns="*.pdf" \ - --ignore-directories \ - --command='ocrmypdf "${watch_src_path}" "out/${watch_src_path}" ' \ - . # don't forget the final dot - -On file servers, you could configure watchmedo as a system service so it -will run all the time. - -For more complex behavior you can write a Python script around to use -the watchdog API. You can refer to the watcher.py script as an example. - Caveats ------- diff --git a/misc/watcher.py b/misc/watcher.py index 58fe35e0..a161ede2 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -14,8 +14,10 @@ # You should have received a copy of the GNU General Public License # along with this program. If not, see . +import json import logging import os +import sys import time from datetime import datetime from pathlib import Path @@ -33,6 +35,7 @@ OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) DESKEW = bool(os.getenv('OCR_DESKEW', False)) +OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '')) POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] @@ -87,7 +90,10 @@ def execute_ocrmypdf(file_path): return log.info(f'Attempting to OCRmyPDF to: {output_path}') exit_code = ocrmypdf.ocr( - input_file=file_path, output_file=output_path, deskew=DESKEW + input_file=file_path, + output_file=output_path, + deskew=DESKEW, + **OCR_JSON_SETTINGS, ) if exit_code == 0 and ON_SUCCESS_DELETE: log.info(f'OCR is done. Deleting: {file_path}') @@ -118,10 +124,15 @@ def main(): f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n" f"DESKEW: {DESKEW}\n" + f"ARGS: {OCR_JSON_SETTINGS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"LOGLEVEL: {LOGLEVEL}\n" ) + if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: + log.error('OCR_JSON_SETTINGS should not specify input file or output file') + sys.exit(1) + handler = HandleObserverEvent(patterns=PATTERNS) observer = Observer() observer.schedule(handler, INPUT_DIRECTORY, recursive=True) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index b11e9a56..e28162e7 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -20,6 +20,8 @@ import argparse from ._version import PROGRAM_NAME as _PROGRAM_NAME from ._version import __version__ as _VERSION +__all__ = ['parser'] + def numeric(basetype, min_=None, max_=None): """Validator for numeric params""" From f35a2303bb9709d62d97b8939d5b7059ebba270f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Mar 2020 22:59:18 -0700 Subject: [PATCH 370/880] info.py: linearize O(n^2) search for use images on a page --- src/ocrmypdf/pdfinfo/info.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 75768639..ec645805 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -18,7 +18,7 @@ import logging import re -from collections import namedtuple +from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum from math import hypot, isclose @@ -98,7 +98,7 @@ XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_dep InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth']) ContentsInfo = namedtuple( - 'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector'] + 'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector', 'name_index'] ) TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt']) @@ -151,6 +151,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): ctm = PdfMatrix(initial_shorthand) xobject_settings = [] inline_images = [] + name_index = defaultdict(lambda: []) found_vector = False vector_ops = set('S s f F f* B B* b b*'.split()) image_ops = set('BI ID EI q Q Do cm'.split()) @@ -185,6 +186,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack) ) xobject_settings.append(settings) + name_index[image_name].append(settings) elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this iimage = operands[0] inline = InlineSettings( @@ -198,6 +200,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): xobject_settings=xobject_settings, inline_images=inline_images, found_vector=found_vector, + name_index=name_index, ) @@ -419,13 +422,9 @@ def _find_regular_images(container, contentsinfo): """ for pdfimage, xobj in _image_xobjects(container): - - # For each image that is drawn on this, check if we drawing the - # current image - yes this is O(n^2), but n == 1 almost always - for draw in contentsinfo.xobject_settings: - if draw.name != xobj: - continue - + if xobj not in contentsinfo.name_index: + continue + for draw in contentsinfo.name_index[xobj]: if draw.stack_depth == 0 and _is_unit_square(draw.shorthand): # At least one PDF in the wild (and test suite) draws an image # when the graphics stack depth is 0, meaning that the image From a4555b1daed4576badd06f92a35f8938516ae9fe Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 1 Oct 2019 00:45:48 -0700 Subject: [PATCH 371/880] Add halftone mask to leptonica --- src/ocrmypdf/lib/_leptonica.py | 10 +++++----- src/ocrmypdf/lib/compile_leptonica.py | 6 ++++++ 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/lib/_leptonica.py b/src/ocrmypdf/lib/_leptonica.py index 549098c1..6d6e2e8b 100644 --- a/src/ocrmypdf/lib/_leptonica.py +++ b/src/ocrmypdf/lib/_leptonica.py @@ -3,9 +3,9 @@ import _cffi_backend ffi = _cffi_backend.FFI('ocrmypdf.lib._leptonica', _version = 0x2601, - _types = b'\x00\x00\x01\x0D\x00\x01\x50\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x51\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x55\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x57\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x56\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x52\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x5B\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x05\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x5E\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x70\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x72\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x11\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x59\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x59\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x59\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x9E\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x47\x0D\x00\x00\x8C\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x8C\x11\x00\x00\x00\x0F\x00\x00\x47\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5D\x0D\x00\x00\x47\x11\x00\x00\x00\x0F\x00\x01\x5D\x0D\x00\x00\x00\x0F\x00\x00\x62\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x34\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x62\x11\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x62\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x53\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD0\x11\x00\x00\xD0\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x18\x11\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x71\x03\x00\x00\x90\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xF9\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x8C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x6F\x03\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x26\x11\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x26\x11\x00\x01\x11\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x25\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\xF9\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x11\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x9E\x11\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x00\x47\x03\x00\x00\x00\x0F\x00\x01\x75\x0D\x00\x01\x75\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x00\x0A\x09\x00\x01\x54\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x06\x09\x00\x00\x07\x09\x00\x00\x04\x09\x00\x01\x5A\x03\x00\x00\x08\x09\x00\x00\x09\x09\x00\x01\x5D\x03\x00\x01\x5E\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x62\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x58\x03\x00\x01\x6D\x03\x00\x01\x6E\x03\x00\x00\x05\x09\x00\x01\x70\x03\x00\x00\x04\x01\x00\x01\x72\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', - _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x38\x23boxDestroy',0,b'\x00\x01\x3B\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x01\x13\x23getImpliedFileFormat',0,b'\x00\x00\xB8\x23getLeptonicaVersion',0,b'\x00\x01\x3E\x23l_CIDataDestroy',0,b'\x00\x01\x16\x23l_generateCIDataForPdf',0,b'\x00\x01\x4D\x23lept_free',0,b'\x00\x00\xBA\x23makePixelSumTab8',0,b'\x00\x00\x2B\x23pixAnd',0,b'\x00\x00\x38\x23pixBackgroundNorm',0,b'\x00\x00\x30\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x22\x23pixClipRectangle',0,b'\x00\x00\xFB\x23pixColorFraction',0,b'\x00\x00\x7F\x23pixColorMagnitude',0,b'\x00\x00\x1F\x23pixConvertRGBToLuminance',0,b'\x00\x00\x76\x23pixConvertTo8',0,b'\x00\x00\xCD\x23pixCorrelationBinary',0,b'\x00\x00\xE7\x23pixCountPixels',0,b'\x00\x00\x92\x23pixDeserializeFromMemory',0,b'\x00\x00\x76\x23pixDeskew',0,b'\x00\x01\x41\x23pixDestroy',0,b'\x00\x00\x44\x23pixDilate',0,b'\x00\x00\x1F\x23pixEndianByteSwapNew',0,b'\x00\x00\xD2\x23pixEqual',0,b'\x00\x00\x44\x23pixErode',0,b'\x00\x00\x96\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xE2\x23pixFindSkew',0,b'\x00\x00\x49\x23pixGammaTRC',0,b'\x00\x00\xF4\x23pixGenerateCIData',0,b'\x00\x00\xD7\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x50\x23pixGlobalNormRGB',0,b'\x00\x00\x44\x23pixHMT',0,b'\x00\x00\x27\x23pixInvert',0,b'\x00\x00\x15\x23pixLocateBarcodes',0,b'\x00\x00\x7A\x23pixMaskOverColorPixels',0,b'\x00\x00\x58\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xEC\x23pixNumSignificantGrayColors',0,b'\x00\x01\x04\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x64\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\x9A\x23pixProcessBarcodes',0,b'\x00\x00\x8B\x23pixRead',0,b'\x00\x00\xA1\x23pixReadBarcodes',0,b'\x00\x00\x8E\x23pixReadMem',0,b'\x00\x00\x1B\x23pixReadStream',0,b'\x00\x00\x76\x23pixRemoveColormap',0,b'\x00\x00\x7A\x23pixRemoveColormapGeneral',0,b'\x00\x00\xC7\x23pixRenderBoxa',0,b'\x00\x00\x27\x23pixRotate180',0,b'\x00\x00\x76\x23pixRotateOrth',0,b'\x00\x00\x71\x23pixScale',0,b'\x00\x01\x0E\x23pixSerializeToMemory',0,b'\x00\x00\x2B\x23pixSubtract',0,b'\x00\x01\x1C\x23pixWriteImpliedFormat',0,b'\x00\x01\x2B\x23pixWriteMem',0,b'\x00\x01\x31\x23pixWriteMemJpeg',0,b'\x00\x01\x25\x23pixWriteMemPng',0,b'\x00\x00\xBC\x23pixWriteStream',0,b'\x00\x00\xC1\x23pixWriteStreamJpeg',0,b'\x00\x01\x44\x23pixaDestroy',0,b'\x00\x00\x10\x23pixaGetBox',0,b'\x00\x00\x86\x23pixaGetPix',0,b'\x00\x01\x47\x23sarrayDestroy',0,b'\x00\x00\xAE\x23selCreateBrick',0,b'\x00\x00\xA8\x23selCreateFromString',0,b'\x00\x01\x4A\x23selDestroy',0,b'\x00\x00\xB5\x23selPrintToString',0,b'\x00\x01\x22\x23setMsgSeverity',0), - _struct_unions = ((b'\x00\x00\x01\x50\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x72\x11refcount'),(b'\x00\x00\x01\x51\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x72\x11refcount',b'\x00\x00\x25\x11box'),(b'\x00\x00\x01\x54\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x6F\x11datacomp',b'\x00\x00\x90\x11nbytescomp',b'\x00\x01\x5D\x11data85',b'\x00\x00\x90\x11nbytes85',b'\x00\x01\x5D\x11cmapdata85',b'\x00\x01\x5D\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x90\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x55\x00\x00\x00\x02Pix',b'\x00\x01\x72\x11w',b'\x00\x01\x72\x11h',b'\x00\x01\x72\x11d',b'\x00\x01\x72\x11spp',b'\x00\x01\x72\x11wpl',b'\x00\x01\x72\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x5D\x11text',b'\x00\x01\x6B\x11colormap',b'\x00\x01\x71\x11data'),(b'\x00\x00\x01\x58\x00\x00\x00\x02PixColormap',b'\x00\x01\x4E\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x6E\x00\x00\x00\x10PixComp',),(b'\x00\x00\x01\x56\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x72\x11refcount',b'\x00\x00\x18\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x57\x00\x00\x00\x02PixaComp',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11offset',b'\x00\x01\x6C\x11pixc',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x5A\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x5C\x11array'),(b'\x00\x00\x01\x5B\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x67\x11data',b'\x00\x01\x5D\x11name'),(b'\x00\x00\x01\x52\x00\x00\x00\x10_IO_FILE',)), - _enums = (b'\x00\x00\x01\x60\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x61\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x62\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x63\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x64\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x65\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x66\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), - _typenames = (b'\x00\x00\x01\x50BOX',b'\x00\x00\x01\x51BOXA',b'\x00\x00\x01\x52FILE',b'\x00\x00\x01\x54L_COMP_DATA',b'\x00\x00\x01\x55PIX',b'\x00\x00\x01\x56PIXA',b'\x00\x00\x01\x57PIXAC',b'\x00\x00\x01\x58PIXCMAP',b'\x00\x00\x01\x5ASARRAY',b'\x00\x00\x01\x5BSEL',b'\x00\x00\x00\x34l_float32',b'\x00\x00\x01\x5Fl_float64',b'\x00\x00\x01\x69l_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x68l_int64',b'\x00\x00\x01\x6Al_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x74l_uint16',b'\x00\x00\x01\x72l_uint32',b'\x00\x00\x01\x73l_uint64',b'\x00\x00\x01\x70l_uint8'), + _types = b'\x00\x00\x01\x0D\x00\x01\x56\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x57\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x5B\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x5D\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x5C\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x58\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x18\x11\x00\x00\x05\x03\x00\x00\x11\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x61\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x64\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x76\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x78\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x11\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5F\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x5F\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5F\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xA4\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x92\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x63\x0D\x00\x00\x4D\x11\x00\x00\x00\x0F\x00\x01\x63\x0D\x00\x00\x00\x0F\x00\x00\x2A\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x3A\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x59\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x18\x11\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x77\x03\x00\x00\x96\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x75\x03\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x25\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x11\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\xA4\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x4D\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x01\x7B\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x00\x0A\x09\x00\x01\x5A\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x06\x09\x00\x00\x07\x09\x00\x00\x04\x09\x00\x01\x60\x03\x00\x00\x08\x09\x00\x00\x09\x09\x00\x01\x63\x03\x00\x01\x64\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x2A\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x5E\x03\x00\x01\x73\x03\x00\x01\x74\x03\x00\x00\x05\x09\x00\x01\x76\x03\x00\x00\x04\x01\x00\x01\x78\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', + _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x3E\x23boxDestroy',0,b'\x00\x01\x41\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x01\x19\x23getImpliedFileFormat',0,b'\x00\x00\xBE\x23getLeptonicaVersion',0,b'\x00\x01\x44\x23l_CIDataDestroy',0,b'\x00\x01\x1C\x23l_generateCIDataForPdf',0,b'\x00\x01\x53\x23lept_free',0,b'\x00\x00\xC0\x23makePixelSumTab8',0,b'\x00\x00\x31\x23pixAnd',0,b'\x00\x00\x3E\x23pixBackgroundNorm',0,b'\x00\x00\x36\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x22\x23pixClipRectangle',0,b'\x00\x01\x01\x23pixColorFraction',0,b'\x00\x00\x85\x23pixColorMagnitude',0,b'\x00\x00\x1F\x23pixConvertRGBToLuminance',0,b'\x00\x00\x7C\x23pixConvertTo8',0,b'\x00\x00\xD3\x23pixCorrelationBinary',0,b'\x00\x00\xED\x23pixCountPixels',0,b'\x00\x00\x98\x23pixDeserializeFromMemory',0,b'\x00\x00\x7C\x23pixDeskew',0,b'\x00\x01\x47\x23pixDestroy',0,b'\x00\x00\x4A\x23pixDilate',0,b'\x00\x00\x1F\x23pixEndianByteSwapNew',0,b'\x00\x00\xD8\x23pixEqual',0,b'\x00\x00\x4A\x23pixErode',0,b'\x00\x00\x9C\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xE8\x23pixFindSkew',0,b'\x00\x00\x4F\x23pixGammaTRC',0,b'\x00\x00\x27\x23pixGenHalftoneMask',0,b'\x00\x00\xFA\x23pixGenerateCIData',0,b'\x00\x00\xDD\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x56\x23pixGlobalNormRGB',0,b'\x00\x00\x4A\x23pixHMT',0,b'\x00\x00\x2D\x23pixInvert',0,b'\x00\x00\x15\x23pixLocateBarcodes',0,b'\x00\x00\x80\x23pixMaskOverColorPixels',0,b'\x00\x00\x5E\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xF2\x23pixNumSignificantGrayColors',0,b'\x00\x01\x0A\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x6A\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\xA0\x23pixProcessBarcodes',0,b'\x00\x00\x91\x23pixRead',0,b'\x00\x00\xA7\x23pixReadBarcodes',0,b'\x00\x00\x94\x23pixReadMem',0,b'\x00\x00\x1B\x23pixReadStream',0,b'\x00\x00\x7C\x23pixRemoveColormap',0,b'\x00\x00\x80\x23pixRemoveColormapGeneral',0,b'\x00\x00\xCD\x23pixRenderBoxa',0,b'\x00\x00\x2D\x23pixRotate180',0,b'\x00\x00\x7C\x23pixRotateOrth',0,b'\x00\x00\x77\x23pixScale',0,b'\x00\x01\x14\x23pixSerializeToMemory',0,b'\x00\x00\x31\x23pixSubtract',0,b'\x00\x01\x22\x23pixWriteImpliedFormat',0,b'\x00\x01\x31\x23pixWriteMem',0,b'\x00\x01\x37\x23pixWriteMemJpeg',0,b'\x00\x01\x2B\x23pixWriteMemPng',0,b'\x00\x00\xC2\x23pixWriteStream',0,b'\x00\x00\xC7\x23pixWriteStreamJpeg',0,b'\x00\x01\x4A\x23pixaDestroy',0,b'\x00\x00\x10\x23pixaGetBox',0,b'\x00\x00\x8C\x23pixaGetPix',0,b'\x00\x01\x4D\x23sarrayDestroy',0,b'\x00\x00\xB4\x23selCreateBrick',0,b'\x00\x00\xAE\x23selCreateFromString',0,b'\x00\x01\x50\x23selDestroy',0,b'\x00\x00\xBB\x23selPrintToString',0,b'\x00\x01\x28\x23setMsgSeverity',0), + _struct_unions = ((b'\x00\x00\x01\x56\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x78\x11refcount'),(b'\x00\x00\x01\x57\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x78\x11refcount',b'\x00\x00\x25\x11box'),(b'\x00\x00\x01\x5A\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x75\x11datacomp',b'\x00\x00\x96\x11nbytescomp',b'\x00\x01\x63\x11data85',b'\x00\x00\x96\x11nbytes85',b'\x00\x01\x63\x11cmapdata85',b'\x00\x01\x63\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x96\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x5B\x00\x00\x00\x02Pix',b'\x00\x01\x78\x11w',b'\x00\x01\x78\x11h',b'\x00\x01\x78\x11d',b'\x00\x01\x78\x11spp',b'\x00\x01\x78\x11wpl',b'\x00\x01\x78\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x63\x11text',b'\x00\x01\x71\x11colormap',b'\x00\x01\x77\x11data'),(b'\x00\x00\x01\x5E\x00\x00\x00\x02PixColormap',b'\x00\x01\x54\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x74\x00\x00\x00\x10PixComp',),(b'\x00\x00\x01\x5C\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x78\x11refcount',b'\x00\x00\x18\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x5D\x00\x00\x00\x02PixaComp',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11offset',b'\x00\x01\x72\x11pixc',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x60\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x62\x11array'),(b'\x00\x00\x01\x61\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x6D\x11data',b'\x00\x01\x63\x11name'),(b'\x00\x00\x01\x58\x00\x00\x00\x10_IO_FILE',)), + _enums = (b'\x00\x00\x01\x66\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x67\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x68\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x69\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x6A\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x6B\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x6C\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), + _typenames = (b'\x00\x00\x01\x56BOX',b'\x00\x00\x01\x57BOXA',b'\x00\x00\x01\x58FILE',b'\x00\x00\x01\x5AL_COMP_DATA',b'\x00\x00\x01\x5BPIX',b'\x00\x00\x01\x5CPIXA',b'\x00\x00\x01\x5DPIXAC',b'\x00\x00\x01\x5EPIXCMAP',b'\x00\x00\x01\x60SARRAY',b'\x00\x00\x01\x61SEL',b'\x00\x00\x00\x3Al_float32',b'\x00\x00\x01\x65l_float64',b'\x00\x00\x01\x6Fl_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x6El_int64',b'\x00\x00\x01\x70l_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x7Al_uint16',b'\x00\x00\x01\x78l_uint32',b'\x00\x00\x01\x79l_uint64',b'\x00\x00\x01\x76l_uint8'), ) diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index 4d50943d..d4ed8226 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -443,6 +443,12 @@ pixReadBarcodes(PIXA *pixa, SARRAY **psaw, l_int32 debugflag); +PIX * +pixGenHalftoneMask(PIX *pixs, + PIX **ppixtext, + l_int32 *phtfound, + PIXA *pixadb); + l_int32 l_generateCIDataForPdf(const char *fname, PIX *pix, From e4cc9fcba73b1614781d398e30ef1aa098fa6841 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 23 Mar 2020 01:06:55 -0700 Subject: [PATCH 372/880] Wrong number of threads to use shown when OMP_THREAD_LIMIT is defined --- src/ocrmypdf/_sync.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a4c18b21..c397b3f4 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -245,6 +245,10 @@ def exec_concurrent(context): if context.options.tesseract_env is None: context.options.tesseract_env = os.environ.copy() context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads)) + try: + tess_threads = int(context.options.tesseract_env['OMP_THREAD_LIMIT']) + except ValueError: # OMP_THREAD_LIMIT initialized to non-numeric + context.log.error("Environment variable OMP_THREAD_LIMIT is not numeric") if tess_threads > 1: context.log.info("Using Tesseract OpenMP thread limit %d", tess_threads) From 00498282f53b77be7820198fafe0df2765cf7ce2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 24 Mar 2020 21:27:18 -0700 Subject: [PATCH 373/880] validation: blacklist Ghostscript 9.51 too --- src/ocrmypdf/_validation.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 8b874ba6..ef440e0f 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -466,11 +466,12 @@ def check_dependency_versions(options): version_checker=ghostscript.version, need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports ) - if ghostscript.version() == '9.24': + gs_version = ghostscript.version() + if gs_version in ('9.24', '9.51'): raise MissingDependencyError( - "Ghostscript 9.24 contains serious regressions and is not " - "supported. Please upgrade to Ghostscript 9.25 or use an older " - "version." + f"Ghostscript {gs_version} contains serious regressions and is not " + "supported. Please upgrade to a newer version, or downgrade to the " + "previous version." ) check_external_program( program='qpdf', From 85e6c6669a53ff439982914539752b19e6215e7a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 21:16:24 -0700 Subject: [PATCH 374/880] docs: Add username to WSL instructions Fixes #519 --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 08345ba2..2015c839 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -574,7 +574,7 @@ Installing on Windows Subsystem for Linux .. code-block:: powershell - wsl sudo ln -s /home/user/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf + wsl sudo ln -s /home/$USER/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf Then confirm that the expected version from PyPI (|latest|) is installed: From 2490be849045d361ba5c9913a42949027e361f6c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 21:53:45 -0700 Subject: [PATCH 375/880] Fix debug.log not being deleted on Windows (probably) Fixes #515 --- src/ocrmypdf/_sync.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c397b3f4..07c355ab 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -361,10 +361,11 @@ def run_pipeline(options, api=False): options.jobs = available_cpu_count() work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + debug_log_handler = None if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get( 'PYTEST_CURRENT_TEST', '' ): - configure_debug_logging(Path(work_folder) / "debug.log") + debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") try: check_requested_output_file(options) @@ -432,6 +433,12 @@ def run_pipeline(options, api=False): log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error finally: + if debug_log_handler: + try: + debug_log_handler.close() + log.removeHandler(debug_log_handler) + except EnvironmentError as e: + print(e, file=sys.stderr) cleanup_working_files(work_folder, options) return ExitCode.ok From dd1cf567dbebb7d62a3b9ff6d5b85d69a294533e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 21:58:31 -0700 Subject: [PATCH 376/880] watcher: Fix JSONDecodeError if OCR_JSON_SETTINGS not set Fixes #516 --- misc/watcher.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/watcher.py b/misc/watcher.py index a161ede2..168c9998 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -35,7 +35,7 @@ OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) DESKEW = bool(os.getenv('OCR_DESKEW', False)) -OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '')) +OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] From 8307832ce9675d3154c0af04bcf9099908e863fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 22:44:16 -0700 Subject: [PATCH 377/880] tests: add force OCR to a file with text that Ghostscript doesn't see For gs 9.52 support. Also refactor use of pikepdf.open() to use with blocks. --- tests/test_graft.py | 29 +++++++++++++++++------------ 1 file changed, 17 insertions(+), 12 deletions(-) diff --git a/tests/test_graft.py b/tests/test_graft.py index a14e3cee..65cba7c8 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -25,25 +25,30 @@ import ocrmypdf def test_no_glyphless_graft(resources, outdir): - pdf = pikepdf.open(resources / 'francais.pdf') - pdf_aspect = pikepdf.open(resources / 'aspect.pdf') - pdf_cmyk = pikepdf.open(resources / 'cmyk.pdf') - pdf.pages.extend(pdf_aspect.pages) - pdf.pages.extend(pdf_cmyk.pages) - pdf.save(outdir / 'test.pdf') + with pikepdf.open(resources / 'francais.pdf') as pdf, pikepdf.open( + resources / 'aspect.pdf' + ) as pdf_aspect, pikepdf.open(resources / 'cmyk.pdf') as pdf_cmyk: + pdf.pages.extend(pdf_aspect.pages) + pdf.pages.extend(pdf_cmyk.pages) + pdf.save(outdir / 'test.pdf') with patch('ocrmypdf._graft.MAX_REPLACE_PAGES', 2): ocrmypdf.ocr( - outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0 + outdir / 'test.pdf', + outdir / 'out.pdf', + deskew=True, + tesseract_timeout=0, + force_ocr=True, ) + # This test needs asserts def test_links(resources, outpdf): ocrmypdf.ocr( resources / 'link.pdf', outpdf, redo_ocr=True, oversample=200, output_type='pdf' ) - pdf = pikepdf.open(outpdf) - p1 = pdf.pages[0] - p2 = pdf.pages[1] - assert p1.Annots[0].A.D[0].objgen == p2.objgen - assert p2.Annots[0].A.D[0].objgen == p1.objgen + with pikepdf.open(outpdf) as pdf: + p1 = pdf.pages[0] + p2 = pdf.pages[1] + assert p1.Annots[0].A.D[0].objgen == p2.objgen + assert p2.Annots[0].A.D[0].objgen == p1.objgen From 23bc3d3a29439a05ab510c734881563638c6b9eb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 22:45:16 -0700 Subject: [PATCH 378/880] tests: workaround for Ghostscript 9.52 txtwrite problem --- src/ocrmypdf/pdfinfo/info.py | 49 ++++++++++++++++++++---------------- tests/test_pdfinfo.py | 4 +++ 2 files changed, 31 insertions(+), 22 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index ec645805..4251eb54 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -30,10 +30,10 @@ import pikepdf from pikepdf import PdfMatrix from tqdm import tqdm -from ocrmypdf.exceptions import EncryptedPdfError, MissingDependencyError - -from . import ghosttext -from .layout import get_page_analysis, get_text_boxes +from ocrmypdf.exceptions import EncryptedPdfError +from ocrmypdf.exec import ghostscript +from ocrmypdf.pdfinfo import ghosttext +from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes logger = logging.getLogger() @@ -618,25 +618,28 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False): pdf = pikepdf.open(infile) # Do not close in this function - if pdf.is_encrypted: - pdf.close() - raise EncryptedPdfError() # Triggered by encryption with empty passwd - if detailed_analysis: - pages_xml = None - else: - pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) + try: + if pdf.is_encrypted: + raise EncryptedPdfError() # Triggered by encryption with empty passwd + if detailed_analysis: + pages_xml = None + else: + pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) - pages = [] - for n, _ in tqdm( - enumerate(pdf.pages), - total=len(pdf.pages), - desc="Scan", - unit='page', - disable=not progbar, - ): - page_xml = pages_xml[n] if pages_xml else None - page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) - pages.append(page) + pages = [] + for n, _ in tqdm( + enumerate(pdf.pages), + total=len(pdf.pages), + desc="Scan", + unit='page', + disable=not progbar, + ): + page_xml = pages_xml[n] if pages_xml else None + page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) + pages.append(page) + except Exception: + pdf.close() + raise return pages, pdf @@ -757,6 +760,8 @@ class PdfInfo: def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False): self._infile = infile + if ghostscript.version() in ('9.52',): + detailed_page_analysis = True # txtwrite doesn't work in these versions self._pages, pdf = _pdf_get_all_pageinfo( infile, detailed_page_analysis, log=log, progbar=progbar ) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index b6455aa6..facf3b6f 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -26,6 +26,7 @@ from PIL import Image from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo +from ocrmypdf.exec import ghostscript from ocrmypdf.pdfinfo import Colorspace, Encoding # pylint: disable=protected-access @@ -183,6 +184,9 @@ def test_ocr_detection(resources): @pytest.mark.parametrize( 'testfile', ('truetype_font_nomapping.pdf', 'type3_font_nomapping.pdf') ) +@pytest.mark.xfail( + ghostscript.version() in ('9.52',), reason="gs 9.52 txtwrite doesn't work" +) def test_corrupt_font_detection(resources, testfile): filename = resources / testfile with pytest.raises(NotImplementedError): From 8de0f9b86f830395aa40a5df428e4c847eed6a15 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Mar 2020 22:45:25 -0700 Subject: [PATCH 379/880] v9.7.0 release notes --- docs/release_notes.rst | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 6db2bae3..c79f1245 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -14,6 +14,22 @@ Note that it is licensed under GPLv3, so scripts that licensed under GPLv3. +v9.7.0 +====== + +- Fixed an error in watcher.py if ``OCR_JSON_SETTINGS`` was not defined. +- Ghostscript 9.51 is now blacklisted, due to numerous problems with this version. +- Added a workaround for a problem with "txtwrite" in Ghostscript 9.52. +- Fixed an issue where the incorrect number of threads used was shown when + ``OMP_THREAD_LIMIT`` was manipulated. +- Removed a possible performance bottlenecks for files that use hundreds to + thousands of images on the same page. +- Documentation improvements. +- Optimization will now be applied to some monochrome images that have a color + profile defined instead of only black and white. +- ICC profiles are consulted when determining the simplified colorspace of an + image. + v9.6.1 ====== From c152710617ffb8e1808886505e74a205dd696525 Mon Sep 17 00:00:00 2001 From: jbarlow83 Date: Sat, 4 Apr 2020 15:41:53 -0700 Subject: [PATCH 380/880] Update issue templates --- .github/ISSUE_TEMPLATE/bug_report.md | 36 +++++++++++++++++++++++ .github/ISSUE_TEMPLATE/feature_request.md | 17 +++++++++++ 2 files changed, 53 insertions(+) create mode 100644 .github/ISSUE_TEMPLATE/bug_report.md create mode 100644 .github/ISSUE_TEMPLATE/feature_request.md diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 00000000..04820098 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,36 @@ +--- +name: Bug report +about: Create a report to help us improve +title: '' +labels: '' +assignees: '' + +--- + +**Describe the bug** +A clear and concise description of what the bug is. + +**To Reproduce** +What command line or API call were you trying to run? + +```bash +ocrmypdf ...arguments... input.pdf output.pdf +``` + +Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful. + +**Example file** +Please include an example *input* PDF (or image). The input file is more helpful. + +If possible, use an input file with no personal or confidential information. At your option you may GPG-encrypt the file for OCRmyPDF's author only. + +**Expected behavior** +A clear and concise description of what you expected to happen. + +**Screenshots** +If applicable, add screenshots to help explain your problem. + +**System** + - OS: [e.g. Linux, Windows, macOS] + - OCRmyPDF Version: ``ocrmypdf --version`` + - How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image? diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md new file mode 100644 index 00000000..59094e26 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.md @@ -0,0 +1,17 @@ +--- +name: Feature request +about: Suggest an idea for this project +title: '' +labels: enhancement +assignees: '' + +--- + +**Is your feature request related to a problem? Please describe.** +A clear and concise description of what the problem is. Ex. I'm always frustrated when [...] + +**Describe the solution you'd like** +A clear and concise description of what you want to happen. + +**Additional context** +Add any other context or screenshots about the feature request here. From 99ef42940c8f81b31eca1c4bd4904fb79dea5850 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Apr 2020 16:29:18 -0700 Subject: [PATCH 381/880] docs: warn that Windows users should use an ifmain guard --- docs/api.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/api.rst b/docs/api.rst index ef429066..c38bb400 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -51,6 +51,14 @@ Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That way your application will survive and remain interactive even if OCRmyPDF does not. +.. warning:: + + On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected + by an "ifmain" guard (``if __name__ == '__main__'``) or you must use + ``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one + of these steps, Windows fork semantics will prevent OCRmyPDF from working + correct. + Logging ------- From 32a88f1bad23fedc555284523c234d953f507ac1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Apr 2020 16:29:37 -0700 Subject: [PATCH 382/880] docs: warn that AWS Lambda doesn't work --- docs/batch.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/batch.rst b/docs/batch.rst index 7144246e..9d484790 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -202,6 +202,13 @@ Alternatives - `Watchman `__ is a more powerful alternative to ``watchmedo``. +AWS Lambda is not viable +------------------------ + +AWS Lambda and its equivalents have low limits on execution time and payload +size, relative to OCRmyPDF's needs. As of this writing, the request/response +payload for AWS Lambda was 6 MB, which means many PDFs will not fit. + macOS Automator =============== From 58ec56180a751c335589661d133494d94da85e5a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 3 Apr 2020 22:04:00 -0700 Subject: [PATCH 383/880] Add a few more type annotations to public APIs --- src/ocrmypdf/api.py | 8 ++++++-- src/ocrmypdf/helpers.py | 8 ++++---- src/ocrmypdf/pdfinfo/info.py | 6 +++--- 3 files changed, 13 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index e48320b5..36be4dd4 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -67,7 +67,11 @@ class Verbosity(IntEnum): debug_all = 2 #: More detailed debugging from ocrmypdf and dependent modules -def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=False): +def configure_logging( + verbosity: Verbosity, + progress_bar_friendly: bool = True, + manage_root_logger: bool = False, +): """Set up logging. Library users may wish to use this function if they want their log output to be @@ -128,7 +132,7 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger= return log -def create_options(*, input_file, output_file, **kwargs): +def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwargs): cmdline = [] deferred = [] diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 70e884d9..ae326331 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -28,7 +28,7 @@ from pathlib import Path log = logging.getLogger(__name__) -def safe_symlink(input_file, soft_link_name, *args, **kwargs): +def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **kwargs): """ Helper function: relinks soft symbolic link if necessary """ @@ -76,12 +76,12 @@ def is_iterable_notstr(thing): return isinstance(thing, Iterable) and not isinstance(thing, str) -def monotonic(L): +def monotonic(L: Iterable): """Does list increase monotonically?""" return all(b > a for a, b in zip(L, L[1:])) -def page_number(input_file): +def page_number(input_file: os.PathLike): """Get one-based page number implied by filename (000002.pdf -> 2)""" return int(os.path.basename(os.fspath(input_file))[0:6]) @@ -97,7 +97,7 @@ def available_cpu_count(): return 1 -def is_file_writable(test_file): +def is_file_writable(test_file: os.PathLike): """Intentionally racy test if target is writable. We intend to write to the output file if and only if we succeed and diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 4251eb54..d304688b 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -22,7 +22,7 @@ from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum from math import hypot, isclose -from os import fspath +from os import PathLike, fspath from pathlib import Path from warnings import warn @@ -40,7 +40,7 @@ logger = logging.getLogger() Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000') Encoding = Enum( - 'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate ' + 'runlength' + 'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate runlength' ) FRIENDLY_COLORSPACE = { @@ -558,7 +558,7 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) -def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): +def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): pageinfo = {} pageinfo['pageno'] = pageno pageinfo['images'] = [] From d13d70fd56419b12786226af0067ef66a2904031 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 10 Apr 2020 01:27:46 -0700 Subject: [PATCH 384/880] Fix version checker failing for qpdf 10.0.0 Fixes #527 --- src/ocrmypdf/exec/__init__.py | 3 ++- tests/test_validation.py | 28 ++++++++++++++++++++++++++++ 2 files changed, 30 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 838697fc..1fb3b047 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -23,6 +23,7 @@ import re import shutil import sys from collections.abc import Mapping +from distutils.version import LooseVersion from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError from subprocess import run as subprocess_run @@ -270,7 +271,7 @@ def check_external_program( raise MissingDependencyError() return - if found_version < need_version: + if LooseVersion(found_version) < LooseVersion(need_version): _error_old_version(program, package, need_version, found_version, required_for) if not recommended: raise MissingDependencyError() diff --git a/tests/test_validation.py b/tests/test_validation.py index fb03de3b..9fc5fe39 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -176,3 +176,31 @@ def test_language_warning(caplog): vd.check_options_languages(opts) assert opts.language == ['eng'] assert 'assuming --language' in caplog.text + + +def test_version_comparison(): + vd.check_external_program( + program="dummy_basic", + package="dummy", + version_checker=lambda: '9.0', + need_version='8.0.2', + ) + vd.check_external_program( + program="dummy_doubledigit", + package="dummy", + version_checker=lambda: '10.0', + need_version='8.0.2', + ) + vd.check_external_program( + program="tesseract", + package="tesseract", + version_checker=lambda: '4.0.0-beta.1', + need_version='4.0.0', + ) + with pytest.raises(MissingDependencyError): + vd.check_external_program( + program="dummy_fails", + package="dummy", + version_checker=lambda: '1.0', + need_version='2.0', + ) From 7fe06c64fc75996e9a6c01e1033bf0af8e149d2c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 10 Apr 2020 12:53:24 -0700 Subject: [PATCH 385/880] v9.7.1 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c79f1245..86a1b9f3 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -14,6 +14,13 @@ Note that it is licensed under GPLv3, so scripts that licensed under GPLv3. +v9.7.1 +====== + +- Fixed version check failing when used with qpdf 10.0.0. +- Added some missing type annotations. +- Updated documentation to warn about need for "ifmain" guard and Windows. + v9.7.0 ====== From 9471bc8921054a58685fcc3317da966cae21e4fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 10 Apr 2020 13:42:33 -0700 Subject: [PATCH 386/880] Fix versions with leading v, e.g. v5.0 --- src/ocrmypdf/exec/__init__.py | 8 ++++++++ tests/test_validation.py | 6 ++++++ 2 files changed, 14 insertions(+) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 1fb3b047..92e4d956 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -271,6 +271,14 @@ def check_external_program( raise MissingDependencyError() return + def remove_leading_v(s): + if s.startswith('v'): + return s[1:] + return s + + found_version = remove_leading_v(found_version) + need_version = remove_leading_v(need_version) + if LooseVersion(found_version) < LooseVersion(need_version): _error_old_version(program, package, need_version, found_version, required_for) if not recommended: diff --git a/tests/test_validation.py b/tests/test_validation.py index 9fc5fe39..af1eadea 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -197,6 +197,12 @@ def test_version_comparison(): version_checker=lambda: '4.0.0-beta.1', need_version='4.0.0', ) + vd.check_external_program( + program="tesseract", + package="tesseract", + version_checker=lambda: 'v5.0.0-alpha.20200201', + need_version='4.0.0', + ) with pytest.raises(MissingDependencyError): vd.check_external_program( program="dummy_fails", From 4a640b8dcd9eb0cd2e74eaf748fd65064cc639e3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Apr 2020 23:18:52 -0700 Subject: [PATCH 387/880] Fix language argument not working as list Fixes #523 --- src/ocrmypdf/api.py | 12 +++++++++--- tests/test_api.py | 5 +++++ 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 36be4dd4..88f9657e 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -21,7 +21,7 @@ import sys from contextlib import suppress from enum import IntEnum from pathlib import Path -from typing import Dict, List +from typing import Dict, Iterable from tqdm import tqdm @@ -154,6 +154,12 @@ def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwarg cmdline.append(f"--{cmd_style_arg}") continue + if isinstance(val, Iterable): + for elem in val: + cmdline.append(f"--{cmd_style_arg}") + cmdline.append(elem) + continue + # We have a parameter cmdline.append(f"--{cmd_style_arg}") if isinstance(val, (int, float)): @@ -184,7 +190,7 @@ def ocr( # pylint: disable=unused-argument input_file: os.PathLike, output_file: os.PathLike, *, - language: List[str] = None, + language: Iterable[str] = None, image_dpi: int = None, output_type=None, sidecar: os.PathLike = None, @@ -214,7 +220,7 @@ def ocr( # pylint: disable=unused-argument jbig2_page_group_size: int = None, pages: str = None, max_image_mpixels: float = None, - tesseract_config: List[str] = None, + tesseract_config: Iterable[str] = None, tesseract_pagesegmode: int = None, tesseract_oem: int = None, pdf_renderer=None, diff --git a/tests/test_api.py b/tests/test_api.py index fd09f262..2842d224 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -59,3 +59,8 @@ def test_tqdm_console(): log.info("done") assert not before_pbar("done") + + +def test_language_list(): + with pytest.raises(ocrmypdf.exceptions.InputFileError): + ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'ita']) From 21cf9029e83b2061257ec01af234c4e6c91fbad9 Mon Sep 17 00:00:00 2001 From: "Lars K.W. Gohlke" Date: Wed, 15 Apr 2020 08:32:01 +0200 Subject: [PATCH 388/880] docs: Set ownership when using docker image (#518) --- docs/docker.rst | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 032ad090..95ad668c 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -90,10 +90,8 @@ Docker volume: .. code-block:: bash - docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf - -(However, when done this way, ``output.pdf`` may be owned by the root -user.) + alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" ocrmypdf' + docker_ocrmypdf /data/input.pdf /data/output.pdf .. _docker-lang-packs: From 4c029e973faaf24e933540ce63221c3b42980730 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Apr 2020 23:53:52 -0700 Subject: [PATCH 389/880] Fix isinstance(..,str) --- src/ocrmypdf/api.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 88f9657e..8fbcc2b9 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -154,7 +154,7 @@ def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwarg cmdline.append(f"--{cmd_style_arg}") continue - if isinstance(val, Iterable): + if isinstance(val, Iterable) and not isinstance(val, str): for elem in val: cmdline.append(f"--{cmd_style_arg}") cmdline.append(elem) From 4ff4ed24a80a4d4c5b84aba8412849993e5fb43c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Mar 2020 21:40:28 -0800 Subject: [PATCH 390/880] Refactor Windows executable shims --- src/ocrmypdf/exec/__init__.py | 34 +++++++++++++++++++++------------- 1 file changed, 21 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 92e4d956..a19b2043 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -62,21 +62,10 @@ def run(args, *, env=None, **kwargs): # Search in spoof path if necessary program = _get_program(args, env) - - # If we are running a .py on Windows, ensure we call it with this Python - # (to support test suite shims) - if os.name == 'nt' and program.lower().endswith('.py'): - args = [sys.executable, program] + args[1:] - else: - args = [program] + args[1:] + args = [program] + args[1:] if os.name == 'nt': - paths = os.pathsep.join(os.get_exec_path(env)) - if not shutil.which(args[0], path=paths): - shimmed_path = shim_paths_with_program_files(env) - new_args0 = shutil.which(args[0], path=shimmed_path) - if new_args0: - args[0] = new_args0 + args = fix_windows_args(program, args, env) process_log = log.getChild(os.path.basename(program)) process_log.debug("Running: %s", args) @@ -95,6 +84,25 @@ def run(args, *, env=None, **kwargs): return proc +def fix_windows_args(program, args, env): + """Adjust our desired program and command line arguments for use on Windows""" + + # If we are running a .py on Windows, ensure we call it with this Python + # (to support test suite shims) + if program.lower().endswith('.py'): + args = [sys.executable] + args + + paths = os.pathsep.join(os.get_exec_path(env)) + if not shutil.which(args[0], path=paths): + # If the program we want is not on the PATH, add some interesting + # locations in %PROGRAMFILES% to the PATH and try again + shimmed_path = shim_paths_with_program_files(env) + new_args0 = shutil.which(args[0], path=shimmed_path) + if new_args0: + args[0] = new_args0 + return args + + def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): """Get the version of the specified program""" args_prog = [program, version_arg] From d146d2b65c7cd99b6fffa4a5aac5247786b4686d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Mar 2020 21:24:13 -0800 Subject: [PATCH 391/880] The Great Logging Refactor Remove all instances of logger object being passed as parameters. This was a holdover from ruffus, and complicated a lot of simple things. --- src/ocrmypdf/__main__.py | 4 +-- src/ocrmypdf/_graft.py | 10 +++--- src/ocrmypdf/_jobcontext.py | 42 ++--------------------- src/ocrmypdf/_pipeline.py | 55 +++++++++++-------------------- src/ocrmypdf/_sync.py | 13 ++++---- src/ocrmypdf/exec/__init__.py | 3 -- src/ocrmypdf/exec/ghostscript.py | 10 +----- src/ocrmypdf/exec/qpdf.py | 8 +++-- src/ocrmypdf/exec/tesseract.py | 42 +++++++++++------------ src/ocrmypdf/exec/unpaper.py | 9 +++-- src/ocrmypdf/optimize.py | 50 ++++++++++++++-------------- src/ocrmypdf/pdfinfo/ghosttext.py | 4 +-- src/ocrmypdf/pdfinfo/info.py | 8 ++--- tests/test_ghostscript.py | 4 --- tests/test_optimize.py | 7 +--- tests/test_preprocessing.py | 11 +------ tests/test_rotation.py | 6 +--- tests/test_stdio.py | 2 +- tests/test_tess4.py | 18 +++------- 19 files changed, 107 insertions(+), 199 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index bcfd6528..459b3040 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,13 +21,14 @@ import os import sys from . import __version__ -from ._jobcontext import make_logger from ._sync import run_pipeline from ._validation import check_closed_streams, check_options from .api import Verbosity, configure_logging from .cli import parser from .exceptions import BadArgsError, ExitCode, MissingDependencyError +log = logging.getLogger('ocrmypdf') + def run(args=None): options = parser.parse_args(args=args) @@ -47,7 +48,6 @@ def run(args=None): configure_logging( verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True ) - log = make_logger('ocrmypdf') log.debug('ocrmypdf ' + __version__) try: check_options(options) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index a535d492..fba24718 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -15,12 +15,14 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import os from contextlib import suppress from pathlib import Path import pikepdf +log = logging.getLogger(__name__) MAX_REPLACE_PAGES = 100 @@ -89,7 +91,7 @@ def strip_invisible_text(pdf, page): def _graft_text_layer( - *, pdf_base, page_num, text, font, font_key, procset, rotation, strip_old_text, log + *, pdf_base, page_num, text, font, font_key, procset, rotation, strip_old_text ): """Insert the text layer from text page 0 on to pdf_base at page_num""" @@ -179,7 +181,6 @@ def _find_font(text, pdf_base): class OcrGrafter: def __init__(self, context): self.context = context - self.log = context.log self.path_base = Path(context.origin).resolve() self.pdf_base = pikepdf.open(self.path_base) @@ -206,7 +207,7 @@ class OcrGrafter: if path_image is not None and path_image != self.path_base: # We are updating the old page with a rasterized PDF of the new # page (without changing objgen, to preserve references) - self.log.debug("Emplacement update") + log.debug("Emplacement update") with pikepdf.open(image) as pdf_image: self.emplacements += 1 foreign_image_page = pdf_image.pages[0] @@ -220,7 +221,7 @@ class OcrGrafter: content_rotation = autorotate_correction text_rotation = autorotate_correction text_misaligned = (text_rotation - content_rotation) % 360 - self.log.debug( + log.debug( f"Rotations for page {pageno}: [text, auto, misalign, content] = " f"{text_rotation}, {autorotate_correction}, " f"{text_misaligned}, {content_rotation}" @@ -238,7 +239,6 @@ class OcrGrafter: rotation=text_misaligned, procset=self.procset, strip_old_text=strip_old, - log=self.log, ) # Correct the rotation if applicable diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 42c96282..59de5cdd 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -21,30 +21,10 @@ import shutil import sys -class PicklableLoggerMixin: - def __init__(self): - self._log = None - - @property - def log(self): - if not self._log: - self._log = self.get_logger() - return self._log - - def __getstate__(self): - # Python 3.6 is incapable of pickling a logger and marshalling it to another - # process (threading._RLock error), so we disconnect it before pickling, - # and create a new logger in the worker process. - state = self.__dict__.copy() - state['_log'] = None - return state - - -class PDFContext(PicklableLoggerMixin): +class PDFContext: """Holds our context for a particular run of the pipeline""" def __init__(self, options, work_folder, origin, pdfinfo): - PicklableLoggerMixin.__init__(self) self.options = options self.work_folder = work_folder self.origin = origin @@ -56,9 +36,6 @@ class PDFContext(PicklableLoggerMixin): if self.name == '-': self.name = 'stdin' - def get_logger(self): - return make_logger(self.options, filename=self.name) - def get_path(self, name): return os.path.join(self.work_folder, name) @@ -68,14 +45,13 @@ class PDFContext(PicklableLoggerMixin): yield PageContext(self, n) -class PageContext(PicklableLoggerMixin): +class PageContext: """Holds our context for a page Must be pickable, so only store intrinsic/simple data elements """ def __init__(self, pdf_context, pageno): - PicklableLoggerMixin.__init__(self) self.work_folder = pdf_context.work_folder self.origin = pdf_context.origin self.options = pdf_context.options @@ -84,9 +60,6 @@ class PageContext(PicklableLoggerMixin): self.pageinfo = pdf_context.pdfinfo[pageno] self._log = None - def get_logger(self): - return make_logger(self.options, filename=self.name, page=self.pageno + 1) - def get_path(self, name): return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name)) @@ -111,14 +84,3 @@ class LogNamePageAdapter(logging.LoggerAdapter): '%4u: %s' % (self.extra['page'], msg), kwargs, ) - - -def make_logger(options=None, prefix='ocrmypdf', filename=None, page=None): - log = logging.getLogger(prefix) - if filename and page: - adapter = LogNamePageAdapter(log, dict(input_filename=filename, page=page)) - elif filename: - adapter = LogNameAdapter(log, dict(input_filename=filename)) - else: - adapter = log - return adapter diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 58ac2495..98ed32fd 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging import os import re import sys @@ -44,10 +45,12 @@ from .optimize import optimize from .pdfa import generate_pdfa_ps from .pdfinfo import Colorspace, Encoding, PdfInfo +log = logging.getLogger(__name__) + VECTOR_PAGE_DPI = 400 -def triage_image_file(input_file, output_file, options, log): +def triage_image_file(input_file, output_file, options): log.info("Input file is not a PDF, checking if it is an image...") try: im = Image.open(input_file) @@ -124,7 +127,7 @@ def _pdf_guess_version(input_file, search_window=1024): return '' -def triage(original_filename, input_file, output_file, options, log): +def triage(original_filename, input_file, output_file, options): try: if _pdf_guess_version(input_file): if options.image_dpi: @@ -140,7 +143,7 @@ def triage(original_filename, input_file, output_file, options, log): msg = str(e).replace(input_file, original_filename) raise InputFileError(msg) from e - triage_image_file(input_file, output_file, options, log) + triage_image_file(input_file, output_file, options) return output_file @@ -156,7 +159,6 @@ def get_pdfinfo(input_file, detailed_page_analysis=False, progbar=False): def validate_pdfinfo_options(context): - log = context.log pdfinfo = context.pdfinfo options = context.options @@ -241,7 +243,6 @@ def get_canvas_square_dpi(pageinfo, options): def is_ocr_required(page_context): pageinfo = page_context.pageinfo options = page_context.options - log = page_context.log ocr_required = True @@ -322,7 +323,6 @@ def rasterize_preview(input_file, page_context): xres=canvas_dpi, yres=canvas_dpi, raster_device='jpeggray', - log=page_context.log, page_dpi=(page_dpi, page_dpi), pageno=page_context.pageinfo.pageno + 1, ) @@ -380,12 +380,11 @@ def get_orientation_correction(preview, page_context): preview, engine_mode=page_context.options.tesseract_oem, timeout=page_context.options.tesseract_timeout, - log=page_context.log, tesseract_env=page_context.options.tesseract_env, ) correction = orient_conf.angle % 360 - page_context.log.info(describe_rotation(page_context, orient_conf, correction)) + log.info(describe_rotation(page_context, orient_conf, correction)) if ( orient_conf.confidence >= page_context.options.rotate_pages_threshold and correction != 0 @@ -426,7 +425,7 @@ def rasterize( device = colorspaces[device_idx] - page_context.log.debug(f"Rasterize with {device}") + log.debug(f"Rasterize with {device}") # Produce the page image with square resolution or else deskew and OCR # will not work properly. @@ -439,7 +438,6 @@ def rasterize( xres=canvas_dpi, yres=canvas_dpi, raster_device=device, - log=page_context.log, page_dpi=(page_dpi, page_dpi), pageno=pageinfo.pageno + 1, rotation=correction, @@ -454,7 +452,7 @@ def preprocess_remove_background(input_file, page_context): leptonica.remove_background(input_file, output_file) return output_file else: - page_context.log.info("background removal skipped on mono page") + log.info("background removal skipped on mono page") return input_file @@ -470,13 +468,7 @@ def preprocess_clean(input_file, page_context): output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - unpaper.clean( - input_file, - output_file, - dpi, - page_context.log, - page_context.options.unpaper_args, - ) + unpaper.clean(input_file, output_file, dpi, page_context.options.unpaper_args) return output_file @@ -496,7 +488,7 @@ def create_ocr_image(image, page_context): draw = ImageDraw.ImageDraw(im) xres, yres = im.info['dpi'] - page_context.log.debug('resolution %r %r' % (xres, yres)) + log.debug('resolution %r %r' % (xres, yres)) if not options.force_ocr: # Do not mask text areas when forcing OCR, because we need to OCR @@ -520,7 +512,7 @@ def create_ocr_image(image, page_context): im.height - bbox[1] * yscale, ] pixcoords = [int(round(c)) for c in pixcoords] - page_context.log.debug('blanking %r', pixcoords) + log.debug('blanking %r', pixcoords) draw.rectangle(pixcoords, fill=white) # draw.rectangle(pixcoords, outline=pink) @@ -551,7 +543,6 @@ def ocr_tesseract_hocr(input_file, page_context): user_words=options.user_words, user_patterns=options.user_patterns, tesseract_env=options.tesseract_env, - log=page_context.log, ) return (hocr_out, hocr_text_out) @@ -591,11 +582,11 @@ def create_pdf_page_from_image(image, page_context): # This create a single page PDF with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: - page_context.log.debug('convert') + log.debug('convert') img2pdf.convert( imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf ) - page_context.log.debug('convert done') + log.debug('convert done') return output_file @@ -631,7 +622,6 @@ def ocr_tesseract_textonly_pdf(input_image, page_context): user_words=options.user_words, user_patterns=options.user_patterns, tesseract_env=options.tesseract_env, - log=page_context.log, ) return (output_pdf, output_text) @@ -697,7 +687,7 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): try: len(pdf_file.docinfo) except TypeError: - context.log.error( + log.error( "File contains a malformed DocumentInfo block - continuing anyway" ) else: @@ -716,7 +706,6 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): pdf_pages=[fix_docinfo_file, input_ps_stub], output_file=output_file, compression=options.pdfa_image_compression, - log=context.log, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 ) @@ -738,22 +727,18 @@ def metadata_fixup(working_file, context): if not missing: return if options.output_type.startswith('pdfa'): - context.log.warning( + log.warning( "Some input metadata could not be copied because it is not " "permitted in PDF/A. You may wish to examine the output " "PDF's XMP metadata." ) - context.log.debug( - "The following metadata fields were not copied: %r", missing - ) + log.debug("The following metadata fields were not copied: %r", missing) else: - context.log.error( + log.error( "Some input metadata could not be copied." "You may wish to examine the output PDF's XMP metadata." ) - context.log.info( - "The following metadata fields were not copied: %r", missing - ) + log.info("The following metadata fields were not copied: %r", missing) with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf: docinfo = get_docinfo(original, options) @@ -819,7 +804,7 @@ def merge_sidecars(txt_files, context): def copy_final(input_file, output_file, context): - context.log.debug('%s -> %s', input_file, output_file) + log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': copyfileobj(input_stream, sys.stdout.buffer) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 07c355ab..dbb7603e 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -30,7 +30,7 @@ import PIL from tqdm import tqdm from ._graft import OcrGrafter -from ._jobcontext import PDFContext, cleanup_working_files, make_logger +from ._jobcontext import PDFContext, cleanup_working_files from ._pipeline import ( convert_to_pdfa, copy_final, @@ -66,6 +66,8 @@ from .exec import qpdf from .helpers import available_cpu_count from .pdfa import file_claims_pdfa +log = logging.getLogger(__name__) + PageResult = namedtuple( 'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction' ) @@ -231,7 +233,7 @@ def exec_concurrent(context): # Run exec_page_sync on every page context max_workers = min(len(context.pdfinfo), context.options.jobs) if max_workers > 1: - context.log.info("Start processing %d pages concurrently", max_workers) + log.info("Start processing %d pages concurrently", max_workers) # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want # to manage how many threads it uses to avoid creating total threads than cores. @@ -250,7 +252,7 @@ def exec_concurrent(context): except ValueError: # OMP_THREAD_LIMIT initialized to non-numeric context.log.error("Environment variable OMP_THREAD_LIMIT is not numeric") if tess_threads > 1: - context.log.info("Using Tesseract OpenMP thread limit %d", tess_threads) + log.info("Using Tesseract OpenMP thread limit %d", tess_threads) if context.options.use_threads: from multiprocessing.dummy import Pool @@ -352,8 +354,6 @@ def configure_debug_logging(log_filename, prefix=''): def run_pipeline(options, api=False): - log = make_logger(options, __name__) - # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example # options.input_file, options.pdf_renderer are already bound.) @@ -377,7 +377,6 @@ def run_pipeline(options, api=False): start_input_file, os.path.join(work_folder, 'origin.pdf'), options, - log, ) # Gather pdfinfo and create context @@ -412,7 +411,7 @@ def run_pipeline(options, api=False): pdfa_info['conformance'], ) return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file, log): + if not qpdf.check(options.output_file): log.warning('Output file: The generated PDF is INVALID') return ExitCode.invalid_output_pdf report_output_file_size(options, start_input_file, options.output_file) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index a19b2043..e2bbf0a5 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -268,9 +268,6 @@ def check_external_program( recommended=False, **kwargs, # To consume log parameter ): - if kwargs: - if not 'log' in kwargs: - log.warning('check_external_program(log=...) is deprecated') try: found_version = version_checker() except (CalledProcessError, FileNotFoundError, MissingDependencyError): diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 856bc0c1..de4cbda2 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -34,7 +34,7 @@ from PIL import Image from ..exceptions import MissingDependencyError, SubprocessOutputError from . import get_version, run -gslog = logging.getLogger() +log = logging.getLogger(__name__) GS = 'gs' if os.name == 'nt': @@ -138,7 +138,6 @@ def rasterize_pdf( xres, yres, raster_device, - log, pageno=1, page_dpi=None, rotation=None, @@ -155,7 +154,6 @@ def rasterize_pdf( :param xres: resolution at which to rasterize page :param yres: :param raster_device: - :param log: :param pageno: page number to rasterize (beginning at page 1) :param page_dpi: resolution tuple (x, y) overriding output image DPI :param rotation: 0, 90, 180, 270: clockwise angle to rotate page @@ -165,8 +163,6 @@ def rasterize_pdf( res = round(xres, 6), round(yres, 6) if not page_dpi: page_dpi = res - if not log: - log = gslog args_gs = ( [ @@ -191,7 +187,6 @@ def rasterize_pdf( ] ) - log.debug(args_gs) try: p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True) except CalledProcessError as e: @@ -224,7 +219,6 @@ def generate_pdfa( pdf_pages, output_file, compression, - log, threads=None, # deprecated parameter pdf_version='1.5', pdfa_part='2', @@ -246,8 +240,6 @@ def generate_pdfa( images entirely. (The feature was added in 9.23 but broken, and the 9.24 release of Ghostscript had regressions, so we don't support it until 9.25.) """ - if not log: - log = gslog if threads is not None: warnings.warn( "use of deprecated parameter 'threads'", category=DeprecationWarning diff --git a/src/ocrmypdf/exec/qpdf.py b/src/ocrmypdf/exec/qpdf.py index d46baf1c..32ef8cb6 100644 --- a/src/ocrmypdf/exec/qpdf.py +++ b/src/ocrmypdf/exec/qpdf.py @@ -17,22 +17,24 @@ """Interface to qpdf executable""" +import logging from io import StringIO import pikepdf +log = logging.getLogger(__name__) + def version(): return pikepdf.__libqpdf_version__ -def check(input_file, log=None): +def check(input_file): pdf = None try: pdf = pikepdf.open(input_file) except pikepdf.PdfError as e: - if log: - log.error(e) + log.error(e) return False else: messages = pdf.check() diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index abcbb2fa..4ebdb0cf 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -33,6 +33,8 @@ from ..exceptions import ( from ..helpers import page_number, safe_symlink from . import get_version, run +log = logging.getLogger(__name__) + OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) HOCR_TEMPLATE = """ @@ -144,7 +146,7 @@ def tess_base_args(langs, engine_mode): return args -def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=None): +def get_orientation(input_file, engine_mode, timeout: float, tesseract_env=None): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', @@ -165,7 +167,7 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env= except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) except CalledProcessError as e: - tesseract_log_output(log, e.output, input_file) + tesseract_log_output(e.output, input_file) if ( b'Too few characters. Skipping this page' in e.output or b'Image too large' in e.output @@ -187,9 +189,9 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env= return oc -def tesseract_log_output(mainlog, stdout, input_file): - log = TesseractLoggerAdapter( - mainlog, extra=mainlog.extra if hasattr(mainlog, 'extra') else None +def tesseract_log_output(stdout, input_file): + tlog = TesseractLoggerAdapter( + log, extra=log.extra if hasattr(log, 'extra') else None ) try: @@ -204,28 +206,28 @@ def tesseract_log_output(mainlog, stdout, input_file): elif line.startswith("Warning in pixReadMem"): continue elif 'diacritics' in line: - log.warning("lots of diacritics - possibly poor OCR") + tlog.warning("lots of diacritics - possibly poor OCR") elif line.startswith('OSD: Weak margin'): - log.warning("unsure about page orientation") + tlog.warning("unsure about page orientation") elif 'Error in pixScanForForeground' in line: pass # Appears to be spurious/problem with nonwhite borders elif 'Error in boxClipToRectangle' in line: pass # Always appears with pixScanForForeground message elif 'parameter not found: ' in line.lower(): - log.error(line.strip()) + tlog.error(line.strip()) problem = line.split('found: ')[1] raise TesseractConfigError(problem) elif 'error' in line.lower() or 'exception' in line.lower(): - log.error(line.strip()) + tlog.error(line.strip()) elif 'warning' in line.lower(): - log.warning(line.strip()) + tlog.warning(line.strip()) elif 'read_params_file' in line.lower(): - log.error(line.strip()) + tlog.error(line.strip()) else: - log.info(line.strip()) + tlog.info(line.strip()) -def page_timedout(log, input_file, timeout): +def page_timedout(input_file, timeout): if timeout == 0: return prefix = f"{(page_number(input_file)):4d}: [tesseract] " @@ -257,7 +259,6 @@ def generate_hocr( user_words, user_patterns, tesseract_env, - log, ): output_hocr = next(o for o in output_files if fspath(o).endswith('.hocr')) @@ -292,17 +293,17 @@ def generate_hocr( # Generate a HOCR file with no recognized text if tesseract times out # Temporary workaround to hocrTransform not being able to function if # it does not have a valid hOCR file. - page_timedout(log, input_file, timeout) + page_timedout(input_file, timeout) _generate_null_hocr(output_hocr, output_sidecar, input_file) except CalledProcessError as e: - tesseract_log_output(log, e.output, input_file) + tesseract_log_output(e.output, input_file) if b'Image too large' in e.output: _generate_null_hocr(output_hocr, output_sidecar, input_file) return raise SubprocessOutputError() from e else: - tesseract_log_output(log, stdout, input_file) + tesseract_log_output(stdout, input_file) # The sidecar text file will get the suffix .txt; rename it to # whatever caller wants it named if os.path.exists(prefix + '.txt'): @@ -340,7 +341,6 @@ def generate_pdf( user_words, user_patterns, tesseract_env, - log, ): """Use Tesseract to render a PDF. @@ -389,13 +389,13 @@ def generate_pdf( if os.path.exists(prefix + '.txt'): shutil.move(prefix + '.txt', output_text) except TimeoutExpired: - page_timedout(log, input_image, timeout) + page_timedout(input_image, timeout) use_skip_page(text_only, skip_pdf, output_pdf, output_text) except CalledProcessError as e: - tesseract_log_output(log, e.output, input_image) + tesseract_log_output(e.output, input_image) if b'Image too large' in e.output: use_skip_page(text_only, skip_pdf, output_pdf, output_text) return raise SubprocessOutputError() from e else: - tesseract_log_output(log, stdout, input_image) + tesseract_log_output(stdout, input_image) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 2984b455..0228beb6 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -20,6 +20,7 @@ """Interface to unpaper executable""" +import logging import os import shlex from functools import lru_cache @@ -32,13 +33,15 @@ from ..exceptions import MissingDependencyError, SubprocessOutputError from . import get_version from . import run as external_run +log = logging.getLogger(__name__) + @lru_cache(maxsize=1) def version(): return get_version('unpaper') -def run(input_file, output_file, dpi, log, mode_args): +def run(input_file, output_file, dpi, mode_args): args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'} @@ -110,7 +113,7 @@ def validate_custom_args(args: str): return unpaper_args -def clean(input_file, output_file, dpi, log, unpaper_args=None): +def clean(input_file, output_file, dpi, unpaper_args=None): default_args = [ '--layout', 'none', @@ -124,4 +127,4 @@ def clean(input_file, output_file, dpi, log, unpaper_args=None): ] if not unpaper_args: unpaper_args = default_args - run(input_file, output_file, dpi, log, unpaper_args) + run(input_file, output_file, dpi, unpaper_args) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 839a8ff9..e6f0bd3f 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -16,6 +16,7 @@ # along with OCRmyPDF. If not, see . import concurrent.futures +import logging import sys import tempfile from collections import defaultdict @@ -33,6 +34,8 @@ from .exceptions import OutputFileAccessError from .exec import jbig2enc, pngquant from .helpers import safe_symlink +log = logging.getLogger(__name__) + DEFAULT_JPEG_QUALITY = 75 DEFAULT_PNG_QUALITY = 70 @@ -53,7 +56,7 @@ def tif_name(root, xref): return img_name(root, xref, '.tif') -def extract_image_filter(pike, root, log, image, xref): +def extract_image_filter(pike, root, image, xref): if image.Subtype != Name.Image: return None if image.Length < 100: @@ -79,8 +82,8 @@ def extract_image_filter(pike, root, log, image, xref): return pim, filtdp -def extract_image_jbig2(*, pike, root, log, image, xref, options): - result = extract_image_filter(pike, root, log, image, xref) +def extract_image_jbig2(*, pike, root, image, xref, options): + result = extract_image_filter(pike, root, image, xref) if result is None: return None pim, filtdp = result @@ -101,8 +104,8 @@ def extract_image_jbig2(*, pike, root, log, image, xref, options): return None -def extract_image_generic(*, pike, root, log, image, xref, options): - result = extract_image_filter(pike, root, log, image, xref) +def extract_image_generic(*, pike, root, image, xref, options): + result = extract_image_filter(pike, root, image, xref) if result is None: return None pim, filtdp = result @@ -170,7 +173,7 @@ def extract_image_generic(*, pike, root, log, image, xref, options): return None -def extract_images(pike, root, log, options, extract_fn): +def extract_images(pike, root, options, extract_fn): """Extract image using extract_fn Enumerate images on each page, lookup their xref/ID number in the PDF. @@ -212,7 +215,7 @@ def extract_images(pike, root, log, options, extract_fn): image = pike.get_object((xref, 0)) try: result = extract_fn( - pike=pike, root=root, log=log, image=image, xref=xref, options=options + pike=pike, root=root, image=image, xref=xref, options=options ) except Exception as e: log.debug("Image xref %s, error %s", xref, repr(e)) @@ -223,12 +226,12 @@ def extract_images(pike, root, log, options, extract_fn): yield pageno_for_xref[xref], xref, ext -def extract_images_generic(pike, root, log, options): +def extract_images_generic(pike, root, options): """Extract any >=2bpp image we think we can improve""" jpegs = [] pngs = [] - for _, xref, ext in extract_images(pike, root, log, options, extract_image_generic): + for _, xref, ext in extract_images(pike, root, options, extract_image_generic): log.debug('xref = %s ext = %s', xref, ext) if ext == '.png': pngs.append(xref) @@ -238,13 +241,11 @@ def extract_images_generic(pike, root, log, options): return jpegs, pngs -def extract_images_jbig2(pike, root, log, options): +def extract_images_jbig2(pike, root, options): """Extract any bitonal image that we think we can improve as JBIG2""" jbig2_groups = defaultdict(list) - for pageno, xref, ext in extract_images( - pike, root, log, options, extract_image_jbig2 - ): + for pageno, xref, ext in extract_images(pike, root, options, extract_image_jbig2): group = pageno // options.jbig2_page_group_size jbig2_groups[group].append((xref, ext)) @@ -256,7 +257,7 @@ def extract_images_jbig2(pike, root, log, options): return jbig2_groups -def _produce_jbig2_images(jbig2_groups, root, log, options): +def _produce_jbig2_images(jbig2_groups, root, options): """Produce JBIG2 images from their groups""" def jbig2_group_futures(executor, root, groups): @@ -304,7 +305,7 @@ def _produce_jbig2_images(jbig2_groups, root, log, options): pbar.update() -def convert_to_jbig2(pike, jbig2_groups, root, log, options): +def convert_to_jbig2(pike, jbig2_groups, root, options): """Convert images to JBIG2 and insert into PDF. When the JBIG2 page group size is > 1 we do several JBIG2 images at once @@ -318,7 +319,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options): and needs no dictionary. Currently this must be lossless JBIG2. """ - _produce_jbig2_images(jbig2_groups, root, log, options) + _produce_jbig2_images(jbig2_groups, root, options) for group, xref_exts in jbig2_groups.items(): prefix = f'group{group:08d}' @@ -342,7 +343,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options): ) -def transcode_jpegs(pike, jpegs, root, log, options): +def transcode_jpegs(pike, jpegs, root, options): for xref in tqdm( jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar ): @@ -365,7 +366,7 @@ def transcode_jpegs(pike, jpegs, root, log, options): im_obj.write(compdata.read(), filter=Name.DCTDecode) -def transcode_pngs(pike, images, image_name_fn, root, log, options): +def transcode_pngs(pike, images, image_name_fn, root, options): modified = set() if options.optimize >= 2: png_quality = ( @@ -500,7 +501,6 @@ def rewrite_png(pike, im_obj, compdata, log): def optimize(input_file, output_file, context, save_settings): - log = context.log options = context.options if options.optimize == 0: safe_symlink(input_file, output_file) @@ -517,15 +517,15 @@ def optimize(input_file, output_file, context, save_settings): root = Path(output_file).parent / 'images' root.mkdir(exist_ok=True) - jpegs, pngs = extract_images_generic(pike, root, log, options) - transcode_jpegs(pike, jpegs, root, log, options) + jpegs, pngs = extract_images_generic(pike, root, options) + transcode_jpegs(pike, jpegs, root, options) # if options.optimize >= 2: # Try pngifying the jpegs - # transcode_pngs(pike, jpegs, jpg_name, root, log, options) - transcode_pngs(pike, pngs, png_name, root, log, options) + # transcode_pngs(pike, jpegs, jpg_name, root, options) + transcode_pngs(pike, pngs, png_name, root, options) - jbig2_groups = extract_images_jbig2(pike, root, log, options) - convert_to_jbig2(pike, jbig2_groups, root, log, options) + jbig2_groups = extract_images_jbig2(pike, root, options) + convert_to_jbig2(pike, jbig2_groups, root, options) target_file = Path(output_file).with_suffix('.opt.pdf') pike.remove_unreferenced_resources() diff --git a/src/ocrmypdf/pdfinfo/ghosttext.py b/src/ocrmypdf/pdfinfo/ghosttext.py index 9626fad7..07e72f19 100644 --- a/src/ocrmypdf/pdfinfo/ghosttext.py +++ b/src/ocrmypdf/pdfinfo/ghosttext.py @@ -21,7 +21,7 @@ import xml.etree.ElementTree as ET from ..exec import ghostscript -gslog = logging.getLogger() +log = logging.getLogger(__name__) # Forgive me for I have sinned # I am using regular expressions to parse XML. However the XML in this case, @@ -77,7 +77,7 @@ def page_get_textblocks(infile, pageno, xmltext, height): return [block for block in joined_blocks()] -def extract_text_xml(infile, pdf, pageno=None, log=gslog): +def extract_text_xml(infile, pdf, pageno=None): existing_text = ghostscript.extract_text(infile, pageno=None) existing_text = regex_remove_char_tags.sub(b' ', existing_text) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index d304688b..329661b3 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -616,7 +616,7 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): return pageinfo -def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False): +def _pdf_get_all_pageinfo(infile, detailed_analysis=False, progbar=False): pdf = pikepdf.open(infile) # Do not close in this function try: if pdf.is_encrypted: @@ -624,7 +624,7 @@ def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=Fal if detailed_analysis: pages_xml = None else: - pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log) + pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None) pages = [] for n, _ in tqdm( @@ -758,12 +758,12 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False): + def __init__(self, infile, detailed_page_analysis=False, progbar=False): self._infile = infile if ghostscript.version() in ('9.52',): detailed_page_analysis = True # txtwrite doesn't work in these versions self._pages, pdf = _pdf_get_all_pageinfo( - infile, detailed_page_analysis, log=log, progbar=progbar + infile, detailed_page_analysis, progbar=progbar ) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 3aed104a..5bb7b76f 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -73,14 +73,12 @@ def test_rasterize_size(francais, outdir, caplog): target_size = Decimal('50.0'), Decimal('30.0') forced_dpi = 42.0, 4242.0 - log = logging.getLogger() rasterize_pdf( path, outdir / 'out.png', target_size[0] / page_size[0], target_size[1] / page_size[1], raster_device='pngmono', - log=log, page_dpi=forced_dpi, ) @@ -97,7 +95,6 @@ def test_rasterize_rotated(francais, outdir, caplog): target_size = Decimal('50.0'), Decimal('30.0') forced_dpi = 42.0, 4242.0 - log = logging.getLogger() caplog.set_level(logging.DEBUG) rasterize_pdf( path, @@ -105,7 +102,6 @@ def test_rasterize_rotated(francais, outdir, caplog): target_size[0] / page_size[0], target_size[1] / page_size[1], raster_device='pngmono', - log=log, page_dpi=forced_dpi, rotation=90, ) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 0c78d653..6cdc1c71 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -43,12 +43,7 @@ def test_mono_not_inverted(resources, outdir): opt.main(infile, outdir / 'out.pdf', level=3) rasterize_pdf( - outdir / 'out.pdf', - outdir / 'im.png', - xres=10, - yres=10, - raster_device='pnggray', - log=logging.getLogger(name='test_mono_not_inverted'), + outdir / 'out.pdf', outdir / 'im.png', xres=10, yres=10, raster_device='pnggray' ) with Image.open(fspath(outdir / 'im.png')) as im: diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 4054e4e4..50eb6b56 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -55,7 +55,6 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): xres=150, yres=150, raster_device='pngmono', - log=log, pageno=1, ) @@ -80,18 +79,10 @@ def test_remove_background(spoof_tesseract_noop, resources, outdir): env=spoof_tesseract_noop, ) - log = logging.getLogger() - output_png = outdir / 'remove_bg.png' ghostscript.rasterize_pdf( - output_pdf, - output_png, - xres=100, - yres=100, - raster_device='png16m', - log=log, - pageno=1, + output_pdf, output_png, xres=100, yres=100, raster_device='png16m', pageno=1 ) # The output image should contain pure white and black diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 05d42300..796c2bfb 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -48,8 +48,6 @@ RENDERERS = ['hocr', 'sandwich'] def check_monochrome_correlation( outdir, reference_pdf, reference_pageno, test_pdf, test_pageno ): - gslog = logging.getLogger() - reference_png = outdir / f'{reference_pdf.name}.ref{reference_pageno:04d}.png' test_png = outdir / f'{test_pdf.name}.test{test_pageno:04d}.png' @@ -63,7 +61,6 @@ def check_monochrome_correlation( xres=100, yres=100, raster_device='pngmono', - log=gslog, pageno=pageno, rotation=0, ) @@ -268,7 +265,6 @@ def test_tesseract_orientation(resources, tmp_path): pix_rotated = pix.rotate_orth(2) # 180 degrees clockwise pix_rotated.write_implied_format(tmp_path / '000001.png') - log = logging.getLogger() tesseract.get_orientation( # Test results of this are unreliable - tmp_path / '000001.png', engine_mode='3', timeout=10, log=log + tmp_path / '000001.png', engine_mode='3', timeout=10 ) diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 250ca6f2..150e40ff 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -74,7 +74,7 @@ def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): ) assert p.returncode == ExitCode.ok - assert qpdf.check(output_file, log=None) + assert qpdf.check(output_file) @pytest.mark.skipif( diff --git a/tests/test_tess4.py b/tests/test_tess4.py index a66f2186..33110a43 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -86,8 +86,6 @@ def test_no_languages(tmp_path): def test_image_too_large_hocr(monkeypatch, resources, outdir): - log = logging.getLogger('test_image_too_large_hocr') - def dummy_run(args, *, env=None, **kwargs): raise subprocess.CalledProcessError(1, 'tesseract', output=b'Image too large') @@ -100,7 +98,6 @@ def test_image_too_large_hocr(monkeypatch, resources, outdir): tessconfig=[], timeout=180.0, pagesegmode=None, - log=log, user_words=None, user_patterns=None, tesseract_env=None, @@ -109,8 +106,6 @@ def test_image_too_large_hocr(monkeypatch, resources, outdir): def test_image_too_large_pdf(monkeypatch, resources, outdir): - log = logging.getLogger('test_image_too_large_pdf') - def dummy_run(args, *, env=None, **kwargs): raise subprocess.CalledProcessError(1, 'tesseract', output=b'Image too large') @@ -126,7 +121,6 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): tessconfig=[], timeout=180.0, pagesegmode=None, - log=log, user_words=None, user_patterns=None, tesseract_env=None, @@ -137,8 +131,7 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): def test_timeout(caplog): - log = logging.getLogger('test_timeout') - tesseract.page_timedout(log, '123456.png', 5) + tesseract.page_timedout('123456.png', 5) assert "123456" in caplog.text assert "took too long" in caplog.text @@ -160,10 +153,8 @@ def test_timeout(caplog): ], ) def test_tesseract_log_output(caplog, in_, logged): - log = logging.getLogger('tesseract_log_output') - log.setLevel(logging.INFO) - - tesseract.tesseract_log_output(log, in_, 'dummy') + caplog.set_level(logging.INFO) + tesseract.tesseract_log_output(in_, 'dummy') if logged == '': assert caplog.text == '' else: @@ -171,7 +162,6 @@ def test_tesseract_log_output(caplog, in_, logged): def test_tesseract_log_output_raises(caplog): - log = logging.getLogger('tesseract_log_output') with pytest.raises(tesseract.TesseractConfigError): - tesseract.tesseract_log_output(log, b'parameter not found: moo', 'dummy') + tesseract.tesseract_log_output(b'parameter not found: moo', 'dummy') assert 'not found' in caplog.text From af914893763fb8620be0cee7bc8b3c700f5025b8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Mar 2020 21:24:40 -0800 Subject: [PATCH 392/880] Remove safe_symlink log= warning --- src/ocrmypdf/helpers.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index ae326331..55707082 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -32,11 +32,6 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, ** """ Helper function: relinks soft symbolic link if necessary """ - if len(args) == 1 and isinstance(args[0], logging.Logger): - log.warning("Deprecated: safe_symlink(,log)") - if 'log' in kwargs: - log.warning('Deprecated: safe_symlink(...log=)') - input_file = os.fspath(input_file) soft_link_name = os.fspath(soft_link_name) From a63d624052fe66b067ce62c4a67c2bc95a46f67a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 00:04:43 -0700 Subject: [PATCH 393/880] Improve logging of subprocess output --- src/ocrmypdf/exec/__init__.py | 25 ++++++++++++++++--------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index e2bbf0a5..80e6ba98 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -23,6 +23,7 @@ import re import shutil import sys from collections.abc import Mapping +from contextlib import suppress from distutils.version import LooseVersion from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError @@ -67,19 +68,25 @@ def run(args, *, env=None, **kwargs): if os.name == 'nt': args = fix_windows_args(program, args, env) - process_log = log.getChild(os.path.basename(program)) - process_log.debug("Running: %s", args) + log.debug("Running: %s", args) + process_log = log.getChild('subprocess.' + os.path.basename(program)) if sys.version_info < (3, 7) and os.name == 'nt': # Can't use close_fds=True on Windows with Python 3.6 or older # https://bugs.python.org/issue19575, etc. kwargs['close_fds'] = False - proc = subprocess_run(args, env=env, **kwargs) - if process_log.isEnabledFor(logging.DEBUG): - try: - stderr = proc.stderr.decode('utf-8', 'replace') - except AttributeError: - stderr = proc.stderr - if stderr: + + stderr = None + try: + proc = subprocess_run(args, env=env, **kwargs) + except CalledProcessError as e: + stderr = getattr(e, 'stderr', None) + raise + else: + stderr = getattr(proc, 'stderr', None) + finally: + if process_log.isEnabledFor(logging.DEBUG) and stderr: + with suppress(AttributeError, UnicodeDecodeError): + stderr = stderr.decode('utf-8', 'replace') process_log.debug("stderr = %s", stderr) return proc From c2919f2e1ca16b838ab1247e79aaff04697b6a16 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 00:05:23 -0700 Subject: [PATCH 394/880] Reinstate logging of page numbers --- src/ocrmypdf/_jobcontext.py | 16 ---------- src/ocrmypdf/_logging.py | 60 +++++++++++++++++++++++++++++++++++++ src/ocrmypdf/_sync.py | 21 ++++++++++++- src/ocrmypdf/api.py | 41 ++++++------------------- 4 files changed, 89 insertions(+), 49 deletions(-) create mode 100644 src/ocrmypdf/_logging.py diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 59de5cdd..a782d596 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -58,7 +58,6 @@ class PageContext: self.name = pdf_context.name self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] - self._log = None def get_path(self, name): return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name)) @@ -69,18 +68,3 @@ def cleanup_working_files(work_folder, options): print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr) else: shutil.rmtree(work_folder, ignore_errors=True) - - -class LogNameAdapter(logging.LoggerAdapter): - def process(self, msg, kwargs): - # return '[%s] %s' % (self.extra['input_filename'], msg), kwargs - return '%s' % (msg,), kwargs - - -class LogNamePageAdapter(logging.LoggerAdapter): - def process(self, msg, kwargs): - return ( - #'[%s:%05u] %s' % (self.extra['input_filename'], self.extra['page'], msg), - '%4u: %s' % (self.extra['page'], msg), - kwargs, - ) diff --git a/src/ocrmypdf/_logging.py b/src/ocrmypdf/_logging.py new file mode 100644 index 00000000..412552fa --- /dev/null +++ b/src/ocrmypdf/_logging.py @@ -0,0 +1,60 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +import sys +from contextlib import suppress + +from tqdm import tqdm + + +class PageNumberFilter(logging.Filter): + def filter(self, record): + pageno = getattr(record, 'pageno', None) + if pageno is not None: + record.pageno = f' [{pageno:5d}]' + else: + record.pageno = '' + return True + + +class TqdmConsole: + """Wrapper to log messages in a way that is compatible with tqdm progress bar + + This routes log messages through tqdm so that it can print them above the + progress bar, and then refresh the progress bar, rather than overwriting + it which looks messy. + + For some reason Python 3.6 prints extra empty messages from time to time, + so we suppress those. + """ + + def __init__(self, file): + self.file = file + self.py36 = sys.version_info[0:2] == (3, 6) + + def write(self, msg): + # When no progress bar is active, tqdm.write() routes to print() + if self.py36: + if msg.strip() != '': + tqdm.write(msg.rstrip(), end='\n', file=self.file) + else: + tqdm.write(msg.rstrip(), end='\n', file=self.file) + + def flush(self): + with suppress(AttributeError): + self.file.flush() diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index dbb7603e..383e8dc7 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -31,6 +31,7 @@ from tqdm import tqdm from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files +from ._logging import PageNumberFilter from ._pipeline import ( convert_to_pdfa, copy_final, @@ -72,6 +73,9 @@ PageResult = namedtuple( 'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction' ) +tls = threading.local() +tls.pageno = None + def preprocess(page_context, image, remove_background, deskew, clean): if remove_background: @@ -83,8 +87,23 @@ def preprocess(page_context, image, remove_background, deskew, clean): return image +old_factory = logging.getLogRecordFactory() + + +def record_factory(*args, **kwargs): + record = old_factory(*args, **kwargs) + if hasattr(tls, 'pageno'): + record.pageno = tls.pageno + return record + + +logging.setLogRecordFactory(record_factory) + + def exec_page_sync(page_context): options = page_context.options + tls.pageno = page_context.pageno + 1 + orientation_correction = 0 pdf_page_from_image_out = None ocr_out = None @@ -346,7 +365,7 @@ def configure_debug_logging(log_filename, prefix=''): log_file_handler = logging.FileHandler(log_filename, delay=True) log_file_handler.setLevel(logging.DEBUG) formatter = logging.Formatter( - '[%(asctime)s] - %(name)s - %(levelname)7s - %(message)s' + '[%(asctime)s] - %(name)s - %(levelname)7s -%(pageno)s %(message)s' ) log_file_handler.setFormatter(formatter) logging.getLogger(prefix).addHandler(log_file_handler) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 8fbcc2b9..0eaaa5b7 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -23,41 +23,12 @@ from enum import IntEnum from pathlib import Path from typing import Dict, Iterable -from tqdm import tqdm - +from ._logging import PageNumberFilter, TqdmConsole from ._sync import run_pipeline from ._validation import check_options from .cli import parser -class TqdmConsole: - """Wrapper to log messages in a way that is compatible with tqdm progress bar - - This routes log messages through tqdm so that it can print them above the - progress bar, and then refresh the progress bar, rather than overwriting - it which looks messy. - - For some reason Python 3.6 prints extra empty messages from time to time, - so we suppress those. - """ - - def __init__(self, file): - self.file = file - self.py36 = sys.version_info[0:2] == (3, 6) - - def write(self, msg): - # When no progress bar is active, tqdm.write() routes to print() - if self.py36: - if msg.strip() != '': - tqdm.write(msg.rstrip(), end='\n', file=self.file) - else: - tqdm.write(msg.rstrip(), end='\n', file=self.file) - - def flush(self): - with suppress(AttributeError): - self.file.flush() - - class Verbosity(IntEnum): """Verbosity level for configure_logging.""" @@ -98,6 +69,7 @@ def configure_logging( """ prefix = '' if manage_root_logger else 'ocrmypdf' + log = logging.getLogger(prefix) log.setLevel(logging.DEBUG) @@ -113,9 +85,14 @@ def configure_logging( else: console.setLevel(logging.INFO) - formatter = logging.Formatter('%(levelname)7s - %(message)s') + console.addFilter(PageNumberFilter()) + if verbosity >= 2: - formatter = logging.Formatter('%(name)s - %(levelname)7s - %(message)s') + fmt = '%(levelname)7s %(name)s -%(pageno)s %(message)s' + else: + fmt = '%(levelname)7s -%(pageno)s %(message)s' + + formatter = logging.Formatter(fmt=fmt) console.setFormatter(formatter) log.addHandler(console) From f4f7946a0c2363a76524090221031df569e788fc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 5 Mar 2020 13:55:48 -0800 Subject: [PATCH 395/880] Add colored logs --- requirements/main.txt | 1 + setup.py | 1 + src/ocrmypdf/api.py | 18 +++++++++++++++++- 3 files changed, 19 insertions(+), 1 deletion(-) diff --git a/requirements/main.txt b/requirements/main.txt index 5b56bd04..81e92848 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -2,6 +2,7 @@ # setup.py lists a separate set of requirements that are looser to simplify # installation cffi == 1.14.0 +coloredlogs == 14.0 # technically optional img2pdf == 0.3.3 pdfminer.six == 20200124 pikepdf == 1.10.2 diff --git a/setup.py b/setup.py index 0b4a61da..d0e05c0d 100644 --- a/setup.py +++ b/setup.py @@ -97,6 +97,7 @@ setup( install_requires=[ 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement + 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six >= 20181108, <= 20200124', 'pikepdf >= 1.8.1, < 2', diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 0eaaa5b7..91cd608a 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -28,6 +28,11 @@ from ._sync import run_pipeline from ._validation import check_options from .cli import parser +try: + import coloredlogs +except ModuleNotFoundError: + coloredlogs = None + class Verbosity(IntEnum): """Verbosity level for configure_logging.""" @@ -92,7 +97,18 @@ def configure_logging( else: fmt = '%(levelname)7s -%(pageno)s %(message)s' - formatter = logging.Formatter(fmt=fmt) + use_colors = progress_bar_friendly + if not coloredlogs: + use_colors = False + if use_colors: + if os.name == 'nt': + use_colors = coloredlogs.enable_ansi_support() + if use_colors: + use_colors = coloredlogs.terminal_supports_colors() + if use_colors: + formatter = coloredlogs.ColoredFormatter(fmt=fmt) + else: + formatter = logging.Formatter(fmt=fmt) console.setFormatter(formatter) log.addHandler(console) From 346da95899e1227d5e04a2d7225b5812086af9c5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 17 Mar 2020 21:06:07 -0700 Subject: [PATCH 396/880] Suppress loglevel since we have color now --- src/ocrmypdf/_logging.py | 2 +- src/ocrmypdf/api.py | 2 +- tests/test_main.py | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_logging.py b/src/ocrmypdf/_logging.py index 412552fa..5126d97c 100644 --- a/src/ocrmypdf/_logging.py +++ b/src/ocrmypdf/_logging.py @@ -26,7 +26,7 @@ class PageNumberFilter(logging.Filter): def filter(self, record): pageno = getattr(record, 'pageno', None) if pageno is not None: - record.pageno = f' [{pageno:5d}]' + record.pageno = f'{pageno:5d} ' else: record.pageno = '' return True diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 91cd608a..efa24c7f 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -95,7 +95,7 @@ def configure_logging( if verbosity >= 2: fmt = '%(levelname)7s %(name)s -%(pageno)s %(message)s' else: - fmt = '%(levelname)7s -%(pageno)s %(message)s' + fmt = '%(pageno)s%(message)s' use_colors = progress_bar_friendly if not coloredlogs: diff --git a/tests/test_main.py b/tests/test_main.py index 39abbe98..2f843536 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -331,7 +331,7 @@ def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf) ) assert p.returncode == ExitCode.child_process_error assert not os.path.exists(no_outpdf) - assert "ERROR" in err + assert "uncaught exception" in err print(out) print(err) From 2155bcacb46ba34bab7d62a9497dfd516e2a2c42 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 00:30:38 -0700 Subject: [PATCH 397/880] Loosen test language requirements - eng/deu --- tests/test_api.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/test_api.py b/tests/test_api.py index 2842d224..038fc778 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -62,5 +62,7 @@ def test_tqdm_console(): def test_language_list(): - with pytest.raises(ocrmypdf.exceptions.InputFileError): - ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'ita']) + with pytest.raises( + [ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError] + ): + ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'deu']) From 9e3e4f2687690cc18e96c48afb7e91672bb09b96 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 02:17:55 -0700 Subject: [PATCH 398/880] Improve help text about aborting due to text --- docs/errors.rst | 6 ++++++ src/ocrmypdf/_pipeline.py | 3 ++- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/errors.rst b/docs/errors.rst index 825cd656..328080ad 100644 --- a/docs/errors.rst +++ b/docs/errors.rst @@ -22,6 +22,12 @@ As the error message suggests, your options are: - ``ocrmypdf --skip-text`` to skip OCR and other processing on any pages that contain text. Text pages will be copied into the output PDF without modification. +- ``ocrmypdf --redo-ocr`` to scan the file for any existing OCR + (non-printing text), remove it, and do OCR again. This is one way + to take advantage of improvements in OCR accuracy. Printable vector + text is excluded from OCR, so this can be used on files that contain + a mix of digital and scanned files. + Input file 'filename' is not a valid PDF ======================================== diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 98ed32fd..7d994d0f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -252,7 +252,8 @@ def is_ocr_required(page_context): elif pageinfo.has_text: if not options.force_ocr and not (options.skip_text or options.redo_ocr): raise PriorOcrFoundError( - "page already has text! - aborting (use --force-ocr to force OCR)" + "page already has text! - aborting (use --force-ocr to force OCR; " + " see also help for the arguments --skip-text and --redo-ocr" ) elif options.force_ocr: log.info("page already has text! - rasterizing text and running OCR anyway") From 957fb1494e4724686ad14e03abb440cfd6e01776 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 02:26:20 -0700 Subject: [PATCH 399/880] pytest picky about list vs tuple --- tests/test_api.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_api.py b/tests/test_api.py index 038fc778..acc24683 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -63,6 +63,6 @@ def test_tqdm_console(): def test_language_list(): with pytest.raises( - [ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError] + (ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError) ): ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'deu']) From 31b5f63f85029944269e6bfda0b6028b03e534f4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 3 Apr 2020 22:04:42 -0700 Subject: [PATCH 400/880] hocrtransform: cleanup/PEP8 Some API breaking changes. --- src/ocrmypdf/_pipeline.py | 8 ++--- src/ocrmypdf/hocrtransform.py | 68 ++++++++++++++++++----------------- tests/test_hocrtransform.py | 2 +- 3 files changed, 40 insertions(+), 38 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 7d994d0f..034b831c 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -597,10 +597,10 @@ def render_hocr_page(hocr, page_context): hocrtransform = HocrTransform(hocr, dpi) hocrtransform.to_pdf( output_file, - imageFileName=None, - showBoundingboxes=False, - invisibleText=True, - interwordSpaces=True, + image_filename=None, + show_bounding_boxes=False, + invisible_text=True, + interword_spaces=True, ) return output_file diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 809f858e..a60fe959 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -64,9 +64,9 @@ class HocrTransform: {'ff': 'ff', 'ffi': 'f‌f‌i', 'ffl': 'f‌f‌l', 'fi': 'fi', 'fl': 'fl'} ) - def __init__(self, hocrFileName, dpi): + def __init__(self, hocr_filename: str, dpi: float): self.dpi = dpi - self.hocr = ElementTree.parse(hocrFileName) + self.hocr = ElementTree.parse(hocr_filename) # if the hOCR file has a namespace, ElementTree requires its use to # find elements @@ -114,12 +114,12 @@ class HocrTransform: return text @classmethod - def element_coordinates(cls, element): + def element_coordinates(cls, element) -> Rect: """ Returns a tuple containing the coordinates of the bounding box around an element """ - out = (0, 0, 0, 0) + out = Rect._make(0 for _ in range(4)) if 'title' in element.attrib: matches = cls.box_pattern.search(element.attrib['title']) if matches: @@ -136,7 +136,7 @@ class HocrTransform: matches = cls.baseline_pattern.search(element.attrib['title']) if matches: return float(matches.group(1)), int(matches.group(2)) - return (0, 0) + return (0.0, 0.0) def pt_from_pixel(self, pxl): """ @@ -145,7 +145,7 @@ class HocrTransform: return Rect._make((c / self.dpi * inch) for c in pxl) @classmethod - def replace_unsupported_chars(cls, s): + def replace_unsupported_chars(cls, s: str): """ Given an input string, returns the corresponding string that: - is available in the helvetica facetype @@ -155,12 +155,12 @@ class HocrTransform: def to_pdf( self, - outFileName, - imageFileName=None, - showBoundingboxes=False, - fontname="Helvetica", - invisibleText=False, - interwordSpaces=False, + out_filename: str, + image_filename: str = None, + show_bounding_boxes: bool = False, + fontname: str = "Helvetica", + invisible_text: bool = False, + interword_spaces: bool = False, ): """ Creates a PDF file with an image superimposed on top of the text. @@ -172,7 +172,9 @@ class HocrTransform: """ # create the PDF file # page size in points (1/72 in.) - pdf = Canvas(outFileName, pagesize=(self.width, self.height), pageCompression=1) + pdf = Canvas( + out_filename, pagesize=(self.width, self.height), pageCompression=1 + ) # draw bounding box for each paragraph # light blue for bounding box of paragraph @@ -190,7 +192,7 @@ class HocrTransform: pt = self.pt_from_pixel(pxl_coords) # draw the bbox border - if showBoundingboxes: # pragma: no cover + if show_bounding_boxes: # pragma: no cover pdf.rect( pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1 ) @@ -205,9 +207,9 @@ class HocrTransform: line, "ocrx_word", fontname, - invisibleText, - interwordSpaces, - showBoundingboxes, + invisible_text, + interword_spaces, + show_bounding_boxes, ) if not found_lines: @@ -218,13 +220,13 @@ class HocrTransform: root, "ocrx_word", fontname, - invisibleText, - interwordSpaces, - showBoundingboxes, + invisible_text, + interword_spaces, + show_bounding_boxes, ) # put the image on the page, scaled to fill the page - if imageFileName is not None: - pdf.drawImage(imageFileName, 0, 0, width=self.width, height=self.height) + if image_filename is not None: + pdf.drawImage(image_filename, 0, 0, width=self.width, height=self.height) # finish up the page and save it pdf.showPage() @@ -236,13 +238,13 @@ class HocrTransform: def _do_line( self, - pdf, + pdf: Canvas, line, - elemclass, - fontname, - invisibleText, - interwordSpaces, - showBoundingboxes, + elemclass: str, + fontname: str, + invisible_text: bool, + interword_spaces: bool, + show_bounding_boxes: bool, ): pxl_line_coords = self.element_coordinates(line) line_box = self.pt_from_pixel(pxl_line_coords) @@ -262,14 +264,14 @@ class HocrTransform: # on a sloped baseline and the edge of the bounding box. fontsize = (line_height - abs(intercept)) / cos_a text.setFont(fontname, fontsize) - if invisibleText: + if invisible_text: text.setTextRenderMode(3) # Invisible (indicates OCR text) # Intercept is normally negative, so this places it above the bottom # of the line box baseline_y2 = self.height - (line_box.y2 + intercept) - if showBoundingboxes: # pragma: no cover + if show_bounding_boxes: # pragma: no cover # draw the baseline in magenta, dashed pdf.setDash() pdf.setStrokeColorRGB(0.95, 0.65, 0.95) @@ -298,7 +300,7 @@ class HocrTransform: pxl_coords = self.element_coordinates(elem) box = self.pt_from_pixel(pxl_coords) - if interwordSpaces: + if interword_spaces: # if `--interword-spaces` is true, append a space # to the end of each text element to allow simpler PDF viewers # such as PDF.js to better recognize words in search and copy @@ -318,7 +320,7 @@ class HocrTransform: font_width = pdf.stringWidth(elemtxt, fontname, fontsize) # draw the bbox border - if showBoundingboxes: # pragma: no cover + if show_bounding_boxes: # pragma: no cover pdf.rect( box.x1, self.height - line_box.y2, box_width, line_height, fill=0 ) @@ -385,5 +387,5 @@ if __name__ == "__main__": args.outputfile, args.image, args.boundingboxes, - interwordSpaces=args.interword_spaces, + interword_spaces=args.interword_spaces, ) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index 19e4684d..13b1f601 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -41,6 +41,6 @@ def test_mono_image(blank_hocr, outdir): im.save(outdir / 'mono.tif', format='TIFF') hocr = hocrtransform.HocrTransform(str(blank_hocr), 300) - hocr.to_pdf(str(outdir / 'mono.pdf'), imageFileName=str(outdir / 'mono.tif')) + hocr.to_pdf(str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif')) qpdf.check(str(outdir / 'mono.pdf')) From 4581027246bcf76a0a06fc9d9d1fe8230af24f3d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Jan 2020 17:51:09 -0800 Subject: [PATCH 401/880] Drop support for pdfminer.six 20181108 This version required a patch that has since been mainlined, and also did not declare its dependency on chardet correctly. We can remove both hacks now. --- setup.py | 3 +-- src/ocrmypdf/pdfinfo/layout.py | 48 ---------------------------------- 2 files changed, 1 insertion(+), 50 deletions(-) diff --git a/setup.py b/setup.py index d0e05c0d..eb0d783a 100644 --- a/setup.py +++ b/setup.py @@ -95,11 +95,10 @@ setup( use_scm_version={'version_scheme': 'post-release'}, cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'], install_requires=[ - 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20200124', + 'pdfminer.six >= 20191110, <= 20200124', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index af9d7961..f763a9eb 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -36,54 +36,6 @@ from ..exceptions import EncryptedPdfError STRIP_NAME = re.compile(r'[0-9]+') -# -# pdfminer 20181108 patches -# - -if pdfminer.__version__ == '20181108': - - def name2unicode(name): - """Fix pdfminer's name2unicode function - - Font cids that are mapped to names of the form /g123 seem to be, by convention - characters with no corresponding Unicode entry. These can be subsetted fonts - or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode, - barring a ToUnicode data structure. - """ - if name in glyphname2unicode: - return glyphname2unicode[name] - if name.startswith('g') or name.startswith('a'): - raise KeyError(name) - if name.startswith('uni'): - try: - return chr(int(name[3:], 16)) - except ValueError: # Not hexadecimal - raise KeyError(name) - m = STRIP_NAME.search(name) - if not m: - raise KeyError(name) - return chr(int(m.group(0))) - - pdfminer.encodingdb.name2unicode = name2unicode - - original_PDFFont_init = PDFFont.__init__ - - def PDFFont__init__(self, descriptor, widths, default_width=None): - original_PDFFont_init(self, descriptor, widths, default_width) - # PDF spec says descent should be negative - # A font with a positive descent implies it floats entirely above the - # baseline, i.e. it's not really a baseline anymore. I have fonts that - # claim a positive descent, but treating descent as positive always seems - # to misposition text. - if self.descent > 0: - self.descent = -self.descent - - PDFFont.__init__ = PDFFont__init__ - -# -# end of pdfminer 20181108 patches -# - original_PDFSimpleFont_init = PDFSimpleFont.__init__ From 0c50eedb2a9bb6674700ebc35771d75403e6fa7c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 02:55:08 -0700 Subject: [PATCH 402/880] Support pdfminer.six 20200402 --- requirements/main.txt | 2 +- setup.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 5b56bd04..8934d40f 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,7 +3,7 @@ # installation cffi == 1.14.0 img2pdf == 0.3.3 -pdfminer.six == 20200124 +pdfminer.six == 20200402 pikepdf == 1.10.2 Pillow == 7.0.0 reportlab == 3.5.34 diff --git a/setup.py b/setup.py index 0b4a61da..84107620 100644 --- a/setup.py +++ b/setup.py @@ -98,7 +98,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20200124', + 'pdfminer.six >= 20181108, <= 20200402', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 509e75eaffffe129d3dff5aef8dce6bdfd55c403 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 02:56:46 -0700 Subject: [PATCH 403/880] v9.7.2 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 86a1b9f3..66f1f576 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,12 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.7.2 +====== + +- Fixed an issue with ``ocrmypdf.ocr(...language=)`` not accepting a list of + languages as documented. +- Updated setup.py to confirm that pdfminer.six version 20200402 is supported. v9.7.1 ====== From 58abb5785cf55d0cfddeee017e81ca4a8250a94c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 15 Apr 2020 02:26:20 -0700 Subject: [PATCH 404/880] pytest picky about list vs tuple --- tests/test_api.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/test_api.py b/tests/test_api.py index 2842d224..acc24683 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -62,5 +62,7 @@ def test_tqdm_console(): def test_language_list(): - with pytest.raises(ocrmypdf.exceptions.InputFileError): - ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'ita']) + with pytest.raises( + (ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError) + ): + ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'deu']) From 57771f06a32f4d540590956e20c6a540f8719ecc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 Apr 2020 15:38:33 -0700 Subject: [PATCH 405/880] Refactor xy-pair for resolution to tuple --- src/ocrmypdf/_pipeline.py | 33 +++++++++++++---------------- src/ocrmypdf/exec/ghostscript.py | 13 ++++++------ src/ocrmypdf/pdfinfo/info.py | 36 +++++++++++++------------------- tests/test_ghostscript.py | 6 ++---- tests/test_main.py | 8 +++---- tests/test_optimize.py | 2 +- tests/test_pdfinfo.py | 10 ++++----- tests/test_preprocessing.py | 22 +++++++------------ tests/test_rotation.py | 3 +-- 9 files changed, 57 insertions(+), 76 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 034b831c..842b4f1a 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -201,12 +201,12 @@ def validate_pdfinfo_options(context): def get_page_dpi(pageinfo, options): "Get the DPI when nonsquare DPI is tolerable" xres = max( - pageinfo.xres or VECTOR_PAGE_DPI, + pageinfo.xyres[0] or VECTOR_PAGE_DPI, options.oversample or 0, VECTOR_PAGE_DPI if pageinfo.has_vector else 0, ) yres = max( - pageinfo.yres or VECTOR_PAGE_DPI, + pageinfo.xyres[1] or VECTOR_PAGE_DPI, options.oversample or 0, VECTOR_PAGE_DPI if pageinfo.has_vector else 0, ) @@ -215,8 +215,8 @@ def get_page_dpi(pageinfo, options): def get_page_square_dpi(pageinfo, options): "Get the DPI when we require xres == yres, scaled to physical units" - xres = pageinfo.xres or 0 - yres = pageinfo.yres or 0 + xres = pageinfo.xyres[0] or 0 + yres = pageinfo.xyres[1] or 0 userunit = pageinfo.userunit or 1 return float( max( @@ -232,8 +232,8 @@ def get_canvas_square_dpi(pageinfo, options): """Get the DPI when we require xres == yres, in Postscript units""" return float( max( - (pageinfo.xres) or VECTOR_PAGE_DPI, - (pageinfo.yres) or VECTOR_PAGE_DPI, + (pageinfo.xyres[0]) or VECTOR_PAGE_DPI, + (pageinfo.xyres[1]) or VECTOR_PAGE_DPI, VECTOR_PAGE_DPI if pageinfo.has_vector else 0, options.oversample or 0, ) @@ -321,9 +321,8 @@ def rasterize_preview(input_file, page_context): ghostscript.rasterize_pdf( input_file, output_file, - xres=canvas_dpi, - yres=canvas_dpi, raster_device='jpeggray', + xyres=(canvas_dpi, canvas_dpi), page_dpi=(page_dpi, page_dpi), pageno=page_context.pageinfo.pageno + 1, ) @@ -436,9 +435,8 @@ def rasterize( ghostscript.rasterize_pdf( input_file, output_file, - xres=canvas_dpi, - yres=canvas_dpi, raster_device=device, + xyres=(canvas_dpi, canvas_dpi), page_dpi=(page_dpi, page_dpi), pageno=pageinfo.pageno + 1, rotation=correction, @@ -488,8 +486,7 @@ def create_ocr_image(image, page_context): # pink = ImageColor.getcolor('#ff0080', im.mode) draw = ImageDraw.ImageDraw(im) - xres, yres = im.info['dpi'] - log.debug('resolution %r %r' % (xres, yres)) + log.debug('resolution %r', im.info['dpi']) if not options.force_ocr: # Do not mask text areas when forcing OCR, because we need to OCR @@ -505,12 +502,12 @@ def create_ocr_image(image, page_context): # without regard whatever resolution is in pageinfo (may differ or # be None) bbox = [float(v) for v in textarea] - xscale, yscale = float(xres) / 72.0, float(yres) / 72.0 + xyscale = tuple(float(coord) / 72.0 for coord in im.info['dpi']) pixcoords = [ - bbox[0] * xscale, - im.height - bbox[3] * yscale, - bbox[2] * xscale, - im.height - bbox[1] * yscale, + bbox[0] * xyscale[0], + im.height - bbox[3] * xyscale[1], + bbox[2] * xyscale[0], + im.height - bbox[1] * xyscale[1], ] pixcoords = [int(round(c)) for c in pixcoords] log.debug('blanking %r', pixcoords) @@ -524,7 +521,7 @@ def create_ocr_image(image, page_context): del draw # Pillow requires integer DPI - dpi = round(xres), round(yres) + dpi = tuple(round(coord) for coord in im.info['dpi']) im.save(output_file, dpi=dpi) return output_file diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index de4cbda2..d85ce801 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -135,32 +135,31 @@ def extract_text(input_file, pageno=1): def rasterize_pdf( input_file, output_file, - xres, - yres, + *, raster_device, + xyres, pageno=1, page_dpi=None, rotation=None, filter_vector=False, ): - """Rasterize one page of a PDF at resolution (xres, yres) in canvas units. + """Rasterize one page of a PDF at resolution xyres in canvas units. The image is sized to match the integer pixels dimensions implied by - (xres, yres) even if those numbers are noninteger. The image's DPI will + (xyres[0], xyres[1]) even if those numbers are noninteger. The image's DPI will be overridden with the values in page_dpi. :param input_file: pathlike :param output_file: pathlike - :param xres: resolution at which to rasterize page - :param yres: :param raster_device: + :param xyres: resolution at which to rasterize page :param pageno: page number to rasterize (beginning at page 1) :param page_dpi: resolution tuple (x, y) overriding output image DPI :param rotation: 0, 90, 180, 270: clockwise angle to rotate page :param filter_vector: if True, remove vector graphics objects :return: """ - res = round(xres, 6), round(yres, 6) + res = round(xyres[0], 6), round(xyres[1], 6) if not page_dpi: page_dpi = res diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 329661b3..bb9605e1 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -356,12 +356,11 @@ class ImageInfo: return self._enc @property - def xres(self): - return _get_dpi(self._shorthand, (self._width, self._height))[0] - - @property - def yres(self): - return _get_dpi(self._shorthand, (self._width, self._height))[1] + def xyres(self): + return ( + _get_dpi(self._shorthand, (self._width, self._height))[0], + _get_dpi(self._shorthand, (self._width, self._height))[1], + ) def __repr__(self): class_locals = { @@ -371,7 +370,7 @@ class ImageInfo: } return ( "" + "{comp} {bpc} {enc} {xyres}>" ).format(**class_locals) @@ -607,9 +606,9 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] if pageinfo['images']: - xres = Decimal(max(image.xres for image in pageinfo['images'])) - yres = Decimal(max(image.yres for image in pageinfo['images'])) - pageinfo['xres'], pageinfo['yres'] = xres, yres + xres = Decimal(max(image.xyres[0] for image in pageinfo['images'])) + yres = Decimal(max(image.xyres[1] for image in pageinfo['images'])) + pageinfo['xyres'] = xres, yres pageinfo['width_pixels'] = int(round(xres * pageinfo['width_inches'])) pageinfo['height_pixels'] = int(round(yres * pageinfo['height_inches'])) @@ -679,11 +678,11 @@ class PageInfo: @property def width_pixels(self): - return int(round(self.width_inches * self.xres)) + return int(round(self.width_inches * self.xyres[0])) @property def height_pixels(self): - return int(round(self.height_inches * self.yres)) + return int(round(self.height_inches * self.xyres[1])) @property def rotation(self): @@ -723,12 +722,8 @@ class PageInfo: ) @property - def xres(self): - return self._pageinfo.get('xres', None) - - @property - def yres(self): - return self._pageinfo.get('yres', None) + def xyres(self): + return self._pageinfo.get('xyres', (0, 0)) @property def userunit(self): @@ -743,14 +738,13 @@ class PageInfo: def __repr__(self): return ( - '' + '' ).format( self.pageno, self.width_inches, self.height_inches, self.rotation, - self.xres, - self.yres, + self.xyres, self.has_text, ) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 5bb7b76f..2e1543ea 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -76,9 +76,8 @@ def test_rasterize_size(francais, outdir, caplog): rasterize_pdf( path, outdir / 'out.png', - target_size[0] / page_size[0], - target_size[1] / page_size[1], raster_device='pngmono', + xyres=(target_size[0] / page_size[0], target_size[1] / page_size[1]), page_dpi=forced_dpi, ) @@ -99,9 +98,8 @@ def test_rasterize_rotated(francais, outdir, caplog): rasterize_pdf( path, outdir / 'out.png', - target_size[0] / page_size[0], - target_size[1] / page_size[1], raster_device='pngmono', + xyres=(target_size[0] / page_size[0], target_size[1] / page_size[1]), page_dpi=forced_dpi, rotation=90, ) diff --git a/tests/test_main.py b/tests/test_main.py index 2f843536..0583abfc 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -74,8 +74,8 @@ def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): pdfinfo = PdfInfo(oversampled_pdf) - print(pdfinfo[0].xres) - assert abs(pdfinfo[0].xres - 350) < 1 + print(pdfinfo[0].xyres[0]) + assert abs(pdfinfo[0].xyres[0] - 350) < 1 def test_repeat_ocr(resources, no_outpdf): @@ -393,8 +393,8 @@ def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): pdfinfo = PdfInfo(outpdf) image = pdfinfo[0].images[0] - assert isclose(image.xres, image.yres) - assert isclose(image.xres, 2400) + assert isclose(image.xyres[0], image.xyres[1]) + assert isclose(image.xyres[0], 2400) def test_overlay(spoof_tesseract_noop, resources, outpdf): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 6cdc1c71..5f198143 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -43,7 +43,7 @@ def test_mono_not_inverted(resources, outdir): opt.main(infile, outdir / 'out.pdf', level=3) rasterize_pdf( - outdir / 'out.pdf', outdir / 'im.png', xres=10, yres=10, raster_device='pnggray' + outdir / 'out.pdf', outdir / 'im.png', raster_device='pnggray', xyres=(10, 10) ) with Image.open(fspath(outdir / 'im.png')) as im: diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index facf3b6f..40f49c94 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -85,8 +85,8 @@ def test_single_page_image(outdir): assert pdfimage.color == Colorspace.gray # DPI in a 1"x1" is the image width - assert isclose(pdfimage.xres, 8) - assert isclose(pdfimage.yres, 8) + assert isclose(pdfimage.xyres[0], 8) + assert isclose(pdfimage.xyres[1], 8) def test_single_page_inline_image(outdir): @@ -105,7 +105,7 @@ def test_single_page_inline_image(outdir): info = pdfinfo.PdfInfo(filename) print(info) pdfimage = info[0].images[0] - assert isclose(pdfimage.xres, 8) + assert isclose(pdfimage.xyres[0], 8) assert pdfimage.color == Colorspace.gray assert pdfimage.width == 8 @@ -117,7 +117,7 @@ def test_jpeg(resources, outdir): pdfimage = pdf[0].images[0] assert pdfimage.enc == Encoding.jpeg - assert isclose(pdfimage.xres, 150) + assert isclose(pdfimage.xyres[0], 150) def test_form_xobject(resources): @@ -139,7 +139,7 @@ def test_no_contents(resources): def test_oversized_page(resources): pdf = pdfinfo.PdfInfo(resources / 'poster.pdf') image = pdf[0].images[0] - assert image.width * image.xres > 200, "this is supposed to be oversized" + assert image.width * image.xyres[0] > 200, "this is supposed to be oversized" def test_pickle(resources): diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 50eb6b56..00e4aeda 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -50,12 +50,7 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): deskewed_png = outdir / 'deskewed.png' ghostscript.rasterize_pdf( - deskewed_pdf, - deskewed_png, - xres=150, - yres=150, - raster_device='pngmono', - pageno=1, + deskewed_pdf, deskewed_png, raster_device='pngmono', xyres=(150, 150), pageno=1 ) pix = Pix.open(deskewed_png) @@ -82,7 +77,7 @@ def test_remove_background(spoof_tesseract_noop, resources, outdir): output_png = outdir / 'remove_bg.png' ghostscript.rasterize_pdf( - output_pdf, output_png, xres=100, yres=100, raster_device='png16m', pageno=1 + output_pdf, output_png, raster_device='png16m', xyres=(100, 100), pageno=1 ) # The output image should contain pure white and black @@ -122,7 +117,7 @@ def test_exotic_image( def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xres != in_pageinfo[0].yres + assert in_pageinfo[0].xyres[0] != in_pageinfo[0].xyres[1] check_ocrmypdf( resources / 'aspect.pdf', @@ -135,8 +130,7 @@ def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpd out_pageinfo = PdfInfo(outpdf) # Confirm resolution was kept the same - assert in_pageinfo[0].xres == out_pageinfo[0].xres - assert in_pageinfo[0].yres == out_pageinfo[0].yres + assert in_pageinfo[0].xyres == out_pageinfo[0].xyres @pytest.mark.parametrize('renderer', RENDERERS) @@ -145,7 +139,7 @@ def test_convert_to_square_resolution( ): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xres != in_pageinfo[0].yres + assert in_pageinfo[0].xyres[0] != in_pageinfo[0].xyres[1] # --force-ocr requires means forced conversion to square resolution check_ocrmypdf( @@ -162,7 +156,7 @@ def test_convert_to_square_resolution( in_p0, out_p0 = in_pageinfo[0], out_pageinfo[0] # Resolution show now be equal - assert out_p0.xres == out_p0.yres + assert out_p0.xyres[0] == out_p0.xyres[1] # Page size should match input page size assert isclose(in_p0.width_inches, out_p0.width_inches) @@ -170,7 +164,7 @@ def test_convert_to_square_resolution( # Because we rasterized the page to produce a new image, it should occupy # the entire page - out_im_w = out_p0.images[0].width / out_p0.images[0].xres - out_im_h = out_p0.images[0].height / out_p0.images[0].yres + out_im_w = out_p0.images[0].width / out_p0.images[0].xyres[0] + out_im_h = out_p0.images[0].height / out_p0.images[0].xyres[1] assert isclose(out_p0.width_inches, out_im_w) assert isclose(out_p0.height_inches, out_im_h) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 796c2bfb..101ee8e7 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -58,9 +58,8 @@ def check_monochrome_correlation( ghostscript.rasterize_pdf( pdf, png, - xres=100, - yres=100, raster_device='pngmono', + xyres=(100, 100), pageno=pageno, rotation=0, ) From 94c52a6fa3d92f7a5f85af6f1fecdd7ae1e76310 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 24 Apr 2020 04:12:05 -0700 Subject: [PATCH 406/880] Refactor 'xyres' into Resolution --- src/ocrmypdf/_pipeline.py | 70 +++++++++++++++++--------------- src/ocrmypdf/exec/ghostscript.py | 48 +++++++++++----------- src/ocrmypdf/helpers.py | 34 ++++++++++++++++ src/ocrmypdf/pdfinfo/info.py | 40 ++++++++---------- tests/test_ghostscript.py | 13 ++++-- tests/test_main.py | 8 ++-- tests/test_optimize.py | 6 ++- tests/test_pdfinfo.py | 10 ++--- tests/test_preprocessing.py | 25 ++++++++---- tests/test_rotation.py | 3 +- 10 files changed, 153 insertions(+), 104 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 842b4f1a..bb2f7e18 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -39,7 +39,7 @@ from .exceptions import ( UnsupportedImageFormatError, ) from .exec import ghostscript, tesseract -from .helpers import safe_symlink +from .helpers import Resolution, safe_symlink from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps @@ -99,7 +99,7 @@ def triage_image_file(input_file, output_file, options): layout_fun = img2pdf.default_layout_fun if options.image_dpi: layout_fun = img2pdf.get_fixed_dpi_layout_fun( - (options.image_dpi, options.image_dpi) + Resolution(options.image_dpi, options.image_dpi) ) with open(output_file, 'wb') as outf: img2pdf.convert( @@ -201,43 +201,45 @@ def validate_pdfinfo_options(context): def get_page_dpi(pageinfo, options): "Get the DPI when nonsquare DPI is tolerable" xres = max( - pageinfo.xyres[0] or VECTOR_PAGE_DPI, - options.oversample or 0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + pageinfo.dpi.x or VECTOR_PAGE_DPI, + options.oversample or 0.0, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, ) yres = max( - pageinfo.xyres[1] or VECTOR_PAGE_DPI, + pageinfo.dpi.y or VECTOR_PAGE_DPI, options.oversample or 0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, ) - return (float(xres), float(yres)) + return Resolution(float(xres), float(yres)) -def get_page_square_dpi(pageinfo, options): +def get_page_square_dpi(pageinfo, options) -> Resolution: "Get the DPI when we require xres == yres, scaled to physical units" - xres = pageinfo.xyres[0] or 0 - yres = pageinfo.xyres[1] or 0 - userunit = pageinfo.userunit or 1 - return float( + xres = pageinfo.dpi.x or 0.0 + yres = pageinfo.dpi.y or 0.0 + userunit = float(pageinfo.userunit) or 1.0 + units = float( max( (xres * userunit) or VECTOR_PAGE_DPI, (yres * userunit) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - options.oversample or 0, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + options.oversample or 0.0, ) ) + return Resolution(units, units) -def get_canvas_square_dpi(pageinfo, options): +def get_canvas_square_dpi(pageinfo, options) -> Resolution: """Get the DPI when we require xres == yres, in Postscript units""" - return float( + units = float( max( - (pageinfo.xyres[0]) or VECTOR_PAGE_DPI, - (pageinfo.xyres[1]) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0, - options.oversample or 0, + (pageinfo.dpi.x) or VECTOR_PAGE_DPI, + (pageinfo.dpi.y) or VECTOR_PAGE_DPI, + VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + options.oversample or 0.0, ) ) + return Resolution(units, units) def is_ocr_required(page_context): @@ -322,8 +324,8 @@ def rasterize_preview(input_file, page_context): input_file, output_file, raster_device='jpeggray', - xyres=(canvas_dpi, canvas_dpi), - page_dpi=(page_dpi, page_dpi), + raster_dpi=canvas_dpi, + page_dpi=page_dpi, pageno=page_context.pageinfo.pageno + 1, ) return output_file @@ -436,8 +438,8 @@ def rasterize( input_file, output_file, raster_device=device, - xyres=(canvas_dpi, canvas_dpi), - page_dpi=(page_dpi, page_dpi), + raster_dpi=canvas_dpi, + page_dpi=page_dpi, pageno=pageinfo.pageno + 1, rotation=correction, filter_vector=remove_vectors, @@ -458,7 +460,7 @@ def preprocess_remove_background(input_file, page_context): def preprocess_deskew(input_file, page_context): output_file = page_context.get_path('pp_deskew.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - leptonica.deskew(input_file, output_file, dpi) + leptonica.deskew(input_file, output_file, dpi.x) return output_file @@ -467,7 +469,7 @@ def preprocess_clean(input_file, page_context): output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - unpaper.clean(input_file, output_file, dpi, page_context.options.unpaper_args) + unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args) return output_file @@ -558,12 +560,14 @@ def create_visible_page_jpg(image, page_context): # square DPI used to rasterize. When the preview image was # rasterized, it was also converted to square resolution, which is # what we want to give tesseract, so keep it square. - fallback_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi)) + if 'dpi' in im.info: + dpi = Resolution(*im.info['dpi']) + else: + # Fallback to page-implied DPI + dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) # Pillow requires integer DPI - dpi = round(dpi[0]), round(dpi[1]) - im.save(output_file, format='JPEG', dpi=dpi) + im.save(output_file, format='JPEG', dpi=dpi.to_int()) return output_file @@ -576,7 +580,7 @@ def create_pdf_page_from_image(image, page_context): # sandwich renderer would be fine. output_file = page_context.get_path('visible.pdf') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi)) + layout_fun = img2pdf.get_fixed_dpi_layout_fun(dpi) # This create a single page PDF with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: @@ -591,7 +595,7 @@ def create_pdf_page_from_image(image, page_context): def render_hocr_page(hocr, page_context): output_file = page_context.get_path('ocr_hocr.pdf') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - hocrtransform = HocrTransform(hocr, dpi) + hocrtransform = HocrTransform(hocr, dpi.x) # square hocrtransform.to_pdf( output_file, image_filename=None, diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index d85ce801..5c27488f 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -21,7 +21,6 @@ import logging import os import re import warnings -from contextlib import suppress from functools import lru_cache from io import BytesIO from os import fspath @@ -31,8 +30,9 @@ from subprocess import PIPE, CalledProcessError from PIL import Image -from ..exceptions import MissingDependencyError, SubprocessOutputError -from . import get_version, run +from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError +from ocrmypdf.exec import get_version, run +from ocrmypdf.helpers import Resolution log = logging.getLogger(__name__) @@ -62,7 +62,7 @@ def version(): return get_version(GS) -def jpeg_passthrough_available(): +def jpeg_passthrough_available() -> bool: """Returns True if the installed version of Ghostscript supports JPEG passthru Prior to 9.23, Ghostscript decode and re-encoded JPEGs internally. In 9.23 @@ -79,7 +79,7 @@ def jpeg_passthrough_available(): return version() >= '9.24' -def _gs_error_reported(stream): +def _gs_error_reported(stream) -> bool: return re.search(r'error', stream, flags=re.IGNORECASE) @@ -133,35 +133,35 @@ def extract_text(input_file, pageno=1): def rasterize_pdf( - input_file, - output_file, + input_file: os.PathLike, + output_file: os.PathLike, *, - raster_device, - xyres, - pageno=1, - page_dpi=None, - rotation=None, - filter_vector=False, + raster_device: str, + raster_dpi: Resolution, + pageno: int = 1, + page_dpi: Resolution = None, + rotation: int = None, + filter_vector: bool = False, ): - """Rasterize one page of a PDF at resolution xyres in canvas units. + """Rasterize one page of a PDF at resolution raster_dpi in canvas units. The image is sized to match the integer pixels dimensions implied by - (xyres[0], xyres[1]) even if those numbers are noninteger. The image's DPI will + raster_dpi even if those numbers are noninteger. The image's DPI will be overridden with the values in page_dpi. :param input_file: pathlike :param output_file: pathlike :param raster_device: - :param xyres: resolution at which to rasterize page + :param raster_dpi: resolution at which to rasterize page :param pageno: page number to rasterize (beginning at page 1) :param page_dpi: resolution tuple (x, y) overriding output image DPI :param rotation: 0, 90, 180, 270: clockwise angle to rotate page :param filter_vector: if True, remove vector graphics objects :return: """ - res = round(xyres[0], 6), round(xyres[1], 6) + raster_dpi = raster_dpi.round(6) if not page_dpi: - page_dpi = res + page_dpi = raster_dpi args_gs = ( [ @@ -173,7 +173,7 @@ def rasterize_pdf( f'-sDEVICE={raster_device}', f'-dFirstPage={pageno}', f'-dLastPage={pageno}', - f'-r{res[0]:f}x{res[1]:f}', + f'-r{raster_dpi.x:f}x{raster_dpi.y:f}', ] + (['-dFILTERVECTOR'] if filter_vector else []) + [ @@ -210,17 +210,17 @@ def rasterize_pdf( elif rotation == 270: im = im.transpose(Image.ROTATE_270) if rotation % 180 == 90: - page_dpi = page_dpi[1], page_dpi[0] + page_dpi = page_dpi.flip_axis() im.save(fspath(output_file), dpi=page_dpi) def generate_pdfa( pdf_pages, - output_file, - compression, + output_file: os.PathLike, + compression: str, threads=None, # deprecated parameter - pdf_version='1.5', - pdfa_part='2', + pdf_version: str = '1.5', + pdfa_part: str = '2', ): """Generate a PDF/A. diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 55707082..c57aca35 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -20,14 +20,48 @@ import multiprocessing import os import shutil import warnings +from collections import namedtuple from collections.abc import Iterable from contextlib import suppress from functools import wraps +from math import inf, isclose from pathlib import Path log = logging.getLogger(__name__) +class Resolution(namedtuple('Resolution', ('x', 'y'))): + __slots__ = () + + def round(self, ndigits): + return Resolution(round(self.x, ndigits), round(self.y, ndigits)) + + def to_int(self): + return Resolution(int(round(self.x)), int(round(self.y))) + + @property + def is_square(self): + return isclose(self.x, self.y, rel_tol=1e-3) + + def take_max(self, vals, yvals=None): + if yvals is not None: + return Resolution(max(self.x, *vals), max(self.y, *yvals)) + max_x, max_y = self.x, self.y + for x, y in vals: + max_x = max(x, max_x) + max_y = max(y, max_y) + return Resolution(max_x, max_y) + + def flip_axis(self): + return Resolution(self.y, self.x) + + def __str__(self): + return f"{self.x:f}x{self.y:f}" + + def __repr__(self): + return f"Resolution({self.x}x{self.y} dpi)" + + def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **kwargs): """ Helper function: relinks soft symbolic link if necessary diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index bb9605e1..565ed745 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -32,6 +32,7 @@ from tqdm import tqdm from ocrmypdf.exceptions import EncryptedPdfError from ocrmypdf.exec import ghostscript +from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import ghosttext from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes @@ -265,7 +266,7 @@ def _get_dpi(ctm_shorthand, image_size): dpi_w = scale_w * 72.0 dpi_h = scale_h * 72.0 - return dpi_w, dpi_h + return Resolution(dpi_w, dpi_h) class ImageInfo: @@ -356,11 +357,8 @@ class ImageInfo: return self._enc @property - def xyres(self): - return ( - _get_dpi(self._shorthand, (self._width, self._height))[0], - _get_dpi(self._shorthand, (self._width, self._height))[1], - ) + def dpi(self): + return _get_dpi(self._shorthand, (self._width, self._height)) def __repr__(self): class_locals = { @@ -370,7 +368,7 @@ class ImageInfo: } return ( "" + "{comp} {bpc} {enc} {dpi}>" ).format(**class_locals) @@ -606,11 +604,10 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] if pageinfo['images']: - xres = Decimal(max(image.xyres[0] for image in pageinfo['images'])) - yres = Decimal(max(image.xyres[1] for image in pageinfo['images'])) - pageinfo['xyres'] = xres, yres - pageinfo['width_pixels'] = int(round(xres * pageinfo['width_inches'])) - pageinfo['height_pixels'] = int(round(yres * pageinfo['height_inches'])) + dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images']) + pageinfo['dpi'] = dpi + pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches']))) + pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches']))) return pageinfo @@ -678,11 +675,11 @@ class PageInfo: @property def width_pixels(self): - return int(round(self.width_inches * self.xyres[0])) + return int(round(float(self.width_inches) * self.dpi.x)) @property def height_pixels(self): - return int(round(self.height_inches * self.xyres[1])) + return int(round(float(self.height_inches) * self.dpi.y)) @property def rotation(self): @@ -722,8 +719,8 @@ class PageInfo: ) @property - def xyres(self): - return self._pageinfo.get('xyres', (0, 0)) + def dpi(self): + return self._pageinfo.get('dpi', Resolution(0.0, 0.0)) @property def userunit(self): @@ -738,14 +735,9 @@ class PageInfo: def __repr__(self): return ( - '' - ).format( - self.pageno, - self.width_inches, - self.height_inches, - self.rotation, - self.xyres, - self.has_text, + f'' ) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 2e1543ea..da04ea84 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -24,6 +24,7 @@ from PIL import Image from ocrmypdf.exceptions import ExitCode from ocrmypdf.exec.ghostscript import rasterize_pdf +from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf @@ -71,13 +72,15 @@ def test_rasterize_size(francais, outdir, caplog): assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) target_size = Decimal('50.0'), Decimal('30.0') - forced_dpi = 42.0, 4242.0 + forced_dpi = Resolution(42.0, 4242.0) rasterize_pdf( path, outdir / 'out.png', raster_device='pngmono', - xyres=(target_size[0] / page_size[0], target_size[1] / page_size[1]), + raster_dpi=Resolution( + target_size[0] / page_size[0], target_size[1] / page_size[1] + ), page_dpi=forced_dpi, ) @@ -92,14 +95,16 @@ def test_rasterize_rotated(francais, outdir, caplog): assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) target_size = Decimal('50.0'), Decimal('30.0') - forced_dpi = 42.0, 4242.0 + forced_dpi = Resolution(42.0, 4242.0) caplog.set_level(logging.DEBUG) rasterize_pdf( path, outdir / 'out.png', raster_device='pngmono', - xyres=(target_size[0] / page_size[0], target_size[1] / page_size[1]), + raster_dpi=Resolution( + target_size[0] / page_size[0], target_size[1] / page_size[1] + ), page_dpi=forced_dpi, rotation=90, ) diff --git a/tests/test_main.py b/tests/test_main.py index 0583abfc..d4146ccf 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -74,8 +74,8 @@ def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): pdfinfo = PdfInfo(oversampled_pdf) - print(pdfinfo[0].xyres[0]) - assert abs(pdfinfo[0].xyres[0] - 350) < 1 + print(pdfinfo[0].dpi.x) + assert abs(pdfinfo[0].dpi.x - 350) < 1 def test_repeat_ocr(resources, no_outpdf): @@ -393,8 +393,8 @@ def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): pdfinfo = PdfInfo(outpdf) image = pdfinfo[0].images[0] - assert isclose(image.xyres[0], image.xyres[1]) - assert isclose(image.xyres[0], 2400) + assert isclose(image.dpi.x, image.dpi.y) + assert isclose(image.dpi.x, 2400) def test_overlay(spoof_tesseract_noop, resources, outpdf): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 5f198143..36e00c5e 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -26,6 +26,7 @@ from PIL import Image from ocrmypdf import optimize as opt from ocrmypdf.exec import jbig2enc, pngquant from ocrmypdf.exec.ghostscript import rasterize_pdf +from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101 @@ -43,7 +44,10 @@ def test_mono_not_inverted(resources, outdir): opt.main(infile, outdir / 'out.pdf', level=3) rasterize_pdf( - outdir / 'out.pdf', outdir / 'im.png', raster_device='pnggray', xyres=(10, 10) + outdir / 'out.pdf', + outdir / 'im.png', + raster_device='pnggray', + raster_dpi=Resolution(10, 10), ) with Image.open(fspath(outdir / 'im.png')) as im: diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 40f49c94..13fb8a8b 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -85,8 +85,8 @@ def test_single_page_image(outdir): assert pdfimage.color == Colorspace.gray # DPI in a 1"x1" is the image width - assert isclose(pdfimage.xyres[0], 8) - assert isclose(pdfimage.xyres[1], 8) + assert isclose(pdfimage.dpi.x, 8) + assert isclose(pdfimage.dpi.y, 8) def test_single_page_inline_image(outdir): @@ -105,7 +105,7 @@ def test_single_page_inline_image(outdir): info = pdfinfo.PdfInfo(filename) print(info) pdfimage = info[0].images[0] - assert isclose(pdfimage.xyres[0], 8) + assert isclose(pdfimage.dpi.x, 8) assert pdfimage.color == Colorspace.gray assert pdfimage.width == 8 @@ -117,7 +117,7 @@ def test_jpeg(resources, outdir): pdfimage = pdf[0].images[0] assert pdfimage.enc == Encoding.jpeg - assert isclose(pdfimage.xyres[0], 150) + assert isclose(pdfimage.dpi.x, 150) def test_form_xobject(resources): @@ -139,7 +139,7 @@ def test_no_contents(resources): def test_oversized_page(resources): pdf = pdfinfo.PdfInfo(resources / 'poster.pdf') image = pdf[0].images[0] - assert image.width * image.xyres[0] > 200, "this is supposed to be oversized" + assert image.width * image.dpi.x > 200, "this is supposed to be oversized" def test_pickle(resources): diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 00e4aeda..b90517eb 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -22,6 +22,7 @@ import pytest from PIL import Image from ocrmypdf.exec import ghostscript +from ocrmypdf.helpers import Resolution from ocrmypdf.leptonica import Pix from ocrmypdf.pdfinfo import PdfInfo @@ -50,7 +51,11 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): deskewed_png = outdir / 'deskewed.png' ghostscript.rasterize_pdf( - deskewed_pdf, deskewed_png, raster_device='pngmono', xyres=(150, 150), pageno=1 + deskewed_pdf, + deskewed_png, + raster_device='pngmono', + raster_dpi=Resolution(150, 150), + pageno=1, ) pix = Pix.open(deskewed_png) @@ -77,7 +82,11 @@ def test_remove_background(spoof_tesseract_noop, resources, outdir): output_png = outdir / 'remove_bg.png' ghostscript.rasterize_pdf( - output_pdf, output_png, raster_device='png16m', xyres=(100, 100), pageno=1 + output_pdf, + output_png, + raster_device='png16m', + raster_dpi=Resolution(100, 100), + pageno=1, ) # The output image should contain pure white and black @@ -117,7 +126,7 @@ def test_exotic_image( def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xyres[0] != in_pageinfo[0].xyres[1] + assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y check_ocrmypdf( resources / 'aspect.pdf', @@ -130,7 +139,7 @@ def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpd out_pageinfo = PdfInfo(outpdf) # Confirm resolution was kept the same - assert in_pageinfo[0].xyres == out_pageinfo[0].xyres + assert in_pageinfo[0].dpi == out_pageinfo[0].dpi @pytest.mark.parametrize('renderer', RENDERERS) @@ -139,7 +148,7 @@ def test_convert_to_square_resolution( ): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') - assert in_pageinfo[0].xyres[0] != in_pageinfo[0].xyres[1] + assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y # --force-ocr requires means forced conversion to square resolution check_ocrmypdf( @@ -156,7 +165,7 @@ def test_convert_to_square_resolution( in_p0, out_p0 = in_pageinfo[0], out_pageinfo[0] # Resolution show now be equal - assert out_p0.xyres[0] == out_p0.xyres[1] + assert out_p0.dpi.x == out_p0.dpi.y # Page size should match input page size assert isclose(in_p0.width_inches, out_p0.width_inches) @@ -164,7 +173,7 @@ def test_convert_to_square_resolution( # Because we rasterized the page to produce a new image, it should occupy # the entire page - out_im_w = out_p0.images[0].width / out_p0.images[0].xyres[0] - out_im_h = out_p0.images[0].height / out_p0.images[0].xyres[1] + out_im_w = out_p0.images[0].width / out_p0.images[0].dpi.x + out_im_h = out_p0.images[0].height / out_p0.images[0].dpi.y assert isclose(out_p0.width_inches, out_im_w) assert isclose(out_p0.height_inches, out_im_h) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 101ee8e7..4ffa59f8 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -27,6 +27,7 @@ from PIL import Image from ocrmypdf import leptonica from ocrmypdf.exec import ghostscript, tesseract +from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import PdfInfo # pytest.helpers is dynamic @@ -59,7 +60,7 @@ def check_monochrome_correlation( pdf, png, raster_device='pngmono', - xyres=(100, 100), + raster_dpi=Resolution(100, 100), pageno=pageno, rotation=0, ) From 0a5108e704aa760d3dbc8746d9c709ee3514f402 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 9 Apr 2020 04:09:44 -0700 Subject: [PATCH 407/880] install: clarify that old ocrmypdf should be removed from Ubuntu 18.04 Closes #526 --- docs/installation.rst | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 2015c839..97eb4df5 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -137,11 +137,12 @@ Installing the latest version on Ubuntu 18.04 LTS ------------------------------------------------- Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but -it is quite old now. To install a more recent version, first install several -system dependencies: +it is quite old now. To install a more recent version, uninstall the old version +of ocrmypdf, and install the following dependencies: .. code-block:: bash + sudo apt-get -y remove ocrmypdf sudo apt-get -y update sudo apt-get -y install \ ghostscript \ From d96867e6ab9a53e72a01c7926aaba1e0aebab8bb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Apr 2020 02:50:39 -0700 Subject: [PATCH 408/880] watcher: add polling and log level adjustment --- misc/watcher.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index 168c9998..c1a3d963 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -25,6 +25,7 @@ from pathlib import Path import pikepdf from watchdog.events import PatternMatchingEventHandler from watchdog.observers import Observer +from watchdog.observers.polling import PollingObserver import ocrmypdf @@ -37,7 +38,8 @@ ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) DESKEW = bool(os.getenv('OCR_DESKEW', False)) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) -LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() +USE_POLLING = bool(os.getenv('OCR_USE_POLLING', False)) +LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] log = logging.getLogger('ocrmypdf-watcher') @@ -112,6 +114,7 @@ def main(): ocrmypdf.configure_logging( verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True ) + log.setLevel(LOGLEVEL) log.info( f"Starting OCRmyPDF watcher with config:\n" f"Input Directory: {INPUT_DIRECTORY}\n" @@ -126,6 +129,7 @@ def main(): f"DESKEW: {DESKEW}\n" f"ARGS: {OCR_JSON_SETTINGS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" + f"USE_POLLING: {USE_POLLING}\n" f"LOGLEVEL: {LOGLEVEL}\n" ) @@ -134,7 +138,10 @@ def main(): sys.exit(1) handler = HandleObserverEvent(patterns=PATTERNS) - observer = Observer() + if USE_POLLING: + observer = PollingObserver() + else: + observer = Observer() observer.schedule(handler, INPUT_DIRECTORY, recursive=True) observer.start() try: From b4c65c57816b9c9434a2e7f2eaf7a3855e46b6fe Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 25 Apr 2020 03:49:34 -0700 Subject: [PATCH 409/880] Update requirements --- requirements/dev.txt | 2 -- requirements/main.txt | 8 ++++---- 2 files changed, 4 insertions(+), 6 deletions(-) delete mode 100644 requirements/dev.txt diff --git a/requirements/dev.txt b/requirements/dev.txt deleted file mode 100644 index ab2e5ff6..00000000 --- a/requirements/dev.txt +++ /dev/null @@ -1,2 +0,0 @@ -twine >= 1.8.1 -coverage >= 4.5 diff --git a/requirements/main.txt b/requirements/main.txt index 8934d40f..a9bc5ee8 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -2,9 +2,9 @@ # setup.py lists a separate set of requirements that are looser to simplify # installation cffi == 1.14.0 -img2pdf == 0.3.3 +img2pdf == 0.3.4 pdfminer.six == 20200402 -pikepdf == 1.10.2 -Pillow == 7.0.0 +pikepdf == 1.11.1 +Pillow == 7.1.1 reportlab == 3.5.34 -tqdm == 4.42.1 +tqdm == 4.45.0 From 43d650e78c47ea67a42442ad54d459ba1da8c823 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 25 Apr 2020 03:50:11 -0700 Subject: [PATCH 410/880] Fix issue where only first PNG-style image would be optimized --- src/ocrmypdf/optimize.py | 4 ++-- tests/test_optimize.py | 42 +++++++++++++++++++++++++++++++++++++++- 2 files changed, 43 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 839a8ff9..99211698 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -421,9 +421,9 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options): ) continue if compdata.type == leptonica.lept.L_FLATE_ENCODE: - return rewrite_png(pike, im_obj, compdata, log) + rewrite_png(pike, im_obj, compdata, log) elif compdata.type == leptonica.lept.L_G4_ENCODE: - return rewrite_png_as_g4(pike, im_obj, compdata, log) + rewrite_png_as_g4(pike, im_obj, compdata, log) def rewrite_png_as_g4(pike, im_obj, compdata, log): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 0c78d653..b9368849 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -18,10 +18,12 @@ import logging from os import fspath from pathlib import Path +from unittest.mock import patch +import img2pdf import pikepdf import pytest -from PIL import Image +from PIL import Image, ImageDraw from ocrmypdf import optimize as opt from ocrmypdf.exec import jbig2enc, pngquant @@ -130,3 +132,41 @@ def test_flate_to_jbig2(resources, outdir, spoof_tesseract_noop): pdf = pikepdf.open(outdir / 'out.pdf') pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values()))) assert pim.filters[0] == '/JBIG2Decode' + + +def test_multiple_pngs(resources, outdir, spoof_tesseract_noop): + with Path.open(outdir / 'in.pdf', 'wb') as inpdf: + img2pdf.convert( + fspath(resources / 'baiona_colormapped.png'), + fspath(resources / 'baiona_gray.png'), + with_pdfrw=False, + outputstream=inpdf, + ) + + def mockquant(input_file, output_file, _quality_min, _quality_max): + with Image.open(input_file) as im: + draw = ImageDraw.Draw(im) + draw.rectangle((0, 0, im.width, im.height), fill=128) + im.save(output_file) + + with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant): + check_ocrmypdf( + outdir / 'in.pdf', + outdir / 'out.pdf', + '--optimize', + '3', + '--jobs', + '1', + '--use-threads', + '--output-type', + 'pdf', + env=spoof_tesseract_noop, + ) + + with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open( + outdir / 'out.pdf' + ) as outpdf: + for n in range(len(inpdf.pages)): + inim = next(iter(inpdf.pages[n].images.values())) + outim = next(iter(outpdf.pages[n].images.values())) + assert len(outim.read_raw_bytes()) < len(inim.read_raw_bytes()), n From 33e982b3fdebb6163dc3b09f4e7b2f246b765962 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 00:37:14 -0700 Subject: [PATCH 411/880] azure: add certifi, openssl for macOS --- azure-pipelines.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 6d92fb39..ac6226ac 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -168,12 +168,13 @@ stages: jbig2enc \ leptonica \ openjpeg \ + openssl \ pngquant \ tesseract \ unpaper displayName: "Install system packages" - bash: | - pip3 install --upgrade pip + pip3 install --upgrade pip certifi pip3 install -r requirements/main.txt -r requirements/test.txt . displayName: "Install Python packages" - bash: | From 3834d1a0bf39bb900b9b933051fd7117e09c9a5e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 00:58:38 -0700 Subject: [PATCH 412/880] azure: use brew python instead --- azure-pipelines.yml | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index ac6226ac..5f88b76b 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -153,12 +153,13 @@ stages: matrix: Python37: python.version: "3.7" - Python38: - python.version: "3.8" + # Python38: + # python.version: "3.8" steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" + # https://github.com/actions/virtual-environments/issues/664 + # - task: UsePythonVersion@0 + # inputs: + # versionSpec: "$(python.version)" - bash: | brew update brew unlink python@2 @@ -168,13 +169,13 @@ stages: jbig2enc \ leptonica \ openjpeg \ - openssl \ pngquant \ + python \ tesseract \ unpaper displayName: "Install system packages" - bash: | - pip3 install --upgrade pip certifi + pip3 install --upgrade pip pip3 install -r requirements/main.txt -r requirements/test.txt . displayName: "Install Python packages" - bash: | From d0d0a98dca15e188d8967e725846028158be8898 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 5 Apr 2020 04:22:51 -0700 Subject: [PATCH 413/880] First cut at concurrent page scan Improvement appears on 168 page file. Needs refactoring --- src/ocrmypdf/pdfinfo/info.py | 87 +++++++++++++++++++++++++++++++----- 1 file changed, 76 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 565ed745..656e94a5 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -17,11 +17,13 @@ # along with OCRmyPDF. If not, see . import logging +import os import re from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum from math import hypot, isclose +from multiprocessing import Pool from os import PathLike, fspath from pathlib import Path from warnings import warn @@ -612,6 +614,77 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): return pageinfo +worker_pdf = None + + +def _pdf_pageinfo_sync(args): + global worker_pdf + + infile, pageno, xmltext, detailed_analysis = args + page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) + return page + + +def _pdf_pageinfo_sync_init(infile): + global worker_pdf + worker_pdf = pikepdf.open(infile) + + +def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar): + pages = [None] * len(pdf.pages) + with tqdm( + total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar + ) as pbar: + pool = Pool( + processes=4, # max_workers, + initializer=_pdf_pageinfo_sync_init, + initargs=(infile,), + ) + contexts = ( + (infile, n, pages_xml[n] if pages_xml else None, detailed_analysis) + for n in range(len(pdf.pages)) + ) + try: + results = pool.imap_unordered(_pdf_pageinfo_sync, contexts, chunksize=1) + while True: + try: + # page = results.next() + page = next(results) + pages[page.pageno] = page + pbar.update() + except StopIteration: + break + except KeyboardInterrupt: + pool.terminate() + raise + except Exception: + if not os.environ.get("PYTEST_CURRENT_TEST", ""): + # Unless inside pytest, exit immediately because no one wants + # to wait for child processes to finalize results that will be + # thrown away. Inside pytest, we want child processes to exit + # cleanly so that they output an error messages or coverage data + # we need from them. + pool.terminate() + raise + finally: + # Terminate log listener + # log_queue.put_nowait(None) + pool.close() + pool.join() + + # for n, _ in tqdm( + # enumerate(pdf.pages), + # total=len(pdf.pages), + # desc="Scan", + # unit='page', + # disable=not progbar, + # ): + # page_xml = pages_xml[n] if pages_xml else None + # page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) + # pages.append(page) + return pages + + def _pdf_get_all_pageinfo(infile, detailed_analysis=False, progbar=False): pdf = pikepdf.open(infile) # Do not close in this function try: @@ -622,17 +695,9 @@ def _pdf_get_all_pageinfo(infile, detailed_analysis=False, progbar=False): else: pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None) - pages = [] - for n, _ in tqdm( - enumerate(pdf.pages), - total=len(pdf.pages), - desc="Scan", - unit='page', - disable=not progbar, - ): - page_xml = pages_xml[n] if pages_xml else None - page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) - pages.append(page) + pages = _pdf_pageinfo_concurrent( + pdf, infile, pages_xml, detailed_analysis, progbar + ) except Exception: pdf.close() raise From ce49fc26dd2a8bd5cf19c82fa446d850e51bae20 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 10 Apr 2020 12:40:30 -0700 Subject: [PATCH 414/880] Do pikepdf.open() once instead of per worker --- src/ocrmypdf/pdfinfo/info.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 656e94a5..7df7f48e 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -620,14 +620,13 @@ worker_pdf = None def _pdf_pageinfo_sync(args): global worker_pdf - infile, pageno, xmltext, detailed_analysis = args + pageno, infile, xmltext, detailed_analysis = args page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) return page -def _pdf_pageinfo_sync_init(infile): - global worker_pdf - worker_pdf = pikepdf.open(infile) +def _pdf_pageinfo_sync_init(): + pass def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar): @@ -635,13 +634,15 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) with tqdm( total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar ) as pbar: + global worker_pdf + worker_pdf = pdf pool = Pool( processes=4, # max_workers, initializer=_pdf_pageinfo_sync_init, - initargs=(infile,), + initargs=tuple(), ) contexts = ( - (infile, n, pages_xml[n] if pages_xml else None, detailed_analysis) + (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) for n in range(len(pdf.pages)) ) try: From db3e75e33ef88471ce11acfbe801e633bf0fabce Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 10 Apr 2020 23:57:09 -0700 Subject: [PATCH 415/880] Refactor multiprocessing pool --- src/ocrmypdf/_concurrent.py | 110 +++++++++++++++++++++++++++++++++++ src/ocrmypdf/_sync.py | 72 +++++++---------------- src/ocrmypdf/pdfinfo/info.py | 2 +- 3 files changed, 133 insertions(+), 51 deletions(-) create mode 100644 src/ocrmypdf/_concurrent.py diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py new file mode 100644 index 00000000..d6e49e91 --- /dev/null +++ b/src/ocrmypdf/_concurrent.py @@ -0,0 +1,110 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +import logging.handlers +import multiprocessing +import os +import signal +import sys +import threading +from multiprocessing import Pool as ProcessPool +from multiprocessing.dummy import Pool as ThreadPool +from pathlib import Path + +from tqdm import tqdm + + +def log_listener(queue): + """Listen to the worker processes and forward the messages to logging + + For simplicity this is a thread rather than a process. Only one process + should actually write to sys.stderr or whatever we're using, so if this is + made into a process the main application needs to be directed to it. + + See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes + """ + + while True: + try: + record = queue.get() + if record is None: + break + logger = logging.getLogger(record.name) + logger.handle(record) + except Exception: + import traceback + + print("Logging problem", file=sys.stderr) + traceback.print_exc(file=sys.stderr) + + +def exec_progress_pool( + *, + use_threads, + max_workers, + tqdm_kwargs, + task_initializer=None, + task_initargs=None, + task=None, + task_arguments=None, + task_finished=None, +): + log_queue = multiprocessing.Queue(-1) + listener = threading.Thread(target=log_listener, args=(log_queue,)) + + if use_threads: + pool_class = ThreadPool + else: + pool_class = ProcessPool + listener.start() + + with tqdm(**tqdm_kwargs) as pbar: + pool = pool_class( + processes=max_workers, + initializer=task_initializer, + initargs=(log_queue, *task_initargs), + ) + try: + results = pool.imap_unordered(task, task_arguments) + while True: + try: + result = results.next() + task_finished(result, pbar) + except StopIteration: + break + except KeyboardInterrupt: + # Terminate pool so we exit instantly + pool.terminate() + # Don't try listener.join() here, will deadlock + raise + except Exception: + if not os.environ.get("PYTEST_CURRENT_TEST", ""): + # Unless inside pytest, exit immediately because no one wants + # to wait for child processes to finalize results that will be + # thrown away. Inside pytest, we want child processes to exit + # cleanly so that they output an error messages or coverage data + # we need from them. + pool.terminate() + raise + finally: + # Terminate log listener + log_queue.put_nowait(None) + pool.close() + pool.join() + + listener.join() diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 383e8dc7..7996f869 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -27,8 +27,8 @@ from pathlib import Path from tempfile import mkdtemp import PIL -from tqdm import tqdm +from ._concurrent import exec_progress_pool from ._graft import OcrGrafter from ._jobcontext import PDFContext, cleanup_working_files from ._logging import PageNumberFilter @@ -274,63 +274,35 @@ def exec_concurrent(context): log.info("Using Tesseract OpenMP thread limit %d", tess_threads) if context.options.use_threads: - from multiprocessing.dummy import Pool - initializer = worker_thread_init else: - Pool = multiprocessing.Pool initializer = worker_init sidecars = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) - log_queue = multiprocessing.Queue(-1) - listener = threading.Thread(target=log_listener, args=(log_queue,)) - listener.start() - with tqdm( - total=(2 * len(context.pdfinfo)), - desc='OCR', - unit='page', - unit_scale=0.5, - disable=not context.options.progress_bar, - ) as pbar: - pool = Pool( - processes=max_workers, - initializer=initializer, - initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS), - ) - try: - results = pool.imap_unordered(exec_page_sync, context.get_page_contexts()) - while True: - try: - page_result = results.next() - sidecars[page_result.pageno] = page_result.text - pbar.update() - ocrgraft.graft_page(page_result) - pbar.update() - except StopIteration: - break - except KeyboardInterrupt: - # Terminate pool so we exit instantly - pool.terminate() - # Don't try listener.join() here, will deadlock - raise - except Exception: - if not os.environ.get("PYTEST_CURRENT_TEST", ""): - # Unless inside pytest, exit immediately because no one wants - # to wait for child processes to finalize results that will be - # thrown away. Inside pytest, we want child processes to exit - # cleanly so that they output an error messages or coverage data - # we need from them. - pool.terminate() - raise - finally: - # Terminate log listener - log_queue.put_nowait(None) - pool.close() - pool.join() + def update_page(result, pbar): + sidecars[result.pageno] = result.text + pbar.update() + ocrgraft.graft_page(result) + pbar.update() - listener.join() + exec_progress_pool( + use_threads=context.options.use_threads, + max_workers=max_workers, + tqdm_kwargs=dict( + total=(2 * len(context.pdfinfo)), + desc='OCR', + unit='page', + unit_scale=0.5, + disable=not context.options.progress_bar, + ), + task_initializer=initializer, + task_initargs=(PIL.Image.MAX_IMAGE_PIXELS,), + task=exec_page_sync, + task_arguments=context.get_page_contexts(), + task_finished=update_page, + ) # Output sidecar text if context.options.sidecar: diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 7df7f48e..b228e769 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -637,7 +637,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) global worker_pdf worker_pdf = pdf pool = Pool( - processes=4, # max_workers, + processes=1, # max_workers, initializer=_pdf_pageinfo_sync_init, initargs=tuple(), ) From af3c3c646606993d3fa6f16486844f0ae2a544e0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 11 Apr 2020 00:49:36 -0700 Subject: [PATCH 416/880] Further refactoring of concurrency concerns --- src/ocrmypdf/_concurrent.py | 30 ++++++++++- src/ocrmypdf/_sync.py | 53 ++------------------ src/ocrmypdf/pdfinfo/info.py | 96 ++++++++++++++---------------------- 3 files changed, 69 insertions(+), 110 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index d6e49e91..724e6484 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -53,6 +53,27 @@ def log_listener(queue): traceback.print_exc(file=sys.stderr) +def process_init(queue, userfn, *userargs): + """Initialize a process pool worker""" + + # Ignore SIGINT (our parent process will kill us gracefully) + signal.signal(signal.SIGINT, signal.SIG_IGN) + + # Reconfigure the root logger for this process to send all messages to a queue + h = logging.handlers.QueueHandler(queue) + root = logging.getLogger() + root.handlers = [] + root.addHandler(h) + + if userfn: + userfn(*userargs) + + +def thread_init(_queue, userfn, *userargs): + if userfn: + userfn(*userargs) + + def exec_progress_pool( *, use_threads, @@ -67,17 +88,22 @@ def exec_progress_pool( log_queue = multiprocessing.Queue(-1) listener = threading.Thread(target=log_listener, args=(log_queue,)) + if not task_initargs: + task_initargs = tuple() + if use_threads: pool_class = ThreadPool + initializer = thread_init else: pool_class = ProcessPool + initializer = process_init listener.start() with tqdm(**tqdm_kwargs) as pbar: pool = pool_class( processes=max_workers, - initializer=task_initializer, - initargs=(log_queue, *task_initargs), + initializer=initializer, + initargs=(log_queue, task_initializer, *task_initargs), ) try: results = pool.imap_unordered(task, task_arguments) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 7996f869..998b67ab 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -199,53 +199,13 @@ def post_process(pdf_file, context): return optimize_pdf(pdf_out, context) -def worker_init(queue, max_pixels): - """Initialize a process pool worker""" - - # Ignore SIGINT (our parent process will kill us gracefully) - signal.signal(signal.SIGINT, signal.SIG_IGN) - - # Reconfigure the root logger for this process to send all messages to a queue - h = logging.handlers.QueueHandler(queue) - root = logging.getLogger() - root.handlers = [] - root.addHandler(h) - +def worker_init(max_pixels): # In Windows, child process will not inherit our change to this value in - # the parent process, so ensure workers get it set + # the parent process, so ensure workers get it set. Not needed when running + # threaded, but harmless to set again. PIL.Image.MAX_IMAGE_PIXELS = max_pixels -def worker_thread_init(_queue, max_pixels): - # This is probably not needed since threads should all see the same memory, - # but done for consistency. - PIL.Image.MAX_IMAGE_PIXELS = max_pixels - - -def log_listener(queue): - """Listen to the worker processes and forward the messages to logging - - For simplicity this is a thread rather than a process. Only one process - should actually write to sys.stderr or whatever we're using, so if this is - made into a process the main application needs to be directed to it. - - See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes - """ - - while True: - try: - record = queue.get() - if record is None: - break - logger = logging.getLogger(record.name) - logger.handle(record) - except Exception: - import traceback - - print("Logging problem", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - - def exec_concurrent(context): """Execute the pipeline concurrently""" @@ -273,11 +233,6 @@ def exec_concurrent(context): if tess_threads > 1: log.info("Using Tesseract OpenMP thread limit %d", tess_threads) - if context.options.use_threads: - initializer = worker_thread_init - else: - initializer = worker_init - sidecars = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) @@ -297,7 +252,7 @@ def exec_concurrent(context): unit_scale=0.5, disable=not context.options.progress_bar, ), - task_initializer=initializer, + task_initializer=worker_init, task_initargs=(PIL.Image.MAX_IMAGE_PIXELS,), task=exec_page_sync, task_arguments=context.get_page_contexts(), diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index b228e769..97ddb1d8 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -23,15 +23,14 @@ from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum from math import hypot, isclose -from multiprocessing import Pool from os import PathLike, fspath from pathlib import Path from warnings import warn import pikepdf from pikepdf import PdfMatrix -from tqdm import tqdm +from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf.exceptions import EncryptedPdfError from ocrmypdf.exec import ghostscript from ocrmypdf.helpers import Resolution @@ -618,71 +617,50 @@ worker_pdf = None def _pdf_pageinfo_sync(args): - global worker_pdf - pageno, infile, xmltext, detailed_analysis = args page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) return page -def _pdf_pageinfo_sync_init(): - pass - - def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar): pages = [None] * len(pdf.pages) - with tqdm( - total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar - ) as pbar: - global worker_pdf - worker_pdf = pdf - pool = Pool( - processes=1, # max_workers, - initializer=_pdf_pageinfo_sync_init, - initargs=tuple(), - ) - contexts = ( - (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) - for n in range(len(pdf.pages)) - ) - try: - results = pool.imap_unordered(_pdf_pageinfo_sync, contexts, chunksize=1) - while True: - try: - # page = results.next() - page = next(results) - pages[page.pageno] = page - pbar.update() - except StopIteration: - break - except KeyboardInterrupt: - pool.terminate() - raise - except Exception: - if not os.environ.get("PYTEST_CURRENT_TEST", ""): - # Unless inside pytest, exit immediately because no one wants - # to wait for child processes to finalize results that will be - # thrown away. Inside pytest, we want child processes to exit - # cleanly so that they output an error messages or coverage data - # we need from them. - pool.terminate() - raise - finally: - # Terminate log listener - # log_queue.put_nowait(None) - pool.close() - pool.join() - # for n, _ in tqdm( - # enumerate(pdf.pages), - # total=len(pdf.pages), - # desc="Scan", - # unit='page', - # disable=not progbar, - # ): - # page_xml = pages_xml[n] if pages_xml else None - # page = PageInfo(pdf, n, infile, page_xml, detailed_analysis) - # pages.append(page) + def update_pageinfo(result, pbar): + page = result + pages[page.pageno] = page + pbar.update() + + contexts = ( + (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) + for n in range(len(pdf.pages)) + ) + global worker_pdf + worker_pdf = pdf + + if os.name == 'nt': + # We can't parallelize on Windows, because Windows cannot fork. + # We are trying to fork, then take advantage of the preloaded pikepdf.Pdf + # object in memory to save time reloading it, hence the silly global + # variable. Hey, it works. Threads are not helpful here because they + # will all just fight over the lock. So on Windows just run sequentially. + use_threads = True + max_workers = 1 + else: + use_threads = False + max_workers = min(len(pages), 16) + + exec_progress_pool( + use_threads=use_threads, + max_workers=1, + tqdm_kwargs=dict( + total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar + ), + task_initializer=None, + task_initargs=None, + task=_pdf_pageinfo_sync, + task_arguments=contexts, + task_finished=update_pageinfo, + ) return pages From 7513f5425c7fbd7c8ef9bcdca7016d5d1cdcd055 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 11 Apr 2020 01:21:07 -0700 Subject: [PATCH 417/880] Fix some broken tests --- src/ocrmypdf/_concurrent.py | 18 +++++++++--------- src/ocrmypdf/pdfinfo/info.py | 2 +- tests/test_validation.py | 2 +- 3 files changed, 11 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 724e6484..7253ab89 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -24,7 +24,7 @@ import sys import threading from multiprocessing import Pool as ProcessPool from multiprocessing.dummy import Pool as ThreadPool -from pathlib import Path +from typing import Callable, Iterable, Optional from tqdm import tqdm @@ -76,14 +76,14 @@ def thread_init(_queue, userfn, *userargs): def exec_progress_pool( *, - use_threads, - max_workers, - tqdm_kwargs, - task_initializer=None, - task_initargs=None, - task=None, - task_arguments=None, - task_finished=None, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + task_initializer: Optional[Callable] = None, + task_initargs: Optional[tuple] = None, + task: Optional[Callable] = None, + task_arguments: Optional[Iterable] = None, + task_finished: Optional[Callable] = None, ): log_queue = multiprocessing.Queue(-1) listener = threading.Thread(target=log_listener, args=(log_queue,)) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 97ddb1d8..c5e0f9db 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -651,7 +651,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) exec_progress_pool( use_threads=use_threads, - max_workers=1, + max_workers=max_workers, tqdm_kwargs=dict( total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar ), diff --git a/tests/test_validation.py b/tests/test_validation.py index af1eadea..864dd05a 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -150,7 +150,7 @@ def test_false_action_store_true(): @pytest.mark.parametrize('progress_bar', [True, False]) def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) - with patch('ocrmypdf.pdfinfo.info.tqdm', autospec=True) as tqdmpatch: + with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: vd.check_options(opts) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) assert pdfinfo is not None From 86145a8c76c714f8d690f53e709b2024cf6526b5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Apr 2020 22:58:59 -0700 Subject: [PATCH 418/880] Some wrong with forking worker_pdf, just open it once per page for now --- src/ocrmypdf/pdfinfo/info.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index c5e0f9db..b4da45c1 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -613,12 +613,13 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): return pageinfo -worker_pdf = None +# worker_pdf = None def _pdf_pageinfo_sync(args): pageno, infile, xmltext, detailed_analysis = args - page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) + with pikepdf.open(infile) as worker_pdf: + page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) return page @@ -634,8 +635,8 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) for n in range(len(pdf.pages)) ) - global worker_pdf - worker_pdf = pdf + # global worker_pdf + # worker_pdf = pdf if os.name == 'nt': # We can't parallelize on Windows, because Windows cannot fork. From 8c381a022729e41c04ef38e74230e350d538e127 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Apr 2020 23:02:48 -0700 Subject: [PATCH 419/880] Replace task_initargs with use of partial() --- src/ocrmypdf/_concurrent.py | 18 +++++++----------- src/ocrmypdf/_sync.py | 4 ++-- src/ocrmypdf/pdfinfo/info.py | 1 - 3 files changed, 9 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 7253ab89..a321f28d 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -53,7 +53,7 @@ def log_listener(queue): traceback.print_exc(file=sys.stderr) -def process_init(queue, userfn, *userargs): +def process_init(queue, user_init): """Initialize a process pool worker""" # Ignore SIGINT (our parent process will kill us gracefully) @@ -65,13 +65,13 @@ def process_init(queue, userfn, *userargs): root.handlers = [] root.addHandler(h) - if userfn: - userfn(*userargs) + if user_init: + user_init() -def thread_init(_queue, userfn, *userargs): - if userfn: - userfn(*userargs) +def thread_init(_queue, user_init): + if user_init: + user_init() def exec_progress_pool( @@ -80,7 +80,6 @@ def exec_progress_pool( max_workers: int, tqdm_kwargs: dict, task_initializer: Optional[Callable] = None, - task_initargs: Optional[tuple] = None, task: Optional[Callable] = None, task_arguments: Optional[Iterable] = None, task_finished: Optional[Callable] = None, @@ -88,9 +87,6 @@ def exec_progress_pool( log_queue = multiprocessing.Queue(-1) listener = threading.Thread(target=log_listener, args=(log_queue,)) - if not task_initargs: - task_initargs = tuple() - if use_threads: pool_class = ThreadPool initializer = thread_init @@ -103,7 +99,7 @@ def exec_progress_pool( pool = pool_class( processes=max_workers, initializer=initializer, - initargs=(log_queue, task_initializer, *task_initargs), + initargs=(log_queue, task_initializer), ) try: results = pool.imap_unordered(task, task_arguments) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 998b67ab..f54c32a6 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -23,6 +23,7 @@ import signal import sys import threading from collections import namedtuple +from functools import partial from pathlib import Path from tempfile import mkdtemp @@ -252,8 +253,7 @@ def exec_concurrent(context): unit_scale=0.5, disable=not context.options.progress_bar, ), - task_initializer=worker_init, - task_initargs=(PIL.Image.MAX_IMAGE_PIXELS,), + task_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS), task=exec_page_sync, task_arguments=context.get_page_contexts(), task_finished=update_page, diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index b4da45c1..26a5ba4e 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -657,7 +657,6 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar ), task_initializer=None, - task_initargs=None, task=_pdf_pageinfo_sync, task_arguments=contexts, task_finished=update_pageinfo, From 27a3b80376533ea39c3c17e8152c89d03caaba27 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 03:25:38 -0700 Subject: [PATCH 420/880] Use once-per-worker pikepdf init --- setup.py | 2 +- src/ocrmypdf/pdfinfo/info.py | 16 ++++++++++------ 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/setup.py b/setup.py index eb0d783a..da695597 100644 --- a/setup.py +++ b/setup.py @@ -98,7 +98,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, <= 20200124', + 'pdfminer.six >= 20191110, <= 20200402', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 26a5ba4e..bb232ee5 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -22,6 +22,7 @@ import re from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum +from functools import partial from math import hypot, isclose from os import PathLike, fspath from pathlib import Path @@ -613,13 +614,18 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): return pageinfo -# worker_pdf = None +worker_pdf = None + + +def _pdf_pageinfo_sync_init(infile): + global worker_pdf # pylint: disable=global-statement + worker_pdf = pikepdf.open(infile) def _pdf_pageinfo_sync(args): + global worker_pdf # pylint: disable=global-statement pageno, infile, xmltext, detailed_analysis = args - with pikepdf.open(infile) as worker_pdf: - page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) + page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) return page @@ -635,8 +641,6 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) for n in range(len(pdf.pages)) ) - # global worker_pdf - # worker_pdf = pdf if os.name == 'nt': # We can't parallelize on Windows, because Windows cannot fork. @@ -656,7 +660,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) tqdm_kwargs=dict( total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar ), - task_initializer=None, + task_initializer=partial(_pdf_pageinfo_sync_init, infile), task=_pdf_pageinfo_sync, task_arguments=contexts, task_finished=update_pageinfo, From 2c07515907da1b071c4a5e64e66249d9ff8ba7aa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 03:33:31 -0700 Subject: [PATCH 421/880] macOS - use spawn for multiprocessing See bpo-33725. This is the default for 3.8, opt-in for 3.7 and older. --- src/ocrmypdf/__main__.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 459b3040..5d3a5d50 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -19,6 +19,7 @@ import logging import os import sys +from multiprocessing import set_start_method from . import __version__ from ._sync import run_pipeline @@ -66,4 +67,6 @@ def run(args=None): if __name__ == '__main__': + if sys.platform == 'darwin' and sys.version_info < (3, 8): + set_start_method('spawn') # see python bpo-33725 sys.exit(run()) From 991db17fdeb212f524e3be499c542b904b3e97af Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 11 Apr 2020 16:03:00 -0700 Subject: [PATCH 422/880] Remove Ghostscript-based text extraction While faster than Python based methods, we've outgrown the limited amount of information Ghostscript provides with this feature, and it repeats an analysis we have to do anyway to learn what images are present. --- src/ocrmypdf/_pipeline.py | 6 +- src/ocrmypdf/_sync.py | 6 +- src/ocrmypdf/exec/ghostscript.py | 49 --------- src/ocrmypdf/pdfinfo/ghosttext.py | 102 ------------------ src/ocrmypdf/pdfinfo/info.py | 58 +++------- tests/cache/manifest.jsonl | 1 + .../hocr.bin | 30 ++++++ .../stderr.bin | 1 + .../stdout.bin | 0 .../txt.bin | 3 + tests/test_main.py | 4 +- tests/test_pdfinfo.py | 25 +---- 12 files changed, 57 insertions(+), 228 deletions(-) delete mode 100644 src/ocrmypdf/pdfinfo/ghosttext.py create mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin create mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin create mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin create mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index bb2f7e18..3be65faa 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -147,11 +147,9 @@ def triage(original_filename, input_file, output_file, options): return output_file -def get_pdfinfo(input_file, detailed_page_analysis=False, progbar=False): +def get_pdfinfo(input_file, progbar=False): try: - return PdfInfo( - input_file, detailed_page_analysis=detailed_page_analysis, progbar=progbar - ) + return PdfInfo(input_file, progbar=progbar) except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index f54c32a6..9f5c65d8 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -326,11 +326,7 @@ def run_pipeline(options, api=False): ) # Gather pdfinfo and create context - pdfinfo = get_pdfinfo( - origin_pdf, - detailed_page_analysis=options.redo_ocr, - progbar=options.progress_bar, - ) + pdfinfo = get_pdfinfo(origin_pdf, progbar=options.progress_bar) context = PDFContext(options, work_folder, origin_pdf, pdfinfo) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 5c27488f..44b81682 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -83,55 +83,6 @@ def _gs_error_reported(stream) -> bool: return re.search(r'error', stream, flags=re.IGNORECASE) -def extract_text(input_file, pageno=1): - """Use the txtwrite device to get text layout information out - - For details on options of -dTextFormat see - https://www.ghostscript.com/doc/current/VectorDevices.htm#TXT - - Format is like - - - - - - :param pageno: number of page to extract, or all pages if None - :return: XML-ish text representation in bytes - """ - - if pageno is not None: - pages = ['-dFirstPage=%i' % pageno, '-dLastPage=%i' % pageno] - else: - pages = [] - - # Note due to bug https://bugs.ghostscript.com/show_bug.cgi?id=701971 - # Ghostscript <= 9.50 will truncate output unless we write to stdout, so - # don't write to a file. - args_gs = ( - [ - GS, - '-dQUIET', - '-dSAFER', - '-dBATCH', - '-dNOPAUSE', - '-sDEVICE=txtwrite', - '-dTextFormat=0', - ] - + pages - + ['-o', '-', fspath(input_file), "-sstdout=%stderr"] - ) - - try: - p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True) - except CalledProcessError as e: - raise SubprocessOutputError( - 'Ghostscript text extraction failed\n%s\n%s' - % (input_file, e.stderr.decode(errors='replace')) - ) - - return p.stdout - - def rasterize_pdf( input_file: os.PathLike, output_file: os.PathLike, diff --git a/src/ocrmypdf/pdfinfo/ghosttext.py b/src/ocrmypdf/pdfinfo/ghosttext.py deleted file mode 100644 index 07e72f19..00000000 --- a/src/ocrmypdf/pdfinfo/ghosttext.py +++ /dev/null @@ -1,102 +0,0 @@ -# © 2018 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -import logging -import re -import xml.etree.ElementTree as ET - -from ..exec import ghostscript - -log = logging.getLogger(__name__) - -# Forgive me for I have sinned -# I am using regular expressions to parse XML. However the XML in this case, -# generated by Ghostscript, is self-consistent enough to be parseable. -regex_remove_char_tags = re.compile( - br""" - ] # anything single character but > - | \">\" # special case: trap ">" - )* - /> # terminate with '/>' -""", - re.VERBOSE, -) - - -def page_get_textblocks(infile, pageno, xmltext, height): - """Get text boxes out of Ghostscript txtwrite xml""" - - root = xmltext - if not hasattr(xmltext, 'findall'): - return [] - - def blocks(): - for span in root.findall('.//span'): - bbox_str = span.attrib['bbox'] - font_size = span.attrib['size'] - pts = [int(pt) for pt in bbox_str.split()] - pts[1] = pts[1] - int(float(font_size) + 0.5) - bbox_topdown = tuple(pts) - bb = bbox_topdown - bbox_bottomup = (bb[0], height - bb[3], bb[2], height - bb[1]) - yield bbox_bottomup - - def joined_blocks(): - prev = None - for bbox in blocks(): - if prev is None: - prev = bbox - if bbox[1] == prev[1] and bbox[3] == prev[3]: - gap = prev[2] - bbox[0] - height = abs(bbox[3] - bbox[1]) - if gap < height: - # Join boxes - prev = (prev[0], prev[1], bbox[2], bbox[3]) - continue - # yield previously joined bboxes and start anew - yield prev - prev = bbox - if prev is not None: - yield prev - - return [block for block in joined_blocks()] - - -def extract_text_xml(infile, pdf, pageno=None): - existing_text = ghostscript.extract_text(infile, pageno=None) - existing_text = regex_remove_char_tags.sub(b' ', existing_text) - - try: - root = ET.fromstringlist([b'\n', existing_text, b'\n']) - page_xml = root.findall('page') - except ET.ParseError as e: - log.error( - "An error occurred while attempting to retrieve existing text in " - "the input file. Will attempt to continue assuming that there is " - "no existing text in the file. The error was:" - ) - log.error(e) - page_xml = [None] * len(pdf.pages) - - page_count_difference = len(pdf.pages) - len(page_xml) - if page_count_difference != 0: - log.error("The number of pages in the input file is inconsistent.") - log.error(f"Expected {len(pdf.pages)}, txtwrite says {len(page_xml)}") - if page_count_difference > 0: - page_xml.extend([None] * page_count_difference) - return page_xml diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index bb232ee5..bd7641c4 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -33,9 +33,7 @@ from pikepdf import PdfMatrix from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf.exceptions import EncryptedPdfError -from ocrmypdf.exec import ghostscript from ocrmypdf.helpers import Resolution -from ocrmypdf.pdfinfo import ghosttext from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes logger = logging.getLogger() @@ -557,7 +555,7 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) -def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): +def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike): pageinfo = {} pageinfo['pageno'] = pageno pageinfo['images'] = [] @@ -567,16 +565,10 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str): width_pt = mediabox[2] - mediabox[0] height_pt = mediabox[3] - mediabox[1] - if xmltext is not None: - bboxes = ghosttext.page_get_textblocks( - fspath(infile), pageno, xmltext=xmltext, height=height_pt - ) - pageinfo['bboxes'] = bboxes - else: - pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') - miner = get_page_analysis(infile, pageno, pscript5_mode) - pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) - bboxes = (box.bbox for box in pageinfo['textboxes']) + pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') + miner = get_page_analysis(infile, pageno, pscript5_mode) + pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) + bboxes = (box.bbox for box in pageinfo['textboxes']) pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) @@ -624,12 +616,12 @@ def _pdf_pageinfo_sync_init(infile): def _pdf_pageinfo_sync(args): global worker_pdf # pylint: disable=global-statement - pageno, infile, xmltext, detailed_analysis = args - page = PageInfo(worker_pdf, pageno, infile, xmltext, detailed_analysis) + pageno, infile = args + page = PageInfo(worker_pdf, pageno, infile) return page -def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar): +def _pdf_pageinfo_concurrent(pdf, infile, progbar): pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -637,11 +629,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) pages[page.pageno] = page pbar.update() - contexts = ( - (n, infile, pages_xml[n] if pages_xml else None, detailed_analysis) - for n in range(len(pdf.pages)) - ) - + contexts = ((n, infile) for n in range(len(pdf.pages))) if os.name == 'nt': # We can't parallelize on Windows, because Windows cannot fork. # We are trying to fork, then take advantage of the preloaded pikepdf.Pdf @@ -668,19 +656,12 @@ def _pdf_pageinfo_concurrent(pdf, infile, pages_xml, detailed_analysis, progbar) return pages -def _pdf_get_all_pageinfo(infile, detailed_analysis=False, progbar=False): +def _pdf_get_all_pageinfo(infile, progbar=False): pdf = pikepdf.open(infile) # Do not close in this function try: if pdf.is_encrypted: raise EncryptedPdfError() # Triggered by encryption with empty passwd - if detailed_analysis: - pages_xml = None - else: - pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None) - - pages = _pdf_pageinfo_concurrent( - pdf, infile, pages_xml, detailed_analysis, progbar - ) + pages = _pdf_pageinfo_concurrent(pdf, infile, progbar) except Exception: pdf.close() raise @@ -689,11 +670,10 @@ def _pdf_get_all_pageinfo(infile, detailed_analysis=False, progbar=False): class PageInfo: - def __init__(self, pdf, pageno, infile, xmltext, detailed_analysis=False): + def __init__(self, pdf, pageno, infile): self._pageno = pageno self._infile = infile - self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, xmltext) - self._detailed_analysis = detailed_analysis + self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile) @property def pageno(self): @@ -705,8 +685,6 @@ class PageInfo: @property def has_corrupt_text(self): - if not self._detailed_analysis: - raise NotImplementedError('Did not do detailed analysis') return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) @property @@ -757,7 +735,7 @@ class PageInfo: if 'textboxes' not in self._pageinfo: if visible is not None and corrupt is not None: - raise NotImplementedError('Ghostscript textboxes cannot be classified') + raise NotImplementedError('Incomplete information on textboxes') return self._pageinfo['bboxes'] return ( @@ -792,13 +770,9 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, detailed_page_analysis=False, progbar=False): + def __init__(self, infile, progbar=False): self._infile = infile - if ghostscript.version() in ('9.52',): - detailed_page_analysis = True # txtwrite doesn't work in these versions - self._pages, pdf = _pdf_get_all_pageinfo( - infile, detailed_page_analysis, progbar=progbar - ) + self._pages, pdf = _pdf_get_all_pageinfo(infile, progbar=progbar) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False if '/AcroForm' in pdf.root: diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 05a0e86c..23e50166 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -69,3 +69,4 @@ {"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.1 leptonica-1.79.0 libgif 5.2.1 : libjpeg 9d : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.1.0 : libopenjp2 2.3.1 Found AVX2 Found AVX Found FMA Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_hocr", "hocr", "txt"]} diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin new file mode 100644 index 00000000..27e769f6 --- /dev/null +++ b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin @@ -0,0 +1,30 @@ + + + + + + + + + + +

+ + diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin new file mode 100644 index 00000000..16b617e5 --- /dev/null +++ b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin new file mode 100644 index 00000000..21e1e995 --- /dev/null +++ b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin @@ -0,0 +1,3 @@ +YOOOxXYOO0O pixels at GOO DPI +oO] megapixels + \ No newline at end of file diff --git a/tests/test_main.py b/tests/test_main.py index d4146ccf..9b0cd17e 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -101,10 +101,10 @@ def test_skip_ocr(spoof_tesseract_cache, resources, outpdf): def test_redo_ocr(resources, outpdf): in_ = resources / 'graph_ocred.pdf' - before = PdfInfo(in_, detailed_page_analysis=True) + before = PdfInfo(in_) out = outpdf out = check_ocrmypdf(in_, out, '--redo-ocr') - after = PdfInfo(out, detailed_page_analysis=True) + after = PdfInfo(out) assert before[0].has_text and after[0].has_text assert ( before[0].get_textareas() != after[0].get_textareas() diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 13fb8a8b..cfa90d94 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -151,22 +151,6 @@ def test_pickle(resources): pickle.dumps(pdf) -def test_regex(): - rx = pdfinfo.ghosttext.regex_remove_char_tags - - must_match = [ - b'', - b'', - b'', - ] - must_not_match = [b'', b'', b'', b''] - - for s in must_match: - assert rx.match(s) - for s in must_not_match: - assert not rx.match(s) - - def test_vector(resources): filename = resources / 'vector.pdf' pdf = pdfinfo.PdfInfo(filename) @@ -184,16 +168,9 @@ def test_ocr_detection(resources): @pytest.mark.parametrize( 'testfile', ('truetype_font_nomapping.pdf', 'type3_font_nomapping.pdf') ) -@pytest.mark.xfail( - ghostscript.version() in ('9.52',), reason="gs 9.52 txtwrite doesn't work" -) def test_corrupt_font_detection(resources, testfile): filename = resources / testfile - with pytest.raises(NotImplementedError): - pdf = pdfinfo.PdfInfo(filename) - pdf[0].has_corrupt_text - - pdf = pdfinfo.PdfInfo(filename, detailed_page_analysis=True) + pdf = pdfinfo.PdfInfo(filename) assert pdf[0].has_corrupt_text From 18c4aa10bf524b864763879e9091f5da7787035f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 04:21:15 -0700 Subject: [PATCH 423/880] Adjust number of workers for concurrent page scanning --- src/ocrmypdf/_pipeline.py | 4 ++-- src/ocrmypdf/_sync.py | 6 +++++- src/ocrmypdf/pdfinfo/info.py | 35 ++++++++++++++++++----------------- 3 files changed, 25 insertions(+), 20 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 3be65faa..2128adf3 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -147,9 +147,9 @@ def triage(original_filename, input_file, output_file, options): return output_file -def get_pdfinfo(input_file, progbar=False): +def get_pdfinfo(input_file, progbar=False, max_workers=None): try: - return PdfInfo(input_file, progbar=progbar) + return PdfInfo(input_file, progbar=progbar, max_workers=max_workers) except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 9f5c65d8..e6e44271 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -326,7 +326,11 @@ def run_pipeline(options, api=False): ) # Gather pdfinfo and create context - pdfinfo = get_pdfinfo(origin_pdf, progbar=options.progress_bar) + pdfinfo = get_pdfinfo( + origin_pdf, + progbar=options.progress_bar, + max_workers=options.jobs if not options.use_threads else 1, # To help debug + ) context = PDFContext(options, work_folder, origin_pdf, pdfinfo) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index bd7641c4..86bb059d 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -33,7 +33,7 @@ from pikepdf import PdfMatrix from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf.exceptions import EncryptedPdfError -from ocrmypdf.helpers import Resolution +from ocrmypdf.helpers import Resolution, available_cpu_count from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes logger = logging.getLogger() @@ -621,7 +621,7 @@ def _pdf_pageinfo_sync(args): return page -def _pdf_pageinfo_concurrent(pdf, infile, progbar): +def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -629,22 +629,21 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar): pages[page.pageno] = page pbar.update() + if max_workers is None: + max_workers = available_cpu_count() + contexts = ((n, infile) for n in range(len(pdf.pages))) - if os.name == 'nt': - # We can't parallelize on Windows, because Windows cannot fork. - # We are trying to fork, then take advantage of the preloaded pikepdf.Pdf - # object in memory to save time reloading it, hence the silly global - # variable. Hey, it works. Threads are not helpful here because they - # will all just fight over the lock. So on Windows just run sequentially. + + use_threads = False # No performance gain if threaded due to GIL + n_workers = min(1 + len(pages) // 4, max_workers) + if n_workers == 1: + # But if we decided on only one worker, there is no point in using + # a separate process. use_threads = True - max_workers = 1 - else: - use_threads = False - max_workers = min(len(pages), 16) exec_progress_pool( use_threads=use_threads, - max_workers=max_workers, + max_workers=n_workers, tqdm_kwargs=dict( total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar ), @@ -656,12 +655,12 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar): return pages -def _pdf_get_all_pageinfo(infile, progbar=False): +def _pdf_get_all_pageinfo(infile, progbar=False, max_workers=None): pdf = pikepdf.open(infile) # Do not close in this function try: if pdf.is_encrypted: raise EncryptedPdfError() # Triggered by encryption with empty passwd - pages = _pdf_pageinfo_concurrent(pdf, infile, progbar) + pages = _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers) except Exception: pdf.close() raise @@ -770,9 +769,11 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, progbar=False): + def __init__(self, infile, progbar=False, max_workers=None): self._infile = infile - self._pages, pdf = _pdf_get_all_pageinfo(infile, progbar=progbar) + self._pages, pdf = _pdf_get_all_pageinfo( + infile, progbar=progbar, max_workers=max_workers + ) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False if '/AcroForm' in pdf.root: From 8b54ce338f1ba0880ae770d944700bd424984fb3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 05:09:42 -0700 Subject: [PATCH 424/880] setup: remove deprecated message about removeal of --force parameter --- setup.py | 16 ---------------- 1 file changed, 16 deletions(-) diff --git a/setup.py b/setup.py index da695597..45e0ae1e 100644 --- a/setup.py +++ b/setup.py @@ -27,22 +27,6 @@ if sys.version_info < (3, 6): print("Python 3.6 or newer is required", file=sys.stderr) sys.exit(1) - -# pylint: disable=w0613 - - -command = next((arg for arg in sys.argv[1:] if not arg.startswith('-')), '') -if command.startswith('install') or command in [ - 'check', - 'test', - 'nosetests', - 'easy_install', -]: - forced = '--force' in sys.argv - if forced: - print("The argument --force is deprecated. Please discontinue use.") - - if 'upload' in sys.argv[1:]: print('Use twine to upload the package - setup.py upload is insecure') sys.exit(1) From c84d0f606d5558b491d1ab7c00434c63f932aef0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 05:11:11 -0700 Subject: [PATCH 425/880] ghostscript: remove deprecated argument from generate_pdfa --- src/ocrmypdf/exec/ghostscript.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 44b81682..b58e30cb 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -169,7 +169,6 @@ def generate_pdfa( pdf_pages, output_file: os.PathLike, compression: str, - threads=None, # deprecated parameter pdf_version: str = '1.5', pdfa_part: str = '2', ): @@ -190,10 +189,6 @@ def generate_pdfa( images entirely. (The feature was added in 9.23 but broken, and the 9.24 release of Ghostscript had regressions, so we don't support it until 9.25.) """ - if threads is not None: - warnings.warn( - "use of deprecated parameter 'threads'", category=DeprecationWarning - ) compression_args = [] if compression == 'jpeg': From 168fc6077478c5aadfbdd00e612ab8c8e7642f68 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 05:14:59 -0700 Subject: [PATCH 426/880] Update release notes with v10 changes --- docs/release_notes.rst | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 86a1b9f3..3c01b831 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,31 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.0.0 (not yet released) +========================== + +**Breaking changes** + +- Support for pdfminer.six version 20181108 has been dropped, along with a + monkeypatch that made this version work. +- Ghostscript is no longer used for finding the location of text in PDFs, and + APIs related to this feature have been removed. +- Output messages are now displayed in color (when supported by the terminal) + and prefixes describing the severity of the message are removed. As such + programs that parse OCRmyPDF's log message will need to be revised. (Please + consider using OCRmyPDF as a library instead.) +- Code describing the resolution in DPI of images was refactored into a + ``ocrmypdf.helpers.Resolution`` class. +- A deprecated parameter in ``ocrmypdf.exec.ghostscript.generate_pdfa`` was + removed. +- The ``ocrmypdf.hocrtransform`` module has been updated to follow PEP8 naming + conventions. + +**New features** + +- PDF page scanning is now parallelized across CPUs, speeding up the "Scan" + phase for files with a high page count. +- Colored log messages. v9.7.1 ====== From 8f5c95f0f4aee6d50ec0c82d75133cd5621cd6cf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Apr 2020 05:33:26 -0700 Subject: [PATCH 427/880] Remove last vestiges of command line usage of qpdf - change to check_pdf --- docs/advanced.rst | 3 +- docs/errors.rst | 8 +-- docs/release_notes.rst | 1 + docs/security.rst | 9 ++-- src/ocrmypdf/_sync.py | 5 +- src/ocrmypdf/_validation.py | 7 --- src/ocrmypdf/exec/qpdf.py | 63 ----------------------- src/ocrmypdf/helpers.py | 37 +++++++++++++ tests/{test_qpdf.py => test_check_pdf.py} | 8 +-- tests/test_hocrtransform.py | 4 +- tests/test_main.py | 7 +-- tests/test_stdio.py | 4 +- tests/test_userunit.py | 2 +- 13 files changed, 63 insertions(+), 95 deletions(-) delete mode 100644 src/ocrmypdf/exec/qpdf.py rename tests/{test_qpdf.py => test_check_pdf.py} (82%) diff --git a/docs/advanced.rst b/docs/advanced.rst index 6f8567ba..9c30f1c5 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -148,7 +148,8 @@ In addition to tesseract, OCRmyPDF uses the following external binaries: - ``gs`` (Ghostscript) - ``unpaper`` -- ``qpdf`` +- ``pngquant`` +- ``jbig2`` In each case OCRmyPDF will search the ``PATH`` environment variable to locate the binaries. diff --git a/docs/errors.rst b/docs/errors.rst index 328080ad..bf0b53d0 100644 --- a/docs/errors.rst +++ b/docs/errors.rst @@ -32,10 +32,10 @@ As the error message suggests, your options are: Input file 'filename' is not a valid PDF ======================================== -OCRmyPDF passes files through qpdf, a program that fixes errors in PDFs, -before it tries to work on them. In most cases this happens because the -PDF is corrupt and truncated (incomplete file copying) and not much can -be done. +OCRmyPDF checks files with pikepdf, a library that in turn uses libqpdf to fixes +errors in PDFs, before it tries to work on them. In most cases this happens +because the PDF is corrupt and truncated (incomplete file copying) and not much +can be done. You can try rewriting the file with Ghostscript: diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 3c01b831..3c0bfa90 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -30,6 +30,7 @@ v10.0.0 (not yet released) ``ocrmypdf.helpers.Resolution`` class. - A deprecated parameter in ``ocrmypdf.exec.ghostscript.generate_pdfa`` was removed. +- The deprecated module ``ocrmypdf.exec.qpdf`` was removed. - The ``ocrmypdf.hocrtransform`` module has been updated to follow PEP8 naming conventions. diff --git a/docs/security.rst b/docs/security.rst index bcc69e8e..36246960 100644 --- a/docs/security.rst +++ b/docs/security.rst @@ -68,7 +68,7 @@ license, OCRmyPDF's GPL license, and any other licenses. Setting aside these concerns, a side effect of OCRmyPDF is it may incidentally sanitize PDFs that contain certain types of malware. It -runs ``qpdf`` to repair the PDF, which could correct malformed PDF +repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF structures that are part of an attack. When PDF/A output is selected (the default), the input PDF is partially reconstructed by Ghostscript. When ``--force-ocr`` is used, all pages are rasterized and reconverted @@ -144,10 +144,9 @@ set, the document cannot be viewed without the password. Either way, OCRmyPDF does not remove passwords from PDFs and exits with an error on encountering them. -``qpdf``, one of OCRmyPDF's dependencies, can remove passwords. If the -owner and user password are set, a password is required for ``qpdf``. If -only the owner password is set, then the password can be stripped, even -if one does not have the owner password. +``qpdf`` can remove passwords. If the owner and user password are set, a +password is required for ``qpdf``. If only the owner password is set, then the +password can be stripped, even if one does not have the owner password. After OCR is applied, password protection is not permitted on PDF/A documents but the file can be converted to regular PDF. diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index e6e44271..77ecfa9b 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -64,8 +64,7 @@ from ._validation import ( report_output_file_size, ) from .exceptions import ExitCode, ExitCodeException -from .exec import qpdf -from .helpers import available_cpu_count +from .helpers import available_cpu_count, check_pdf from .pdfa import file_claims_pdfa log = logging.getLogger(__name__) @@ -357,7 +356,7 @@ def run_pipeline(options, api=False): pdfa_info['conformance'], ) return ExitCode.pdfa_conversion_failed - if not qpdf.check(options.output_file): + if not check_pdf(options.output_file): log.warning('Output file: The generated PDF is INVALID') return ExitCode.invalid_output_pdf report_output_file_size(options, start_input_file, options.output_file) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index ef440e0f..8be0b16e 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -38,7 +38,6 @@ from .exec import ( ghostscript, jbig2enc, pngquant, - qpdf, tesseract, unpaper, ) @@ -473,9 +472,3 @@ def check_dependency_versions(options): "supported. Please upgrade to a newer version, or downgrade to the " "previous version." ) - check_external_program( - program='qpdf', - package='qpdf', - version_checker=qpdf.version, - need_version='8.0.2', - ) diff --git a/src/ocrmypdf/exec/qpdf.py b/src/ocrmypdf/exec/qpdf.py deleted file mode 100644 index 32ef8cb6..00000000 --- a/src/ocrmypdf/exec/qpdf.py +++ /dev/null @@ -1,63 +0,0 @@ -# © 2017 James R. Barlow: github.com/jbarlow83 -# -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . - -"""Interface to qpdf executable""" - -import logging -from io import StringIO - -import pikepdf - -log = logging.getLogger(__name__) - - -def version(): - return pikepdf.__libqpdf_version__ - - -def check(input_file): - pdf = None - try: - pdf = pikepdf.open(input_file) - except pikepdf.PdfError as e: - log.error(e) - return False - else: - messages = pdf.check() - for msg in messages: - if 'error' in msg.lower(): - log.error(msg) - else: - log.warning(msg) - - sio = StringIO() - linearize = None - try: - pdf.check_linearization(sio) - except RuntimeError: - pass - else: - linearize = sio.getvalue() - if linearize: - log.warning(linearize) - - if not messages and not linearize: - return True - return False - finally: - if pdf: - pdf.close() diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index c57aca35..69f4d9bf 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -24,9 +24,12 @@ from collections import namedtuple from collections.abc import Iterable from contextlib import suppress from functools import wraps +from io import StringIO from math import inf, isclose from pathlib import Path +import pikepdf + log = logging.getLogger(__name__) @@ -165,6 +168,40 @@ def is_file_writable(test_file: os.PathLike): return False +def check_pdf(input_file): + pdf = None + try: + pdf = pikepdf.open(input_file) + except pikepdf.PdfError as e: + log.error(e) + return False + else: + messages = pdf.check() + for msg in messages: + if 'error' in msg.lower(): + log.error(msg) + else: + log.warning(msg) + + sio = StringIO() + linearize = None + try: + pdf.check_linearization(sio) + except RuntimeError: + pass + else: + linearize = sio.getvalue() + if linearize: + log.warning(linearize) + + if not messages and not linearize: + return True + return False + finally: + if pdf: + pdf.close() + + def deprecated(func): """Warn that function is deprecated""" diff --git a/tests/test_qpdf.py b/tests/test_check_pdf.py similarity index 82% rename from tests/test_qpdf.py rename to tests/test_check_pdf.py index 0e925249..b3516e90 100644 --- a/tests/test_qpdf.py +++ b/tests/test_check_pdf.py @@ -17,9 +17,9 @@ import pytest -import ocrmypdf.exec.qpdf as qpdf +from ocrmypdf.helpers import check_pdf -def test_qpdf_error(resources): - assert qpdf.check(resources / 'blank.pdf') - assert not qpdf.check(__file__) +def test_pdf_error(resources): + assert check_pdf(resources / 'blank.pdf') + assert not check_pdf(__file__) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index 13b1f601..e00f8365 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -21,8 +21,8 @@ import pytest from PIL import Image from ocrmypdf import hocrtransform -from ocrmypdf.exec import qpdf from ocrmypdf.exec.tesseract import HOCR_TEMPLATE +from ocrmypdf.helpers import check_pdf # pylint: disable=redefined-outer-name @@ -43,4 +43,4 @@ def test_mono_image(blank_hocr, outdir): hocr = hocrtransform.HocrTransform(str(blank_hocr), 300) hocr.to_pdf(str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif')) - qpdf.check(str(outdir / 'mono.pdf')) + check_pdf(str(outdir / 'mono.pdf')) diff --git a/tests/test_main.py b/tests/test_main.py index 9b0cd17e..e69ed192 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -29,7 +29,8 @@ from PIL import Image import ocrmypdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import ghostscript, qpdf, tesseract +from ocrmypdf.exec import ghostscript, tesseract +from ocrmypdf.helpers import check_pdf from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo @@ -529,8 +530,8 @@ def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): @pytest.mark.skipif( - '8.0.0' <= qpdf.version() <= '8.0.1', - reason="qpdf regression on pages with no contents", + '8.0.0' <= pikepdf.__libqpdf_version__ <= '8.0.1', + reason="libqpdf regression on pages with no contents", ) def test_no_contents(spoof_tesseract_noop, resources, outpdf): check_ocrmypdf( diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 150e40ff..e57c11f0 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -23,7 +23,7 @@ from subprocess import DEVNULL, PIPE, CalledProcessError, Popen, run import pytest from ocrmypdf.exceptions import ExitCode -from ocrmypdf.exec import qpdf +from ocrmypdf.helpers import check_pdf # pytest.helpers is dynamic # pylint: disable=no-member,redefined-outer-name @@ -74,7 +74,7 @@ def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): ) assert p.returncode == ExitCode.ok - assert qpdf.check(output_file) + assert check_pdf(output_file) @pytest.mark.skipif( diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 83ad01d4..41d21aa2 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -39,7 +39,7 @@ def test_userunit_ghostscript_fails(poster, no_outpdf, caplog): assert 'not supported by Ghostscript' in caplog.text -def test_userunit_qpdf_passes(spoof_tesseract_cache, poster, outpdf): +def test_userunit_pdf_passes(spoof_tesseract_cache, poster, outpdf): before = PdfInfo(poster) check_ocrmypdf(poster, outpdf, '--output-type=pdf', env=spoof_tesseract_cache) From b840b16c82c3845bcece3c5f8e360857e5f40571 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 Apr 2020 02:35:23 -0700 Subject: [PATCH 428/880] Remove tesseract_badutf8.py Should have been removed in 9db01c7 --- tests/spoof/tesseract_badutf8.py | 80 -------------------------------- tests/test_stdio.py | 5 -- 2 files changed, 85 deletions(-) delete mode 100755 tests/spoof/tesseract_badutf8.py diff --git a/tests/spoof/tesseract_badutf8.py b/tests/spoof/tesseract_badutf8.py deleted file mode 100755 index 3cbb6625..00000000 --- a/tests/spoof/tesseract_badutf8.py +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env python3 -# © 2017 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -import sys - -"""Tesseract bad utf8 spoof - -In 'hocr' mode or 'pdf' mode, return error code 1 and some non-Unicode -text because tesseract seems to do that in some cases related to -language pack version mismatches - -""" - - -VERSION_STRING = '''tesseract 4.0.0 - leptonica-1.77.0 - libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 - Found AVX2 - Found AVX - Found SSE -SPOOFED -''' - -# Japanese "Invalid UTF-8" encoded in Shift JIS -BAD_UTF8 = b'\x96\xb3\x8c\xf8\x82\xc8UTF-8\x0a' - - -def main(): - if sys.argv[1] == '--version': - print(VERSION_STRING, file=sys.stderr) - sys.exit(0) - elif sys.argv[1] == '--list-langs': - print('List of available languages (1):\neng', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == '--print-parameters': - print("Some parameters", file=sys.stderr) - print("textonly_pdf\t1\tSome help text") - sys.exit(0) - elif sys.argv[-2] in ('hocr', 'pdf'): - sys.stdout.buffer.write(BAD_UTF8) - sys.exit(1) - elif sys.argv[-1] == 'stdout': - # input file is at sys.argv[-2] but we don't look at it - print( - """Orientation: 0 -Orientation in degrees: 0 -Orientation confidence: 100.00 -Script: 1 -Script confidence: 100.00""", - file=sys.stderr, - ) - else: - print("Spoof doesn't understand arguments", file=sys.stderr) - print(sys.argv, file=sys.stderr) - sys.exit(1) - - sys.exit(0) - - -if __name__ == '__main__': - main() diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 250ca6f2..21112468 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -33,11 +33,6 @@ run_ocrmypdf_api = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -@pytest.fixture -def spoof_tess_bad_utf8(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') - - def test_stdin(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): input_file = str(resources / 'francais.pdf') output_file = str(outpdf) From 17cd655752fba4e261a4a1260fa33bef22428b11 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 Apr 2020 02:37:17 -0700 Subject: [PATCH 429/880] Don't utf-8 decode tesseract --print-parameters Output not guaranteed to be UTF-8. Fixes #543. --- src/ocrmypdf/exec/tesseract.py | 12 +++--------- 1 file changed, 3 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index abcbb2fa..c8a16f42 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -77,20 +77,14 @@ def has_textonly_pdf(tesseract_env=None, langs=None): args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf'] params = '' try: - proc = run( - args_tess, - check=True, - universal_newlines=True, - stdout=PIPE, - stderr=STDOUT, - env=tesseract_env, - ) + # print-parameters can return non-UTF8 if the parameters are so initialized + proc = run(args_tess, check=True, stdout=PIPE, stderr=STDOUT, env=tesseract_env) params = proc.stdout except CalledProcessError as e: raise MissingDependencyError( "Could not --print-parameters from tesseract" ) from e - if 'textonly_pdf' in params: + if b'textonly_pdf' in params: return True return False From b59e761a14af27a079a286b4fe52359d671ab8e3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 28 Apr 2020 02:40:17 -0700 Subject: [PATCH 430/880] v9.8.0 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 66f1f576..a5f49460 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,14 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.8.0 +====== + +- Fixed issue where only the first PNG (FlateDecode) image in a file would be + considered for optimization. File sizes should be improved from here on. +- Fixed a startup crash when the chosen language was Japanese (#543). +- Added options to configure polling and log level to watcher.py. + v9.7.2 ====== From 016dfd420c7d42d01583ac6902c8926bd6a96126 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 Apr 2020 04:11:38 -0700 Subject: [PATCH 431/880] Add warning if problematic --tesseract-pagesegmode is selected Fixes #549 --- src/ocrmypdf/_validation.py | 5 +++++ tests/test_validation.py | 6 ++++++ 2 files changed, 11 insertions(+) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 8be0b16e..bfc25d02 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -276,6 +276,11 @@ def check_options_advanced(options): "Tesseract 4.0 ignores --user-words and --user-patterns, so these " "arguments have no effect." ) + if options.tesseract_pagesegmode in (0, 2): + log.warning( + "The --tesseract-pagesegmode argument you select will disable OCR. " + "This may cause processing to fail." + ) def check_options_metadata(options): diff --git a/tests/test_validation.py b/tests/test_validation.py index 864dd05a..f183aa46 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -210,3 +210,9 @@ def test_version_comparison(): version_checker=lambda: '1.0', need_version='2.0', ) + + +def test_pagesegmode_warning(caplog): + opts = make_opts(tesseract_pagesegmode='0') + vd.check_options_advanced(opts) + assert 'disable OCR' in caplog.text From 82bce463aece0c2e423a2fc7c0d4319546e77b24 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 May 2020 02:15:23 -0700 Subject: [PATCH 432/880] Start pluggy-based plugin system --- setup.cfg | 2 +- src/ocrmypdf/__init__.py | 13 +++++--- src/ocrmypdf/_jobcontext.py | 3 +- src/ocrmypdf/_pipeline.py | 1 + src/ocrmypdf/_pluginspec.py | 66 +++++++++++++++++++++++++++++++++++++ src/ocrmypdf/_sync.py | 49 +++++++++++++++++++++------ src/ocrmypdf/cli.py | 6 ++++ src/ocrmypdf/example.py | 6 ++++ 8 files changed, 130 insertions(+), 16 deletions(-) create mode 100644 src/ocrmypdf/_pluginspec.py create mode 100644 src/ocrmypdf/example.py diff --git a/setup.cfg b/setup.cfg index f307a2e5..3cb3db9d 100644 --- a/setup.cfg +++ b/setup.cfg @@ -23,7 +23,7 @@ force_grid_wrap=0 use_parentheses=True line_length=88 known_first_party = ocrmypdf -known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug +known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug [metadata] license_file = LICENSE diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 2f76bf4f..2326efb2 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -15,10 +15,13 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from . import helpers, hocrtransform, leptonica, pdfa, pdfinfo -from ._version import PROGRAM_NAME, __version__ -from .api import Verbosity, configure_logging, ocr -from .exceptions import ( + +from pluggy import HookimplMarker + +from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo +from ocrmypdf._version import PROGRAM_NAME, __version__ +from ocrmypdf.api import Verbosity, configure_logging, ocr +from ocrmypdf.exceptions import ( BadArgsError, DpiError, EncryptedPdfError, @@ -33,3 +36,5 @@ from .exceptions import ( TesseractConfigError, UnsupportedImageFormatError, ) + +hookimpl = HookimplMarker('ocrmypdf') diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index a782d596..eac75d5e 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -24,11 +24,12 @@ import sys class PDFContext: """Holds our context for a particular run of the pipeline""" - def __init__(self, options, work_folder, origin, pdfinfo): + def __init__(self, options, work_folder, origin, pdfinfo, plugin_manager): self.options = options self.work_folder = work_folder self.origin = origin self.pdfinfo = pdfinfo + self.plugin_manager = plugin_manager if options: self.name = os.path.basename(options.input_file) else: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 2128adf3..6a691b81 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -194,6 +194,7 @@ def validate_pdfinfo_options(context): "form and all filled form fields. The output PDF will be " "'flattened' and will no longer be fillable." ) + context.plugin_manager.hook.prepare(options=options) def get_page_dpi(pageinfo, options): diff --git a/src/ocrmypdf/_pluginspec.py b/src/ocrmypdf/_pluginspec.py new file mode 100644 index 00000000..c3aac6f2 --- /dev/null +++ b/src/ocrmypdf/_pluginspec.py @@ -0,0 +1,66 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from argparse import Namespace + +import pluggy +from PIL import Image + +from ocrmypdf.pdfinfo import PdfInfo + +hookspec = pluggy.HookspecMarker('ocrmypdf') + +# pylint: disable=unused-argument + + +@hookspec +def prepare(options: Namespace) -> None: + """Called to notify a plugin that a file will be processed. + + The plugin may modify the options. All objects that are in options must + be picklable so they can be marshalled to child worker processes. + + Typically, a plugin will call ``registry.register_plugin(__name__)`` to register + all of its public functions with the plugin registry. Functions that are + not intended for registration should be prefixed with an underscore. + Functions that imported from other modules will be ignored by + ``.register_plugin()``. For example if you use ``from os import basename``, + ``basename`` will not be registered. + """ + + +@hookspec +def validate(pdfinfo: PdfInfo, options: Namespace) -> None: + """Called to give a plugin an opportunity to review options and pdfinfo. + + options contains the "work order" to process a particular file. pdfinfo + contains information about the input file obtained after loading and + parsing. + + The plugin may raise InputFileError or any ExitCodeException to request + normal termination. If the plugin raises another exception type, ocrmypdf + will abort with an error and hold the plugin responsible. + """ + + +@hookspec +def filter_ocr_image(image: Image) -> Image: + """Called to filter the image before it is sent to OCR. + + This is the image that OCR sees, not what the user sees when they view the + PDF. + """ diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 77ecfa9b..69da25bc 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import importlib import logging import logging.handlers import multiprocessing @@ -28,12 +29,14 @@ from pathlib import Path from tempfile import mkdtemp import PIL +import pluggy -from ._concurrent import exec_progress_pool -from ._graft import OcrGrafter -from ._jobcontext import PDFContext, cleanup_working_files -from ._logging import PageNumberFilter -from ._pipeline import ( +from ocrmypdf import _pluginspec +from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._graft import OcrGrafter +from ocrmypdf._jobcontext import PDFContext, cleanup_working_files +from ocrmypdf._logging import PageNumberFilter +from ocrmypdf._pipeline import ( convert_to_pdfa, copy_final, create_ocr_image, @@ -58,14 +61,14 @@ from ._pipeline import ( triage, validate_pdfinfo_options, ) -from ._validation import ( +from ocrmypdf._validation import ( check_requested_output_file, create_input_file, report_output_file_size, ) -from .exceptions import ExitCode, ExitCodeException -from .helpers import available_cpu_count, check_pdf -from .pdfa import file_claims_pdfa +from ocrmypdf.exceptions import ExitCode, ExitCodeException +from ocrmypdf.helpers import available_cpu_count, check_pdf +from ocrmypdf.pdfa import file_claims_pdfa log = logging.getLogger(__name__) @@ -298,6 +301,24 @@ def configure_debug_logging(log_filename, prefix=''): return log_file_handler +def _load_object_from_module(location): + """Load a object given a module location + + For location=a.b.c, will effectively run "from a.b import c" + + Example: + _load_object_from_module("a.b.c") + + """ + module_parts = location.split('.') + module_name = '.'.join(module_parts[:-1]) + object_name = module_parts[-1] + module = importlib.import_module(module_name) + obj = getattr(module, object_name) + log.debug(f"Loaded object: from {module_name} import {object_name}") + return obj + + def run_pipeline(options, api=False): # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example @@ -312,6 +333,14 @@ def run_pipeline(options, api=False): ): debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") + pm = pluggy.PluginManager('ocrmypdf') + pm.add_hookspecs(_pluginspec) + + for name in options.plugins: + # module = _load_object_from_module(name) + module = importlib.import_module(name) + pm.register(module) + try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) @@ -331,7 +360,7 @@ def run_pipeline(options, api=False): max_workers=options.jobs if not options.use_threads else 1, # To help debug ) - context = PDFContext(options, work_folder, origin_pdf, pdfinfo) + context = PDFContext(options, work_folder, origin_pdf, pdfinfo, pm) # Validate options are okay for this pdf validate_pdfinfo_options(context) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index e28162e7..48a90339 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -480,6 +480,12 @@ advanced.add_argument( "which do not benefit. If the threshold is 0 it will be apply to all files. " "Set the threshold very high to disable.", ) +advanced.add_argument( + '--plugins', + action='append', + default=[], + help="Path to a folder than contains plugins.", +) debugging = parser.add_argument_group( "Debugging", "Arguments to help with troubleshooting and debugging" diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py new file mode 100644 index 00000000..a890bb3e --- /dev/null +++ b/src/ocrmypdf/example.py @@ -0,0 +1,6 @@ +import ocrmypdf + + +@ocrmypdf.hookimpl +def prepare(options): + raise ValueError('foo') From d8ff4485f8482411480431c7774833f8cc916439 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 May 2020 02:18:11 -0700 Subject: [PATCH 433/880] Move samefile to helpers --- src/ocrmypdf/_sync.py | 9 +-------- src/ocrmypdf/helpers.py | 7 +++++++ 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 69da25bc..0497f5f1 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -67,7 +67,7 @@ from ocrmypdf._validation import ( report_output_file_size, ) from ocrmypdf.exceptions import ExitCode, ExitCodeException -from ocrmypdf.helpers import available_cpu_count, check_pdf +from ocrmypdf.helpers import available_cpu_count, check_pdf, samefile from ocrmypdf.pdfa import file_claims_pdfa log = logging.getLogger(__name__) @@ -283,13 +283,6 @@ class NeverRaise(Exception): pass # pylint: disable=unnecessary-pass -def samefile(f1, f2): - if os.name == 'nt': - return f1 == f2 - else: - return os.path.samefile(f1, f2) - - def configure_debug_logging(log_filename, prefix=''): log_file_handler = logging.FileHandler(log_filename, delay=True) log_file_handler.setLevel(logging.DEBUG) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 69f4d9bf..0bfe694a 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -104,6 +104,13 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, ** os.symlink(os.path.abspath(input_file), soft_link_name) +def samefile(f1, f2): + if os.name == 'nt': + return f1 == f2 + else: + return os.path.samefile(f1, f2) + + def is_iterable_notstr(thing): return isinstance(thing, Iterable) and not isinstance(thing, str) From 5eb4fe00525dfec915b2b394eae05a7aa12e44fc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 May 2020 02:18:31 -0700 Subject: [PATCH 434/880] Refactor plugin setup to get_plugin_manager --- src/ocrmypdf/_sync.py | 33 ++++++++------------------------- 1 file changed, 8 insertions(+), 25 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 0497f5f1..ac5a8bef 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,9 +18,7 @@ import importlib import logging import logging.handlers -import multiprocessing import os -import signal import sys import threading from collections import namedtuple @@ -294,22 +292,14 @@ def configure_debug_logging(log_filename, prefix=''): return log_file_handler -def _load_object_from_module(location): - """Load a object given a module location +def get_plugin_manager(options): + pm = pluggy.PluginManager('ocrmypdf') + pm.add_hookspecs(_pluginspec) - For location=a.b.c, will effectively run "from a.b import c" - - Example: - _load_object_from_module("a.b.c") - - """ - module_parts = location.split('.') - module_name = '.'.join(module_parts[:-1]) - object_name = module_parts[-1] - module = importlib.import_module(module_name) - obj = getattr(module, object_name) - log.debug(f"Loaded object: from {module_name} import {object_name}") - return obj + for name in options.plugins: + module = importlib.import_module(name) + pm.register(module) + return pm def run_pipeline(options, api=False): @@ -326,14 +316,7 @@ def run_pipeline(options, api=False): ): debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") - pm = pluggy.PluginManager('ocrmypdf') - pm.add_hookspecs(_pluginspec) - - for name in options.plugins: - # module = _load_object_from_module(name) - module = importlib.import_module(name) - pm.register(module) - + pm = get_plugin_manager(options) try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) From 8d2535e327d98ac8dc3995880117b026e5a43edd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 May 2020 02:39:50 -0700 Subject: [PATCH 435/880] Get pluggy to work with forking workers --- src/ocrmypdf/_jobcontext.py | 15 +++++++++++++++ src/ocrmypdf/_pipeline.py | 3 +++ src/ocrmypdf/_plugin_manager.py | 15 +++++++++++++++ src/ocrmypdf/_sync.py | 13 ++----------- src/ocrmypdf/example.py | 11 ++++++++--- src/ocrmypdf/{_pluginspec.py => pluginspec.py} | 0 6 files changed, 43 insertions(+), 14 deletions(-) create mode 100644 src/ocrmypdf/_plugin_manager.py rename src/ocrmypdf/{_pluginspec.py => pluginspec.py} (100%) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index eac75d5e..ea48380b 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -19,6 +19,9 @@ import logging import os import shutil import sys +from functools import partial + +from ocrmypdf._plugin_manager import get_plugin_manager class PDFContext: @@ -59,10 +62,22 @@ class PageContext: self.name = pdf_context.name self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] + self.plugin_manager = pdf_context.plugin_manager def get_path(self, name): return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name)) + def __getstate__(self): + state = self.__dict__.copy() + del state['plugin_manager'] + state['construct_plugin_manager'] = partial(get_plugin_manager, self.options) + return state + + def __setstate__(self, state): + self.__dict__.update(state) + self.plugin_manager = self.__dict__['construct_plugin_manager']() + del self.__dict__['construct_plugin_manager'] + def cleanup_working_files(work_folder, options): if options.keep_temporary_files: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6a691b81..182a2888 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -521,6 +521,9 @@ def create_ocr_image(image, page_context): im = pix.topil() del draw + + im = page_context.plugin_manager.hook.filter_ocr_image(image=im) + # Pillow requires integer DPI dpi = tuple(round(coord) for coord in im.info['dpi']) im.save(output_file, dpi=dpi) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py new file mode 100644 index 00000000..a7f7987e --- /dev/null +++ b/src/ocrmypdf/_plugin_manager.py @@ -0,0 +1,15 @@ +import importlib + +import pluggy + +from ocrmypdf import pluginspec + + +def get_plugin_manager(options): + pm = pluggy.PluginManager('ocrmypdf') + pm.add_hookspecs(pluginspec) + + for name in options.plugins: + module = importlib.import_module(name) + pm.register(module) + return pm diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index ac5a8bef..6f21bb09 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -29,7 +29,7 @@ from tempfile import mkdtemp import PIL import pluggy -from ocrmypdf import _pluginspec +from ocrmypdf import pluginspec from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._graft import OcrGrafter from ocrmypdf._jobcontext import PDFContext, cleanup_working_files @@ -59,6 +59,7 @@ from ocrmypdf._pipeline import ( triage, validate_pdfinfo_options, ) +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._validation import ( check_requested_output_file, create_input_file, @@ -292,16 +293,6 @@ def configure_debug_logging(log_filename, prefix=''): return log_file_handler -def get_plugin_manager(options): - pm = pluggy.PluginManager('ocrmypdf') - pm.add_hookspecs(_pluginspec) - - for name in options.plugins: - module = importlib.import_module(name) - pm.register(module) - return pm - - def run_pipeline(options, api=False): # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py index a890bb3e..755e0f57 100644 --- a/src/ocrmypdf/example.py +++ b/src/ocrmypdf/example.py @@ -1,6 +1,11 @@ -import ocrmypdf +from ocrmypdf import hookimpl -@ocrmypdf.hookimpl +@hookimpl def prepare(options): - raise ValueError('foo') + pass + + +@hookimpl +def filter_ocr_image(image): + return image diff --git a/src/ocrmypdf/_pluginspec.py b/src/ocrmypdf/pluginspec.py similarity index 100% rename from src/ocrmypdf/_pluginspec.py rename to src/ocrmypdf/pluginspec.py From be107b4fedb838b1c4b4599abd75488edf6f6fa1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 May 2020 02:56:41 -0700 Subject: [PATCH 436/880] Set up filter_ocr_image hook --- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_plugin_manager.py | 17 +++++++++++++++++ src/ocrmypdf/_sync.py | 2 ++ src/ocrmypdf/example.py | 9 +++++++++ src/ocrmypdf/pluginspec.py | 9 +-------- 5 files changed, 30 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 182a2888..142c2983 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -194,7 +194,7 @@ def validate_pdfinfo_options(context): "form and all filled form fields. The output PDF will be " "'flattened' and will no longer be fillable." ) - context.plugin_manager.hook.prepare(options=options) + context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options) def get_page_dpi(pageinfo, options): diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index a7f7987e..2c3df5a1 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -1,3 +1,20 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + import importlib import pluggy diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 6f21bb09..1691eb00 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -320,6 +320,8 @@ def run_pipeline(options, api=False): options, ) + pm.hook.prepare(options=options) + # Gather pdfinfo and create context pdfinfo = get_pdfinfo( origin_pdf, diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py index 755e0f57..dda37bf0 100644 --- a/src/ocrmypdf/example.py +++ b/src/ocrmypdf/example.py @@ -1,11 +1,20 @@ +import logging + from ocrmypdf import hookimpl +log = logging.getLogger(__name__) + @hookimpl def prepare(options): pass +@hookimpl +def validate(pdfinfo, options): + pass + + @hookimpl def filter_ocr_image(image): return image diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index c3aac6f2..86657c9d 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -33,13 +33,6 @@ def prepare(options: Namespace) -> None: The plugin may modify the options. All objects that are in options must be picklable so they can be marshalled to child worker processes. - - Typically, a plugin will call ``registry.register_plugin(__name__)`` to register - all of its public functions with the plugin registry. Functions that are - not intended for registration should be prefixed with an underscore. - Functions that imported from other modules will be ignored by - ``.register_plugin()``. For example if you use ``from os import basename``, - ``basename`` will not be registered. """ @@ -57,7 +50,7 @@ def validate(pdfinfo: PdfInfo, options: Namespace) -> None: """ -@hookspec +@hookspec(firstresult=True) def filter_ocr_image(image: Image) -> Image: """Called to filter the image before it is sent to OCR. From 23d558ad8c4149a503aed237893cf6b8af6761e9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 2 May 2020 01:37:24 -0700 Subject: [PATCH 437/880] Allow plugins to add command line arguments --- src/ocrmypdf/__main__.py | 18 ++++++++++++------ src/ocrmypdf/_pipeline.py | 4 +++- src/ocrmypdf/cli.py | 10 ++++++++++ src/ocrmypdf/example.py | 10 +++++++++- src/ocrmypdf/pluginspec.py | 10 ++++++++-- 5 files changed, 42 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 5d3a5d50..c62c9409 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -21,17 +21,23 @@ import os import sys from multiprocessing import set_start_method -from . import __version__ -from ._sync import run_pipeline -from ._validation import check_closed_streams, check_options -from .api import Verbosity, configure_logging -from .cli import parser -from .exceptions import BadArgsError, ExitCode, MissingDependencyError +from ocrmypdf import __version__ +from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf._sync import run_pipeline +from ocrmypdf._validation import check_closed_streams, check_options +from ocrmypdf.api import Verbosity, configure_logging +from ocrmypdf.cli import parser, plugins_only_parser +from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError log = logging.getLogger('ocrmypdf') def run(args=None): + pre_options, _unused = plugins_only_parser.parse_known_args(args=args) + if pre_options.plugins: + pm = get_plugin_manager(pre_options) + pm.hook.install_cli(parser=parser) + options = parser.parse_args(args=args) if not check_closed_streams(options): diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 142c2983..a8d6365f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -522,7 +522,9 @@ def create_ocr_image(image, page_context): del draw - im = page_context.plugin_manager.hook.filter_ocr_image(image=im) + im = page_context.plugin_manager.hook.filter_ocr_image( + page=page_context, image=im + ) # Pillow requires integer DPI dpi = tuple(round(coord) for coord in im.info['dpi']) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 48a90339..d243364a 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -497,3 +497,13 @@ debugging.add_argument( help="Keep temporary files (helpful for debugging)", ) debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) + +plugins_only_parser = ArgumentParser( + prog=_PROGRAM_NAME, fromfile_prefix_chars='@', add_help=False, allow_abbrev=False +) +plugins_only_parser.add_argument( + '--plugins', + action='append', + default=[], + help="Path to a folder than contains plugins.", +) diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py index dda37bf0..73b1b544 100644 --- a/src/ocrmypdf/example.py +++ b/src/ocrmypdf/example.py @@ -5,6 +5,11 @@ from ocrmypdf import hookimpl log = logging.getLogger(__name__) +@hookimpl +def install_cli(parser): + parser.add_argument('--invert', action='store_true') + + @hookimpl def prepare(options): pass @@ -16,5 +21,8 @@ def validate(pdfinfo, options): @hookimpl -def filter_ocr_image(image): +def filter_ocr_image(page, image): + if page.options.invert: + log.info("inverting") + return image.invert() return image diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 86657c9d..5b5d7493 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -15,11 +15,12 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from argparse import Namespace +from argparse import ArgumentParser, Namespace import pluggy from PIL import Image +from ocrmypdf._jobcontext import PageContext from ocrmypdf.pdfinfo import PdfInfo hookspec = pluggy.HookspecMarker('ocrmypdf') @@ -27,6 +28,11 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument +@hookspec +def install_cli(parser: ArgumentParser) -> None: + """Allows the plugin to add its own command line arguments.""" + + @hookspec def prepare(options: Namespace) -> None: """Called to notify a plugin that a file will be processed. @@ -51,7 +57,7 @@ def validate(pdfinfo: PdfInfo, options: Namespace) -> None: @hookspec(firstresult=True) -def filter_ocr_image(image: Image) -> Image: +def filter_ocr_image(page: PageContext, image: Image) -> Image: """Called to filter the image before it is sent to OCR. This is the image that OCR sees, not what the user sees when they view the From 8c9a8fc85cb0cafdd41ab305d20077cc21e8540a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 2 May 2020 03:32:55 -0700 Subject: [PATCH 438/880] pluginspec: avoid circular reference --- src/ocrmypdf/pluginspec.py | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 5b5d7493..e95241e0 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -20,9 +20,6 @@ from argparse import ArgumentParser, Namespace import pluggy from PIL import Image -from ocrmypdf._jobcontext import PageContext -from ocrmypdf.pdfinfo import PdfInfo - hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument @@ -43,7 +40,7 @@ def prepare(options: Namespace) -> None: @hookspec -def validate(pdfinfo: PdfInfo, options: Namespace) -> None: +def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: """Called to give a plugin an opportunity to review options and pdfinfo. options contains the "work order" to process a particular file. pdfinfo @@ -57,7 +54,7 @@ def validate(pdfinfo: PdfInfo, options: Namespace) -> None: @hookspec(firstresult=True) -def filter_ocr_image(page: PageContext, image: Image) -> Image: +def filter_ocr_image(page: 'PageContext', image: Image) -> Image: """Called to filter the image before it is sent to OCR. This is the image that OCR sees, not what the user sees when they view the From e02f6c1e97c4353834f7c982ec2d79c15b60aef7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 2 May 2020 03:34:31 -0700 Subject: [PATCH 439/880] Support plugin invocation with API --- src/ocrmypdf/__main__.py | 11 +- src/ocrmypdf/_jobcontext.py | 12 +- src/ocrmypdf/_pipeline.py | 4 +- src/ocrmypdf/_plugin_manager.py | 6 +- src/ocrmypdf/_sync.py | 9 +- src/ocrmypdf/api.py | 32 +- src/ocrmypdf/cli.py | 791 ++++++++++++++++---------------- src/ocrmypdf/optimize.py | 2 +- tests/conftest.py | 8 +- tests/test_metadata.py | 17 +- tests/test_unpaper.py | 4 +- tests/test_validation.py | 5 +- 12 files changed, 472 insertions(+), 429 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index c62c9409..b6dc0466 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -26,7 +26,7 @@ from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_closed_streams, check_options from ocrmypdf.api import Verbosity, configure_logging -from ocrmypdf.cli import parser, plugins_only_parser +from ocrmypdf.cli import get_parser, plugins_only_parser from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError log = logging.getLogger('ocrmypdf') @@ -34,9 +34,10 @@ log = logging.getLogger('ocrmypdf') def run(args=None): pre_options, _unused = plugins_only_parser.parse_known_args(args=args) - if pre_options.plugins: - pm = get_plugin_manager(pre_options) - pm.hook.install_cli(parser=parser) + plugin_manager = get_plugin_manager(pre_options.plugins) + + parser = get_parser() + plugin_manager.hook.install_cli(parser=parser) options = parser.parse_args(args=args) @@ -68,7 +69,7 @@ def run(args=None): log.error(e) return ExitCode.missing_dependency - result = run_pipeline(options=options) + result = run_pipeline(options=options, plugin_manager=plugin_manager) return result diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index ea48380b..7158abe8 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -69,13 +69,19 @@ class PageContext: def __getstate__(self): state = self.__dict__.copy() - del state['plugin_manager'] - state['construct_plugin_manager'] = partial(get_plugin_manager, self.options) + if state['plugin_manager'] is not None: + del state['plugin_manager'] + state['construct_plugin_manager'] = partial( + get_plugin_manager, self.options.plugins + ) return state def __setstate__(self, state): self.__dict__.update(state) - self.plugin_manager = self.__dict__['construct_plugin_manager']() + if 'construct_plugin_manager' in state: + self.plugin_manager = state['construct_plugin_manager']() + else: + self.plugin_manager = None del self.__dict__['construct_plugin_manager'] diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index a8d6365f..cd82fea4 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -522,9 +522,11 @@ def create_ocr_image(image, page_context): del draw - im = page_context.plugin_manager.hook.filter_ocr_image( + filter_im = page_context.plugin_manager.hook.filter_ocr_image( page=page_context, image=im ) + if filter_im is not None: + im = filter_im # Pillow requires integer DPI dpi = tuple(round(coord) for coord in im.info['dpi']) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 2c3df5a1..295e955c 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -16,17 +16,17 @@ # along with OCRmyPDF. If not, see . import importlib +from typing import List import pluggy from ocrmypdf import pluginspec -def get_plugin_manager(options): +def get_plugin_manager(plugins: List[str]): pm = pluggy.PluginManager('ocrmypdf') pm.add_hookspecs(pluginspec) - - for name in options.plugins: + for name in plugins: module = importlib.import_module(name) pm.register(module) return pm diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 1691eb00..0834993d 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -293,12 +293,14 @@ def configure_debug_logging(log_filename, prefix=''): return log_file_handler -def run_pipeline(options, api=False): +def run_pipeline(options, *, plugin_manager, api=False): # Any changes to options will not take effect for options that are already # bound to function parameters in the pipeline. (For example # options.input_file, options.pdf_renderer are already bound.) if not options.jobs: options.jobs = available_cpu_count() + if not plugin_manager: + plugin_manager = get_plugin_manager([]) work_folder = mkdtemp(prefix="com.github.ocrmypdf.") debug_log_handler = None @@ -307,7 +309,6 @@ def run_pipeline(options, api=False): ): debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") - pm = get_plugin_manager(options) try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) @@ -320,7 +321,7 @@ def run_pipeline(options, api=False): options, ) - pm.hook.prepare(options=options) + plugin_manager.hook.prepare(options=options) # Gather pdfinfo and create context pdfinfo = get_pdfinfo( @@ -329,7 +330,7 @@ def run_pipeline(options, api=False): max_workers=options.jobs if not options.use_threads else 1, # To help debug ) - context = PDFContext(options, work_folder, origin_pdf, pdfinfo, pm) + context = PDFContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager) # Validate options are okay for this pdf validate_pdfinfo_options(context) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index efa24c7f..cc7e103f 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -18,15 +18,17 @@ import logging import os import sys +from argparse import ArgumentParser from contextlib import suppress from enum import IntEnum from pathlib import Path from typing import Dict, Iterable -from ._logging import PageNumberFilter, TqdmConsole -from ._sync import run_pipeline -from ._validation import check_options -from .cli import parser +from ocrmypdf._logging import PageNumberFilter, TqdmConsole +from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf._sync import run_pipeline +from ocrmypdf._validation import check_options +from ocrmypdf.cli import get_parser, plugins_only_parser try: import coloredlogs @@ -125,7 +127,13 @@ def configure_logging( return log -def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwargs): +def create_options( + *, + input_file: os.PathLike, + output_file: os.PathLike, + parser: ArgumentParser, + **kwargs, +): cmdline = [] deferred = [] @@ -223,9 +231,11 @@ def ocr( # pylint: disable=unused-argument user_words: os.PathLike = None, user_patterns: os.PathLike = None, fast_web_view: float = None, + plugins: Iterable[str] = None, keep_temporary_files: bool = None, progress_bar: bool = None, tesseract_env: Dict[str, str] = None, + **kwargs, ): """Run OCRmyPDF on one PDF or image. @@ -260,7 +270,15 @@ def ocr( # pylint: disable=unused-argument Returns: :class:`ocrmypdf.ExitCode` """ + if not plugins: + plugins = [] - options = create_options(**locals()) + parser = get_parser() + _plugin_manager = get_plugin_manager(plugins) + _plugin_manager.hook.install_cli(parser=parser) + + options = create_options( + **{k: v for k, v in locals().items() if not k.startswith('_')} + ) check_options(options) - return run_pipeline(options, api=True) + return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index d243364a..3edf6cd7 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -17,10 +17,8 @@ import argparse -from ._version import PROGRAM_NAME as _PROGRAM_NAME -from ._version import __version__ as _VERSION - -__all__ = ['parser'] +from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME +from ocrmypdf._version import __version__ as _VERSION def numeric(basetype, min_=None, max_=None): @@ -56,18 +54,19 @@ class ArgumentParser(argparse.ArgumentParser): raise ValueError(message) -parser = ArgumentParser( - prog=_PROGRAM_NAME, - fromfile_prefix_chars='@', - formatter_class=argparse.RawDescriptionHelpFormatter, - description="""\ +def get_parser(): + parser = ArgumentParser( + prog=_PROGRAM_NAME, + fromfile_prefix_chars='@', + formatter_class=argparse.RawDescriptionHelpFormatter, + description="""\ Generates a searchable PDF or PDF/A from a regular PDF. OCRmyPDF rasterizes each page of the input PDF, optionally corrects page rotation and performs image processing, runs the Tesseract OCR engine on the image, and then creates a PDF from the OCR information. """, - epilog="""\ + epilog="""\ OCRmyPDF attempts to keep the output file at about the same size. If a file contains losslessly compressed images, and output file will be losslessly compressed as well. @@ -108,395 +107,409 @@ Online documentation is located at: https://ocrmypdf.readthedocs.io/en/latest/introduction.html """, -) + ) -parser.add_argument( - 'input_file', - metavar="input_pdf_or_image", - help="PDF file containing the images to be OCRed (or '-' to read from " - "standard input)", -) -parser.add_argument( - 'output_file', - metavar="output_pdf", - help="Output searchable PDF file (or '-' to write to standard output). " - "Existing files will be ovewritten. If same as input file, the " - "input file will be updated only if processing is successful.", -) -parser.add_argument( - '-l', - '--language', - action='append', - help="Language(s) of the file to be OCRed (see tesseract --list-langs for " - "all language packs installed in your system). Use -l eng+deu for " - "multiple languages.", -) -parser.add_argument( - '--image-dpi', - metavar='DPI', - type=int, - help="For input image instead of PDF, use this DPI instead of file's.", -) -parser.add_argument( - '--output-type', - choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'], - default='pdfa', - help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for " - "long term archiving (default, recommended) but may not suitable " - "for users who want their file altered as little as possible. 'pdfa' " - "also has problems with full Unicode text. 'pdf' attempts to " - "preserve file contents as much as possible. 'pdf-a1' creates a " - "PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a " - "PDF/A3-b file.", -) + parser.add_argument( + 'input_file', + metavar="input_pdf_or_image", + help="PDF file containing the images to be OCRed (or '-' to read from " + "standard input)", + ) + parser.add_argument( + 'output_file', + metavar="output_pdf", + help="Output searchable PDF file (or '-' to write to standard output). " + "Existing files will be ovewritten. If same as input file, the " + "input file will be updated only if processing is successful.", + ) + parser.add_argument( + '-l', + '--language', + action='append', + help="Language(s) of the file to be OCRed (see tesseract --list-langs for " + "all language packs installed in your system). Use -l eng+deu for " + "multiple languages.", + ) + parser.add_argument( + '--image-dpi', + metavar='DPI', + type=int, + help="For input image instead of PDF, use this DPI instead of file's.", + ) + parser.add_argument( + '--output-type', + choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'], + default='pdfa', + help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for " + "long term archiving (default, recommended) but may not suitable " + "for users who want their file altered as little as possible. 'pdfa' " + "also has problems with full Unicode text. 'pdf' attempts to " + "preserve file contents as much as possible. 'pdf-a1' creates a " + "PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a " + "PDF/A3-b file.", + ) -# Use null string '\0' as sentinel to indicate the user supplied no argument, -# since that is the only invalid character for filepaths on all platforms -# bool('\0') is True in Python -parser.add_argument( - '--sidecar', - nargs='?', - const='\0', - default=None, - metavar='FILE', - help="Generate sidecar text files that contain the same text recognized " - "by Tesseract. This may be useful for building a OCR text database. " - "If FILE is omitted, the sidecar file be named {output_file}.txt " - "If FILE is set to '-', the sidecar is written to stdout (a " - "convenient way to preview OCR quality). The output file and sidecar " - "may not both use stdout at the same time.", -) + # Use null string '\0' as sentinel to indicate the user supplied no argument, + # since that is the only invalid character for filepaths on all platforms + # bool('\0') is True in Python + parser.add_argument( + '--sidecar', + nargs='?', + const='\0', + default=None, + metavar='FILE', + help="Generate sidecar text files that contain the same text recognized " + "by Tesseract. This may be useful for building a OCR text database. " + "If FILE is omitted, the sidecar file be named {output_file}.txt " + "If FILE is set to '-', the sidecar is written to stdout (a " + "convenient way to preview OCR quality). The output file and sidecar " + "may not both use stdout at the same time.", + ) -parser.add_argument( - '--version', - action='version', - version=_VERSION, - help="Print program version and exit", -) + parser.add_argument( + '--version', + action='version', + version=_VERSION, + help="Print program version and exit", + ) -jobcontrol = parser.add_argument_group("Job control options") -jobcontrol.add_argument( - '-j', - '--jobs', - metavar='N', - type=numeric(int, 0, 256), - help="Use up to N CPU cores simultaneously (default: use all).", -) -jobcontrol.add_argument( - '-q', '--quiet', action='store_true', help="Suppress INFO messages" -) -jobcontrol.add_argument( - '-v', - '--verbose', - type=numeric(int, 0, 2), - default=0, - const=1, - nargs='?', - help="Print more verbose messages for each additional verbose level. Use " - "`-v 1` typically for much more detailed logging. Higher numbers " - "are probably only useful in debugging.", -) -jobcontrol.add_argument( - '--no-progress-bar', - action='store_false', - dest='progress_bar', - help=argparse.SUPPRESS, -) -jobcontrol.add_argument('--use-threads', action='store_true', help=argparse.SUPPRESS) + jobcontrol = parser.add_argument_group("Job control options") + jobcontrol.add_argument( + '-j', + '--jobs', + metavar='N', + type=numeric(int, 0, 256), + help="Use up to N CPU cores simultaneously (default: use all).", + ) + jobcontrol.add_argument( + '-q', '--quiet', action='store_true', help="Suppress INFO messages" + ) + jobcontrol.add_argument( + '-v', + '--verbose', + type=numeric(int, 0, 2), + default=0, + const=1, + nargs='?', + help="Print more verbose messages for each additional verbose level. Use " + "`-v 1` typically for much more detailed logging. Higher numbers " + "are probably only useful in debugging.", + ) + jobcontrol.add_argument( + '--no-progress-bar', + action='store_false', + dest='progress_bar', + help=argparse.SUPPRESS, + ) + jobcontrol.add_argument( + '--use-threads', action='store_true', help=argparse.SUPPRESS + ) -metadata = parser.add_argument_group( - "Metadata options", - "Set output PDF/A metadata (default: copy input document's metadata)", -) -metadata.add_argument( - '--title', type=str, help="Set document title (place multiple words in quotes)" -) -metadata.add_argument('--author', type=str, help="Set document author") -metadata.add_argument('--subject', type=str, help="Set document subject description") -metadata.add_argument('--keywords', type=str, help="Set document keywords") + metadata = parser.add_argument_group( + "Metadata options", + "Set output PDF/A metadata (default: copy input document's metadata)", + ) + metadata.add_argument( + '--title', type=str, help="Set document title (place multiple words in quotes)" + ) + metadata.add_argument('--author', type=str, help="Set document author") + metadata.add_argument( + '--subject', type=str, help="Set document subject description" + ) + metadata.add_argument('--keywords', type=str, help="Set document keywords") -preprocessing = parser.add_argument_group( - "Image preprocessing options", - "Options to improve the quality of the final PDF and OCR", -) -preprocessing.add_argument( - '-r', - '--rotate-pages', - action='store_true', - help="Automatically rotate pages based on detected text orientation", -) -preprocessing.add_argument( - '--remove-background', - action='store_true', - help="Attempt to remove background from gray or color pages, setting it " - "to white ", -) -preprocessing.add_argument( - '-d', '--deskew', action='store_true', help="Deskew each page before performing OCR" -) -preprocessing.add_argument( - '-c', - '--clean', - action='store_true', - help="Clean pages from scanning artifacts before performing OCR, and send " - "the cleaned page to OCR, but do not include the cleaned page in " - "the output", -) -preprocessing.add_argument( - '-i', - '--clean-final', - action='store_true', - help="Clean page as above, and incorporate the cleaned image in the final " - "PDF. Might remove desired content.", -) -preprocessing.add_argument( - '--unpaper-args', - type=str, - default=None, - help="A quoted string of arguments to pass to unpaper. Requires --clean. " - "Example: --unpaper-args '--layout double'.", -) -preprocessing.add_argument( - '--oversample', - metavar='DPI', - type=numeric(int, 0, 5000), - default=0, - help="Oversample images to at least the specified DPI, to improve OCR " - "results slightly", -) -preprocessing.add_argument( - '--remove-vectors', - action='store_true', - help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they " - "will not be included in OCR. This can eliminate false characters.", -) -preprocessing.add_argument( - '--threshold', - action='store_true', - help="EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract for OCR. Can " - "improve OCR quality compared to Tesseract's thresholder.", -) + preprocessing = parser.add_argument_group( + "Image preprocessing options", + "Options to improve the quality of the final PDF and OCR", + ) + preprocessing.add_argument( + '-r', + '--rotate-pages', + action='store_true', + help="Automatically rotate pages based on detected text orientation", + ) + preprocessing.add_argument( + '--remove-background', + action='store_true', + help="Attempt to remove background from gray or color pages, setting it " + "to white ", + ) + preprocessing.add_argument( + '-d', + '--deskew', + action='store_true', + help="Deskew each page before performing OCR", + ) + preprocessing.add_argument( + '-c', + '--clean', + action='store_true', + help="Clean pages from scanning artifacts before performing OCR, and send " + "the cleaned page to OCR, but do not include the cleaned page in " + "the output", + ) + preprocessing.add_argument( + '-i', + '--clean-final', + action='store_true', + help="Clean page as above, and incorporate the cleaned image in the final " + "PDF. Might remove desired content.", + ) + preprocessing.add_argument( + '--unpaper-args', + type=str, + default=None, + help="A quoted string of arguments to pass to unpaper. Requires --clean. " + "Example: --unpaper-args '--layout double'.", + ) + preprocessing.add_argument( + '--oversample', + metavar='DPI', + type=numeric(int, 0, 5000), + default=0, + help="Oversample images to at least the specified DPI, to improve OCR " + "results slightly", + ) + preprocessing.add_argument( + '--remove-vectors', + action='store_true', + help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they " + "will not be included in OCR. This can eliminate false characters.", + ) + preprocessing.add_argument( + '--threshold', + action='store_true', + help=( + "EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract " + "for OCR. Can improve OCR quality compared to Tesseract's thresholder." + ), + ) -ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied") -ocrsettings.add_argument( - '-f', - '--force-ocr', - action='store_true', - help="Rasterize any text or vector objects on each page, apply OCR, and " - "save the rastered output (this rewrites the PDF)", -) -ocrsettings.add_argument( - '-s', - '--skip-text', - action='store_true', - help="Skip OCR on any pages that already contain text, but include the " - "page in final output; useful for PDFs that contain a mix of " - "images, text pages, and/or previously OCRed pages", -) -ocrsettings.add_argument( - '--redo-ocr', - action='store_true', - help="Attempt to detect and remove the hidden OCR layer from files that " - "were previously OCRed with OCRmyPDF or another program. Apply OCR " - "to text found in raster images. Existing visible text objects will " - "not be changed. If there is no existing OCR, OCR will be added.", -) -ocrsettings.add_argument( - '--skip-big', - type=numeric(float, 0, 5000), - metavar='MPixels', - help="Skip OCR on pages larger than the specified amount of megapixels, " - "but include skipped pages in final output", -) + ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied") + ocrsettings.add_argument( + '-f', + '--force-ocr', + action='store_true', + help="Rasterize any text or vector objects on each page, apply OCR, and " + "save the rastered output (this rewrites the PDF)", + ) + ocrsettings.add_argument( + '-s', + '--skip-text', + action='store_true', + help="Skip OCR on any pages that already contain text, but include the " + "page in final output; useful for PDFs that contain a mix of " + "images, text pages, and/or previously OCRed pages", + ) + ocrsettings.add_argument( + '--redo-ocr', + action='store_true', + help="Attempt to detect and remove the hidden OCR layer from files that " + "were previously OCRed with OCRmyPDF or another program. Apply OCR " + "to text found in raster images. Existing visible text objects will " + "not be changed. If there is no existing OCR, OCR will be added.", + ) + ocrsettings.add_argument( + '--skip-big', + type=numeric(float, 0, 5000), + metavar='MPixels', + help="Skip OCR on pages larger than the specified amount of megapixels, " + "but include skipped pages in final output", + ) -optimizing = parser.add_argument_group( - "Optimization options", "Control how the PDF is optimized after OCR" -) -optimizing.add_argument( - '-O', - '--optimize', - type=int, - choices=range(0, 4), - default=1, - help=( - "Control how PDF is optimized after processing:" - "0 - do not optimize; " - "1 - do safe, lossless optimizations (default); " - "2 - do some lossy optimizations; " - "3 - do aggressive lossy optimizations (including lossy JBIG2)" - ), -) -optimizing.add_argument( - '--jpeg-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - help=( - "Adjust JPEG quality level for JPEG optimization. " - "100 is best quality and largest output size; " - "1 is lowest quality and smallest output; " - "0 uses the default." - ), -) -optimizing.add_argument( - '--jpg-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - dest='jpeg_quality', - help=argparse.SUPPRESS, # Alias for --jpeg-quality -) -optimizing.add_argument( - '--png-quality', - type=numeric(int, 0, 100), - default=0, - metavar='Q', - help=( - "Adjust PNG quality level to use when quantizing PNGs. " - "Values have same meaning as with --jpeg-quality" - ), -) -optimizing.add_argument( - '--jbig2-lossy', - action='store_true', - help=( - "Enable JBIG2 lossy mode (better compression, not suitable for some " - "use cases - see documentation)." - ), -) -optimizing.add_argument( - '--jbig2-page-group-size', - type=numeric(int, 1, 10000), - default=0, - metavar='N', - # Adjust number of pages to consider at once for JBIG2 compression - help=argparse.SUPPRESS, -) + optimizing = parser.add_argument_group( + "Optimization options", "Control how the PDF is optimized after OCR" + ) + optimizing.add_argument( + '-O', + '--optimize', + type=int, + choices=range(0, 4), + default=1, + help=( + "Control how PDF is optimized after processing:" + "0 - do not optimize; " + "1 - do safe, lossless optimizations (default); " + "2 - do some lossy optimizations; " + "3 - do aggressive lossy optimizations (including lossy JBIG2)" + ), + ) + optimizing.add_argument( + '--jpeg-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + help=( + "Adjust JPEG quality level for JPEG optimization. " + "100 is best quality and largest output size; " + "1 is lowest quality and smallest output; " + "0 uses the default." + ), + ) + optimizing.add_argument( + '--jpg-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + dest='jpeg_quality', + help=argparse.SUPPRESS, # Alias for --jpeg-quality + ) + optimizing.add_argument( + '--png-quality', + type=numeric(int, 0, 100), + default=0, + metavar='Q', + help=( + "Adjust PNG quality level to use when quantizing PNGs. " + "Values have same meaning as with --jpeg-quality" + ), + ) + optimizing.add_argument( + '--jbig2-lossy', + action='store_true', + help=( + "Enable JBIG2 lossy mode (better compression, not suitable for some " + "use cases - see documentation)." + ), + ) + optimizing.add_argument( + '--jbig2-page-group-size', + type=numeric(int, 1, 10000), + default=0, + metavar='N', + # Adjust number of pages to consider at once for JBIG2 compression + help=argparse.SUPPRESS, + ) -advanced = parser.add_argument_group( - "Advanced", "Advanced options to control Tesseract's OCR behavior" -) -advanced.add_argument( - '--pages', - type=str, - help="Limit OCR to the specified pages (ranges or comma separated), skipping others", -) -advanced.add_argument( - '--max-image-mpixels', - action='store', - type=numeric(float, 0), - metavar='MPixels', - help="Set maximum number of pixels to unpack before treating an image as a " - "decompression bomb", - default=128.0, -) -advanced.add_argument( - '--tesseract-config', - action='append', - metavar='CFG', - default=[], - help="Additional Tesseract configuration files -- see documentation", -) -advanced.add_argument( - '--tesseract-pagesegmode', - action='store', - type=int, - metavar='PSM', - choices=range(0, 14), - help="Set Tesseract page segmentation mode (see tesseract --help)", -) -advanced.add_argument( - '--tesseract-oem', - action='store', - type=int, - metavar='MODE', - choices=range(0, 4), - help=( - "Set Tesseract 4.0 OCR engine mode: " - "0 - original Tesseract only; " - "1 - neural nets LSTM only; " - "2 - Tesseract + LSTM; " - "3 - default." - ), -) -advanced.add_argument( - '--pdf-renderer', - choices=['auto', 'hocr', 'sandwich'], - default='auto', - help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " - "choose. See documentation for discussion.", -) -advanced.add_argument( - '--tesseract-timeout', - default=180.0, - type=numeric(float, 0), - metavar='SECONDS', - help='Give up on OCR after the timeout, but copy the preprocessed page ' - 'into the final output', -) -advanced.add_argument( - '--rotate-pages-threshold', - default=14.0, - type=numeric(float, 0, 1000), - metavar='CONFIDENCE', - help="Only rotate pages when confidence is above this value (arbitrary " - "units reported by tesseract)", -) -advanced.add_argument( - '--pdfa-image-compression', - choices=['auto', 'jpeg', 'lossless'], - default='auto', - help="Specify how to compress images in the output PDF/A. 'auto' lets " - "OCRmyPDF decide. 'jpeg' changes all grayscale and color images to " - "JPEG compression. 'lossless' uses PNG-style lossless compression " - "for all images. Monochrome images are always compressed using a " - "lossless codec. Compression settings " - "are applied to all pages, including those for which OCR was " - "skipped. Not supported for --output-type=pdf ; that setting " - "preserves the original compression of all images.", -) -advanced.add_argument( - '--user-words', - metavar='FILE', - help="Specify the location of the Tesseract user words file. This is a " - "list of words Tesseract should consider while performing OCR in " - "addition to its standard language dictionaries. This can improve " - "OCR quality especially for specialized and technical documents.", -) -advanced.add_argument( - '--user-patterns', - metavar='FILE', - help="Specify the location of the Tesseract user patterns file.", -) -advanced.add_argument( - '--fast-web-view', - type=numeric(float, 0), - default=1.0, - metavar="MEGABYTES", - help="If the size of file is more than this threshold (in MB), then " - "linearize the PDF for fast web viewing. This allows the PDF to be " - "displayed before it is fully downloaded in web browsers, but increases " - "the space required slightly. By default we skip this for small files " - "which do not benefit. If the threshold is 0 it will be apply to all files. " - "Set the threshold very high to disable.", -) -advanced.add_argument( - '--plugins', - action='append', - default=[], - help="Path to a folder than contains plugins.", -) + advanced = parser.add_argument_group( + "Advanced", "Advanced options to control Tesseract's OCR behavior" + ) + advanced.add_argument( + '--pages', + type=str, + help=( + "Limit OCR to the specified pages (ranges or comma separated), " + "skipping others", + ), + ) + advanced.add_argument( + '--max-image-mpixels', + action='store', + type=numeric(float, 0), + metavar='MPixels', + help="Set maximum number of pixels to unpack before treating an image as a " + "decompression bomb", + default=128.0, + ) + advanced.add_argument( + '--tesseract-config', + action='append', + metavar='CFG', + default=[], + help="Additional Tesseract configuration files -- see documentation", + ) + advanced.add_argument( + '--tesseract-pagesegmode', + action='store', + type=int, + metavar='PSM', + choices=range(0, 14), + help="Set Tesseract page segmentation mode (see tesseract --help)", + ) + advanced.add_argument( + '--tesseract-oem', + action='store', + type=int, + metavar='MODE', + choices=range(0, 4), + help=( + "Set Tesseract 4.0 OCR engine mode: " + "0 - original Tesseract only; " + "1 - neural nets LSTM only; " + "2 - Tesseract + LSTM; " + "3 - default." + ), + ) + advanced.add_argument( + '--pdf-renderer', + choices=['auto', 'hocr', 'sandwich'], + default='auto', + help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " + "choose. See documentation for discussion.", + ) + advanced.add_argument( + '--tesseract-timeout', + default=180.0, + type=numeric(float, 0), + metavar='SECONDS', + help='Give up on OCR after the timeout, but copy the preprocessed page ' + 'into the final output', + ) + advanced.add_argument( + '--rotate-pages-threshold', + default=14.0, + type=numeric(float, 0, 1000), + metavar='CONFIDENCE', + help="Only rotate pages when confidence is above this value (arbitrary " + "units reported by tesseract)", + ) + advanced.add_argument( + '--pdfa-image-compression', + choices=['auto', 'jpeg', 'lossless'], + default='auto', + help="Specify how to compress images in the output PDF/A. 'auto' lets " + "OCRmyPDF decide. 'jpeg' changes all grayscale and color images to " + "JPEG compression. 'lossless' uses PNG-style lossless compression " + "for all images. Monochrome images are always compressed using a " + "lossless codec. Compression settings " + "are applied to all pages, including those for which OCR was " + "skipped. Not supported for --output-type=pdf ; that setting " + "preserves the original compression of all images.", + ) + advanced.add_argument( + '--user-words', + metavar='FILE', + help="Specify the location of the Tesseract user words file. This is a " + "list of words Tesseract should consider while performing OCR in " + "addition to its standard language dictionaries. This can improve " + "OCR quality especially for specialized and technical documents.", + ) + advanced.add_argument( + '--user-patterns', + metavar='FILE', + help="Specify the location of the Tesseract user patterns file.", + ) + advanced.add_argument( + '--fast-web-view', + type=numeric(float, 0), + default=1.0, + metavar="MEGABYTES", + help="If the size of file is more than this threshold (in MB), then " + "linearize the PDF for fast web viewing. This allows the PDF to be " + "displayed before it is fully downloaded in web browsers, but increases " + "the space required slightly. By default we skip this for small files " + "which do not benefit. If the threshold is 0 it will be apply to all files. " + "Set the threshold very high to disable.", + ) + advanced.add_argument( + '--plugins', + action='append', + default=[], + help="Path to a folder than contains plugins.", + ) + + debugging = parser.add_argument_group( + "Debugging", "Arguments to help with troubleshooting and debugging" + ) + debugging.add_argument( + '-k', + '--keep-temporary-files', + action='store_true', + help="Keep temporary files (helpful for debugging)", + ) + debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) + return parser -debugging = parser.add_argument_group( - "Debugging", "Arguments to help with troubleshooting and debugging" -) -debugging.add_argument( - '-k', - '--keep-temporary-files', - action='store_true', - help="Keep temporary files (helpful for debugging)", -) -debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) plugins_only_parser = ArgumentParser( prog=_PROGRAM_NAME, fromfile_prefix_chars='@', add_help=False, allow_abbrev=False diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index e6f0bd3f..9459b70e 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -582,7 +582,7 @@ def main(infile, outfile, level, jobs=1): ) with TemporaryDirectory() as td: - context = PDFContext(options, td, infile, None) + context = PDFContext(options, td, infile, None, None) tmpout = Path(td) / 'out.pdf' optimize( infile, diff --git a/tests/conftest.py b/tests/conftest.py index 8724ead5..54960cdc 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -214,7 +214,7 @@ def no_outpdf(tmp_path): def check_ocrmypdf(input_file, output_file, *args, env=None): """Run ocrmypdf and confirmed that a valid file was created""" - options = cli.parser.parse_args( + options = cli.get_parser().parse_args( [str(input_file), str(output_file)] + [str(arg) for arg in args if arg is not None] ) @@ -222,7 +222,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): if env: options.tesseract_env = env options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - result = api.run_pipeline(options, api=True) + result = api.run_pipeline(options, plugin_manager=None, api=True) assert result == 0 assert os.path.exists(str(output_file)), "Output file not created" @@ -238,7 +238,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): Does not currently have a way to manipulate the PATH except for Tesseract. """ - options = cli.parser.parse_args( + options = cli.get_parser().parse_args( [str(input_file), str(output_file)] + [str(arg) for arg in args if arg is not None] ) @@ -253,7 +253,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) - return api.run_pipeline(options, api=False) + return api.run_pipeline(options, plugin_manager=None, api=False) @pytest.helpers.register diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 270a8b62..79e63604 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -32,7 +32,7 @@ from pikepdf.models.metadata import decode_pdf_date from ocrmypdf._jobcontext import PDFContext from ocrmypdf._pipeline import convert_to_pdfa -from ocrmypdf.cli import parser +from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps from ocrmypdf.pdfinfo import PdfInfo @@ -290,16 +290,15 @@ def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): def test_metadata_fixup_warning(resources, outdir, caplog): - from ocrmypdf.__main__ import parser from ocrmypdf._pipeline import metadata_fixup - options = parser.parse_args( + options = get_parser().parse_args( args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf'] ) copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') - context = PDFContext(options, outdir, outdir / 'graph.pdf', None) + context = PDFContext(options, outdir, outdir / 'graph.pdf', None, None) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) for record in caplog.records: assert record.levelname != 'WARNING' @@ -310,7 +309,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog): meta['prism2:publicationName'] = 'OCRmyPDF Test' graph.save(outdir / 'graph_mod.pdf') - context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None) + context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None, None) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) assert any(record.levelname == 'WARNING' for record in caplog.records) @@ -326,11 +325,11 @@ def test_prevent_gs_invalid_xml(resources, outdir): Title=b'String with trailing nul\x00' ) - options = parser.parse_args( + options = get_parser().parse_args( args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo) + context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context @@ -357,11 +356,11 @@ def test_malformed_docinfo(caplog, resources, outdir): pike.trailer.Info = pikepdf.Stream(pike, b"") pike.save(outdir / 'layers.rendered.pdf', fix_metadata_version=False) - options = parser.parse_args( + options = get_parser().parse_args( args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo) + context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index f0e90b80..7753d340 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -21,7 +21,7 @@ from unittest.mock import patch import pytest from ocrmypdf._validation import check_options -from ocrmypdf.cli import parser +from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError from ocrmypdf.exec import unpaper @@ -51,7 +51,7 @@ def spoof_unpaper_oldversion(tmp_path_factory): def test_no_unpaper(resources, no_outpdf): input_ = fspath(resources / "c02-22.pdf") output = fspath(no_outpdf) - options = parser.parse_args(args=["--clean", input_, output]) + options = get_parser().parse_args(args=["--clean", input_, output]) with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") diff --git a/tests/test_validation.py b/tests/test_validation.py index f183aa46..1a453e78 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -23,6 +23,7 @@ import pytest import ocrmypdf._validation as vd from ocrmypdf.api import create_options +from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.pdfinfo import PdfInfo @@ -30,7 +31,9 @@ from ocrmypdf.pdfinfo import PdfInfo def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): if language is not None: kwargs['language'] = language - return create_options(input_file=input_file, output_file=output_file, **kwargs) + return create_options( + input_file=input_file, output_file=output_file, parser=get_parser(), **kwargs + ) def test_hocr_notlatin_warning(caplog): From 5dbc080fa034702a076db37b4d60091cb71dfedf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 2 May 2020 04:32:46 -0700 Subject: [PATCH 440/880] Rename PDFContext->PdfContext --- src/ocrmypdf/_jobcontext.py | 2 +- src/ocrmypdf/_sync.py | 4 ++-- src/ocrmypdf/optimize.py | 4 ++-- tests/test_metadata.py | 10 +++++----- 4 files changed, 10 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 7158abe8..938c908b 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -24,7 +24,7 @@ from functools import partial from ocrmypdf._plugin_manager import get_plugin_manager -class PDFContext: +class PdfContext: """Holds our context for a particular run of the pipeline""" def __init__(self, options, work_folder, origin, pdfinfo, plugin_manager): diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 0834993d..a6bc3093 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -32,7 +32,7 @@ import pluggy from ocrmypdf import pluginspec from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._graft import OcrGrafter -from ocrmypdf._jobcontext import PDFContext, cleanup_working_files +from ocrmypdf._jobcontext import PdfContext, cleanup_working_files from ocrmypdf._logging import PageNumberFilter from ocrmypdf._pipeline import ( convert_to_pdfa, @@ -330,7 +330,7 @@ def run_pipeline(options, *, plugin_manager, api=False): max_workers=options.jobs if not options.use_threads else 1, # To help debug ) - context = PDFContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager) + context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager) # Validate options are okay for this pdf validate_pdfinfo_options(context) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 9459b70e..e4fc1048 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -29,7 +29,7 @@ from PIL import Image from tqdm import tqdm from . import leptonica -from ._jobcontext import PDFContext +from ._jobcontext import PdfContext from .exceptions import OutputFileAccessError from .exec import jbig2enc, pngquant from .helpers import safe_symlink @@ -582,7 +582,7 @@ def main(infile, outfile, level, jobs=1): ) with TemporaryDirectory() as td: - context = PDFContext(options, td, infile, None, None) + context = PdfContext(options, td, infile, None, None) tmpout = Path(td) / 'out.pdf' optimize( infile, diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 79e63604..102c9f28 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -30,7 +30,7 @@ import pikepdf import pytest from pikepdf.models.metadata import decode_pdf_date -from ocrmypdf._jobcontext import PDFContext +from ocrmypdf._jobcontext import PdfContext from ocrmypdf._pipeline import convert_to_pdfa from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode @@ -298,7 +298,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog): copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') - context = PDFContext(options, outdir, outdir / 'graph.pdf', None, None) + context = PdfContext(options, outdir, outdir / 'graph.pdf', None, None) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) for record in caplog.records: assert record.levelname != 'WARNING' @@ -309,7 +309,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog): meta['prism2:publicationName'] = 'OCRmyPDF Test' graph.save(outdir / 'graph_mod.pdf') - context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None, None) + context = PdfContext(options, outdir, outdir / 'graph_mod.pdf', None, None) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) assert any(record.levelname == 'WARNING' for record in caplog.records) @@ -329,7 +329,7 @@ def test_prevent_gs_invalid_xml(resources, outdir): args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) + context = PdfContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context @@ -360,7 +360,7 @@ def test_malformed_docinfo(caplog, resources, outdir): args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PDFContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) + context = PdfContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context From c85278b31d546d85f425803df761b258502b2183 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 May 2020 00:51:17 -0700 Subject: [PATCH 441/880] Delinting --- src/ocrmypdf/__main__.py | 2 +- src/ocrmypdf/_jobcontext.py | 1 - src/ocrmypdf/_pipeline.py | 42 +++++++++++++++------------------- src/ocrmypdf/_sync.py | 6 +---- src/ocrmypdf/_validation.py | 3 +-- src/ocrmypdf/api.py | 3 +-- src/ocrmypdf/exec/tesseract.py | 21 ++++++++--------- src/ocrmypdf/helpers.py | 2 +- src/ocrmypdf/leptonica.py | 18 ++++----------- src/ocrmypdf/optimize.py | 14 ++++++------ src/ocrmypdf/pdfinfo/info.py | 6 ++--- tests/conftest.py | 18 ++++++++------- tests/test_main.py | 21 ++++++++--------- tests/test_metadata.py | 16 ++++--------- tests/test_optimize.py | 1 - tests/test_tess4.py | 8 +++---- tests/test_unpaper.py | 13 ++--------- tests/test_validation.py | 4 ++-- 18 files changed, 79 insertions(+), 120 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index b6dc0466..411948e2 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -56,7 +56,7 @@ def run(args=None): configure_logging( verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True ) - log.debug('ocrmypdf ' + __version__) + log.debug('ocrmypdf %s', __version__) try: check_options(options) except ValueError as e: diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 938c908b..c5ffa5b1 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import logging import os import shutil import sys diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index cd82fea4..196a8d70 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -20,30 +20,29 @@ import os import re import sys from datetime import datetime, timezone -from pathlib import Path from shutil import copyfileobj import img2pdf import pikepdf from pikepdf.models.metadata import encode_pdf_date -from PIL import Image +from PIL import Image, ImageColor, ImageDraw -from . import leptonica -from ._version import PROGRAM_NAME -from ._version import __version__ as VERSION -from .exceptions import ( +from ocrmypdf import leptonica +from ocrmypdf._version import PROGRAM_NAME +from ocrmypdf._version import __version__ as VERSION +from ocrmypdf.exceptions import ( DpiError, EncryptedPdfError, InputFileError, PriorOcrFoundError, UnsupportedImageFormatError, ) -from .exec import ghostscript, tesseract -from .helpers import Resolution, safe_symlink -from .hocrtransform import HocrTransform -from .optimize import optimize -from .pdfa import generate_pdfa_ps -from .pdfinfo import Colorspace, Encoding, PdfInfo +from ocrmypdf.exec import ghostscript, tesseract, unpaper +from ocrmypdf.helpers import Resolution, safe_symlink +from ocrmypdf.hocrtransform import HocrTransform +from ocrmypdf.optimize import optimize +from ocrmypdf.pdfa import generate_pdfa_ps +from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo log = logging.getLogger(__name__) @@ -63,8 +62,8 @@ def triage_image_file(input_file, output_file, options): log.info("Input file is an image") if 'dpi' in im.info: if im.info['dpi'] <= (96, 96) and not options.image_dpi: - log.info("Image size: (%d, %d)" % im.size) - log.info("Image resolution: (%d, %d)" % im.info['dpi']) + log.info("Image size: (%d, %d)", *im.size) + log.info("Image resolution: (%d, %d)", *im.info['dpi']) log.error( "Input file is an image, but the resolution (DPI) is " "not credible. Estimate the resolution at which the " @@ -72,7 +71,7 @@ def triage_image_file(input_file, output_file, options): ) raise DpiError() elif not options.image_dpi: - log.info("Image size: (%d, %d)" % im.size) + log.info("Image size: (%d, %d)", *im.size) log.error( "Input file is an image, but has no resolution (DPI) " "in its metadata. Estimate the resolution at which " @@ -261,7 +260,7 @@ def is_ocr_required(page_context): ocr_required = True elif options.redo_ocr: if pageinfo.has_corrupt_text: - log.warn( + log.warning( "some text on this page cannot be mapped to characters: " "consider using --force-ocr instead" ) @@ -288,7 +287,7 @@ def is_ocr_required(page_context): ) elif options.force_ocr: # Warn the user they might not want to do this - log.warn( + log.warning( "page has no images - " "all vector content will be " f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely " @@ -308,7 +307,7 @@ def is_ocr_required(page_context): pixel_count = pageinfo.width_pixels * pageinfo.height_pixels if pixel_count > (options.skip_big * 1_000_000): ocr_required = False - log.warn( + log.warning( "page too big, skipping OCR " f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)" ) @@ -464,8 +463,6 @@ def preprocess_deskew(input_file, page_context): def preprocess_clean(input_file, page_context): - from .exec import unpaper - output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args) @@ -480,9 +477,6 @@ def create_ocr_image(image, page_context): output_file = page_context.get_path('ocr.png') options = page_context.options with Image.open(image) as im: - from PIL import ImageColor - from PIL import ImageDraw - white = ImageColor.getcolor('#ffffff', im.mode) # pink = ImageColor.getcolor('#ff0080', im.mode) draw = ImageDraw.ImageDraw(im) @@ -811,7 +805,7 @@ def merge_sidecars(txt_files, context): return output_file -def copy_final(input_file, output_file, context): +def copy_final(input_file, output_file, _context): log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a6bc3093..1d256c92 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import importlib import logging import logging.handlers import os @@ -27,13 +26,10 @@ from pathlib import Path from tempfile import mkdtemp import PIL -import pluggy -from ocrmypdf import pluginspec from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._graft import OcrGrafter from ocrmypdf._jobcontext import PdfContext, cleanup_working_files -from ocrmypdf._logging import PageNumberFilter from ocrmypdf._pipeline import ( convert_to_pdfa, copy_final, @@ -372,7 +368,7 @@ def run_pipeline(options, *, plugin_manager, api=False): else: log.error(type(e).__name__) return e.exit_code - except (Exception if not api else NeverRaise) as e: + except (Exception if not api else NeverRaise) as e: # pylint: disable=broad-except log.exception("An exception occurred while executing the pipeline") return ExitCode.other_error finally: diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index bfc25d02..0a755d3c 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -21,6 +21,7 @@ import locale import logging import os import sys +import unicodedata from pathlib import Path from shutil import copyfileobj @@ -284,8 +285,6 @@ def check_options_advanced(options): def check_options_metadata(options): - import unicodedata - docinfo = [options.title, options.author, options.keywords, options.subject] for s in (m for m in docinfo if m): for c in s: diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index cc7e103f..16e151b2 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -19,7 +19,6 @@ import logging import os import sys from argparse import ArgumentParser -from contextlib import suppress from enum import IntEnum from pathlib import Path from typing import Dict, Iterable @@ -28,7 +27,7 @@ from ocrmypdf._logging import PageNumberFilter, TqdmConsole from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_options -from ocrmypdf.cli import get_parser, plugins_only_parser +from ocrmypdf.cli import get_parser try: import coloredlogs diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 4ebdb0cf..731746e4 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -25,13 +25,15 @@ from contextlib import suppress from os import fspath from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired -from ..exceptions import ( +from PIL import Image + +from ocrmypdf.exceptions import ( MissingDependencyError, SubprocessOutputError, TesseractConfigError, ) -from ..helpers import page_number, safe_symlink -from . import get_version, run +from ocrmypdf.exec import get_version, run +from ocrmypdf.helpers import safe_symlink log = logging.getLogger(__name__) @@ -133,7 +135,7 @@ def languages(tesseract_env=None): for line in output.splitlines(): if line.startswith('Error'): raise MissingDependencyError(lang_error(output)) - header, *rest = output.splitlines() + _header, *rest = output.splitlines() return set(lang.strip() for lang in rest) @@ -227,18 +229,15 @@ def tesseract_log_output(stdout, input_file): tlog.info(line.strip()) -def page_timedout(input_file, timeout): +def page_timedout(timeout): if timeout == 0: return - prefix = f"{(page_number(input_file)):4d}: [tesseract] " - log.warning(prefix + " took too long to OCR - skipping") + log.warning("[tesseract] took too long to OCR - skipping") def _generate_null_hocr(output_hocr, output_sidecar, image): """Produce a .hocr file that reports no text detected on a page that is the same size as the input image.""" - from PIL import Image - with Image.open(image) as im: w, h = im.size @@ -293,7 +292,7 @@ def generate_hocr( # Generate a HOCR file with no recognized text if tesseract times out # Temporary workaround to hocrTransform not being able to function if # it does not have a valid hOCR file. - page_timedout(input_file, timeout) + page_timedout(timeout) _generate_null_hocr(output_hocr, output_sidecar, input_file) except CalledProcessError as e: tesseract_log_output(e.output, input_file) @@ -389,7 +388,7 @@ def generate_pdf( if os.path.exists(prefix + '.txt'): shutil.move(prefix + '.txt', output_text) except TimeoutExpired: - page_timedout(input_image, timeout) + page_timedout(timeout) use_skip_page(text_only, skip_pdf, output_pdf, output_text) except CalledProcessError as e: tesseract_log_output(e.output, input_image) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 0bfe694a..46a9ea80 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -25,7 +25,7 @@ from collections.abc import Iterable from contextlib import suppress from functools import wraps from io import StringIO -from math import inf, isclose +from math import isclose from pathlib import Path import pikepdf diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 328b0630..0a8f6cba 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -29,7 +29,7 @@ from collections.abc import Sequence from contextlib import suppress from ctypes.util import find_library from functools import lru_cache -from io import BytesIO +from io import BytesIO, UnsupportedOperation from os import fspath from tempfile import TemporaryFile @@ -96,7 +96,6 @@ class _LeptonicaErrorTrap: self.no_stderr = False def __enter__(self): - from io import UnsupportedOperation self.tmpfile = TemporaryFile() @@ -351,7 +350,7 @@ class Pix(LeptonicaObject): py_file.write(buffer) @classmethod - def frompil(self, pillow_image): + def frompil(cls, pillow_image): """Create a copy of a PIL.Image from this Pix""" bio = BytesIO() pillow_image.save(bio, format='png', compress_level=1) @@ -363,7 +362,7 @@ class Pix(LeptonicaObject): def topil(self): """Returns a PIL.Image version of this Pix""" - from PIL import Image + from PIL import Image # pylint: disable=import-outside-toplevel # Leptonica manages data in words, so it implicitly does an endian # swap. Tell Pillow about this when it reads the data. @@ -534,16 +533,7 @@ class Pix(LeptonicaObject): ) return Pix(thresh_pix) - def crop_to_foreground( - self, - threshold=128, - mindist=70, - erasedist=30, - pagenum=0, - showmorph=0, - display=0, - pdfdir=ffi.NULL, - ): + def crop_to_foreground(self, threshold=128, mindist=70, erasedist=30, showmorph=0): if get_leptonica_version() < 'leptonica-1.76': # Leptonica 1.76 changed the API for pixFindPageForeground; we don't # support the old version diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index e4fc1048..7438c180 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -217,7 +217,7 @@ def extract_images(pike, root, options, extract_fn): result = extract_fn( pike=pike, root=root, image=image, xref=xref, options=options ) - except Exception as e: + except Exception as e: # pylint: disable=broad-except log.debug("Image xref %s, error %s", xref, repr(e)) errors += 1 else: @@ -422,12 +422,12 @@ def transcode_pngs(pike, images, image_name_fn, root, options): ) continue if compdata.type == leptonica.lept.L_FLATE_ENCODE: - return rewrite_png(pike, im_obj, compdata, log) + return rewrite_png(pike, im_obj, compdata) elif compdata.type == leptonica.lept.L_G4_ENCODE: - return rewrite_png_as_g4(pike, im_obj, compdata, log) + return rewrite_png_as_g4(pike, im_obj, compdata) -def rewrite_png_as_g4(pike, im_obj, compdata, log): +def rewrite_png_as_g4(pike, im_obj, compdata): im_obj.BitsPerComponent = 1 im_obj.Width = compdata.w im_obj.Height = compdata.h @@ -447,7 +447,7 @@ def rewrite_png_as_g4(pike, im_obj, compdata, log): return -def rewrite_png(pike, im_obj, compdata, log): +def rewrite_png(pike, im_obj, compdata): # When a PNG is inserted into a PDF, we more or less copy the IDAT section from # the PDF and transfer the rest of the PNG headers to PDF image metadata. # One thing we have to do is tell the PDF reader whether a predictor was used @@ -553,8 +553,8 @@ def optimize(input_file, output_file, context, save_settings): def main(infile, outfile, level, jobs=1): - from tempfile import TemporaryDirectory - from shutil import copy + from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel + from shutil import copy # pylint: disable=import-outside-toplevel class OptimizeOptions: """Emulate ocrmypdf's options""" diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 86bb059d..42ed372f 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -17,14 +17,13 @@ # along with OCRmyPDF. If not, see . import logging -import os import re from collections import defaultdict, namedtuple from decimal import Decimal from enum import Enum from functools import partial from math import hypot, isclose -from os import PathLike, fspath +from os import PathLike from pathlib import Path from warnings import warn @@ -821,13 +820,14 @@ class PdfInfo: def main(): + # pylint: disable=import-outside-toplevel import argparse + from pprint import pprint parser = argparse.ArgumentParser() parser.add_argument('infile') args = parser.parse_args() pagesinfo, pdfinfo = _pdf_get_all_pageinfo(args.infile) - from pprint import pprint pprint(pdfinfo) for page in pagesinfo: diff --git a/tests/conftest.py b/tests/conftest.py index 54960cdc..d1cc9ef1 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -24,7 +24,8 @@ from subprocess import PIPE, run import pytest -from ocrmypdf import api, cli +from ocrmypdf import api, cli, pdfinfo +from ocrmypdf.exec import unpaper pytest_plugins = ['helpers_namespace'] @@ -62,10 +63,8 @@ def running_in_travis(): @pytest.helpers.register def have_unpaper(): try: - from ocrmypdf.exec import unpaper - unpaper.version() - except Exception: + except Exception: # pylint: disable=broad-except return False return True @@ -95,7 +94,7 @@ assert ast.parse(WINDOWS_SHIM_TEMPLATE.format(spoofer=repr(r"C:\\Temp\\file.py") @pytest.helpers.register def spoof(tmp_path_factory, **kwargs): - """Modify PATH to override subprocess executables + r"""Modify PATH to override subprocess executables spoof(tmp_path_factory, program1='replacement', ...) @@ -277,7 +276,12 @@ def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=Tr env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) p = run( - p_args, stdout=PIPE, stderr=PIPE, universal_newlines=universal_newlines, env=env + p_args, + stdout=PIPE, + stderr=PIPE, + universal_newlines=universal_newlines, + env=env, + check=False, ) # print(p.stderr) return p, p.stdout, p.stderr @@ -285,8 +289,6 @@ def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=Tr @pytest.helpers.register def first_page_dimensions(pdf): - from ocrmypdf import pdfinfo - info = pdfinfo.PdfInfo(pdf) page0 = info[0] return (page0.width_inches, page0.height_inches) diff --git a/tests/test_main.py b/tests/test_main.py index e69ed192..cca9c4eb 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -29,8 +29,7 @@ from PIL import Image import ocrmypdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import ghostscript, tesseract -from ocrmypdf.helpers import check_pdf +from ocrmypdf.exec import get_version, ghostscript, tesseract from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo @@ -311,7 +310,7 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) -def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf, caplog): +def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): p, _, err = run_ocrmypdf( resources / 'ccitt.pdf', no_outpdf, @@ -410,7 +409,7 @@ def test_destination_not_writable(spoof_tesseract_noop, resources, outdir): protected_file = outdir / 'protected.pdf' protected_file.touch() protected_file.chmod(0o400) # Read-only - p, out, err = run_ocrmypdf( + p, _out, _err = run_ocrmypdf( resources / 'jbig2.pdf', protected_file, env=spoof_tesseract_noop ) assert p.returncode == ExitCode.file_access_error, "Expected error" @@ -448,7 +447,7 @@ THIS FILE IS INVALID ''' ) - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'ccitt.pdf', outdir / 'out.pdf', '--pdf-renderer', @@ -568,6 +567,7 @@ def test_compression_preserved( stdin=input_stream, universal_newlines=True, env=spoof_tesseract_noop, + check=False, ) if im.mode in ('RGBA', 'LA'): @@ -629,6 +629,7 @@ def test_compression_changed( stdin=input_stream, universal_newlines=True, env=spoof_tesseract_noop, + check=False, ) assert p.returncode == ExitCode.ok, p.stderr @@ -711,10 +712,10 @@ def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf): ) @pytest.mark.slow def test_decompression_bomb(resources, outpdf): - p, out, err = run_ocrmypdf(resources / 'hugemono.pdf', outpdf) + p, _out, err = run_ocrmypdf(resources / 'hugemono.pdf', outpdf) assert 'decompression bomb' in err - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'hugemono.pdf', outpdf, '--max-image-mpixels', '2000' ) assert p.returncode == 0 @@ -736,7 +737,7 @@ def test_text_curves(spoof_tesseract_noop, resources, outpdf): def test_output_is_dir(spoof_tesseract_noop, resources, outdir): - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'trivial.pdf', outdir, '--force-ocr', env=spoof_tesseract_noop ) assert p.returncode == ExitCode.file_access_error @@ -747,7 +748,7 @@ def test_output_is_dir(spoof_tesseract_noop, resources, outdir): def test_output_is_symlink(spoof_tesseract_noop, resources, outdir): sym = Path(outdir / 'this_is_a_symlink') sym.symlink_to(outdir / 'out.pdf') - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'trivial.pdf', sym, '--force-ocr', env=spoof_tesseract_noop ) assert p.returncode == ExitCode.ok, err @@ -761,8 +762,6 @@ def test_livecycle(resources, no_outpdf): def test_version_check(): - from ocrmypdf.exec import get_version - with pytest.raises(MissingDependencyError): get_version('NOT_FOUND_UNLIKELY_ON_PATH') diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 102c9f28..9a92c65a 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -17,21 +17,18 @@ import datetime -import logging import mmap -import os from datetime import timezone from os import fspath -from pathlib import Path -from shutil import copyfile, move -from unittest.mock import MagicMock, patch +from shutil import copyfile +from unittest.mock import patch import pikepdf import pytest from pikepdf.models.metadata import decode_pdf_date from ocrmypdf._jobcontext import PdfContext -from ocrmypdf._pipeline import convert_to_pdfa +from ocrmypdf._pipeline import convert_to_pdfa, metadata_fixup from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps @@ -192,9 +189,8 @@ def test_xml_metadata_preserved(spoof_tesseract_noop, output_type, resources, ou input_file = resources / 'graph.pdf' try: - from libxmp import consts - from libxmp.utils import file_to_dict - except Exception: + from libxmp.utils import file_to_dict # pylint: disable=import-outside-toplevel + except Exception: # pylint: disable=broad-except pytest.skip("libxmp not available or libexempi3 not installed") before = file_to_dict(str(input_file)) @@ -290,8 +286,6 @@ def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): def test_metadata_fixup_warning(resources, outdir, caplog): - from ocrmypdf._pipeline import metadata_fixup - options = get_parser().parse_args( args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf'] ) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 36e00c5e..f9f6747c 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import logging from os import fspath from pathlib import Path diff --git a/tests/test_tess4.py b/tests/test_tess4.py index 33110a43..4fada7d4 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -18,7 +18,6 @@ import logging import os import subprocess -from contextlib import contextmanager from os import fspath from pathlib import Path @@ -60,8 +59,8 @@ def test_skip_pages_does_not_replicate(resources, basename, outdir): for page in info: assert len(page.images) == 1, "skipped page was replicated" - for n in range(len(info_in)): - assert info[n].width_inches == info_in[n].width_inches + for n, info_out_n in enumerate(info): + assert info_out_n.width_inches == info_in[n].width_inches def test_content_preservation(resources, outpdf): @@ -131,8 +130,7 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): def test_timeout(caplog): - tesseract.page_timedout('123456.png', 5) - assert "123456" in caplog.text + tesseract.page_timedout(5) assert "took too long" in caplog.text diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 7753d340..836ef0a8 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -23,24 +23,15 @@ import pytest from ocrmypdf._validation import check_options from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import unpaper # pytest.helpers is dynamic -# pylint: disable=no-member +# pylint: disable=no-member,redefined-outer-name # pylint: disable=w0612 check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof - - -def have_unpaper(): - try: - unpaper.version() - except Exception: - return False - else: - return True +have_unpaper = pytest.helpers.have_unpaper @pytest.fixture diff --git a/tests/test_validation.py b/tests/test_validation.py index 1a453e78..35c21fa0 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -145,9 +145,9 @@ def test_report_file_size(tmp_path, caplog): def test_false_action_store_true(): opts = make_opts(keep_temporary_files=True) - assert opts.keep_temporary_files == True + assert opts.keep_temporary_files opts = make_opts(keep_temporary_files=False) - assert opts.keep_temporary_files == False + assert not opts.keep_temporary_files @pytest.mark.parametrize('progress_bar', [True, False]) From fe4296c53b151a2a408f093b0700c537d82a7dd2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 May 2020 00:53:47 -0700 Subject: [PATCH 442/880] safe_symlink: remove deprecated params --- src/ocrmypdf/helpers.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 46a9ea80..ff4bb13a 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -65,7 +65,7 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))): return f"Resolution({self.x}x{self.y} dpi)" -def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **kwargs): +def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): """ Helper function: relinks soft symbolic link if necessary """ From 75c34b873a46a2561f8c4fb75c9f1010567d0a70 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 May 2020 02:04:57 -0700 Subject: [PATCH 443/880] optimize: convert from executor to progress pool --- src/ocrmypdf/_concurrent.py | 7 ++- src/ocrmypdf/optimize.py | 88 +++++++++++++++++++------------------ 2 files changed, 51 insertions(+), 44 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index a321f28d..cf46bae3 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -46,7 +46,7 @@ def log_listener(queue): break logger = logging.getLogger(record.name) logger.handle(record) - except Exception: + except Exception: # pylint: disable=broad-except import traceback print("Logging problem", file=sys.stderr) @@ -106,7 +106,10 @@ def exec_progress_pool( while True: try: result = results.next() - task_finished(result, pbar) + if task_finished: + task_finished(result, pbar) + else: + pbar.update() except StopIteration: break except KeyboardInterrupt: diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 7438c180..41bf7c60 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -15,11 +15,11 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import concurrent.futures import logging import sys import tempfile from collections import defaultdict +from functools import partial from os import fspath from pathlib import Path @@ -28,11 +28,12 @@ from pikepdf import Dictionary, Name from PIL import Image from tqdm import tqdm -from . import leptonica -from ._jobcontext import PdfContext -from .exceptions import OutputFileAccessError -from .exec import jbig2enc, pngquant -from .helpers import safe_symlink +from ocrmypdf import leptonica +from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._jobcontext import PdfContext +from ocrmypdf.exceptions import OutputFileAccessError +from ocrmypdf.exec import jbig2enc, pngquant +from ocrmypdf.helpers import safe_symlink log = logging.getLogger(__name__) @@ -260,49 +261,49 @@ def extract_images_jbig2(pike, root, options): def _produce_jbig2_images(jbig2_groups, root, options): """Produce JBIG2 images from their groups""" - def jbig2_group_futures(executor, root, groups): + def jbig2_group_args(root, groups): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' - future = executor.submit( - jbig2enc.convert_group, + yield dict( cwd=fspath(root), infiles=(img_name(root, xref, ext) for xref, ext in xref_exts), out_prefix=prefix, ) - yield future - def jbig2_single_futures(executor, root, groups): + def jbig2_single_args(root, groups): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' # Second loop is to ensure multiple images per page are unpacked for n, xref_ext in enumerate(xref_exts): xref, ext = xref_ext - future = executor.submit( - jbig2enc.convert_single, + yield dict( cwd=fspath(root), infile=img_name(root, xref, ext), outfile=root / f'{prefix}.{n:04d}', ) - yield future + + def convert_generic(fn, kwargs_dict): + return fn(**kwargs_dict) if options.jbig2_page_group_size > 1: - jbig2_futures = jbig2_group_futures + jbig2_args = jbig2_group_args + jbig2_convert = partial(convert_generic, jbig2enc.convert_group) else: - jbig2_futures = jbig2_single_futures + jbig2_args = jbig2_single_args + jbig2_convert = partial(convert_generic, jbig2enc.convert_single) - with concurrent.futures.ThreadPoolExecutor(max_workers=options.jobs) as executor: - futures = jbig2_futures(executor, root, jbig2_groups) - with tqdm( + exec_progress_pool( + use_threads=True, + max_workers=options.jobs, + tqdm_kwargs=dict( total=len(jbig2_groups), desc="JBIG2", unit='item', disable=not options.progress_bar, - ) as pbar: - for future in concurrent.futures.as_completed(futures): - proc = future.result() - if proc.stderr: - log.debug(proc.stderr.decode()) - pbar.update() + ), + task=jbig2_convert, + task_arguments=jbig2_args(root, jbig2_groups), + ) def convert_to_jbig2(pike, jbig2_groups, root, options): @@ -373,30 +374,33 @@ def transcode_pngs(pike, images, image_name_fn, root, options): max(10, options.png_quality - 10), min(100, options.png_quality + 10), ) - with concurrent.futures.ThreadPoolExecutor( - max_workers=options.jobs - ) as executor: - futures = [] + + def pngquant_args(): for xref in images: log.debug(image_name_fn(root, xref)) - futures.append( - executor.submit( - pngquant.quantize, - image_name_fn(root, xref), - png_name(root, xref), - png_quality[0], - png_quality[1], - ) + yield ( + image_name_fn(root, xref), + png_name(root, xref), + png_quality[0], + png_quality[1], ) modified.add(xref) - with tqdm( + + def pngquant_fn(args): + pngquant.quantize(*args) + + exec_progress_pool( + use_threads=True, + max_workers=options.jobs, + tqdm_kwargs=dict( desc="PNGs", - total=len(futures), + total=len(images), unit='image', disable=not options.progress_bar, - ) as pbar: - for _future in concurrent.futures.as_completed(futures): - pbar.update() + ), + task=pngquant_fn, + task_arguments=pngquant_args(), + ) for xref in modified: im_obj = pike.get_object(xref, 0) From 1f3665f6144f731fe6bfca9bbda64c14276ac436 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 May 2020 16:10:26 -0700 Subject: [PATCH 444/880] docs: remove reference to brewfile --- docs/installation.rst | 15 ++++++--------- 1 file changed, 6 insertions(+), 9 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 97eb4df5..2bec6bf9 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -443,7 +443,10 @@ languages you can optionally install them all: Manual installation on macOS ---------------------------- -These instructions probably work on all macOS supported by Homebrew. +These instructions probably work on all macOS supported by Homebrew, and are +for installing a more current version of OCRmyPDF than is available from +Homebrew. Note that the Homebrew versions usually track the release versions +fairly closely. If it's not already present, `install Homebrew `__. @@ -454,14 +457,8 @@ Update Homebrew: brew update Install or upgrade the required Homebrew packages, if any are missing. -To do this, download the ``Brewfile`` that lists all of the dependencies -to the current directory, and run ``brew bundle`` to process them -(installing or upgrading as needed). ``Brewfile`` is a plain text file. - -.. code-block:: bash - - wget https://github.com/jbarlow83/OCRmyPDF/raw/master/.travis/Brewfile - brew bundle +To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew +dependencies. You could also check the ``azure-pipelines.yml``. This will include the English, French, German and Spanish language packs. If you need other languages you can optionally install them all: From 51b54893ceed111ca74d3db12c03ea12b5711a36 Mon Sep 17 00:00:00 2001 From: Peter Hogg Date: Mon, 4 May 2020 01:37:58 -0700 Subject: [PATCH 445/880] docs: update Arch Linux install instructions (#540) The python-pdfminer.six package is now available in the official Arch repositories. The dependency will be automatically resolved when installing the OCRmyPDF AUR package. --- docs/installation.rst | 24 +++++------------------- 1 file changed, 5 insertions(+), 19 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 97eb4df5..78351748 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -327,20 +327,7 @@ standard tooling needed to build packages, such as a compiler and binary tools. sudo pacman -S base-devel -The OCRmyPDF package depends on `the python-pdfminer.six AUR package -`__. Dependencies on -AUR packages are not automatically resolved, so this package must be manually -installed first. - -.. code-block:: bash - - curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/python-pdfminer.six.tar.gz - tar xvzf python-pdfminer.six.tar.gz - cd python-pdfminer.six - makepkg -sri - -With that complete you can then repeat the same series of steps for the -OCRmyPDF package. +Now you are ready to install the OCRmyPDF package. .. code-block:: bash @@ -374,11 +361,10 @@ page. fine without it but will produce larger output files. The encoder is available from `the jbig2enc-git AUR package `__ and may be installed - using the same series of steps as for the installation of the pdfminer.six - and OCRmyPDF AUR packages. Alternatively, it may be built manually from - source following the instructions in `Installing the JBIG2 encoder - `__. If JBIG2 is installed, OCRmyPDF 7.0.0 and later will - automatically detect it. + using the same series of steps as for the installation OCRmyPDF AUR + package. Alternatively, it may be built manually from source following the + instructions in `Installing the JBIG2 encoder `__. If JBIG2 is + installed, OCRmyPDF 7.0.0 and later will automatically detect it. Alpine Linux ------------ From 32759c902522a80d59572fc3cbd4de148a6023a1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 00:43:40 -0700 Subject: [PATCH 446/880] Change argument from --plugins to --plugin --- src/ocrmypdf/cli.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 3edf6cd7..d6ba9202 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -492,10 +492,11 @@ Online documentation is located at: "Set the threshold very high to disable.", ) advanced.add_argument( - '--plugins', + '--plugin', + dest='plugins', action='append', default=[], - help="Path to a folder than contains plugins.", + help="Name of plugin to import.", ) debugging = parser.add_argument_group( @@ -515,8 +516,9 @@ plugins_only_parser = ArgumentParser( prog=_PROGRAM_NAME, fromfile_prefix_chars='@', add_help=False, allow_abbrev=False ) plugins_only_parser.add_argument( - '--plugins', + '--plugin', + dest='plugins', action='append', default=[], - help="Path to a folder than contains plugins.", + help="Name of plugin to import.", ) From dd361ecd059a5fb0eaae160a093eef71c10bcd77 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 00:44:40 -0700 Subject: [PATCH 447/880] Support importing plugin by filename --- src/ocrmypdf/_plugin_manager.py | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 295e955c..9ccfc3b0 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -16,6 +16,9 @@ # along with OCRmyPDF. If not, see . import importlib +import importlib.util +import sys +from pathlib import Path from typing import List import pluggy @@ -27,6 +30,15 @@ def get_plugin_manager(plugins: List[str]): pm = pluggy.PluginManager('ocrmypdf') pm.add_hookspecs(pluginspec) for name in plugins: - module = importlib.import_module(name) + if name.endswith('.py'): + # Import by filename + module_name = Path(name).stem + spec = importlib.util.spec_from_file_location(module_name, name) + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + spec.loader.exec_module(module) + else: + # Import by dotted module name + module = importlib.import_module(name) pm.register(module) return pm From 39888ae8c9485806df9d93bbb7bc431cd391c366 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 01:10:09 -0700 Subject: [PATCH 448/880] Rename install_cli to add_options --- src/ocrmypdf/__main__.py | 2 +- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/pluginspec.py | 8 ++++++-- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 411948e2..01f0a478 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -37,7 +37,7 @@ def run(args=None): plugin_manager = get_plugin_manager(pre_options.plugins) parser = get_parser() - plugin_manager.hook.install_cli(parser=parser) + plugin_manager.hook.add_options(parser=parser) options = parser.parse_args(args=args) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 16e151b2..42638a58 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -274,7 +274,7 @@ def ocr( # pylint: disable=unused-argument parser = get_parser() _plugin_manager = get_plugin_manager(plugins) - _plugin_manager.hook.install_cli(parser=parser) + _plugin_manager.hook.add_options(parser=parser) options = create_options( **{k: v for k, v in locals().items() if not k.startswith('_')} diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index e95241e0..76eed1cc 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -26,8 +26,12 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') @hookspec -def install_cli(parser: ArgumentParser) -> None: - """Allows the plugin to add its own command line arguments.""" +def add_options(parser: ArgumentParser) -> None: + """Allows the plugin to add its own command line arguments. + + Even if you do not intend to use plugins in a command line context, you + should use this function to create your options. + """ @hookspec From 6f4286e1b11b580c0d31e476906e659d828749b6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 01:10:32 -0700 Subject: [PATCH 449/880] New hook: filter_page_image --- src/ocrmypdf/_sync.py | 6 ++++++ src/ocrmypdf/example.py | 20 +++++++++++++++----- src/ocrmypdf/pluginspec.py | 28 +++++++++++++++++++++++++--- 3 files changed, 46 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 1d256c92..f15dbbaf 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -165,6 +165,12 @@ def exec_page_sync(page_context): visible_image_out = create_visible_page_jpg( visible_image_out, page_context ) + visible_image_out = ( + page_context.plugin_manager.hook.filter_page_image( + page=page_context, image_filename=Path(visible_image_out) + ) + or visible_image_out + ) pdf_page_from_image_out = create_pdf_page_from_image( visible_image_out, page_context ) diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py index 73b1b544..926831c3 100644 --- a/src/ocrmypdf/example.py +++ b/src/ocrmypdf/example.py @@ -1,13 +1,15 @@ import logging +from PIL import Image + from ocrmypdf import hookimpl log = logging.getLogger(__name__) @hookimpl -def install_cli(parser): - parser.add_argument('--invert', action='store_true') +def add_options(parser): + parser.add_argument('--grayscale-ocr', action='store_true') @hookimpl @@ -22,7 +24,15 @@ def validate(pdfinfo, options): @hookimpl def filter_ocr_image(page, image): - if page.options.invert: - log.info("inverting") - return image.invert() + if page.options.grayscale_ocr: + log.info("graying") + return image.convert('L') return image + + +@hookimpl +def filter_page_image(page, image_filename): + output = image_filename.with_suffix('.jpg') + with Image.open(image_filename) as im: + im.save(output) + return output diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 76eed1cc..3a9620c4 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -16,6 +16,8 @@ # along with OCRmyPDF. If not, see . from argparse import ArgumentParser, Namespace +from pathlib import Path +from typing import Optional import pluggy from PIL import Image @@ -49,11 +51,15 @@ def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: options contains the "work order" to process a particular file. pdfinfo contains information about the input file obtained after loading and - parsing. + parsing. The plugin may modify the options. For example, you could decide + that a certain type of file should be treated with ``options.force_ocr = True`` + based on information in its pdfinfo. The plugin may raise InputFileError or any ExitCodeException to request - normal termination. If the plugin raises another exception type, ocrmypdf - will abort with an error and hold the plugin responsible. + normal termination. ocrmypdf will hold the plugin responsible for raising + exceptions of any other type. + + The return value is ignored. To abort processing, raise an ExitCodeException. """ @@ -64,3 +70,19 @@ def filter_ocr_image(page: 'PageContext', image: Image) -> Image: This is the image that OCR sees, not what the user sees when they view the PDF. """ + + +@hookspec(firstresult=True) +def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: + """Called to filter the whole page before it is inserted into the PDF. + + A whole page image is only produced when preprocessing command line arguments + are issued or when ``--force-ocr`` is issued. If no whole page is image is + produced for a given page, this function will not be called. This is not + the image that will be shown to OCR. + + ocrmypdf will create the PDF page based on the image format used. If you + convert the image to a JPEG, the output page will be created as a JPEG, etc. + Note that the ocrmypdf image optimization stage may ultimately chose a + different format. + """ From 85cbf94a6e76338b764708a26075a49e2639ce8e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 02:53:47 -0700 Subject: [PATCH 450/880] Convert many uses of str paths to Path --- src/ocrmypdf/_concurrent.py | 2 +- src/ocrmypdf/_graft.py | 11 ++++++----- src/ocrmypdf/_jobcontext.py | 19 +++++++++++-------- src/ocrmypdf/_pipeline.py | 9 ++++++--- src/ocrmypdf/_sync.py | 7 ++----- src/ocrmypdf/_validation.py | 6 +++--- src/ocrmypdf/exec/unpaper.py | 7 ++++--- src/ocrmypdf/hocrtransform.py | 14 ++++++++++---- src/ocrmypdf/pdfinfo/info.py | 5 ++--- tests/conftest.py | 14 +++++++------- tests/test_main.py | 6 +++--- 11 files changed, 55 insertions(+), 45 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index cf46bae3..6d608eb6 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -47,7 +47,7 @@ def log_listener(queue): logger = logging.getLogger(record.name) logger.handle(record) except Exception: # pylint: disable=broad-except - import traceback + import traceback # pylint: disable=import-outside-toplevel print("Logging problem", file=sys.stderr) traceback.print_exc(file=sys.stderr) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index fba24718..667067b2 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -16,7 +16,6 @@ # along with OCRmyPDF. If not, see . import logging -import os from contextlib import suppress from pathlib import Path @@ -181,7 +180,7 @@ def _find_font(text, pdf_base): class OcrGrafter: def __init__(self, context): self.context = context - self.path_base = Path(context.origin).resolve() + self.path_base = context.origin self.pdf_base = pikepdf.open(self.path_base) self.font, self.font_key = None, None @@ -264,12 +263,14 @@ class OcrGrafter: # {interim_count} is the opened file we were updateing # {interim_count - 1} can be deleted # {interim_count + 1} is the new file will produce and open - old_file = self.output_file + f'_working{self.interim_count - 1}.pdf' + old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf') if not self.context.options.keep_temporary_files: with suppress(FileNotFoundError): - os.unlink(old_file) + old_file.unlink() - next_file = self.output_file + f'_working{self.interim_count + 1}.pdf' + next_file = self.output_file.with_suffix( + f'.working{self.interim_count + 1}.pdf' + ) self.pdf_base.save(next_file) self.pdf_base.close() diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index c5ffa5b1..3091fbc8 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -19,6 +19,7 @@ import os import shutil import sys from functools import partial +from pathlib import Path from ocrmypdf._plugin_manager import get_plugin_manager @@ -26,10 +27,12 @@ from ocrmypdf._plugin_manager import get_plugin_manager class PdfContext: """Holds our context for a particular run of the pipeline""" - def __init__(self, options, work_folder, origin, pdfinfo, plugin_manager): + def __init__( + self, options, work_folder: Path, origin: Path, pdfinfo, plugin_manager + ): self.options = options - self.work_folder = work_folder - self.origin = origin + self.work_folder = Path(work_folder) + self.origin = Path(origin) self.pdfinfo = pdfinfo self.plugin_manager = plugin_manager if options: @@ -39,8 +42,8 @@ class PdfContext: if self.name == '-': self.name = 'stdin' - def get_path(self, name): - return os.path.join(self.work_folder, name) + def get_path(self, name: str) -> Path: + return self.work_folder / name def get_page_contexts(self): npages = len(self.pdfinfo) @@ -54,7 +57,7 @@ class PageContext: Must be pickable, so only store intrinsic/simple data elements """ - def __init__(self, pdf_context, pageno): + def __init__(self, pdf_context: PdfContext, pageno): self.work_folder = pdf_context.work_folder self.origin = pdf_context.origin self.options = pdf_context.options @@ -63,8 +66,8 @@ class PageContext: self.pageinfo = pdf_context.pdfinfo[pageno] self.plugin_manager = pdf_context.plugin_manager - def get_path(self, name): - return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name)) + def get_path(self, name: str) -> Path: + return self.work_folder / ("%06d_%s" % (self.pageno + 1, name)) def __getstate__(self): state = self.__dict__.copy() diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 196a8d70..9f2fa423 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -55,7 +55,7 @@ def triage_image_file(input_file, output_file, options): im = Image.open(input_file) except EnvironmentError as e: # Recover the original filename - log.error(str(e).replace(input_file, options.input_file)) + log.error(str(e).replace(str(input_file), str(options.input_file))) raise UnsupportedImageFormatError() from e with im: @@ -102,7 +102,10 @@ def triage_image_file(input_file, output_file, options): ) with open(output_file, 'wb') as outf: img2pdf.convert( - input_file, layout_fun=layout_fun, with_pdfrw=False, outputstream=outf + os.fspath(input_file), + layout_fun=layout_fun, + with_pdfrw=False, + outputstream=outf, ) log.info("Successfully converted to PDF, processing...") except img2pdf.ImageOpenError as e: @@ -139,7 +142,7 @@ def triage(original_filename, input_file, output_file, options): return output_file except EnvironmentError as e: log.debug(f"Temporary file was at: {input_file}") - msg = str(e).replace(input_file, original_filename) + msg = str(e).replace(str(input_file), original_filename) raise InputFileError(msg) from e triage_image_file(input_file, output_file, options) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index f15dbbaf..fc5a4ea4 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -304,7 +304,7 @@ def run_pipeline(options, *, plugin_manager, api=False): if not plugin_manager: plugin_manager = get_plugin_manager([]) - work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf.")) debug_log_handler = None if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get( 'PYTEST_CURRENT_TEST', '' @@ -317,10 +317,7 @@ def run_pipeline(options, *, plugin_manager, api=False): # Triage image or pdf origin_pdf = triage( - original_filename, - start_input_file, - os.path.join(work_folder, 'origin.pdf'), - options, + original_filename, start_input_file, work_folder / 'origin.pdf', options ) plugin_manager.hook.prepare(options=options) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 0a755d3c..d7acbf09 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -376,17 +376,17 @@ def log_page_orientations(pdfinfo): log.info('Page orientations detected: %s', ' '.join(orientations)) -def create_input_file(options, work_folder): +def create_input_file(options, work_folder: Path) -> (Path, str): if options.input_file == '-': # stdin log.info('reading file from standard input') - target = os.path.join(work_folder, 'stdin') + target = work_folder / 'stdin' with open(target, 'wb') as stream_buffer: copyfileobj(sys.stdin.buffer, stream_buffer) return target, "" else: try: - target = os.path.join(work_folder, 'origin') + target = work_folder / 'origin' safe_symlink(options.input_file, target) return target, os.fspath(options.input_file) except FileNotFoundError: diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 0228beb6..80243129 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -24,6 +24,7 @@ import logging import os import shlex from functools import lru_cache +from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory @@ -67,8 +68,8 @@ def run(input_file, output_file, dpi, mode_args): "Failed to convert image to a supported format." ) from e - input_pnm = os.path.join(tmpdir, f'input{suffix}') - output_pnm = os.path.join(tmpdir, f'output{suffix}') + input_pnm = Path(tmpdir) / f'input{suffix}' + output_pnm = Path(tmpdir) / f'output{suffix}' im.save(input_pnm, format='PPM') # To prevent any shenanigans from accepting arbitrary parameters in @@ -78,7 +79,7 @@ def run(input_file, output_file, dpi, mode_args): # 3) append absolute paths for the input and output file # This should ensure that a user cannot clobber some other file with # their unpaper arguments (whether intentionally or otherwise) - args_unpaper.extend([input_pnm, output_pnm]) + args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)]) try: proc = external_run( args_unpaper, diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index a60fe959..6240d7ee 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -29,9 +29,11 @@ # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. import argparse +import os import re from collections import namedtuple from math import atan, cos, sin +from pathlib import Path from xml.etree import ElementTree from reportlab.lib.units import inch @@ -155,8 +157,8 @@ class HocrTransform: def to_pdf( self, - out_filename: str, - image_filename: str = None, + out_filename: Path, + image_filename: Path = None, show_bounding_boxes: bool = False, fontname: str = "Helvetica", invisible_text: bool = False, @@ -173,7 +175,9 @@ class HocrTransform: # create the PDF file # page size in points (1/72 in.) pdf = Canvas( - out_filename, pagesize=(self.width, self.height), pageCompression=1 + os.fspath(out_filename), + pagesize=(self.width, self.height), + pageCompression=1, ) # draw bounding box for each paragraph @@ -226,7 +230,9 @@ class HocrTransform: ) # put the image on the page, scaled to fill the page if image_filename is not None: - pdf.drawImage(image_filename, 0, 0, width=self.width, height=self.height) + pdf.drawImage( + os.fspath(image_filename), 0, 0, width=self.width, height=self.height + ) # finish up the page and save it pdf.showPage() diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 42ed372f..fa2dcc05 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -820,9 +820,8 @@ class PdfInfo: def main(): - # pylint: disable=import-outside-toplevel - import argparse - from pprint import pprint + import argparse # pylint: disable=import-outside-toplevel + from pprint import pprint # pylint: disable=import-outside-toplevel parser = argparse.ArgumentParser() parser.add_argument('infile') diff --git a/tests/conftest.py b/tests/conftest.py index d1cc9ef1..9599369a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -52,7 +52,7 @@ def is_macos(): def running_in_docker(): # Docker creates a file named /.dockerenv (newer versions) or # /.dockerinit (older) -- this is undocumented, not an offical test - return os.path.exists('/.dockerenv') or os.path.exists('/.dockerinit') + return Path('/.dockerenv').exists() or Path('/.dockerinit').exists() @pytest.helpers.register @@ -69,9 +69,9 @@ def have_unpaper(): return True -TESTS_ROOT = os.path.abspath(os.path.dirname(__file__)) -SPOOF_PATH = os.path.join(TESTS_ROOT, 'spoof') -PROJECT_ROOT = os.path.dirname(TESTS_ROOT) +TESTS_ROOT = Path(__file__).parent.resolve() +SPOOF_PATH = TESTS_ROOT / 'spoof' +PROJECT_ROOT = TESTS_ROOT OCRMYPDF = [sys.executable, '-m', 'ocrmypdf'] @@ -146,7 +146,7 @@ def spoof(tmp_path_factory, **kwargs): tmpdir.mkdir(parents=True) for replace_program, with_spoof in kwargs.items(): - spoofer = Path(SPOOF_PATH) / with_spoof + spoofer = SPOOF_PATH / with_spoof if os.name != 'nt': spoofer.chmod(0o755) (tmpdir / replace_program).symlink_to(spoofer) @@ -224,8 +224,8 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): result = api.run_pipeline(options, plugin_manager=None, api=True) assert result == 0 - assert os.path.exists(str(output_file)), "Output file not created" - assert os.stat(str(output_file)).st_size > 100, "PDF too small or empty" + assert output_file.exists(), "Output file not created" + assert output_file.stat().st_size > 100, "PDF too small or empty" return output_file diff --git a/tests/test_main.py b/tests/test_main.py index cca9c4eb..9fcb49d7 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -210,7 +210,7 @@ def test_force_ocr_on_pdf_with_no_images(spoof_tesseract_crash, resources, no_ou resources / 'blank.pdf', no_outpdf, '--force-ocr', env=spoof_tesseract_crash ) assert p.returncode == ExitCode.child_process_error - assert not os.path.exists(no_outpdf) + assert not no_outpdf.exists() @pytest.mark.skipif( @@ -321,7 +321,7 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): env=spoof_tesseract_crash, ) assert p.returncode == ExitCode.child_process_error - assert not os.path.exists(no_outpdf) + assert not no_outpdf.exists() assert "SubprocessOutputError" in err @@ -330,7 +330,7 @@ def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf) resources / 'ccitt.pdf', no_outpdf, '-r', env=spoof_tesseract_crash ) assert p.returncode == ExitCode.child_process_error - assert not os.path.exists(no_outpdf) + assert not no_outpdf.exists() assert "uncaught exception" in err print(out) print(err) From 1b086f60a9b82e86aac214aab5390bda1b6d0d43 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 May 2020 12:37:44 -0700 Subject: [PATCH 451/880] tesseract.py: api cleanup --- src/ocrmypdf/_pipeline.py | 3 +- src/ocrmypdf/exec/tesseract.py | 58 ++++++++++++++++------------------ tests/test_tess4.py | 7 ++-- 3 files changed, 33 insertions(+), 35 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 9f2fa423..477ea703 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -537,7 +537,8 @@ def ocr_tesseract_hocr(input_file, page_context): options = page_context.options tesseract.generate_hocr( input_file=input_file, - output_files=[hocr_out, hocr_text_out], + output_hocr=hocr_out, + output_sidecar=hocr_text_out, language=options.language, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 731746e4..19e78739 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -23,7 +23,9 @@ import shutil from collections import namedtuple from contextlib import suppress from os import fspath +from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired +from typing import List, Optional from PIL import Image @@ -139,7 +141,7 @@ def languages(tesseract_env=None): return set(lang.strip() for lang in rest) -def tess_base_args(langs, engine_mode): +def tess_base_args(langs: List[str], engine_mode) -> List[str]: args = ['tesseract'] if langs: args.extend(['-l', '+'.join(langs)]) @@ -148,7 +150,7 @@ def tess_base_args(langs, engine_mode): return args -def get_orientation(input_file, engine_mode, timeout: float, tesseract_env=None): +def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env=None): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', @@ -169,7 +171,7 @@ def get_orientation(input_file, engine_mode, timeout: float, tesseract_env=None) except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) except CalledProcessError as e: - tesseract_log_output(e.output, input_file) + tesseract_log_output(e.output) if ( b'Too few characters. Skipping this page' in e.output or b'Image too large' in e.output @@ -191,7 +193,7 @@ def get_orientation(input_file, engine_mode, timeout: float, tesseract_env=None) return oc -def tesseract_log_output(stdout, input_file): +def tesseract_log_output(stdout): tlog = TesseractLoggerAdapter( log, extra=log.extra if hasattr(log, 'extra') else None ) @@ -241,15 +243,14 @@ def _generate_null_hocr(output_hocr, output_sidecar, image): with Image.open(image) as im: w, h = im.size - with open(output_hocr, 'w', encoding="utf-8") as f: - f.write(HOCR_TEMPLATE.format(w, h)) - with open(output_sidecar, 'w', encoding='utf-8') as f: - f.write('[skipped page]') + output_hocr.write_text(HOCR_TEMPLATE.format(w, h), encoding='utf-8') + output_sidecar.write_text('[skipped page]', encoding='utf-8') def generate_hocr( - input_file, - output_files, + input_file: Path, + output_hocr: Path, + output_sidecar: Path, language: list, engine_mode, tessconfig: list, @@ -259,10 +260,7 @@ def generate_hocr( user_patterns, tesseract_env, ): - - output_hocr = next(o for o in output_files if fspath(o).endswith('.hocr')) - output_sidecar = next(o for o in output_files if fspath(o).endswith('.txt')) - prefix = os.path.splitext(output_hocr)[0] + prefix = output_hocr.with_suffix('') args_tesseract = tess_base_args(language, engine_mode) @@ -295,46 +293,44 @@ def generate_hocr( page_timedout(timeout) _generate_null_hocr(output_hocr, output_sidecar, input_file) except CalledProcessError as e: - tesseract_log_output(e.output, input_file) + tesseract_log_output(e.output) if b'Image too large' in e.output: _generate_null_hocr(output_hocr, output_sidecar, input_file) return raise SubprocessOutputError() from e else: - tesseract_log_output(stdout, input_file) + tesseract_log_output(stdout) # The sidecar text file will get the suffix .txt; rename it to # whatever caller wants it named - if os.path.exists(prefix + '.txt'): - shutil.move(prefix + '.txt', output_sidecar) + if prefix.with_suffix('.txt').exists(): + shutil.move(prefix.with_suffix('.txt'), output_sidecar) def use_skip_page(text_only, skip_pdf, output_pdf, output_text): - with open(output_text, 'w') as f: - f.write('[skipped page]') + output_text.write_text('[skipped page]', encoding='utf-8') if skip_pdf and not text_only: # Substitute a "skipped page" with suppress(FileNotFoundError): - os.remove(output_pdf) # In case it was partially created + output_pdf.unlink() # In case it was partially created safe_symlink(skip_pdf, output_pdf) return # Or normally, just write a 0 byte file to the output to indicate a skip - with open(output_pdf, 'wb') as out: - out.write(b'') + output_pdf.write_bytes(b'') def generate_pdf( *, - input_image, - skip_pdf=None, - output_pdf, - output_text, - language: list, + input_image: Path, + skip_pdf: Optional[Path] = None, + output_pdf: Path, + output_text: Path, + language: List[str], engine_mode, text_only: bool, - tessconfig: list, + tessconfig: List[str], timeout: float, pagesegmode: int, user_words, @@ -391,10 +387,10 @@ def generate_pdf( page_timedout(timeout) use_skip_page(text_only, skip_pdf, output_pdf, output_text) except CalledProcessError as e: - tesseract_log_output(e.output, input_image) + tesseract_log_output(e.output) if b'Image too large' in e.output: use_skip_page(text_only, skip_pdf, output_pdf, output_text) return raise SubprocessOutputError() from e else: - tesseract_log_output(stdout, input_image) + tesseract_log_output(stdout) diff --git a/tests/test_tess4.py b/tests/test_tess4.py index 4fada7d4..6a524d64 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -91,7 +91,8 @@ def test_image_too_large_hocr(monkeypatch, resources, outdir): monkeypatch.setattr(tesseract, 'run', dummy_run) tesseract.generate_hocr( input_file=resources / 'crom.png', - output_files=[outdir / 'out.hocr', outdir / 'out.txt'], + output_hocr=outdir / 'out.hocr', + output_sidecar=outdir / 'out.txt', language=['eng'], engine_mode=None, tessconfig=[], @@ -152,7 +153,7 @@ def test_timeout(caplog): ) def test_tesseract_log_output(caplog, in_, logged): caplog.set_level(logging.INFO) - tesseract.tesseract_log_output(in_, 'dummy') + tesseract.tesseract_log_output(in_) if logged == '': assert caplog.text == '' else: @@ -161,5 +162,5 @@ def test_tesseract_log_output(caplog, in_, logged): def test_tesseract_log_output_raises(caplog): with pytest.raises(tesseract.TesseractConfigError): - tesseract.tesseract_log_output(b'parameter not found: moo', 'dummy') + tesseract.tesseract_log_output(b'parameter not found: moo') assert 'not found' in caplog.text From e760622a5c4ea69cb90e9e147daaa62e4ed03f8a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 02:03:42 -0700 Subject: [PATCH 452/880] graft: refactor --- src/ocrmypdf/_graft.py | 198 +++++++++++++++++++++-------------------- src/ocrmypdf/_sync.py | 7 +- 2 files changed, 108 insertions(+), 97 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 667067b2..768b94ca 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -89,94 +89,6 @@ def strip_invisible_text(pdf, page): page.Contents = pikepdf.Stream(pdf, content_stream) -def _graft_text_layer( - *, pdf_base, page_num, text, font, font_key, procset, rotation, strip_old_text -): - """Insert the text layer from text page 0 on to pdf_base at page_num""" - - log.debug("Grafting") - if Path(text).stat().st_size == 0: - return - - # This is a pointer indicating a specific page in the base file - pdf_text = pikepdf.open(text) - pdf_text_contents = pdf_text.pages[0].Contents.read_bytes() - - base_page = pdf_base.pages.p(page_num) - - # The text page always will be oriented up by this stage but the original - # content may have a rotation applied. Wrap the text stream with a rotation - # so it will be oriented the same way as the rest of the page content. - # (Previous versions OCRmyPDF rotated the content layer to match the text.) - mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)] - wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] - - mediabox = [float(base_page.MediaBox[v]) for v in range(4)] - wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] - - translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2) - untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2) - corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) - # -rotation because the input is a clockwise angle and this formula - # uses CCW - rotation = -rotation % 360 - rotate = pikepdf.PdfMatrix().rotated(rotation) - - # Because of rounding of DPI, we might get a text layer that is not - # identically sized to the target page. Scale to adjust. Normally this - # is within 0.998. - if rotation in (90, 270): - wt, ht = ht, wt - scale_x = wp / wt - scale_y = hp / ht - - # log.debug('%r', scale_x, scale_y) - scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) - - # Translate the text so it is centered at (0, 0), rotate it there, adjust - # for a size different between initial and text PDF, then untranslate, and - # finally move the lower left corner to match the mediabox - ctm = translate @ rotate @ scale @ untranslate @ corner - - pdf_text_contents = b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' - - new_text_layer = pikepdf.Stream(pdf_base, pdf_text_contents) - - if strip_old_text: - strip_invisible_text(pdf_base, base_page) - - base_page.page_contents_add(new_text_layer, prepend=True) - - _update_page_resources( - page=base_page, font=font, font_key=font_key, procset=procset - ) - pdf_text.close() - - -def _find_font(text, pdf_base): - """Copy a font from the filename text into pdf_base""" - - font, font_key = None, None - possible_font_names = ('/f-0-0', '/F1') - try: - with pikepdf.open(text) as pdf_text: - try: - pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {}) - except (AttributeError, IndexError, KeyError): - return None, None - for f in possible_font_names: - pdf_text_font = pdf_text_fonts.get(f, None) - if pdf_text_font is not None: - font_key = f - break - if pdf_text_font: - font = pdf_base.copy_foreign(pdf_text_font) - return font, font_key - except (FileNotFoundError, pikepdf.PdfError): - # PdfError occurs if a 0-length file is written e.g. due to OCR timeout - return None, None - - class OcrGrafter: def __init__(self, context): self.context = context @@ -195,10 +107,11 @@ class OcrGrafter: self.emplacements = 1 self.interim_count = 0 - def graft_page(self, page_result): - pageno, image, text, _sidecar, autorotate_correction = page_result - if text and not self.font: - self.font, self.font_key = _find_font(text, self.pdf_base) + def graft_page( + self, *, pageno: int, image: Path, textpdf: Path, autorotate_correction: int + ): + if textpdf and not self.font: + self.font, self.font_key = self._find_font(textpdf) emplaced_page = False content_rotation = self.pdfinfo[pageno].rotation @@ -226,13 +139,12 @@ class OcrGrafter: f"{text_misaligned}, {content_rotation}" ) - if text and self.font: + if textpdf and self.font: # Graft the text layer onto this page, whether new or old strip_old = self.context.options.redo_ocr - _graft_text_layer( - pdf_base=self.pdf_base, + self._graft_text_layer( page_num=pageno + 1, - text=text, + textpdf=textpdf, font=self.font, font_key=self.font_key, rotation=text_misaligned, @@ -283,3 +195,97 @@ class OcrGrafter: self.pdf_base.save(self.output_file) self.pdf_base.close() return self.output_file + + def _find_font(self, text): + """Copy a font from the filename text into pdf_base""" + + font, font_key = None, None + possible_font_names = ('/f-0-0', '/F1') + try: + with pikepdf.open(text) as pdf_text: + try: + pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {}) + except (AttributeError, IndexError, KeyError): + return None, None + for f in possible_font_names: + pdf_text_font = pdf_text_fonts.get(f, None) + if pdf_text_font is not None: + font_key = f + break + if pdf_text_font: + font = self.pdf_base.copy_foreign(pdf_text_font) + return font, font_key + except (FileNotFoundError, pikepdf.PdfError): + # PdfError occurs if a 0-length file is written e.g. due to OCR timeout + return None, None + + def _graft_text_layer( + self, + *, + page_num: int, + textpdf: Path, + font: pikepdf.Object, + font_key: pikepdf.Object, + procset: pikepdf.Object, + rotation: int, + strip_old_text: bool, + ): + """Insert the text layer from text page 0 on to pdf_base at page_num""" + + log.debug("Grafting") + if Path(textpdf).stat().st_size == 0: + return + + # This is a pointer indicating a specific page in the base file + pdf_text = pikepdf.open(textpdf) + pdf_text_contents = pdf_text.pages[0].Contents.read_bytes() + + base_page = self.pdf_base.pages.p(page_num) + + # The text page always will be oriented up by this stage but the original + # content may have a rotation applied. Wrap the text stream with a rotation + # so it will be oriented the same way as the rest of the page content. + # (Previous versions OCRmyPDF rotated the content layer to match the text.) + mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)] + wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] + + mediabox = [float(base_page.MediaBox[v]) for v in range(4)] + wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] + + translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2) + untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2) + corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) + # -rotation because the input is a clockwise angle and this formula + # uses CCW + rotation = -rotation % 360 + rotate = pikepdf.PdfMatrix().rotated(rotation) + + # Because of rounding of DPI, we might get a text layer that is not + # identically sized to the target page. Scale to adjust. Normally this + # is within 0.998. + if rotation in (90, 270): + wt, ht = ht, wt + scale_x = wp / wt + scale_y = hp / ht + + # log.debug('%r', scale_x, scale_y) + scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) + + # Translate the text so it is centered at (0, 0), rotate it there, adjust + # for a size different between initial and text PDF, then untranslate, and + # finally move the lower left corner to match the mediabox + ctm = translate @ rotate @ scale @ untranslate @ corner + + pdf_text_contents = b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' + + new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents) + + if strip_old_text: + strip_invisible_text(self.pdf_base, base_page) + + base_page.page_contents_add(new_text_layer, prepend=True) + + _update_page_resources( + page=base_page, font=font, font_key=font_key, procset=procset + ) + pdf_text.close() diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index fc5a4ea4..710bda5c 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -243,7 +243,12 @@ def exec_concurrent(context): def update_page(result, pbar): sidecars[result.pageno] = result.text pbar.update() - ocrgraft.graft_page(result) + ocrgraft.graft_page( + pageno=result.pageno, + image=result.pdf_page_from_image, + textpdf=result.ocr, + autorotate_correction=result.orientation_correction, + ) pbar.update() exec_progress_pool( From 9462f0a28fd73f235ee4a79b440b144f4406f673 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 02:59:24 -0700 Subject: [PATCH 453/880] graft: more refactoring --- src/ocrmypdf/_graft.py | 92 ++++++++++++++++++++++-------------------- 1 file changed, 48 insertions(+), 44 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 768b94ca..de9ebda0 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -161,10 +161,13 @@ class OcrGrafter: self.save_and_reload() def save_and_reload(self): - # Periodically save and reload the Pdf object. This will keep a - # lid on our memory usage for very large files. Attach the font to - # page 1 even if page 1 doesn't use it, so we have a way to get it - # back. + """Save and reload the Pdf. + + This will keep a lid on our memory usage for very large files. Attach + the font to page 1 even if page 1 doesn't use it, so we have a way to get it + back. + """ + page0 = self.pdf_base.pages[0] _update_page_resources( page=page0, font=self.font, font_key=self.font_key, procset=self.procset @@ -237,55 +240,56 @@ class OcrGrafter: return # This is a pointer indicating a specific page in the base file - pdf_text = pikepdf.open(textpdf) - pdf_text_contents = pdf_text.pages[0].Contents.read_bytes() + with pikepdf.open(textpdf) as pdf_text: + pdf_text_contents = pdf_text.pages[0].Contents.read_bytes() - base_page = self.pdf_base.pages.p(page_num) + base_page = self.pdf_base.pages.p(page_num) - # The text page always will be oriented up by this stage but the original - # content may have a rotation applied. Wrap the text stream with a rotation - # so it will be oriented the same way as the rest of the page content. - # (Previous versions OCRmyPDF rotated the content layer to match the text.) - mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)] - wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] + # The text page always will be oriented up by this stage but the original + # content may have a rotation applied. Wrap the text stream with a rotation + # so it will be oriented the same way as the rest of the page content. + # (Previous versions OCRmyPDF rotated the content layer to match the text.) + mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)] + wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] - mediabox = [float(base_page.MediaBox[v]) for v in range(4)] - wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] + mediabox = [float(base_page.MediaBox[v]) for v in range(4)] + wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1] - translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2) - untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2) - corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) - # -rotation because the input is a clockwise angle and this formula - # uses CCW - rotation = -rotation % 360 - rotate = pikepdf.PdfMatrix().rotated(rotation) + translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2) + untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2) + corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) + # -rotation because the input is a clockwise angle and this formula + # uses CCW + rotation = -rotation % 360 + rotate = pikepdf.PdfMatrix().rotated(rotation) - # Because of rounding of DPI, we might get a text layer that is not - # identically sized to the target page. Scale to adjust. Normally this - # is within 0.998. - if rotation in (90, 270): - wt, ht = ht, wt - scale_x = wp / wt - scale_y = hp / ht + # Because of rounding of DPI, we might get a text layer that is not + # identically sized to the target page. Scale to adjust. Normally this + # is within 0.998. + if rotation in (90, 270): + wt, ht = ht, wt + scale_x = wp / wt + scale_y = hp / ht - # log.debug('%r', scale_x, scale_y) - scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) + # log.debug('%r', scale_x, scale_y) + scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y) - # Translate the text so it is centered at (0, 0), rotate it there, adjust - # for a size different between initial and text PDF, then untranslate, and - # finally move the lower left corner to match the mediabox - ctm = translate @ rotate @ scale @ untranslate @ corner + # Translate the text so it is centered at (0, 0), rotate it there, adjust + # for a size different between initial and text PDF, then untranslate, and + # finally move the lower left corner to match the mediabox + ctm = translate @ rotate @ scale @ untranslate @ corner - pdf_text_contents = b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' + pdf_text_contents = ( + b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' + ) - new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents) + new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents) - if strip_old_text: - strip_invisible_text(self.pdf_base, base_page) + if strip_old_text: + strip_invisible_text(self.pdf_base, base_page) - base_page.page_contents_add(new_text_layer, prepend=True) + base_page.page_contents_add(new_text_layer, prepend=True) - _update_page_resources( - page=base_page, font=font, font_key=font_key, procset=procset - ) - pdf_text.close() + _update_page_resources( + page=base_page, font=font, font_key=font_key, procset=procset + ) From 7a12908db904cfbda9a43e5ea805d210d5c0652c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 03:27:39 -0700 Subject: [PATCH 454/880] Relocate example plugin --- misc/example_plugin.py | 53 +++++++++++++++++++++++++++++++++++++++++ src/ocrmypdf/example.py | 38 ----------------------------- 2 files changed, 53 insertions(+), 38 deletions(-) create mode 100644 misc/example_plugin.py delete mode 100644 src/ocrmypdf/example.py diff --git a/misc/example_plugin.py b/misc/example_plugin.py new file mode 100644 index 00000000..d6c93363 --- /dev/null +++ b/misc/example_plugin.py @@ -0,0 +1,53 @@ +# © 2020 James R Barlow: https://github.com/jbarlow83 +# +# This program is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with this program. If not, see . + +import logging + +from PIL import Image + +from ocrmypdf import hookimpl + +log = logging.getLogger(__name__) + + +@hookimpl +def add_options(parser): + parser.add_argument('--grayscale-ocr', action='store_true') + + +@hookimpl +def prepare(options): + pass + + +@hookimpl +def validate(pdfinfo, options): + pass + + +@hookimpl +def filter_ocr_image(page, image): + if page.options.grayscale_ocr: + log.info("graying") + return image.convert('L') + return image + + +@hookimpl +def filter_page_image(page, image_filename): + output = image_filename.with_suffix('.jpg') + with Image.open(image_filename) as im: + im.save(output) + return output diff --git a/src/ocrmypdf/example.py b/src/ocrmypdf/example.py deleted file mode 100644 index 926831c3..00000000 --- a/src/ocrmypdf/example.py +++ /dev/null @@ -1,38 +0,0 @@ -import logging - -from PIL import Image - -from ocrmypdf import hookimpl - -log = logging.getLogger(__name__) - - -@hookimpl -def add_options(parser): - parser.add_argument('--grayscale-ocr', action='store_true') - - -@hookimpl -def prepare(options): - pass - - -@hookimpl -def validate(pdfinfo, options): - pass - - -@hookimpl -def filter_ocr_image(page, image): - if page.options.grayscale_ocr: - log.info("graying") - return image.convert('L') - return image - - -@hookimpl -def filter_page_image(page, image_filename): - output = image_filename.with_suffix('.jpg') - with Image.open(image_filename) as im: - im.save(output) - return output From 417dbd43f6d4dfa4f87e939547c8e77e485187f1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 03:53:37 -0700 Subject: [PATCH 455/880] docs: plugin documentation --- docs/api.rst | 6 +-- docs/index.rst | 1 + docs/plugins.rst | 80 ++++++++++++++++++++++++++++++++------ src/ocrmypdf/pluginspec.py | 15 +++---- 4 files changed, 80 insertions(+), 22 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index c38bb400..b724b116 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -105,8 +105,8 @@ Reference :members: :undoc-members: -.. autoclass:: ocrmypdf.ExitCode +.. autofunction:: ocrmypdf.configure_logging + +.. automodule:: ocrmypdf.exceptions :members: :undoc-members: - -.. autofunction:: ocrmypdf.configure_logging diff --git a/docs/index.rst b/docs/index.rst index e28f3234..8e9cd991 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -34,6 +34,7 @@ image processing and OCR to existing PDFs. :maxdepth: 2 api + plugins contributing Indices and tables diff --git a/docs/plugins.rst b/docs/plugins.rst index e22cea36..f2ac0b94 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -2,23 +2,79 @@ Plugins ======= -You can use plugins to customize the behavior of OCRmyPDF at certain -points of interest. +You can use plugins to customize the behavior of OCRmyPDF at certain points of +interest. -Currently, it is possible to: - override the decision for whether or not -to perform OCR on a particular file - modify the image is about to be -sent for OCR +Currently, it is possible to: + +- add new command line arguments +- override the decision for whether or not to perform OCR on a particular file +- modify the image is about to be sent for OCR +- modify the page image before it is converted to PDF + +OCRmyPDF plugins are based on the Python ``pluggy`` package and conform to its +conventions. Note that: plugins installed with as setuptools entrypoints are +not checked currently, because OCRmyPDF assumes you may not want to enable +plugins for all files. Also, plugins must be functions, not classes. How plugins are imported ======================== -Plugins are imported on demand, by the OCRmyPDF worker process that -needs to use them. As such, plugins cannot share state with each other, -and will be imported many times, once for each worker process. +Plugins are imported on demand, by the OCRmyPDF worker process that needs to use +them. As such, plugins cannot share state with other plugins, cannot rely on +their module's or the interpreter's global state, and should expect asynchronous +copies of themselves to be running. Plugins can write intermediate files to the +folder specified in ``options.work_folder``. -Plugins currently cannot override the same hook. +Plugins should work whether executed in threads or processes. -How plugins are invoked -======================= +Script plugins +============== -Plugins may be called from the command line: +Script plugins may be called from the command line, by specifying the name of a file. + +.. code-block:: bash + + ocrmypdf --plugin example_plugin.py input.pdf output.pdf + +Multiple plugins may be called by issuing the ``--plugin`` argument multiple times. + +Packaged plugins +================ + +Installed plugins may be installed into the same virtual environment as OCRmyPDF +is installed into. They may be invoked using Python standard module naming. + +.. code-block:: bash + + ocrmypdf --plugin ocrmypdf_fancypants.pockets.contents input.pdf output.pdf + +OCRmyPDF does not automatically import plugins, because the assumption is that +plugins affect different files differently and you may not want them activated +all the time. The command line or ``ocrmypdf.ocr(plugin='...')`` must call +for them. + +Third parties that wish to distribute packages for ocrmypdf should package them +as packaged plugins, and these modules should begin with the name ``ocrmypdf_`` +similar to ``pytest`` packages such as ``pytest-cov`` (the package) and +``pytest_cov`` (the module). + +Plugin hooks +============ + +A plugin may provide the following hooks. Hooks should be decorated with +``ocrmypdf.hookimpl``, for example: + +.. code-block:: python + + from ocrmpydf import hookimpl + + @hookimpl + def prepare(options): + pass + +The following is a complete list of hooks that may be installed and when +they are called. + +.. automodule:: ocrmypdf.pluginspec + :members: diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 3a9620c4..6063ce54 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -40,26 +40,27 @@ def add_options(parser: ArgumentParser) -> None: def prepare(options: Namespace) -> None: """Called to notify a plugin that a file will be processed. - The plugin may modify the options. All objects that are in options must + The plugin may modify the *options*. All objects that are in options must be picklable so they can be marshalled to child worker processes. """ @hookspec def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: - """Called to give a plugin an opportunity to review options and pdfinfo. + """Called to give a plugin an opportunity to review *options* and *pdfinfo*. - options contains the "work order" to process a particular file. pdfinfo + *options* contains the "work order" to process a particular file. *pdfinfo* contains information about the input file obtained after loading and - parsing. The plugin may modify the options. For example, you could decide + parsing. The plugin may modify the *options*. For example, you could decide that a certain type of file should be treated with ``options.force_ocr = True`` - based on information in its pdfinfo. + based on information in its *pdfinfo*. - The plugin may raise InputFileError or any ExitCodeException to request + The plugin may raise :class:`ocrmypdf.exceptions.InputFileError` or any + :class:`ocrmypdf.exceptions.ExitCodeException` to request normal termination. ocrmypdf will hold the plugin responsible for raising exceptions of any other type. - The return value is ignored. To abort processing, raise an ExitCodeException. + The return value is ignored. To abort processing, raise an ``ExitCodeException``. """ From 4b98ce391b161b92c37f724cb85b755023367896 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 03:54:27 -0700 Subject: [PATCH 456/880] docs: rename security->pdfsecurity so github won't misinterpret it --- docs/index.rst | 2 +- docs/{security.rst => pdfsecurity.rst} | 0 2 files changed, 1 insertion(+), 1 deletion(-) rename docs/{security.rst => pdfsecurity.rst} (100%) diff --git a/docs/index.rst b/docs/index.rst index 8e9cd991..e7f69f7e 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -26,7 +26,7 @@ image processing and OCR to existing PDFs. docker advanced batch - security + pdfsecurity errors .. toctree:: diff --git a/docs/security.rst b/docs/pdfsecurity.rst similarity index 100% rename from docs/security.rst rename to docs/pdfsecurity.rst From 790ff58f675bab850fca41a8ac0c2918bc8dbd96 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 7 May 2020 22:19:21 -0700 Subject: [PATCH 457/880] Add fix for bug in Windows Python 3.6/3.7 TypeError: argument of type 'WindowsPath' is not iterable --- docs/api.rst | 4 ++-- src/ocrmypdf/exec/__init__.py | 5 +++++ 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index b724b116..ce6583fc 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -56,8 +56,8 @@ OCRmyPDF does not. On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected by an "ifmain" guard (``if __name__ == '__main__'``) or you must use ``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one - of these steps, Windows fork semantics will prevent OCRmyPDF from working - correct. + of these steps, Windows process semantics will prevent OCRmyPDF from working + correctly. Logging ------- diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 80e6ba98..d400820f 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -94,6 +94,11 @@ def run(args, *, env=None, **kwargs): def fix_windows_args(program, args, env): """Adjust our desired program and command line arguments for use on Windows""" + if sys.version_info < (3, 8): + # bpo-33617 - Windows needs manual Path -> str conversion + args = [os.fspath(arg) for arg in args] + program = os.fspath(program) + # If we are running a .py on Windows, ensure we call it with this Python # (to support test suite shims) if program.lower().endswith('.py'): From fd7497f00d2bdc9263fc348a6a868c46ec3479ef Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 May 2020 03:44:39 -0700 Subject: [PATCH 458/880] Remove old function tesseract.v4() --- src/ocrmypdf/exec/tesseract.py | 5 ----- tests/{test_tess4.py => test_tesseract.py} | 6 +----- 2 files changed, 1 insertion(+), 10 deletions(-) rename tests/{test_tess4.py => test_tesseract.py} (98%) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 19e78739..aea9a0ce 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -69,11 +69,6 @@ def version(tesseract_env=None): return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env) -def v4(tesseract_env=None): - "Is this Tesseract v4.0?" - return version(tesseract_env) >= '4' - - def has_textonly_pdf(tesseract_env=None, langs=None): """Does Tesseract have textonly_pdf capability? diff --git a/tests/test_tess4.py b/tests/test_tesseract.py similarity index 98% rename from tests/test_tess4.py rename to tests/test_tesseract.py index 6a524d64..3f921de8 100644 --- a/tests/test_tess4.py +++ b/tests/test_tesseract.py @@ -27,17 +27,13 @@ from ocrmypdf import pdfinfo from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.exec import tesseract -# pylint: disable=no-member,w0621 +# pylint: disable=no-member,redefined-outer-name check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -def test_tesseract_v4(): - assert tesseract.v4() - - @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) def test_skip_pages_does_not_replicate(resources, basename, outdir): infile = resources / basename From 977665d2b6175f0ac27759bd26df779feb2e2635 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 May 2020 03:49:33 -0700 Subject: [PATCH 459/880] Delint some tests --- src/ocrmypdf/pdfinfo/layout.py | 3 +-- tests/test_graft.py | 1 - tests/test_metadata.py | 7 +++---- 3 files changed, 4 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index f763a9eb..db159c2e 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -25,10 +25,9 @@ import pdfminer.encodingdb import pdfminer.pdfdevice import pdfminer.pdfinterp from pdfminer.converter import PDFLayoutAnalyzer -from pdfminer.glyphlist import glyphname2unicode from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox from pdfminer.pdfdocument import PDFTextExtractionNotAllowed -from pdfminer.pdffont import PDFFont, PDFSimpleFont, PDFUnicodeNotDefined +from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined from pdfminer.pdfpage import PDFPage from pdfminer.utils import bbox2str, matrix2str diff --git a/tests/test_graft.py b/tests/test_graft.py index 65cba7c8..1329e869 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import os from unittest.mock import patch import pikepdf diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 9a92c65a..acb8f31c 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -41,7 +41,6 @@ except ImportError: # pytest.helpers is dynamic # pylint: disable=no-member -# pylint: disable=w0612 pytestmark = pytest.mark.filterwarnings('ignore:.*XMLParser.*:DeprecationWarning') @@ -77,7 +76,7 @@ def test_override_metadata(spoof_tesseract_noop, output_type, resources, outpdf) german = 'Du siehst den Wald vor lauter Bäumen nicht.' chinese = '孔子' - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( input_file, outpdf, '--title', @@ -113,7 +112,7 @@ def test_high_unicode(spoof_tesseract_noop, resources, no_outpdf): input_file = resources / 'c02-22.pdf' high_unicode = 'U+1030C is: 𐌌' - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( input_file, no_outpdf, '--subject', @@ -275,7 +274,7 @@ def test_srgb_in_unicode_path(tmp_path): def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): - output = check_ocrmypdf( + _output = check_ocrmypdf( resources / 'kcs.pdf', outpdf, '--output-type', 'pdf', env=spoof_tesseract_noop ) From 33b68454f35b483aac5edc349d90253143b0ac18 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 May 2020 03:49:49 -0700 Subject: [PATCH 460/880] watcher: cleanup getenv casting --- misc/watcher.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index 168c9998..097865fa 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -32,11 +32,11 @@ import ocrmypdf INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') -OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False)) -ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False)) -DESKEW = bool(os.getenv('OCR_DESKEW', False)) +OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', '')) +ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', '')) +DESKEW = bool(os.getenv('OCR_DESKEW', '')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) -POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1) +POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] From 2541f6cf899f3e7e282c92de403598dabf3fecd4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:05:57 -0700 Subject: [PATCH 461/880] Fix missing jbig2enc reported as error with -O3 instead of warning Fixes #558 --- src/ocrmypdf/exec/__init__.py | 8 ++++---- tests/test_validation.py | 21 +++++++++++++++++++++ 2 files changed, 25 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 92e4d956..f235b6e8 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -233,10 +233,10 @@ def _error_trailer(program, package, **kwargs): def _error_missing_program(program, package, required_for, recommended): - if required_for: + if recommended: + log.warning(missing_recommend_program.format(**locals())) + elif required_for: log.error(missing_optional_program.format(**locals())) - elif recommended: - log.info(missing_recommend_program.format(**locals())) else: log.error(missing_program.format(**locals())) _error_trailer(**locals()) @@ -279,7 +279,7 @@ def check_external_program( found_version = remove_leading_v(found_version) need_version = remove_leading_v(need_version) - if LooseVersion(found_version) < LooseVersion(need_version): + if found_version and LooseVersion(found_version) < LooseVersion(need_version): _error_old_version(program, package, need_version, found_version, required_for) if not recommended: raise MissingDependencyError() diff --git a/tests/test_validation.py b/tests/test_validation.py index af1eadea..199cfcd2 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -210,3 +210,24 @@ def test_version_comparison(): version_checker=lambda: '1.0', need_version='2.0', ) + + +def test_optional_program_recommended(caplog): + caplog.clear() + + def raiser(): + raise FileNotFoundError('jbig2') + + with caplog.at_level(logging.WARNING): + vd.check_external_program( + program="jbig2", + package="jbig2enc", + version_checker=raiser, + need_version='42', + required_for='this test case', + recommended=True, + ) + assert any( + (loglevel == logging.WARNING and "recommended" in msg) + for _logger_name, loglevel, msg in caplog.record_tuples + ) From 2fae9b655e9c5c7271cee74b869511858a42a9b5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:07:01 -0700 Subject: [PATCH 462/880] Remove **kwargs from check_external_program; deprecated --- src/ocrmypdf/exec/__init__.py | 1 - 1 file changed, 1 deletion(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index d400820f..44ef5e3e 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -278,7 +278,6 @@ def check_external_program( need_version, required_for=None, recommended=False, - **kwargs, # To consume log parameter ): try: found_version = version_checker() From 4b986a5943e578c2a0efa187792333757c024c8f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:28:36 -0700 Subject: [PATCH 463/880] cli: make ArgumentParser._api_mode private --- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/cli.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 42638a58..f53107fa 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -174,7 +174,7 @@ def create_options( cmdline.append(str(input_file)) cmdline.append(str(output_file)) - parser.api_mode = True + parser._api_mode = True options = parser.parse_args(cmdline) for keyword, val in deferred: setattr(options, keyword, val) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index d6ba9202..cea8e351 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -45,10 +45,10 @@ class ArgumentParser(argparse.ArgumentParser): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) - self.api_mode = False + self._api_mode = False def error(self, message): - if not self.api_mode: + if not self._api_mode: super().error(message) return raise ValueError(message) From a87c81a64fc0347f2a2624b69f18d257f16dc239 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:28:50 -0700 Subject: [PATCH 464/880] helpers: remove unnecessary isinstance test --- src/ocrmypdf/helpers.py | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index ff4bb13a..964f7670 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -144,11 +144,7 @@ def is_file_writable(test_file: os.PathLike): the location is writable. """ try: - if not isinstance(test_file, Path): - p = Path(test_file) - else: - p = test_file - + p = Path(test_file) if p.is_symlink(): p = p.resolve(strict=False) From db8c37e58cb6d7f572f8437da5cad24f78d68def Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:34:10 -0700 Subject: [PATCH 465/880] Refactor ocrmypdf.exec.__init__.py --- src/ocrmypdf/exec/__init__.py | 292 +------------------------------- src/ocrmypdf/exec/_support.py | 302 ++++++++++++++++++++++++++++++++++ src/ocrmypdf/leptonica.py | 6 +- 3 files changed, 312 insertions(+), 288 deletions(-) create mode 100644 src/ocrmypdf/exec/_support.py diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 44ef5e3e..13b4e48c 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -1,4 +1,4 @@ -# © 2016 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # This file is part of OCRmyPDF. # @@ -17,287 +17,9 @@ """Wrappers to manage subprocess calls""" -import logging -import os -import re -import shutil -import sys -from collections.abc import Mapping -from contextlib import suppress -from distutils.version import LooseVersion -from functools import lru_cache -from subprocess import PIPE, STDOUT, CalledProcessError -from subprocess import run as subprocess_run - -from ..exceptions import ExitCode, MissingDependencyError - -log = logging.getLogger(__name__) - - -def _get_program(args, env=None): - program = args[0] - test_path = env.get('_OCRMYPDF_TEST_PATH', '') - if test_path: - program = shutil.which(program, path=test_path) - return program - - -def run(args, *, env=None, **kwargs): - """Wrapper around subprocess.run() - - The main purpose of this wrapper is to allow us to substitute the main program - for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces - the main PATH as a location to check for programs to run. - - Secondly we have to account for behavioral differences in Windows in particular. - Creating symbolic links in Windows requires administrator privileges and - may not work if for some reason we're using a FAT file system or the temporary - folder is on a different drive from the working folder. The test suite - works around this by creating shim Python scripts that perform the same function - as a symbolic link, but those shims require support on this side, to ensure - we call them with Python. - - """ - if not env: - env = os.environ - - # Search in spoof path if necessary - program = _get_program(args, env) - args = [program] + args[1:] - - if os.name == 'nt': - args = fix_windows_args(program, args, env) - - log.debug("Running: %s", args) - process_log = log.getChild('subprocess.' + os.path.basename(program)) - if sys.version_info < (3, 7) and os.name == 'nt': - # Can't use close_fds=True on Windows with Python 3.6 or older - # https://bugs.python.org/issue19575, etc. - kwargs['close_fds'] = False - - stderr = None - try: - proc = subprocess_run(args, env=env, **kwargs) - except CalledProcessError as e: - stderr = getattr(e, 'stderr', None) - raise - else: - stderr = getattr(proc, 'stderr', None) - finally: - if process_log.isEnabledFor(logging.DEBUG) and stderr: - with suppress(AttributeError, UnicodeDecodeError): - stderr = stderr.decode('utf-8', 'replace') - process_log.debug("stderr = %s", stderr) - return proc - - -def fix_windows_args(program, args, env): - """Adjust our desired program and command line arguments for use on Windows""" - - if sys.version_info < (3, 8): - # bpo-33617 - Windows needs manual Path -> str conversion - args = [os.fspath(arg) for arg in args] - program = os.fspath(program) - - # If we are running a .py on Windows, ensure we call it with this Python - # (to support test suite shims) - if program.lower().endswith('.py'): - args = [sys.executable] + args - - paths = os.pathsep.join(os.get_exec_path(env)) - if not shutil.which(args[0], path=paths): - # If the program we want is not on the PATH, add some interesting - # locations in %PROGRAMFILES% to the PATH and try again - shimmed_path = shim_paths_with_program_files(env) - new_args0 = shutil.which(args[0], path=shimmed_path) - if new_args0: - args[0] = new_args0 - return args - - -def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): - """Get the version of the specified program""" - args_prog = [program, version_arg] - try: - proc = run( - args_prog, - close_fds=True, - universal_newlines=True, - stdout=PIPE, - stderr=STDOUT, - check=True, - env=env, - ) - output = proc.stdout - except FileNotFoundError as e: - raise MissingDependencyError( - f"Could not find program '{program}' on the PATH" - ) from e - except CalledProcessError as e: - if e.returncode != 0: - raise MissingDependencyError( - f"Ran program '{program}' but it exited with an error:\n{e.output}" - ) from e - raise MissingDependencyError( - f"Could not find program '{program}' on the PATH" - ) from e - try: - version = re.match(regex, output.strip()).group(1) - except AttributeError as e: - raise MissingDependencyError( - f"The program '{program}' did not report its version. " - f"Message was:\n{output}" - ) - - return version - - -def shim_paths_with_program_files(env=None): - if not env: - env = os.environ - program_files = env.get('PROGRAMFILES', '') - if not program_files: - return env.get('PATH', '') - paths = [] - try: - for dirname in os.listdir(program_files): - if dirname.lower() == 'tesseract-ocr': - paths.append(os.path.join(program_files, dirname)) - elif dirname.lower() == 'gs': - try: - latest_gs = max( - os.listdir(os.path.join(program_files, dirname)), - key=lambda d: float(d[2:]), - ) - except (FileNotFoundError, NotADirectoryError): - continue - paths.append(os.path.join(program_files, dirname, latest_gs, 'bin')) - except EnvironmentError: - pass - paths.extend(path for path in os.get_exec_path(env) if path not in set(paths)) - return os.pathsep.join(paths) - - -missing_program = ''' -The program '{program}' could not be executed or was not found on your -system PATH. -''' - -missing_optional_program = ''' -The program '{program}' could not be executed or was not found on your -system PATH. This program is required when you use the -{required_for} arguments. You could try omitting these arguments, or install -the package. -''' - -missing_recommend_program = ''' -The program '{program}' could not be executed or was not found on your -system PATH. This program is recommended when using the {required_for} arguments, -but not required, so we will proceed. For best results, install the program. -''' - -old_version = ''' -OCRmyPDF requires '{program}' {need_version} or higher. Your system appears -to have {found_version}. Please update this program. -''' - -old_version_required_for = ''' -OCRmyPDF requires '{program}' {need_version} or higher when run with the -{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to -proceed. For best results, install the program. -''' - -osx_install_advice = ''' -If you have homebrew installed, try these command to install the missing -package: - brew install {package} -''' - -linux_install_advice = ''' -On systems with the aptitude package manager (Debian, Ubuntu), try these -commands: - sudo apt-get update - sudo apt-get install {package} - -On RPM-based systems (Red Hat, Fedora), search for instructions on -installing the RPM for {program}. -''' - -windows_install_advice = ''' -If not already installed, install the Chocolatey package manager. Then use -a command prompt to install the missing package: - choco install {package} -''' - - -def _get_platform(): - if sys.platform.startswith('freebsd'): - return 'freebsd' - elif sys.platform.startswith('linux'): - return 'linux' - elif sys.platform.startswith('win'): - return 'windows' - return sys.platform - - -def _error_trailer(program, package, **kwargs): - if isinstance(package, Mapping): - package = package.get(_get_platform(), program) - - if _get_platform() == 'darwin': - log.info(osx_install_advice.format(**locals())) - elif _get_platform() == 'linux': - log.info(linux_install_advice.format(**locals())) - elif _get_platform() == 'windows': - log.info(windows_install_advice.format(**locals())) - - -def _error_missing_program(program, package, required_for, recommended): - if required_for: - log.error(missing_optional_program.format(**locals())) - elif recommended: - log.info(missing_recommend_program.format(**locals())) - else: - log.error(missing_program.format(**locals())) - _error_trailer(**locals()) - - -def _error_old_version(program, package, need_version, found_version, required_for): - if required_for: - log.error(old_version_required_for.format(**locals())) - else: - log.error(old_version.format(**locals())) - _error_trailer(**locals()) - - -def check_external_program( - *, - program, - package, - version_checker, - need_version, - required_for=None, - recommended=False, -): - try: - found_version = version_checker() - except (CalledProcessError, FileNotFoundError, MissingDependencyError): - _error_missing_program(program, package, required_for, recommended) - if not recommended: - raise MissingDependencyError() - return - - def remove_leading_v(s): - if s.startswith('v'): - return s[1:] - return s - - found_version = remove_leading_v(found_version) - need_version = remove_leading_v(need_version) - - if LooseVersion(found_version) < LooseVersion(need_version): - _error_old_version(program, package, need_version, found_version, required_for) - if not recommended: - raise MissingDependencyError() - - log.debug('Found %s %s', program, found_version) +from ocrmypdf.exec._support import ( + check_external_program, + get_version, + run, + shim_paths_with_program_files, +) diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/exec/_support.py new file mode 100644 index 00000000..eba2a472 --- /dev/null +++ b/src/ocrmypdf/exec/_support.py @@ -0,0 +1,302 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +"""Wrappers to manage subprocess calls""" + +import logging +import os +import re +import shutil +import sys +from collections.abc import Mapping +from contextlib import suppress +from distutils.version import LooseVersion +from subprocess import PIPE, STDOUT, CalledProcessError +from subprocess import run as subprocess_run + +from ..exceptions import MissingDependencyError + +log = logging.getLogger(__name__) + + +def _get_program(args, env=None): + program = args[0] + test_path = env.get('_OCRMYPDF_TEST_PATH', '') + if test_path: + program = shutil.which(program, path=test_path) + return program + + +def run(args, *, env=None, **kwargs): + """Wrapper around subprocess.run() + + The main purpose of this wrapper is to allow us to substitute the main program + for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces + the main PATH as a location to check for programs to run. + + Secondly we have to account for behavioral differences in Windows in particular. + Creating symbolic links in Windows requires administrator privileges and + may not work if for some reason we're using a FAT file system or the temporary + folder is on a different drive from the working folder. The test suite + works around this by creating shim Python scripts that perform the same function + as a symbolic link, but those shims require support on this side, to ensure + we call them with Python. + + """ + if not env: + env = os.environ + + # Search in spoof path if necessary + program = _get_program(args, env) + args = [program] + args[1:] + + if os.name == 'nt': + args = fix_windows_args(program, args, env) + + log.debug("Running: %s", args) + process_log = log.getChild('subprocess.' + os.path.basename(program)) + if sys.version_info < (3, 7) and os.name == 'nt': + # Can't use close_fds=True on Windows with Python 3.6 or older + # https://bugs.python.org/issue19575, etc. + kwargs['close_fds'] = False + + stderr = None + try: + proc = subprocess_run(args, env=env, **kwargs) + except CalledProcessError as e: + stderr = getattr(e, 'stderr', None) + raise + else: + stderr = getattr(proc, 'stderr', None) + finally: + if process_log.isEnabledFor(logging.DEBUG) and stderr: + with suppress(AttributeError, UnicodeDecodeError): + stderr = stderr.decode('utf-8', 'replace') + process_log.debug("stderr = %s", stderr) + return proc + + +def fix_windows_args(program, args, env): + """Adjust our desired program and command line arguments for use on Windows""" + + if sys.version_info < (3, 8): + # bpo-33617 - Windows needs manual Path -> str conversion + args = [os.fspath(arg) for arg in args] + program = os.fspath(program) + + # If we are running a .py on Windows, ensure we call it with this Python + # (to support test suite shims) + if program.lower().endswith('.py'): + args = [sys.executable] + args + + paths = os.pathsep.join(os.get_exec_path(env)) + if not shutil.which(args[0], path=paths): + # If the program we want is not on the PATH, add some interesting + # locations in %PROGRAMFILES% to the PATH and try again + shimmed_path = shim_paths_with_program_files(env) + new_args0 = shutil.which(args[0], path=shimmed_path) + if new_args0: + args[0] = new_args0 + return args + + +def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): + """Get the version of the specified program""" + args_prog = [program, version_arg] + try: + proc = run( + args_prog, + close_fds=True, + universal_newlines=True, + stdout=PIPE, + stderr=STDOUT, + check=True, + env=env, + ) + output = proc.stdout + except FileNotFoundError as e: + raise MissingDependencyError( + f"Could not find program '{program}' on the PATH" + ) from e + except CalledProcessError as e: + if e.returncode != 0: + raise MissingDependencyError( + f"Ran program '{program}' but it exited with an error:\n{e.output}" + ) from e + raise MissingDependencyError( + f"Could not find program '{program}' on the PATH" + ) from e + try: + version = re.match(regex, output.strip()).group(1) + except AttributeError as e: + raise MissingDependencyError( + f"The program '{program}' did not report its version. " + f"Message was:\n{output}" + ) + + return version + + +def shim_paths_with_program_files(env=None): + if not env: + env = os.environ + program_files = env.get('PROGRAMFILES', '') + if not program_files: + return env.get('PATH', '') + paths = [] + try: + for dirname in os.listdir(program_files): + if dirname.lower() == 'tesseract-ocr': + paths.append(os.path.join(program_files, dirname)) + elif dirname.lower() == 'gs': + try: + latest_gs = max( + os.listdir(os.path.join(program_files, dirname)), + key=lambda d: float(d[2:]), + ) + except (FileNotFoundError, NotADirectoryError): + continue + paths.append(os.path.join(program_files, dirname, latest_gs, 'bin')) + except EnvironmentError: + pass + paths.extend(path for path in os.get_exec_path(env) if path not in set(paths)) + return os.pathsep.join(paths) + + +missing_program = ''' +The program '{program}' could not be executed or was not found on your +system PATH. +''' + +missing_optional_program = ''' +The program '{program}' could not be executed or was not found on your +system PATH. This program is required when you use the +{required_for} arguments. You could try omitting these arguments, or install +the package. +''' + +missing_recommend_program = ''' +The program '{program}' could not be executed or was not found on your +system PATH. This program is recommended when using the {required_for} arguments, +but not required, so we will proceed. For best results, install the program. +''' + +old_version = ''' +OCRmyPDF requires '{program}' {need_version} or higher. Your system appears +to have {found_version}. Please update this program. +''' + +old_version_required_for = ''' +OCRmyPDF requires '{program}' {need_version} or higher when run with the +{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to +proceed. For best results, install the program. +''' + +osx_install_advice = ''' +If you have homebrew installed, try these command to install the missing +package: + brew install {package} +''' + +linux_install_advice = ''' +On systems with the aptitude package manager (Debian, Ubuntu), try these +commands: + sudo apt-get update + sudo apt-get install {package} + +On RPM-based systems (Red Hat, Fedora), search for instructions on +installing the RPM for {program}. +''' + +windows_install_advice = ''' +If not already installed, install the Chocolatey package manager. Then use +a command prompt to install the missing package: + choco install {package} +''' + + +def _get_platform(): + if sys.platform.startswith('freebsd'): + return 'freebsd' + elif sys.platform.startswith('linux'): + return 'linux' + elif sys.platform.startswith('win'): + return 'windows' + return sys.platform + + +def _error_trailer(program, package, **kwargs): + if isinstance(package, Mapping): + package = package.get(_get_platform(), program) + + if _get_platform() == 'darwin': + log.info(osx_install_advice.format(**locals())) + elif _get_platform() == 'linux': + log.info(linux_install_advice.format(**locals())) + elif _get_platform() == 'windows': + log.info(windows_install_advice.format(**locals())) + + +def _error_missing_program(program, package, required_for, recommended): + if required_for: + log.error(missing_optional_program.format(**locals())) + elif recommended: + log.info(missing_recommend_program.format(**locals())) + else: + log.error(missing_program.format(**locals())) + _error_trailer(**locals()) + + +def _error_old_version(program, package, need_version, found_version, required_for): + if required_for: + log.error(old_version_required_for.format(**locals())) + else: + log.error(old_version.format(**locals())) + _error_trailer(**locals()) + + +def check_external_program( + *, + program, + package, + version_checker, + need_version, + required_for=None, + recommended=False, +): + try: + found_version = version_checker() + except (CalledProcessError, FileNotFoundError, MissingDependencyError): + _error_missing_program(program, package, required_for, recommended) + if not recommended: + raise MissingDependencyError() + return + + def remove_leading_v(s): + if s.startswith('v'): + return s[1:] + return s + + found_version = remove_leading_v(found_version) + need_version = remove_leading_v(need_version) + + if LooseVersion(found_version) < LooseVersion(need_version): + _error_old_version(program, package, need_version, found_version, required_for) + if not recommended: + raise MissingDependencyError() + + log.debug('Found %s %s', program, found_version) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 0a8f6cba..50e9b294 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -33,9 +33,9 @@ from io import BytesIO, UnsupportedOperation from os import fspath from tempfile import TemporaryFile -from .exceptions import MissingDependencyError -from .exec import shim_paths_with_program_files -from .lib._leptonica import ffi +from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.exec import shim_paths_with_program_files +from ocrmypdf.lib._leptonica import ffi # pylint: disable=protected-access From 7f67556995568ddef50bef76c21266dc0c1201b4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 01:35:45 -0700 Subject: [PATCH 466/880] ocrmypdf.__init__: Hide _HookimplMarker --- src/ocrmypdf/__init__.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 2326efb2..08da4d6f 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -16,7 +16,7 @@ # along with OCRmyPDF. If not, see . -from pluggy import HookimplMarker +from pluggy import HookimplMarker as _HookimplMarker from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo from ocrmypdf._version import PROGRAM_NAME, __version__ @@ -37,4 +37,4 @@ from ocrmypdf.exceptions import ( UnsupportedImageFormatError, ) -hookimpl = HookimplMarker('ocrmypdf') +hookimpl = _HookimplMarker('ocrmypdf') From a2d3e0b53ea8b4d5430788dbd9f72ff42fda1165 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 02:12:08 -0700 Subject: [PATCH 467/880] Convert remaining imports to absolute --- src/ocrmypdf/_validation.py | 13 +++++++++---- src/ocrmypdf/exec/_support.py | 2 +- src/ocrmypdf/exec/jbig2enc.py | 4 ++-- src/ocrmypdf/exec/pngquant.py | 4 ++-- src/ocrmypdf/exec/unpaper.py | 6 +++--- src/ocrmypdf/pdfinfo/__init__.py | 2 +- src/ocrmypdf/pdfinfo/layout.py | 2 +- 7 files changed, 19 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index d7acbf09..16096aa3 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -27,14 +27,14 @@ from shutil import copyfileobj import PIL -from ._unicodefun import verify_python3_env -from .exceptions import ( +from ocrmypdf._unicodefun import verify_python3_env +from ocrmypdf.exceptions import ( BadArgsError, InputFileError, MissingDependencyError, OutputFileAccessError, ) -from .exec import ( +from ocrmypdf.exec import ( check_external_program, ghostscript, jbig2enc, @@ -42,7 +42,12 @@ from .exec import ( tesseract, unpaper, ) -from .helpers import is_file_writable, is_iterable_notstr, monotonic, safe_symlink +from ocrmypdf.helpers import ( + is_file_writable, + is_iterable_notstr, + monotonic, + safe_symlink, +) # ------------- # External dependencies diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/exec/_support.py index eba2a472..bd8ed0e6 100644 --- a/src/ocrmypdf/exec/_support.py +++ b/src/ocrmypdf/exec/_support.py @@ -28,7 +28,7 @@ from distutils.version import LooseVersion from subprocess import PIPE, STDOUT, CalledProcessError from subprocess import run as subprocess_run -from ..exceptions import MissingDependencyError +from ocrmypdf.exceptions import MissingDependencyError log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index 5218edbd..979b8eb7 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -20,8 +20,8 @@ from functools import lru_cache from subprocess import PIPE -from ..exceptions import MissingDependencyError -from . import get_version, run +from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.exec import get_version, run @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index 17721065..f99fd421 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -23,8 +23,8 @@ from tempfile import NamedTemporaryFile from PIL import Image -from ..exceptions import MissingDependencyError -from . import get_version +from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.exec import get_version @lru_cache(maxsize=1) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 80243129..2f77701b 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -30,9 +30,9 @@ from tempfile import TemporaryDirectory from PIL import Image -from ..exceptions import MissingDependencyError, SubprocessOutputError -from . import get_version -from . import run as external_run +from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError +from ocrmypdf.exec import get_version +from ocrmypdf.exec import run as external_run log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index 093fea5e..0e8b8750 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -16,4 +16,4 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from .info import Colorspace, Encoding, PdfInfo +from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index db159c2e..2c3d2b2a 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -31,7 +31,7 @@ from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined from pdfminer.pdfpage import PDFPage from pdfminer.utils import bbox2str, matrix2str -from ..exceptions import EncryptedPdfError +from ocrmypdf.exceptions import EncryptedPdfError STRIP_NAME = re.compile(r'[0-9]+') From 6f5b75bcd04040b262600f9eaff781f99aadb679 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 02:23:56 -0700 Subject: [PATCH 468/880] Remove lru_cache on get_version Does not play well with forking. --- src/ocrmypdf/exec/_support.py | 1 + src/ocrmypdf/exec/ghostscript.py | 2 -- src/ocrmypdf/exec/jbig2enc.py | 2 -- src/ocrmypdf/exec/pngquant.py | 2 -- src/ocrmypdf/exec/unpaper.py | 2 -- 5 files changed, 1 insertion(+), 8 deletions(-) diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/exec/_support.py index bd8ed0e6..60d22a83 100644 --- a/src/ocrmypdf/exec/_support.py +++ b/src/ocrmypdf/exec/_support.py @@ -25,6 +25,7 @@ import sys from collections.abc import Mapping from contextlib import suppress from distutils.version import LooseVersion +from functools import lru_cache from subprocess import PIPE, STDOUT, CalledProcessError from subprocess import run as subprocess_run diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index b58e30cb..d7407726 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -21,7 +21,6 @@ import logging import os import re import warnings -from functools import lru_cache from io import BytesIO from os import fspath from pathlib import Path @@ -57,7 +56,6 @@ if os.name == 'nt': GS = Path(GS).stem -@lru_cache(maxsize=1) def version(): return get_version(GS) diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index 979b8eb7..c027e28d 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -17,14 +17,12 @@ """Interface to jbig2 executable""" -from functools import lru_cache from subprocess import PIPE from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.exec import get_version, run -@lru_cache(maxsize=1) def version(): return get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*') diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index f99fd421..ad00560f 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -17,7 +17,6 @@ """Interface to pngquant executable""" -from functools import lru_cache from subprocess import run from tempfile import NamedTemporaryFile @@ -27,7 +26,6 @@ from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.exec import get_version -@lru_cache(maxsize=1) def version(): return get_version('pngquant', regex=r'(\d+(\.\d+)*).*') diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index 2f77701b..a1aa749e 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -23,7 +23,6 @@ import logging import os import shlex -from functools import lru_cache from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory @@ -37,7 +36,6 @@ from ocrmypdf.exec import run as external_run log = logging.getLogger(__name__) -@lru_cache(maxsize=1) def version(): return get_version('unpaper') From d372f1f7fa70ce29b49cb29efa475d10243de5c9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 12 May 2020 04:09:29 -0700 Subject: [PATCH 469/880] Remove "skip page" from tesseract interface Breaks tests/test_main.py::test_tesseract_missing_tessdata because conftest.py does not update options.tesseract_env before testing options for some reason, and tesseract.has_textonly_pdf raises an exception instead of returning False as the test assumes. --- src/ocrmypdf/_pipeline.py | 2 -- src/ocrmypdf/exec/tesseract.py | 23 ++++++----------------- tests/test_tesseract.py | 4 +--- 3 files changed, 7 insertions(+), 22 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 477ea703..162677df 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -616,12 +616,10 @@ def ocr_tesseract_textonly_pdf(input_image, page_context): options = page_context.options tesseract.generate_pdf( input_image=input_image, - skip_pdf=None, output_pdf=output_pdf, output_text=output_text, language=options.language, engine_mode=options.tesseract_oem, - text_only=True, tessconfig=options.tesseract_config, timeout=options.tesseract_timeout, pagesegmode=options.tesseract_pagesegmode, diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index aea9a0ce..57450937 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -302,29 +302,20 @@ def generate_hocr( shutil.move(prefix.with_suffix('.txt'), output_sidecar) -def use_skip_page(text_only, skip_pdf, output_pdf, output_text): +def use_skip_page(output_pdf, output_text): output_text.write_text('[skipped page]', encoding='utf-8') - if skip_pdf and not text_only: - # Substitute a "skipped page" - with suppress(FileNotFoundError): - output_pdf.unlink() # In case it was partially created - safe_symlink(skip_pdf, output_pdf) - return - - # Or normally, just write a 0 byte file to the output to indicate a skip + # A 0 byte file to the output to indicate a skip output_pdf.write_bytes(b'') def generate_pdf( *, input_image: Path, - skip_pdf: Optional[Path] = None, output_pdf: Path, output_text: Path, language: List[str], engine_mode, - text_only: bool, tessconfig: List[str], timeout: float, pagesegmode: int, @@ -335,12 +326,10 @@ def generate_pdf( """Use Tesseract to render a PDF. input_image -- image to analyze - skip_pdf -- if we time out, use this file as output output_pdf -- file to generate output_text -- OCR text file language -- list of languages to consider engine_mode -- engine mode argument for tess v4 - text_only -- enable tesseract text only mode? tessconfig -- tesseract configuration timeout -- timeout (seconds) log -- logger object @@ -351,8 +340,8 @@ def generate_pdf( if pagesegmode is not None: args_tesseract.extend(['--psm', str(pagesegmode)]) - if text_only and has_textonly_pdf(tesseract_env, language): - args_tesseract.extend(['-c', 'textonly_pdf=1']) + # has_textonly_pdf(tesseract_env=tesseract_env, langs=language) + args_tesseract.extend(['-c', 'textonly_pdf=1']) if user_words: args_tesseract.extend(['--user-words', user_words]) @@ -380,11 +369,11 @@ def generate_pdf( shutil.move(prefix + '.txt', output_text) except TimeoutExpired: page_timedout(timeout) - use_skip_page(text_only, skip_pdf, output_pdf, output_text) + use_skip_page(output_pdf, output_text) except CalledProcessError as e: tesseract_log_output(e.output) if b'Image too large' in e.output: - use_skip_page(text_only, skip_pdf, output_pdf, output_text) + use_skip_page(output_pdf, output_text) return raise SubprocessOutputError() from e else: diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index 3f921de8..9c37dde1 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -108,12 +108,10 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): monkeypatch.setattr(tesseract, 'run', dummy_run) tesseract.generate_pdf( input_image=resources / 'crom.png', - skip_pdf=resources / 'blank.pdf', output_pdf=outdir / 'pdf.pdf', output_text=outdir / 'txt.txt', language=['eng'], engine_mode=None, - text_only=False, tessconfig=[], timeout=180.0, pagesegmode=None, @@ -123,7 +121,7 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): ) assert Path(outdir / 'txt.txt').read_text() == '[skipped page]' if os.name != 'nt': # different semantics - assert Path(outdir / 'pdf.pdf').samefile(resources / 'blank.pdf') + assert Path(outdir / 'pdf.pdf').stat().st_size == 0 def test_timeout(caplog): From 12a2f78c4dda83c50ba1b484a0f29f5adc8a0a6e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 14 May 2020 03:19:22 -0700 Subject: [PATCH 470/880] Fix validation of languages not using tesseract_env And some related issues. --- src/ocrmypdf/_validation.py | 4 ++-- tests/conftest.py | 3 ++- tests/test_main.py | 8 ++++---- 3 files changed, 8 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 16096aa3..2fc85928 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -84,12 +84,12 @@ def check_options_languages(options): options.language = options.language[0].split('+') languages = set(options.language) - if not languages.issubset(tesseract.languages()): + if not languages.issubset(tesseract.languages(options.tesseract_env)): msg = ( "The installed version of tesseract does not have language " "data for the following requested languages: \n" ) - for lang in languages - tesseract.languages(): + for lang in languages - tesseract.languages(options.tesseract_env): msg += lang + '\n' raise MissingDependencyError(msg) diff --git a/tests/conftest.py b/tests/conftest.py index 9599369a..6ea95e96 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -241,7 +241,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): [str(input_file), str(output_file)] + [str(arg) for arg in args if arg is not None] ) - api.check_options(options) + if env: options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) @@ -252,6 +252,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) + api.check_options(options) return api.run_pipeline(options, plugin_manager=None, api=False) diff --git a/tests/test_main.py b/tests/test_main.py index 9fcb49d7..0fc11e76 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -186,10 +186,10 @@ def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir): env = os.environ.copy() env['TESSDATA_PREFIX'] = os.fspath(tmpdir) - returncode = run_ocrmypdf_api( - resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env - ) - assert returncode == ExitCode.missing_dependency + with pytest.raises(MissingDependencyError): + run_ocrmypdf_api( + resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env + ) def test_invalid_input_pdf(resources, no_outpdf): From 41eb54cc0a59855aaa2d5d03001f80507d2fb2b6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 14 May 2020 03:23:25 -0700 Subject: [PATCH 471/880] Standardize tesseract.generate_hocr and _pdf parameters --- src/ocrmypdf/_pipeline.py | 8 ++++---- src/ocrmypdf/_validation.py | 4 ++-- src/ocrmypdf/exec/tesseract.py | 35 +++++++++++++++++----------------- tests/conftest.py | 1 - tests/test_main.py | 2 +- tests/test_tesseract.py | 10 +++++----- 6 files changed, 29 insertions(+), 31 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 162677df..6350ea69 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -538,8 +538,8 @@ def ocr_tesseract_hocr(input_file, page_context): tesseract.generate_hocr( input_file=input_file, output_hocr=hocr_out, - output_sidecar=hocr_text_out, - language=options.language, + output_text=hocr_text_out, + languages=options.language, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, timeout=options.tesseract_timeout, @@ -615,10 +615,10 @@ def ocr_tesseract_textonly_pdf(input_image, page_context): output_text = page_context.get_path('ocr_tess.txt') options = page_context.options tesseract.generate_pdf( - input_image=input_image, + input_file=input_image, output_pdf=output_pdf, output_text=output_text, - language=options.language, + languages=options.language, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, timeout=options.tesseract_timeout, diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 2fc85928..4ef6fcdf 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -84,12 +84,12 @@ def check_options_languages(options): options.language = options.language[0].split('+') languages = set(options.language) - if not languages.issubset(tesseract.languages(options.tesseract_env)): + if not languages.issubset(tesseract.get_languages(options.tesseract_env)): msg = ( "The installed version of tesseract does not have language " "data for the following requested languages: \n" ) - for lang in languages - tesseract.languages(options.tesseract_env): + for lang in languages - tesseract.get_languages(options.tesseract_env): msg += lang + '\n' raise MissingDependencyError(msg) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 57450937..39046ce5 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -89,7 +89,8 @@ def has_textonly_pdf(tesseract_env=None, langs=None): params = proc.stdout except CalledProcessError as e: raise MissingDependencyError( - "Could not --print-parameters from tesseract" + "Could not --print-parameters from tesseract. This can happen if the " + "TESSDATA_PREFIX environment is not set to a valid tessdata folder. " ) from e if 'textonly_pdf' in params: return True @@ -105,7 +106,7 @@ def has_user_words(tesseract_env=None): return version(tesseract_env) >= '4.1' -def languages(tesseract_env=None): +def get_languages(tesseract_env=None): def lang_error(output): msg = ( "Tesseract failed to report available languages.\n" @@ -232,21 +233,21 @@ def page_timedout(timeout): log.warning("[tesseract] took too long to OCR - skipping") -def _generate_null_hocr(output_hocr, output_sidecar, image): +def _generate_null_hocr(output_hocr, output_text, image): """Produce a .hocr file that reports no text detected on a page that is the same size as the input image.""" with Image.open(image) as im: w, h = im.size output_hocr.write_text(HOCR_TEMPLATE.format(w, h), encoding='utf-8') - output_sidecar.write_text('[skipped page]', encoding='utf-8') + output_text.write_text('[skipped page]', encoding='utf-8') def generate_hocr( input_file: Path, output_hocr: Path, - output_sidecar: Path, - language: list, + output_text: Path, + languages: list, engine_mode, tessconfig: list, timeout: float, @@ -257,7 +258,7 @@ def generate_hocr( ): prefix = output_hocr.with_suffix('') - args_tesseract = tess_base_args(language, engine_mode) + args_tesseract = tess_base_args(languages, engine_mode) if pagesegmode is not None: args_tesseract.extend(['--psm', str(pagesegmode)]) @@ -286,11 +287,11 @@ def generate_hocr( # Temporary workaround to hocrTransform not being able to function if # it does not have a valid hOCR file. page_timedout(timeout) - _generate_null_hocr(output_hocr, output_sidecar, input_file) + _generate_null_hocr(output_hocr, output_text, input_file) except CalledProcessError as e: tesseract_log_output(e.output) if b'Image too large' in e.output: - _generate_null_hocr(output_hocr, output_sidecar, input_file) + _generate_null_hocr(output_hocr, output_text, input_file) return raise SubprocessOutputError() from e @@ -299,7 +300,7 @@ def generate_hocr( # The sidecar text file will get the suffix .txt; rename it to # whatever caller wants it named if prefix.with_suffix('.txt').exists(): - shutil.move(prefix.with_suffix('.txt'), output_sidecar) + shutil.move(prefix.with_suffix('.txt'), output_text) def use_skip_page(output_pdf, output_text): @@ -311,10 +312,10 @@ def use_skip_page(output_pdf, output_text): def generate_pdf( *, - input_image: Path, + input_file: Path, output_pdf: Path, output_text: Path, - language: List[str], + languages: List[str], engine_mode, tessconfig: List[str], timeout: float, @@ -325,22 +326,20 @@ def generate_pdf( ): """Use Tesseract to render a PDF. - input_image -- image to analyze + input_file -- image to analyze output_pdf -- file to generate output_text -- OCR text file - language -- list of languages to consider + languages -- list of languages to consider engine_mode -- engine mode argument for tess v4 tessconfig -- tesseract configuration timeout -- timeout (seconds) - log -- logger object """ - args_tesseract = tess_base_args(language, engine_mode) + args_tesseract = tess_base_args(languages, engine_mode) if pagesegmode is not None: args_tesseract.extend(['--psm', str(pagesegmode)]) - # has_textonly_pdf(tesseract_env=tesseract_env, langs=language) args_tesseract.extend(['-c', 'textonly_pdf=1']) if user_words: @@ -354,7 +353,7 @@ def generate_pdf( # Reminder: test suite tesseract spoofers might break after any changes # to the number of order parameters here - args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig) + args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig) try: p = run( args_tesseract, diff --git a/tests/conftest.py b/tests/conftest.py index 6ea95e96..8eaede2a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -241,7 +241,6 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): [str(input_file), str(output_file)] + [str(arg) for arg in args if arg is not None] ) - if env: options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) diff --git a/tests/test_main.py b/tests/test_main.py index 0fc11e76..c3a715ce 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -233,7 +233,7 @@ def test_german(spoof_tesseract_cache, resources, outdir): env=spoof_tesseract_cache, ) except MissingDependencyError: - if 'deu' not in tesseract.languages(): + if 'deu' not in tesseract.get_languages(): pytest.xfail(reason="tesseract-deu language pack not installed") raise diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index 9c37dde1..ffe3fe31 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -77,7 +77,7 @@ def test_no_languages(tmp_path): env['TESSDATA_PREFIX'] = fspath(tmp_path) with pytest.raises(MissingDependencyError): - tesseract.languages(tesseract_env=env) + tesseract.get_languages(tesseract_env=env) def test_image_too_large_hocr(monkeypatch, resources, outdir): @@ -88,8 +88,8 @@ def test_image_too_large_hocr(monkeypatch, resources, outdir): tesseract.generate_hocr( input_file=resources / 'crom.png', output_hocr=outdir / 'out.hocr', - output_sidecar=outdir / 'out.txt', - language=['eng'], + output_text=outdir / 'out.txt', + languages=['eng'], engine_mode=None, tessconfig=[], timeout=180.0, @@ -107,10 +107,10 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): monkeypatch.setattr(tesseract, 'run', dummy_run) tesseract.generate_pdf( - input_image=resources / 'crom.png', + input_file=resources / 'crom.png', output_pdf=outdir / 'pdf.pdf', output_text=outdir / 'txt.txt', - language=['eng'], + languages=['eng'], engine_mode=None, tessconfig=[], timeout=180.0, From 8174089c8be254152b4f9b80f2833f108bdb0ff9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 14 May 2020 03:54:21 -0700 Subject: [PATCH 472/880] Begin transforming Tesseract into pluggable OCR engine --- src/ocrmypdf/_pipeline.py | 37 ++++------ src/ocrmypdf/_plugin_manager.py | 4 +- src/ocrmypdf/_sync.py | 10 ++- src/ocrmypdf/builtin_plugins/__init__.py | 18 +++++ src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 68 +++++++++++++++++++ src/ocrmypdf/pluginspec.py | 36 +++++++++- 6 files changed, 140 insertions(+), 33 deletions(-) create mode 100644 src/ocrmypdf/builtin_plugins/__init__.py create mode 100644 src/ocrmypdf/builtin_plugins/tesseract_ocr.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6350ea69..21a92202 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -379,11 +379,8 @@ def get_orientation_correction(preview, page_context): """ - orient_conf = tesseract.get_orientation( - preview, - engine_mode=page_context.options.tesseract_oem, - timeout=page_context.options.tesseract_timeout, - tesseract_env=page_context.options.tesseract_env, + orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation( + preview, page_context.options ) correction = orient_conf.angle % 360 @@ -531,22 +528,17 @@ def create_ocr_image(image, page_context): return output_file -def ocr_tesseract_hocr(input_file, page_context): +def ocr_engine_hocr(input_file, page_context): hocr_out = page_context.get_path('ocr_hocr.hocr') hocr_text_out = page_context.get_path('ocr_hocr.txt') options = page_context.options - tesseract.generate_hocr( + + ocr_engine = page_context.plugin_manager.hook.get_ocr_engine() + ocr_engine.generate_hocr( input_file=input_file, output_hocr=hocr_out, output_text=hocr_text_out, - languages=options.language, - engine_mode=options.tesseract_oem, - tessconfig=options.tesseract_config, - timeout=options.tesseract_timeout, - pagesegmode=options.tesseract_pagesegmode, - user_words=options.user_words, - user_patterns=options.user_patterns, - tesseract_env=options.tesseract_env, + options=options, ) return (hocr_out, hocr_text_out) @@ -610,22 +602,17 @@ def render_hocr_page(hocr, page_context): return output_file -def ocr_tesseract_textonly_pdf(input_image, page_context): +def ocr_engine_textonly_pdf(input_image, page_context): output_pdf = page_context.get_path('ocr_tess.pdf') output_text = page_context.get_path('ocr_tess.txt') options = page_context.options - tesseract.generate_pdf( + + ocr_engine = page_context.plugin_manager.hook.get_ocr_engine() + ocr_engine.generate_pdf( input_file=input_image, output_pdf=output_pdf, output_text=output_text, - languages=options.language, - engine_mode=options.tesseract_oem, - tessconfig=options.tesseract_config, - timeout=options.tesseract_timeout, - pagesegmode=options.tesseract_pagesegmode, - user_words=options.user_words, - user_patterns=options.user_patterns, - tesseract_env=options.tesseract_env, + options=options, ) return (output_pdf, output_text) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 9ccfc3b0..1edd2483 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -26,9 +26,11 @@ import pluggy from ocrmypdf import pluginspec -def get_plugin_manager(plugins: List[str]): +def get_plugin_manager(plugins: List[str], builtins=True): pm = pluggy.PluginManager('ocrmypdf') pm.add_hookspecs(pluginspec) + if builtins: + plugins.insert(0, 'ocrmypdf.builtin_plugins') for name in plugins: if name.endswith('.py'): # Import by filename diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 710bda5c..3f97a301 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -42,8 +42,8 @@ from ocrmypdf._pipeline import ( is_ocr_required, merge_sidecars, metadata_fixup, - ocr_tesseract_hocr, - ocr_tesseract_textonly_pdf, + ocr_engine_hocr, + ocr_engine_textonly_pdf, optimize_pdf, preprocess_clean, preprocess_deskew, @@ -176,13 +176,11 @@ def exec_page_sync(page_context): ) if options.pdf_renderer == 'hocr': - (hocr_out, text_out) = ocr_tesseract_hocr(ocr_image_out, page_context) + (hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context) ocr_out = render_hocr_page(hocr_out, page_context) if options.pdf_renderer == 'sandwich': - (ocr_out, text_out) = ocr_tesseract_textonly_pdf( - ocr_image_out, page_context - ) + (ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context) return PageResult( pageno=page_context.pageno, diff --git a/src/ocrmypdf/builtin_plugins/__init__.py b/src/ocrmypdf/builtin_plugins/__init__.py new file mode 100644 index 00000000..e5fd494e --- /dev/null +++ b/src/ocrmypdf/builtin_plugins/__init__.py @@ -0,0 +1,18 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from ocrmypdf.builtin_plugins.tesseract_ocr import * diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py new file mode 100644 index 00000000..12e8d667 --- /dev/null +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -0,0 +1,68 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from ocrmypdf import hookimpl +from ocrmypdf.exec import tesseract +from ocrmypdf.pluginspec import OcrEngine + + +class TesseractOcrEngine(OcrEngine): + def languages(self): + return tesseract.get_languages() + + def get_orientation(self, input_file, options): + return tesseract.get_orientation( + input_file, + engine_mode=options.tesseract_oem, + timeout=options.tesseract_timeout, + tesseract_env=options.tesseract_env, + ) + + def generate_hocr(self, input_file, output_hocr, output_text, options): + tesseract.generate_hocr( + input_file=input_file, + output_hocr=output_hocr, + output_text=output_text, + languages=options.language, + engine_mode=options.tesseract_oem, + tessconfig=options.tesseract_config, + timeout=options.tesseract_timeout, + pagesegmode=options.tesseract_pagesegmode, + user_words=options.user_words, + user_patterns=options.user_patterns, + tesseract_env=options.tesseract_env, + ) + + def generate_pdf(self, input_file, output_pdf, output_text, options): + tesseract.generate_pdf( + input_file=input_file, + output_pdf=output_pdf, + output_text=output_text, + languages=options.language, + engine_mode=options.tesseract_oem, + tessconfig=options.tesseract_config, + timeout=options.tesseract_timeout, + pagesegmode=options.tesseract_pagesegmode, + user_words=options.user_words, + user_patterns=options.user_patterns, + tesseract_env=options.tesseract_env, + ) + + +@hookimpl +def get_ocr_engine(): + return TesseractOcrEngine() diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 6063ce54..0261ba75 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -15,9 +15,11 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +from abc import ABC, abstractmethod from argparse import ArgumentParser, Namespace +from collections import namedtuple from pathlib import Path -from typing import Optional +from typing import AbstractSet, Optional import pluggy from PIL import Image @@ -87,3 +89,35 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: Note that the ocrmypdf image optimization stage may ultimately chose a different format. """ + + +OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) + + +class OcrEngine(ABC): + @abstractmethod + def languages(self) -> AbstractSet[str]: + """Returns set of languages that are supported.""" + + @abstractmethod + def get_orientation( + self, input_file: Path, options: Namespace + ) -> OrientationConfidence: + """Returns the orientation of the image.""" + + @abstractmethod + def generate_hocr( + self, input_file: Path, output_hocr: Path, output_text: Path, options: Namespace + ) -> None: + pass + + @abstractmethod + def generate_pdf( + self, input_file: Path, output_pdf: Path, output_text: Path, options: Namespace + ) -> None: + pass + + +@hookspec(firstresult=True) +def get_ocr_engine() -> OcrEngine: + pass From 9af94ac9b7b75601689443f146052e9e45876998 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 14 May 2020 04:23:23 -0700 Subject: [PATCH 473/880] pipeline: use OCR engine abstraction instead of Tesseract --- src/ocrmypdf/_pipeline.py | 36 ++++++++----------- src/ocrmypdf/_plugin_manager.py | 7 ++-- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 21 ++++++++--- src/ocrmypdf/pluginspec.py | 32 ++++++++++------- tests/test_metadata.py | 9 +++-- 5 files changed, 63 insertions(+), 42 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 21a92202..0806880e 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -37,7 +37,7 @@ from ocrmypdf.exceptions import ( PriorOcrFoundError, UnsupportedImageFormatError, ) -from ocrmypdf.exec import ghostscript, tesseract, unpaper +from ocrmypdf.exec import ghostscript, unpaper from ocrmypdf.helpers import Resolution, safe_symlink from ocrmypdf.hocrtransform import HocrTransform from ocrmypdf.optimize import optimize @@ -362,21 +362,19 @@ def describe_rotation(page_context, orient_conf, correction): def get_orientation_correction(preview, page_context): - """ - Work out orientation correct for each page. + """Work out orientation correct for each page. We ask Ghostscript to draw a preview page, which will rasterize with the - current /Rotate applied, and then ask Tesseract which way the page is + current /Rotate applied, and then ask OCR which way the page is oriented. If the value of /Rotate is correct (e.g., a user already - manually fixed rotation), then Tesseract will say the page is pointing + manually fixed rotation), then OCR will say the page is pointing up and the correction is zero. Otherwise, the orientation found by - Tesseract represents the clockwise rotation, or the counterclockwise + OCR represents the clockwise rotation, or the counterclockwise correction to rotation. When we draw the real page for OCR, we rotate it by the CCW correction, which points it (hopefully) upright. _graft.py takes care of the orienting the image and text layers. - """ orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation( @@ -555,7 +553,7 @@ def create_visible_page_jpg(image, page_context): # might have removed the DPI information. In this case, fall back to # square DPI used to rasterize. When the preview image was # rasterized, it was also converted to square resolution, which is - # what we want to give tesseract, so keep it square. + # what we want to give to the OCR engine, so keep it square. if 'dpi' in im.info: dpi = Resolution(*im.info['dpi']) else: @@ -617,7 +615,9 @@ def ocr_engine_textonly_pdf(input_image, page_context): return (output_pdf, output_text) -def get_docinfo(base_pdf, options): +def get_docinfo(base_pdf, context): + options = context.options + def from_document_info(key): try: s = base_pdf.docinfo[key] @@ -629,7 +629,6 @@ def get_docinfo(base_pdf, options): k: from_document_info(k) for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate') } - renderer_tag = 'OCR' if options is not None: if options.title: pdfmark['/Title'] = options.title @@ -640,12 +639,9 @@ def get_docinfo(base_pdf, options): if options.subject: pdfmark['/Subject'] = options.subject - if options.pdf_renderer == 'sandwich': - renderer_tag = 'OCR-PDF' + creator_tag = context.plugin_manager.hook.get_ocr_engine().creator_tag(options) - pdfmark['/Creator'] = ( - f'{PROGRAM_NAME} {VERSION} / ' f'Tesseract {renderer_tag} {tesseract.version()}' - ) + pdfmark['/Creator'] = f'{PROGRAM_NAME} {VERSION} / {creator_tag}' pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}' if 'OCRMYPDF_CREATOR' in os.environ: pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR'] @@ -732,7 +728,7 @@ def metadata_fixup(working_file, context): log.info("The following metadata fields were not copied: %r", missing) with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf: - docinfo = get_docinfo(original, options) + docinfo = get_docinfo(original, context) with pdf.open_metadata() as meta: meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False) # If xmp:CreateDate is missing, set it to the modify date to @@ -780,11 +776,9 @@ def merge_sidecars(txt_files, context): if txt_file: with open(txt_file, 'r', encoding="utf-8") as in_: txt = in_.read() - # Tesseract v4 alpha started adding form feeds in - # commit aa6eb6b - # No obvious way to detect what binaries will do this, so - # for consistency just ignore its form feeds and insert our - # own + # Some OCR engines (e.g. Tesseract v4 alpha) add form feeds + # between pages, and some do not. For consistency, we ignore + # any added by the OCR engine and them on our own. if txt.endswith('\f'): stream.write(txt[:-1]) else: diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 1edd2483..e42ef94b 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -29,9 +29,12 @@ from ocrmypdf import pluginspec def get_plugin_manager(plugins: List[str], builtins=True): pm = pluggy.PluginManager('ocrmypdf') pm.add_hookspecs(pluginspec) + if builtins: - plugins.insert(0, 'ocrmypdf.builtin_plugins') - for name in plugins: + all_plugins = ['ocrmypdf.builtin_plugins'] + plugins + else: + all_plugins = plugins + for name in all_plugins: if name.endswith('.py'): # Import by filename module_name = Path(name).stem diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 12e8d667..d68a2946 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -21,10 +21,21 @@ from ocrmypdf.pluginspec import OcrEngine class TesseractOcrEngine(OcrEngine): - def languages(self): + @staticmethod + def version(): + return tesseract.version() + + @staticmethod + def creator_tag(options): + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}" + + @staticmethod + def languages(): return tesseract.get_languages() - def get_orientation(self, input_file, options): + @staticmethod + def get_orientation(input_file, options): return tesseract.get_orientation( input_file, engine_mode=options.tesseract_oem, @@ -32,7 +43,8 @@ class TesseractOcrEngine(OcrEngine): tesseract_env=options.tesseract_env, ) - def generate_hocr(self, input_file, output_hocr, output_text, options): + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): tesseract.generate_hocr( input_file=input_file, output_hocr=output_hocr, @@ -47,7 +59,8 @@ class TesseractOcrEngine(OcrEngine): tesseract_env=options.tesseract_env, ) - def generate_pdf(self, input_file, output_pdf, output_text, options): + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): tesseract.generate_pdf( input_file=input_file, output_pdf=output_pdf, diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 0261ba75..366c51d3 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -15,7 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from abc import ABC, abstractmethod +from abc import ABC, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple from pathlib import Path @@ -95,27 +95,33 @@ OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidenc class OcrEngine(ABC): - @abstractmethod - def languages(self) -> AbstractSet[str]: + @abstractstaticmethod + def version() -> str: + """Returns the version of the OCR engine.""" + + @abstractstaticmethod + def creator_tag(options) -> str: + """Returns the creator tag to identify this software's role in creating the PDF.""" + + @abstractstaticmethod + def languages() -> AbstractSet[str]: """Returns set of languages that are supported.""" - @abstractmethod - def get_orientation( - self, input_file: Path, options: Namespace - ) -> OrientationConfidence: + @abstractstaticmethod + def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence: """Returns the orientation of the image.""" - @abstractmethod + @abstractstaticmethod def generate_hocr( - self, input_file: Path, output_hocr: Path, output_text: Path, options: Namespace + input_file: Path, output_hocr: Path, output_text: Path, options: Namespace ) -> None: - pass + """Called to produce a hOCR file.""" - @abstractmethod + @abstractstaticmethod def generate_pdf( - self, input_file: Path, output_pdf: Path, output_text: Path, options: Namespace + input_file: Path, output_pdf: Path, output_text: Path, options: Namespace ) -> None: - pass + """Called to produce a text only PDF (no image, invisible text).""" @hookspec(firstresult=True) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index acb8f31c..795dcc43 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -29,6 +29,7 @@ from pikepdf.models.metadata import decode_pdf_date from ocrmypdf._jobcontext import PdfContext from ocrmypdf._pipeline import convert_to_pdfa, metadata_fixup +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps @@ -291,7 +292,9 @@ def test_metadata_fixup_warning(resources, outdir, caplog): copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') - context = PdfContext(options, outdir, outdir / 'graph.pdf', None, None) + context = PdfContext( + options, outdir, outdir / 'graph.pdf', None, get_plugin_manager([]) + ) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) for record in caplog.records: assert record.levelname != 'WARNING' @@ -302,7 +305,9 @@ def test_metadata_fixup_warning(resources, outdir, caplog): meta['prism2:publicationName'] = 'OCRmyPDF Test' graph.save(outdir / 'graph_mod.pdf') - context = PdfContext(options, outdir, outdir / 'graph_mod.pdf', None, None) + context = PdfContext( + options, outdir, outdir / 'graph_mod.pdf', None, get_plugin_manager([]) + ) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) assert any(record.levelname == 'WARNING' for record in caplog.records) From 2bd586e093a191fb6e9e939b56eb801de5069035 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 May 2020 01:50:37 -0700 Subject: [PATCH 474/880] Compare requested languages to OCR engine instead of tesseract directly Also refactoring to facilitating validation needing the plugin manager. --- src/ocrmypdf/__main__.py | 2 +- src/ocrmypdf/_validation.py | 16 +++++++++------- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 7 +++++-- src/ocrmypdf/pluginspec.py | 8 ++++++-- tests/conftest.py | 9 ++++++--- tests/test_unpaper.py | 6 ++++-- tests/test_validation.py | 9 ++++++--- 8 files changed, 38 insertions(+), 21 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 01f0a478..1cf84daf 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -58,7 +58,7 @@ def run(args=None): ) log.debug('ocrmypdf %s', __version__) try: - check_options(options) + check_options(options, plugin_manager) except ValueError as e: log.error(e) return ExitCode.bad_args diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 4ef6fcdf..bfd52e69 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -27,6 +27,7 @@ from shutil import copyfileobj import PIL +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._unicodefun import verify_python3_env from ocrmypdf.exceptions import ( BadArgsError, @@ -72,7 +73,7 @@ def check_platform(): ) -def check_options_languages(options): +def check_options_languages(options, plugin_manager): if not options.language: options.language = [DEFAULT_LANGUAGE] system_lang = locale.getlocale()[0] @@ -84,12 +85,13 @@ def check_options_languages(options): options.language = options.language[0].split('+') languages = set(options.language) - if not languages.issubset(tesseract.get_languages(options.tesseract_env)): + ocr_engine = plugin_manager.hook.get_ocr_engine() + if not languages.issubset(ocr_engine.languages(options)): msg = ( - "The installed version of tesseract does not have language " - "data for the following requested languages: \n" + f"{ocr_engine} does not have language data for the following " + "requested languages: \n" ) - for lang in languages - tesseract.get_languages(options.tesseract_env): + for lang in languages - ocr_engine.languages(options): msg += lang + '\n' raise MissingDependencyError(msg) @@ -308,9 +310,9 @@ def check_options_pillow(options): PIL.Image.MAX_IMAGE_PIXELS = None -def check_options(options): +def check_options(options, plugin_manager): check_platform() - check_options_languages(options) + check_options_languages(options, plugin_manager) check_options_metadata(options) check_options_output(options) check_options_sidecar(options) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index f53107fa..7edce482 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -279,5 +279,5 @@ def ocr( # pylint: disable=unused-argument options = create_options( **{k: v for k, v in locals().items() if not k.startswith('_')} ) - check_options(options) + check_options(options, _plugin_manager) return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index d68a2946..85b2e4c8 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -30,9 +30,12 @@ class TesseractOcrEngine(OcrEngine): tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}" + def __str__(self): + return f"Tesseract OCR {TesseractOcrEngine.version()}" + @staticmethod - def languages(): - return tesseract.get_languages() + def languages(options): + return tesseract.get_languages(options.tesseract_env) @staticmethod def get_orientation(input_file, options): diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 366c51d3..5e0a4f88 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -100,11 +100,15 @@ class OcrEngine(ABC): """Returns the version of the OCR engine.""" @abstractstaticmethod - def creator_tag(options) -> str: + def creator_tag(options: Namespace) -> str: """Returns the creator tag to identify this software's role in creating the PDF.""" @abstractstaticmethod - def languages() -> AbstractSet[str]: + def __str__(self): + """Returns name of OCR engine and version.""" + + @abstractstaticmethod + def languages(options: Namespace) -> AbstractSet[str]: """Returns set of languages that are supported.""" @abstractstaticmethod diff --git a/tests/conftest.py b/tests/conftest.py index 8eaede2a..55bd8376 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -25,6 +25,7 @@ from subprocess import PIPE, run import pytest from ocrmypdf import api, cli, pdfinfo +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.exec import unpaper pytest_plugins = ['helpers_namespace'] @@ -217,11 +218,12 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): [str(input_file), str(output_file)] + [str(arg) for arg in args if arg is not None] ) - api.check_options(options) + plugin_manager = get_plugin_manager(options.plugins) + api.check_options(options, plugin_manager) if env: options.tesseract_env = env options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - result = api.run_pipeline(options, plugin_manager=None, api=True) + result = api.run_pipeline(options, plugin_manager=plugin_manager, api=True) assert result == 0 assert output_file.exists(), "Output file not created" @@ -251,7 +253,8 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) - api.check_options(options) + plugin_manager = get_plugin_manager(options.plugins) + api.check_options(options, plugin_manager) return api.run_pipeline(options, plugin_manager=None, api=False) diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 836ef0a8..e87800c9 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -20,6 +20,7 @@ from unittest.mock import patch import pytest +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._validation import check_options from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError @@ -43,11 +44,12 @@ def test_no_unpaper(resources, no_outpdf): input_ = fspath(resources / "c02-22.pdf") output = fspath(no_outpdf) options = get_parser().parse_args(args=["--clean", input_, output]) - + plugin_manager = get_plugin_manager(options.plugins) with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") + with pytest.raises(MissingDependencyError): - check_options(options) + check_options(options, plugin_manager) def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): diff --git a/tests/test_validation.py b/tests/test_validation.py index 35c21fa0..3d3221a4 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -22,6 +22,7 @@ from unittest.mock import patch import pytest import ocrmypdf._validation as vd +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.api import create_options from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import BadArgsError, MissingDependencyError @@ -153,8 +154,9 @@ def test_false_action_store_true(): @pytest.mark.parametrize('progress_bar', [True, False]) def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) + plugin_manager = get_plugin_manager(opts.plugins) with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: - vd.check_options(opts) + vd.check_options(opts, plugin_manager) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) assert pdfinfo is not None assert tqdmpatch.called @@ -164,11 +166,12 @@ def test_no_progress_bar(progress_bar, resources): def test_language_warning(caplog): opts = make_opts(language=None) + plugin_manager = get_plugin_manager(opts.plugins) caplog.set_level(logging.DEBUG) with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') ): - vd.check_options_languages(opts) + vd.check_options_languages(opts, plugin_manager) assert opts.language == ['eng'] assert '' in caplog.text @@ -176,7 +179,7 @@ def test_language_warning(caplog): with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') ): - vd.check_options_languages(opts) + vd.check_options_languages(opts, plugin_manager) assert opts.language == ['eng'] assert 'assuming --language' in caplog.text From 9bccff4f885b4cbf3e38aa36437073c9003997d4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 May 2020 03:24:31 -0700 Subject: [PATCH 475/880] Move Tesseract specific arguments to plugin --- src/ocrmypdf/__main__.py | 10 +--- src/ocrmypdf/_plugin_manager.py | 12 ++++ src/ocrmypdf/_sync.py | 2 +- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 56 +++++++++++++++++++ src/ocrmypdf/cli.py | 55 +----------------- tests/conftest.py | 20 +++---- tests/test_validation.py | 5 +- 7 files changed, 87 insertions(+), 73 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 1cf84daf..69d68db4 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -22,7 +22,7 @@ import sys from multiprocessing import set_start_method from ocrmypdf import __version__ -from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_closed_streams, check_options from ocrmypdf.api import Verbosity, configure_logging @@ -33,13 +33,7 @@ log = logging.getLogger('ocrmypdf') def run(args=None): - pre_options, _unused = plugins_only_parser.parse_known_args(args=args) - plugin_manager = get_plugin_manager(pre_options.plugins) - - parser = get_parser() - plugin_manager.hook.add_options(parser=parser) - - options = parser.parse_args(args=args) + parser, options, plugin_manager = get_parser_options_plugins(args=args) if not check_closed_streams(options): return ExitCode.bad_args diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index e42ef94b..6216a44b 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -24,6 +24,7 @@ from typing import List import pluggy from ocrmypdf import pluginspec +from ocrmypdf.cli import get_parser, plugins_only_parser def get_plugin_manager(plugins: List[str], builtins=True): @@ -47,3 +48,14 @@ def get_plugin_manager(plugins: List[str], builtins=True): module = importlib.import_module(name) pm.register(module) return pm + + +def get_parser_options_plugins(args): + pre_options, _unused = plugins_only_parser.parse_known_args(args=args) + plugin_manager = get_plugin_manager(pre_options.plugins) + + parser = get_parser() + plugin_manager.hook.add_options(parser=parser) + + options = parser.parse_args(args=args) + return parser, options, plugin_manager diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 3f97a301..c25bf8e8 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -305,7 +305,7 @@ def run_pipeline(options, *, plugin_manager, api=False): if not options.jobs: options.jobs = available_cpu_count() if not plugin_manager: - plugin_manager = get_plugin_manager([]) + plugin_manager = get_plugin_manager(options.plugins) work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf.")) debug_log_handler = None diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 85b2e4c8..ee97845b 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -16,10 +16,66 @@ # along with OCRmyPDF. If not, see . from ocrmypdf import hookimpl +from ocrmypdf.cli import numeric from ocrmypdf.exec import tesseract from ocrmypdf.pluginspec import OcrEngine +@hookimpl +def add_options(parser): + tess = parser.add_argument_group("Tesseract", "Advanced control of Tesseract OCR") + tess.add_argument( + '--tesseract-config', + action='append', + metavar='CFG', + default=[], + help="Additional Tesseract configuration files -- see documentation", + ) + tess.add_argument( + '--tesseract-pagesegmode', + action='store', + type=int, + metavar='PSM', + choices=range(0, 14), + help="Set Tesseract page segmentation mode (see tesseract --help)", + ) + tess.add_argument( + '--tesseract-oem', + action='store', + type=int, + metavar='MODE', + choices=range(0, 4), + help=( + "Set Tesseract 4.0 OCR engine mode: " + "0 - original Tesseract only; " + "1 - neural nets LSTM only; " + "2 - Tesseract + LSTM; " + "3 - default." + ), + ) + tess.add_argument( + '--tesseract-timeout', + default=180.0, + type=numeric(float, 0), + metavar='SECONDS', + help='Give up on OCR after the timeout, but copy the preprocessed page ' + 'into the final output', + ) + tess.add_argument( + '--user-words', + metavar='FILE', + help="Specify the location of the Tesseract user words file. This is a " + "list of words Tesseract should consider while performing OCR in " + "addition to its standard language dictionaries. This can improve " + "OCR quality especially for specialized and technical documents.", + ) + tess.add_argument( + '--user-patterns', + metavar='FILE', + help="Specify the location of the Tesseract user patterns file.", + ) + + class TesseractOcrEngine(OcrEngine): @staticmethod def version(): diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index cea8e351..0675300c 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -57,6 +57,7 @@ class ArgumentParser(argparse.ArgumentParser): def get_parser(): parser = ArgumentParser( prog=_PROGRAM_NAME, + allow_abbrev=True, fromfile_prefix_chars='@', formatter_class=argparse.RawDescriptionHelpFormatter, description="""\ @@ -382,14 +383,14 @@ Online documentation is located at: ) advanced = parser.add_argument_group( - "Advanced", "Advanced options to control Tesseract's OCR behavior" + "Advanced", "Advanced options to control OCRmyPDF" ) advanced.add_argument( '--pages', type=str, help=( "Limit OCR to the specified pages (ranges or comma separated), " - "skipping others", + "skipping others" ), ) advanced.add_argument( @@ -401,35 +402,6 @@ Online documentation is located at: "decompression bomb", default=128.0, ) - advanced.add_argument( - '--tesseract-config', - action='append', - metavar='CFG', - default=[], - help="Additional Tesseract configuration files -- see documentation", - ) - advanced.add_argument( - '--tesseract-pagesegmode', - action='store', - type=int, - metavar='PSM', - choices=range(0, 14), - help="Set Tesseract page segmentation mode (see tesseract --help)", - ) - advanced.add_argument( - '--tesseract-oem', - action='store', - type=int, - metavar='MODE', - choices=range(0, 4), - help=( - "Set Tesseract 4.0 OCR engine mode: " - "0 - original Tesseract only; " - "1 - neural nets LSTM only; " - "2 - Tesseract + LSTM; " - "3 - default." - ), - ) advanced.add_argument( '--pdf-renderer', choices=['auto', 'hocr', 'sandwich'], @@ -437,14 +409,6 @@ Online documentation is located at: help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " "choose. See documentation for discussion.", ) - advanced.add_argument( - '--tesseract-timeout', - default=180.0, - type=numeric(float, 0), - metavar='SECONDS', - help='Give up on OCR after the timeout, but copy the preprocessed page ' - 'into the final output', - ) advanced.add_argument( '--rotate-pages-threshold', default=14.0, @@ -466,19 +430,6 @@ Online documentation is located at: "skipped. Not supported for --output-type=pdf ; that setting " "preserves the original compression of all images.", ) - advanced.add_argument( - '--user-words', - metavar='FILE', - help="Specify the location of the Tesseract user words file. This is a " - "list of words Tesseract should consider while performing OCR in " - "addition to its standard language dictionaries. This can improve " - "OCR quality especially for specialized and technical documents.", - ) - advanced.add_argument( - '--user-patterns', - metavar='FILE', - help="Specify the location of the Tesseract user patterns file.", - ) advanced.add_argument( '--fast-web-view', type=numeric(float, 0), diff --git a/tests/conftest.py b/tests/conftest.py index 55bd8376..d7d4b437 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -25,7 +25,7 @@ from subprocess import PIPE, run import pytest from ocrmypdf import api, cli, pdfinfo -from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf.exec import unpaper pytest_plugins = ['helpers_namespace'] @@ -213,12 +213,11 @@ def no_outpdf(tmp_path): @pytest.helpers.register def check_ocrmypdf(input_file, output_file, *args, env=None): """Run ocrmypdf and confirmed that a valid file was created""" + args = [str(input_file), str(output_file)] + [ + str(arg) for arg in args if arg is not None + ] - options = cli.get_parser().parse_args( - [str(input_file), str(output_file)] - + [str(arg) for arg in args if arg is not None] - ) - plugin_manager = get_plugin_manager(options.plugins) + _parser, options, plugin_manager = get_parser_options_plugins(args=args) api.check_options(options, plugin_manager) if env: options.tesseract_env = env @@ -239,10 +238,10 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): Does not currently have a way to manipulate the PATH except for Tesseract. """ - options = cli.get_parser().parse_args( - [str(input_file), str(output_file)] - + [str(arg) for arg in args if arg is not None] - ) + args = [str(input_file), str(output_file)] + [ + str(arg) for arg in args if arg is not None + ] + _parser, options, plugin_manager = get_parser_options_plugins(args=args) if env: options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) @@ -253,7 +252,6 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) - plugin_manager = get_plugin_manager(options.plugins) api.check_options(options, plugin_manager) return api.run_pipeline(options, plugin_manager=None, api=False) diff --git a/tests/test_validation.py b/tests/test_validation.py index 3d3221a4..857dbda0 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -32,8 +32,11 @@ from ocrmypdf.pdfinfo import PdfInfo def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): if language is not None: kwargs['language'] = language + parser = get_parser() + pm = get_plugin_manager(kwargs.get('plugins', [])) + pm.hook.add_options(parser=parser) return create_options( - input_file=input_file, output_file=output_file, parser=get_parser(), **kwargs + input_file=input_file, output_file=output_file, parser=parser, **kwargs ) From 03da34ee2473de1121707d795dcec146bccfef82 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 16 May 2020 17:04:44 -0700 Subject: [PATCH 476/880] Test files needed! --- .github/ISSUE_TEMPLATE/bug_report.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md index 04820098..daf2e24d 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -20,9 +20,13 @@ ocrmypdf ...arguments... input.pdf output.pdf Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful. **Example file** -Please include an example *input* PDF (or image). The input file is more helpful. +Include an input PDF or image that demonstrates your issue. -If possible, use an input file with no personal or confidential information. At your option you may GPG-encrypt the file for OCRmyPDF's author only. +Please provide an input file with no personal or confidential information. At your option you may `GPG-encrypt the file ` for OCRmyPDF's author only. + +Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue. + +(Exceptions: Issues with installation, command line argument parsing, test suite failures.Issues without example files usually cannot be resolved.) **Expected behavior** A clear and concise description of what you expected to happen. From f656c00f41a02e395fa8e40b6588b0fccb07136a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 18 May 2020 01:27:45 -0700 Subject: [PATCH 477/880] docs: Note about OCRmyPDF speed --- docs/index.rst | 1 + docs/performance.rst | 22 ++++++++++++++++++++++ 2 files changed, 23 insertions(+) create mode 100644 docs/performance.rst diff --git a/docs/index.rst b/docs/index.rst index e28f3234..bf042ae1 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -27,6 +27,7 @@ image processing and OCR to existing PDFs. advanced batch security + performance errors .. toctree:: diff --git a/docs/performance.rst b/docs/performance.rst new file mode 100644 index 00000000..82625fc0 --- /dev/null +++ b/docs/performance.rst @@ -0,0 +1,22 @@ +=========== +Performance +=========== + +Some users have noticed that current versions of OCRmyPDF do not run as quickly +as some older versions (specifically 6.x and older). This is because OCRmyPDF +added image optimization as a postprocessing step, and it is enabled by default. + +Speed +===== + +If running OCRmyPDF quickly is your main goal, you can use settings such as: + +* ``--optimize 0`` to disable file size optimization +* ``--output-type pdf`` to disable PDF/A generation +* ``--fast-web-view 0`` to disable fast web view optimization +* ``--skip-big`` to skip large images, if some pages have large images + +You can also avoid: + +* ``--force-ocr`` +* Image preprocessing From 0cefe886ec816632a963d1d88a57dca72436cf51 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 19 May 2020 16:12:36 -0700 Subject: [PATCH 478/880] Update email --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 84107620..63cead80 100644 --- a/setup.py +++ b/setup.py @@ -62,7 +62,7 @@ setup( long_description_content_type='text/markdown', url='https://github.com/jbarlow83/OCRmyPDF', author='James R. Barlow', - author_email='jim@purplerock.ca', + author_email='james@purplerock.ca', packages=find_packages('src', exclude=["tests", "tests.*"]), package_dir={'': 'src'}, keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'], From a0f9ca3a30d3de8b3b4f555985f2fd5decee0d7f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 May 2020 01:31:46 -0700 Subject: [PATCH 479/880] Move Tesseract options validation into plugin --- src/ocrmypdf/_sync.py | 2 - src/ocrmypdf/_validation.py | 31 +------------- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 41 ++++++++++++++++++- src/ocrmypdf/pluginspec.py | 2 +- tests/test_validation.py | 19 ++++++--- 5 files changed, 55 insertions(+), 40 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c25bf8e8..87302438 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -323,8 +323,6 @@ def run_pipeline(options, *, plugin_manager, api=False): original_filename, start_input_file, work_folder / 'origin.pdf', options ) - plugin_manager.hook.prepare(options=options) - # Gather pdfinfo and create context pdfinfo = get_pdfinfo( origin_pdf, diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index bfd52e69..785ca13a 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -123,18 +123,6 @@ def check_options_output(options): msg += f"Found Ghostscript {ghostscript.version()}" log.warning(msg) - # Decide on what renderer to use - if options.pdf_renderer == 'auto': - options.pdf_renderer = 'sandwich' - - if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( - options.tesseract_env, languages - ): - raise MissingDependencyError( - "You are using an alpha version of Tesseract 4.0 that does not support " - "the textonly_pdf parameter. We don't support versions this old." - ) - if options.output_type == 'pdfa': options.output_type = 'pdfa-2' @@ -277,18 +265,6 @@ def check_options_advanced(options): "--pdfa-image-compression argument has no effect when " "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" ) - if not tesseract.has_user_words(options.tesseract_env) and ( - options.user_words or options.user_patterns - ): - log.warning( - "Tesseract 4.0 ignores --user-words and --user-patterns, so these " - "arguments have no effect." - ) - if options.tesseract_pagesegmode in (0, 2): - log.warning( - "The --tesseract-pagesegmode argument you select will disable OCR. " - "This may cause processing to fail." - ) def check_options_metadata(options): @@ -322,6 +298,7 @@ def check_options(options, plugin_manager): check_options_advanced(options) check_options_pillow(options) check_dependency_versions(options) + plugin_manager.hook.check_options(options=options) def check_closed_streams(options): # pragma: no cover @@ -464,12 +441,6 @@ def report_output_file_size(options, input_file, output_file): def check_dependency_versions(options): - check_external_program( - program='tesseract', - package={'linux': 'tesseract-ocr'}, - version_checker=tesseract.version, - need_version='4.0.0', # using backport for Travis CI - ) check_external_program( program='gs', package='ghostscript', diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index ee97845b..41cc830a 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -15,11 +15,16 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import logging + from ocrmypdf import hookimpl from ocrmypdf.cli import numeric -from ocrmypdf.exec import tesseract +from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.exec import check_external_program, tesseract from ocrmypdf.pluginspec import OcrEngine +log = logging.getLogger(__name__) + @hookimpl def add_options(parser): @@ -76,6 +81,40 @@ def add_options(parser): ) +@hookimpl +def check_options(options): + check_external_program( + program='tesseract', + package={'linux': 'tesseract-ocr'}, + version_checker=tesseract.version, + need_version='4.0.0', # using backport for Travis CI + ) + + # Decide on what renderer to use + if options.pdf_renderer == 'auto': + options.pdf_renderer = 'sandwich' + + if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( + options.tesseract_env, set(options.language) + ): + raise MissingDependencyError( + "You are using an alpha version of Tesseract 4.0 that does not support " + "the textonly_pdf parameter. We don't support versions this old." + ) + if not tesseract.has_user_words(options.tesseract_env) and ( + options.user_words or options.user_patterns + ): + log.warning( + "Tesseract 4.0 ignores --user-words and --user-patterns, so these " + "arguments have no effect." + ) + if options.tesseract_pagesegmode in (0, 2): + log.warning( + "The --tesseract-pagesegmode argument you select will disable OCR. " + "This may cause processing to fail." + ) + + class TesseractOcrEngine(OcrEngine): @staticmethod def version(): diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 5e0a4f88..432b1946 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -39,7 +39,7 @@ def add_options(parser: ArgumentParser) -> None: @hookspec -def prepare(options: Namespace) -> None: +def check_options(options: Namespace) -> None: """Called to notify a plugin that a file will be processed. The plugin may modify the *options*. All objects that are in options must diff --git a/tests/test_validation.py b/tests/test_validation.py index 857dbda0..babc301d 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -69,7 +69,8 @@ def test_old_tesseract_error(): with patch('ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=False): with pytest.raises(MissingDependencyError): opts = make_opts(pdf_renderer='sandwich', language='eng') - vd.check_options_output(opts) + plugin_manager = get_plugin_manager(opts.plugins) + vd.check_options(opts, plugin_manager) def test_lossless_redo(): @@ -96,12 +97,17 @@ def test_optimizing(caplog): def test_user_words(caplog): - with patch('ocrmypdf.exec.tesseract.version', return_value='4.0.0'): - vd.check_options_advanced(make_opts(user_words='foo')) + + with patch('ocrmypdf.exec.tesseract.has_user_words', return_value=False): + opts = make_opts(user_words='foo') + plugin_manager = get_plugin_manager(opts.plugins) + vd.check_options(opts, plugin_manager) assert '4.0 ignores --user-words' in caplog.text caplog.clear() - with patch('ocrmypdf.exec.tesseract.version', return_value='4.1.0'): - vd.check_options_advanced(make_opts(user_patterns='foo')) + with patch('ocrmypdf.exec.tesseract.has_user_words', return_value=True): + opts = make_opts(user_patterns='foo') + plugin_manager = get_plugin_manager(opts.plugins) + vd.check_options(opts, plugin_manager) assert '4.0 ignores --user-words' not in caplog.text @@ -223,5 +229,6 @@ def test_version_comparison(): def test_pagesegmode_warning(caplog): opts = make_opts(tesseract_pagesegmode='0') - vd.check_options_advanced(opts) + plugin_manager = get_plugin_manager(opts.plugins) + vd.check_options(opts, plugin_manager) assert 'disable OCR' in caplog.text From d43212d30b6e26dd272efffcbe4650c9b4c94970 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 25 May 2020 03:20:10 -0700 Subject: [PATCH 480/880] Refactor --language argument into set --- src/ocrmypdf/_validation.py | 16 +++++----------- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 6 +++--- src/ocrmypdf/cli.py | 17 ++++++++++++++++- tests/test_validation.py | 4 ++-- 4 files changed, 26 insertions(+), 17 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 785ca13a..941e3522 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -74,24 +74,19 @@ def check_platform(): def check_options_languages(options, plugin_manager): - if not options.language: - options.language = [DEFAULT_LANGUAGE] + if not options.languages: + options.languages = {DEFAULT_LANGUAGE} system_lang = locale.getlocale()[0] if system_lang and not system_lang.startswith('en'): log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE) - # Support v2.x "eng+deu" language syntax - if '+' in options.language[0]: - options.language = options.language[0].split('+') - - languages = set(options.language) ocr_engine = plugin_manager.hook.get_ocr_engine() - if not languages.issubset(ocr_engine.languages(options)): + if not options.languages.issubset(ocr_engine.languages(options)): msg = ( f"{ocr_engine} does not have language data for the following " "requested languages: \n" ) - for lang in languages - ocr_engine.languages(options): + for lang in options.languages - ocr_engine.languages(options): msg += lang + '\n' raise MissingDependencyError(msg) @@ -101,8 +96,7 @@ def check_options_output(options): # 1. Ghostscript < 9.20 mangles multibyte Unicode # 2. hocr doesn't work on non-Latin languages (so don't select it) - languages = set(options.language) - is_latin = languages.issubset(HOCR_OK_LANGS) + is_latin = options.languages.issubset(HOCR_OK_LANGS) if options.pdf_renderer == 'hocr' and not is_latin: msg = ( diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 41cc830a..e2fcfb6d 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -95,7 +95,7 @@ def check_options(options): options.pdf_renderer = 'sandwich' if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( - options.tesseract_env, set(options.language) + options.tesseract_env, set(options.languages) ): raise MissingDependencyError( "You are using an alpha version of Tesseract 4.0 that does not support " @@ -147,7 +147,7 @@ class TesseractOcrEngine(OcrEngine): input_file=input_file, output_hocr=output_hocr, output_text=output_text, - languages=options.language, + languages=options.languages, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, timeout=options.tesseract_timeout, @@ -163,7 +163,7 @@ class TesseractOcrEngine(OcrEngine): input_file=input_file, output_pdf=output_pdf, output_text=output_text, - languages=options.language, + languages=options.languages, engine_mode=options.tesseract_oem, tessconfig=options.tesseract_config, timeout=options.tesseract_timeout, diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 0675300c..ccf66d99 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -54,6 +54,20 @@ class ArgumentParser(argparse.ArgumentParser): raise ValueError(message) +class LanguageSetAction(argparse.Action): + def __init__(self, option_strings, dest, default=None, **kwargs): + if default is None: + default = set() + super().__init__(option_strings, dest, default=default, **kwargs) + + def __call__(self, parser, namespace, values, option_string=None): + dest = getattr(namespace, self.dest) + if '+' in values: + dest.add(lang for lang in values.split('+')) + else: + dest.add(values) + + def get_parser(): parser = ArgumentParser( prog=_PROGRAM_NAME, @@ -126,7 +140,8 @@ Online documentation is located at: parser.add_argument( '-l', '--language', - action='append', + dest='languages', + action=LanguageSetAction, help="Language(s) of the file to be OCRed (see tesseract --list-langs for " "all language packs installed in your system). Use -l eng+deu for " "multiple languages.", diff --git a/tests/test_validation.py b/tests/test_validation.py index babc301d..f6fcc775 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -181,7 +181,7 @@ def test_language_warning(caplog): 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') ): vd.check_options_languages(opts, plugin_manager) - assert opts.language == ['eng'] + assert opts.languages == {'eng'} assert '' in caplog.text opts = make_opts(language=None) @@ -189,7 +189,7 @@ def test_language_warning(caplog): 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') ): vd.check_options_languages(opts, plugin_manager) - assert opts.language == ['eng'] + assert opts.languages == {'eng'} assert 'assuming --language' in caplog.text From aa060db5bc7f2c4017902360cc2b039bfca4d8bc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 May 2020 02:13:17 -0700 Subject: [PATCH 481/880] Refactor tesseract_env variable into the plugin Removed all cases except one in api.py, which isn't worth solving because it should be removed anyway. This also fixes a logic error in the OMP_THREAD_LIMIT decision, api.py did not use pass kwargs correctly so they never worked before. --- src/ocrmypdf/_plugin_manager.py | 5 +++- src/ocrmypdf/_sync.py | 19 -------------- src/ocrmypdf/api.py | 19 ++++++-------- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 25 +++++++++++++++++++ src/ocrmypdf/cli.py | 1 - src/ocrmypdf/pluginspec.py | 2 +- tests/test_unpaper.py | 8 +++--- 7 files changed, 42 insertions(+), 37 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 6216a44b..1fbc1b76 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import argparse import importlib import importlib.util import sys @@ -50,7 +51,9 @@ def get_plugin_manager(plugins: List[str], builtins=True): return pm -def get_parser_options_plugins(args): +def get_parser_options_plugins( + args, +) -> (argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager): pre_options, _unused = plugins_only_parser.parse_known_args(args=args) plugin_manager = get_plugin_manager(pre_options.plugins) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 87302438..60eed5de 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -216,25 +216,6 @@ def exec_concurrent(context): if max_workers > 1: log.info("Start processing %d pages concurrently", max_workers) - # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want - # to manage how many threads it uses to avoid creating total threads than cores. - # Performance testing shows we're better off - # parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we - # get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the - # input file is small, then we allow Tesseract to use threads, subject to the - # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers. - # As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system. - tess_threads = min(3, context.options.jobs // max_workers) - if context.options.tesseract_env is None: - context.options.tesseract_env = os.environ.copy() - context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads)) - try: - tess_threads = int(context.options.tesseract_env['OMP_THREAD_LIMIT']) - except ValueError: # OMP_THREAD_LIMIT initialized to non-numeric - context.log.error("Environment variable OMP_THREAD_LIMIT is not numeric") - if tess_threads > 1: - log.info("Using Tesseract OpenMP thread limit %d", tess_threads) - sidecars = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 7edce482..68ede13a 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -15,6 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import inspect import logging import os import sys @@ -178,11 +179,6 @@ def create_options( options = parser.parse_args(cmdline) for keyword, val in deferred: setattr(options, keyword, val) - - # If we are running a Tesseract spoof, ensure it knows what the input file is - if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env: - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - return options @@ -233,7 +229,6 @@ def ocr( # pylint: disable=unused-argument plugins: Iterable[str] = None, keep_temporary_files: bool = None, progress_bar: bool = None, - tesseract_env: Dict[str, str] = None, **kwargs, ): """Run OCRmyPDF on one PDF or image. @@ -245,7 +240,6 @@ def ocr( # pylint: disable=unused-argument use_threads (bool): Use worker threads instead of processes. This reduces performance but may make debugging easier since it is easier to set breakpoints. - tesseract_env (dict): Override environment variables for Tesseract Raises: ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging with the OCR layer. @@ -274,10 +268,13 @@ def ocr( # pylint: disable=unused-argument parser = get_parser() _plugin_manager = get_plugin_manager(plugins) - _plugin_manager.hook.add_options(parser=parser) + _plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member - options = create_options( - **{k: v for k, v in locals().items() if not k.startswith('_')} - ) + create_options_kwargs = { + k: v for k, v in locals().items() if not k.startswith('_') and k != 'kwargs' + } + create_options_kwargs.update(kwargs) + + options = create_options(**create_options_kwargs) check_options(options, _plugin_manager) return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index e2fcfb6d..4991baff 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -15,7 +15,9 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import argparse import logging +import os from ocrmypdf import hookimpl from ocrmypdf.cli import numeric @@ -79,6 +81,7 @@ def add_options(parser): metavar='FILE', help="Specify the location of the Tesseract user patterns file.", ) + tess.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) @hookimpl @@ -115,6 +118,28 @@ def check_options(options): ) +@hookimpl +def validate(pdfinfo, options): + # If we are running a Tesseract spoof, ensure it knows what the input file is + if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env: + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(options.input_file) + + # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want + # to manage how many threads it uses to avoid creating total threads than cores. + # Performance testing shows we're better off + # parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we + # get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the + # input file is small, then we allow Tesseract to use threads, subject to the + # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers. + # As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system. + if not options.tesseract_env.get('OMP_THREAD_LIMIT', '').isnumeric(): + tess_threads = min(3, options.jobs // len(pdfinfo), len(pdfinfo)) + options.tesseract_env['OMP_THREAD_LIMIT'] = str(tess_threads) + + if tess_threads > 1: + log.info("Using Tesseract OpenMP thread limit %d", tess_threads) + + class TesseractOcrEngine(OcrEngine): @staticmethod def version(): diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index ccf66d99..a34a2108 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -474,7 +474,6 @@ Online documentation is located at: action='store_true', help="Keep temporary files (helpful for debugging)", ) - debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) return parser diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 432b1946..dd659458 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -40,7 +40,7 @@ def add_options(parser: ArgumentParser) -> None: @hookspec def check_options(options: Namespace) -> None: - """Called to notify a plugin that a file will be processed. + """Called to ask the plugin to check all of its options. The plugin may modify the *options*. All objects that are in options must be picklable so they can be marshalled to child worker processes. diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index e87800c9..bd04da2c 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -20,7 +20,7 @@ from unittest.mock import patch import pytest -from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._validation import check_options from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError @@ -43,13 +43,13 @@ def spoof_unpaper_oldversion(tmp_path_factory): def test_no_unpaper(resources, no_outpdf): input_ = fspath(resources / "c02-22.pdf") output = fspath(no_outpdf) - options = get_parser().parse_args(args=["--clean", input_, output]) - plugin_manager = get_plugin_manager(options.plugins) + + _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") with pytest.raises(MissingDependencyError): - check_options(options, plugin_manager) + check_options(options, pm) def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): From df9f5157bd4dd5cb447f18031d35ee030c15ff66 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 May 2020 14:58:41 -0700 Subject: [PATCH 482/880] Fix shim_paths to account for unexpected files in Program Files\gs Fixes #565 --- src/ocrmypdf/exec/__init__.py | 38 ++++++++++++++++++----------------- tests/test_helpers.py | 19 ++++++++++++++++++ 2 files changed, 39 insertions(+), 18 deletions(-) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index f235b6e8..145cc43b 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -25,6 +25,7 @@ import sys from collections.abc import Mapping from distutils.version import LooseVersion from functools import lru_cache +from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError from subprocess import run as subprocess_run @@ -138,24 +139,25 @@ def shim_paths_with_program_files(env=None): program_files = env.get('PROGRAMFILES', '') if not program_files: return env.get('PATH', '') - paths = [] - try: - for dirname in os.listdir(program_files): - if dirname.lower() == 'tesseract-ocr': - paths.append(os.path.join(program_files, dirname)) - elif dirname.lower() == 'gs': - try: - latest_gs = max( - os.listdir(os.path.join(program_files, dirname)), - key=lambda d: float(d[2:]), - ) - except (FileNotFoundError, NotADirectoryError): - continue - paths.append(os.path.join(program_files, dirname, latest_gs, 'bin')) - except EnvironmentError: - pass - paths.extend(path for path in os.get_exec_path(env) if path not in set(paths)) - return os.pathsep.join(paths) + + def path_walker(): + for path in Path(program_files).iterdir(): + if not path.is_dir(): + continue + if path.name.lower() == 'tesseract-ocr': + yield path + elif path.name.lower() == 'gs': + yield from (p for p in path.glob('**/bin') if p.is_dir()) + + paths = sorted( + (p for p in path_walker()), key=lambda p: (p.name, p.parent.name), reverse=True + ) + paths.extend( + Path(str_path) + for str_path in os.get_exec_path(env) + if Path(str_path) not in set(paths) + ) + return os.pathsep.join(str(p) for p in paths) missing_program = ''' diff --git a/tests/test_helpers.py b/tests/test_helpers.py index f3c964d7..b7355903 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -17,6 +17,7 @@ import logging import multiprocessing +import os from pathlib import Path from unittest.mock import MagicMock @@ -95,3 +96,21 @@ class TestFileIsWritable: pathmock.exists.return_value = True pathmock.is_file.side_effect = PermissionError assert not helpers.is_file_writable(pathmock) + + +def test_shim_paths(tmp_path): + progfiles = tmp_path / 'Program Files' + progfiles.mkdir() + (progfiles / 'tesseract-ocr').mkdir() + (progfiles / 'gs' / '9.51' / 'bin').mkdir(parents=True) + (progfiles / 'gs' / '9.52' / 'bin').mkdir(parents=True) + syspath = tmp_path / 'bin' + env = {'PROGRAMFILES': str(progfiles), 'PATH': str(syspath)} + from ocrmypdf.exec import shim_paths_with_program_files + + result_str = shim_paths_with_program_files(env=env) + results = result_str.split(os.pathsep) + assert results[0].endswith('tesseract-ocr') + assert results[1].endswith('gs/9.52/bin') + assert results[2].endswith('gs/9.51/bin') + assert results[3] == str(syspath) From 3754185f56b3d8f0b978d2fc91b83281f4b4b493 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 May 2020 15:01:51 -0700 Subject: [PATCH 483/880] Mark pdfminer.six 20200517 as supported --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 63cead80..419a0c97 100644 --- a/setup.py +++ b/setup.py @@ -98,7 +98,7 @@ setup( 'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20181108, <= 20200402', + 'pdfminer.six >= 20181108, <= 20200517', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 74fdfeea3f3ad06107ec1923d73326811f7de39c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 May 2020 15:04:23 -0700 Subject: [PATCH 484/880] v9.8.1 notes --- docs/release_notes.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index a5f49460..3ed82045 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,16 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.8.1 +====== + +- Fixed an issue where unexpected files in the ``%PROGRAMFILES%\gs`` directory + (Windows) caused an exception. +- Mark pdfminer.six 20200517 as supported. +- If jbig2enc is missing and optimization is requested, a warning is issued + instead of an error, which was the intended behavior. +- Documentation updates. + v9.8.0 ====== From 642ebc6098da30a1aa41006fee809ff3b654d55a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 May 2020 15:52:00 -0700 Subject: [PATCH 485/880] Fix test that failed on Windows --- tests/test_helpers.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_helpers.py b/tests/test_helpers.py index b7355903..c211477f 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -111,6 +111,6 @@ def test_shim_paths(tmp_path): result_str = shim_paths_with_program_files(env=env) results = result_str.split(os.pathsep) assert results[0].endswith('tesseract-ocr') - assert results[1].endswith('gs/9.52/bin') - assert results[2].endswith('gs/9.51/bin') + assert results[1].endswith(os.path.join('gs', '9.52', 'bin')) + assert results[2].endswith(os.path.join('gs', '9.51', 'bin')) assert results[3] == str(syspath) From 6528234608b66216843fe0ed17e78ef2b3e967dd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 1 Jun 2020 02:27:27 -0700 Subject: [PATCH 486/880] Fix tesseract_ocr.py errors --- src/ocrmypdf/__init__.py | 1 + src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 7 ++++++- 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 08da4d6f..d64253b8 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -36,5 +36,6 @@ from ocrmypdf.exceptions import ( TesseractConfigError, UnsupportedImageFormatError, ) +from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence hookimpl = _HookimplMarker('ocrmypdf') diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 4991baff..d5f5eb7b 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -120,8 +120,11 @@ def check_options(options): @hookimpl def validate(pdfinfo, options): + if not options.tesseract_env: + return + # If we are running a Tesseract spoof, ensure it knows what the input file is - if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env: + if os.environ.get('PYTEST_CURRENT_TEST'): options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(options.input_file) # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want @@ -135,6 +138,8 @@ def validate(pdfinfo, options): if not options.tesseract_env.get('OMP_THREAD_LIMIT', '').isnumeric(): tess_threads = min(3, options.jobs // len(pdfinfo), len(pdfinfo)) options.tesseract_env['OMP_THREAD_LIMIT'] = str(tess_threads) + else: + tess_threads = int(options.tesseract_env['OMP_THREAD_LIMIT']) if tess_threads > 1: log.info("Using Tesseract OpenMP thread limit %d", tess_threads) From 2b23f7ec73121214c91496f32b8669538d911d94 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 1 Jun 2020 02:45:49 -0700 Subject: [PATCH 487/880] tesseract_noop: begin implementing with plugin --- tests/conftest.py | 27 +++++--- tests/plugins/tesseract_noop.py | 111 ++++++++++++++++++++++++++++++++ tests/test_main.py | 10 +-- 3 files changed, 133 insertions(+), 15 deletions(-) create mode 100644 tests/plugins/tesseract_noop.py diff --git a/tests/conftest.py b/tests/conftest.py index d7d4b437..7be5f1f8 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -220,8 +220,12 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): _parser, options, plugin_manager = get_parser_options_plugins(args=args) api.check_options(options, plugin_manager) if env: - options.tesseract_env = env - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] + if 'tesseract_noop' in first: + options.plugins = ['tests/plugins/tesseract_noop.py'] + else: + options.tesseract_env = env + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) result = api.run_pipeline(options, plugin_manager=plugin_manager, api=True) assert result == 0 @@ -243,12 +247,19 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): ] _parser, options, plugin_manager = get_parser_options_plugins(args=args) if env: - options.tesseract_env = env.copy() - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] - if 'spoof' in first_path: - assert 'gs' not in first_path, "use run_ocrmypdf() for gs" - assert 'tesseract' in first_path + try: + first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] + if 'tesseract_noop' in first: + options.plugins = ['tests/plugins/tesseract_noop.py'] + else: + options.tesseract_env = env.copy() + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] + if 'spoof' in first_path: + assert 'gs' not in first_path, "use run_ocrmypdf() for gs" + assert 'tesseract' in first_path + except KeyError: + pass if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) diff --git a/tests/plugins/tesseract_noop.py b/tests/plugins/tesseract_noop.py new file mode 100644 index 00000000..1db91fe4 --- /dev/null +++ b/tests/plugins/tesseract_noop.py @@ -0,0 +1,111 @@ +#!/usr/bin/env python3 +# © 2016 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +"""Tesseract no-op spoof + +To quickly run tests where getting OCR output is not necessary. + +In 'hocr' mode, create a .hocr file that specifies no text found. + +In 'pdf' mode, convert the image to PDF using another program. + +In orientation check mode, report the orientation is upright. +""" + +import sys +from pathlib import Path + +import img2pdf +import pikepdf +from PIL import Image + +from ocrmypdf import OcrEngine, OrientationConfidence, hookimpl + +HOCR_TEMPLATE = ''' + + + + + + + + + +
+
+

+ + +

+
+
+ +''' + + +class NoopOcrEngine(OcrEngine): + @staticmethod + def version(): + return '4.0.0' + + @staticmethod + def creator_tag(options): + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + return f"NO-OP {tag} {NoopOcrEngine.version()}" + + def __str__(self): + return f"NO-OP {NoopOcrEngine.version()}" + + @staticmethod + def languages(options): + return {'eng'} + + @staticmethod + def get_orientation(input_file, options): + return OrientationConfidence(angle=0, confidence=0.0) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with Image.open(input_file) as im, open( + output_hocr, 'w', encoding='utf-8' + ) as f: + w, h = im.size + f.write(HOCR_TEMPLATE.format(str(w), str(h))) + with open(output_text, 'w') as f: + f.write('') + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with Image.open(input_file) as im: + dpi = im.info['dpi'] + pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1] + ptsize = pagesize[0] * 72, pagesize[1] * 72 + pdf = pikepdf.new() + pdf.add_blank_page(page_size=ptsize) + pdf.save(output_pdf, static_id=True) + output_text.write_text('') + + +@hookimpl +def get_ocr_engine(): + return NoopOcrEngine() diff --git a/tests/test_main.py b/tests/test_main.py index c3a715ce..8c417d76 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -182,14 +182,10 @@ def test_maximum_options( ) -def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir): - env = os.environ.copy() - env['TESSDATA_PREFIX'] = os.fspath(tmpdir) - +def test_tesseract_missing_tessdata(monkeypatch, resources, no_outpdf, tmpdir): + monkeypatch.setenv("TESSDATA_PREFIX", os.fspath(tmpdir)) with pytest.raises(MissingDependencyError): - run_ocrmypdf_api( - resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env - ) + run_ocrmypdf_api(resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text') def test_invalid_input_pdf(resources, no_outpdf): From 1598f2f0e5f1a07f3af3c292adf0622925c8110d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 1 Jun 2020 03:06:40 -0700 Subject: [PATCH 488/880] Abolish spoof_tesseract_noop --- src/ocrmypdf/api.py | 2 +- tests/conftest.py | 9 +-- tests/spoof/tesseract_noop.py | 134 ---------------------------------- tests/test_acroform.py | 4 +- tests/test_ghostscript.py | 58 +++++++++------ tests/test_image_input.py | 14 +++- tests/test_main.py | 131 +++++++++++++++++++++------------ tests/test_metadata.py | 51 ++++++++----- tests/test_optimize.py | 16 ++-- tests/test_preprocessing.py | 16 ++-- tests/test_stdio.py | 56 +++++++------- tests/test_unpaper.py | 25 +++++-- 12 files changed, 234 insertions(+), 282 deletions(-) delete mode 100755 tests/spoof/tesseract_noop.py diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 68ede13a..008d284d 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -143,7 +143,7 @@ def create_options( # These arguments with special handling for which we bypass # argparse - if arg in {'tesseract_env', 'progress_bar'}: + if arg in {'tesseract_env', 'progress_bar', 'plugins'}: deferred.append((arg, val)) continue diff --git a/tests/conftest.py b/tests/conftest.py index 7be5f1f8..bc188bf9 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -169,11 +169,6 @@ def spoof(tmp_path_factory, **kwargs): return env -@pytest.fixture -def spoof_tesseract_noop(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_noop.py') - - @pytest.fixture def spoof_tesseract_cache(tmp_path_factory): if running_in_docker(): @@ -222,7 +217,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): if env: first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] if 'tesseract_noop' in first: - options.plugins = ['tests/plugins/tesseract_noop.py'] + raise ValueError('noop') else: options.tesseract_env = env options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) @@ -250,7 +245,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): try: first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] if 'tesseract_noop' in first: - options.plugins = ['tests/plugins/tesseract_noop.py'] + raise ValueError('noop') else: options.tesseract_env = env.copy() options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) diff --git a/tests/spoof/tesseract_noop.py b/tests/spoof/tesseract_noop.py deleted file mode 100755 index 30f97209..00000000 --- a/tests/spoof/tesseract_noop.py +++ /dev/null @@ -1,134 +0,0 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -"""Tesseract no-op spoof - -To quickly run tests where getting OCR output is not necessary. - -In 'hocr' mode, create a .hocr file that specifies no text found. - -In 'pdf' mode, convert the image to PDF using another program. - -In orientation check mode, report the orientation is upright. -""" - -import sys -from pathlib import Path - -import img2pdf -import pikepdf -from PIL import Image - -VERSION_STRING = '''tesseract 4.0.0 - leptonica-1.77.0 - libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 - Found AVX2 - Found AVX - Found SSE -SPOOFED -''' - -HOCR_TEMPLATE = ''' - - - - - - - - - -
-
-

- - -

-
-
- -''' - - -def main(): - if sys.argv[1] == '--version': - print(VERSION_STRING, file=sys.stderr) - sys.exit(0) - elif sys.argv[1] == '--list-langs': - print('List of available languages (1):\neng', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == '--print-parameters': - print("Some parameters", file=sys.stderr) - print("textonly_pdf\t1\tSome help text") - sys.exit(0) - elif sys.argv[-2] == 'hocr': - inputf = sys.argv[-4] - output = sys.argv[-3] - with Image.open(inputf) as im, open( - output + '.hocr', 'w', encoding='utf-8' - ) as f: - w, h = im.size - f.write(HOCR_TEMPLATE.format(str(w), str(h))) - with open(output + '.txt', 'w') as f: - f.write('') - elif sys.argv[-2] == 'pdf': - if 'textonly_pdf=1' in sys.argv: - inputf = sys.argv[-4] - output = sys.argv[-3] - with Image.open(inputf) as im: - dpi = im.info['dpi'] - pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1] - ptsize = pagesize[0] * 72, pagesize[1] * 72 - - pdf_out = pikepdf.new() - pdf_out.add_blank_page(page_size=ptsize) - pdf_out.save(Path(output).with_suffix('.pdf'), static_id=True) - Path(output).with_suffix('.txt').write_text('') - else: - inputf = sys.argv[-4] - output = sys.argv[-3] - pdf_bytes = img2pdf.convert([inputf], dpi=300) - with open(output + '.pdf', 'wb') as f: - f.write(pdf_bytes) - with open(output + '.txt', 'w') as f: - f.write('') - elif sys.argv[-1] == 'stdout': - inputf = sys.argv[-2] - print( - """Orientation: 0 -Orientation in degrees: 0 -Orientation confidence: 100.00 -Script: 1 -Script confidence: 100.00""", - file=sys.stderr, - ) - else: - print("Spoof doesn't understand arguments", file=sys.stderr) - print(sys.argv, file=sys.stderr) - sys.exit(1) - - sys.exit(0) - - -if __name__ == '__main__': - main() diff --git a/tests/test_acroform.py b/tests/test_acroform.py index 44de63da..4ab52406 100644 --- a/tests/test_acroform.py +++ b/tests/test_acroform.py @@ -35,8 +35,8 @@ def test_acroform_and_redo(acroform, caplog, no_outpdf): assert '--redo-ocr is not currently possible' in caplog.text -def test_acroform_message(acroform, caplog, spoof_tesseract_noop, outpdf): +def test_acroform_message(acroform, caplog, outpdf): caplog.set_level(logging.INFO) - check_ocrmypdf(acroform, outpdf, env=spoof_tesseract_noop) + check_ocrmypdf(acroform, outpdf, '--plugin', 'tests/plugins/tesseract_noop.py') assert 'fillable form' in caplog.text assert '--force-ocr' in caplog.text diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index da04ea84..0e6931df 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -33,31 +33,23 @@ spoof = pytest.helpers.spoof @pytest.fixture -def spoof_no_tess_gs_render_fail(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py' - ) +def spoof_gs_render_fail(tmp_path_factory): + return spoof(tmp_path_factory, gs='gs_render_failure.py') @pytest.fixture -def spoof_no_tess_gs_raster_fail(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py' - ) +def spoof_gs_raster_fail(tmp_path_factory): + return spoof(tmp_path_factory, gs='gs_raster_failure.py') @pytest.fixture -def spoof_no_tess_no_pdfa(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py' - ) +def spoof_no_pdfa(tmp_path_factory): + return spoof(tmp_path_factory, gs='gs_pdfa_failure.py') @pytest.fixture -def spoof_no_tess_pdfa_warning(tmp_path_factory): - return spoof( - tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py' - ) +def spoof_pdfa_warning(tmp_path_factory): + return spoof(tmp_path_factory, gs='gs_feature_elision.py') @pytest.fixture @@ -114,30 +106,48 @@ def test_rasterize_rotated(francais, outdir, caplog): assert im.info['dpi'] == (forced_dpi[1], forced_dpi[0]) -def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf): +def test_gs_render_failure(spoof_gs_render_fail, resources, outpdf): p, out, err = run_ocrmypdf( - resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail + resources / 'blank.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + env=spoof_gs_render_fail, ) assert 'Casper is not a friendly ghost' in err assert p.returncode == ExitCode.child_process_error -def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf): +def test_gs_raster_failure(spoof_gs_raster_fail, resources, outpdf): p, out, err = run_ocrmypdf( - resources / 'francais.pdf', outpdf, env=spoof_no_tess_gs_raster_fail + resources / 'francais.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + env=spoof_gs_raster_fail, ) assert 'Ghost story archive not found' in err assert p.returncode == ExitCode.child_process_error -def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf): +def test_ghostscript_pdfa_failure(spoof_no_pdfa, resources, outpdf): p, out, err = run_ocrmypdf( - resources / 'francais.pdf', outpdf, env=spoof_no_tess_no_pdfa + resources / 'francais.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + env=spoof_no_pdfa, ) assert ( p.returncode == ExitCode.pdfa_conversion_failed ), "Unexpected return when PDF/A fails" -def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf): - check_ocrmypdf(resources / 'francais.pdf', outpdf, env=spoof_no_tess_pdfa_warning) +def test_ghostscript_feature_elision(spoof_pdfa_warning, resources, outpdf): + check_ocrmypdf( + resources / 'francais.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + env=spoof_pdfa_warning, + ) diff --git a/tests/test_image_input.py b/tests/test_image_input.py index ceb94cbe..c3636952 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -33,9 +33,14 @@ def baiona(resources): return Image.open(resources / 'baiona_gray.png') -def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): +def test_image_to_pdf(resources, outpdf): check_ocrmypdf( - resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop + resources / 'crom.png', + outpdf, + '--image-dpi', + '200', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @@ -77,7 +82,7 @@ def test_img2pdf_fails(resources, no_outpdf): assert rc == ocrmypdf.ExitCode.input_file -def test_jpeg_in_jpeg_out(resources, outpdf, spoof_tesseract_noop): +def test_jpeg_in_jpeg_out(resources, outpdf): check_ocrmypdf( resources / 'congress.jpg', outpdf, @@ -86,7 +91,8 @@ def test_jpeg_in_jpeg_out(resources, outpdf, spoof_tesseract_noop): '--output-type', 'pdf', # specifically check pdf because Ghostscript may convert to JPEG '--remove-background', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) with pikepdf.open(outpdf) as pdf: assert next(pdf.pages[0].images.values()).Filter == pikepdf.Name.DCTDecode diff --git a/tests/test_main.py b/tests/test_main.py index 8c417d76..e161730e 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -111,7 +111,7 @@ def test_redo_ocr(resources, outpdf): ), "Expected text to be different after re-OCR" -def test_argsfile(spoof_tesseract_noop, resources, outdir): +def test_argsfile(resources, outdir): path_argsfile = outdir / 'test_argsfile.txt' with open(str(path_argsfile), 'w') as argsfile: print( @@ -119,15 +119,14 @@ def test_argsfile(spoof_tesseract_noop, resources, outdir): 'ArgsFile Test', '--author', 'Test Cases', + '--plugin', + 'tests/plugins/tesseract_noop.py', sep='\n', end='\n', file=argsfile, ) check_ocrmypdf( - resources / 'graph.pdf', - path_argsfile, - '@' + str(outdir / 'test_argsfile.txt'), - env=spoof_tesseract_noop, + resources / 'graph.pdf', path_argsfile, '@' + str(outdir / 'test_argsfile.txt') ) @@ -239,23 +238,27 @@ def test_klingon(resources, outpdf): assert p.returncode == ExitCode.missing_dependency -def test_missing_docinfo(spoof_tesseract_noop, resources, outpdf): +def test_missing_docinfo(resources, outpdf): result = run_ocrmypdf_api( resources / 'missing_docinfo.pdf', outpdf, '-l', 'eng', '--skip-text', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert result == ExitCode.ok -def test_uppercase_extension(spoof_tesseract_noop, resources, outdir): +def test_uppercase_extension(resources, outdir): shutil.copy(str(resources / "skew.pdf"), str(outdir / "UPPERCASE.PDF")) check_ocrmypdf( - outdir / "UPPERCASE.PDF", outdir / "UPPERCASE_OUT.PDF", env=spoof_tesseract_noop + outdir / "UPPERCASE.PDF", + outdir / "UPPERCASE_OUT.PDF", + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @@ -349,9 +352,12 @@ def test_tesseract_image_too_big( ) -def test_algo4(resources, spoof_tesseract_noop, outpdf): +def test_algo4(resources, outpdf): p, _, _ = run_ocrmypdf( - resources / 'encrypted_algo4.pdf', outpdf, env=spoof_tesseract_noop + resources / 'encrypted_algo4.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.encrypted_pdf @@ -370,17 +376,19 @@ def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): assert out_pageinfo[0].images[0].enc == Encoding.jbig2 -def test_masks(spoof_tesseract_noop, resources, outpdf): +def test_masks(resources, outpdf): assert ( ocrmypdf.ocr( - resources / 'masks.pdf', outpdf, tesseract_env=spoof_tesseract_noop + resources / 'masks.pdf', outpdf, plugins=['tests/plugins/tesseract_noop.py'] ) == ExitCode.ok ) -def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf(resources / 'epson.pdf', outpdf, env=spoof_tesseract_noop) +def test_linearized_pdf_and_indirect_object(resources, outpdf): + check_ocrmypdf( + resources / 'epson.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_noop.py' + ) def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): @@ -393,20 +401,27 @@ def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): assert isclose(image.dpi.x, 2400) -def test_overlay(spoof_tesseract_noop, resources, outpdf): +def test_overlay(resources, outpdf): check_ocrmypdf( - resources / 'overlay.pdf', outpdf, '--skip-text', env=spoof_tesseract_noop + resources / 'overlay.pdf', + outpdf, + '--skip-text', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) -def test_destination_not_writable(spoof_tesseract_noop, resources, outdir): +def test_destination_not_writable(resources, outdir): if os.name != 'nt' and (os.getuid() == 0 or os.geteuid() == 0): pytest.xfail(reason="root can write to anything") protected_file = outdir / 'protected.pdf' protected_file.touch() protected_file.chmod(0o400) # Read-only p, _out, _err = run_ocrmypdf( - resources / 'jbig2.pdf', protected_file, env=spoof_tesseract_noop + resources / 'jbig2.pdf', + protected_file, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.file_access_error, "Expected error" @@ -479,9 +494,13 @@ def test_user_words_ocr(resources, outdir): ) -def test_form_xobject(spoof_tesseract_noop, resources, outpdf): +def test_form_xobject(resources, outpdf): check_ocrmypdf( - resources / 'formxobject.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop + resources / 'formxobject.pdf', + outpdf, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @@ -513,14 +532,15 @@ def test_pagesize_consistency(renderer, resources, outpdf): assert isclose(before_dims[1], after_dims[1], rel_tol=1e-4) -def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): +def test_skip_big_with_no_images(resources, outpdf): check_ocrmypdf( resources / 'blank.pdf', outpdf, '--skip-big', '5', '--force-ocr', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @@ -528,18 +548,20 @@ def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): '8.0.0' <= pikepdf.__libqpdf_version__ <= '8.0.1', reason="libqpdf regression on pages with no contents", ) -def test_no_contents(spoof_tesseract_noop, resources, outpdf): +def test_no_contents(resources, outpdf): check_ocrmypdf( - resources / 'no_contents.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop + resources / 'no_contents.pdf', + outpdf, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @pytest.mark.parametrize( 'image', ['baiona.png', 'baiona_gray.png', 'baiona_alpha.png', 'congress.jpg'] ) -def test_compression_preserved( - spoof_tesseract_noop, ocrmypdf_exec, resources, image, outpdf -): +def test_compression_preserved(ocrmypdf_exec, resources, image, outpdf): input_file = str(resources / image) output_file = str(outpdf) @@ -553,6 +575,8 @@ def test_compression_preserved( '150', '--output-type', 'pdf', + '--plugin', + 'tests/plugins/tesseract_noop.py', '-', output_file, ] @@ -562,7 +586,6 @@ def test_compression_preserved( stderr=PIPE, stdin=input_stream, universal_newlines=True, - env=spoof_tesseract_noop, check=False, ) @@ -596,9 +619,7 @@ def test_compression_preserved( ('congress.jpg', 'lossless'), ], ) -def test_compression_changed( - spoof_tesseract_noop, ocrmypdf_exec, resources, image, compression, outpdf -): +def test_compression_changed(ocrmypdf_exec, resources, image, compression, outpdf): input_file = str(resources / image) output_file = str(outpdf) @@ -615,6 +636,8 @@ def test_compression_changed( '0', '--pdfa-image-compression', compression, + '--plugin', + 'tests/plugins/tesseract_noop.py', '-', output_file, ] @@ -624,7 +647,6 @@ def test_compression_changed( stderr=PIPE, stdin=input_stream, universal_newlines=True, - env=spoof_tesseract_noop, check=False, ) assert p.returncode == ExitCode.ok, p.stderr @@ -717,35 +739,52 @@ def test_decompression_bomb(resources, outpdf): assert p.returncode == 0 -def test_text_curves(spoof_tesseract_noop, resources, outpdf): +def test_text_curves(resources, outpdf): with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100): - check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop) + check_ocrmypdf( + resources / 'vector.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + ) info = PdfInfo(outpdf) assert len(info.pages[0].images) == 0, "added images to the vector PDF" check_ocrmypdf( - resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop + resources / 'vector.pdf', + outpdf, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) info = PdfInfo(outpdf) assert len(info.pages[0].images) != 0, "force did not rasterize" -def test_output_is_dir(spoof_tesseract_noop, resources, outdir): +def test_output_is_dir(resources, outdir): p, _out, err = run_ocrmypdf( - resources / 'trivial.pdf', outdir, '--force-ocr', env=spoof_tesseract_noop + resources / 'trivial.pdf', + outdir, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.file_access_error assert 'is not a writable file' in err @pytest.mark.skipif(os.name == 'nt', reason="symlink needs admin permissions") -def test_output_is_symlink(spoof_tesseract_noop, resources, outdir): +def test_output_is_symlink(resources, outdir): sym = Path(outdir / 'this_is_a_symlink') sym.symlink_to(outdir / 'out.pdf') p, _out, err = run_ocrmypdf( - resources / 'trivial.pdf', sym, '--force-ocr', env=spoof_tesseract_noop + resources / 'trivial.pdf', + sym, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.ok, err assert (outdir / 'out.pdf').stat().st_size > 0, 'target file not created' @@ -781,9 +820,7 @@ def test_version_check(): [0.0, 1, 'pdf', True], ], ) -def test_fast_web_view( - spoof_tesseract_noop, resources, outpdf, threshold, optimize, output_type, expected -): +def test_fast_web_view(resources, outpdf, threshold, optimize, output_type, expected): check_ocrmypdf( resources / 'trivial.pdf', outpdf, @@ -793,18 +830,20 @@ def test_fast_web_view( optimize, '--output-type', output_type, - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) with pikepdf.open(outpdf) as pdf: assert pdf.is_linearized == expected -def test_image_dpi_not_image(caplog, spoof_tesseract_noop, resources, outpdf): +def test_image_dpi_not_image(caplog, resources, outpdf): check_ocrmypdf( resources / 'trivial.pdf', outpdf, '--image-dpi', '100', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert '--image-dpi is being ignored' in caplog.text diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 795dcc43..59250eac 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -51,7 +51,7 @@ spoof = pytest.helpers.spoof @pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) -def test_preserve_metadata(spoof_tesseract_noop, output_type, resources, outpdf): +def test_preserve_metadata(output_type, resources, outpdf): pdf_before = pikepdf.open(resources / 'graph.pdf') output = check_ocrmypdf( @@ -59,7 +59,8 @@ def test_preserve_metadata(spoof_tesseract_noop, output_type, resources, outpdf) outpdf, '--output-type', output_type, - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) pdf_after = pikepdf.open(output) @@ -72,7 +73,7 @@ def test_preserve_metadata(spoof_tesseract_noop, output_type, resources, outpdf) @pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) -def test_override_metadata(spoof_tesseract_noop, output_type, resources, outpdf): +def test_override_metadata(output_type, resources, outpdf): input_file = resources / 'c02-22.pdf' german = 'Du siehst den Wald vor lauter Bäumen nicht.' chinese = '孔子' @@ -86,7 +87,8 @@ def test_override_metadata(spoof_tesseract_noop, output_type, resources, outpdf) chinese, '--output-type', output_type, - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.ok, err @@ -106,7 +108,7 @@ def test_override_metadata(spoof_tesseract_noop, output_type, resources, outpdf) assert pdfa_info['output'] == output_type -def test_high_unicode(spoof_tesseract_noop, resources, no_outpdf): +def test_high_unicode(resources, no_outpdf): # Ghostscript doesn't support high Unicode, so neither do we, to be # safe @@ -120,7 +122,8 @@ def test_high_unicode(spoof_tesseract_noop, resources, no_outpdf): high_unicode, '--output-type', 'pdfa', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == ExitCode.bad_args, err @@ -129,9 +132,7 @@ def test_high_unicode(spoof_tesseract_noop, resources, no_outpdf): @pytest.mark.skipif(not fitz, reason="test uses fitz") @pytest.mark.parametrize('ocr_option', ['--skip-text', '--force-ocr']) @pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) -def test_bookmarks_preserved( - spoof_tesseract_noop, output_type, ocr_option, resources, outpdf -): +def test_bookmarks_preserved(output_type, ocr_option, resources, outpdf): input_file = resources / 'toc.pdf' before_toc = fitz.Document(str(input_file)).getToC() @@ -141,7 +142,8 @@ def test_bookmarks_preserved( ocr_option, '--output-type', output_type, - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) after_toc = fitz.Document(str(outpdf)).getToC() @@ -156,13 +158,16 @@ def seconds_between_dates(date1, date2): @pytest.mark.parametrize('infile', ['trivial.pdf', 'jbig2.pdf']) @pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) -def test_creation_date_preserved( - spoof_tesseract_noop, output_type, resources, infile, outpdf -): +def test_creation_date_preserved(output_type, resources, infile, outpdf): input_file = resources / infile check_ocrmypdf( - input_file, outpdf, '--output-type', output_type, env=spoof_tesseract_noop + input_file, + outpdf, + '--output-type', + output_type, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) pdf_before = pikepdf.open(input_file) @@ -185,7 +190,7 @@ def test_creation_date_preserved( @pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) -def test_xml_metadata_preserved(spoof_tesseract_noop, output_type, resources, outpdf): +def test_xml_metadata_preserved(output_type, resources, outpdf): input_file = resources / 'graph.pdf' try: @@ -196,7 +201,12 @@ def test_xml_metadata_preserved(spoof_tesseract_noop, output_type, resources, ou before = file_to_dict(str(input_file)) check_ocrmypdf( - input_file, outpdf, '--output-type', output_type, env=spoof_tesseract_noop + input_file, + outpdf, + '--output-type', + output_type, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) after = file_to_dict(str(outpdf)) @@ -274,9 +284,14 @@ def test_srgb_in_unicode_path(tmp_path): generate_pdfa_ps(dstdir / 'out.ps') -def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): +def test_kodak_toc(resources, outpdf): _output = check_ocrmypdf( - resources / 'kcs.pdf', outpdf, '--output-type', 'pdf', env=spoof_tesseract_noop + resources / 'kcs.pdf', + outpdf, + '--output-type', + 'pdf', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) p = pikepdf.open(outpdf) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index f9f6747c..e8cf0717 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -54,7 +54,7 @@ def test_mono_not_inverted(resources, outdir): @pytest.mark.skipif(not pngquant.available(), reason='need pngquant') -def test_jpg_png_params(resources, outpdf, spoof_tesseract_noop): +def test_jpg_png_params(resources, outpdf): check_ocrmypdf( resources / 'crom.png', outpdf, @@ -66,13 +66,14 @@ def test_jpg_png_params(resources, outpdf, spoof_tesseract_noop): '50', '--png-quality', '20', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc') @pytest.mark.parametrize('lossy', [False, True]) -def test_jbig2_lossy(lossy, resources, outpdf, spoof_tesseract_noop): +def test_jbig2_lossy(lossy, resources, outpdf): args = [ resources / 'ccitt.pdf', outpdf, @@ -84,11 +85,13 @@ def test_jbig2_lossy(lossy, resources, outpdf, spoof_tesseract_noop): '50', '--png-quality', '20', + '--plugin', + 'tests/plugins/tesseract_noop.py', ] if lossy: args.append('--jbig2-lossy') - check_ocrmypdf(*args, env=spoof_tesseract_noop) + check_ocrmypdf(*args) pdf = pikepdf.open(outpdf) pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values()))) @@ -104,7 +107,7 @@ def test_jbig2_lossy(lossy, resources, outpdf, spoof_tesseract_noop): not jbig2enc.available() or not pngquant.available(), reason='need jbig2enc and pngquant', ) -def test_flate_to_jbig2(resources, outdir, spoof_tesseract_noop): +def test_flate_to_jbig2(resources, outdir): # This test requires an image that pngquant is capable of converting to # to 1bpp - so use an existing 1bpp image, convert up, confirm it can # convert down @@ -122,7 +125,8 @@ def test_flate_to_jbig2(resources, outdir, spoof_tesseract_noop): '50', '--optimize', '3', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) pdf = pikepdf.open(outdir / 'out.pdf') diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index b90517eb..08e34bec 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import logging from math import isclose import pytest @@ -38,16 +37,18 @@ spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] -def test_deskew(spoof_tesseract_noop, resources, outdir): +def test_deskew(resources, outdir): # Run with deskew deskewed_pdf = check_ocrmypdf( - resources / 'skew.pdf', outdir / 'skew.pdf', '-d', env=spoof_tesseract_noop + resources / 'skew.pdf', + outdir / 'skew.pdf', + '-d', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) # Now render as an image again and use Leptonica to find the skew angle # to confirm that it was deskewed - log = logging.getLogger() - deskewed_png = outdir / 'deskewed.png' ghostscript.rasterize_pdf( @@ -65,7 +66,7 @@ def test_deskew(spoof_tesseract_noop, resources, outdir): assert -0.5 < skew_angle < 0.5, "Deskewing failed" -def test_remove_background(spoof_tesseract_noop, resources, outdir): +def test_remove_background(resources, outdir): # Ensure the input image does not contain pure white/black with Image.open(resources / 'congress.jpg') as im: assert im.getextrema() != ((0, 255), (0, 255), (0, 255)) @@ -76,7 +77,8 @@ def test_remove_background(spoof_tesseract_noop, resources, outdir): '--remove-background', '--image-dpi', '150', - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) output_png = outdir / 'remove_bg.png' diff --git a/tests/test_stdio.py b/tests/test_stdio.py index e57c11f0..d78b1644 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -38,25 +38,24 @@ def spoof_tess_bad_utf8(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') -def test_stdin(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): +def test_stdin(ocrmypdf_exec, resources, outpdf): input_file = str(resources / 'francais.pdf') output_file = str(outpdf) # Runs: ocrmypdf - output.pdf < testfile.pdf with open(input_file, 'rb') as input_stream: - p_args = ocrmypdf_exec + ['-', output_file] - p = run( - p_args, - stdout=PIPE, - stderr=PIPE, - stdin=input_stream, - env=spoof_tesseract_noop, - ) + p_args = ocrmypdf_exec + [ + '-', + output_file, + '--plugin', + 'tests/plugins/tesseract_noop.py', + ] + p = run(p_args, stdout=PIPE, stderr=PIPE, stdin=input_stream) assert p.returncode == ExitCode.ok -def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): - if 'COV_CORE_DATAFILE' in spoof_tesseract_noop: +def test_stdout(ocrmypdf_exec, resources, outpdf): + if 'COV_CORE_DATAFILE' in os.environ: pytest.skip(msg="Coverage uses stdout") input_file = str(resources / 'francais.pdf') @@ -64,14 +63,13 @@ def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): # Runs: ocrmypdf francais.pdf - > test_stdout.pdf with open(output_file, 'wb') as output_stream: - p_args = ocrmypdf_exec + [input_file, '-'] - p = run( - p_args, - stdout=output_stream, - stderr=PIPE, - stdin=DEVNULL, - env=spoof_tesseract_noop, - ) + p_args = ocrmypdf_exec + [ + input_file, + '-', + '--plugin', + 'tests/plugins/tesseract_noop.py', + ] + p = run(p_args, stdout=output_stream, stderr=PIPE, stdin=DEVNULL) assert p.returncode == ExitCode.ok assert check_pdf(output_file) @@ -81,7 +79,7 @@ def test_stdout(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): sys.version_info[0:3] >= (3, 6, 4), reason="issue fixed in Python 3.6.4" ) @pytest.mark.skipif(os.name == 'nt', reason="POSIX problem") -def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): +def test_closed_streams(ocrmypdf_exec, resources, outpdf): input_file = str(resources / 'francais.pdf') output_file = str(outpdf) @@ -89,14 +87,18 @@ def test_closed_streams(spoof_tesseract_noop, ocrmypdf_exec, resources, outpdf): os.close(0) os.close(1) - p_args = ocrmypdf_exec + [input_file, output_file] + p_args = ocrmypdf_exec + [ + input_file, + output_file, + '--plugin', + 'tests/plugins/tesseract_noop.py', + ] p = Popen( # pylint: disable=subprocess-popen-preexec-fn p_args, close_fds=True, stdout=None, stderr=PIPE, stdin=None, - env=spoof_tesseract_noop, preexec_fn=evil_closer, ) out, err = p.communicate() @@ -123,12 +125,16 @@ def test_bad_locale(): os.name == 'nt' and sys.version_info < (3, 8), reason="Windows does not like this; not sure how to fix", ) -def test_dev_null(spoof_tesseract_noop, resources): - if 'COV_CORE_DATAFILE' in spoof_tesseract_noop: +def test_dev_null(resources): + if 'COV_CORE_DATAFILE' in os.environ: pytest.skip(msg="Coverage uses stdout") p, out, err = run_ocrmypdf( - resources / 'trivial.pdf', os.devnull, '--force-ocr', env=spoof_tesseract_noop + resources / 'trivial.pdf', + os.devnull, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert p.returncode == 0, "could not send output to /dev/null" assert len(out) == 0, "wrote to stdout" diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index bd04da2c..5ef9fac3 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -60,45 +60,54 @@ def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") -def test_clean(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf(resources / "skew.pdf", outpdf, "-c", env=spoof_tesseract_noop) +def test_clean(resources, outpdf): + check_ocrmypdf( + resources / "skew.pdf", + outpdf, + "-c", + '--plugin', + 'tests/plugins/tesseract_noop.py', + ) @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") -def test_unpaper_args_valid(spoof_tesseract_noop, resources, outpdf): +def test_unpaper_args_valid(resources, outpdf): check_ocrmypdf( resources / "skew.pdf", outpdf, "-c", "--unpaper-args", "--layout double", # Spaces required here - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") -def test_unpaper_args_invalid_filename(spoof_tesseract_noop, resources, outpdf): +def test_unpaper_args_invalid_filename(resources, outpdf): p, out, err = run_ocrmypdf( resources / "skew.pdf", outpdf, "-c", "--unpaper-args", "/etc/passwd", - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) assert "No filenames allowed" in err assert p.returncode == ExitCode.bad_args @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") -def test_unpaper_args_invalid(spoof_tesseract_noop, resources, outpdf): +def test_unpaper_args_invalid(resources, outpdf): p, out, err = run_ocrmypdf( resources / "skew.pdf", outpdf, "-c", "--unpaper-args", "unpaper is not going to like these arguments", - env=spoof_tesseract_noop, + '--plugin', + 'tests/plugins/tesseract_noop.py', ) # Can't tell difference between unpaper choking on bad arguments or some # other unpaper failure From daca9197755e2872b26d010ebf4b27a5b2661b6f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 28 May 2020 15:01:51 -0700 Subject: [PATCH 489/880] Mark pdfminer.six 20200517 as supported --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 45e0ae1e..b91e182d 100644 --- a/setup.py +++ b/setup.py @@ -82,7 +82,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, <= 20200402', + 'pdfminer.six >= 20191110, <= 20200517', 'pikepdf >= 1.8.1, < 2', 'Pillow >= 6.2.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 1d0b8641a0447d5f120921c336ee46b6a6207d7f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jun 2020 00:35:49 -0700 Subject: [PATCH 490/880] Improve file size increase warning to account for changes to small files Fixes #569 --- src/ocrmypdf/_validation.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index ef440e0f..9006e115 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -24,6 +24,7 @@ import sys from pathlib import Path from shutil import copyfileobj +import pikepdf import PIL from ._unicodefun import verify_python3_env @@ -410,8 +411,15 @@ def report_output_file_size(options, input_file, output_file): input_size = Path(input_file).stat().st_size except FileNotFoundError: return # Outputting to stream or something + with pikepdf.open(output_file) as p: + # Overhead constants obtained by estimating amount of data added by OCR + # PDF/A conversion, and possible XMP metadata addition, with compression + FILE_OVERHEAD = 4000 + OCR_PER_PAGE_OVERHEAD = 3000 + reasonable_overhead = FILE_OVERHEAD + OCR_PER_PAGE_OVERHEAD * len(p.pages) ratio = output_size / input_size - if ratio < 1.35 or input_size < 25000: + reasonable_ratio = output_size / (input_size + reasonable_overhead) + if reasonable_ratio < 1.35 or input_size < 25000: return # Seems fine reasons = [] From 4f4ad0fb7602f4c10e9acf17ae67e2d0b778fe2c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jun 2020 01:49:47 -0700 Subject: [PATCH 491/880] Convert tesseract_big_image_error to plugin --- tests/plugins/tesseract_big_image_error.py | 61 +++++++++++++++++ tests/spoof/tesseract_big_image_error.py | 77 ---------------------- tests/test_main.py | 12 +--- 3 files changed, 64 insertions(+), 86 deletions(-) create mode 100644 tests/plugins/tesseract_big_image_error.py delete mode 100755 tests/spoof/tesseract_big_image_error.py diff --git a/tests/plugins/tesseract_big_image_error.py b/tests/plugins/tesseract_big_image_error.py new file mode 100644 index 00000000..f040855c --- /dev/null +++ b/tests/plugins/tesseract_big_image_error.py @@ -0,0 +1,61 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +from subprocess import CalledProcessError +from unittest.mock import patch + +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins.tesseract_ocr import TesseractOcrEngine + + +def raise_size_exception(*args, **kwargs): + raise CalledProcessError( + 1, + 'tesseract', + output=b"Image too large: (33830, 14959)\nError during processing.", + stderr=b"", + ) + + +class BigImageErrorOcrEngine(TesseractOcrEngine): + @staticmethod + def get_orientation(input_file, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + return TesseractOcrEngine.get_orientation(input_file, options) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + TesseractOcrEngine.generate_hocr( + input_file, output_hocr, output_text, options + ) + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + TesseractOcrEngine.generate_pdf( + input_file, output_pdf, output_text, options + ) + + +@hookimpl +def get_ocr_engine(): + return BigImageErrorOcrEngine() diff --git a/tests/spoof/tesseract_big_image_error.py b/tests/spoof/tesseract_big_image_error.py deleted file mode 100755 index 8b710bee..00000000 --- a/tests/spoof/tesseract_big_image_error.py +++ /dev/null @@ -1,77 +0,0 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - - -import sys - -VERSION_STRING = '''tesseract 4.0.0 - leptonica-1.77.0 - libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 - Found AVX2 - Found AVX - Found SSE -SPOOFED: return error claiming image too big -''' - -"""Simulates an error of Tesseract failing on attempts to process large images - -""" - - -def main(): - if sys.argv[1] == '--version': - print(VERSION_STRING, file=sys.stderr) - sys.exit(0) - elif sys.argv[1] == '--list-langs': - print('List of available languages (1):\neng\n', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == '--print-parameters': - print('A parameter list would go here\ntextonly_pdf 0\n', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == 'hocr': - print( - "Image too large: (33830, 14959)\n" "Error during processing.", - file=sys.stderr, - ) - sys.exit(1) - elif sys.argv[-2] == 'pdf': - print( - "Image too large: (33830, 14959)\n" "Error during processing.", - file=sys.stderr, - ) - sys.exit(1) - elif sys.argv[-1] == 'stdout': - print( - "Image too large: (33830, 14959)\n" "Error during processing.", - file=sys.stderr, - ) - sys.exit(1) - else: - print("Spoof doesn't understand arguments", file=sys.stderr) - print(sys.argv, file=sys.stderr) - sys.exit(1) - - sys.exit(0) - - -if __name__ == '__main__': - main() diff --git a/tests/test_main.py b/tests/test_main.py index e161730e..bf95bd76 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -50,11 +50,6 @@ def spoof_tesseract_crash(tmp_path_factory): return spoof(tmp_path_factory, tesseract='tesseract_crash.py') -@pytest.fixture -def spoof_tesseract_big_image_error(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_big_image_error.py') - - def test_quick(spoof_tesseract_cache, resources, outpdf): check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) @@ -337,9 +332,7 @@ def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf) @pytest.mark.parametrize('renderer', RENDERERS) @pytest.mark.slow -def test_tesseract_image_too_big( - renderer, spoof_tesseract_big_image_error, resources, outpdf -): +def test_tesseract_image_too_big(renderer, resources, outpdf): check_ocrmypdf( resources / 'hugemono.pdf', outpdf, @@ -348,7 +341,8 @@ def test_tesseract_image_too_big( renderer, '--max-image-mpixels', '0', - env=spoof_tesseract_big_image_error, + '--plugin', + 'tests/plugins/tesseract_big_image_error.py', ) From 82e7eb91d2d3f56fd627bc3ac48699c17b09e771 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jun 2020 01:50:02 -0700 Subject: [PATCH 492/880] Tidy tesseract_noop --- tests/plugins/tesseract_noop.py | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/tests/plugins/tesseract_noop.py b/tests/plugins/tesseract_noop.py index 1db91fe4..7f78d3e0 100644 --- a/tests/plugins/tesseract_noop.py +++ b/tests/plugins/tesseract_noop.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # Permission is hereby granted, free of charge, to any person obtaining a # copy of this software and associated documentation files (the @@ -31,10 +30,6 @@ In 'pdf' mode, convert the image to PDF using another program. In orientation check mode, report the orientation is upright. """ -import sys -from pathlib import Path - -import img2pdf import pikepdf from PIL import Image From 1b92f447c3c44440641e87acd1af2bc13bd13b37 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jun 2020 02:36:41 -0700 Subject: [PATCH 493/880] Convert tesseract_crash to plugin --- src/ocrmypdf/exec/tesseract.py | 4 +- tests/plugins/tesseract_crash.py | 64 ++++++++++++++++++++++++++ tests/spoof/tesseract_crash.py | 78 -------------------------------- tests/test_main.py | 26 ++++++----- 4 files changed, 82 insertions(+), 90 deletions(-) create mode 100755 tests/plugins/tesseract_crash.py delete mode 100755 tests/spoof/tesseract_crash.py diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 39046ce5..b1ae3c3c 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -167,7 +167,9 @@ def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) except CalledProcessError as e: - tesseract_log_output(e.output) + # breakpoint() + tesseract_log_output(e.stdout) + tesseract_log_output(e.stderr) if ( b'Too few characters. Skipping this page' in e.output or b'Image too large' in e.output diff --git a/tests/plugins/tesseract_crash.py b/tests/plugins/tesseract_crash.py new file mode 100755 index 00000000..806af41b --- /dev/null +++ b/tests/plugins/tesseract_crash.py @@ -0,0 +1,64 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +import signal +import sys +from subprocess import CalledProcessError +from unittest.mock import patch + +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins.tesseract_ocr import TesseractOcrEngine + + +def raise_crash(*args, **kwargs): + raise CalledProcessError( + 128 + signal.SIGABRT, + 'tesseract', + output=b"", + stderr=b"libc++abi.dylib: terminating with uncaught exception of type " + + b"std::bad_alloc: std::bad_alloc", + ) + + +class CrashOcrEngine(TesseractOcrEngine): + @staticmethod + def get_orientation(input_file, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + return TesseractOcrEngine.get_orientation(input_file, options) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + TesseractOcrEngine.generate_hocr( + input_file, output_hocr, output_text, options + ) + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + TesseractOcrEngine.generate_pdf( + input_file, output_pdf, output_text, options + ) + + +@hookimpl +def get_ocr_engine(): + return CrashOcrEngine() diff --git a/tests/spoof/tesseract_crash.py b/tests/spoof/tesseract_crash.py deleted file mode 100755 index 03c7dbde..00000000 --- a/tests/spoof/tesseract_crash.py +++ /dev/null @@ -1,78 +0,0 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -import signal -import sys - -VERSION_STRING = '''tesseract 4.0.0 - leptonica-1.77.0 - libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 - Found AVX2 - Found AVX - Found SSE -SPOOFED: CRASH ON OCR or --psm 0 -''' - -"""Simulates a Tesseract crash when asked to run OCR - -It isn't strictly necessary to crash the process and that has unwanted -side effects like triggering core dumps or error reporting, logging and such. -It's enough to dump some text to stderr and return an error code. - -Follows the POSIX(?) convention of returning 128 + signal number. - -""" - - -def main(): - if sys.argv[1] == '--version': - print(VERSION_STRING, file=sys.stderr) - sys.exit(0) - elif sys.argv[1] == '--list-langs': - print('List of available languages (1):\neng', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == '--print-parameters': - print('A parameter list would go here\ntextonly_pdf 0\n', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == 'hocr': - print("KABOOM! Tesseract failed for some reason", file=sys.stderr) - sys.exit(128 + signal.SIGSEGV) - elif sys.argv[-2] == 'pdf': - print("KABOOM! Tesseract failed for some reason", file=sys.stderr) - sys.exit(128 + signal.SIGSEGV) - elif sys.argv[-1] == 'stdout': - print( - "libc++abi.dylib: terminating with uncaught exception of type " - "std::bad_alloc: std::bad_alloc", - file=sys.stderr, - ) - sys.exit(128 + signal.SIGABRT) - else: - print("Spoof doesn't understand arguments", file=sys.stderr) - print(sys.argv, file=sys.stderr) - sys.exit(1) - - sys.exit(0) - - -if __name__ == '__main__': - main() diff --git a/tests/test_main.py b/tests/test_main.py index bf95bd76..b16b31d8 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -45,11 +45,6 @@ spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] -@pytest.fixture -def spoof_tesseract_crash(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_crash.py') - - def test_quick(spoof_tesseract_cache, resources, outpdf): check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) @@ -192,12 +187,16 @@ def test_blank_input_pdf(resources, outpdf): assert result == ExitCode.ok -def test_force_ocr_on_pdf_with_no_images(spoof_tesseract_crash, resources, no_outpdf): +def test_force_ocr_on_pdf_with_no_images(resources, no_outpdf): # As a correctness test, make sure that --force-ocr on a PDF with no # content still triggers tesseract. If tesseract crashes, then it was # called. p, _, _ = run_ocrmypdf( - resources / 'blank.pdf', no_outpdf, '--force-ocr', env=spoof_tesseract_crash + resources / 'blank.pdf', + no_outpdf, + '--force-ocr', + '--plugin', + 'tests/plugins/tesseract_crash.py', ) assert p.returncode == ExitCode.child_process_error assert not no_outpdf.exists() @@ -304,7 +303,7 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) -def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): +def test_tesseract_crash(renderer, resources, no_outpdf): p, _, err = run_ocrmypdf( resources / 'ccitt.pdf', no_outpdf, @@ -312,16 +311,21 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf): '1', '--pdf-renderer', renderer, - env=spoof_tesseract_crash, + '--plugin', + 'tests/plugins/tesseract_crash.py', ) assert p.returncode == ExitCode.child_process_error assert not no_outpdf.exists() assert "SubprocessOutputError" in err -def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf): +def test_tesseract_crash_autorotate(resources, no_outpdf): p, out, err = run_ocrmypdf( - resources / 'ccitt.pdf', no_outpdf, '-r', env=spoof_tesseract_crash + resources / 'ccitt.pdf', + no_outpdf, + '-r', + '--plugin', + 'tests/plugins/tesseract_crash.py', ) assert p.returncode == ExitCode.child_process_error assert not no_outpdf.exists() From c6b2fa8851df3c1b23c3c2f2c525700d76ddb0e3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 2 Jun 2020 02:42:14 -0700 Subject: [PATCH 494/880] Remove unpaper spoof; no plugin needed --- tests/spoof/unpaper_oldversion.py | 37 ------------------------------- tests/test_unpaper.py | 20 ++++++++--------- 2 files changed, 10 insertions(+), 47 deletions(-) delete mode 100755 tests/spoof/unpaper_oldversion.py diff --git a/tests/spoof/unpaper_oldversion.py b/tests/spoof/unpaper_oldversion.py deleted file mode 100755 index ff2e27ea..00000000 --- a/tests/spoof/unpaper_oldversion.py +++ /dev/null @@ -1,37 +0,0 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - - -import sys - - -def main(): - if sys.argv[1] == '--version': - print('0.5') - sys.exit(0) - - print("Only supports --version") - sys.exit(1) - - -if __name__ == '__main__': - main() diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 5ef9fac3..3b9f155e 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -35,11 +35,6 @@ spoof = pytest.helpers.spoof have_unpaper = pytest.helpers.have_unpaper -@pytest.fixture -def spoof_unpaper_oldversion(tmp_path_factory): - return spoof(tmp_path_factory, unpaper="unpaper_oldversion.py") - - def test_no_unpaper(resources, no_outpdf): input_ = fspath(resources / "c02-22.pdf") output = fspath(no_outpdf) @@ -52,11 +47,16 @@ def test_no_unpaper(resources, no_outpdf): check_options(options, pm) -def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): - p, out, err = run_ocrmypdf( - resources / "c02-22.pdf", no_outpdf, "--clean", env=spoof_unpaper_oldversion - ) - assert p.returncode == ExitCode.missing_dependency +def test_old_unpaper(resources, no_outpdf): + input_ = fspath(resources / "c02-22.pdf") + output = fspath(no_outpdf) + + _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) + with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: + mock_unpaper_version.return_value = '0.5' + + with pytest.raises(MissingDependencyError): + check_options(options, pm) @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") From 5f47aac36f9fb0efa87b562039bcb1d5073303b2 Mon Sep 17 00:00:00 2001 From: jhgarrison Date: Wed, 3 Jun 2020 13:16:23 -0700 Subject: [PATCH 495/880] Add installation instructions for Windows/Cygwin64 (#571) Co-authored-by: Jim Garrison --- docs/installation.rst | 48 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/docs/installation.rst b/docs/installation.rst index 6c2b1f3e..f0211077 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -575,6 +575,54 @@ Docker You can also :ref:`Install the Docker ` container on Windows. Ensure that your command prompt can run the docker "hello world" container. + +Installing on Cygwin64 under Windows +==================================== + +First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``:: + + python36 (or later) + python3?-devel + python3?-pip + python3?-lxml + python3?-imaging + + (where 3? means match the version of python3 you installed) + + gcc-g++ + ghostscript (<=9.50 or >=9.52-2 see note below) + libexempi3 + libexempi-devel + libffi6 + libffi-devel + pngquant + qpdf + libqpdf-devel + tesseract-ocr + tesseract-ocr-devel + + Note: The Cygwin package for Ghostscript in versions 9.52 and + 9.52-1 contained a bug that caused an exception to occur when + ocrmypdf invoked gs. Make sure you have either 9.50 (or earlier) + or 9.52-2 (or later). + +Then open a Cygwin terminal (i.e. ``mintty``), run the following commands. +Note that if you are using the version of ``pip`` that was installed with the +Cygwin Python package, the command name will be ``pip3``. If you have since +updated ``pip`` (with, for instance ``pip3 install --upgrade pip``) the the +command is likely just ``pip`` instead of ``pip3``: + +.. code-block:: bash + + pip3 install wheel + pip3 install ocrmypdf + +There is one optional dependency, "unpaper" that is currently not +available under Cygwin. Without it, certain options such as --clean +will produce an error message. However, the OCR-to-text-layer +functionality is available. + + Installing with Python pip ========================== From d118132fa6f0c52210276acfde80d9f91be37f4f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jun 2020 13:16:55 -0700 Subject: [PATCH 496/880] layout: look for text in XObjects too --- src/ocrmypdf/pdfinfo/layout.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index af9d7961..1368a75b 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -269,7 +269,9 @@ class TextPositionTracker(PDFLayoutAnalyzer): def get_page_analysis(infile, pageno, pscript5_mode): rman = pdfminer.pdfinterp.PDFResourceManager(caching=True) - dev = TextPositionTracker(rman, laparams=LAParams()) + dev = TextPositionTracker( + rman, laparams=LAParams(all_texts=True, detect_vertical=True) + ) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) if pscript5_mode: From 5e14d5b0dd853e8958a8e81c866dbce1687f7889 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jun 2020 13:24:55 -0700 Subject: [PATCH 497/880] Fix test_report_file_size Use more realistic test data --- tests/test_validation.py | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index 199cfcd2..01f393ba 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -19,6 +19,7 @@ import logging import os from unittest.mock import patch +import pikepdf import pytest import ocrmypdf._validation as vd @@ -111,15 +112,20 @@ def test_output_tty(): def test_report_file_size(tmp_path, caplog): in_ = tmp_path / 'a.pdf' out = tmp_path / 'b.pdf' - in_.write_bytes(b'123') - out.write_bytes(b'') + pdf = pikepdf.new() + pdf.save(in_) + pdf.save(out) opts = make_opts() vd.report_output_file_size(opts, in_, out) assert caplog.text == '' caplog.clear() - os.truncate(in_, 25001) - os.truncate(out, 50000) + waste_of_space = b'Dummy' * 5000 + pdf.root.Dummy = waste_of_space + pdf.save(in_) + pdf.root.Dummy2 = waste_of_space + waste_of_space + pdf.save(out) + with patch('ocrmypdf._validation.jbig2enc.available', return_value=True), patch( 'ocrmypdf._validation.pngquant.available', return_value=True ): From e60f4d3f43254e6040af4271877a4aad2c54f642 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jun 2020 13:27:05 -0700 Subject: [PATCH 498/880] docs: tidy Cygwin install --- docs/installation.rst | 29 ++++++++++++++--------------- 1 file changed, 14 insertions(+), 15 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index f0211077..375ebc0b 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -575,7 +575,6 @@ Docker You can also :ref:`Install the Docker ` container on Windows. Ensure that your command prompt can run the docker "hello world" container. - Installing on Cygwin64 under Windows ==================================== @@ -601,27 +600,27 @@ First install the the following prerequisite Cygwin packages using ``setup-x86_6 tesseract-ocr tesseract-ocr-devel - Note: The Cygwin package for Ghostscript in versions 9.52 and - 9.52-1 contained a bug that caused an exception to occur when - ocrmypdf invoked gs. Make sure you have either 9.50 (or earlier) - or 9.52-2 (or later). +.. note:: -Then open a Cygwin terminal (i.e. ``mintty``), run the following commands. -Note that if you are using the version of ``pip`` that was installed with the -Cygwin Python package, the command name will be ``pip3``. If you have since -updated ``pip`` (with, for instance ``pip3 install --upgrade pip``) the the -command is likely just ``pip`` instead of ``pip3``: + The Cygwin package for Ghostscript in versions 9.52 and + 9.52-1 contained a bug that caused an exception to occur when + ocrmypdf invoked gs. Make sure you have either 9.50 (or earlier) + or 9.52-2 (or later). + +Then open a Cygwin terminal (i.e. ``mintty``), run the following commands. Note +that if you are using the version of ``pip`` that was installed with the Cygwin +Python package, the command name will be ``pip3``. If you have since updated +``pip`` (with, for instance ``pip3 install --upgrade pip``) the the command is +likely just ``pip`` instead of ``pip3``: .. code-block:: bash pip3 install wheel pip3 install ocrmypdf -There is one optional dependency, "unpaper" that is currently not -available under Cygwin. Without it, certain options such as --clean -will produce an error message. However, the OCR-to-text-layer -functionality is available. - +The optional dependency "unpaper" that is currently not available under Cygwin. +Without it, certain options such as ``--clean`` will produce an error message. +However, the OCR-to-text-layer functionality is available. Installing with Python pip ========================== From 00daa51a7393cf4cf0db36241515b6835b2205dd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Jun 2020 13:28:35 -0700 Subject: [PATCH 499/880] v9.8.2 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 3ed82045..9a6eeb3b 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,14 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v9.8.2 +====== + +- Fixed an issue where OCRmyPDF would ignore text inside Form XObject when + making certain decisions about whether a document already had text. +- Fixed file size increase warning to take overhead of small files into account. +- Added instructions for installing on Cygwin. + v9.8.1 ====== From ec3f506500175b37cb76154e4a6c4dc33edd58bd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 5 Jun 2020 16:36:11 -0700 Subject: [PATCH 500/880] Convert tesseract_badutf8 to plugin --- tests/plugins/tesseract_badutf8.py | 63 +++++++++++++++++++++++ tests/spoof/tesseract_badutf8.py | 80 ------------------------------ tests/test_stdio.py | 5 -- 3 files changed, 63 insertions(+), 85 deletions(-) create mode 100644 tests/plugins/tesseract_badutf8.py delete mode 100755 tests/spoof/tesseract_badutf8.py diff --git a/tests/plugins/tesseract_badutf8.py b/tests/plugins/tesseract_badutf8.py new file mode 100644 index 00000000..b87a3dac --- /dev/null +++ b/tests/plugins/tesseract_badutf8.py @@ -0,0 +1,63 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +"""Tesseract bad utf8 + +In some cases, some versions of Tesseract can output binary gibberish or data +that is not UTF-8 compatible, so we are forced to check that we can convert it +and present it to the user. +""" + +from subprocess import CalledProcessError +from unittest.mock import patch + +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins.tesseract_ocr import TesseractOcrEngine + + +def bad_utf8(*args, **kwargs): + raise CalledProcessError( + 1, + 'tesseract', + output=b'\x96\xb3\x8c\xf8\x82\xc8UTF-8\x0a', # "Invalid UTF-8" in Shift JIS + stderr=b"", + ) + + +class BadUtf8OcrEngine(TesseractOcrEngine): + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=bad_utf8): + TesseractOcrEngine.generate_hocr( + input_file, output_hocr, output_text, options + ) + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=bad_utf8): + TesseractOcrEngine.generate_pdf( + input_file, output_pdf, output_text, options + ) + + +@hookimpl +def get_ocr_engine(): + return BadUtf8OcrEngine() diff --git a/tests/spoof/tesseract_badutf8.py b/tests/spoof/tesseract_badutf8.py deleted file mode 100755 index 3cbb6625..00000000 --- a/tests/spoof/tesseract_badutf8.py +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env python3 -# © 2017 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -import sys - -"""Tesseract bad utf8 spoof - -In 'hocr' mode or 'pdf' mode, return error code 1 and some non-Unicode -text because tesseract seems to do that in some cases related to -language pack version mismatches - -""" - - -VERSION_STRING = '''tesseract 4.0.0 - leptonica-1.77.0 - libjpeg 9c : libpng 1.6.35 : libtiff 4.0.10 : zlib 1.2.11 : libopenjp2 2.3.0 - Found AVX2 - Found AVX - Found SSE -SPOOFED -''' - -# Japanese "Invalid UTF-8" encoded in Shift JIS -BAD_UTF8 = b'\x96\xb3\x8c\xf8\x82\xc8UTF-8\x0a' - - -def main(): - if sys.argv[1] == '--version': - print(VERSION_STRING, file=sys.stderr) - sys.exit(0) - elif sys.argv[1] == '--list-langs': - print('List of available languages (1):\neng', file=sys.stderr) - sys.exit(0) - elif sys.argv[-2] == '--print-parameters': - print("Some parameters", file=sys.stderr) - print("textonly_pdf\t1\tSome help text") - sys.exit(0) - elif sys.argv[-2] in ('hocr', 'pdf'): - sys.stdout.buffer.write(BAD_UTF8) - sys.exit(1) - elif sys.argv[-1] == 'stdout': - # input file is at sys.argv[-2] but we don't look at it - print( - """Orientation: 0 -Orientation in degrees: 0 -Orientation confidence: 100.00 -Script: 1 -Script confidence: 100.00""", - file=sys.stderr, - ) - else: - print("Spoof doesn't understand arguments", file=sys.stderr) - print(sys.argv, file=sys.stderr) - sys.exit(1) - - sys.exit(0) - - -if __name__ == '__main__': - main() diff --git a/tests/test_stdio.py b/tests/test_stdio.py index d78b1644..2f58306a 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -33,11 +33,6 @@ run_ocrmypdf_api = pytest.helpers.run_ocrmypdf spoof = pytest.helpers.spoof -@pytest.fixture -def spoof_tess_bad_utf8(tmp_path_factory): - return spoof(tmp_path_factory, tesseract='tesseract_badutf8.py') - - def test_stdin(ocrmypdf_exec, resources, outpdf): input_file = str(resources / 'francais.pdf') output_file = str(outpdf) From 6268e2fafff02469381f29550626b059e4eb9256 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 5 Jun 2020 17:27:10 -0700 Subject: [PATCH 501/880] Begin replacing tests/spoof/tesseract_cache with plugin --- tests/plugins/tesseract_cache.py | 175 +++++++++++++++++++++++++++++++ tests/test_main.py | 6 +- 2 files changed, 179 insertions(+), 2 deletions(-) create mode 100644 tests/plugins/tesseract_cache.py diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py new file mode 100644 index 00000000..37b865f1 --- /dev/null +++ b/tests/plugins/tesseract_cache.py @@ -0,0 +1,175 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +import argparse +import json +import logging +import platform +import re +import shutil +from functools import partial +from pathlib import Path +from subprocess import PIPE, CalledProcessError, CompletedProcess +from unittest.mock import patch + +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins.tesseract_ocr import TesseractOcrEngine +from ocrmypdf.exec import run + +log = logging.getLogger(__name__) + +TESTS_ROOT = Path(__file__).resolve().parent.parent +CACHE_ROOT = TESTS_ROOT / 'cache' + + +parser = argparse.ArgumentParser( + prog='tesseract-cache', description='cache output of tesseract' +) +parser.add_argument('-l', '--language', action='append') +parser.add_argument('imagename') +parser.add_argument('outputbase') +parser.add_argument('configfiles', nargs='*') +parser.add_argument('--user-words', type=str) +parser.add_argument('--user-patterns', type=str) +parser.add_argument('-c', action='append') +parser.add_argument('--psm', type=int) +parser.add_argument('--oem', type=int) + + +def get_cache_folder(source_pdf, run_args, parsed_args): + def slugs(): + yield '' # so we don't start with a '-' which makes rm difficult + for arg in run_args[1:]: + if arg == parsed_args.imagename: + yield Path(parsed_args.imagename).name + elif arg == parsed_args.outputbase: + yield Path(parsed_args.outputbase).name + elif arg == '-c' or arg.startswith('textonly'): + pass + else: + yield arg + + argv_slug = '__'.join(slugs()) + argv_slug = argv_slug.replace('/', '___') + + return Path(CACHE_ROOT) / Path(source_pdf).stem / argv_slug + + +def cached_run(options, run_args, **run_kwargs): + run_args = [str(arg) for arg in run_args] # flatten PosixPaths + args = parser.parse_args(run_args[1:]) + + if args.imagename in ('stdin', '-'): + return run(run_args, **run_kwargs) + + source_file = options.input_file + cache_folder = get_cache_folder(source_file, run_args, args) + cache_folder.mkdir(parents=True, exist_ok=True) + + log.debug("Using Tesseract cache {cache_folder}") + + if (cache_folder / 'stderr.bin').exists(): + log.debug("Cache HIT") + + # Replicate stdout/err + if args.outputbase != 'stdout': + if not args.configfiles: + args.configfiles.append('txt') + for configfile in args.configfiles: + # cp cache -> output + tessfile = args.outputbase + '.' + configfile + shutil.copy(str(cache_folder / configfile) + '.bin', tessfile) + return CompletedProcess( + args=run_args, + returncode=0, + stdout=(cache_folder / 'stdout.bin').read_bytes(), + stderr=(cache_folder / 'stderr.bin').read_bytes(), + ) + + log.debug("Cache MISS") + + cache_kwargs = { + k: v for k, v in run_kwargs.items() if k not in ('stdout', 'stderr') + } + assert cache_kwargs['check'] + try: + p = run(run_args, stdout=PIPE, stderr=PIPE, **cache_kwargs) + except CalledProcessError as e: + log.exception(e) + raise # Pass exception onward + + # Update cache + (cache_folder / 'stdout.bin').write_bytes(p.stdout) + (cache_folder / 'stderr.bin').write_bytes(p.stderr) + + if args.outputbase != 'stdout': + if not args.configfiles: + args.configfiles.append('txt') + + for configfile in args.configfiles: + if configfile not in ('hocr', 'pdf', 'txt'): + continue + # cp pwd/{outputbase}.{configfile} -> {cache}/{configfile} + tessfile = args.outputbase + '.' + configfile + shutil.copy(tessfile, str(cache_folder / configfile) + '.bin') + + manifest = {} + manifest['tesseract_version'] = TesseractOcrEngine.version().replace('\n', ' ') + manifest['platform'] = platform.platform() + manifest['python'] = platform.python_version() + manifest['argv_slug'] = cache_folder.name + manifest['sourcefile'] = str(Path(source_file).relative_to(TESTS_ROOT)) + + def clean_sys_argv(): + for arg in run_args[1:]: + yield re.sub(r'.*/com.github.ocrmypdf[^/]+[/](.*)', r'$TMPDIR/\1', arg) + + manifest['args'] = list(clean_sys_argv()) + with (Path(CACHE_ROOT) / 'manifest.jsonl').open('a') as f: + json.dump(manifest, f) + f.write('\n') + f.flush() + + +class CacheOcrEngine(TesseractOcrEngine): + @staticmethod + def get_orientation(input_file, options): + with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + return TesseractOcrEngine.get_orientation(input_file, options) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + TesseractOcrEngine.generate_hocr( + input_file, output_hocr, output_text, options + ) + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + TesseractOcrEngine.generate_pdf( + input_file, output_pdf, output_text, options + ) + + +@hookimpl +def get_ocr_engine(): + return CacheOcrEngine() diff --git a/tests/test_main.py b/tests/test_main.py index b16b31d8..ceead7de 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -45,8 +45,10 @@ spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] -def test_quick(spoof_tesseract_cache, resources, outpdf): - check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache) +def test_quick(resources, outpdf): + check_ocrmypdf( + resources / 'ccitt.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_cache.py' + ) @pytest.mark.parametrize('renderer', RENDERERS) From a9a473f2e5d544082f70946f53b26b75f4a3496e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 5 Jun 2020 17:45:11 -0700 Subject: [PATCH 502/880] Convert all tesseract cache usages to plugin --- src/ocrmypdf/exec/tesseract.py | 9 +- tests/conftest.py | 7 -- tests/plugins/tesseract_cache.py | 26 ++++ tests/plugins/tesseract_noop.py | 2 +- tests/spoof/tesseract_cache.py | 198 ------------------------------- tests/test_main.py | 80 +++++++++---- tests/test_page_numbers.py | 4 +- tests/test_preprocessing.py | 19 ++- tests/test_rotation.py | 12 +- tests/test_tesseract.py | 1 - tests/test_unpaper.py | 1 - tests/test_userunit.py | 20 +++- 12 files changed, 118 insertions(+), 261 deletions(-) delete mode 100755 tests/spoof/tesseract_cache.py diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index b1ae3c3c..8e0f4608 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -167,7 +167,6 @@ def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) except CalledProcessError as e: - # breakpoint() tesseract_log_output(e.stdout) tesseract_log_output(e.stderr) if ( @@ -191,15 +190,17 @@ def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env return oc -def tesseract_log_output(stdout): +def tesseract_log_output(stream): tlog = TesseractLoggerAdapter( log, extra=log.extra if hasattr(log, 'extra') else None ) + if not stream: + return try: - text = stdout.decode() + text = stream.decode() except UnicodeDecodeError: - text = stdout.decode('utf-8', 'ignore') + text = stream.decode('utf-8', 'ignore') lines = text.splitlines() for line in lines: diff --git a/tests/conftest.py b/tests/conftest.py index bc188bf9..0b88a070 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -169,13 +169,6 @@ def spoof(tmp_path_factory, **kwargs): return env -@pytest.fixture -def spoof_tesseract_cache(tmp_path_factory): - if running_in_docker(): - return os.environ.copy() - return spoof(tmp_path_factory, tesseract="tesseract_cache.py") - - @pytest.fixture def resources(): return Path(TESTS_ROOT) / 'resources' diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py index 37b865f1..807d65b0 100644 --- a/tests/plugins/tesseract_cache.py +++ b/tests/plugins/tesseract_cache.py @@ -19,6 +19,31 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +"""Cache output of tesseract to speed up test suite + +The cache is keyed by by the input test file The input arguments are slugged +into a hideous filename that more or less represents them literally. Joined +together, this becomes the name of the cache folder. A few name files like +stdout, stderr, hocr, pdf, describe the output to reproduce. + +Changes to tests/resources/ or image processing algorithms don't trigger a +cache miss. By design, an input image that varies according to platform +differences (e.g. JPEG decoders are allowed to produce differing outputs, +and in practice they do) will still be a cache hit. By design, an +invocation of tesseract with the same parameters from a different test case +will be a hit. It's fragile. + +The tests/cache/manifest.jsonl is a JSON lines file that contains +information about the system that produced the results used when cache was +generated. This mainly a log to answer questions about how the files +were produced. + +Certain operations are not cached and routed to Tesseract OCR directly. + +Assumes Tesseract 4.0.0-alpha or higher. + +""" + import argparse import json import logging @@ -147,6 +172,7 @@ def cached_run(options, run_args, **run_kwargs): json.dump(manifest, f) f.write('\n') f.flush() + return p class CacheOcrEngine(TesseractOcrEngine): diff --git a/tests/plugins/tesseract_noop.py b/tests/plugins/tesseract_noop.py index 7f78d3e0..26bfe1df 100644 --- a/tests/plugins/tesseract_noop.py +++ b/tests/plugins/tesseract_noop.py @@ -19,7 +19,7 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -"""Tesseract no-op spoof +"""Tesseract no-op plugin To quickly run tests where getting OCR output is not necessary. diff --git a/tests/spoof/tesseract_cache.py b/tests/spoof/tesseract_cache.py deleted file mode 100755 index adf3e257..00000000 --- a/tests/spoof/tesseract_cache.py +++ /dev/null @@ -1,198 +0,0 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -"""Cache output of tesseract to speed up test suite - -The cache is keyed by an environment variable that slips the input test file -from tests/resources/ to us. The input arguments are slugged into a hideous -filename that more or less represents them literally. Joined together, this -becomes the name of the cache folder. A few name files like stdout, stderr, -hocr, pdf, describe the output to reproduce. - -Changes to tests/resources/ or image processing algorithms don't trigger a -cache miss. By design, an input image that varies according to platform -differences (e.g. JPEG decoders are allowed to produce differing outputs, -and in practice they do) will still be a cache hit. By design, an -invocation of tesseract with the same parameters from a different test case -will be a hit. It's fragile. - -The tests/cache/manifest.jsonl is a JSON lines file that contains -information about the system that produced the results used when cache was -generated. This mainly a log to answer questions about how the files -were produced. - -For performance reasons, especially the slow performance of Tesseract on -machines with AVX2, the cache is now bundled. - -Certain operations are not cached and routed to tesseract directly. - -Assumes Tesseract 4.0.0-alpha or higher. - -""" - -import argparse -import json -import os -import platform -import re -import shutil -import subprocess -import sys -from pathlib import Path - -__version__ = subprocess.check_output( - ['tesseract', '--version'], stderr=subprocess.STDOUT -).decode() - - -parser = argparse.ArgumentParser( - prog='tesseract-cache', description='cache output of tesseract' -) -parser.add_argument('-l', '--language', action='append') -parser.add_argument('imagename') -parser.add_argument('outputbase') -parser.add_argument('configfiles', nargs='*') -parser.add_argument('--user-words', type=str) -parser.add_argument('--user-patterns', type=str) -parser.add_argument('-c', action='append') -parser.add_argument('--psm', type=int) -parser.add_argument('--oem', type=int) - -TESTS_ROOT = Path(__file__).resolve().parent.parent -CACHE_ROOT = TESTS_ROOT / 'cache' - - -def real_tesseract(): - tess_args = ['tesseract'] + sys.argv[1:] - os.execvp("tesseract", tess_args) - return # Not reachable - - -def main(): - if any( - opt in sys.argv[1:] - for opt in ('--print-parameters', '--list-langs', '--version') - ): - real_tesseract() # jump into real tesseract, replacing this process - - # Convert non-standard but supported -psm to --psm - sys.argv = ['--psm' if arg == '-psm' else arg for arg in sys.argv] - - if '_OCRMYPDF_TEST_INFILE' not in os.environ: - real_tesseract() # test not properly set up - source = os.environ['_OCRMYPDF_TEST_INFILE'] # required - args = parser.parse_args() - - cache_disabled = os.environ.get('_OCRMYPDF_CACHE_DISABLED', False) - - if args.imagename == 'stdin': - real_tesseract() - - def slugs(): - yield '' # so we don't start with a '-' which makes rm difficult - for arg in sys.argv[1:]: - if arg == args.imagename: - yield Path(args.imagename).name - elif arg == args.outputbase: - yield Path(args.outputbase).name - elif arg == '-c' or arg.startswith('textonly'): - pass - else: - yield arg - - argv_slug = '__'.join(slugs()) - argv_slug = argv_slug.replace('/', '___') - - cache_folder = Path(CACHE_ROOT) / Path(source).stem / argv_slug - cache_folder.mkdir(parents=True, exist_ok=True) - - print(f"Tesseract cache folder {cache_folder} - ", end='', file=sys.stderr) - - if (cache_folder / 'stderr.bin').exists() and not cache_disabled: - # Cache hit - print("HIT", file=sys.stderr) - - # Replicate stdout/err - sys.stdout.buffer.write((cache_folder / 'stdout.bin').read_bytes()) - sys.stderr.buffer.write((cache_folder / 'stderr.bin').read_bytes()) - if args.outputbase != 'stdout': - if not args.configfiles: - args.configfiles.append('txt') - for configfile in args.configfiles: - # cp cache -> output - tessfile = args.outputbase + '.' + configfile - shutil.copy(str(cache_folder / configfile) + '.bin', tessfile) - sys.exit(0) - - # Cache miss - print("MISS", file=sys.stderr) - - # Call tesseract - print(sys.argv[1:]) - p = subprocess.run( - ['tesseract'] + sys.argv[1:], stdout=subprocess.PIPE, stderr=subprocess.PIPE - ) - sys.stdout.buffer.write(p.stdout) - sys.stderr.buffer.write(p.stderr) - - if p.returncode != 0: - # Do not cache errors or crashes - print("Tesseract error", file=sys.stderr) - return p.returncode - - (cache_folder / 'stdout.bin').write_bytes(p.stdout) - - if args.outputbase != 'stdout': - if not args.configfiles: - args.configfiles.append('txt') - - for configfile in args.configfiles: - if configfile not in ('hocr', 'pdf', 'txt'): - continue - # cp pwd/{outputbase}.{configfile} -> {cache}/{configfile} - tessfile = args.outputbase + '.' + configfile - shutil.copy(tessfile, str(cache_folder / configfile) + '.bin') - - (cache_folder / 'stderr.bin').write_bytes(p.stderr) - - manifest = {} - manifest['tesseract_version'] = __version__.replace('\n', ' ') - manifest['platform'] = platform.platform() - manifest['python'] = platform.python_version() - manifest['argv_slug'] = argv_slug - manifest['sourcefile'] = str(Path(source).relative_to(TESTS_ROOT)) - - def clean_sys_argv(): - for arg in sys.argv[1:]: - yield re.sub(r'.*/com.github.ocrmypdf[^/]+[/](.*)', r'$TMPDIR/\1', arg) - - manifest['args'] = list(clean_sys_argv()) - - # pylint: disable=E1101 - with (Path(CACHE_ROOT) / 'manifest.jsonl').open('a') as f: - json.dump(manifest, f) - f.write('\n') - f.flush() - - -if __name__ == '__main__': - main() diff --git a/tests/test_main.py b/tests/test_main.py index ceead7de..6f9d7b50 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -52,7 +52,7 @@ def test_quick(resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) -def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): +def test_oversample(renderer, resources, outpdf): oversampled_pdf = check_ocrmypdf( resources / 'skew.pdf', outpdf, @@ -61,7 +61,8 @@ def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf): '-f', '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfinfo = PdfInfo(oversampled_pdf) @@ -75,17 +76,25 @@ def test_repeat_ocr(resources, no_outpdf): assert result == ExitCode.already_done_ocr -def test_force_ocr(spoof_tesseract_cache, resources, outpdf): +def test_force_ocr(resources, outpdf): out = check_ocrmypdf( - resources / 'graph_ocred.pdf', outpdf, '-f', env=spoof_tesseract_cache + resources / 'graph_ocred.pdf', + outpdf, + '-f', + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfinfo = PdfInfo(out) assert pdfinfo[0].has_text -def test_skip_ocr(spoof_tesseract_cache, resources, outpdf): +def test_skip_ocr(resources, outpdf): out = check_ocrmypdf( - resources / 'graph_ocred.pdf', outpdf, '-s', env=spoof_tesseract_cache + resources / 'graph_ocred.pdf', + outpdf, + '-s', + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfinfo = PdfInfo(out) assert pdfinfo[0].has_text @@ -136,9 +145,14 @@ def test_ocr_timeout(renderer, resources, outpdf): assert not pdfinfo[0].has_text -def test_skip_big(spoof_tesseract_cache, resources, outpdf): +def test_skip_big(resources, outpdf): out = check_ocrmypdf( - resources / 'jbig2.pdf', outpdf, '--skip-big', '1', env=spoof_tesseract_cache + resources / 'jbig2.pdf', + outpdf, + '--skip-big', + '1', + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfinfo = PdfInfo(out) assert not pdfinfo[0].has_text @@ -146,9 +160,7 @@ def test_skip_big(spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) @pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) -def test_maximum_options( - spoof_tesseract_cache, renderer, output_type, resources, outpdf -): +def test_maximum_options(renderer, output_type, resources, outpdf): check_ocrmypdf( resources / 'multipage.pdf', outpdf, @@ -169,7 +181,8 @@ def test_maximum_options( renderer, '--output-type', output_type, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) @@ -208,7 +221,7 @@ def test_force_ocr_on_pdf_with_no_images(resources, no_outpdf): pytest.helpers.is_macos() and pytest.helpers.running_in_travis(), reason="takes too long to install language packs in Travis macOS homebrew", ) -def test_german(spoof_tesseract_cache, resources, outdir): +def test_german(resources, outdir): # Produce a sidecar too - implicit test that system locale is set up # properly. It is fine that we are testing -l deu on a French file because # we are exercising the functionality not going for accuracy. @@ -221,7 +234,8 @@ def test_german(spoof_tesseract_cache, resources, outdir): 'deu', # more commonly installed '--sidecar', sidecar, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) except MissingDependencyError: if 'deu' not in tesseract.get_languages(): @@ -290,7 +304,7 @@ def test_encrypted(resources, caplog, no_outpdf): @pytest.mark.parametrize('renderer', RENDERERS) -def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): +def test_pagesegmode(renderer, resources, outpdf): check_ocrmypdf( resources / 'skew.pdf', outpdf, @@ -300,7 +314,8 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf): '1', '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) @@ -362,7 +377,7 @@ def test_algo4(resources, outpdf): assert p.returncode == ExitCode.encrypted_pdf -def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): +def test_jbig2_passthrough(resources, outpdf): out = check_ocrmypdf( resources / 'jbig2.pdf', outpdf, @@ -370,7 +385,8 @@ def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): 'pdf', '--pdf-renderer', 'hocr', - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) out_pageinfo = PdfInfo(out) assert out_pageinfo[0].images[0].enc == Encoding.jbig2 @@ -391,9 +407,14 @@ def test_linearized_pdf_and_indirect_object(resources, outpdf): ) -def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf): +def test_very_high_dpi(resources, outpdf): "Checks for a Decimal quantize error with high DPI, etc" - check_ocrmypdf(resources / '2400dpi.pdf', outpdf, env=spoof_tesseract_cache) + check_ocrmypdf( + resources / '2400dpi.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_cache.py', + ) pdfinfo = PdfInfo(outpdf) image = pdfinfo[0].images[0] @@ -673,7 +694,7 @@ def test_compression_changed(ocrmypdf_exec, resources, image, compression, outpd im.close() -def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): +def test_sidecar_pagecount(resources, outpdf): sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( resources / '3small.pdf', @@ -681,7 +702,8 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): '--skip-text', '--sidecar', sidecar, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfinfo = PdfInfo(resources / '3small.pdf') @@ -697,10 +719,15 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): ), "Sidecar page count does not match PDF page count" -def test_sidecar_nonempty(spoof_tesseract_cache, resources, outpdf): +def test_sidecar_nonempty(resources, outpdf): sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( - resources / 'ccitt.pdf', outpdf, '--sidecar', sidecar, env=spoof_tesseract_cache + resources / 'ccitt.pdf', + outpdf, + '--sidecar', + sidecar, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) with open(sidecar, 'r', encoding='utf-8') as f: @@ -709,7 +736,7 @@ def test_sidecar_nonempty(spoof_tesseract_cache, resources, outpdf): @pytest.mark.parametrize('pdfa_level', ['1', '2', '3']) -def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf): +def test_pdfa_n(pdfa_level, resources, outpdf): if pdfa_level == '3' and ghostscript.version() < '9.19': pytest.xfail(reason='Ghostscript >= 9.19 required') @@ -718,7 +745,8 @@ def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf): outpdf, '--output-type', 'pdfa-' + pdfa_level, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) pdfa_info = file_claims_pdfa(outpdf) diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index 1fb494c4..733153fc 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -58,7 +58,7 @@ def test_list_range(): assert _pages_from_ranges([0, 1, 2]) == {0, 1, 2} -def test_limited_pages(resources, outpdf, spoof_tesseract_cache): +def test_limited_pages(resources, outpdf): multi = resources / 'multipage.pdf' ocrmypdf.ocr( multi, @@ -66,7 +66,7 @@ def test_limited_pages(resources, outpdf, spoof_tesseract_cache): pages='5-6', optimize=0, output_type='pdf', - tesseract_env=spoof_tesseract_cache, + plugins=['tests/plugins/tesseract_cache.py'], ) pi = PdfInfo(outpdf) assert not pi.pages[0].has_text diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 08e34bec..e24c3b7d 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -102,9 +102,7 @@ def test_remove_background(resources, outdir): ) @pytest.mark.parametrize("renderer", ['sandwich', 'hocr']) @pytest.mark.parametrize("output_type", ['pdf', 'pdfa']) -def test_exotic_image( - spoof_tesseract_cache, pdf, renderer, output_type, resources, outdir -): +def test_exotic_image(pdf, renderer, output_type, resources, outdir): outfile = outdir / f'test_{pdf}_{renderer}.pdf' check_ocrmypdf( resources / pdf, @@ -118,14 +116,15 @@ def test_exotic_image( '--skip-text', '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) assert outfile.with_suffix('.pdf.txt').exists() @pytest.mark.parametrize('renderer', RENDERERS) -def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf): +def test_non_square_resolution(renderer, resources, outpdf): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y @@ -135,7 +134,8 @@ def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpd outpdf, '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) out_pageinfo = PdfInfo(outpdf) @@ -145,9 +145,7 @@ def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpd @pytest.mark.parametrize('renderer', RENDERERS) -def test_convert_to_square_resolution( - renderer, spoof_tesseract_cache, resources, outpdf -): +def test_convert_to_square_resolution(renderer, resources, outpdf): # Confirm input image is non-square resolution in_pageinfo = PdfInfo(resources / 'aspect.pdf') assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y @@ -159,7 +157,8 @@ def test_convert_to_square_resolution( '--force-ocr', '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) out_pageinfo = PdfInfo(outpdf) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 4ffa59f8..0c8b5dd3 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -97,7 +97,7 @@ def test_monochrome_correlation(resources, outdir): @pytest.mark.slow @pytest.mark.parametrize('renderer', RENDERERS) -def test_autorotate(spoof_tesseract_cache, renderer, resources, outdir): +def test_autorotate(renderer, resources, outdir): # cardinal.pdf contains four copies of an image rotated in each cardinal # direction - these ones are "burned in" not tagged with /Rotate out = check_ocrmypdf( @@ -108,7 +108,8 @@ def test_autorotate(spoof_tesseract_cache, renderer, resources, outdir): '1', '--pdf-renderer', renderer, - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) for n in range(1, 4 + 1): correlation = check_monochrome_correlation( @@ -128,9 +129,7 @@ def test_autorotate(spoof_tesseract_cache, renderer, resources, outdir): ('99', 'correlation < 0.10'), # High thres -> never rotate -> low corr ], ) -def test_autorotate_threshold( - spoof_tesseract_cache, threshold, correlation_test, resources, outdir -): +def test_autorotate_threshold(threshold, correlation_test, resources, outdir): out = check_ocrmypdf( resources / 'cardinal.pdf', outdir / 'out.pdf', @@ -139,7 +138,8 @@ def test_autorotate_threshold( '-r', # '-v', # '1', - env=spoof_tesseract_cache, + '--plugin', + 'tests/plugins/tesseract_cache.py', ) correlation = check_monochrome_correlation( diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index ffe3fe31..99222d5a 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -31,7 +31,6 @@ from ocrmypdf.exec import tesseract check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf -spoof = pytest.helpers.spoof @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 3b9f155e..9a6dc248 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -31,7 +31,6 @@ from ocrmypdf.exceptions import ExitCode, MissingDependencyError check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf -spoof = pytest.helpers.spoof have_unpaper = pytest.helpers.have_unpaper diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 41d21aa2..60f97d08 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -25,7 +25,6 @@ from ocrmypdf.pdfinfo import PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api -spoof = pytest.helpers.spoof @pytest.fixture @@ -39,15 +38,26 @@ def test_userunit_ghostscript_fails(poster, no_outpdf, caplog): assert 'not supported by Ghostscript' in caplog.text -def test_userunit_pdf_passes(spoof_tesseract_cache, poster, outpdf): +def test_userunit_pdf_passes(poster, outpdf): before = PdfInfo(poster) - check_ocrmypdf(poster, outpdf, '--output-type=pdf', env=spoof_tesseract_cache) + check_ocrmypdf( + poster, + outpdf, + '--output-type=pdf', + '--plugin', + 'tests/plugins/tesseract_cache.py', + ) after = PdfInfo(outpdf) assert isclose(before[0].width_inches, after[0].width_inches) -def test_rotate_interaction(spoof_tesseract_cache, poster, outpdf): +def test_rotate_interaction(poster, outpdf): check_ocrmypdf( - poster, outpdf, '--output-type=pdf', '--rotate-pages', env=spoof_tesseract_cache + poster, + outpdf, + '--output-type=pdf', + '--rotate-pages', + '--plugin', + 'tests/plugins/tesseract_cache.py', ) From c6c70c21712af36d065467abed1c7dc0278568cc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jun 2020 07:42:13 -0700 Subject: [PATCH 503/880] docs: Ubuntu 20.04 install instructions --- docs/installation.rst | 48 ++++++++++++++++++++++++++++++++++--------- 1 file changed, 38 insertions(+), 10 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 375ebc0b..570045e4 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -45,14 +45,8 @@ Debian and Ubuntu 18.04 or newer .. |ubu-1804| image:: https://repology.org/badge/version-for-repo/ubuntu_18_04/ocrmypdf.svg :alt: Ubuntu 18.04 LTS -.. |ubu-1810| image:: https://repology.org/badge/version-for-repo/ubuntu_18_10/ocrmypdf.svg - :alt: Ubuntu 18.10 - -.. |ubu-1904| image:: https://repology.org/badge/version-for-repo/ubuntu_19_04/ocrmypdf.svg - :alt: Ubuntu 19.04 - -.. |ubu-1910| image:: https://repology.org/badge/version-for-repo/ubuntu_19_10/ocrmypdf.svg - :alt: Ubuntu 19.10 +.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg + :alt: Ubuntu 20.04 LTS +-----------------------------------------------+ | **OCRmyPDF versions in Debian & Ubuntu** | @@ -61,7 +55,7 @@ Debian and Ubuntu 18.04 or newer +-----------------------------------------------+ | |deb-stable| |deb-testing| |deb-unstable| | +-----------------------------------------------+ -| |ubu-1804| |ubu-1810| |ubu-1904| |ubu-1910| | +| |ubu-1804| |ubu-2004| | +-----------------------------------------------+ Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users @@ -133,7 +127,41 @@ from sources <#installing-head-revision-from-sources>`__. .. _ubuntu-lts-latest: -Installing the latest version on Ubuntu 18.04 LTS +Installing the latest version on Ubuntu 20.04 LTS +------------------------------------------------- + +Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To +install a more recent version, uninstall the system-provided version of +ocrmypdf, and install the following dependencies: + +.. code-block:: bash + + sudo apt-get -y remove ocrmypdf # remove system ocrmypdf, if installed + sudo apt-get -y update + sudo apt-get -y install \ + ghostscript \ + icc-profiles-free \ + liblept5 \ + libxml2 \ + pngquant \ + python3-pip \ + tesseract-ocr \ + zlib1g + +To install ocrmypdf for the system: + +.. code-block:: bash + + sudo pip3 install ocrmypdf + +To install for the current user only: + +.. code-block:: bash + + export PATH=$HOME/.local/bin:$PATH + pip3 install --user ocrmypdf + +Ubuntu 18.04 LTS ------------------------------------------------- Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but From fd1cd8e50afaaada707f15d10af00f3be8f2c609 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jun 2020 07:46:55 -0700 Subject: [PATCH 504/880] docs: explain --rotate-pages-threshold --- docs/cookbook.rst | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 3daab1ca..59e7b267 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -56,7 +56,11 @@ portrait pages. ocrmypdf --rotate-pages myfile.pdf myfile.pdf You can increase (decrease) the parameter ``--rotate-pages-threshold`` -to make page rotation more (less) aggressive. +to make page rotation more (less) aggressive. The threshold number is the ratio +of how confidence the OCR engine is that the document image should be changed, +compared to kept the same. A value of ``15.0`` is the default, and is fairly +conservative. A value of ``2.0`` will produce more rotations, and more false +positives. If the page is "just a little off horizontal", like a crooked picture, then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal From b109445215155fdd5d8707f96266885ca4706bbb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jun 2020 17:10:27 -0700 Subject: [PATCH 505/880] Move Ghostscript rasterize_pdf to plugin --- src/ocrmypdf/_pipeline.py | 12 +-- src/ocrmypdf/_plugin_manager.py | 5 +- src/ocrmypdf/_validation.py | 50 +----------- src/ocrmypdf/builtin_plugins/__init__.py | 2 - src/ocrmypdf/builtin_plugins/ghostscript.py | 90 +++++++++++++++++++++ src/ocrmypdf/exec/_support.py | 5 +- src/ocrmypdf/exec/ghostscript.py | 18 +---- src/ocrmypdf/pluginspec.py | 31 +++++++ tests/test_main.py | 1 - tests/test_metadata.py | 1 - tests/test_validation.py | 25 ++++-- 11 files changed, 154 insertions(+), 86 deletions(-) create mode 100644 src/ocrmypdf/builtin_plugins/ghostscript.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 0806880e..6e67d808 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -321,9 +321,9 @@ def rasterize_preview(input_file, page_context): output_file = page_context.get_path('rasterize_preview.jpg') canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - ghostscript.rasterize_pdf( - input_file, - output_file, + page_context.plugin_manager.hook.rasterize_pdf_page( + input_file=input_file, + output_file=output_file, raster_device='jpeggray', raster_dpi=canvas_dpi, page_dpi=page_dpi, @@ -430,9 +430,9 @@ def rasterize( canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options) page_dpi = get_page_square_dpi(pageinfo, page_context.options) - ghostscript.rasterize_pdf( - input_file, - output_file, + page_context.plugin_manager.hook.rasterize_pdf_page( + input_file=input_file, + output_file=output_file, raster_device=device, raster_dpi=canvas_dpi, page_dpi=page_dpi, diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 1fbc1b76..9f328c69 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -33,7 +33,10 @@ def get_plugin_manager(plugins: List[str], builtins=True): pm.add_hookspecs(pluginspec) if builtins: - all_plugins = ['ocrmypdf.builtin_plugins'] + plugins + all_plugins = [ + 'ocrmypdf.builtin_plugins.ghostscript', + 'ocrmypdf.builtin_plugins.tesseract_ocr', + ] + plugins else: all_plugins = plugins for name in all_plugins: diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 941e3522..3a92d34c 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -27,7 +27,6 @@ from shutil import copyfileobj import PIL -from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._unicodefun import verify_python3_env from ocrmypdf.exceptions import ( BadArgsError, @@ -35,14 +34,7 @@ from ocrmypdf.exceptions import ( MissingDependencyError, OutputFileAccessError, ) -from ocrmypdf.exec import ( - check_external_program, - ghostscript, - jbig2enc, - pngquant, - tesseract, - unpaper, -) +from ocrmypdf.exec import check_external_program, jbig2enc, pngquant, unpaper from ocrmypdf.helpers import ( is_file_writable, is_iterable_notstr, @@ -92,10 +84,6 @@ def check_options_languages(options, plugin_manager): def check_options_output(options): - # We have these constraints to check for. - # 1. Ghostscript < 9.20 mangles multibyte Unicode - # 2. hocr doesn't work on non-Latin languages (so don't select it) - is_latin = options.languages.issubset(HOCR_OK_LANGS) if options.pdf_renderer == 'hocr' and not is_latin: @@ -106,25 +94,6 @@ def check_options_output(options): ) log.warning(msg) - if ghostscript.version() < '9.20' and options.output_type != 'pdf' and not is_latin: - # https://bugs.ghostscript.com/show_bug.cgi?id=696874 - # Ghostscript < 9.20 fails to encode multibyte characters properly - msg = ( - "The installed version of Ghostscript does not work correctly " - "with the OCR languages you specified. Use --output-type pdf or " - "upgrade to Ghostscript 9.20 or later to avoid this issue." - ) - msg += f"Found Ghostscript {ghostscript.version()}" - log.warning(msg) - - if options.output_type == 'pdfa': - options.output_type = 'pdfa-2' - - if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19': - raise MissingDependencyError( - "--output-type pdfa-3 requires Ghostscript 9.19 or later" - ) - lossless_reconstruction = False if not any( ( @@ -291,7 +260,6 @@ def check_options(options, plugin_manager): check_options_optimizing(options) check_options_advanced(options) check_options_pillow(options) - check_dependency_versions(options) plugin_manager.hook.check_options(options=options) @@ -432,19 +400,3 @@ def report_output_file_size(options, input_file, output_file): f"The output file size is {ratio:.2f}× larger than the input file.\n" f"{explanation}" ) - - -def check_dependency_versions(options): - check_external_program( - program='gs', - package='ghostscript', - version_checker=ghostscript.version, - need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports - ) - gs_version = ghostscript.version() - if gs_version in ('9.24', '9.51'): - raise MissingDependencyError( - f"Ghostscript {gs_version} contains serious regressions and is not " - "supported. Please upgrade to a newer version, or downgrade to the " - "previous version." - ) diff --git a/src/ocrmypdf/builtin_plugins/__init__.py b/src/ocrmypdf/builtin_plugins/__init__.py index e5fd494e..0ed32bc2 100644 --- a/src/ocrmypdf/builtin_plugins/__init__.py +++ b/src/ocrmypdf/builtin_plugins/__init__.py @@ -14,5 +14,3 @@ # # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . - -from ocrmypdf.builtin_plugins.tesseract_ocr import * diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py new file mode 100644 index 00000000..92f39008 --- /dev/null +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -0,0 +1,90 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +from pathlib import Path + +from ocrmypdf import hookimpl +from ocrmypdf._validation import HOCR_OK_LANGS +from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.exec import check_external_program, ghostscript +from ocrmypdf.helpers import Resolution + +log = logging.getLogger(__name__) + + +@hookimpl +def check_options(options): + gs_version = ghostscript.version() + check_external_program( + program='gs', + package='ghostscript', + version_checker=gs_version, + need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports + ) + if gs_version in ('9.24', '9.51'): + raise MissingDependencyError( + f"Ghostscript {gs_version} contains serious regressions and is not " + "supported. Please upgrade to a newer version, or downgrade to the " + "previous version." + ) + + # We have these constraints to check for. + # 1. Ghostscript < 9.20 mangles multibyte Unicode + # 2. hocr doesn't work on non-Latin languages (so don't select it) + is_latin = options.languages.issubset(HOCR_OK_LANGS) + if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin: + # https://bugs.ghostscript.com/show_bug.cgi?id=696874 + # Ghostscript < 9.20 fails to encode multibyte characters properly + msg = ( + "The installed version of Ghostscript does not work correctly " + "with the OCR languages you specified. Use --output-type pdf or " + "upgrade to Ghostscript 9.20 or later to avoid this issue." + ) + msg += f"Found Ghostscript {gs_version}" + log.warning(msg) + + if options.output_type == 'pdfa': + options.output_type = 'pdfa-2' + + if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19': + raise MissingDependencyError( + "--output-type pdfa-3 requires Ghostscript 9.19 or later" + ) + + +@hookimpl +def rasterize_pdf_page( + input_file: Path, + output_file: Path, + raster_device: str, + raster_dpi: Resolution, + pageno: int, + page_dpi: Resolution = None, + rotation: int = None, + filter_vector: bool = False, +): + return ghostscript.rasterize_pdf( + input_file, + output_file, + raster_device=raster_device, + raster_dpi=raster_dpi, + pageno=pageno, + page_dpi=page_dpi, + rotation=rotation, + filter_vector=filter_vector, + ) diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/exec/_support.py index 60d22a83..5f52ac87 100644 --- a/src/ocrmypdf/exec/_support.py +++ b/src/ocrmypdf/exec/_support.py @@ -280,7 +280,10 @@ def check_external_program( recommended=False, ): try: - found_version = version_checker() + if callable(version_checker): + found_version = version_checker() + else: + found_version = version_checker except (CalledProcessError, FileNotFoundError, MissingDependencyError): _error_missing_program(program, package, required_for, recommended) if not recommended: diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index d7407726..a43b972b 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -20,7 +20,6 @@ import logging import os import re -import warnings from io import BytesIO from os import fspath from pathlib import Path @@ -92,22 +91,7 @@ def rasterize_pdf( rotation: int = None, filter_vector: bool = False, ): - """Rasterize one page of a PDF at resolution raster_dpi in canvas units. - - The image is sized to match the integer pixels dimensions implied by - raster_dpi even if those numbers are noninteger. The image's DPI will - be overridden with the values in page_dpi. - - :param input_file: pathlike - :param output_file: pathlike - :param raster_device: - :param raster_dpi: resolution at which to rasterize page - :param pageno: page number to rasterize (beginning at page 1) - :param page_dpi: resolution tuple (x, y) overriding output image DPI - :param rotation: 0, 90, 180, 270: clockwise angle to rotate page - :param filter_vector: if True, remove vector graphics objects - :return: - """ + """Rasterize one page of a PDF at resolution raster_dpi in canvas units.""" raster_dpi = raster_dpi.round(6) if not page_dpi: page_dpi = raster_dpi diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index dd659458..fd2498b2 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -24,6 +24,8 @@ from typing import AbstractSet, Optional import pluggy from PIL import Image +from ocrmypdf.helpers import Resolution + hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument @@ -66,6 +68,35 @@ def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: """ +@hookspec(firstresult=True) +def rasterize_pdf_page( + input_file: Path, + output_file: Path, + raster_device: str, + raster_dpi: Resolution, + pageno: int, + page_dpi: Resolution = None, + rotation: int = None, + filter_vector: bool = False, +) -> None: + """Rasterize one page of a PDF at resolution raster_dpi in canvas units. + + The image is sized to match the integer pixels dimensions implied by + raster_dpi even if those numbers are noninteger. The image's DPI will + be overridden with the values in page_dpi. + + Args: + raster_device: type of image to produce at output_file + raster_dpi: resolution at which to rasterize page + pageno: page number to rasterize (beginning at page 1) + page_dpi: resolution, overriding output image DPI + rotation: cardinal angle, clockwise, to rotate page + filter_vector: if True, remove vector graphics objects + Returns: + None + """ + + @hookspec(firstresult=True) def filter_ocr_image(page: 'PageContext', image: Image) -> Image: """Called to filter the image before it is sent to OCR. diff --git a/tests/test_main.py b/tests/test_main.py index 6f9d7b50..0b913a14 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -39,7 +39,6 @@ from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api -spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 59250eac..16175140 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -47,7 +47,6 @@ pytestmark = pytest.mark.filterwarnings('ignore:.*XMLParser.*:DeprecationWarning check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf -spoof = pytest.helpers.spoof @pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) diff --git a/tests/test_validation.py b/tests/test_validation.py index f6fcc775..3e6272af 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -29,19 +29,29 @@ from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.pdfinfo import PdfInfo -def make_opts(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): +def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): if language is not None: kwargs['language'] = language parser = get_parser() pm = get_plugin_manager(kwargs.get('plugins', [])) pm.hook.add_options(parser=parser) - return create_options( - input_file=input_file, output_file=output_file, parser=parser, **kwargs + return ( + create_options( + input_file=input_file, output_file=output_file, parser=parser, **kwargs + ), + pm, ) +def make_opts(*args, **kwargs): + opts, _pm = make_opts_pm(*args, **kwargs) + return opts + + def test_hocr_notlatin_warning(caplog): - vd.check_options_output(make_opts(language='chi_sim', pdf_renderer='hocr')) + vd.check_options( + *make_opts_pm(language='chi_sim', pdf_renderer='hocr', output_type='pdfa') + ) assert 'PDF renderer is known to cause' in caplog.text @@ -49,20 +59,20 @@ def test_old_ghostscript(caplog): with patch('ocrmypdf.exec.ghostscript.version', return_value='9.19'), patch( 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True ): - vd.check_options_output(make_opts(language='chi_sim', output_type='pdfa')) + vd.check_options(*make_opts_pm(language='chi_sim', output_type='pdfa')) assert 'Ghostscript does not work correctly' in caplog.text with patch('ocrmypdf.exec.ghostscript.version', return_value='9.18'), patch( 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): - vd.check_options_output(make_opts(output_type='pdfa-3')) + vd.check_options(*make_opts_pm(output_type='pdfa-3')) with patch('ocrmypdf.exec.ghostscript.version', return_value='9.24'), patch( 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): - vd.check_dependency_versions(make_opts()) + vd.check_options(*make_opts_pm()) def test_old_tesseract_error(): @@ -97,7 +107,6 @@ def test_optimizing(caplog): def test_user_words(caplog): - with patch('ocrmypdf.exec.tesseract.has_user_words', return_value=False): opts = make_opts(user_words='foo') plugin_manager = get_plugin_manager(opts.plugins) From 7b9025f3977bf7dc423ff4bc1eb374f06eb6bb6b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jun 2020 22:28:38 -0700 Subject: [PATCH 506/880] Convert generate_pdfa to plugin --- src/ocrmypdf/_pipeline.py | 5 +-- src/ocrmypdf/builtin_plugins/ghostscript.py | 27 +++++++++++----- src/ocrmypdf/exec/ghostscript.py | 18 ----------- src/ocrmypdf/pluginspec.py | 34 +++++++++++++++++++-- tests/test_metadata.py | 8 +++-- tests/test_preprocessing.py | 1 - tests/test_stdio.py | 1 - 7 files changed, 59 insertions(+), 35 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6e67d808..094b0858 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -688,9 +688,10 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): else: safe_symlink(input_pdf, fix_docinfo_file) - ghostscript.generate_pdfa( + context.plugin_manager.hook.generate_pdfa( pdf_version=input_pdfinfo.min_version, - pdf_pages=[fix_docinfo_file, input_ps_stub], + pdf_pages=[fix_docinfo_file], + pdfmark=input_ps_stub, output_file=output_file, compression=options.pdfa_image_compression, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 92f39008..90cf41a3 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -69,14 +69,14 @@ def check_options(options): @hookimpl def rasterize_pdf_page( - input_file: Path, - output_file: Path, - raster_device: str, - raster_dpi: Resolution, - pageno: int, - page_dpi: Resolution = None, - rotation: int = None, - filter_vector: bool = False, + input_file, + output_file, + raster_device, + raster_dpi, + pageno, + page_dpi=None, + rotation=None, + filter_vector=False, ): return ghostscript.rasterize_pdf( input_file, @@ -88,3 +88,14 @@ def rasterize_pdf_page( rotation=rotation, filter_vector=filter_vector, ) + + +@hookimpl +def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): + return ghostscript.generate_pdfa( + pdf_pages=[*pdf_pages, pdfmark], + output_file=output_file, + compression=compression, + pdf_version=pdf_version, + pdfa_part=pdfa_part, + ) diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index a43b972b..9fe690b9 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -154,24 +154,6 @@ def generate_pdfa( pdf_version: str = '1.5', pdfa_part: str = '2', ): - """Generate a PDF/A. - - The pdf_pages, a list files, will be merged into output_file. One or more - PDF files may be merged. One of the files in this list must be a pdfmark - file that provides Ghostscript with details on how to perform the PDF/A - conversion. By default with we pick PDF/A-2b, but this works for 1 or 3. - - compression can be 'jpeg', 'lossless', or an empty string. In 'jpeg', - Ghostscript is instructed to convert color and grayscale images to DCT - (JPEG encoding). In 'lossless' Ghostscript is told to convert images to - Flate (lossless/PNG). If the parameter is omitted Ghostscript is left to - make its own decisions about how to encode images; it appears to use a - heuristic to decide how to encode images. As of Ghostscript 9.25, we - support passthrough JPEG which allows Ghostscript to avoid transcoding - images entirely. (The feature was added in 9.23 but broken, and the 9.24 - release of Ghostscript had regressions, so we don't support it until 9.25.) - """ - compression_args = [] if compression == 'jpeg': compression_args = [ diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index fd2498b2..b83aaaa2 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -19,7 +19,7 @@ from abc import ABC, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple from pathlib import Path -from typing import AbstractSet, Optional +from typing import AbstractSet, List, Optional import pluggy from PIL import Image @@ -75,8 +75,8 @@ def rasterize_pdf_page( raster_device: str, raster_dpi: Resolution, pageno: int, - page_dpi: Resolution = None, - rotation: int = None, + page_dpi: Optional[Resolution] = None, + rotation: Optional[int] = None, filter_vector: bool = False, ) -> None: """Rasterize one page of a PDF at resolution raster_dpi in canvas units. @@ -162,3 +162,31 @@ class OcrEngine(ABC): @hookspec(firstresult=True) def get_ocr_engine() -> OcrEngine: pass + + +@hookspec(firstresult=True) +def generate_pdfa( + pdf_pages: List[Path], + pdfmark: Path, + output_file: Path, + compression: str, + pdf_version: str, + pdfa_part: str, +): + """Generate a PDF/A. + + The pdf_pages, a list of files, will be merged into output_file. One or more + PDF files may be merged. The pdfmark file is a PostScript.ps file that + provides Ghostscript with details on how to perform the PDF/A + conversion. By default with we pick PDF/A-2b, but this works for 1 or 3. + + compression can be 'jpeg', 'lossless', or an empty string. In 'jpeg', + Ghostscript is instructed to convert color and grayscale images to DCT + (JPEG encoding). In 'lossless' Ghostscript is told to convert images to + Flate (lossless/PNG). If the parameter is omitted Ghostscript is left to + make its own decisions about how to encode images; it appears to use a + heuristic to decide how to encode images. As of Ghostscript 9.25, we + support passthrough JPEG which allows Ghostscript to avoid transcoding + images entirely. (The feature was added in 9.23 but broken, and the 9.24 + release of Ghostscript had regressions, so we don't support it until 9.25.) + """ diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 16175140..1d310107 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -341,7 +341,9 @@ def test_prevent_gs_invalid_xml(resources, outdir): args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PdfContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) + context = PdfContext( + options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, get_plugin_manager([]) + ) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context @@ -372,7 +374,9 @@ def test_malformed_docinfo(caplog, resources, outdir): args=['-j', '1', '--output-type', 'pdfa-2', 'a.pdf', 'b.pdf'] ) pdfinfo = PdfInfo(outdir / 'layers.rendered.pdf') - context = PdfContext(options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, None) + context = PdfContext( + options, outdir, outdir / 'layers.rendered.pdf', pdfinfo, get_plugin_manager([]) + ) convert_to_pdfa( str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index e24c3b7d..865b5e58 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -31,7 +31,6 @@ from ocrmypdf.pdfinfo import PdfInfo check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api -spoof = pytest.helpers.spoof RENDERERS = ['hocr', 'sandwich'] diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 2f58306a..883d48de 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -30,7 +30,6 @@ from ocrmypdf.helpers import check_pdf run_ocrmypdf = pytest.helpers.run_ocrmypdf run_ocrmypdf_api = pytest.helpers.run_ocrmypdf -spoof = pytest.helpers.spoof def test_stdin(ocrmypdf_exec, resources, outpdf): From c22f2456064d70cbae6e25c9ac799fd6e76c47a8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 8 Jun 2020 23:48:45 -0700 Subject: [PATCH 507/880] Plugins must return not-None if they intend to stop builtin --- src/ocrmypdf/builtin_plugins/ghostscript.py | 6 ++++-- src/ocrmypdf/pluginspec.py | 9 ++++++--- 2 files changed, 10 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 90cf41a3..c74bb360 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -78,7 +78,7 @@ def rasterize_pdf_page( rotation=None, filter_vector=False, ): - return ghostscript.rasterize_pdf( + ghostscript.rasterize_pdf( input_file, output_file, raster_device=raster_device, @@ -88,14 +88,16 @@ def rasterize_pdf_page( rotation=rotation, filter_vector=filter_vector, ) + return output_file @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - return ghostscript.generate_pdfa( + ghostscript.generate_pdfa( pdf_pages=[*pdf_pages, pdfmark], output_file=output_file, compression=compression, pdf_version=pdf_version, pdfa_part=pdfa_part, ) + return output_file diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index b83aaaa2..e3961fbf 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -78,7 +78,7 @@ def rasterize_pdf_page( page_dpi: Optional[Resolution] = None, rotation: Optional[int] = None, filter_vector: bool = False, -) -> None: +) -> Path: """Rasterize one page of a PDF at resolution raster_dpi in canvas units. The image is sized to match the integer pixels dimensions implied by @@ -93,7 +93,7 @@ def rasterize_pdf_page( rotation: cardinal angle, clockwise, to rotate page filter_vector: if True, remove vector graphics objects Returns: - None + output_file """ @@ -172,7 +172,7 @@ def generate_pdfa( compression: str, pdf_version: str, pdfa_part: str, -): +) -> Path: """Generate a PDF/A. The pdf_pages, a list of files, will be merged into output_file. One or more @@ -189,4 +189,7 @@ def generate_pdfa( support passthrough JPEG which allows Ghostscript to avoid transcoding images entirely. (The feature was added in 9.23 but broken, and the 9.24 release of Ghostscript had regressions, so we don't support it until 9.25.) + + Returns: + output_file """ From 2059e916da89bbfc2fc45b41707a40ac08c166d7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 00:00:25 -0700 Subject: [PATCH 508/880] Convert all ghostscript spoofs to test plugins --- setup.cfg | 2 +- .../{spoof => plugins}/gs_feature_elision.py | 46 +++++++-------- tests/{spoof => plugins}/gs_pdfa_failure.py | 55 ++++++++---------- tests/{spoof => plugins}/gs_raster_failure.py | 58 +++++++++++-------- tests/{spoof => plugins}/gs_render_failure.py | 46 +++++++-------- tests/test_ghostscript.py | 40 ++++--------- 6 files changed, 114 insertions(+), 133 deletions(-) rename tests/{spoof => plugins}/gs_feature_elision.py (59%) mode change 100755 => 100644 rename tests/{spoof => plugins}/gs_pdfa_failure.py (59%) mode change 100755 => 100644 rename tests/{spoof => plugins}/gs_raster_failure.py (51%) mode change 100755 => 100644 rename tests/{spoof => plugins}/gs_render_failure.py (55%) mode change 100755 => 100644 diff --git a/setup.cfg b/setup.cfg index 3cb3db9d..487ed30d 100644 --- a/setup.cfg +++ b/setup.cfg @@ -23,7 +23,7 @@ force_grid_wrap=0 use_parentheses=True line_length=88 known_first_party = ocrmypdf -known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug +known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug [metadata] license_file = LICENSE diff --git a/tests/spoof/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py old mode 100755 new mode 100644 similarity index 59% rename from tests/spoof/gs_feature_elision.py rename to tests/plugins/gs_feature_elision.py index a06deaf3..84f5e6e1 --- a/tests/spoof/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # Permission is hereby granted, free of charge, to any person obtaining a # copy of this software and associated documentation files (the @@ -20,34 +19,31 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +from unittest.mock import patch -import os -import sys -from subprocess import check_call - -from gs import real_ghostscript - -"""Replicate one type of Ghostscript feature elision warning during -PDF/A creation.""" - +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins import ghostscript +from ocrmypdf.exec import run elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 not permitted in PDF/A-2, overprint mode not set""" -def main(): - if '--version' in sys.argv: - print('9.20') - print('SPOOFED: ' + os.path.basename(__file__)) - sys.exit(0) - gs_args = ['gs'] + sys.argv[1:] - check_call(gs_args) - - if '-sDEVICE=pdfwrite' in sys.argv[1:]: - print(elision_warning) - - sys.exit(0) +def run_append_stderr(*args, **kwargs): + proc = run(*args, **kwargs) + proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')]) + return proc -if __name__ == '__main__': - main() +@hookimpl +def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): + with patch('ocrmypdf.exec.ghostscript.run', new=run_append_stderr): + ghostscript.generate_pdfa( + pdf_pages=pdf_pages, + pdfmark=pdfmark, + output_file=output_file, + compression=compression, + pdf_version=pdf_version, + pdfa_part=pdfa_part, + ) + return output_file diff --git a/tests/spoof/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py old mode 100755 new mode 100644 similarity index 59% rename from tests/spoof/gs_pdfa_failure.py rename to tests/plugins/gs_pdfa_failure.py index 1d9fdf7d..f9093224 --- a/tests/spoof/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # Permission is hereby granted, free of charge, to any person obtaining a # copy of this software and associated documentation files (the @@ -20,41 +19,33 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -import os -import sys +from unittest.mock import patch -from gs import real_ghostscript +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins import ghostscript +from ocrmypdf.exec import run -"""Replicate Ghostscript PDF/A conversion failure by suppressing some -arguments""" - - -def main(): - if '--version' in sys.argv: - print('9.20') - print('SPOOFED: ' + os.path.basename(__file__)) - sys.exit(0) - - # Unless some argument is calling for PDFA generation, forward to - # real ghostscript - if not any(arg.startswith('-dPDFA') for arg in sys.argv): - real_ghostscript(sys.argv) - return - +def run_rig_args(args, **kwargs): # Remove the two arguments that tell ghostscript to create a PDF/A # Does not remove the Postscript definition file - not necessary # to cause PDF/A creation failure - argv = [] - for arg in sys.argv: - if arg.startswith('-dPDFA'): - continue - elif arg.startswith('-dPDFACompatibilityPolicy'): - continue - argv.append(arg) - - real_ghostscript(argv) + new_args = [ + arg for arg in args if not arg.startswith('-dPDFA') and not arg.endswith('.ps') + ] + proc = run(new_args, **kwargs) + return proc -if __name__ == '__main__': - main() +@hookimpl +def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): + with patch('ocrmypdf.exec.ghostscript.run', new=run_rig_args): + ghostscript.generate_pdfa( + pdf_pages=pdf_pages, + pdfmark=pdfmark, + output_file=output_file, + compression=compression, + pdf_version=pdf_version, + pdfa_part=pdfa_part, + ) + return output_file diff --git a/tests/spoof/gs_raster_failure.py b/tests/plugins/gs_raster_failure.py old mode 100755 new mode 100644 similarity index 51% rename from tests/spoof/gs_raster_failure.py rename to tests/plugins/gs_raster_failure.py index 7619aae2..cab3268e --- a/tests/spoof/gs_raster_failure.py +++ b/tests/plugins/gs_raster_failure.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -# © 2016 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # Permission is hereby granted, free of charge, to any person obtaining a # copy of this software and associated documentation files (the @@ -20,30 +19,41 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +from pathlib import Path +from subprocess import CalledProcessError +from unittest.mock import patch -import os -import sys - -from gs import real_ghostscript - -"""Replicate Ghostscript raster failure while allowing rendering""" +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins import ghostscript +from ocrmypdf.exec import run -def main(): - if '--version' in sys.argv: - print('9.20') - print('SPOOFED: ' + os.path.basename(__file__)) - sys.exit(0) - - # For non-image rastering calls, use real ghostscript - if '-sDEVICE=pdfwrite' in sys.argv or '-sDEVICE=txtwrite' in sys.argv: - real_ghostscript(sys.argv) - return - - # Fail - print("ERROR: Ghost story archive not found", file=sys.stderr) - sys.exit(1) +def raise_gs_fail(*args, **kwargs): + raise CalledProcessError( + 1, 'gs', output=b"", stderr=b"ERROR: Ghost story archive not found" + ) -if __name__ == '__main__': - main() +@hookimpl +def rasterize_pdf_page( + input_file, + output_file, + raster_device, + raster_dpi, + pageno, + page_dpi=None, + rotation=None, + filter_vector=False, +) -> Path: + with patch('ocrmypdf.exec.ghostscript.run', new=raise_gs_fail): + ghostscript.rasterize_pdf_page( + input_file=input_file, + output_file=output_file, + raster_device=raster_device, + raster_dpi=raster_dpi, + pageno=pageno, + page_dpi=page_dpi, + rotation=rotation, + filter_vector=filter_vector, + ) + return output_file diff --git a/tests/spoof/gs_render_failure.py b/tests/plugins/gs_render_failure.py old mode 100755 new mode 100644 similarity index 55% rename from tests/spoof/gs_render_failure.py rename to tests/plugins/gs_render_failure.py index d0c1d60d..e1d934ce --- a/tests/spoof/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -# © 2016-18 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # Permission is hereby granted, free of charge, to any person obtaining a # copy of this software and associated documentation files (the @@ -20,29 +19,30 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -"""Replicate Ghostscript render failure while allowing rasterizing""" +from pathlib import Path +from subprocess import CalledProcessError +from unittest.mock import patch -import os -import sys - -from gs import real_ghostscript +from ocrmypdf import hookimpl +from ocrmypdf.builtin_plugins import ghostscript +from ocrmypdf.exec import run -def main(): - if '--version' in sys.argv: - print('9.20') - print('SPOOFED: ' + os.path.basename(__file__)) - sys.exit(0) - - # For any rasterize calls (device != pdfwrite) call real ghostscript - if '-sDEVICE=pdfwrite' not in sys.argv: - real_ghostscript(sys.argv) - return - - # Fail - print("ERROR: Casper is not a friendly ghost", file=sys.stderr) - sys.exit(1) +def raise_gs_fail(*args, **kwargs): + raise CalledProcessError( + 1, 'gs', output=b"", stderr=b"ERROR: Casper is not a friendly ghost" + ) -if __name__ == '__main__': - main() +@hookimpl +def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): + with patch('ocrmypdf.exec.ghostscript.run', new=raise_gs_fail): + ghostscript.generate_pdfa( + pdf_pages=pdf_pages, + pdfmark=pdfmark, + output_file=output_file, + compression=compression, + pdf_version=pdf_version, + pdfa_part=pdfa_part, + ) + return output_file diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 0e6931df..a58a3c04 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -32,26 +32,6 @@ run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api spoof = pytest.helpers.spoof -@pytest.fixture -def spoof_gs_render_fail(tmp_path_factory): - return spoof(tmp_path_factory, gs='gs_render_failure.py') - - -@pytest.fixture -def spoof_gs_raster_fail(tmp_path_factory): - return spoof(tmp_path_factory, gs='gs_raster_failure.py') - - -@pytest.fixture -def spoof_no_pdfa(tmp_path_factory): - return spoof(tmp_path_factory, gs='gs_pdfa_failure.py') - - -@pytest.fixture -def spoof_pdfa_warning(tmp_path_factory): - return spoof(tmp_path_factory, gs='gs_feature_elision.py') - - @pytest.fixture def francais(resources): path = resources / 'francais.pdf' @@ -106,48 +86,52 @@ def test_rasterize_rotated(francais, outdir, caplog): assert im.info['dpi'] == (forced_dpi[1], forced_dpi[0]) -def test_gs_render_failure(spoof_gs_render_fail, resources, outpdf): +def test_gs_render_failure(resources, outpdf): p, out, err = run_ocrmypdf( resources / 'blank.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_noop.py', - env=spoof_gs_render_fail, + '--plugin', + 'tests/plugins/gs_render_failure.py', ) assert 'Casper is not a friendly ghost' in err assert p.returncode == ExitCode.child_process_error -def test_gs_raster_failure(spoof_gs_raster_fail, resources, outpdf): +def test_gs_raster_failure(resources, outpdf): p, out, err = run_ocrmypdf( resources / 'francais.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_noop.py', - env=spoof_gs_raster_fail, + '--plugin', + 'tests/plugins/gs_raster_failure.py', ) assert 'Ghost story archive not found' in err assert p.returncode == ExitCode.child_process_error -def test_ghostscript_pdfa_failure(spoof_no_pdfa, resources, outpdf): +def test_ghostscript_pdfa_failure(resources, outpdf): p, out, err = run_ocrmypdf( resources / 'francais.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_noop.py', - env=spoof_no_pdfa, + '--plugin', + 'tests/plugins/gs_pdfa_failure.py', ) assert ( p.returncode == ExitCode.pdfa_conversion_failed ), "Unexpected return when PDF/A fails" -def test_ghostscript_feature_elision(spoof_pdfa_warning, resources, outpdf): +def test_ghostscript_feature_elision(resources, outpdf): check_ocrmypdf( resources / 'francais.pdf', outpdf, '--plugin', 'tests/plugins/tesseract_noop.py', - env=spoof_pdfa_warning, + '--plugin', + 'tests/plugins/gs_feature_elision.py', ) From ebbf68bd08dd25ea2afd5f674f9a7f3bb2903685 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 00:08:20 -0700 Subject: [PATCH 509/880] The big payoff: abolishing spoofing machinery --- src/ocrmypdf/exec/tesseract.py | 4 +- tests/conftest.py | 115 +-------------------------------- tests/test_ghostscript.py | 1 - 3 files changed, 4 insertions(+), 116 deletions(-) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 8e0f4608..591445b7 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -272,7 +272,7 @@ def generate_hocr( if user_patterns: args_tesseract.extend(['--user-patterns', user_patterns]) - # Reminder: test suite tesseract spoofers will break after any changes + # Reminder: test suite tesseract test plugins will break after any changes # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) try: @@ -353,7 +353,7 @@ def generate_pdf( prefix = os.path.splitext(output_pdf)[0] # Tesseract appends suffixes - # Reminder: test suite tesseract spoofers might break after any changes + # Reminder: test suite tesseract test plugins might break after any changes # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig) diff --git a/tests/conftest.py b/tests/conftest.py index 0b88a070..75789771 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -71,104 +71,10 @@ def have_unpaper(): TESTS_ROOT = Path(__file__).parent.resolve() -SPOOF_PATH = TESTS_ROOT / 'spoof' PROJECT_ROOT = TESTS_ROOT OCRMYPDF = [sys.executable, '-m', 'ocrmypdf'] -WINDOWS_SHIM_TEMPLATE = """ -# This is a shim for Windows that has the same effect as a symlink to the target .py -# file -import os -import subprocess -import sys - -args = [sys.executable, {spoofer}, *sys.argv[1:]] -p = subprocess.run(args, check=False, stdout=subprocess.PIPE, stderr=subprocess.PIPE) -sys.stdout.buffer.write(p.stdout) -sys.stderr.buffer.write(p.stderr) -sys.exit(p.returncode) -""" - -assert ast.parse(WINDOWS_SHIM_TEMPLATE.format(spoofer=repr(r"C:\\Temp\\file.py"))) - - -@pytest.helpers.register -def spoof(tmp_path_factory, **kwargs): - r"""Modify PATH to override subprocess executables - - spoof(tmp_path_factory, program1='replacement', ...) - - For the test suite we need a way override executables, so that we can - substitute desired results such as errors or just speed up OCR. - - On POSIXish platforms we create a temporary folder with overrides that - are symlinks to the executables we want to run. We do not actually override - PATH. We also set an environment variable _OCRMYPDF_TEST_PATH, which - OCRmyPDF's subprocess wrapper will check before they use regular PATH. The - output is a folder full of executables we are overriding. We can override - multiple executables. The end result is a folder we can use in a PATH-style - lookup to override some executables: - - /tmp/abcxyz/tesseract -> ocrmypdf/tests/resources/spoof/tesseract_crash.py - /tmp/abcxyz/gs -> ocrmypdf/tests/resources/spoof/gs_backflip.py - - Windows needs extra help from us because usually, only the Administrator - can create symlinks. Instead we create small Python scripts that call - the programs we want, implementing the effect of a symlink. This is cleaner - than creating Windows executables or trying to use non-Python scripts. - The temporary folder generated for Windows could like: - - %TEMP%\abcxyz\tesseract.py: - (script that runs ocrmypdf/tests/resources/spoof/tesseract_crash.py) - %TEMP%\abcxyz\gswin32c.py: - (script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py) - %TEMP%\abcxyz\gswin64c.py: - (script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py) - - We also address one quirk here, that Ghostscript may be known as gswin32c - or gswin64c, depending on what the user installed (regardless of Windows - itself). On POSIX, Ghostscript is just 'gs'. We handle the special case here - too. - - All of this is intimately dependent on the machinery in ocrmypdf.exec.run(). - In particular, for Windows, that code has to know that if there is a .py - file, it needs to run it with Python, since Windows does not like being - asked to execute files. - - We don't overload PATH directly because we have some tests where we call - ocrmypdf as a subprocess (to exercise the command line interface) and some - tests where we call it as an API. - """ - env = os.environ.copy() - slug = '-'.join(v.replace('.py', '') for v in sorted(kwargs.values())) - spoofer_base = tmp_path_factory.mktemp('spoofers') - tmpdir = Path(spoofer_base / slug) - tmpdir.mkdir(parents=True) - - for replace_program, with_spoof in kwargs.items(): - spoofer = SPOOF_PATH / with_spoof - if os.name != 'nt': - spoofer.chmod(0o755) - (tmpdir / replace_program).symlink_to(spoofer) - else: - py_file = WINDOWS_SHIM_TEMPLATE.format( - spoofer=repr(os.fspath(spoofer.absolute())) - ) - if replace_program == 'gs': - programs = ['gswin64c', 'gswin32c'] - else: - programs = [replace_program] - for prog in programs: - (tmpdir / f'{prog}.py').write_text(py_file, encoding='utf-8') - - env['_OCRMYPDF_TEST_PATH'] = str(tmpdir) + os.pathsep + env['PATH'] - if os.name == 'nt': - if '.py' not in env['PATHEXT'].lower(): - raise EnvironmentError("PATHEXT is not configured to support .py") - return env - - @pytest.fixture def resources(): return Path(TESTS_ROOT) / 'resources' @@ -208,12 +114,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): _parser, options, plugin_manager = get_parser_options_plugins(args=args) api.check_options(options, plugin_manager) if env: - first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] - if 'tesseract_noop' in first: - raise ValueError('noop') - else: - options.tesseract_env = env - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + assert False, 'env set' result = api.run_pipeline(options, plugin_manager=plugin_manager, api=True) assert result == 0 @@ -235,19 +136,7 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): ] _parser, options, plugin_manager = get_parser_options_plugins(args=args) if env: - try: - first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] - if 'tesseract_noop' in first: - raise ValueError('noop') - else: - options.tesseract_env = env.copy() - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] - if 'spoof' in first_path: - assert 'gs' not in first_path, "use run_ocrmypdf() for gs" - assert 'tesseract' in first_path - except KeyError: - pass + assert False, 'env set' if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index a58a3c04..34472e91 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -29,7 +29,6 @@ from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf run_ocrmypdf = pytest.helpers.run_ocrmypdf run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api -spoof = pytest.helpers.spoof @pytest.fixture From 21c0e045cbdd2a7a7253532ff34085a248239f03 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 00:30:13 -0700 Subject: [PATCH 510/880] Remove _OCRMYPDF_TEST_PATH environment variable --- src/ocrmypdf/exec/_support.py | 15 ++------------- tests/conftest.py | 14 ++++---------- tests/test_stdio.py | 8 +++----- 3 files changed, 9 insertions(+), 28 deletions(-) diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/exec/_support.py index 5f52ac87..3ff10558 100644 --- a/src/ocrmypdf/exec/_support.py +++ b/src/ocrmypdf/exec/_support.py @@ -34,20 +34,10 @@ from ocrmypdf.exceptions import MissingDependencyError log = logging.getLogger(__name__) -def _get_program(args, env=None): - program = args[0] - test_path = env.get('_OCRMYPDF_TEST_PATH', '') - if test_path: - program = shutil.which(program, path=test_path) - return program - - def run(args, *, env=None, **kwargs): """Wrapper around subprocess.run() - The main purpose of this wrapper is to allow us to substitute the main program - for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces - the main PATH as a location to check for programs to run. + The main purpose of this wrapper is to log subprocess output. Secondly we have to account for behavioral differences in Windows in particular. Creating symbolic links in Windows requires administrator privileges and @@ -62,8 +52,7 @@ def run(args, *, env=None, **kwargs): env = os.environ # Search in spoof path if necessary - program = _get_program(args, env) - args = [program] + args[1:] + program = args[0] if os.name == 'nt': args = fix_windows_args(program, args, env) diff --git a/tests/conftest.py b/tests/conftest.py index 75789771..b520d73b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -105,7 +105,7 @@ def no_outpdf(tmp_path): @pytest.helpers.register -def check_ocrmypdf(input_file, output_file, *args, env=None): +def check_ocrmypdf(input_file, output_file, *args): """Run ocrmypdf and confirmed that a valid file was created""" args = [str(input_file), str(output_file)] + [ str(arg) for arg in args if arg is not None @@ -113,8 +113,6 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): _parser, options, plugin_manager = get_parser_options_plugins(args=args) api.check_options(options, plugin_manager) - if env: - assert False, 'env set' result = api.run_pipeline(options, plugin_manager=plugin_manager, api=True) assert result == 0 @@ -125,7 +123,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): @pytest.helpers.register -def run_ocrmypdf_api(input_file, output_file, *args, env=None): +def run_ocrmypdf_api(input_file, output_file, *args): """Run ocrmypdf via API and let caller deal with results Does not currently have a way to manipulate the PATH except for Tesseract. @@ -135,8 +133,6 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): str(arg) for arg in args if arg is not None ] _parser, options, plugin_manager = get_parser_options_plugins(args=args) - if env: - assert False, 'env set' if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) @@ -145,12 +141,9 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): @pytest.helpers.register -def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=True): +def run_ocrmypdf(input_file, output_file, *args, universal_newlines=True): "Run ocrmypdf and let caller deal with results" - if env is None: - env = os.environ.copy() - p_args = ( OCRMYPDF + [str(arg) for arg in args if arg is not None] @@ -162,6 +155,7 @@ def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=Tr # Details: https://coverage.readthedocs.io/en/coverage-5.0/subprocess.html coverage_rc = Path(__file__).parent.parent / '.coveragerc' assert coverage_rc.exists() + env = os.environ.copy() env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) p = run( diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 883d48de..478661c2 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -105,11 +105,9 @@ def test_closed_streams(ocrmypdf_exec, resources, outpdf): Path('/etc/alpine-release').exists(), reason="invalid test on alpine" ) @pytest.mark.skipif(os.name == 'nt', reason="invalid test on Windows") -def test_bad_locale(): - env = os.environ.copy() - env['LC_ALL'] = 'C' - - p, out, err = run_ocrmypdf('a', 'b', env=env) +def test_bad_locale(monkeypatch): + monkeypatch.setenv('LC_ALL', 'C') + p, out, err = run_ocrmypdf('a', 'b') assert out == '', "stdout not clean" assert p.returncode != 0 assert 'configured to use ASCII as encoding' in err, "should whine" From 3b6f6782f0a3950d0c0f4e6f19f16ec2f2acbd9a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 00:39:53 -0700 Subject: [PATCH 511/880] Remove tesseract_env, --tesseract-env --- src/ocrmypdf/api.py | 2 +- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 25 ++------ src/ocrmypdf/exec/tesseract.py | 57 ++++--------------- tests/conftest.py | 2 - tests/test_tesseract.py | 10 +--- 5 files changed, 22 insertions(+), 74 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 008d284d..7e389d88 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -143,7 +143,7 @@ def create_options( # These arguments with special handling for which we bypass # argparse - if arg in {'tesseract_env', 'progress_bar', 'plugins'}: + if arg in {'progress_bar', 'plugins'}: deferred.append((arg, val)) continue diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index d5f5eb7b..150fcedf 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -81,7 +81,6 @@ def add_options(parser): metavar='FILE', help="Specify the location of the Tesseract user patterns file.", ) - tess.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS) @hookimpl @@ -98,15 +97,13 @@ def check_options(options): options.pdf_renderer = 'sandwich' if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( - options.tesseract_env, set(options.languages) + set(options.languages) ): raise MissingDependencyError( "You are using an alpha version of Tesseract 4.0 that does not support " "the textonly_pdf parameter. We don't support versions this old." ) - if not tesseract.has_user_words(options.tesseract_env) and ( - options.user_words or options.user_patterns - ): + if not tesseract.has_user_words() and (options.user_words or options.user_patterns): log.warning( "Tesseract 4.0 ignores --user-words and --user-patterns, so these " "arguments have no effect." @@ -120,13 +117,6 @@ def check_options(options): @hookimpl def validate(pdfinfo, options): - if not options.tesseract_env: - return - - # If we are running a Tesseract spoof, ensure it knows what the input file is - if os.environ.get('PYTEST_CURRENT_TEST'): - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(options.input_file) - # Tesseract 4.x can be multithreaded, and we also run multiple workers. We want # to manage how many threads it uses to avoid creating total threads than cores. # Performance testing shows we're better off @@ -135,11 +125,11 @@ def validate(pdfinfo, options): # input file is small, then we allow Tesseract to use threads, subject to the # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers. # As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system. - if not options.tesseract_env.get('OMP_THREAD_LIMIT', '').isnumeric(): + if not os.environ.get('OMP_THREAD_LIMIT', '').isnumeric(): tess_threads = min(3, options.jobs // len(pdfinfo), len(pdfinfo)) - options.tesseract_env['OMP_THREAD_LIMIT'] = str(tess_threads) + os.environ['OMP_THREAD_LIMIT'] = str(tess_threads) else: - tess_threads = int(options.tesseract_env['OMP_THREAD_LIMIT']) + tess_threads = int(os.environ['OMP_THREAD_LIMIT']) if tess_threads > 1: log.info("Using Tesseract OpenMP thread limit %d", tess_threads) @@ -160,7 +150,7 @@ class TesseractOcrEngine(OcrEngine): @staticmethod def languages(options): - return tesseract.get_languages(options.tesseract_env) + return tesseract.get_languages() @staticmethod def get_orientation(input_file, options): @@ -168,7 +158,6 @@ class TesseractOcrEngine(OcrEngine): input_file, engine_mode=options.tesseract_oem, timeout=options.tesseract_timeout, - tesseract_env=options.tesseract_env, ) @staticmethod @@ -184,7 +173,6 @@ class TesseractOcrEngine(OcrEngine): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, - tesseract_env=options.tesseract_env, ) @staticmethod @@ -200,7 +188,6 @@ class TesseractOcrEngine(OcrEngine): pagesegmode=options.tesseract_pagesegmode, user_words=options.user_words, user_patterns=options.user_patterns, - tesseract_env=options.tesseract_env, ) diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 591445b7..d163d305 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -65,11 +65,11 @@ class TesseractLoggerAdapter(logging.LoggerAdapter): return '[tesseract] %s' % (msg), kwargs -def version(tesseract_env=None): - return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env) +def version(): + return get_version('tesseract', regex=r'tesseract\s(.+)') -def has_textonly_pdf(tesseract_env=None, langs=None): +def has_textonly_pdf(langs=None): """Does Tesseract have textonly_pdf capability? Available in v4.00.00alpha since January 2017. Best to @@ -79,12 +79,7 @@ def has_textonly_pdf(tesseract_env=None, langs=None): params = '' try: proc = run( - args_tess, - check=True, - universal_newlines=True, - stdout=PIPE, - stderr=STDOUT, - env=tesseract_env, + args_tess, check=True, universal_newlines=True, stdout=PIPE, stderr=STDOUT ) params = proc.stdout except CalledProcessError as e: @@ -97,16 +92,16 @@ def has_textonly_pdf(tesseract_env=None, langs=None): return False -def has_user_words(tesseract_env=None): +def has_user_words(): """Does Tesseract have --user-words capability? Not available in 4.0, but available in 4.1. Also available in 3.x, but we no longer support 3.x. """ - return version(tesseract_env) >= '4.1' + return version() >= '4.1' -def get_languages(tesseract_env=None): +def get_languages(): def lang_error(output): msg = ( "Tesseract failed to report available languages.\n" @@ -119,12 +114,7 @@ def get_languages(tesseract_env=None): args_tess = ['tesseract', '--list-langs'] try: proc = run( - args_tess, - universal_newlines=True, - stdout=PIPE, - stderr=STDOUT, - check=True, - env=tesseract_env, + args_tess, universal_newlines=True, stdout=PIPE, stderr=STDOUT, check=True ) output = proc.stdout except CalledProcessError as e: @@ -146,7 +136,7 @@ def tess_base_args(langs: List[str], engine_mode) -> List[str]: return args -def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env=None): +def get_orientation(input_file: Path, engine_mode, timeout: float): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', @@ -155,14 +145,7 @@ def get_orientation(input_file: Path, engine_mode, timeout: float, tesseract_env ] try: - p = run( - args_tesseract, - stdout=PIPE, - stderr=STDOUT, - timeout=timeout, - check=True, - env=tesseract_env, - ) + p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True) stdout = p.stdout except TimeoutExpired: return OrientationConfidence(angle=0, confidence=0.0) @@ -257,7 +240,6 @@ def generate_hocr( pagesegmode: int, user_words, user_patterns, - tesseract_env, ): prefix = output_hocr.with_suffix('') @@ -276,14 +258,7 @@ def generate_hocr( # to the number of order parameters here args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig) try: - p = run( - args_tesseract, - stdout=PIPE, - stderr=STDOUT, - timeout=timeout, - check=True, - env=tesseract_env, - ) + p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True) stdout = p.stdout except TimeoutExpired: # Generate a HOCR file with no recognized text if tesseract times out @@ -325,7 +300,6 @@ def generate_pdf( pagesegmode: int, user_words, user_patterns, - tesseract_env, ): """Use Tesseract to render a PDF. @@ -358,14 +332,7 @@ def generate_pdf( args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig) try: - p = run( - args_tesseract, - stdout=PIPE, - stderr=STDOUT, - timeout=timeout, - check=True, - env=tesseract_env, - ) + p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True) stdout = p.stdout if os.path.exists(prefix + '.txt'): shutil.move(prefix + '.txt', output_text) diff --git a/tests/conftest.py b/tests/conftest.py index b520d73b..34bbba86 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -133,8 +133,6 @@ def run_ocrmypdf_api(input_file, output_file, *args): str(arg) for arg in args if arg is not None ] _parser, options, plugin_manager = get_parser_options_plugins(args=args) - if options.tesseract_env: - assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) api.check_options(options, plugin_manager) return api.run_pipeline(options, plugin_manager=None, api=False) diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index 99222d5a..a8a28dc6 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -70,13 +70,11 @@ def test_content_preservation(resources, outpdf): assert len(page.images) > 1, "masks were rasterized" -def test_no_languages(tmp_path): - env = os.environ.copy() +def test_no_languages(tmp_path, monkeypatch): (tmp_path / 'tessdata').mkdir() - env['TESSDATA_PREFIX'] = fspath(tmp_path) - + monkeypatch.setenv('TESSDATA_PREFIX', fspath(tmp_path)) with pytest.raises(MissingDependencyError): - tesseract.get_languages(tesseract_env=env) + tesseract.get_languages() def test_image_too_large_hocr(monkeypatch, resources, outdir): @@ -95,7 +93,6 @@ def test_image_too_large_hocr(monkeypatch, resources, outdir): pagesegmode=None, user_words=None, user_patterns=None, - tesseract_env=None, ) assert "name='ocr-capabilities'" in Path(outdir / 'out.hocr').read_text() @@ -116,7 +113,6 @@ def test_image_too_large_pdf(monkeypatch, resources, outdir): pagesegmode=None, user_words=None, user_patterns=None, - tesseract_env=None, ) assert Path(outdir / 'txt.txt').read_text() == '[skipped page]' if os.name != 'nt': # different semantics From be8ca589d42a980e0b3d3c2c9fdfb871adb0c764 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 14:53:10 -0700 Subject: [PATCH 512/880] Move ocrmypdf.exec.run and friends to ocrmypdf.subprocess --- src/ocrmypdf/_validation.py | 3 ++- src/ocrmypdf/builtin_plugins/ghostscript.py | 3 ++- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 3 ++- src/ocrmypdf/exec/__init__.py | 9 +-------- src/ocrmypdf/exec/ghostscript.py | 2 +- src/ocrmypdf/exec/jbig2enc.py | 2 +- src/ocrmypdf/exec/pngquant.py | 3 +-- src/ocrmypdf/exec/tesseract.py | 2 +- src/ocrmypdf/exec/unpaper.py | 4 ++-- src/ocrmypdf/leptonica.py | 2 +- src/ocrmypdf/{exec/_support.py => subprocess.py} | 4 ++-- tests/plugins/gs_feature_elision.py | 2 +- tests/plugins/gs_pdfa_failure.py | 2 +- tests/plugins/gs_raster_failure.py | 2 +- tests/plugins/gs_render_failure.py | 2 +- tests/plugins/tesseract_cache.py | 2 +- tests/test_main.py | 3 ++- 17 files changed, 23 insertions(+), 27 deletions(-) rename src/ocrmypdf/{exec/_support.py => subprocess.py} (99%) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 3a92d34c..ca5dedda 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -34,13 +34,14 @@ from ocrmypdf.exceptions import ( MissingDependencyError, OutputFileAccessError, ) -from ocrmypdf.exec import check_external_program, jbig2enc, pngquant, unpaper +from ocrmypdf.exec import jbig2enc, pngquant, unpaper from ocrmypdf.helpers import ( is_file_writable, is_iterable_notstr, monotonic, safe_symlink, ) +from ocrmypdf.subprocess import check_external_program # ------------- # External dependencies diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index c74bb360..76b86d94 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -21,8 +21,9 @@ from pathlib import Path from ocrmypdf import hookimpl from ocrmypdf._validation import HOCR_OK_LANGS from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import check_external_program, ghostscript +from ocrmypdf.exec import ghostscript from ocrmypdf.helpers import Resolution +from ocrmypdf.subprocess import check_external_program log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 150fcedf..2d7ae3e2 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -22,8 +22,9 @@ import os from ocrmypdf import hookimpl from ocrmypdf.cli import numeric from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import check_external_program, tesseract +from ocrmypdf.exec import tesseract from ocrmypdf.pluginspec import OcrEngine +from ocrmypdf.subprocess import check_external_program log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/exec/__init__.py index 13b4e48c..8c6d0bb3 100644 --- a/src/ocrmypdf/exec/__init__.py +++ b/src/ocrmypdf/exec/__init__.py @@ -15,11 +15,4 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -"""Wrappers to manage subprocess calls""" - -from ocrmypdf.exec._support import ( - check_external_program, - get_version, - run, - shim_paths_with_program_files, -) +"""Manage third party executables""" diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index 9fe690b9..0fb65b1b 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -29,8 +29,8 @@ from subprocess import PIPE, CalledProcessError from PIL import Image from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError -from ocrmypdf.exec import get_version, run from ocrmypdf.helpers import Resolution +from ocrmypdf.subprocess import get_version, run log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/exec/jbig2enc.py index c027e28d..deced89a 100644 --- a/src/ocrmypdf/exec/jbig2enc.py +++ b/src/ocrmypdf/exec/jbig2enc.py @@ -20,7 +20,7 @@ from subprocess import PIPE from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import get_version, run +from ocrmypdf.subprocess import get_version, run def version(): diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/exec/pngquant.py index ad00560f..61f197fe 100644 --- a/src/ocrmypdf/exec/pngquant.py +++ b/src/ocrmypdf/exec/pngquant.py @@ -17,13 +17,12 @@ """Interface to pngquant executable""" -from subprocess import run from tempfile import NamedTemporaryFile from PIL import Image from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import get_version +from ocrmypdf.subprocess import get_version, run def version(): diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index d163d305..3beb6ff2 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -34,8 +34,8 @@ from ocrmypdf.exceptions import ( SubprocessOutputError, TesseractConfigError, ) -from ocrmypdf.exec import get_version, run from ocrmypdf.helpers import safe_symlink +from ocrmypdf.subprocess import get_version, run log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/exec/unpaper.py index a1aa749e..e1a58746 100644 --- a/src/ocrmypdf/exec/unpaper.py +++ b/src/ocrmypdf/exec/unpaper.py @@ -30,8 +30,8 @@ from tempfile import TemporaryDirectory from PIL import Image from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError -from ocrmypdf.exec import get_version -from ocrmypdf.exec import run as external_run +from ocrmypdf.subprocess import get_version +from ocrmypdf.subprocess import run as external_run log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 50e9b294..6af38089 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -34,8 +34,8 @@ from os import fspath from tempfile import TemporaryFile from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import shim_paths_with_program_files from ocrmypdf.lib._leptonica import ffi +from ocrmypdf.subprocess import shim_paths_with_program_files # pylint: disable=protected-access diff --git a/src/ocrmypdf/exec/_support.py b/src/ocrmypdf/subprocess.py similarity index 99% rename from src/ocrmypdf/exec/_support.py rename to src/ocrmypdf/subprocess.py index 3ff10558..36c36cae 100644 --- a/src/ocrmypdf/exec/_support.py +++ b/src/ocrmypdf/subprocess.py @@ -55,7 +55,7 @@ def run(args, *, env=None, **kwargs): program = args[0] if os.name == 'nt': - args = fix_windows_args(program, args, env) + args = _fix_windows_args(program, args, env) log.debug("Running: %s", args) process_log = log.getChild('subprocess.' + os.path.basename(program)) @@ -80,7 +80,7 @@ def run(args, *, env=None, **kwargs): return proc -def fix_windows_args(program, args, env): +def _fix_windows_args(program, args, env): """Adjust our desired program and command line arguments for use on Windows""" if sys.version_info < (3, 8): diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index 84f5e6e1..329eecf1 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -23,7 +23,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.exec import run +from ocrmypdf.subprocess import run elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 not permitted in PDF/A-2, overprint mode not set""" diff --git a/tests/plugins/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py index f9093224..b14c5ea3 100644 --- a/tests/plugins/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -23,7 +23,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.exec import run +from ocrmypdf.subprocess import run def run_rig_args(args, **kwargs): diff --git a/tests/plugins/gs_raster_failure.py b/tests/plugins/gs_raster_failure.py index cab3268e..aa032970 100644 --- a/tests/plugins/gs_raster_failure.py +++ b/tests/plugins/gs_raster_failure.py @@ -25,7 +25,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.exec import run +from ocrmypdf.subprocess import run def raise_gs_fail(*args, **kwargs): diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index e1d934ce..eadee7b3 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -25,7 +25,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.exec import run +from ocrmypdf.subprocess import run def raise_gs_fail(*args, **kwargs): diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py index 807d65b0..74d3ab03 100644 --- a/tests/plugins/tesseract_cache.py +++ b/tests/plugins/tesseract_cache.py @@ -57,7 +57,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins.tesseract_ocr import TesseractOcrEngine -from ocrmypdf.exec import run +from ocrmypdf.subprocess import run log = logging.getLogger(__name__) diff --git a/tests/test_main.py b/tests/test_main.py index 0b913a14..b2cd735f 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -29,9 +29,10 @@ from PIL import Image import ocrmypdf from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import get_version, ghostscript, tesseract +from ocrmypdf.exec import ghostscript, tesseract from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo +from ocrmypdf.subprocess import get_version # pytest.helpers is dynamic # pylint: disable=no-member,redefined-outer-name From 0f942fb714d5a5b7d8ea868f1221899ce85b4626 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 14:55:54 -0700 Subject: [PATCH 513/880] Rename ocrmypdf.exec -> ocrmypdf._exec --- src/ocrmypdf/{exec => _exec}/__init__.py | 0 src/ocrmypdf/{exec => _exec}/ghostscript.py | 0 src/ocrmypdf/{exec => _exec}/jbig2enc.py | 0 src/ocrmypdf/{exec => _exec}/pngquant.py | 0 src/ocrmypdf/{exec => _exec}/tesseract.py | 0 src/ocrmypdf/{exec => _exec}/unpaper.py | 0 src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/builtin_plugins/ghostscript.py | 2 +- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 2 +- src/ocrmypdf/optimize.py | 2 +- tests/conftest.py | 2 +- tests/plugins/gs_feature_elision.py | 2 +- tests/plugins/gs_pdfa_failure.py | 2 +- tests/plugins/gs_raster_failure.py | 2 +- tests/plugins/gs_render_failure.py | 2 +- tests/plugins/tesseract_badutf8.py | 4 ++-- tests/plugins/tesseract_big_image_error.py | 6 +++--- tests/plugins/tesseract_cache.py | 6 +++--- tests/plugins/tesseract_crash.py | 6 +++--- tests/test_ghostscript.py | 2 +- tests/test_hocrtransform.py | 2 +- tests/test_main.py | 2 +- tests/test_optimize.py | 4 ++-- tests/test_pdfinfo.py | 2 +- tests/test_preprocessing.py | 2 +- tests/test_rotation.py | 2 +- tests/test_tesseract.py | 2 +- tests/test_unpaper.py | 4 ++-- tests/test_validation.py | 18 +++++++++--------- 30 files changed, 41 insertions(+), 41 deletions(-) rename src/ocrmypdf/{exec => _exec}/__init__.py (100%) rename src/ocrmypdf/{exec => _exec}/ghostscript.py (100%) rename src/ocrmypdf/{exec => _exec}/jbig2enc.py (100%) rename src/ocrmypdf/{exec => _exec}/pngquant.py (100%) rename src/ocrmypdf/{exec => _exec}/tesseract.py (100%) rename src/ocrmypdf/{exec => _exec}/unpaper.py (100%) diff --git a/src/ocrmypdf/exec/__init__.py b/src/ocrmypdf/_exec/__init__.py similarity index 100% rename from src/ocrmypdf/exec/__init__.py rename to src/ocrmypdf/_exec/__init__.py diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py similarity index 100% rename from src/ocrmypdf/exec/ghostscript.py rename to src/ocrmypdf/_exec/ghostscript.py diff --git a/src/ocrmypdf/exec/jbig2enc.py b/src/ocrmypdf/_exec/jbig2enc.py similarity index 100% rename from src/ocrmypdf/exec/jbig2enc.py rename to src/ocrmypdf/_exec/jbig2enc.py diff --git a/src/ocrmypdf/exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py similarity index 100% rename from src/ocrmypdf/exec/pngquant.py rename to src/ocrmypdf/_exec/pngquant.py diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py similarity index 100% rename from src/ocrmypdf/exec/tesseract.py rename to src/ocrmypdf/_exec/tesseract.py diff --git a/src/ocrmypdf/exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py similarity index 100% rename from src/ocrmypdf/exec/unpaper.py rename to src/ocrmypdf/_exec/unpaper.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 094b0858..20bc3dfb 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -28,6 +28,7 @@ from pikepdf.models.metadata import encode_pdf_date from PIL import Image, ImageColor, ImageDraw from ocrmypdf import leptonica +from ocrmypdf._exec import ghostscript, unpaper from ocrmypdf._version import PROGRAM_NAME from ocrmypdf._version import __version__ as VERSION from ocrmypdf.exceptions import ( @@ -37,7 +38,6 @@ from ocrmypdf.exceptions import ( PriorOcrFoundError, UnsupportedImageFormatError, ) -from ocrmypdf.exec import ghostscript, unpaper from ocrmypdf.helpers import Resolution, safe_symlink from ocrmypdf.hocrtransform import HocrTransform from ocrmypdf.optimize import optimize diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index ca5dedda..02483fb1 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -27,6 +27,7 @@ from shutil import copyfileobj import PIL +from ocrmypdf._exec import jbig2enc, pngquant, unpaper from ocrmypdf._unicodefun import verify_python3_env from ocrmypdf.exceptions import ( BadArgsError, @@ -34,7 +35,6 @@ from ocrmypdf.exceptions import ( MissingDependencyError, OutputFileAccessError, ) -from ocrmypdf.exec import jbig2enc, pngquant, unpaper from ocrmypdf.helpers import ( is_file_writable, is_iterable_notstr, diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 76b86d94..e451c771 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -19,9 +19,9 @@ import logging from pathlib import Path from ocrmypdf import hookimpl +from ocrmypdf._exec import ghostscript from ocrmypdf._validation import HOCR_OK_LANGS from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import ghostscript from ocrmypdf.helpers import Resolution from ocrmypdf.subprocess import check_external_program diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 2d7ae3e2..bd15ddbe 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -20,9 +20,9 @@ import logging import os from ocrmypdf import hookimpl +from ocrmypdf._exec import tesseract from ocrmypdf.cli import numeric from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import tesseract from ocrmypdf.pluginspec import OcrEngine from ocrmypdf.subprocess import check_external_program diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 41bf7c60..35e1d731 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -30,9 +30,9 @@ from tqdm import tqdm from ocrmypdf import leptonica from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._jobcontext import PdfContext from ocrmypdf.exceptions import OutputFileAccessError -from ocrmypdf.exec import jbig2enc, pngquant from ocrmypdf.helpers import safe_symlink log = logging.getLogger(__name__) diff --git a/tests/conftest.py b/tests/conftest.py index 34bbba86..adcd1354 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -25,8 +25,8 @@ from subprocess import PIPE, run import pytest from ocrmypdf import api, cli, pdfinfo +from ocrmypdf._exec import unpaper from ocrmypdf._plugin_manager import get_parser_options_plugins -from ocrmypdf.exec import unpaper pytest_plugins = ['helpers_namespace'] diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index 329eecf1..419855cb 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -37,7 +37,7 @@ def run_append_stderr(*args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf.exec.ghostscript.run', new=run_append_stderr): + with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, diff --git a/tests/plugins/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py index b14c5ea3..dcad94f6 100644 --- a/tests/plugins/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -39,7 +39,7 @@ def run_rig_args(args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf.exec.ghostscript.run', new=run_rig_args): + with patch('ocrmypdf._exec.ghostscript.run', new=run_rig_args): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, diff --git a/tests/plugins/gs_raster_failure.py b/tests/plugins/gs_raster_failure.py index aa032970..98b1984c 100644 --- a/tests/plugins/gs_raster_failure.py +++ b/tests/plugins/gs_raster_failure.py @@ -45,7 +45,7 @@ def rasterize_pdf_page( rotation=None, filter_vector=False, ) -> Path: - with patch('ocrmypdf.exec.ghostscript.run', new=raise_gs_fail): + with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail): ghostscript.rasterize_pdf_page( input_file=input_file, output_file=output_file, diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index eadee7b3..c27a5801 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -36,7 +36,7 @@ def raise_gs_fail(*args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf.exec.ghostscript.run', new=raise_gs_fail): + with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, diff --git a/tests/plugins/tesseract_badutf8.py b/tests/plugins/tesseract_badutf8.py index b87a3dac..3511938d 100644 --- a/tests/plugins/tesseract_badutf8.py +++ b/tests/plugins/tesseract_badutf8.py @@ -45,14 +45,14 @@ def bad_utf8(*args, **kwargs): class BadUtf8OcrEngine(TesseractOcrEngine): @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=bad_utf8): + with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=bad_utf8): + with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/plugins/tesseract_big_image_error.py b/tests/plugins/tesseract_big_image_error.py index f040855c..04d0e0cd 100644 --- a/tests/plugins/tesseract_big_image_error.py +++ b/tests/plugins/tesseract_big_image_error.py @@ -38,19 +38,19 @@ def raise_size_exception(*args, **kwargs): class BigImageErrorOcrEngine(TesseractOcrEngine): @staticmethod def get_orientation(input_file, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): return TesseractOcrEngine.get_orientation(input_file, options) @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_size_exception): + with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py index 74d3ab03..1df3fd98 100644 --- a/tests/plugins/tesseract_cache.py +++ b/tests/plugins/tesseract_cache.py @@ -178,19 +178,19 @@ def cached_run(options, run_args, **run_kwargs): class CacheOcrEngine(TesseractOcrEngine): @staticmethod def get_orientation(input_file, options): - with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + with patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)): return TesseractOcrEngine.get_orientation(input_file, options) @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + with patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=partial(cached_run, options)): + with patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/plugins/tesseract_crash.py b/tests/plugins/tesseract_crash.py index 806af41b..74c3970a 100755 --- a/tests/plugins/tesseract_crash.py +++ b/tests/plugins/tesseract_crash.py @@ -41,19 +41,19 @@ def raise_crash(*args, **kwargs): class CrashOcrEngine(TesseractOcrEngine): @staticmethod def get_orientation(input_file, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): return TesseractOcrEngine.get_orientation(input_file, options) @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf.exec.tesseract.run', new=raise_crash): + with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 34472e91..a2dd90d7 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -22,8 +22,8 @@ import pikepdf import pytest from PIL import Image +from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.exceptions import ExitCode -from ocrmypdf.exec.ghostscript import rasterize_pdf from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index e00f8365..1a1f817a 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -21,7 +21,7 @@ import pytest from PIL import Image from ocrmypdf import hocrtransform -from ocrmypdf.exec.tesseract import HOCR_TEMPLATE +from ocrmypdf._exec.tesseract import HOCR_TEMPLATE from ocrmypdf.helpers import check_pdf # pylint: disable=redefined-outer-name diff --git a/tests/test_main.py b/tests/test_main.py index b2cd735f..289cbbcc 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -28,8 +28,8 @@ import pytest from PIL import Image import ocrmypdf +from ocrmypdf._exec import ghostscript, tesseract from ocrmypdf.exceptions import ExitCode, MissingDependencyError -from ocrmypdf.exec import ghostscript, tesseract from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo from ocrmypdf.subprocess import get_version diff --git a/tests/test_optimize.py b/tests/test_optimize.py index e8cf0717..a957af5f 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -23,8 +23,8 @@ import pytest from PIL import Image from ocrmypdf import optimize as opt -from ocrmypdf.exec import jbig2enc, pngquant -from ocrmypdf.exec.ghostscript import rasterize_pdf +from ocrmypdf._exec import jbig2enc, pngquant +from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101 diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index cfa90d94..558995fc 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -26,7 +26,7 @@ from PIL import Image from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo -from ocrmypdf.exec import ghostscript +from ocrmypdf._exec import ghostscript from ocrmypdf.pdfinfo import Colorspace, Encoding # pylint: disable=protected-access diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 865b5e58..7cabe827 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -20,7 +20,7 @@ from math import isclose import pytest from PIL import Image -from ocrmypdf.exec import ghostscript +from ocrmypdf._exec import ghostscript from ocrmypdf.helpers import Resolution from ocrmypdf.leptonica import Pix from ocrmypdf.pdfinfo import PdfInfo diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 0c8b5dd3..f0e7fcfd 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -26,7 +26,7 @@ import pytest from PIL import Image from ocrmypdf import leptonica -from ocrmypdf.exec import ghostscript, tesseract +from ocrmypdf._exec import ghostscript, tesseract from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import PdfInfo diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index a8a28dc6..0db09110 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -24,8 +24,8 @@ from pathlib import Path import pytest from ocrmypdf import pdfinfo +from ocrmypdf._exec import tesseract from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.exec import tesseract # pylint: disable=no-member,redefined-outer-name diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 9a6dc248..6e28235a 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -39,7 +39,7 @@ def test_no_unpaper(resources, no_outpdf): output = fspath(no_outpdf) _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) - with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: + with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") with pytest.raises(MissingDependencyError): @@ -51,7 +51,7 @@ def test_old_unpaper(resources, no_outpdf): output = fspath(no_outpdf) _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) - with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: + with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.return_value = '0.5' with pytest.raises(MissingDependencyError): diff --git a/tests/test_validation.py b/tests/test_validation.py index 3e6272af..3c5ebac6 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -56,27 +56,27 @@ def test_hocr_notlatin_warning(caplog): def test_old_ghostscript(caplog): - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.19'), patch( - 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'), patch( + 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True ): vd.check_options(*make_opts_pm(language='chi_sim', output_type='pdfa')) assert 'Ghostscript does not work correctly' in caplog.text - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.18'), patch( - 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'), patch( + 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): vd.check_options(*make_opts_pm(output_type='pdfa-3')) - with patch('ocrmypdf.exec.ghostscript.version', return_value='9.24'), patch( - 'ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=True + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.24'), patch( + 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): vd.check_options(*make_opts_pm()) def test_old_tesseract_error(): - with patch('ocrmypdf.exec.tesseract.has_textonly_pdf', return_value=False): + with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=False): with pytest.raises(MissingDependencyError): opts = make_opts(pdf_renderer='sandwich', language='eng') plugin_manager = get_plugin_manager(opts.plugins) @@ -107,13 +107,13 @@ def test_optimizing(caplog): def test_user_words(caplog): - with patch('ocrmypdf.exec.tesseract.has_user_words', return_value=False): + with patch('ocrmypdf._exec.tesseract.has_user_words', return_value=False): opts = make_opts(user_words='foo') plugin_manager = get_plugin_manager(opts.plugins) vd.check_options(opts, plugin_manager) assert '4.0 ignores --user-words' in caplog.text caplog.clear() - with patch('ocrmypdf.exec.tesseract.has_user_words', return_value=True): + with patch('ocrmypdf._exec.tesseract.has_user_words', return_value=True): opts = make_opts(user_patterns='foo') plugin_manager = get_plugin_manager(opts.plugins) vd.check_options(opts, plugin_manager) From 64891c2fc33712c5f9c6d69098469a8543a175e4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Jun 2020 15:27:14 -0700 Subject: [PATCH 514/880] Pre-release delinting --- misc/watcher.py | 2 +- src/ocrmypdf/__main__.py | 3 +-- src/ocrmypdf/_exec/tesseract.py | 4 +--- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/_plugin_manager.py | 2 +- src/ocrmypdf/api.py | 3 +-- src/ocrmypdf/builtin_plugins/ghostscript.py | 2 -- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 1 - src/ocrmypdf/pluginspec.py | 4 ++-- tests/conftest.py | 3 +-- tests/plugins/gs_raster_failure.py | 1 - tests/plugins/gs_render_failure.py | 2 -- tests/plugins/tesseract_crash.py | 1 - tests/test_ghostscript.py | 16 +++++++++------- tests/test_helpers.py | 1 - tests/test_hocrtransform.py | 2 -- tests/test_pdfinfo.py | 6 ++---- tests/test_rotation.py | 2 -- tests/test_stdio.py | 6 +++--- tests/test_unpaper.py | 1 - tests/test_userunit.py | 8 +++++--- tests/test_validation.py | 3 +-- 22 files changed, 29 insertions(+), 46 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index d2381050..4b5e4675 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -38,7 +38,7 @@ ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', '')) DESKEW = bool(os.getenv('OCR_DESKEW', '')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) -USE_POLLING = bool(os.getenv('OCR_USE_POLLING', False)) +USE_POLLING = bool(os.getenv('OCR_USE_POLLING', '')) LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() PATTERNS = ['*.pdf'] diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 69d68db4..474c9678 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -26,14 +26,13 @@ from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_closed_streams, check_options from ocrmypdf.api import Verbosity, configure_logging -from ocrmypdf.cli import get_parser, plugins_only_parser from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError log = logging.getLogger('ocrmypdf') def run(args=None): - parser, options, plugin_manager = get_parser_options_plugins(args=args) + _parser, options, plugin_manager = get_parser_options_plugins(args=args) if not check_closed_streams(options): return ExitCode.bad_args diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index a253db74..bfa6305d 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -21,11 +21,10 @@ import logging import os import shutil from collections import namedtuple -from contextlib import suppress from os import fspath from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired -from typing import List, Optional +from typing import List from PIL import Image @@ -34,7 +33,6 @@ from ocrmypdf.exceptions import ( SubprocessOutputError, TesseractConfigError, ) -from ocrmypdf.helpers import safe_symlink from ocrmypdf.subprocess import get_version, run log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 20bc3dfb..689d2542 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -28,7 +28,7 @@ from pikepdf.models.metadata import encode_pdf_date from PIL import Image, ImageColor, ImageDraw from ocrmypdf import leptonica -from ocrmypdf._exec import ghostscript, unpaper +from ocrmypdf._exec import unpaper from ocrmypdf._version import PROGRAM_NAME from ocrmypdf._version import __version__ as VERSION from ocrmypdf.exceptions import ( diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 9f328c69..baa25e4a 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -61,7 +61,7 @@ def get_parser_options_plugins( plugin_manager = get_plugin_manager(pre_options.plugins) parser = get_parser() - plugin_manager.hook.add_options(parser=parser) + plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member options = parser.parse_args(args=args) return parser, options, plugin_manager diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 7e389d88..a0cb19f5 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -15,14 +15,13 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import inspect import logging import os import sys from argparse import ArgumentParser from enum import IntEnum from pathlib import Path -from typing import Dict, Iterable +from typing import Iterable from ocrmypdf._logging import PageNumberFilter, TqdmConsole from ocrmypdf._plugin_manager import get_plugin_manager diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index e451c771..63f79f5e 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -16,13 +16,11 @@ # along with OCRmyPDF. If not, see . import logging -from pathlib import Path from ocrmypdf import hookimpl from ocrmypdf._exec import ghostscript from ocrmypdf._validation import HOCR_OK_LANGS from ocrmypdf.exceptions import MissingDependencyError -from ocrmypdf.helpers import Resolution from ocrmypdf.subprocess import check_external_program log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index bd15ddbe..dced0c9b 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import argparse import logging import os diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index e3961fbf..ed6d0eda 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -15,7 +15,7 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from abc import ABC, abstractstaticmethod +from abc import ABC, abstractmethod, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple from pathlib import Path @@ -134,7 +134,7 @@ class OcrEngine(ABC): def creator_tag(options: Namespace) -> str: """Returns the creator tag to identify this software's role in creating the PDF.""" - @abstractstaticmethod + @abstractmethod def __str__(self): """Returns name of OCR engine and version.""" diff --git a/tests/conftest.py b/tests/conftest.py index adcd1354..849e29d0 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -15,7 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import ast import os import platform import sys @@ -24,7 +23,7 @@ from subprocess import PIPE, run import pytest -from ocrmypdf import api, cli, pdfinfo +from ocrmypdf import api, pdfinfo from ocrmypdf._exec import unpaper from ocrmypdf._plugin_manager import get_parser_options_plugins diff --git a/tests/plugins/gs_raster_failure.py b/tests/plugins/gs_raster_failure.py index 98b1984c..fbf3d5cd 100644 --- a/tests/plugins/gs_raster_failure.py +++ b/tests/plugins/gs_raster_failure.py @@ -25,7 +25,6 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.subprocess import run def raise_gs_fail(*args, **kwargs): diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index c27a5801..e3cee162 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -19,13 +19,11 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -from pathlib import Path from subprocess import CalledProcessError from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.subprocess import run def raise_gs_fail(*args, **kwargs): diff --git a/tests/plugins/tesseract_crash.py b/tests/plugins/tesseract_crash.py index 74c3970a..c76bafd5 100755 --- a/tests/plugins/tesseract_crash.py +++ b/tests/plugins/tesseract_crash.py @@ -20,7 +20,6 @@ # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. import signal -import sys from subprocess import CalledProcessError from unittest.mock import patch diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index a2dd90d7..af14f3b3 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -26,9 +26,11 @@ from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.exceptions import ExitCode from ocrmypdf.helpers import Resolution -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member +run_ocrmypdf = pytest.helpers.run_ocrmypdf # pylint: disable=no-member +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api # pylint: disable=no-member + +# pylint: disable=redefined-outer-name @pytest.fixture @@ -37,7 +39,7 @@ def francais(resources): return path, pikepdf.open(path) -def test_rasterize_size(francais, outdir, caplog): +def test_rasterize_size(francais, outdir): path, pdf = francais page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3]) assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 @@ -86,7 +88,7 @@ def test_rasterize_rotated(francais, outdir, caplog): def test_gs_render_failure(resources, outpdf): - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'blank.pdf', outpdf, '--plugin', @@ -99,7 +101,7 @@ def test_gs_render_failure(resources, outpdf): def test_gs_raster_failure(resources, outpdf): - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / 'francais.pdf', outpdf, '--plugin', @@ -112,7 +114,7 @@ def test_gs_raster_failure(resources, outpdf): def test_ghostscript_pdfa_failure(resources, outpdf): - p, out, err = run_ocrmypdf( + p, _out, _err = run_ocrmypdf( resources / 'francais.pdf', outpdf, '--plugin', diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 47139bec..8b3f6748 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -18,7 +18,6 @@ import logging import multiprocessing import os -from pathlib import Path from unittest.mock import MagicMock import pytest diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index 1a1f817a..e8ed04bb 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -15,8 +15,6 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from pathlib import Path - import pytest from PIL import Image diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 558995fc..8bb2cc01 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -17,7 +17,6 @@ import pickle from math import isclose -from tempfile import NamedTemporaryFile import img2pdf import pikepdf @@ -26,7 +25,6 @@ from PIL import Image from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo -from ocrmypdf._exec import ghostscript from ocrmypdf.pdfinfo import Colorspace, Encoding # pylint: disable=protected-access @@ -110,7 +108,7 @@ def test_single_page_inline_image(outdir): assert pdfimage.width == 8 -def test_jpeg(resources, outdir): +def test_jpeg(resources): filename = resources / 'c02-22.pdf' pdf = pdfinfo.PdfInfo(filename) @@ -133,7 +131,7 @@ def test_no_contents(resources): pdf = pdfinfo.PdfInfo(filename) assert len(pdf[0].images) == 0 - assert pdf[0].has_text == False + assert not pdf[0].has_text def test_oversized_page(resources): diff --git a/tests/test_rotation.py b/tests/test_rotation.py index f0e7fcfd..2a8056ba 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -15,10 +15,8 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -import logging from io import BytesIO from os import fspath -from unittest.mock import Mock import img2pdf import pikepdf diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 478661c2..39a3fade 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -18,7 +18,7 @@ import os import sys from pathlib import Path -from subprocess import DEVNULL, PIPE, CalledProcessError, Popen, run +from subprocess import DEVNULL, PIPE, Popen, run import pytest @@ -95,7 +95,7 @@ def test_closed_streams(ocrmypdf_exec, resources, outpdf): stdin=None, preexec_fn=evil_closer, ) - out, err = p.communicate() + _out, err = p.communicate() print(err.decode()) assert p.returncode == ExitCode.ok @@ -121,7 +121,7 @@ def test_dev_null(resources): if 'COV_CORE_DATAFILE' in os.environ: pytest.skip(msg="Coverage uses stdout") - p, out, err = run_ocrmypdf( + p, out, _err = run_ocrmypdf( resources / 'trivial.pdf', os.devnull, '--force-ocr', diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 6e28235a..ca54eb34 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -22,7 +22,6 @@ import pytest from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._validation import check_options -from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import ExitCode, MissingDependencyError # pytest.helpers is dynamic diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 60f97d08..462c396b 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -22,9 +22,11 @@ import pytest from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfinfo import PdfInfo -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member +run_ocrmypdf = pytest.helpers.run_ocrmypdf # pylint: disable=no-member +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api # pylint: disable=no-member + +# pylint: disable=redefined-outer-name @pytest.fixture diff --git a/tests/test_validation.py b/tests/test_validation.py index 52dfb74b..d7955869 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -16,7 +16,6 @@ # along with OCRmyPDF. If not, see . import logging -import os from unittest.mock import patch import pikepdf @@ -35,7 +34,7 @@ def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwar kwargs['language'] = language parser = get_parser() pm = get_plugin_manager(kwargs.get('plugins', [])) - pm.hook.add_options(parser=parser) + pm.hook.add_options(parser=parser) # pylint: disable=no-member return ( create_options( input_file=input_file, output_file=output_file, parser=parser, **kwargs From f6257c21839e93bf7fee9ff06196fabe71980010 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 00:32:06 -0700 Subject: [PATCH 515/880] subprocess: lru_cache version checks --- src/ocrmypdf/subprocess.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index 96a0afb6..5cc540fa 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -105,6 +105,7 @@ def _fix_windows_args(program, args, env): return args +@lru_cache(maxsize=None) def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): """Get the version of the specified program""" args_prog = [program, version_arg] From a4e88eb8f083b285771cc203f21aa1bb0ff243ff Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 00:41:19 -0700 Subject: [PATCH 516/880] Simplify plugin_manager pickling --- src/ocrmypdf/_jobcontext.py | 12 ++---------- 1 file changed, 2 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 3091fbc8..3c91e856 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -18,7 +18,6 @@ import os import shutil import sys -from functools import partial from pathlib import Path from ocrmypdf._plugin_manager import get_plugin_manager @@ -72,19 +71,12 @@ class PageContext: def __getstate__(self): state = self.__dict__.copy() if state['plugin_manager'] is not None: - del state['plugin_manager'] - state['construct_plugin_manager'] = partial( - get_plugin_manager, self.options.plugins - ) + state['plugin_manager'] = None return state def __setstate__(self, state): self.__dict__.update(state) - if 'construct_plugin_manager' in state: - self.plugin_manager = state['construct_plugin_manager']() - else: - self.plugin_manager = None - del self.__dict__['construct_plugin_manager'] + self.plugin_manager = get_plugin_manager(self.options.plugins) def cleanup_working_files(work_folder, options): From b6eebadf054b53480ff1f32ff92b6721683576d2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 11:32:46 -0700 Subject: [PATCH 517/880] Use pikepdf.open with block to manage PdfInfo --- src/ocrmypdf/pdfinfo/info.py | 26 +++++++------------------- 1 file changed, 7 insertions(+), 19 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index fa2dcc05..5c8175f3 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -654,19 +654,6 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): return pages -def _pdf_get_all_pageinfo(infile, progbar=False, max_workers=None): - pdf = pikepdf.open(infile) # Do not close in this function - try: - if pdf.is_encrypted: - raise EncryptedPdfError() # Triggered by encryption with empty passwd - pages = _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers) - except Exception: - pdf.close() - raise - - return pages, pdf - - class PageInfo: def __init__(self, pdf, pageno, infile): self._pageno = pageno @@ -770,9 +757,11 @@ class PdfInfo: def __init__(self, infile, progbar=False, max_workers=None): self._infile = infile - self._pages, pdf = _pdf_get_all_pageinfo( - infile, progbar=progbar, max_workers=max_workers - ) + + with pikepdf.open(infile) as pdf: + if pdf.is_encrypted: + raise EncryptedPdfError() # Triggered by encryption with empty passwd + self._pages = _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False if '/AcroForm' in pdf.root: @@ -780,7 +769,6 @@ class PdfInfo: self._has_acroform = True elif '/XFA' in pdf.root.AcroForm: self._has_acroform = True - pdf.close() @property def pages(self): @@ -826,10 +814,10 @@ def main(): parser = argparse.ArgumentParser() parser.add_argument('infile') args = parser.parse_args() - pagesinfo, pdfinfo = _pdf_get_all_pageinfo(args.infile) + pdfinfo = PdfInfo(args.infile) pprint(pdfinfo) - for page in pagesinfo: + for page in pdfinfo.pages: pprint(page) for im in page.images: pprint(im) From 8599400445d160c97bdb00ee1ae7d69ff4d0f5a2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 11:33:27 -0700 Subject: [PATCH 518/880] Only do page analysis on pages we will do OCR on --- src/ocrmypdf/_pipeline.py | 9 +++-- src/ocrmypdf/_sync.py | 1 + src/ocrmypdf/pdfinfo/info.py | 69 +++++++++++++++++++++--------------- 3 files changed, 49 insertions(+), 30 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 689d2542..3789fdca 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -149,9 +149,14 @@ def triage(original_filename, input_file, output_file, options): return output_file -def get_pdfinfo(input_file, progbar=False, max_workers=None): +def get_pdfinfo(input_file, progbar=False, max_workers=None, check_pages=None): try: - return PdfInfo(input_file, progbar=progbar, max_workers=max_workers) + return PdfInfo( + input_file, + progbar=progbar, + max_workers=max_workers, + check_pages=check_pages, + ) except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 60eed5de..931cd320 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -309,6 +309,7 @@ def run_pipeline(options, *, plugin_manager, api=False): origin_pdf, progbar=options.progress_bar, max_workers=options.jobs if not options.use_threads else 1, # To help debug + check_pages=options.pages, ) context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 5c8175f3..4ef35d3c 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -554,7 +554,7 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) -def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike): +def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, check_pages): pageinfo = {} pageinfo['pageno'] = pageno pageinfo['images'] = [] @@ -564,12 +564,18 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike): width_pt = mediabox[2] - mediabox[0] height_pt = mediabox[3] - mediabox[1] - pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') - miner = get_page_analysis(infile, pageno, pscript5_mode) - pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) - bboxes = (box.bbox for box in pageinfo['textboxes']) + check_this_page = not check_pages or pageno in check_pages - pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) + if check_this_page: + pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') + miner = get_page_analysis(infile, pageno, pscript5_mode) + pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) + bboxes = (box.bbox for box in pageinfo['textboxes']) + + pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) + else: + pageinfo['textboxes'] = [] + pageinfo['has_text'] = None userunit = page.get('/UserUnit', Decimal(1.0)) if not isinstance(userunit, Decimal): @@ -584,12 +590,16 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike): pageinfo['rotate'] = 0 userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) - contentsinfo = [ - ci - for ci in _process_content_streams( - pdf=pdf, container=page, shorthand=userunit_shorthand - ) - ] + + if check_this_page: + contentsinfo = [ + ci + for ci in _process_content_streams( + pdf=pdf, container=page, shorthand=userunit_shorthand + ) + ] + else: + contentsinfo = [] pageinfo['has_vector'] = False if any(isinstance(ci, VectorInfo) for ci in contentsinfo): @@ -615,12 +625,12 @@ def _pdf_pageinfo_sync_init(infile): def _pdf_pageinfo_sync(args): global worker_pdf # pylint: disable=global-statement - pageno, infile = args - page = PageInfo(worker_pdf, pageno, infile) + pageno, infile, check_pages = args + page = PageInfo(worker_pdf, pageno, infile, check_pages) return page -def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): +def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers, check_pages): pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -631,7 +641,8 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): if max_workers is None: max_workers = available_cpu_count() - contexts = ((n, infile) for n in range(len(pdf.pages))) + total = len(pdf.pages) + contexts = ((n, infile, check_pages) for n in range(total)) use_threads = False # No performance gain if threaded due to GIL n_workers = min(1 + len(pages) // 4, max_workers) @@ -644,7 +655,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): use_threads=use_threads, max_workers=n_workers, tqdm_kwargs=dict( - total=len(pdf.pages), desc="Scan", unit='page', disable=not progbar + total=total, desc="Searching for text", unit='page', disable=not progbar ), task_initializer=partial(_pdf_pageinfo_sync_init, infile), task=_pdf_pageinfo_sync, @@ -655,10 +666,10 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers): class PageInfo: - def __init__(self, pdf, pageno, infile): + def __init__(self, pdf, pageno, infile, check_pages): self._pageno = pageno self._infile = infile - self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile) + self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, check_pages) @property def pageno(self): @@ -755,20 +766,22 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, progbar=False, max_workers=None): + def __init__(self, infile, progbar=False, max_workers=None, check_pages=None): self._infile = infile with pikepdf.open(infile) as pdf: if pdf.is_encrypted: raise EncryptedPdfError() # Triggered by encryption with empty passwd - self._pages = _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers) - self._needs_rendering = pdf.root.get('/NeedsRendering', False) - self._has_acroform = False - if '/AcroForm' in pdf.root: - if len(pdf.root.AcroForm.get('/Fields', [])) > 0: - self._has_acroform = True - elif '/XFA' in pdf.root.AcroForm: - self._has_acroform = True + self._pages = _pdf_pageinfo_concurrent( + pdf, infile, progbar, max_workers, check_pages=check_pages + ) + self._needs_rendering = pdf.root.get('/NeedsRendering', False) + self._has_acroform = False + if '/AcroForm' in pdf.root: + if len(pdf.root.AcroForm.get('/Fields', [])) > 0: + self._has_acroform = True + elif '/XFA' in pdf.root.AcroForm: + self._has_acroform = True @property def pages(self): From 872bafad4b3bc50c2dcd40b82b4098ad0acd6296 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 11:53:04 -0700 Subject: [PATCH 519/880] Reinstate quick test for text/no text Partial revert of commit 991db17 --- src/ocrmypdf/_pipeline.py | 9 ++++- src/ocrmypdf/_sync.py | 1 + src/ocrmypdf/pdfinfo/info.py | 72 +++++++++++++++++++++++++++--------- tests/test_pdfinfo.py | 2 +- 4 files changed, 64 insertions(+), 20 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 3789fdca..8c27127f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -149,10 +149,17 @@ def triage(original_filename, input_file, output_file, options): return output_file -def get_pdfinfo(input_file, progbar=False, max_workers=None, check_pages=None): +def get_pdfinfo( + input_file, + detailed_analysis=False, + progbar=False, + max_workers=None, + check_pages=None, +): try: return PdfInfo( input_file, + detailed_analysis=detailed_analysis, progbar=progbar, max_workers=max_workers, check_pages=check_pages, diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 931cd320..e83a0231 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -307,6 +307,7 @@ def run_pipeline(options, *, plugin_manager, api=False): # Gather pdfinfo and create context pdfinfo = get_pdfinfo( origin_pdf, + detailed_analysis=options.redo_ocr, progbar=options.progress_bar, max_workers=options.jobs if not options.use_threads else 1, # To help debug check_pages=options.pages, diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 4ef35d3c..c3061533 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -98,15 +98,19 @@ XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_dep InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth']) ContentsInfo = namedtuple( - 'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector', 'name_index'] + 'ContentsInfo', + ['xobject_settings', 'inline_images', 'found_vector', 'found_text', 'name_index'], ) TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt']) -class VectorInfo: - def __init__(self): - pass +class VectorMarker: + pass + + +class TextMarker: + pass def _normalize_stack(graphobjs): @@ -153,9 +157,11 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): inline_images = [] name_index = defaultdict(lambda: []) found_vector = False + found_text = False vector_ops = set('S s f F f* B B* b b*'.split()) + text_showing_ops = set("""TJ Tj " '""".split()) image_ops = set('BI ID EI q Q Do cm'.split()) - operator_whitelist = ' '.join(vector_ops | image_ops) + operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops) for n, graphobj in enumerate( _normalize_stack( @@ -195,11 +201,14 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): inline_images.append(inline) elif operator in vector_ops: found_vector = True + elif operator in text_showing_ops: + found_text = True return ContentsInfo( xobject_settings=xobject_settings, inline_images=inline_images, found_vector=found_vector, + found_text=found_text, name_index=name_index, ) @@ -505,7 +514,9 @@ def _process_content_streams(*, pdf, container, shorthand=None): contentsinfo = _interpret_contents(container, initial_shorthand) if contentsinfo.found_vector: - yield VectorInfo() + yield VectorMarker() + if contentsinfo.found_text: + yield TextMarker() yield from _find_inline_images(contentsinfo) yield from _find_regular_images(container, contentsinfo) yield from _find_form_xobject_images(pdf, container, contentsinfo) @@ -554,7 +565,9 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) -def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, check_pages): +def _pdf_get_pageinfo( + pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis +): pageinfo = {} pageinfo['pageno'] = pageno pageinfo['images'] = [] @@ -564,9 +577,9 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, check_pages): width_pt = mediabox[2] - mediabox[0] height_pt = mediabox[3] - mediabox[1] - check_this_page = not check_pages or pageno in check_pages + check_this_page = pageno in check_pages - if check_this_page: + if check_this_page and detailed_analysis: pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') miner = get_page_analysis(infile, pageno, pscript5_mode) pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) @@ -602,8 +615,10 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, check_pages): contentsinfo = [] pageinfo['has_vector'] = False - if any(isinstance(ci, VectorInfo) for ci in contentsinfo): + if any(isinstance(ci, VectorMarker) for ci in contentsinfo): pageinfo['has_vector'] = True + if any(isinstance(ci, TextMarker) for ci in contentsinfo): + pageinfo['has_text'] = True pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] if pageinfo['images']: @@ -625,12 +640,14 @@ def _pdf_pageinfo_sync_init(infile): def _pdf_pageinfo_sync(args): global worker_pdf # pylint: disable=global-statement - pageno, infile, check_pages = args - page = PageInfo(worker_pdf, pageno, infile, check_pages) + pageno, infile, check_pages, detailed_analysis = args + page = PageInfo(worker_pdf, pageno, infile, check_pages, detailed_analysis) return page -def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers, check_pages): +def _pdf_pageinfo_concurrent( + pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False +): pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -642,7 +659,7 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers, check_pages): max_workers = available_cpu_count() total = len(pdf.pages) - contexts = ((n, infile, check_pages) for n in range(total)) + contexts = ((n, infile, check_pages, detailed_analysis) for n in range(total)) use_threads = False # No performance gain if threaded due to GIL n_workers = min(1 + len(pages) // 4, max_workers) @@ -666,10 +683,13 @@ def _pdf_pageinfo_concurrent(pdf, infile, progbar, max_workers, check_pages): class PageInfo: - def __init__(self, pdf, pageno, infile, check_pages): + def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False): self._pageno = pageno self._infile = infile - self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, check_pages) + self._detailed_analysis = detailed_analysis + self._pageinfo = _pdf_get_pageinfo( + pdf, pageno, infile, check_pages, detailed_analysis + ) @property def pageno(self): @@ -681,6 +701,8 @@ class PageInfo: @property def has_corrupt_text(self): + if not self._detailed_analysis: + raise NotImplementedError('Did not do detailed analysis') return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) @property @@ -766,14 +788,28 @@ class PageInfo: class PdfInfo: """Get summary information about a PDF""" - def __init__(self, infile, progbar=False, max_workers=None, check_pages=None): + def __init__( + self, + infile, + detailed_analysis=False, + progbar=False, + max_workers=None, + check_pages=None, + ): self._infile = infile + if check_pages is None: + check_pages = range(0, 1_000_000_000) with pikepdf.open(infile) as pdf: if pdf.is_encrypted: raise EncryptedPdfError() # Triggered by encryption with empty passwd self._pages = _pdf_pageinfo_concurrent( - pdf, infile, progbar, max_workers, check_pages=check_pages + pdf, + infile, + progbar, + max_workers, + check_pages=check_pages, + detailed_analysis=detailed_analysis, ) self._needs_rendering = pdf.root.get('/NeedsRendering', False) self._has_acroform = False diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 8bb2cc01..a7937e4b 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -168,7 +168,7 @@ def test_ocr_detection(resources): ) def test_corrupt_font_detection(resources, testfile): filename = resources / testfile - pdf = pdfinfo.PdfInfo(filename) + pdf = pdfinfo.PdfInfo(filename, detailed_analysis=True) assert pdf[0].has_corrupt_text From f59a757e8b2752fa99e607225051e45edaf7d308 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 12:09:24 -0700 Subject: [PATCH 520/880] info: tidy handling of content streams --- src/ocrmypdf/pdfinfo/info.py | 33 ++++++++++++++++++--------------- 1 file changed, 18 insertions(+), 15 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index c3061533..6a3c4b6a 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -588,7 +588,7 @@ def _pdf_get_pageinfo( pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) else: pageinfo['textboxes'] = [] - pageinfo['has_text'] = None + pageinfo['has_text'] = None # i.e. "no information" userunit = page.get('/UserUnit', Decimal(1.0)) if not isinstance(userunit, Decimal): @@ -605,22 +605,25 @@ def _pdf_get_pageinfo( userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) if check_this_page: - contentsinfo = [ - ci - for ci in _process_content_streams( - pdf=pdf, container=page, shorthand=userunit_shorthand - ) - ] + pageinfo['has_vector'] = False + pageinfo['has_text'] = False + pageinfo['images'] = [] + for ci in _process_content_streams( + pdf=pdf, container=page, shorthand=userunit_shorthand + ): + if isinstance(ci, VectorMarker): + pageinfo['has_vector'] = True + elif isinstance(ci, TextMarker): + pageinfo['has_text'] = True + elif isinstance(ci, ImageInfo): + pageinfo['images'].append(ci) + else: + raise NotImplementedError() else: - contentsinfo = [] + pageinfo['has_vector'] = None # i.e. "no information" + pageinfo['has_text'] = None + pageinfo['images'] = None - pageinfo['has_vector'] = False - if any(isinstance(ci, VectorMarker) for ci in contentsinfo): - pageinfo['has_vector'] = True - if any(isinstance(ci, TextMarker) for ci in contentsinfo): - pageinfo['has_text'] = True - - pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)] if pageinfo['images']: dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images']) pageinfo['dpi'] = dpi From 7caf1e85ff430dbe6fb45fd08e399b179b2cbdfc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 12:11:37 -0700 Subject: [PATCH 521/880] info: change "Scan" message --- src/ocrmypdf/pdfinfo/info.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 6a3c4b6a..1800f1e4 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -675,7 +675,7 @@ def _pdf_pageinfo_concurrent( use_threads=use_threads, max_workers=n_workers, tqdm_kwargs=dict( - total=total, desc="Searching for text", unit='page', disable=not progbar + total=total, desc="Scanning contents", unit='page', disable=not progbar ), task_initializer=partial(_pdf_pageinfo_sync_init, infile), task=_pdf_pageinfo_sync, From 17a4831745a31151db130f47a6cd792c1e0a42b1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 14:27:47 -0700 Subject: [PATCH 522/880] v10 release notes and dependencies --- docs/release_notes.rst | 40 +++++++++++++++++++++++++++------------- requirements/main.txt | 3 ++- requirements/test.txt | 2 +- setup.py | 5 +++-- 4 files changed, 33 insertions(+), 17 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f494063f..bc7c5928 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,32 +13,46 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. -v10.0.0 (not yet released) -========================== +v10.0.0 +======= **Breaking changes** - Support for pdfminer.six version 20181108 has been dropped, along with a monkeypatch that made this version work. -- Ghostscript is no longer used for finding the location of text in PDFs, and - APIs related to this feature have been removed. - Output messages are now displayed in color (when supported by the terminal) and prefixes describing the severity of the message are removed. As such programs that parse OCRmyPDF's log message will need to be revised. (Please consider using OCRmyPDF as a library instead.) +- The minimum version for certain dependencies has increased. +- Many API changes; see developer changes. +- The Python libraries pluggy and coloredlogs are now required. + +**New features and improvements** + +- PDF page scanning is now parallelized across CPUs, speeding up this phase + for files with a high page count. +- PDF page scanning is optimized, addressing some performance regressions. +- A plugin architecture has been added, currently allowing one to more easily + use a different OCR engine or PDF renderer from Tesseract and Ghostscript, + respectively. A plugin can also override some decisions, such changing + the OCR settings after initial scanning. +- Colored log messages. + +**Developer changes** + +- The test spoofing mechanism, used to correct handling of failures in + Tesseract and Ghostscript, has been removed in favor of using plugins for + testing. The spoofing mechanism was fairly complex and required many special + hacks for Windows. - Code describing the resolution in DPI of images was refactored into a ``ocrmypdf.helpers.Resolution`` class. -- A deprecated parameter in ``ocrmypdf.exec.ghostscript.generate_pdfa`` was - removed. -- The deprecated module ``ocrmypdf.exec.qpdf`` was removed. +- The module ``ocrmypdf._exec`` is now private to OCRmyPDF. - The ``ocrmypdf.hocrtransform`` module has been updated to follow PEP8 naming conventions. - -**New features** - -- PDF page scanning is now parallelized across CPUs, speeding up the "Scan" - phase for files with a high page count. -- Colored log messages. +- Ghostscript is no longer used for finding the location of text in PDFs, and + APIs related to this feature have been removed. +- Lots of internal reorganization to support plugins. v9.8.2 ====== diff --git a/requirements/main.txt b/requirements/main.txt index 7dc37803..24210fed 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,8 @@ cffi == 1.14.0 coloredlogs == 14.0 # technically optional img2pdf == 0.3.4 pdfminer.six == 20200517 -pikepdf == 1.11.1 +pikepdf == 1.14.0 +pluggy == 0.13.1 Pillow == 7.1.1 reportlab == 3.5.34 tqdm == 4.45.0 diff --git a/requirements/test.txt b/requirements/test.txt index aeda7a7c..531bea96 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,7 +1,7 @@ pytest >= 5.0.0 pytest-helpers-namespace >= 2019.1.8 pytest-xdist >= 1.31.0 -pytest-cov >= 2.8.0 +pytest-cov >= 2.9.0 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi #PyMuPDF == 1.13.4 # optional diff --git a/setup.py b/setup.py index 9dd07aff..5fd923f4 100644 --- a/setup.py +++ b/setup.py @@ -83,8 +83,9 @@ setup( 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six >= 20191110, <= 20200517', - 'pikepdf >= 1.8.1, < 2', - 'Pillow >= 6.2.0', + 'pikepdf >= 1.14.0, < 2', + 'Pillow >= 7.0.0', + 'pluggy >= 0.13.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling 'tqdm >= 4', ], From c6b9a49cbb2188e8036e3ba01229459de41b1e57 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 10 Jun 2020 17:08:00 -0700 Subject: [PATCH 523/880] Fix tests that fail in CI --- tests/test_validation.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index d7955869..b905b594 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -49,16 +49,19 @@ def make_opts(*args, **kwargs): def test_hocr_notlatin_warning(caplog): - vd.check_options( - *make_opts_pm(language='chi_sim', pdf_renderer='hocr', output_type='pdfa') - ) + # Bypass the test to see if the language is installed; we just want to pretend + # that a non-Latin language is installed + with patch('ocrmypdf._validation.check_options_languages', return_value=None): + vd.check_options( + *make_opts_pm(language='chi_sim', pdf_renderer='hocr', output_type='pdfa') + ) assert 'PDF renderer is known to cause' in caplog.text def test_old_ghostscript(caplog): with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'), patch( 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True - ): + ), patch('ocrmypdf._validation.check_options_languages', return_value=None): vd.check_options(*make_opts_pm(language='chi_sim', output_type='pdfa')) assert 'Ghostscript does not work correctly' in caplog.text From 393c5a9ea4240593a2c6a90ed10a039588109a5f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 12 Jun 2020 12:09:46 -0700 Subject: [PATCH 524/880] Fix error on -l lang1+lang2 --- src/ocrmypdf/cli.py | 2 +- tests/test_main.py | 5 +++++ 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index a34a2108..fd76f7cc 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -63,7 +63,7 @@ class LanguageSetAction(argparse.Action): def __call__(self, parser, namespace, values, option_string=None): dest = getattr(namespace, self.dest) if '+' in values: - dest.add(lang for lang in values.split('+')) + dest.update(lang for lang in values.split('+')) else: dest.add(values) diff --git a/tests/test_main.py b/tests/test_main.py index 289cbbcc..5df4a703 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -875,3 +875,8 @@ def test_image_dpi_not_image(caplog, resources, outpdf): 'tests/plugins/tesseract_noop.py', ) assert '--image-dpi is being ignored' in caplog.text + + +def test_two_languages(resources, no_outpdf): + p, _, _ = run_ocrmypdf(resources / 'graph_ocred.pdf', no_outpdf, '-l', 'eng+deu') + assert p.returncode == ExitCode.already_done_ocr From 863835f66060c933cfb7b7922d650d9128531342 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 12 Jun 2020 12:11:13 -0700 Subject: [PATCH 525/880] v10.0.1 release notes --- docs/release_notes.rst | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index bc7c5928..f44ca427 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,11 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.0.1 +======= + +- Fix regression when ``-l lang1+lang2`` is used from command line. + v10.0.0 ======= From eeb44f78ccaa6eaa0d38295bf2048dbb96c85c09 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 12 Jun 2020 12:59:46 -0700 Subject: [PATCH 526/880] Fix tests that failed on other platforms from previous fix --- src/ocrmypdf/_validation.py | 21 +++++++++++++-------- tests/test_validation.py | 32 +++++++++++++++++--------------- 2 files changed, 30 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 2d688afa..06a16670 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -67,20 +67,20 @@ def check_platform(): ) -def check_options_languages(options, plugin_manager): +def check_options_languages(options, ocr_engine_languages): if not options.languages: options.languages = {DEFAULT_LANGUAGE} system_lang = locale.getlocale()[0] if system_lang and not system_lang.startswith('en'): log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE) - - ocr_engine = plugin_manager.hook.get_ocr_engine() - if not options.languages.issubset(ocr_engine.languages(options)): + if not ocr_engine_languages: + return + if not options.languages.issubset(ocr_engine_languages): msg = ( - f"{ocr_engine} does not have language data for the following " + f"OCR engine does not have language data for the following " "requested languages: \n" ) - for lang in options.languages - ocr_engine.languages(options): + for lang in options.languages - ocr_engine_languages: msg += lang + '\n' raise MissingDependencyError(msg) @@ -251,9 +251,9 @@ def check_options_pillow(options): PIL.Image.MAX_IMAGE_PIXELS = None -def check_options(options, plugin_manager): +def _check_options(options, plugin_manager, ocr_engine_languages): check_platform() - check_options_languages(options, plugin_manager) + check_options_languages(options, ocr_engine_languages) check_options_metadata(options) check_options_output(options) check_options_sidecar(options) @@ -265,6 +265,11 @@ def check_options(options, plugin_manager): plugin_manager.hook.check_options(options=options) +def check_options(options, plugin_manager): + ocr_engine_languages = plugin_manager.hook.get_ocr_engine().languages(options) + _check_options(options, plugin_manager, ocr_engine_languages) + + def check_closed_streams(options): # pragma: no cover """Work around Python issue with multiprocessing forking on closed streams diff --git a/tests/test_validation.py b/tests/test_validation.py index b905b594..99d29792 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -51,31 +51,33 @@ def make_opts(*args, **kwargs): def test_hocr_notlatin_warning(caplog): # Bypass the test to see if the language is installed; we just want to pretend # that a non-Latin language is installed - with patch('ocrmypdf._validation.check_options_languages', return_value=None): - vd.check_options( - *make_opts_pm(language='chi_sim', pdf_renderer='hocr', output_type='pdfa') - ) + vd._check_options( + *make_opts_pm(language='chi_sim', pdf_renderer='hocr', output_type='pdfa'), + {'chi_sim'}, + ) assert 'PDF renderer is known to cause' in caplog.text def test_old_ghostscript(caplog): with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'), patch( 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True - ), patch('ocrmypdf._validation.check_options_languages', return_value=None): - vd.check_options(*make_opts_pm(language='chi_sim', output_type='pdfa')) + ): + vd._check_options( + *make_opts_pm(language='chi_sim', output_type='pdfa'), {'chi_sim'} + ) assert 'Ghostscript does not work correctly' in caplog.text with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'), patch( 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): - vd.check_options(*make_opts_pm(output_type='pdfa-3')) + vd._check_options(*make_opts_pm(output_type='pdfa-3'), set()) with patch('ocrmypdf._exec.ghostscript.version', return_value='9.24'), patch( 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True ): with pytest.raises(MissingDependencyError): - vd.check_options(*make_opts_pm()) + vd._check_options(*make_opts_pm(), set()) def test_old_tesseract_error(): @@ -83,7 +85,7 @@ def test_old_tesseract_error(): with pytest.raises(MissingDependencyError): opts = make_opts(pdf_renderer='sandwich', language='eng') plugin_manager = get_plugin_manager(opts.plugins) - vd.check_options(opts, plugin_manager) + vd._check_options(opts, plugin_manager, {'eng'}) def test_lossless_redo(): @@ -113,13 +115,13 @@ def test_user_words(caplog): with patch('ocrmypdf._exec.tesseract.has_user_words', return_value=False): opts = make_opts(user_words='foo') plugin_manager = get_plugin_manager(opts.plugins) - vd.check_options(opts, plugin_manager) + vd._check_options(opts, plugin_manager, set()) assert '4.0 ignores --user-words' in caplog.text caplog.clear() with patch('ocrmypdf._exec.tesseract.has_user_words', return_value=True): opts = make_opts(user_patterns='foo') plugin_manager = get_plugin_manager(opts.plugins) - vd.check_options(opts, plugin_manager) + vd._check_options(opts, plugin_manager, set()) assert '4.0 ignores --user-words' not in caplog.text @@ -182,7 +184,7 @@ def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) plugin_manager = get_plugin_manager(opts.plugins) with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: - vd.check_options(opts, plugin_manager) + vd._check_options(opts, plugin_manager, set()) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) assert pdfinfo is not None assert tqdmpatch.called @@ -197,7 +199,7 @@ def test_language_warning(caplog): with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') ): - vd.check_options_languages(opts, plugin_manager) + vd.check_options_languages(opts, {'eng'}) assert opts.languages == {'eng'} assert '' in caplog.text @@ -205,7 +207,7 @@ def test_language_warning(caplog): with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') ): - vd.check_options_languages(opts, plugin_manager) + vd.check_options_languages(opts, {'eng'}) assert opts.languages == {'eng'} assert 'assuming --language' in caplog.text @@ -268,5 +270,5 @@ def test_optional_program_recommended(caplog): def test_pagesegmode_warning(caplog): opts = make_opts(tesseract_pagesegmode='0') plugin_manager = get_plugin_manager(opts.plugins) - vd.check_options(opts, plugin_manager) + vd._check_options(opts, plugin_manager, set()) assert 'disable OCR' in caplog.text From 892db88f0eacf39cde12fe2084e5f1ce7de8c11a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 12 Jun 2020 14:33:02 -0700 Subject: [PATCH 527/880] test_two_languages: use narrower test --- tests/test_main.py | 5 ----- tests/test_validation.py | 7 +++++++ 2 files changed, 7 insertions(+), 5 deletions(-) diff --git a/tests/test_main.py b/tests/test_main.py index 5df4a703..289cbbcc 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -875,8 +875,3 @@ def test_image_dpi_not_image(caplog, resources, outpdf): 'tests/plugins/tesseract_noop.py', ) assert '--image-dpi is being ignored' in caplog.text - - -def test_two_languages(resources, no_outpdf): - p, _, _ = run_ocrmypdf(resources / 'graph_ocred.pdf', no_outpdf, '-l', 'eng+deu') - assert p.returncode == ExitCode.already_done_ocr diff --git a/tests/test_validation.py b/tests/test_validation.py index 99d29792..bd9fe098 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -272,3 +272,10 @@ def test_pagesegmode_warning(caplog): plugin_manager = get_plugin_manager(opts.plugins) vd._check_options(opts, plugin_manager, set()) assert 'disable OCR' in caplog.text + + +def test_two_languages(): + with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True): + vd._check_options( + *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} + ) From 862861e3ca59e92110db66e2f63de90714008833 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 13 Jun 2020 14:50:58 -0700 Subject: [PATCH 528/880] Fix error message in logging from repeated filtering If logging somehow triggers PageNumberFilter multiple times, it would fail on the second occurrence. --- src/ocrmypdf/_logging.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_logging.py b/src/ocrmypdf/_logging.py index 5126d97c..af28742d 100644 --- a/src/ocrmypdf/_logging.py +++ b/src/ocrmypdf/_logging.py @@ -25,9 +25,9 @@ from tqdm import tqdm class PageNumberFilter(logging.Filter): def filter(self, record): pageno = getattr(record, 'pageno', None) - if pageno is not None: + if isinstance(pageno, int): record.pageno = f'{pageno:5d} ' - else: + elif pageno is None: record.pageno = '' return True From 2d2a4894ab62ddedc25bd5c87ac9bb959a221460 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 12:51:28 -0700 Subject: [PATCH 529/880] Some corrections to release notes --- docs/release_notes.rst | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f44ca427..c8dee0d9 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -5,9 +5,12 @@ Release notes OCRmyPDF uses `semantic versioning `__ for its command line interface and its public API. -The ``ocrmypdf`` package may now be imported. The public API may be -useful in scripts that launch OCRmyPDF processes or that wish to use -some of its features for working with PDFs. +OCRmyPDF's output messages are not considered part of the stable interface - +that is, output messages may be improved at any release level, so parsing them +may be unreliable. Use the API to depend on precise behavior. + +The public API may be useful in scripts that launch OCRmyPDF processes or that +wish to use some of its features for working with PDFs. Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be @@ -36,8 +39,12 @@ v10.0.0 **New features and improvements** - PDF page scanning is now parallelized across CPUs, speeding up this phase - for files with a high page count. + dramatically for files with a high page counts. - PDF page scanning is optimized, addressing some performance regressions. +- PDF page scanning is no longer run on pages that are not selected when the + ``--pages`` argument is used. +- PDF page scanning is now independent of Ghostscript, ending our past reliance + on this occasionally unstable feature in Ghostscript. - A plugin architecture has been added, currently allowing one to more easily use a different OCR engine or PDF renderer from Tesseract and Ghostscript, respectively. A plugin can also override some decisions, such changing @@ -46,7 +53,7 @@ v10.0.0 **Developer changes** -- The test spoofing mechanism, used to correct handling of failures in +- The test spoofing mechanism, used to test correct handling of failures in Tesseract and Ghostscript, has been removed in favor of using plugins for testing. The spoofing mechanism was fairly complex and required many special hacks for Windows. From 9d127d354c78fa97910224068ea4fcac58e91e0c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 12:51:49 -0700 Subject: [PATCH 530/880] docs: improve description of plugins --- docs/plugins.rst | 2 +- src/ocrmypdf/pluginspec.py | 125 ++++++++++++++++++++++++++----------- 2 files changed, 90 insertions(+), 37 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index f2ac0b94..9dd145c2 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -70,7 +70,7 @@ A plugin may provide the following hooks. Hooks should be decorated with from ocrmpydf import hookimpl @hookimpl - def prepare(options): + def add_options(parser): pass The following is a complete list of hooks that may be installed and when diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index ed6d0eda..3bab37e1 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -33,19 +33,37 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') @hookspec def add_options(parser: ArgumentParser) -> None: - """Allows the plugin to add its own command line arguments. + """Allows the plugin to add its own command line and API arguments. - Even if you do not intend to use plugins in a command line context, you - should use this function to create your options. + OCRmyPDF converts command line arguments to API arguments, so adding + arguments here will cause new arguments to be processed for API calls + to ``ocrmypdf.ocr``, or when invoked on the command line. + + Note: + This hook will be called from the main process, and may modify global state + before child worker processes are forked. """ @hookspec def check_options(options: Namespace) -> None: - """Called to ask the plugin to check all of its options. + """Called to ask the plugin to check all of the options. - The plugin may modify the *options*. All objects that are in options must - be picklable so they can be marshalled to child worker processes. + The plugin may check if options that it added are valid. + + Warnings or other messages may be passed to the user by creating a logger + object using ``log = logging.getLogger(__name__)`` and logging to this. + + The plugin may also modify the *options*. All objects that are in options + must be picklable so they can be marshalled to child worker processes. + + Raises: + ocrmypdf.exceptions.ExitCodeException: If options are not acceptable + and the application should terminate gracefully with an informative + message and error code. + Note: + This hook will be called from the main process, and may modify global state + before child worker processes are forked. """ @@ -59,12 +77,13 @@ def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: that a certain type of file should be treated with ``options.force_ocr = True`` based on information in its *pdfinfo*. - The plugin may raise :class:`ocrmypdf.exceptions.InputFileError` or any - :class:`ocrmypdf.exceptions.ExitCodeException` to request - normal termination. ocrmypdf will hold the plugin responsible for raising - exceptions of any other type. - - The return value is ignored. To abort processing, raise an ``ExitCodeException``. + Raises: + ocrmypdf.exceptions.ExitCodeException: If options or pdfinfo are not acceptable + and the application should terminate gracefully with an informative + message and error code. + Note: + This hook will be called from the main process, and may modify global state + before child worker processes are forked. """ @@ -86,14 +105,19 @@ def rasterize_pdf_page( be overridden with the values in page_dpi. Args: - raster_device: type of image to produce at output_file - raster_dpi: resolution at which to rasterize page - pageno: page number to rasterize (beginning at page 1) - page_dpi: resolution, overriding output image DPI - rotation: cardinal angle, clockwise, to rotate page - filter_vector: if True, remove vector graphics objects + input_file: The PDF to rasterize. + output_file: The desired name of the rasterized image. + raster_device: Type of image to produce at output_file + raster_dpi: Resolution at which to rasterize page + pageno: Page number to rasterize (beginning at page 1) + page_dpi: Resolution, overriding output image DPI + rotation: Cardinal angle, clockwise, to rotate page + filter_vector: If True, remove vector graphics objects Returns: output_file + Note: + This hook will be called from child processes. Modifying global state + will not affect the main process or other child processes. """ @@ -103,6 +127,10 @@ def filter_ocr_image(page: 'PageContext', image: Image) -> Image: This is the image that OCR sees, not what the user sees when they view the PDF. + + Note: + This hook will be called from child processes. Modifying global state + will not affect the main process or other child processes. """ @@ -119,6 +147,10 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: convert the image to a JPEG, the output page will be created as a JPEG, etc. Note that the ocrmypdf image optimization stage may ultimately chose a different format. + + Note: + This hook will be called from child processes. Modifying global state + will not affect the main process or other child processes. """ @@ -140,7 +172,10 @@ class OcrEngine(ABC): @abstractstaticmethod def languages(options: Namespace) -> AbstractSet[str]: - """Returns set of languages that are supported.""" + """Returns the set of all languages that are supported by the engine. + + Languages are typically given in 3-letter ISO 3166-1 codes, but actually + can be any value understood by the OCR engine.""" @abstractstaticmethod def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence: @@ -150,18 +185,31 @@ class OcrEngine(ABC): def generate_hocr( input_file: Path, output_hocr: Path, output_text: Path, options: Namespace ) -> None: - """Called to produce a hOCR file.""" + """Called to produce a hOCR file and sidecar text file.""" @abstractstaticmethod def generate_pdf( input_file: Path, output_pdf: Path, output_text: Path, options: Namespace ) -> None: - """Called to produce a text only PDF (no image, invisible text).""" + """Called to produce a text only PDF. + + Args: + input_file: A page image on which to perform OCR. + output_pdf: The expected name of the output PDF, which must be + a single page PDF with no visible content of any kind, sized + to the dimensions implied by the input_file's width, height + and DPI. The image will be grafted onto the input PDF page. + """ @hookspec(firstresult=True) def get_ocr_engine() -> OcrEngine: - pass + """Returns an OcrEngine to use for processing this file. + + The OcrEngine may be instantiated multiple times, by both the main process + and child process. As such, it must be obtain store any state in ``options`` + or some common location. + """ @hookspec(firstresult=True) @@ -175,21 +223,26 @@ def generate_pdfa( ) -> Path: """Generate a PDF/A. - The pdf_pages, a list of files, will be merged into output_file. One or more - PDF files may be merged. The pdfmark file is a PostScript.ps file that - provides Ghostscript with details on how to perform the PDF/A - conversion. By default with we pick PDF/A-2b, but this works for 1 or 3. + This API strongly assumes a PDF/A generator with Ghostscript's semantics. - compression can be 'jpeg', 'lossless', or an empty string. In 'jpeg', - Ghostscript is instructed to convert color and grayscale images to DCT - (JPEG encoding). In 'lossless' Ghostscript is told to convert images to - Flate (lossless/PNG). If the parameter is omitted Ghostscript is left to - make its own decisions about how to encode images; it appears to use a - heuristic to decide how to encode images. As of Ghostscript 9.25, we - support passthrough JPEG which allows Ghostscript to avoid transcoding - images entirely. (The feature was added in 9.23 but broken, and the 9.24 - release of Ghostscript had regressions, so we don't support it until 9.25.) + OCRmyPDF will modify the metadata and possibly linearize the PDF/A after it + is generated. + + Arguments: + pdf_pages: A list of one or more filenames, will be merged into output_file. + pdfmark: A PostScript file intended for Ghostscript with details on + how to perform the PDF/A conversion. + output_file: The name of the desired output file. + compression: One of ``'jpeg'``, ``'lossless'``, ``''``. For ``'jpeg'``, + the PDF/A generator should convert all images to JPEG encoding where + possible. For lossless, all images should be converted to FlateEncode + (lossless PNG). If an empty string, the PDF generator should make its + own decisions about how to encode images. + pdf_version: The minimum PDF version that the output file should be. + At its own discretion, the PDF/A generator may raise the version, + but should not lower it. + pdfa_part: The desired PDF/A compliance level, such as ``'2B'``. Returns: - output_file + output_file: If successful, the hook should return ``output_file``. """ From ddedf7cd2e807335ec47cd82a7f5483f865ff5ba Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 13:48:49 -0700 Subject: [PATCH 531/880] For --clean-final, use same image as --clean if possible --- src/ocrmypdf/_sync.py | 131 +++++++++++++++++++++++------------------- 1 file changed, 72 insertions(+), 59 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index e83a0231..c85f15ec 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -102,52 +102,67 @@ def exec_page_sync(page_context): options = page_context.options tls.pageno = page_context.pageno + 1 - orientation_correction = 0 - pdf_page_from_image_out = None - ocr_out = None - text_out = None - if is_ocr_required(page_context): - if options.rotate_pages: - # Rasterize - rasterize_preview_out = rasterize_preview(page_context.origin, page_context) - orientation_correction = get_orientation_correction( - rasterize_preview_out, page_context - ) - - rasterize_out = rasterize( - page_context.origin, - page_context, - correction=orientation_correction, - remove_vectors=False, + if not is_ocr_required(page_context): + return PageResult( + pageno=page_context.pageno, + pdf_page_from_image=None, + ocr=None, + text=None, + orientation_correction=0, ) - if not any([options.clean, options.clean_final, options.remove_vectors]): - ocr_image = preprocess_out = preprocess( + orientation_correction = 0 + if options.rotate_pages: + # Rasterize + rasterize_preview_out = rasterize_preview(page_context.origin, page_context) + orientation_correction = get_orientation_correction( + rasterize_preview_out, page_context + ) + + rasterize_out = rasterize( + page_context.origin, + page_context, + correction=orientation_correction, + remove_vectors=False, + ) + + ocr_image = preprocess_out = None + if not any([options.clean, options.clean_final, options.remove_vectors]): + ocr_image = preprocess_out = preprocess( + page_context, + rasterize_out, + options.remove_background, + options.deskew, + clean=False, + ) + else: + if not options.lossless_reconstruction: + preprocess_out = preprocess( page_context, rasterize_out, options.remove_background, options.deskew, - clean=False, + clean=options.clean_final, + ) + if options.remove_vectors: + rasterize_ocr_out = rasterize( + page_context.origin, + page_context, + correction=orientation_correction, + remove_vectors=True, + output_tag='_ocr', ) else: - if not options.lossless_reconstruction: - preprocess_out = preprocess( - page_context, - rasterize_out, - options.remove_background, - options.deskew, - clean=options.clean_final, - ) - if options.remove_vectors: - rasterize_ocr_out = rasterize( - page_context.origin, - page_context, - correction=orientation_correction, - remove_vectors=True, - output_tag='_ocr', - ) - else: - rasterize_ocr_out = rasterize_out + rasterize_ocr_out = rasterize_out + + if ( + preprocess_out + and rasterize_ocr_out == rasterize_out + and options.clean == options.clean_final + ): + # Optimization: image for OCR is identical to presentation image + ocr_image = preprocess_out + else: ocr_image = preprocess( page_context, rasterize_ocr_out, @@ -156,31 +171,29 @@ def exec_page_sync(page_context): clean=options.clean, ) - ocr_image_out = create_ocr_image(ocr_image, page_context) + ocr_image_out = create_ocr_image(ocr_image, page_context) - pdf_page_from_image_out = None - if not options.lossless_reconstruction: - visible_image_out = preprocess_out - if should_visible_page_image_use_jpg(page_context.pageinfo): - visible_image_out = create_visible_page_jpg( - visible_image_out, page_context - ) - visible_image_out = ( - page_context.plugin_manager.hook.filter_page_image( - page=page_context, image_filename=Path(visible_image_out) - ) - or visible_image_out - ) - pdf_page_from_image_out = create_pdf_page_from_image( - visible_image_out, page_context + pdf_page_from_image_out = None + if not options.lossless_reconstruction: + visible_image_out = preprocess_out + if should_visible_page_image_use_jpg(page_context.pageinfo): + visible_image_out = create_visible_page_jpg(visible_image_out, page_context) + visible_image_out = ( + page_context.plugin_manager.hook.filter_page_image( + page=page_context, image_filename=Path(visible_image_out) ) + or visible_image_out + ) + pdf_page_from_image_out = create_pdf_page_from_image( + visible_image_out, page_context + ) - if options.pdf_renderer == 'hocr': - (hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context) - ocr_out = render_hocr_page(hocr_out, page_context) + if options.pdf_renderer == 'hocr': + (hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context) + ocr_out = render_hocr_page(hocr_out, page_context) - if options.pdf_renderer == 'sandwich': - (ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context) + if options.pdf_renderer == 'sandwich': + (ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context) return PageResult( pageno=page_context.pageno, From 34231ac667decc772a03a366317cd9843987f3f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 15:02:28 -0700 Subject: [PATCH 532/880] sync: refactor intermediate image production --- src/ocrmypdf/_sync.py | 79 ++++++++++++++++++++++++++----------------- 1 file changed, 48 insertions(+), 31 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c85f15ec..cdfff74e 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -24,12 +24,13 @@ from collections import namedtuple from functools import partial from pathlib import Path from tempfile import mkdtemp +from typing import List, NamedTuple, Optional, Tuple import PIL from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._graft import OcrGrafter -from ocrmypdf._jobcontext import PdfContext, cleanup_working_files +from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files from ocrmypdf._pipeline import ( convert_to_pdfa, copy_final, @@ -75,16 +76,6 @@ tls = threading.local() tls.pageno = None -def preprocess(page_context, image, remove_background, deskew, clean): - if remove_background: - image = preprocess_remove_background(image, page_context) - if deskew: - image = preprocess_deskew(image, page_context) - if clean: - image = preprocess_clean(image, page_context) - return image - - old_factory = logging.getLogRecordFactory() @@ -98,27 +89,28 @@ def record_factory(*args, **kwargs): logging.setLogRecordFactory(record_factory) -def exec_page_sync(page_context): +def preprocess( + page_context: PageContext, + image: Path, + remove_background: bool, + deskew: bool, + clean: bool, +) -> Path: + if remove_background: + image = preprocess_remove_background(image, page_context) + if deskew: + image = preprocess_deskew(image, page_context) + if clean: + image = preprocess_clean(image, page_context) + return image + + +def make_intermediate_images( + page_context: PageContext, orientation_correction: int +) -> Tuple[Path, Optional[Path]]: options = page_context.options - tls.pageno = page_context.pageno + 1 - - if not is_ocr_required(page_context): - return PageResult( - pageno=page_context.pageno, - pdf_page_from_image=None, - ocr=None, - text=None, - orientation_correction=0, - ) - - orientation_correction = 0 - if options.rotate_pages: - # Rasterize - rasterize_preview_out = rasterize_preview(page_context.origin, page_context) - orientation_correction = get_orientation_correction( - rasterize_preview_out, page_context - ) + ocr_image = preprocess_out = None rasterize_out = rasterize( page_context.origin, page_context, @@ -126,7 +118,6 @@ def exec_page_sync(page_context): remove_vectors=False, ) - ocr_image = preprocess_out = None if not any([options.clean, options.clean_final, options.remove_vectors]): ocr_image = preprocess_out = preprocess( page_context, @@ -170,7 +161,33 @@ def exec_page_sync(page_context): options.deskew, clean=options.clean, ) + return ocr_image, preprocess_out + +def exec_page_sync(page_context: PageContext): + options = page_context.options + tls.pageno = page_context.pageno + 1 + + if not is_ocr_required(page_context): + return PageResult( + pageno=page_context.pageno, + pdf_page_from_image=None, + ocr=None, + text=None, + orientation_correction=0, + ) + + orientation_correction = 0 + if options.rotate_pages: + # Rasterize + rasterize_preview_out = rasterize_preview(page_context.origin, page_context) + orientation_correction = get_orientation_correction( + rasterize_preview_out, page_context + ) + + ocr_image, preprocess_out = make_intermediate_images( + page_context, orientation_correction + ) ocr_image_out = create_ocr_image(ocr_image, page_context) pdf_page_from_image_out = None From 698aab4f75b44836b778f8a2975bd9fe0ecae2c0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 15:10:59 -0700 Subject: [PATCH 533/880] Add a lot of type annotations --- src/ocrmypdf/_graft.py | 8 ++++- src/ocrmypdf/_pipeline.py | 53 +++++++++++++++++++-------------- src/ocrmypdf/_plugin_manager.py | 4 +-- src/ocrmypdf/_sync.py | 22 ++++++++------ src/ocrmypdf/_validation.py | 3 +- src/ocrmypdf/api.py | 2 ++ src/ocrmypdf/helpers.py | 15 +++++----- src/ocrmypdf/hocrtransform.py | 5 ++-- src/ocrmypdf/pluginspec.py | 6 +++- 9 files changed, 72 insertions(+), 46 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index de9ebda0..14c09af1 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -18,6 +18,7 @@ import logging from contextlib import suppress from pathlib import Path +from typing import Optional import pikepdf @@ -108,7 +109,12 @@ class OcrGrafter: self.interim_count = 0 def graft_page( - self, *, pageno: int, image: Path, textpdf: Path, autorotate_correction: int + self, + *, + pageno: int, + image: Optional[Path], + textpdf: Optional[Path], + autorotate_correction: int, ): if textpdf and not self.font: self.font, self.font_key = self._find_font(textpdf) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 8c27127f..40a27e3c 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -20,7 +20,9 @@ import os import re import sys from datetime import datetime, timezone +from pathlib import Path from shutil import copyfileobj +from typing import Dict, Iterable, Optional import img2pdf import pikepdf @@ -29,6 +31,7 @@ from PIL import Image, ImageColor, ImageDraw from ocrmypdf import leptonica from ocrmypdf._exec import unpaper +from ocrmypdf._jobcontext import PageContext, PdfContext from ocrmypdf._version import PROGRAM_NAME from ocrmypdf._version import __version__ as VERSION from ocrmypdf.exceptions import ( @@ -170,7 +173,7 @@ def get_pdfinfo( raise InputFileError() -def validate_pdfinfo_options(context): +def validate_pdfinfo_options(context: PdfContext): pdfinfo = context.pdfinfo options = context.options @@ -255,7 +258,7 @@ def get_canvas_square_dpi(pageinfo, options) -> Resolution: return Resolution(units, units) -def is_ocr_required(page_context): +def is_ocr_required(page_context: PageContext): pageinfo = page_context.pageinfo options = page_context.options @@ -329,7 +332,7 @@ def is_ocr_required(page_context): return ocr_required -def rasterize_preview(input_file, page_context): +def rasterize_preview(input_file: Path, page_context: PageContext): output_file = page_context.get_path('rasterize_preview.jpg') canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options) page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) @@ -344,7 +347,7 @@ def rasterize_preview(input_file, page_context): return output_file -def describe_rotation(page_context, orient_conf, correction): +def describe_rotation(page_context: PageContext, orient_conf, correction: int): """ Describe the page rotation we are going to perform. """ @@ -373,7 +376,7 @@ def describe_rotation(page_context, orient_conf, correction): return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}" -def get_orientation_correction(preview, page_context): +def get_orientation_correction(preview: Path, page_context: PageContext): """Work out orientation correct for each page. We ask Ghostscript to draw a preview page, which will rasterize with the @@ -405,7 +408,11 @@ def get_orientation_correction(preview, page_context): def rasterize( - input_file, page_context, correction=0, output_tag='', remove_vectors=None + input_file: Path, + page_context: PageContext, + correction: int = 0, + output_tag: str = '', + remove_vectors=None, ): colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m'] device_idx = 0 @@ -455,7 +462,7 @@ def rasterize( return output_file -def preprocess_remove_background(input_file, page_context): +def preprocess_remove_background(input_file: Path, page_context: PageContext): if any(image.bpc > 1 for image in page_context.pageinfo.images): output_file = page_context.get_path('pp_rm_bg.png') leptonica.remove_background(input_file, output_file) @@ -465,21 +472,21 @@ def preprocess_remove_background(input_file, page_context): return input_file -def preprocess_deskew(input_file, page_context): +def preprocess_deskew(input_file: Path, page_context: PageContext): output_file = page_context.get_path('pp_deskew.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) leptonica.deskew(input_file, output_file, dpi.x) return output_file -def preprocess_clean(input_file, page_context): +def preprocess_clean(input_file: Path, page_context: PageContext): output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args) return output_file -def create_ocr_image(image, page_context): +def create_ocr_image(image: Path, page_context: PageContext): """Create the image we send for OCR. May not be the same as the display image depending on preprocessing. This image will never be shown to the user.""" @@ -538,7 +545,7 @@ def create_ocr_image(image, page_context): return output_file -def ocr_engine_hocr(input_file, page_context): +def ocr_engine_hocr(input_file: Path, page_context: PageContext): hocr_out = page_context.get_path('ocr_hocr.hocr') hocr_text_out = page_context.get_path('ocr_hocr.txt') options = page_context.options @@ -558,7 +565,7 @@ def should_visible_page_image_use_jpg(pageinfo): return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images) -def create_visible_page_jpg(image, page_context): +def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path: output_file = page_context.get_path('visible.jpg') with Image.open(image) as im: # At this point the image should be a .png, but deskew, unpaper @@ -577,7 +584,7 @@ def create_visible_page_jpg(image, page_context): return output_file -def create_pdf_page_from_image(image, page_context): +def create_pdf_page_from_image(image: Path, page_context: PageContext): # We rasterize a square DPI version of each page because most image # processing tools don't support rectangular DPI. Use the square DPI as it # accurately describes the image. It would be possible to resample the image @@ -598,7 +605,7 @@ def create_pdf_page_from_image(image, page_context): return output_file -def render_hocr_page(hocr, page_context): +def render_hocr_page(hocr: Path, page_context: PageContext): output_file = page_context.get_path('ocr_hocr.pdf') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) hocrtransform = HocrTransform(hocr, dpi.x) # square @@ -612,7 +619,7 @@ def render_hocr_page(hocr, page_context): return output_file -def ocr_engine_textonly_pdf(input_image, page_context): +def ocr_engine_textonly_pdf(input_image: Path, page_context: PageContext): output_pdf = page_context.get_path('ocr_tess.pdf') output_text = page_context.get_path('ocr_tess.txt') options = page_context.options @@ -627,7 +634,7 @@ def ocr_engine_textonly_pdf(input_image, page_context): return (output_pdf, output_text) -def get_docinfo(base_pdf, context): +def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]: options = context.options def from_document_info(key): @@ -664,13 +671,13 @@ def get_docinfo(base_pdf, context): return pdfmark -def generate_postscript_stub(context): +def generate_postscript_stub(context: PdfContext): output_file = context.get_path('pdfa.ps') generate_pdfa_ps(output_file) return output_file -def convert_to_pdfa(input_pdf, input_ps_stub, context): +def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext): options = context.options input_pdfinfo = context.pdfinfo fix_docinfo_file = context.get_path('fix_docinfo.pdf') @@ -712,14 +719,14 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): return output_file -def should_linearize(working_file, context): +def should_linearize(working_file: Path, context: PdfContext): filesize = os.stat(working_file).st_size if filesize > (context.options.fast_web_view * 1_000_000): return True return False -def metadata_fixup(working_file, context): +def metadata_fixup(working_file: Path, context: PdfContext): output_file = context.get_path('metafix.pdf') options = context.options @@ -768,7 +775,7 @@ def metadata_fixup(working_file, context): return output_file -def optimize_pdf(input_file, context): +def optimize_pdf(input_file: Path, context: PdfContext): output_file = context.get_path('optimize.pdf') save_settings = dict( compress_streams=True, @@ -780,7 +787,7 @@ def optimize_pdf(input_file, context): return output_file -def merge_sidecars(txt_files, context): +def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext): output_file = context.get_path('sidecar.txt') with open(output_file, 'w', encoding="utf-8") as stream: for page_num, txt_file in enumerate(txt_files): @@ -801,7 +808,7 @@ def merge_sidecars(txt_files, context): return output_file -def copy_final(input_file, output_file, _context): +def copy_final(input_file: Path, output_file: Path, _context: PdfContext): log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index baa25e4a..5bc1ab60 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -20,7 +20,7 @@ import importlib import importlib.util import sys from pathlib import Path -from typing import List +from typing import List, Tuple import pluggy @@ -56,7 +56,7 @@ def get_plugin_manager(plugins: List[str], builtins=True): def get_parser_options_plugins( args, -) -> (argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager): +) -> Tuple[argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager]: pre_options, _unused = plugins_only_parser.parse_known_args(args=args) plugin_manager = get_plugin_manager(pre_options.plugins) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index cdfff74e..b55ace4f 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -20,7 +20,6 @@ import logging.handlers import os import sys import threading -from collections import namedtuple from functools import partial from pathlib import Path from tempfile import mkdtemp @@ -68,9 +67,14 @@ from ocrmypdf.pdfa import file_claims_pdfa log = logging.getLogger(__name__) -PageResult = namedtuple( - 'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction' -) + +class PageResult(NamedTuple): + pageno: int + pdf_page_from_image: Optional[Path] + ocr: Optional[Path] + text: Optional[Path] + orientation_correction: int + tls = threading.local() tls.pageno = None @@ -221,7 +225,7 @@ def exec_page_sync(page_context: PageContext): ) -def post_process(pdf_file, context): +def post_process(pdf_file, context: PdfContext): pdf_out = pdf_file if context.options.output_type.startswith('pdfa'): ps_stub_out = generate_postscript_stub(context) @@ -231,14 +235,14 @@ def post_process(pdf_file, context): return optimize_pdf(pdf_out, context) -def worker_init(max_pixels): +def worker_init(max_pixels: int): # In Windows, child process will not inherit our change to this value in # the parent process, so ensure workers get it set. Not needed when running # threaded, but harmless to set again. PIL.Image.MAX_IMAGE_PIXELS = max_pixels -def exec_concurrent(context): +def exec_concurrent(context: PdfContext): """Execute the pipeline concurrently""" # Run exec_page_sync on every page context @@ -246,10 +250,10 @@ def exec_concurrent(context): if max_workers > 1: log.info("Start processing %d pages concurrently", max_workers) - sidecars = [None] * len(context.pdfinfo) + sidecars: List[Optional[Path]] = [None] * len(context.pdfinfo) ocrgraft = OcrGrafter(context) - def update_page(result, pbar): + def update_page(result: PageResult, pbar): sidecars[result.pageno] = result.text pbar.update() ocrgraft.graft_page( diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 06a16670..254410c1 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -24,6 +24,7 @@ import sys import unicodedata from pathlib import Path from shutil import copyfileobj +from typing import Tuple import pikepdf import PIL @@ -329,7 +330,7 @@ def log_page_orientations(pdfinfo): log.info('Page orientations detected: %s', ' '.join(orientations)) -def create_input_file(options, work_folder: Path) -> (Path, str): +def create_input_file(options, work_folder: Path) -> Tuple[Path, str]: if options.input_file == '-': # stdin log.info('reading file from standard input') diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index a0cb19f5..a9fb047c 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -264,6 +264,8 @@ def ocr( # pylint: disable=unused-argument """ if not plugins: plugins = [] + else: + plugins = list(plugins) parser = get_parser() _plugin_manager = get_plugin_manager(plugins) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 964f7670..c55ac5f9 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -27,6 +27,7 @@ from functools import wraps from io import StringIO from math import isclose from pathlib import Path +from typing import Any, Sequence import pikepdf @@ -104,28 +105,28 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): os.symlink(os.path.abspath(input_file), soft_link_name) -def samefile(f1, f2): +def samefile(f1: os.PathLike, f2: os.PathLike): if os.name == 'nt': return f1 == f2 else: return os.path.samefile(f1, f2) -def is_iterable_notstr(thing): +def is_iterable_notstr(thing: Any) -> bool: return isinstance(thing, Iterable) and not isinstance(thing, str) -def monotonic(L: Iterable): +def monotonic(L: Sequence) -> bool: """Does list increase monotonically?""" return all(b > a for a, b in zip(L, L[1:])) -def page_number(input_file: os.PathLike): +def page_number(input_file: os.PathLike) -> int: """Get one-based page number implied by filename (000002.pdf -> 2)""" return int(os.path.basename(os.fspath(input_file))[0:6]) -def available_cpu_count(): +def available_cpu_count() -> int: try: return multiprocessing.cpu_count() except NotImplementedError: @@ -136,7 +137,7 @@ def available_cpu_count(): return 1 -def is_file_writable(test_file: os.PathLike): +def is_file_writable(test_file: os.PathLike) -> bool: """Intentionally racy test if target is writable. We intend to write to the output file if and only if we succeed and @@ -171,7 +172,7 @@ def is_file_writable(test_file: os.PathLike): return False -def check_pdf(input_file): +def check_pdf(input_file: Path) -> bool: pdf = None try: pdf = pikepdf.open(input_file) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 6240d7ee..f1446535 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -34,6 +34,7 @@ import re from collections import namedtuple from math import atan, cos, sin from pathlib import Path +from typing import Union from xml.etree import ElementTree from reportlab.lib.units import inch @@ -66,9 +67,9 @@ class HocrTransform: {'ff': 'ff', 'ffi': 'f‌f‌i', 'ffl': 'f‌f‌l', 'fi': 'fi', 'fl': 'fl'} ) - def __init__(self, hocr_filename: str, dpi: float): + def __init__(self, hocr_filename: Union[str, Path], dpi: float): self.dpi = dpi - self.hocr = ElementTree.parse(hocr_filename) + self.hocr = ElementTree.parse(os.fspath(hocr_filename)) # if the hOCR file has a namespace, ElementTree requires its use to # find elements diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 3bab37e1..a96178f9 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -19,13 +19,17 @@ from abc import ABC, abstractmethod, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple from pathlib import Path -from typing import AbstractSet, List, Optional +from typing import TYPE_CHECKING, AbstractSet, List, Optional import pluggy from PIL import Image from ocrmypdf.helpers import Resolution +if TYPE_CHECKING: + from ocrmypdf._jobcontext import PageContext + from ocrmypdf.pdfinfo import PdfInfo + hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument From 642998ead6a2f54894fca0c795951c3cabe4e6bb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 15:26:41 -0700 Subject: [PATCH 534/880] sync: refactor preprocess image filtering --- src/ocrmypdf/_sync.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index b55ace4f..6c730025 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -196,15 +196,15 @@ def exec_page_sync(page_context: PageContext): pdf_page_from_image_out = None if not options.lossless_reconstruction: + assert preprocess_out visible_image_out = preprocess_out if should_visible_page_image_use_jpg(page_context.pageinfo): visible_image_out = create_visible_page_jpg(visible_image_out, page_context) - visible_image_out = ( - page_context.plugin_manager.hook.filter_page_image( - page=page_context, image_filename=Path(visible_image_out) - ) - or visible_image_out + filtered_image = page_context.plugin_manager.hook.filter_page_image( + page=page_context, image_filename=visible_image_out ) + if filtered_image: + visible_image_out = filtered_image pdf_page_from_image_out = create_pdf_page_from_image( visible_image_out, page_context ) From 0b5a20e593114c0397288896ab8eb7247a580de2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Jun 2020 15:55:39 -0700 Subject: [PATCH 535/880] coverage: ignore type checking --- .coveragerc | 1 + 1 file changed, 1 insertion(+) diff --git a/.coveragerc b/.coveragerc index 703ae66d..51fe525c 100644 --- a/.coveragerc +++ b/.coveragerc @@ -23,3 +23,4 @@ exclude_lines = if 0: if False: if __name__ == .__main__.: + if TYPE_CHECKING: From e802896d4dc34ae64bdd8f5055a3b30c8dda5c8d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 16 Jun 2020 00:50:18 -0700 Subject: [PATCH 536/880] unpaper: use PNG input where possible Unpaper accepts PNG as input now, so avoid generating a huge temporary PPM file if we can. If we must create a PNG, compress it lightly to keep our temp usage down. --- src/ocrmypdf/_exec/unpaper.py | 44 +++++++++++++++++++++-------------- 1 file changed, 27 insertions(+), 17 deletions(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index e1a58746..b3154ef6 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -26,6 +26,7 @@ import shlex from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError from tempfile import TemporaryDirectory +from typing import Tuple from PIL import Image @@ -40,13 +41,11 @@ def version(): return get_version('unpaper') -def run(input_file, output_file, dpi, mode_args): - args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args - +def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'} - - with TemporaryDirectory() as tmpdir, Image.open(input_file) as im: - if im.mode not in SUFFIXES.keys(): + with Image.open(input_file) as im: + im_modified = False + if im.mode not in SUFFIXES: log.info("Converting image to other colorspace") try: if im.mode == 'P' and len(im.getcolors()) == 2: @@ -54,11 +53,11 @@ def run(input_file, output_file, dpi, mode_args): else: im = im.convert(mode='RGB') except IOError as e: - im.close() raise MissingDependencyError( "Could not convert image with type " + im.mode ) from e - + else: + im_modified = True try: suffix = SUFFIXES[im.mode] except KeyError: @@ -66,9 +65,21 @@ def run(input_file, output_file, dpi, mode_args): "Failed to convert image to a supported format." ) from e - input_pnm = Path(tmpdir) / f'input{suffix}' - output_pnm = Path(tmpdir) / f'output{suffix}' - im.save(input_pnm, format='PPM') + if im_modified or input_file.suffix != '.png': + input_png = tmpdir / 'input.png' + im.save(input_png, format='PNG', compress_level=1) + else: + # No changes, PNG input, just use the file we already have + input_png = input_file + output_pnm = tmpdir / f'output{suffix}' + return input_png, output_pnm + + +def run(input_file, output_file, dpi, mode_args): + args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args + + with TemporaryDirectory() as tmpdir: + input_png, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file) # To prevent any shenanigans from accepting arbitrary parameters in # --unpaper-args, we: @@ -77,23 +88,22 @@ def run(input_file, output_file, dpi, mode_args): # 3) append absolute paths for the input and output file # This should ensure that a user cannot clobber some other file with # their unpaper arguments (whether intentionally or otherwise) - args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)]) + args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)]) try: proc = external_run( args_unpaper, check=True, close_fds=True, universal_newlines=True, - stderr=STDOUT, - cwd=tmpdir, + stderr=STDOUT, # unpaper writes logging output to stdout and stderr + cwd=tmpdir, # and cannot send file output to stdout stdout=PIPE, ) except CalledProcessError as e: - log.debug(e.output) + log.debug(e.stderr) raise e from e else: - log.debug(proc.stdout) - # unpaper sets dpi to 72; fix this + log.debug(proc.stderr) try: with Image.open(output_pnm) as imout: imout.save(output_file, dpi=(dpi, dpi)) From 24b6a4ad502875986fa85ec184b71d511d783202 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 16 Jun 2020 00:55:28 -0700 Subject: [PATCH 537/880] v10.1.0 notes --- docs/release_notes.rst | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c8dee0d9..f9401ead 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,17 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.1.0 +======= + +- Previously, we ``--clean-final`` would cause an unpaper-cleaned page image to + be produced twice, which was necessary in some cases but not in general. We + now take this optimization opportunity and reuse the image if possible. +- We now provide PNG files as input to unpaper, since it accepts them, instead + of generating PPM files which can be very large. This can improve performance + and temporary disk usage. +- Documentation updated for plugins. + v10.0.1 ======= From 6ac50646f07ad3ab85900b6201f80bdabf913a15 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 17 Jun 2020 14:43:19 -0700 Subject: [PATCH 538/880] Fix OMP_THREAD_LIMIT rounded down to 0 in some cases --- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 3 ++- src/ocrmypdf/helpers.py | 13 ++++++++++--- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index dced0c9b..867e232c 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -22,6 +22,7 @@ from ocrmypdf import hookimpl from ocrmypdf._exec import tesseract from ocrmypdf.cli import numeric from ocrmypdf.exceptions import MissingDependencyError +from ocrmypdf.helpers import clamp from ocrmypdf.pluginspec import OcrEngine from ocrmypdf.subprocess import check_external_program @@ -126,7 +127,7 @@ def validate(pdfinfo, options): # constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers. # As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system. if not os.environ.get('OMP_THREAD_LIMIT', '').isnumeric(): - tess_threads = min(3, options.jobs // len(pdfinfo), len(pdfinfo)) + tess_threads = clamp(options.jobs // len(pdfinfo), 1, 3) os.environ['OMP_THREAD_LIMIT'] = str(tess_threads) else: tess_threads = int(os.environ['OMP_THREAD_LIMIT']) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index c55ac5f9..aca40b6e 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -27,7 +27,7 @@ from functools import wraps from io import StringIO from math import isclose from pathlib import Path -from typing import Any, Sequence +from typing import Any, Sequence, TypeVar import pikepdf @@ -37,14 +37,14 @@ log = logging.getLogger(__name__) class Resolution(namedtuple('Resolution', ('x', 'y'))): __slots__ = () - def round(self, ndigits): + def round(self, ndigits: int): return Resolution(round(self.x, ndigits), round(self.y, ndigits)) def to_int(self): return Resolution(int(round(self.x)), int(round(self.y))) @property - def is_square(self): + def is_square(self) -> bool: return isclose(self.x, self.y, rel_tol=1e-3) def take_max(self, vals, yvals=None): @@ -206,6 +206,13 @@ def check_pdf(input_file: Path) -> bool: pdf.close() +T = TypeVar('T') + + +def clamp(n: T, smallest: T, largest: T) -> T: + return max(smallest, min(n, largest)) + + def deprecated(func): """Warn that function is deprecated""" From ad22977c84432ce125c25ca05d7172a28bbbf31d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 17 Jun 2020 14:45:32 -0700 Subject: [PATCH 539/880] v10.1.1 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f9401ead..40b71d41 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,13 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.1.1 +======= + +- Fixed ``OMP_THREAD_LIMIT`` set to invalid value error messages on some input + files. (The error was harmless, apart from less than optimal performance in + some cases.) + v10.1.0 ======= From ebfe4f0d29d038248137882859025b346ba115e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 20 Jun 2020 02:01:16 -0700 Subject: [PATCH 540/880] Fix issue #582 - PDF/A acquires title "Untitled" after conversion --- src/ocrmypdf/_pipeline.py | 8 ++++++++ tests/test_metadata.py | 30 ++++++++++++++++++++++++------ 2 files changed, 32 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 40a27e3c..688fc5a7 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -756,6 +756,14 @@ def metadata_fixup(working_file: Path, context: PdfContext): if 'xmp:CreateDate' not in meta: meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') + # Ghostscript likes to set title to Untitled if omitted from input. + # Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1 + # and the XMP Spec do not make this recommendation. + if meta.get('dc:title') == 'Untitled': + with original.open_metadata() as original_meta: + if 'dc:title' not in original_meta: + del meta['dc:title'] + meta_original = original.open_metadata() missing = set(meta_original.keys()) - set(meta.keys()) report_on_metadata(missing) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 1d310107..1c851cef 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -50,7 +50,7 @@ run_ocrmypdf = pytest.helpers.run_ocrmypdf @pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) -def test_preserve_metadata(output_type, resources, outpdf): +def test_preserve_docinfo(output_type, resources, outpdf): pdf_before = pikepdf.open(resources / 'graph.pdf') output = check_ocrmypdf( @@ -188,9 +188,17 @@ def test_creation_date_preserved(output_type, resources, infile, outpdf): assert seconds_between_dates(date_after, datetime.datetime.now(timezone.utc)) < 1000 -@pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) -def test_xml_metadata_preserved(output_type, resources, outpdf): - input_file = resources / 'graph.pdf' +@pytest.mark.parametrize( + 'test_file,output_type', + [ + ('graph.pdf', 'pdf'), # PDF with full metadata + ('graph.pdf', 'pdfa'), # PDF/A with full metadata + ('overlay.pdf', 'pdfa'), # /Title() + ('3small.pdf', 'pdfa'), + ], +) +def test_xml_metadata_preserved(test_file, output_type, resources, outpdf): + input_file = resources / test_file try: from libxmp.utils import file_to_dict # pylint: disable=import-outside-toplevel @@ -204,6 +212,7 @@ def test_xml_metadata_preserved(output_type, resources, outpdf): outpdf, '--output-type', output_type, + '--skip-text', '--plugin', 'tests/plugins/tesseract_noop.py', ) @@ -227,6 +236,7 @@ def test_xml_metadata_preserved(output_type, resources, outpdf): 'dc:type', 'pdf:keywords', ] + acquired_properties = ['dc:format'] might_change_properties = [ 'dc:date', 'pdf:pdfversion', @@ -260,8 +270,10 @@ def test_xml_metadata_preserved(output_type, resources, outpdf): assert prop in after, f'{prop} dropped from xmp' assert before[prop] == after[prop] - # Certain entries like title appear as dc:title[1], with the possibility - # of several + # libxmp presents multivalued entries (e.g. dc:title) as: + # 'dc:title': '' <- there's a title + # 'dc:title[1]: 'The Title' <- the actual title + # 'dc:title[1]/?xml:lang': 'x-default' <- language info propidx = f'{prop}[1]' if propidx in before: assert ( @@ -269,6 +281,12 @@ def test_xml_metadata_preserved(output_type, resources, outpdf): or after.get(prop) == before[propidx] ) + if prop in after and prop not in before: + assert prop in acquired_properties, ( + f"acquired unexpected property {prop} with value " + f"{after.get(propidx) or after.get(prop)}" + ) + def test_srgb_in_unicode_path(tmp_path): """Test that we can produce pdfmark when install path is not ASCII""" From 06d52326db2119c0446705d5c3eb29ddf336e7fd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 21 Jun 2020 01:24:23 -0700 Subject: [PATCH 541/880] Fix deleted path in .coveragerc --- .coveragerc | 2 -- 1 file changed, 2 deletions(-) diff --git a/.coveragerc b/.coveragerc index 51fe525c..42b8e78f 100644 --- a/.coveragerc +++ b/.coveragerc @@ -11,8 +11,6 @@ concurrency = multiprocessing source = src/ocrmypdf -omit = - tests/spoof/* [report] exclude_lines = From e182c5f63e82a69e481724a4a30521c7462c256c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 21 Jun 2020 01:25:59 -0700 Subject: [PATCH 542/880] Update and sync .dockerignore, .gitignore Also blacklist .* and whitelist the ones we want. --- .dockerignore | 57 +++++++++++++++++++++++++++++++---------------- .gitignore | 61 +++++++++++++++++++++++---------------------------- 2 files changed, 66 insertions(+), 52 deletions(-) diff --git a/.dockerignore b/.dockerignore index 2ebfdc48..2879b33a 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,25 +1,44 @@ +# dotfiles +.* +!.coveragerc +!.dockerignore +!.git_archival.txt +!.gitattributes +!.gitignore +!.pre-commit-config.yaml +!.readthedocs.yml + +# Dev scratch *.ipynb -*.pdf -*.pyc -*.rst -*.sublime* **/*.pyc -.*/ -!.git/ -.ruffus_history.sqlite -bin/ -build/ -docs/ -dist/ -htmlcov/ -include/ -lib/ -MANIFEST.in -ocrmypdf.egg-info/ -staging/ -tests/cache/ -tests/output/ +/*.pdf +/*.qdf +/*.png +/scratch.py +IDEAS +log/ tests/resources/private/ tmp/ venv*/ +/debug_tests.py +*.traineddata +/private + +# Package building +*.egg-info/ +build/ +dist/ wheelhouse/ +pip-wheel-metadata/ + +# Code coverage +htmlcov/ + +# Docker specific +bin/ +docs/ +include/ +lib/ + +# Docker include .git/ +!.git/ diff --git a/.gitignore b/.gitignore index bf6e896c..90884590 100644 --- a/.gitignore +++ b/.gitignore @@ -1,47 +1,42 @@ -# Development environment -.bash_history -.pylintrc -.pytest_cache/ -.ruffus_history.sqlite -.venv*/ -*.pyc -*.sublime-* -*.DS_Store -.mypy_cache/ +# dotfiles +.* +!.coveragerc +!.dockerignore +!.git_archival.txt +!.gitattributes +!.gitignore +!.pre-commit-config.yaml +!.readthedocs.yml + +# Dev scratch +*.ipynb +**/*.pyc +/*.pdf +/*.qdf +/*.png +/scratch.py +IDEAS +log/ +tests/resources/private/ +tmp/ +venv*/ +/debug_tests.py +*.traineddata +/private # Package building -.eggs/ *.egg-info/ build/ dist/ wheelhouse/ pip-wheel-metadata/ +# Code coverage +htmlcov/ + # Automatically generated files docs/_build/ docs/_static/ docs/_templates/ docs/Makefile ocrmypdf/lib/_*.py - -# Code coverage -.coverage* -htmlcov/ - -# Testing -.ipynb_checkpoints/ -.vscode/ -*.ipynb -*.profile -/*.pdf -/*.qdf -/*.png -/scratch.py -IDEAS -log/ -tests/output/ -tests/resources/private/ -tmp/ -/debug_tests.py -*.traineddata -/private From 48e2750551ba095962292a94212049a7987b192a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 21 Jun 2020 01:48:13 -0700 Subject: [PATCH 543/880] Fix some tests that were failing in Docker --- tests/conftest.py | 8 +++++--- tests/test_helpers.py | 6 ++++++ tests/test_main.py | 4 +++- 3 files changed, 14 insertions(+), 4 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 849e29d0..7d495171 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -148,12 +148,14 @@ def run_ocrmypdf(input_file, output_file, *args, universal_newlines=True): ) # Tell subprocess where to find coverage.py configuration - # This has no effect except when coverage is running + # This has no unless except when coverage is running # Details: https://coverage.readthedocs.io/en/coverage-5.0/subprocess.html coverage_rc = Path(__file__).parent.parent / '.coveragerc' - assert coverage_rc.exists() env = os.environ.copy() - env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) + if coverage_rc.exists(): + env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) + elif not running_in_docker(): + assert False, "could not find .coveragerc" p = run( p_args, diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 8b3f6748..397f068e 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -64,6 +64,11 @@ def test_deprecated(): assert old_function() == 42 +skipif_docker = pytest.mark.skipif( + pytest.helpers.running_in_docker(), reason="fails on Docker" +) + + class TestFileIsWritable: @pytest.fixture def non_existent(self, tmp_path): @@ -83,6 +88,7 @@ class TestFileIsWritable: loop.symlink_to(loop) assert not helpers.is_file_writable(loop) + @skipif_docker def test_chmod(self, basic_file): assert helpers.is_file_writable(basic_file) basic_file.chmod(0o400) diff --git a/tests/test_main.py b/tests/test_main.py index 289cbbcc..712753fe 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -279,7 +279,9 @@ def test_input_file_not_found(caplog, no_outpdf): assert input_file in caplog.text -@pytest.mark.skipif(os.name == 'nt', reason="chmod") +@pytest.mark.skipif( + os.name == 'nt' or pytest.helpers.running_in_docker(), reason="chmod" +) def test_input_file_not_readable(caplog, resources, outdir, no_outpdf): input_file = outdir / 'trivial.pdf' shutil.copy(resources / 'trivial.pdf', input_file) From 24d64b04c33448ef92ac8eecd64959597233916f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 21 Jun 2020 01:48:31 -0700 Subject: [PATCH 544/880] Update Docker to Ubuntu 20.04 and jbig2-latest --- .docker/Dockerfile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 527fed5d..2378adb3 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -1,6 +1,6 @@ # OCRmyPDF # -FROM ubuntu:19.10 as base +FROM ubuntu:20.04 as base FROM base as builder @@ -24,7 +24,7 @@ RUN \ # Needs libleptonica-dev, zlib1g-dev RUN \ mkdir jbig2 \ - && curl -L https://github.com/agl/jbig2enc/archive/0.29.tar.gz | \ + && curl -L https://github.com/agl/jbig2enc/archive/ea6a40a.tar.gz | \ tar xz -C jbig2 --strip-components=1 \ && cd jbig2 \ && ./autogen.sh && ./configure && make && make install \ From 800c75c4e582f75a87b68b8451e295dcb7ba4453 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 21 Jun 2020 01:58:53 -0700 Subject: [PATCH 545/880] Bump requirements (mainly for Docker's benefit) --- requirements/main.txt | 10 +++++----- requirements/test.txt | 2 +- requirements/watcher.txt | 2 +- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 24210fed..74be53e4 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -3,10 +3,10 @@ # installation cffi == 1.14.0 coloredlogs == 14.0 # technically optional -img2pdf == 0.3.4 +img2pdf == 0.3.6 pdfminer.six == 20200517 -pikepdf == 1.14.0 +pikepdf == 1.15.1 pluggy == 0.13.1 -Pillow == 7.1.1 -reportlab == 3.5.34 -tqdm == 4.45.0 +Pillow == 7.1.2 +reportlab == 3.5.42 +tqdm == 4.46.1 diff --git a/requirements/test.txt b/requirements/test.txt index 531bea96..8bdf1fe6 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,7 +1,7 @@ pytest >= 5.0.0 pytest-helpers-namespace >= 2019.1.8 pytest-xdist >= 1.31.0 -pytest-cov >= 2.9.0 +pytest-cov >= 2.10.0 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi #PyMuPDF == 1.13.4 # optional diff --git a/requirements/watcher.txt b/requirements/watcher.txt index e8ddcd19..cdbc5325 100644 --- a/requirements/watcher.txt +++ b/requirements/watcher.txt @@ -1 +1 @@ -watchdog >= 0.8.2, < 1.0 +watchdog == 0.10.2 From 5b10ec9d39c984d360ba12b5ceae2f039e1e1c3c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 00:34:58 -0700 Subject: [PATCH 546/880] jobcontext.PdfContext: remove dead code, add annotations --- src/ocrmypdf/_jobcontext.py | 29 +++++++++++++++++------------ 1 file changed, 17 insertions(+), 12 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 3c91e856..37e541ce 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -18,33 +18,38 @@ import os import shutil import sys +from argparse import Namespace from pathlib import Path +from typing import Iterator from ocrmypdf._plugin_manager import get_plugin_manager +from ocrmypdf.pdfinfo import PdfInfo class PdfContext: """Holds our context for a particular run of the pipeline""" def __init__( - self, options, work_folder: Path, origin: Path, pdfinfo, plugin_manager + self, + options: Namespace, + work_folder: Path, + origin: Path, + pdfinfo: PdfInfo, + plugin_manager, ): self.options = options - self.work_folder = Path(work_folder) - self.origin = Path(origin) + self.work_folder = work_folder + self.origin = origin self.pdfinfo = pdfinfo self.plugin_manager = plugin_manager - if options: - self.name = os.path.basename(options.input_file) - else: - self.name = 'origin.pdf' + self.name = os.path.basename(options.input_file) if self.name == '-': self.name = 'stdin' def get_path(self, name: str) -> Path: return self.work_folder / name - def get_page_contexts(self): + def get_page_contexts(self) -> Iterator[PageContext]: npages = len(self.pdfinfo) for n in range(npages): yield PageContext(self, n) @@ -53,7 +58,8 @@ class PdfContext: class PageContext: """Holds our context for a page - Must be pickable, so only store intrinsic/simple data elements + Must be pickable, so stores only intrinsic/simple data elements or those + capable of their serializing themselves via __getstate__. """ def __init__(self, pdf_context: PdfContext, pageno): @@ -70,8 +76,7 @@ class PageContext: def __getstate__(self): state = self.__dict__.copy() - if state['plugin_manager'] is not None: - state['plugin_manager'] = None + state['plugin_manager'] = None # We cannot serialize the plugin manager... return state def __setstate__(self, state): @@ -79,7 +84,7 @@ class PageContext: self.plugin_manager = get_plugin_manager(self.options.plugins) -def cleanup_working_files(work_folder, options): +def cleanup_working_files(work_folder: Path, options: Namespace): if options.keep_temporary_files: print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr) else: From 86ec63f215a0e7d3167c4fa9dbffeb5bda9aaace Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 01:07:22 -0700 Subject: [PATCH 547/880] Decouple plugin manager forking from PdfContext/Pagecontext --- src/ocrmypdf/_jobcontext.py | 12 +-------- src/ocrmypdf/_plugin_manager.py | 48 ++++++++++++++++++++++++++++++--- 2 files changed, 46 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 37e541ce..4cb910a1 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -22,7 +22,6 @@ from argparse import Namespace from pathlib import Path from typing import Iterator -from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.pdfinfo import PdfInfo @@ -49,7 +48,7 @@ class PdfContext: def get_path(self, name: str) -> Path: return self.work_folder / name - def get_page_contexts(self) -> Iterator[PageContext]: + def get_page_contexts(self) -> Iterator['PageContext']: npages = len(self.pdfinfo) for n in range(npages): yield PageContext(self, n) @@ -74,15 +73,6 @@ class PageContext: def get_path(self, name: str) -> Path: return self.work_folder / ("%06d_%s" % (self.pageno + 1, name)) - def __getstate__(self): - state = self.__dict__.copy() - state['plugin_manager'] = None # We cannot serialize the plugin manager... - return state - - def __setstate__(self, state): - self.__dict__.update(state) - self.plugin_manager = get_plugin_manager(self.options.plugins) - def cleanup_working_files(work_folder: Path, options: Namespace): if options.keep_temporary_files: diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 5bc1ab60..ba867525 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -19,8 +19,9 @@ import argparse import importlib import importlib.util import sys +from functools import partial from pathlib import Path -from typing import List, Tuple +from typing import Callable, List, Tuple import pluggy @@ -28,8 +29,42 @@ from ocrmypdf import pluginspec from ocrmypdf.cli import get_parser, plugins_only_parser -def get_plugin_manager(plugins: List[str], builtins=True): - pm = pluggy.PluginManager('ocrmypdf') +class OcrmypdfPluginManager(pluggy.PluginManager): + """pluggy.PluginManager that can fork. + + Capable of reconstructing itself in child workers. + + Arguments: + setup_func: callback that initializes the plugin manager with all + standard plugins + """ + + def __init__( + self, *args, setup_func: Callable[[pluggy.PluginManager], None], **kwargs + ): + self._init_args = args + self._setup_func = setup_func + self._init_kwargs = kwargs + super().__init__(*args, **kwargs) + setup_func(self) + + def __getstate__(self): + state = dict( + _init_args=self._init_args, + _setup_func=self._setup_func, + _init_kwargs=self._init_kwargs, + ) + return state + + def __setstate__(self, state): + self.__init__( + *state['_init_args'], + setup_func=state['_setup_func'], + **state['_init_kwargs'], + ) + + +def _setup_plugins(pm: pluggy.PluginManager, plugins: List[str], builtins: bool = True): pm.add_hookspecs(pluginspec) if builtins: @@ -51,6 +86,13 @@ def get_plugin_manager(plugins: List[str], builtins=True): # Import by dotted module name module = importlib.import_module(name) pm.register(module) + + +def get_plugin_manager(plugins: List[str], builtins=True): + pm = OcrmypdfPluginManager( + project_name='ocrmypdf', + setup_func=partial(_setup_plugins, plugins=plugins, builtins=builtins), + ) return pm From fef14778d52be895bb690c8ca01012413c7b28ed Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 00:10:24 -0700 Subject: [PATCH 548/880] Fix missing f-string in log message --- tests/plugins/tesseract_cache.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py index 1df3fd98..9d75f574 100644 --- a/tests/plugins/tesseract_cache.py +++ b/tests/plugins/tesseract_cache.py @@ -109,7 +109,7 @@ def cached_run(options, run_args, **run_kwargs): cache_folder = get_cache_folder(source_file, run_args, args) cache_folder.mkdir(parents=True, exist_ok=True) - log.debug("Using Tesseract cache {cache_folder}") + log.debug(f"Using Tesseract cache {cache_folder}") if (cache_folder / 'stderr.bin').exists(): log.debug("Cache HIT") From f4cb4244518bbc43137832d6b4381530eacc599b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 30 Apr 2020 03:38:27 -0700 Subject: [PATCH 549/880] Support input/output streams at API level --- src/ocrmypdf/_jobcontext.py | 16 ++++++++++++---- src/ocrmypdf/_pipeline.py | 5 +++++ src/ocrmypdf/_sync.py | 4 ++++ src/ocrmypdf/_validation.py | 13 ++++++++++++- src/ocrmypdf/api.py | 31 +++++++++++++++++++++---------- tests/test_api.py | 11 ++++++++++- 6 files changed, 64 insertions(+), 16 deletions(-) diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 4cb910a1..ae7b5f52 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -19,6 +19,8 @@ import os import shutil import sys from argparse import Namespace +from copy import copy +from io import IOBase from pathlib import Path from typing import Iterator @@ -41,9 +43,6 @@ class PdfContext: self.origin = origin self.pdfinfo = pdfinfo self.plugin_manager = plugin_manager - self.name = os.path.basename(options.input_file) - if self.name == '-': - self.name = 'stdin' def get_path(self, name: str) -> Path: return self.work_folder / name @@ -65,7 +64,6 @@ class PageContext: self.work_folder = pdf_context.work_folder self.origin = pdf_context.origin self.options = pdf_context.options - self.name = pdf_context.name self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] self.plugin_manager = pdf_context.plugin_manager @@ -73,6 +71,16 @@ class PageContext: def get_path(self, name: str) -> Path: return self.work_folder / ("%06d_%s" % (self.pageno + 1, name)) + def __getstate__(self): + state = self.__dict__.copy() + + state['options'] = copy(self.options) + if not isinstance(state['options'].input_file, (str, bytes, os.PathLike)): + state['options'].input_file = 'stream' + if not isinstance(state['options'].output_file, (str, bytes, os.PathLike)): + state['options'].output_file = 'stream' + return state + def cleanup_working_files(work_folder: Path, options: Namespace): if options.keep_temporary_files: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 688fc5a7..609e1e72 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -19,6 +19,7 @@ import logging import os import re import sys +from contextlib import suppress from datetime import datetime, timezone from pathlib import Path from shutil import copyfileobj @@ -822,6 +823,10 @@ def copy_final(input_file: Path, output_file: Path, _context: PdfContext): if output_file == '-': copyfileobj(input_stream, sys.stdout.buffer) sys.stdout.flush() + elif hasattr(output_file, 'writable'): + copyfileobj(input_stream, output_file) + with suppress(AttributeError): + output_file.flush() else: # At this point we overwrite the output_file specified by the user # use copyfileobj because then we use open() to create the file and diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 6c730025..f7d594a6 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -357,6 +357,10 @@ def run_pipeline(options, *, plugin_manager, api=False): if options.output_file == '-': log.info("Output sent to stdout") + elif ( + hasattr(options.output_file, 'writable') and options.output_file.writable() + ): + log.info("Output written to stream") elif samefile(options.output_file, os.devnull): pass # Say nothing when sending to dev null else: diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 254410c1..de7ba97c 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -337,7 +337,15 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]: target = work_folder / 'stdin' with open(target, 'wb') as stream_buffer: copyfileobj(sys.stdin.buffer, stream_buffer) - return target, "" + return target, "stdin" + elif hasattr(options.input_file, 'readable'): + if not options.input_file.readable(): + raise InputFileError("Input file stream is not readable") + log.info('reading file from input stream') + target = os.path.join(work_folder, 'stream') + with open(target, 'wb') as stream_buffer: + copyfileobj(options.input_file, stream_buffer) + return target, "stream" else: try: target = work_folder / 'origin' @@ -355,6 +363,9 @@ def check_requested_output_file(options): "is connected to a terminal. Please redirect stdout to a " "file." ) + elif hasattr(options.output_file, 'writable'): + if not options.output_file.writable(): + raise OutputFileAccessError("Output stream is not writable") elif not is_file_writable(options.output_file): raise OutputFileAccessError( f"Output file location ({options.output_file}) is not a writable file." diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index a9fb047c..9df821f2 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -21,7 +21,7 @@ import sys from argparse import ArgumentParser from enum import IntEnum from pathlib import Path -from typing import Iterable +from typing import BinaryIO, Iterable, Union from ocrmypdf._logging import PageNumberFilter, TqdmConsole from ocrmypdf._plugin_manager import get_plugin_manager @@ -35,6 +35,9 @@ except ModuleNotFoundError: coloredlogs = None +PathOrIO = Union[BinaryIO, os.PathLike] + + class Verbosity(IntEnum): """Verbosity level for configure_logging.""" @@ -127,11 +130,7 @@ def configure_logging( def create_options( - *, - input_file: os.PathLike, - output_file: os.PathLike, - parser: ArgumentParser, - **kwargs, + *, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs ): cmdline = [] deferred = [] @@ -171,19 +170,31 @@ def create_options( else: raise TypeError(f"{arg}: {val} ({type(val)})") - cmdline.append(str(input_file)) - cmdline.append(str(output_file)) + try: + cmdline.append(os.fspath(input_file)) + except TypeError: + cmdline.append('stream://input_file') + try: + cmdline.append(os.fspath(output_file)) + except TypeError: + cmdline.append('stream://output_file') parser._api_mode = True options = parser.parse_args(cmdline) for keyword, val in deferred: setattr(options, keyword, val) + + if options.input_file == 'stream://input_file': + options.input_file = input_file + if options.output_file == 'stream://output_file': + options.output_file = output_file + return options def ocr( # pylint: disable=unused-argument - input_file: os.PathLike, - output_file: os.PathLike, + input_file: PathOrIO, + output_file: PathOrIO, *, language: Iterable[str] = None, image_dpi: int = None, diff --git a/tests/test_api.py b/tests/test_api.py index acc24683..c71c81cd 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -16,7 +16,7 @@ # along with OCRmyPDF. If not, see . import logging -from io import StringIO +from io import BytesIO, StringIO import pytest from tqdm import tqdm @@ -66,3 +66,12 @@ def test_language_list(): (ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError) ): ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'deu']) + + +def test_stream_api(resources): + in_ = (resources / 'graph.pdf').open('rb') + out = BytesIO() + + ocrmypdf.ocr(in_, out, tesseract_timeout=0.0) + out.seek(0) + assert b'%PDF' in out.read(1024) From c9bd87254e4fc4abaf07dfd184ff7fbca7c06ab4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 02:31:53 -0700 Subject: [PATCH 550/880] A few minor typing issues --- src/ocrmypdf/_pipeline.py | 9 +++++---- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/pdfinfo/info.py | 3 ++- 3 files changed, 8 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 609e1e72..36105fb7 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -23,7 +23,7 @@ from contextlib import suppress from datetime import datetime, timezone from pathlib import Path from shutil import copyfileobj -from typing import Dict, Iterable, Optional +from typing import BinaryIO, Dict, Iterable, Optional, Union, cast import img2pdf import pikepdf @@ -817,16 +817,17 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext): return output_file -def copy_final(input_file: Path, output_file: Path, _context: PdfContext): +def copy_final(input_file, output_file, _context: PdfContext): log.debug('%s -> %s', input_file, output_file) with open(input_file, 'rb') as input_stream: if output_file == '-': copyfileobj(input_stream, sys.stdout.buffer) sys.stdout.flush() elif hasattr(output_file, 'writable'): - copyfileobj(input_stream, output_file) + output_stream = output_file + copyfileobj(input_stream, output_stream) with suppress(AttributeError): - output_file.flush() + output_stream.flush() else: # At this point we overwrite the output_file specified by the user # use copyfileobj because then we use open() to create the file and diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index de7ba97c..f64ed497 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -342,7 +342,7 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]: if not options.input_file.readable(): raise InputFileError("Input file stream is not readable") log.info('reading file from input stream') - target = os.path.join(work_folder, 'stream') + target = work_folder / 'stream' with open(target, 'wb') as stream_buffer: copyfileobj(options.input_file, stream_buffer) return target, "stream" diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 1800f1e4..b9aad6d3 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -25,6 +25,7 @@ from functools import partial from math import hypot, isclose from os import PathLike from pathlib import Path +from typing import Any, Dict, List from warnings import warn import pikepdf @@ -568,7 +569,7 @@ def simplify_textboxes(miner, textbox_getter): def _pdf_get_pageinfo( pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis ): - pageinfo = {} + pageinfo: Dict[str, Any] = {} pageinfo['pageno'] = pageno pageinfo['images'] = [] From ad8dead7df652e6007ce6f4b1dce74c161bc539e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 14:27:27 -0700 Subject: [PATCH 551/880] Document that API accepts streams now --- src/ocrmypdf/api.py | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 9df821f2..4fac0159 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -35,7 +35,7 @@ except ModuleNotFoundError: coloredlogs = None -PathOrIO = Union[BinaryIO, os.PathLike] +PathOrIO = Union[BinaryIO, os.PathLike, str, bytes] class Verbosity(IntEnum): @@ -247,9 +247,24 @@ def ocr( # pylint: disable=unused-argument A few specific arguments are discussed here: Args: - use_threads (bool): Use worker threads instead of processes. This reduces + use_threads: Use worker threads instead of processes. This reduces performance but may make debugging easier since it is easier to set breakpoints. + input_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is + interpreted as file system path to the input file. If the object + appears to be a readable stream (with methods such as ``.read()`` + and ``.seek()``), the object will be read in its entirety and saved to + a temporary file. If ``input_file`` is ``"-"``, standard input will be + read. + output_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is + interpreted as file system path to the output file. If the object + appears to be a writable stream (with methods such as ``.read()`` and + ``.seek()``), the output will be written to this stream. If + ``output_file`` is ``"-"``, the output will be written to ``sys.stdout`` + (provided that standard output does not seem to be a terminal device). + When a stream is used as output, whether via a writable object or + ``"-"``, some final validation steps are not performed (we do not read + back the stream after it is written). Raises: ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging with the OCR layer. From c8b581ac3127e53234c727da209239390c687415 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 14:28:18 -0700 Subject: [PATCH 552/880] hoctransform: remove deprecated element.getchildren() Breaks Python 3.9. --- src/ocrmypdf/hocrtransform.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index f1446535..8ad25aeb 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -110,7 +110,7 @@ class HocrTransform: text = '' if element.text is not None: text += element.text - for child in element.getchildren(): + for child in element: text += self._get_element_text(child) if element.tail is not None: text += element.tail From 2d64e1536db759518b7f44172e1ff56e7182a0d1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 14:44:34 -0700 Subject: [PATCH 553/880] hocrtransform: refactor xpath manipulations --- src/ocrmypdf/hocrtransform.py | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 8ad25aeb..6b5166ee 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -80,7 +80,7 @@ class HocrTransform: # get dimension in pt (not pixel!!!!) of the OCRed image self.width, self.height = None, None - for div in self.hocr.findall(".//%sdiv[@class='ocr_page']" % (self.xmlns)): + for div in self.hocr.findall(self._child_xpath('div', 'ocr_page')): coords = self.element_coordinates(div) pt_coords = self.pt_from_pixel(coords) self.width = pt_coords.x2 - pt_coords.x1 @@ -97,7 +97,7 @@ class HocrTransform: """ if self.hocr is None: return '' - body = self.hocr.find(".//%sbody" % (self.xmlns)) + body = self.hocr.find(self._child_xpath('body')) if body: return self._get_element_text(body) else: @@ -147,6 +147,12 @@ class HocrTransform: """ return Rect._make((c / self.dpi * inch) for c in pxl) + def _child_xpath(self, html_tag, html_class=None): + xpath = f".//{self.xmlns}{html_tag}" + if html_class: + xpath += f"[@class='{html_class}']" + return xpath + @classmethod def replace_unsupported_chars(cls, s: str): """ @@ -187,8 +193,7 @@ class HocrTransform: # light blue for bounding box of paragraph pdf.setFillColorRGB(0, 1, 1) pdf.setLineWidth(0) # no line for bounding box - for elem in self.hocr.findall(".//%sp[@class='%s']" % (self.xmlns, "ocr_par")): - + for elem in self.hocr.findall(self._child_xpath('p', 'ocr_par')): elemtxt = self._get_element_text(elem).rstrip() if len(elemtxt) == 0: continue @@ -203,9 +208,7 @@ class HocrTransform: ) found_lines = False - for line in self.hocr.findall( - ".//%sspan[@class='%s']" % (self.xmlns, "ocr_line") - ): + for line in self.hocr.findall(self._child_xpath('span', 'ocr_line')): found_lines = True self._do_line( pdf, @@ -219,7 +222,7 @@ class HocrTransform: if not found_lines: # Tesseract did not report any lines (just words) - root = self.hocr.find(".//%sdiv[@class='%s']" % (self.xmlns, "ocr_page")) + root = self.hocr.find(self._child_xpath('div', 'ocr_page')) self._do_line( pdf, root, @@ -298,7 +301,7 @@ class HocrTransform: text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, line_box.x1, baseline_y2) pdf.setFillColorRGB(0, 0, 0) # text in black - elements = line.findall(".//%sspan[@class='%s']" % (self.xmlns, elemclass)) + elements = line.findall(self._child_xpath('span', elemclass)) for elem in elements: elemtxt = self._get_element_text(elem).strip() elemtxt = self.replace_unsupported_chars(elemtxt) From d4b704a0ae6c71e80f948c90239398e145bf3cb6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 15:22:48 -0700 Subject: [PATCH 554/880] hocrtransform: refactor colors --- src/ocrmypdf/hocrtransform.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 6b5166ee..cc7cdc28 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -37,6 +37,7 @@ from pathlib import Path from typing import Union from xml.etree import ElementTree +from reportlab.lib.colors import black, cyan, magenta, red from reportlab.lib.units import inch from reportlab.pdfgen.canvas import Canvas @@ -189,9 +190,9 @@ class HocrTransform: # draw bounding box for each paragraph # light blue for bounding box of paragraph - pdf.setStrokeColorRGB(0, 1, 1) + pdf.setStrokeColor(cyan) # light blue for bounding box of paragraph - pdf.setFillColorRGB(0, 1, 1) + pdf.setFillColor(cyan) pdf.setLineWidth(0) # no line for bounding box for elem in self.hocr.findall(self._child_xpath('p', 'ocr_par')): elemtxt = self._get_element_text(elem).rstrip() @@ -284,7 +285,7 @@ class HocrTransform: if show_bounding_boxes: # pragma: no cover # draw the baseline in magenta, dashed pdf.setDash() - pdf.setStrokeColorRGB(0.95, 0.65, 0.95) + pdf.setStrokeColor(magenta) pdf.setLineWidth(0.5) # negate slope because it is defined as a rise/run in pixel # coordinates and page coordinates have the y axis flipped @@ -296,10 +297,10 @@ class HocrTransform: ) # light green for bounding box of word/line pdf.setDash(6, 3) - pdf.setStrokeColorRGB(1, 0, 0) + pdf.setStrokeColor(red) text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, line_box.x1, baseline_y2) - pdf.setFillColorRGB(0, 0, 0) # text in black + pdf.setFillColor(black) # text in black elements = line.findall(self._child_xpath('span', elemclass)) for elem in elements: From 1ce8edbdfe8c8d1da968f8515a0875421a331503 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 15:48:23 -0700 Subject: [PATCH 555/880] hocrtransform: some text not included in output after Tesseract changes --- src/ocrmypdf/hocrtransform.py | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index cc7cdc28..7275dd17 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -32,6 +32,7 @@ import argparse import os import re from collections import namedtuple +from itertools import chain from math import atan, cos, sin from pathlib import Path from typing import Union @@ -163,6 +164,11 @@ class HocrTransform: """ return s.translate(cls.ligatures) + def topdown_position(self, element): + pxl_line_coords = self.element_coordinates(element) + line_box = self.pt_from_pixel(pxl_line_coords) + return -line_box.y2 + def to_pdf( self, out_filename: Path, @@ -194,7 +200,7 @@ class HocrTransform: # light blue for bounding box of paragraph pdf.setFillColor(cyan) pdf.setLineWidth(0) # no line for bounding box - for elem in self.hocr.findall(self._child_xpath('p', 'ocr_par')): + for elem in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')): elemtxt = self._get_element_text(elem).rstrip() if len(elemtxt) == 0: continue @@ -209,7 +215,13 @@ class HocrTransform: ) found_lines = False - for line in self.hocr.findall(self._child_xpath('span', 'ocr_line')): + for line in sorted( + chain( + self.hocr.iterfind(self._child_xpath('span', 'ocr_header')), + self.hocr.iterfind(self._child_xpath('span', 'ocr_line')), + ), + key=self.topdown_position, + ): found_lines = True self._do_line( pdf, From 30404f53f033e78111a255b888cef1028f01f264 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 16:18:38 -0700 Subject: [PATCH 556/880] Add test to sanity check our pdf renderers --- src/ocrmypdf/hocrtransform.py | 1 + tests/test_hocrtransform.py | 58 ++++++++++++++++++++++++++++++++++- 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 7275dd17..606a2850 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -219,6 +219,7 @@ class HocrTransform: chain( self.hocr.iterfind(self._child_xpath('span', 'ocr_header')), self.hocr.iterfind(self._child_xpath('span', 'ocr_line')), + self.hocr.iterfind(self._child_xpath('span', 'ocr_textfloat')), ), key=self.topdown_position, ): diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index e8ed04bb..cf128a52 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -15,20 +15,46 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import re +from io import StringIO +from pathlib import Path + import pytest +from pdfminer.converter import TextConverter +from pdfminer.layout import LAParams +from pdfminer.pdfdocument import PDFDocument +from pdfminer.pdfinterp import PDFPageInterpreter, PDFResourceManager +from pdfminer.pdfpage import PDFPage +from pdfminer.pdfparser import PDFParser from PIL import Image from ocrmypdf import hocrtransform from ocrmypdf._exec.tesseract import HOCR_TEMPLATE from ocrmypdf.helpers import check_pdf + +def text_from_pdf(filename): + output_string = StringIO() + with open(filename, 'rb') as in_file: + parser = PDFParser(in_file) + doc = PDFDocument(parser) + rsrcmgr = PDFResourceManager() + device = TextConverter(rsrcmgr, output_string, laparams=LAParams()) + interpreter = PDFPageInterpreter(rsrcmgr, device) + for page in PDFPage.create_pages(doc): + interpreter.process_page(page) + return output_string.getvalue() + + # pylint: disable=redefined-outer-name +check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member + @pytest.fixture def blank_hocr(tmp_path): filename = tmp_path / "blank.hocr" - filename.write_text(HOCR_TEMPLATE) # pylint: disable=E1101 + filename.write_text(HOCR_TEMPLATE) return filename @@ -42,3 +68,33 @@ def test_mono_image(blank_hocr, outdir): hocr.to_pdf(str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif')) check_pdf(str(outdir / 'mono.pdf')) + + +def test_hocrtransform_matches_sandwich(resources, outdir): + check_ocrmypdf( + resources / 'ccitt.pdf', + outdir / 'hocr.pdf', + '--pdf-renderer=hocr', + # '--plugin', + # 'tests/plugins/tesseract_cache.py', + ) + check_ocrmypdf( + resources / 'ccitt.pdf', + outdir / 'tess.pdf', + '--pdf-renderer=sandwich', + # '--plugin', + # 'tests/plugins/tesseract_cache.py', + ) + + def clean(s): + s = re.sub(r'[ ]+', ' ', s) + s = re.sub(r'[ ]?[\n]+', r'\n', s) + return s + + hocr_txt = clean(text_from_pdf(outdir / 'hocr.pdf')) + tess_txt = clean(text_from_pdf(outdir / 'tess.pdf')) + + # Path('hocr.txt').write_text(hocr_txt) + # Path('tess.txt').write_text(tess_txt) + + assert hocr_txt == tess_txt From 1257419465c04de4a3c54924bdcde38ca0f7dce3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 16:31:06 -0700 Subject: [PATCH 557/880] test_hocrtransform: this test is worth not caching --- tests/test_hocrtransform.py | 14 ++------------ 1 file changed, 2 insertions(+), 12 deletions(-) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index cf128a52..f4b0a1a1 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -71,19 +71,9 @@ def test_mono_image(blank_hocr, outdir): def test_hocrtransform_matches_sandwich(resources, outdir): + check_ocrmypdf(resources / 'ccitt.pdf', outdir / 'hocr.pdf', '--pdf-renderer=hocr') check_ocrmypdf( - resources / 'ccitt.pdf', - outdir / 'hocr.pdf', - '--pdf-renderer=hocr', - # '--plugin', - # 'tests/plugins/tesseract_cache.py', - ) - check_ocrmypdf( - resources / 'ccitt.pdf', - outdir / 'tess.pdf', - '--pdf-renderer=sandwich', - # '--plugin', - # 'tests/plugins/tesseract_cache.py', + resources / 'ccitt.pdf', outdir / 'tess.pdf', '--pdf-renderer=sandwich' ) def clean(s): From 06ab114aa80b046807188443bc2db4de7be69ffa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 16:31:34 -0700 Subject: [PATCH 558/880] Update test cache --- .../pdf.bin | Bin 4010 -> 4036 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 3501 -> 3501 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 2962 -> 2962 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 4068 -> 4068 bytes .../stderr.bin | 2 +- .../hocr.bin | 10 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 2986 -> 2989 bytes .../stderr.bin | 2 +- .../hocr.bin | 316 +-- .../stderr.bin | 2 +- .../pdf.bin | Bin 10249 -> 10251 bytes .../stderr.bin | 2 +- .../hocr.bin | 1784 +++++++-------- .../stderr.bin | 2 +- .../txt.bin | 6 +- .../pdf.bin | Bin 11243 -> 11311 bytes .../stderr.bin | 2 +- .../hocr.bin | 1992 +++++++++-------- .../stderr.bin | 2 +- .../txt.bin | 169 +- .../pdf.bin | Bin 11513 -> 11615 bytes .../stderr.bin | 2 +- .../hocr.bin | 1982 ++++++++-------- .../stderr.bin | 2 +- .../txt.bin | 165 +- .../pdf.bin | Bin 12456 -> 12553 bytes .../stderr.bin | 2 +- .../stderr.bin | 1 - .../stderr.bin | 1 - .../stderr.bin | 1 - .../stderr.bin | 1 - .../hocr.bin | 358 +-- .../stderr.bin | 2 +- .../pdf.bin | Bin 10242 -> 10251 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 3610 -> 0 bytes .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 13 - .../hocr.bin | 54 - .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 9 - .../pdf.bin | Bin 3073 -> 0 bytes .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 9 - .../pdf.bin | Bin 3610 -> 3626 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 3610 -> 0 bytes .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 13 - .../pdf.bin | Bin 3309 -> 3310 bytes .../stderr.bin | 2 +- .../hocr.bin | 762 ++++--- .../stderr.bin | 2 +- .../txt.bin | 18 +- .../pdf.bin | Bin 5909 -> 5972 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 3610 -> 0 bytes .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 13 - .../hocr.bin | 18 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 2860 -> 2853 bytes .../stderr.bin | 2 +- tests/cache/manifest.jsonl | 136 +- .../hocr.bin | 44 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 5173 -> 5225 bytes .../stderr.bin | 2 +- .../hocr.bin | 30 - .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 3 - .../pdf.bin | Bin 2998 -> 0 bytes .../stderr.bin | 1 - .../stdout.bin | 0 .../txt.bin | 3 - .../hocr.bin | 126 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 4229 -> 4291 bytes .../stderr.bin | 2 +- .../hocr.bin | 72 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 3986 -> 4042 bytes .../stderr.bin | 2 +- .../hocr.bin | 36 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 8535 -> 8641 bytes .../stderr.bin | 2 +- .../hocr.bin | 976 ++++---- .../stderr.bin | 2 +- .../pdf.bin | Bin 8060 -> 8230 bytes .../stderr.bin | 2 +- .../txt.bin | 4 +- .../hocr.bin | 112 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 5744 -> 5766 bytes .../stderr.bin | 2 +- .../pdf.bin | Bin 10842 -> 11558 bytes .../stderr.bin | 4 +- .../txt.bin | 2 + .../stderr.bin | 4 +- .../stdout.bin | 2 +- .../hocr.bin | 10 +- .../stderr.bin | 2 +- .../pdf.bin | Bin 2798 -> 2798 bytes .../stderr.bin | 2 +- .../hocr.bin | 412 ++-- .../stderr.bin | 2 +- .../pdf.bin | Bin 12674 -> 12736 bytes .../stderr.bin | 2 +- 119 files changed, 4818 insertions(+), 4936 deletions(-) delete mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin delete mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin delete mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin delete mode 100644 tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin delete mode 100644 tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin delete mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin delete mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin delete mode 100644 tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin delete mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin delete mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin delete mode 100644 tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin delete mode 100644 tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin diff --git a/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index fc70a6311190482b159ffcb0654ea99c0175220f..86270f4985ce464c4149388899fd2a2a7b326a26 100644 GIT binary patch delta 1349 zcmZ1_e?)%6dJZNtqsbdNRO(OXshIQZ`K|qhd9lZmc@6>zw%ho3vd*}j*W2H$)eAEubG zJn7Q0Rmb_i-u^y&r=+X)zSSJBK6=-N?w1x`W0>$S;Qp1z{!dM(Jh%N9crBv0Pq%)z zzxU!}A=7uQ*~?$Mvj6v`7cVYfR=@E4_P_lf&dGiJ^v~kR8BJ8)``x!thwDefL$j2N7r#C z_V6X{^}Xp9IWCchQP#CWH&E%3M-TjHzbYvrGFUOFdWu5_re(?W=KVn*Jl zvM}qE_Py5*^VEhtTbiYDj&IWCp8eZr-K?$edcd&7z0`TGv-K2luhlDWbg69ivGjOq z>AUIP3xTG3jy-GMhwPlYYVD0j9KJUEzKJ`T5=*`awM#Mad$u{>o~?E8b%L8k^wpXx zQ|>I^`fibuPgmR$`-v?YrxnA)O%L2rH+A5c^Wx#2HffIclYHMxGptw}eMR)${?&VD zdYTpNpLu;!hE#y+;v>_ZWv;xw?^k9~Q)0zxp1$^3&-O_CYoA`fp=EI?`%Cjz=3jRR zzhsta(E3<-@ww3DG@HWPNt3^w5xll>Vr1xo8GjeV&JNLJe`lC?Vy?Yl^jc+0SF3HC z#5z;1XHRYq-hH!hFH6?9P0y|!^xVyQa0_pXfppfA-zs6*Xl=r z@Ml3$tKFq3vzN{bihZxvdaUo?%48dx^wKP;ZF45s z9d*4KtUV?6ROmUS(59=Q3j=l2*KO8!)}Q40f>$+CbK|7c3Fr3e_b6<(;5uzsKf82B zidf}@5;eZtPDbIqE!HMKnz_tG+aq~H4vQwO64g(wmMC7iU`5boqSpgYRi19=XoA@m2>u?ongJ>hF^BQ@fQm0zcR=#pIB^~^>wZ1?PEzh zdmpO$>?=QhVEN^YYo8-dw*I`oJa*xoSp_0p5{}}BCHbQ5JRVQx7VEXQc&Gc6bJ11X zn-T__#IpT&PBpgs`kT#SfyK6&TXjT(n_`4Z9;WS>vG>)Vqc1(*UYx*cIy>yv95dwu zhnW+cH`R;3{4v#k>co?uOY}7E+2%ez=^(#pNAFs_y48xSO+NbPvK`v^HSQk&%SU%^ zTL)?u%`~0<>vgly$AgM94xc^uXvTr(HxKIbEnwj>uq@&>zp8z6Z?Waaou2V;YfoJ| z^wRH`p2nQ1SGjqO;UWlGsvhlAAm%sNGeyOAP zS)W;?L>5f1^Znkl?>txD?D8}DlViR!?C4V46d3-cG2O;+y0vp6XSwLbm2a-eE-Ngu zAW1y+ukR-vaX27Y|vQnmSYoZj?*rG$^DzZ^ZaLw zS1z5NnwL^sQk0sQ%axipc_VK&v!R~hWO+Vu9wP$-Gb1BIGeb)whHz{0Cx(M(EtDd delta 1329 zcmX>ize;|?dJZOIv&kDdRO)y4shIPe`L6wiS^2eNkWhlejq5Duczn_5%CPShy@&7Vh%Ipu>vqkmlaQaY^1sn7p>6m7{mHAVuB_hU^7+T@A8-Bd9Z5L2 zBxkO9|NCdl^UceP+jRFy@+ZjspK<@}`EwKhAN`Xkw_j?@Lb-20PFL`}S>khVYJKD; zLv!O(W)C(T3RyeH`@sAY^Be8$zwN)6{<-dFjZR@xZ^nunBB65v9_ysekxOiO>Am~@ zV>tI z=(oqAOkVwbIgRfozVbb*tZU$vaPaCAOM~~{>$m7FVw~MQ=k1~yDmg3VgW681{JX!< zt8#VY_oG(o9|JQY@7~sodGz7lW-qhS6FUR=(zgGujC$jmd3d(ifyFX{hdtJpEE6g! zcr|T%S@+}fVUda~?B)q4AAPM&@0HdpTaS%Y5zd zb4;3bStfI9y)0|v-2Uk8DxDc~MCUHOalXhbdEVx{^4Ae6eBa)tIIFb?-|zkZeHy#X z+aM3NJ!?ZOR&X8)VY%UD^F*nfr9s;0a;f;YHTFy6I+~AVHy`=?{6+AjQ&+SfWmzvf z#L%+1KPhyIXk+GDKR?;&{0TUmasA1)ebak%1a;56-Fl~f2SfT9-pTRq$~7Ft z+&3l6XL?k(Rla&)sy1zEpIboDBmXBVpDuPCy&$RW?s|3h!RK3(A4C|xdsz2R?3JSR zoWP4o#wsRHA~~CVxF=m`zP9l-SEPB(iL|3{ymQm$8)UezjtdLgyNby%cd?dkYI&jY zqDwX{LY>x~|4LfkR%rz}*01`uWQz6NhEqX}s~QZ~1#Jxe+`(}Frn-XmQ8lR`Ytx0FT~P7ZZKO3al7iHr~T}jzE~!;>P(6KJ5$p<{ldxjnEH)k z=kwd9>{${Qwsl_L?|yZ+XSo5opLlk$|BBa=ojfZ$>+DSKW!K|nZu|6x_21N6;8Yf_ z99_77g48?ZIowtG_W4aU@7NC|w!dh7AANqxhDTc3e#<^DQH4Baahz16$Do8{c*HTf>;`k!8!Iw;A%4x04Z zL-MrGC-zg9CtrAFG3C>)WV7~9b)7jzhYbCG_U5oxG-VoTfH-dmA{ACg49=9Vs)wPc0P>Pt4_Pv3H{ z_xsG<@$`^E*PHA4Lig@&iDvt&du_jZn}*xu{k-LB26_gXTnY;M&PAz-C7JnoE{P?n z3K}j}Mh1qK2BwB)h9;&aCI*vZ_+pt14JPmA+sJHcZaO)i-WEvfqD(A delta 52 zcmZ20y;gdIB9FR(CYOSOzH?D(Vo7Fxo=aj$s)B}#m63s=rJ<35rJ=E*p`n?sfyHJI Ho>WEvfoBb; diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index d4784957..16b617e5 100644 --- a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.1.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin index 2e6ae52c9cea302efb0e22932a631b176b420af8..71e9d0fd8badde53666eb0c1ba25af490729f3fb 100644 GIT binary patch delta 52 zcmbOvK1qB78<)DFCYOSOzH?D(Vo7Fxo=aj$s)B}#m63swfq|Kkk)fHPrIER=f%#@# Hu2e< - - + + - - -
+ + +

diff --git a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 8f965ffd680a9ddf70de6a811922f679238f0a67..4601a61056deed113e83ae5dd0077b8590e8ef60 100644 GIT binary patch delta 276 zcmZ1_zE*t01`bBk$s0LT>d$(y8uGM$|1473C_VYva!CP4{app>CF<*@?B2~C`|C@Z z)nZ)_PRl8C^q2o-DQtQ>rQ_cH^&d9bY;j#s%lKWJP0juDl($^w(}m}9%vl(_KqHa2 z{TuK40`HKF)z%fw-*`?Z|8_TGE!VM{xMT7Rk?cv<{9xnJN~zN<3-@O-#j)VD_Z zR{7hR8_spc@NIr#cWt9f-}LFu^UE9V_}Bkt{&PAueDYGxa%Mw4!^!eo;ygwM24+S^ khGvGACdQMUxZ)TsCokmM%xE$>l-rlXgiBS`)!&T^06QIau>b%7 delta 291 zcmZ20zDj(<1`bA}$s0LT>JN4?8S=Ef|146;n0(13lT9o!{LbNfE%%O0ess23?)%HL zSFDshI60sG-~aC=?-7n=lkY^^$0pkUx#GCsUgK?1=E)1AdcJbKKHU|@ax6f9!rVab zwqN}0Ka{+-G%8=Qmni$%yh!ql!BcZCb1v?OCyZT;CstnZ2wB3MdPMAjOxDrBkV`uH zm!9r7%>I%a^3E(&<-MKWi#MCk2Tdu971Eh&Z#3(M(qp|u*Yn;iUsK;Y32Xgo+44L` zaP^*~8GCv@^64M*ymHN__MY9@Pk+Dl^Phekv2*eY&T=&aJp)ZH1qFTQqSVBa%=|o; z#FA764HqjT14By#Q$sUD6H^l-)5)$}ag1h@mvC)nG@Km4?aN`vrK;-c@5TiH_&0wR diff --git a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index b7414ebc..ac662824 100644 --- a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -4,19 +4,19 @@ - - + + - - -

+ + +

- + The LinnSequencer - + 32 Track MIDI @@ -27,27 +27,27 @@

- + The - LinnSequencer - is + LinnSequencer + is a state-of-the-art composition and - performance - tool - for + performance + tool + for the professional - musician. - It + musician. + It is

- + extremely powerful, yet @@ -66,7 +66,7 @@

- + ¢ Operation is @@ -81,18 +81,18 @@ RECORD, FAST - + FORWARD, REWIND, and - LOCATE - controls. + LOCATE + controls.

- + e Each of @@ -108,7 +108,7 @@ track may - + be assigned to @@ -135,7 +135,7 @@

- + ¢ Ultra-fast 3%” @@ -147,8 +147,8 @@ in seconds and - holds - over + holds + over 110,000 notes @@ -164,23 +164,23 @@

- + ¢ - One - or - all + One + or + all tracks may be - TRANSPOSED - at + TRANSPOSED + at the touch - of - a + of + a key. - + e Exclusive real-time @@ -190,7 +190,7 @@ editing FAST. - + * Exclusive REPEAT @@ -216,11 +216,11 @@

- + ¢ TIMING - CORRECTION - works + CORRECTION + works during playback and @@ -233,11 +233,11 @@

- + ¢ Optional - SMPTE - time + SMPTE + time code synchronization. @@ -278,9 +278,9 @@ then play your - MIDI - keyboard - in + MIDI + keyboard + in time to the @@ -294,14 +294,14 @@ sequence loops back - around - to - bar - 1, + around + to + bar + 1, - you’ - ll + you’ + ll hear what you @@ -316,7 +316,7 @@

- + corrected! (Timing correction @@ -343,8 +343,8 @@ track - — - existing + — + existing notes are not @@ -358,10 +358,10 @@ FAST FORWARD, - REWIND, - and - LOCATE - controls + REWIND, + and + LOCATE + controls may @@ -385,8 +385,8 @@ To overdub a - new - part, + new + part, select @@ -401,8 +401,8 @@ record, the - first - track + first + track will play in @@ -412,12 +412,12 @@ you - MUTE - it, + MUTE + it, or SOLO - another - track). + another + track). In this way, @@ -431,8 +431,8 @@ be overdubbed! All - MIDI - effects + MIDI + effects are recorded @@ -440,8 +440,8 @@ including pitch bend, - modulation, - velocity, + modulation, + velocity, aftertouch, @@ -465,13 +465,13 @@ To erase a - wrong - note, + wrong + note, simply hold - ERASE - and - press + ERASE + and + press the @@ -480,16 +480,16 @@ be erased just - before - it + before + it plays in the sequence— - when - played + when + played back, it will @@ -512,15 +512,15 @@ using the SINGLE - STEP - func- + STEP + func- tion. To overdub - notes - at + notes + at specific points within @@ -544,10 +544,10 @@ use LOCATE, FAST - FORWARD, - or - REWIND - to + FORWARD, + or + REWIND + to find @@ -564,8 +564,8 @@

The - INSERT/COPY - function + INSERT/COPY + function allows you to @@ -580,8 +580,8 @@ another—in the same - sequence - or + sequence + or a @@ -603,15 +603,15 @@ between the second - chorus - and - the + chorus + and + the bridge. DELETE - BARS - operates + BARS + operates the same way @@ -619,8 +619,8 @@ remove - unwanted - sections, + unwanted + sections,

@@ -636,12 +636,12 @@

One - way - to + way + to create a - song - is + song + is to record each @@ -657,8 +657,8 @@ 999 bars). Another - way - is + way + is to record @@ -667,8 +667,8 @@ basic section (verse, - chorus, - etc.) + chorus, + etc.) in individual @@ -678,8 +678,8 @@ use the CREATE - SONG - function + SONG + function to “chain” @@ -687,14 +687,14 @@ them together. CREATE - SONG - will + SONG + will then automatically - copy - all + copy + all the parts into @@ -714,8 +714,8 @@ few bars to - repeat - infinitely, + repeat + infinitely, for a fadeout. @@ -757,8 +757,8 @@ the - LinnSequencer - is + LinnSequencer + is designed to let @@ -767,8 +767,8 @@ record - and - edit + and + edit while devoting your @@ -792,7 +792,7 @@

- + * Simple, easy @@ -806,8 +806,8 @@ clearly guides you - through - all + through + all operations. If needed, @@ -818,8 +818,8 @@

- HELP - button + HELP + button displays additional explanations. @@ -828,7 +828,7 @@

- + * Non-destructive recording—existing @@ -839,14 +839,14 @@ while recording. - + ¢ Two FOOTSWITCH - INPUTS - may - be - assigned + INPUTS + may + be + assigned to remotely control @@ -873,12 +873,12 @@

- + ¢ Iwo TRIGGER - OUTPUTS - may + OUTPUTS + may be programmed to @@ -904,14 +904,14 @@ or Linn 9000 - sync - tone. + sync + tone.

- + © Utilizes ultra @@ -927,17 +927,17 @@ FAST operation. - + * - TEMPO - may - be - specified + TEMPO + may + be + specified in BEATS-PER-MINUTE or - FRAMES-PER-BEAT - at + FRAMES-PER-BEAT + at 24, 25, or @@ -959,11 +959,11 @@

- + ¢ - TEMPO - may - be + TEMPO + may + be entered numerically, adjustable @@ -987,36 +987,36 @@ on the TAP - TEMPO - button. + TEMPO + button.

- + ¢ TEMPO - CHANGES - may + CHANGES + may be - programmed - into + programmed + into a sequence, with - smooth - transitions + smooth + transitions if desired. - + ¢ Any TIME - SIGNATURE - may - be + SIGNATURE + may + be used, and may diff --git a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 6e958fc5a44e28c784e79c0ee82bd52d8d8efb88..ec8b419ea1e3a4a16e88c88f5196802e2d7d3b59 100644 GIT binary patch delta 7615 zcmeAS=nmMhk%Q4}@Nb$pvK=3AR=5*G5&OFDPpZH%l|` z{&*?geo~MKzou)-lecn*Gh*(%TGi8j-J;)`oC%WX0{vm$Jf`UY@fRS&;8@|Hvf*!sDIGg z>Hd@Rn8oMy{r_wJ{Mz1L9^LoUZ2!OCiM_XTiXMOdx46b^`h5BG^EcS9_uclV<@x;E z`!C+N-tj)aKmXdVs`vTlaxPwHzp(n|udI_fWj0y*+i!dYf06V?Ur&I?-({k6yM#oN!9RJ_Rd?$pT7isJTI7eXZv36r5U{Y z9QUT)dbrqqYQ^sqgHMyi?53K1{}W)fjpcV({?tF853hVb{ddi~#Gdo^>M>u^|F5b3 zXS;9X+kTB{1v2$(+*Bq-J7ivm^{*mD)4XZX8v5gX4>DiUU_wUe`Ov2{H>Gc@8Fjq>Ufd->K}^z}-R0&MAP5Q5pFa5bJ0Eg?7J!7-(O1jmT~%V_rW*OWda{_nU(9>Yk$6TQ8@8L<8j-= zSM|$3O%)NCcCXSxNdBO+Ut8>zTy^=h>0Ec~>OVaDtk=4f)sO4!j8`ix+^V;1>#da( zIqhfY%aL5%r#|7(@u+(WCOqXlAEq4qaXHo@XpZv2_M~XW5@W|6n>v47RAZM!B`+}(c8TwR6lGw$t;jct6@|8@6d zJ(+^HmmXSLKbUp5JD25O&E(_f_xW#ozvY9*&RZXyUb5x=V$8a|(S5VgmJ`Ae(cko^ zRh(ULZClHN7IXQO`kK}axt6jGvHS@pPSTeTXtLgCKQN8ixoM-?uMER$wqDH953(+O zmN+{1`zNKRwztlUxN%h&L|V<(zR~+{mBNdSKbZdZ6(??4`{G^HokbOKLg_q7)0ZFN zK9C@Im2sm>OU;+0B{d7JMWY%5s?TxsC2e#}OuV$<&j0@m2W~8sfA3x|RVDkmETU6q zi~QUvMeBcb#~nFvHT;yz#ZvLv8;35AqI2|5P2RWLE` z7Up2z%lzc;)s+k{qFRkNO_=h;`|p$8pAYe>USZ{JxU^p9{S(nynk)N@N;V!!pa1l0 zwX!S6MGo-`tTrOkzokDtY@=Ry^lAOC*pJNlHx9R~3zm?YG{v@NLmP*4*_z)gy;w`^ zZS;?ZR$8j zGs*q`ru+G~6)CHZ&z9SpzEPcNXHkOIr8^IU8BLmMTpnvGNvt$D;j_?I-DLKwg|Uq6 z_1}cQD)6+=^QvJ}{@hcs)cR4*${58njrTUhiETAJ{&0Ku=B<%!%$JswdL2xOnbzk} ztZzHt^F@hbisolufv>BoC$mjrUl6j&r?GJ9srFB=cek8mS)1b1YI2Ca* zrYiHyHm0`|-kw^)r66E;QYWk_LYtfSQq;do`)V6`&xWZ#7CTlpX{y-y{mdNSPH%Y5 z5q>7oe_fa9ZYS#p7doUTyUbf&5Ou+w!~DWIC8jhXd2H@74Ak6k1|@z4!*_roe!C4RRH0q$3Y= z%-k#>Ua~l->;CRt>2r;2_w7_#^C)tL%S0_Lu~`$O9XNl#^Pjj?@?xNMuL{$dAnYI?kj`X=3t&biNOZwl>A>Sk}9&9m!Z z$?7d#pA&3?o;2Nipq->yaMVNCv$3~uhw`qgdbJA+f+yx`tvu2i(2=;**J{DEA8yf$ zCeIZ0jxOt6GGWh_&+ke!au&*o5~I0m$tua5k}=gp}Q zHrHgH%TJ-yLvL_qW9C zu+m#2>+0^txJV`TNg99M%x_M~&*v>&!0H%%ddt7oxA#M0qBHmFXXQ{vv@+m4`2_4n2CSdfM!q{ZFNArJ17AdL{`Zz4eYb zc%_j4vd;^L2K5vUldcO7(yRmRY?3!J`lMZb!l0R=@^~xvS!td64Qo2PmohKg_i#hv z;zO)!ZCqr$>wYRFru#iPz`y^UfAp55kdMcwCDvviefW*(*otTRuUalYI4T#zuXA(B zZnc<})g^t|yFcuRx)`e^!?N%DHjQq-CDY}S0)I7iU2K<>O*wn3wKba0`l#kZxp|HD zCuVfDyDc}^GWo}JP7Wuo68HM`T8d_SC*asE4*` z&llag#Ge>fw`tu%!OPa0OFQ39bjdPV+AKMvX{qp`MHe)@#8%{J7golxg^T~?l*_lcd!DT;H(GyhQoZDw z6(O+-9mO6Yi&h*~+3d(Qmxa;I;f_L3o(r4x?&oKsp7r$wtSigk-h6bTY2WGT5f?41 z39%uA^>al%!X}Njo`;)F+ zOm8QB$!?bvU02L^%P^%r{?VximsGlH10=Jhl0*5m*tQj24EY%6nLkDPv&<`x$xShb z@Z{4l$s7{}{CPw?Ga6rh4KWpR#n7T`gBp#P3FG!!0>8+I7b7p4zHYTOarvmJI zJr=#{61=+a4eOobGuRd#3$Jb1alDs9s#>S!k$jJQVu`~;9~X;2vxaYun)Y=)%(1g( zo?dc6{<-h$jhAMx30`jgVeN$14&NtiIpbOLOs`+1;95c0h5957Tjn1Z_gqllSulfn zClClnWvx5zSCxJtHU;@JEl6|Q2g2Hhi;tZn2>Ji%*?Xo z)x?E@{i~AZo9~_SBD;T0%Jn-}W|%QqhsG+o{C8J(eR$oX&eczLh0I=VCDoTXVT{)b z=6K$kxpl?O52q&-o$N52R6pnH3Im;h-~Tlanu+XLAil=^(l__Ro4r+*$-lBxeR0g( z@L(~+r`T=Ur=~5x_-=js9fkNag@?hknEr(Buqu4eMhXvqp)w&8%vJBdTb zCEPF9$tOM5_<7i{?PR?)n@8BxU3EqtQDF-% zg|$^D&t;iq1ibvu!ZEA7@!@(V{kv&ORW@(dTs(8|NoHA|zkqOf_|cY&&s(?4i*ytz z6#aH%^e(vi`idai1F_;>D>>g>wexctgWFEU`mQUK^03LgEM*;aVNM|5%nH%FoF}HP zO`gBRzP@Q&N6g}hf~ziuOp!gboMD0E$Mz5=FTKg4wZTjLnUcS&`h;=)h}u5GO@Uoz z_L3K!S6=8WnRobYA)nLki8iul6cYuKs=R^@Tv_ir{nf$ew{{rMICT4mrgvmNk7?ul z1M{jBgdH=NCkoEGvd=OsK;S~&Ea|Y)Pm@w)6{7j0YP^2b|M{kKWvhuOhE4@imQs#A^nYD4!q4Z#$Vz&>A7Mn{}tTuKI}Woo%VDDSp-A ztyA}=BwAd!w`PO6$J^?}S0A3#E_8imn%~KMqH1xt!j~}D?{PkVn4~>~I@ovBgmO3( z%X>CdiCJdb#3`={Ev%ouW}}T0%P~d2gMU23kBF|H_+H)Wcfdxg-)c;UR3|Td@hnU% ze`9LJN1YE3GiOE#tczbG#IW0XYM=d+!+krGtUFT+CYfq)UZwfTf2F;`_vr`bvi6?* z`A5hjM$fV$t^YprB8yjcvl7bRzG|%f8N_?p_2#v&I*GqlPO%S7v0L)Kx8BEP*N0o| z8d3>`WoKQ_g&ip8TD3D*a7*gP3DGGt^@Z8b`CQEI_$51gt5MEY0VdtUx?61$WnU|X zbsGyh*$XGS-gPosCH3|0@*=jKsu_2bjW=Ikeaz!U;x1ozu?K6P{9sf4*?MW(t5@M~ z{xvL=TbxI9#VG@7XcCr<`{8U&xp6OxbtC ze#aLs)6IpuuP>kQbkTC2#)BdnkX&>8! z3ssNO#1(nu+#Y?nvUBsJSz34c*BQEGhd*`Sw6>~^Rl(pgbc+|Lt@!prZ#gV0}U%aa;+5P*8 z#vMITtp^*L-c8fKw09M==4zKeZgU+>RrbbCT6HnGe6NQ1@>LgpIUF*)lss2FL{d{g zHQcxS_3X(HzZf?(NiaWtv36GbyK~DQ{+WMDwjpov<(q!>e}BC0+il%u5aZDL?qWdY8eQm(ryV|M+Rm>OZ?K zVqs^~`@KC@8-9IY4N&)s_p0^$oN)RC%R;5!t$BB6EBxS*(3z7nq3=WQt>-a2Ezb6= zk-IgQ>vwX#$GL7%kEiM zE|nJi;#Md-@pi#+x92T2R?~xc4HxrGojU#Q)3eqlMy(|(XA&!7L!JiotPxvriDw$a zugJTHWUXE-p5p6n_kllu%X?W*v2KmeEoN%6zh?4?dBpZGTeUUM_$60fe={rZc-?YQ z!&SD@o9Eo?2`TToZrHKxh*XQL-@5V?ua_@Y@%98qy*co)FEP!cds0@}(|0lI4;Gp4 z+5BzW^?f^6v|Z%6bmWl1`3l)7Zrxj^#%!H;>GH!pQzKUWPs;OSs%|p*Qnh$%=_1nz zzqOA{)0?zIOk{7AO^kl48h^S_D?_G!5!=qgiwy7hE7zY9+b{RidY<#M1MNDidA{yA zwypQIQ|AZE)0<-oHfZ0y-c#XjJZb)&lUo9v0%z*(5wxFkOm+3&jgK?f@7iqC-~RWW zfnQwqp1kjo#fqCeHrTZB?=F959w+;H?{0~Josv4U?bA&DM)`6qla>4#P}LyOJKz2D<6>REiSZj@lxTw`}jdW zU&C(S-#_o{>^_nkwbi^}**96Q_5hQtKir!>^4`ek=Ggdlk*u{-@h$b<|I8mPcb`1q z%|5jw*k8c3e(Kc3%lkJ4+Z>slkk@L&9>^i|vW2aYY z(w4D&nZ@!S_3HJ!7P~*@y??dJ|4UPI$(&7(w=@4)$UILyFrPDs6L z+I#+*LZiQy32*la8*zP3ndj$3EM-=$@Aa8w|Htm<0{6pjyWAM(h}O5CEk?A4d zS8m?j?~@K~Hohuw?6BnA#GvxPLko8$Fl-J#KD}pE+lw9x$(<+ljodd#TPC_M|Iqi3 zb(X@XS(cH@4(g__&8__;qj353qD!kMGw=Cp>U^=}!I_e%9i=^@UpCCPdQ$#N%HvCa zaY$&}8UtlLPOYuxR*{$bVedE56WcX05>Mp0{{$Q@K_-7?f;b&9LE&3w`OESVI2%U*{NJyxE z61k0y(^AK8o8~z&|3~xZ`z~WOvwgU_)_q!#>hAe)A_!g_>E8N zF8?VM>k``Nq_+Hi55wL z;Z?DEVWnl3OSRR`4%I2UZQt&1n90I(?nSD=v-;b~hrc%L?5vL$1+OJtNp6aw`xklH%pZDb3kxrY>iLtk&Q_?ct&6#-YIUl#m z(wgI1AEUo~@A;D7b*kXSlSi*bR(85=Y5vIh{`>NS@ykmjpPJqKJ=wIWh&52txp${w zThO7E#lP=dujgJDB4fvUw#9FYac91*b<8aK&beflS^&FJ?X!w$8Osn!Os zEPnQKJ-Kn+PI_-wLF4Vo`fg$$E4Z%IsI1sf)cAs9;)RT_o5LcUvwJ_CK6>qaFt65^ zuMU+~?K=COdN6wLJu2Qi>DEk@!^^|gXze+meV?HLQF8-?86<67$e(;3F;*KdLlP5*|YGccoo9pa$$}*xm*YDp0zD9*x=O3SJ z30qRtkRCnlndHZgqRkeQ5|?iJ6`gkMp!DyWdAp}OJh{pk#jIU%QO{|n$Ue@ykCyVq zRHR&H)4NxG@+kkI#3v>>m+H5D3%(_Jly}x!8{ZtKW6f?Kmps^g+@r^C?V0|#q)jaD zF7iP)^{ak=zxnjve~WFe>?X8cJ8|{9*q+OEufHvGyf2ll5Uf-v@I&h&_mw!+4%T^A z+qQZ!cG`WNvfVD`{_l&`kIS`#-}pa2z`yl^mdftsZtK@SoGj~dSzyV((na^YV(NdX z%b4!)kH6y^`KvU{B1-=K)c(aYbT7WOpT#owwe!~N-{ct>82__GwyC<+rHL+Wyq$V_x0^AvdsR-mi+Zftcct$ z;f|RzEButdsBj)z+ZFaD>YUWF^RsVT z7PBn-?|*)g!UlfJTGig)J2+*|{Gw`;k^^12>b9p`V{WZ7-PtK}VCRylTkf-6E*B0? zoGg=Ff6G>ANlf*2<7cy8_uN)FSpL?kJ#2%|r?NE(ulLRlyBK`KE&A=&{QSrAXF9)s zO3=A-UDw;Qo^5^lpJ~g~{)lp^?%eHhCde|-HDmG{*T%-^|LddoUtp2a#xMO;u?rLsR$}ZOX$1HjbcnWzM_Iir`t(PsEeExWp z&&!iPZ@m2}@X0=;Xwyr<3%}m4^qecKQhAxhH7a+OQp^6pJIm`6qOZl7FJoJ{m^C?R zNrl74%nL<^_ghyus5l9}+oRa?=GwZo1%Go)PQU(^I_J$SkuTF`em^+vA@Ae^XRj^I zXpT4<$vV-iBE{-nL&?tt3No==+cfXnDjFqRUU+fRxh0Vr}AN_h=KX;W+ zu)8fWEnC36;L`84>$^UbGtM*g44KAUUaT>*VzT@6gASZq`R3(!9X`k_^TAv_>Ay8C*=g+jTW9_R~r)i zA3U&|7J4Xm?en>M??2kL=DU4bByqqyVVxYO;A`#79lfri>nuF1?p^;Q8Cf_!9N&E=UDUn z`yluCMB?qEOMk9(p7Z>6nd{fGFR#q6Z|ba_KH++}~R)^?jz*?c1MDURRdu-jaMZGyYN8-Zr>C1E&8@wh@cqv(RRP7F1$cFlZ!pB>4 zJ9CfCTCUcbGxeZ1tJ<%Rr#?JoJaaNG^5U*dy9}pnoqlBYRC$Ff-h4Zb#<<&UGjIO& zFu8JZbdnwn2u cs20m)YB))KGo#t$E$Y4;W?ZVOuKsRZ00au>V*mgE delta 7616 zcmeAU=nUAfk%Q4>@!_ z&*!fbQzphtd2-zHuH%IrJFbS_-sbah;|`Bg_0?;y?e+N=f2)80|Ih2!*Z+Qe{(k+B z>+k>9-Tbxw&FRS9Yi?_|e)#qG`gQxiPtWd|Io){u{rbwt=i=Xg{k30TU;lTrp<%tm z=DeT&f2Lor-&6Phwf2tG_|LX=|F55p+AMM2p7;IF^y+_~EB|!vi%Ij@^wZzw-_C#L zzrL@#`>Fr>?5zv$d^%^-L%~E ze>TT%nsv0btoM62@26c4mH(f7FW~fZcHMiiUpqPLPm4VioWA~s-u=T#X1nb6drY*9 zeSgdJ>G$~DoeRa!KYyS2f7>4S+u3((>nd*^*vobLnCh{wHSBM^(~!i~p4U=%#6J(wb}?JbB`8iznOG z9k*AnyZ-I|8^y~a^%>fq%XUv@-C`SYy4L45|NS@xlY>+K|MS#;D}Gpia;nsa)L9E& z>@?HeXj|T&Ve-{;hmlas|6@!EK@IKe<-hU3%5?C#OC>j9Ko{-p+NB+R`(anKwtBOkzpPzHS=??9`g#~ zdV5u7rnqEd%dFhsQ)la*?A46?yx`w-jlJ=?8~<-V?4rVPyNi2UsUYK@qBm>HzqwU% z*9XjtpIffH-}Xn~w!fwE=a||fZFWj;Kk{VRuE_s&>VlJf7VWtA)y16ka>{#|>jpdK zYi-+B!DSMhBdE-AL}zy00@aBfZPuJeZ?Dh~-=Uzq@xTry!{|~ik*#JI{n%`OZj^2i zD0cZbAzZxtgKcC{-_-j$3VWG&xYlV@$LLM>W_n*g`D|6r+GAFV|1Ex$&9zPti|<}^ z`n}pEj_l}|TCEqa(@s~Xe2B=|A2TiVR_1M%GQa=9I^7*{(FPBn z??{NsKN#}-+{wsB_Xkr~U)Vp(C1vt5&O{26(CQbW_4q=uvX1`5xnu~YJZ2|-X4PHR$<@8=iTZklQYT6Gj{N`i%p75iS34eFuHQH1jbOQ}^=`lO zo32ite`4E<1J?6bo;+2Zb-3!I!p@}R=Tkp^wsT49JH)yt^*Hwf-}i33+Y7fzpQ;m0 z^4-AfxJTbTGu5It#gnhbq@&HevAJ=fN%48Xi|dykGp^4w<@hnR;YBlyYOaE6w|6}hY4x`y7k{_GD&$kKcB z_qslrpX;^oT*JYw8!Q)|yLtMb@$cD|uX{>2)rC!dH@7rw*}2oIMV=E_ikW_UH$>bu ztCm+f({H{Fh$g$6uul&3I z_8V85q-_ol1Z_K>2t}Wpyy#zLY{bqK*Xo#?>+{@>$}Lmkn5D_X_h!<`e@9~icyvE| z-d5>I*(0|5g+tclfcxcanXwG(Ul{~mT=O%tr(fd9Di4bdoL}?=;w_hPUQlGX@RD=& z-|F(;b5={A-0;P%?qZ+99P^|}mRbxNYt3#3G}c$EmoRWJYBb4I3){WiY^z%`+tT#Z zr0D&R^3tDO+?nUupw{4)KmF1zYi?f4nDB1tGsY`2%uimnk`eu7uq60iq4N*l7LSeF zPM z-1XSIhULbVSG)h*EGa)e<EnJ2mA*ufCyJSpK4-P#%TnQE z@UT+FrNu9L{By$^Y%QQzM=kB=?MLaW%U zdq#s%_GO;aPD%O_3Hl9|ejzG@e~_gx8qLB6D|v2mU_C?p!~&N%}QQR zOY1K=E?QSrY;|l8?Mu`u6-wFsyFW7j&xw$fDZPu<@ZXU-qITS?MB(s^sYO@LYNgB! zyIHQpA!nemQ6yMyaq&)i|JuQE3U*>35h>lM8th={xKEqjGb*|TvZ+oT*G&Fiy zO$@rT(k0F_aEZd?vn9{Hi?f{M3MYqNnYv+)S7G7-(Q*WXshW~-cRxOl7Vc3mLH!H}?+S1fF^y|*on zJ){+{bEm(4>Mb7$)wOXRQwr9Ye6;2eJK?wWea?f*ic#a?y2POZK^t7XsCL;0;s8M}O@#2eWB z4~}bl^nKaP@ALc<{GZKX5%D3xMHfEi0jAdj|(07U4=`;-nT_ptm@np$<>{7 zc9+YEC+QJflQ?hWUNc%Ge6GH0%F8u#6errZCbD|&NJyM;#9uUz&-eDsW_N|l$L6NI zWudcEcuk$Cj_L)V_IuxS1iT$7{`zd`7 z&X?)WcPsCHv^D$b!SZcwUp3CgI>=|w6}h$XV9P1yB?^33{Ux@UoXC0~&N8LjvVP^B zScYzyO)_a=rat+%Fa2!Qk9apBlBxccqx9xuiNBZKjyhvMdCGQ~3vVw!|K8Jbf8h`N zZ32lJUFAW|d;7GO+<3h7dc4yyo=md~KD$j6wEW7&SBV{D+B)xxt%uI*zmh3u-X-7u zHFvASik3!^##Kt+%06&^e{zGz?78fQNti;V4KJ((<3~Onz-BJG+x+&d`wYhTg z_G(dtdQX|W=eCaQtE1%2Kr=}l&w}T~Yj_mF%&Ett%XUx)G zDQ2|P_HLi&kMCtAuQT+Q-!El)@N-$RBqJ{;C%b*#8NHrUX2+z0`Zc(}y>|KNZ+&>d z;$ScLrR5 zX`uzf!|<~q0V=wa*);jq)!g%Lbtn~9Y&!mHrL=nyvyJPV7ko{zcTql>6pehDixGPH(td;`vgq_M)5qb?FypoS5s|zPTD}weL~-vh?{I z{-ur=Zq4SruFN=JI??I5ouTI1=tF#b=D&0br6D^De})I>v5LUGR-7^`+6iM83*WrIM424NmbMIeDw{X_0#u=j>UUIzR-7H>rzhmd|D_6El z%xI}_>b>@0N6)E=f3_dGF!B1FW7|c4^scUXbwfpw` z7jAUZ@M-N$dhE(CVzkmTN6TqbYTZJkL(RJ)g=7NfY>JuD;K;*VA+^P;U|zcW>VRhP zLdk`Bp|__oo;}hY=u{qi$mFu-iUdF570NmF3)e*jOza78=bzoRAmd*@YvIx*cV3<` zd34dX=a~l63b`$jJR$sBVl+>^>S;_9D-LfvzjKdK)r86BL8~t?_%8Q0FyE}SYTvB* ziJR*09glKRO1^j3z|u-KdX~w-4PQ1j&N_Ek&~`!8g{9Y{&dN9%Tq``n_;JyxvnGvR z(LQ{;)~nPPHL)0JGp}g4oMA4oV%t$|@5fDX;Xk(>?<@NB`otCkf?c;Qt%zxT7@}}W}6@r&pzWJR!=ux`Xc~|?}HzxHz z7pPk$7vyVgdeb#e)sbCa@ZGF!|12tAO|DvUaj)X$swHaA+3N2(med%Xwsc(C<>bkk z(!KhajN1RtX=yy27g@b|-W=1|x@%QLAEU!|{?lEDwk)W)b))^?|5baB6m6bR+Er`0 zdRAEcG^y8HS&Lk2ryrJF-)7*?UCi|P-KLM~g0H3bh;~{uTSY8Un&p}mBYVqze%Ix< z0aH`vJg&|$>ol6%Ka;Paen#pfo$N24&(2Z!y;5f8mS~&vw;G?UITEN#p6O2e{&NDCtHC`bT2o%5br=FF6rW~K_En~20;-5ouTUWldqR@O1rAJ+11!DU85J^M~yohPQk zY$p6dMZw1MS>m>U`UNK^TuarM_kOlsQrsc^OuiYC-*cuc<5|?ebY9zV#(s{FgNNJw zx&m|5_E`%)*d`)6;~VG8Bb-SaS`!y!F7+x(K6bGBddXMTd5g_Hc0UhzdLy)l$?*B% zBAsxJGS_;ZkBov-%#AJ^T&nCi*8WYlbb?RV=Ox^0I;DI)b!FeJ{qpnO?Cn2AY zE#4`nEz=%uH|JUC3yXy_+xKQiCb>N{+kRqq!`}2O`D>n~EvzqDD+c47fam_ zZr1FX-zs=zwP(aV3+s-HPEWsESG(qNW}xq*`R_v5c`pjE+;s?OcPh+SB&3r4 zJL4{Y_56_D`9XRaatAfC)px4$u~{;OB+tuUAEC)$BEI;@f#0Pi@4WtYRo5FIx-v8H zsrI6*dz)_BC)x=}&9WBSSa(X@cW+1m)5S+i;yRAI*v`7&H1Aw~fW^m^x7P&wZhL(x z=ue|T?DK<;8^r!Bk-u{AK;+LKkCIw+f@0^f2xVB_xwyQjxM6vkUv#g+;TOWKdY`!W z%d36&9^Bm{0?q${C^^R>USy6nj`Qk-+lU9&ZT@NZz}FM&N98uW39JO zU;0GN@r7S^8~<^XnEd&Eo=rvj1otvIJ}ItQ-;W&Tc^kZ-{p!8b6V+K}-ni-Ve)Xxm z0OyJu*PYV-ZD9VVrx#kUw~u$v+cU?`rsTiam({0q(3;7Xxy#p9Ac*V#v9yHXc(ViF z-1CmqKYqD+W>4Df8;T3AtmEN%{ce`!rVFg$X1n(4%WONe zL;H_vu!E6SX`EBznNy|v4G%R5UUZaE+mw(x|IjkGh70wt@AyuBsLiIhc|H%vw)iIR zy$iO_dvtgKvx2|Rx4kDiCWZ*Je;0Vj$uR9F`}#{$-RB(aEq%CfmBF!pc_pk1-f7O+ zb5&;LsavMIoL1I%Tx2+GGe`HX*3UJQ-;{0Gc=EI(+yCu{zWrP0dH(DxzU`k^NP4Nu zo~SEP@?`B2V^oBh`% z9L6Ve)gIq2-`e0GIA`P9P2VmAn&~j>tTL$cI;7O!;C{qbMaZf>-|_G&$5(57`|aI2 zb&a>a*_e7{&7bOB6Rf74{pM*Wqt_<#G4S_!`<}N-Vnqs%%Nt)Vu`L_y7mwL{R zPwe|O?fefF4gJ-9dViQ?=d;eg%JH@3>cWcZ16(4@auv_seW}zUX;xGw_uy}r@43F( z3+em6Zn0-Sws!pwCv{oIRJAF>7BLBL>gA<=UapG1@5vf;#oc9XhKq3c&G~CPlpicJ z-WAF3K7G`l1>?{3Se>G@V`RvnWUiXO>sa88!RAR?;NWT6kTYu*Lt!9pwtX?|4X!`f%K*of_ zxl;~RIHw(&c=A!Wq2AxFIW7Xrjr;fNhi~l*-E^!yCgAP#mCwx#b@sk^fA@;vV%fqY zu_DP{t5zIy|Gaq@^Nep@Odqx^mMQr*dvmSUW<}rGOAgmZnLp?_xc9eC(BX?ECzsDW z9&KT;I{l!YwafDNA>~2kZOk3>Y9}9eoXi&He7w zn-A|RKD3xAKEPeB?C8{M_ty8%dD_^}t~}FxrH|#(x=W`IK4zWIvZQxiSHl0BzW=NG z65jpT7G1uD%_r;cqt<#yW6>W@rPF?|ap{_yFe~d(48tsg`G?C&{~UkG-n^t~vaMX* z%{ND-R&Q2r@TJk@;dVC@|t%mN@ZkU?$29l!Bg)!^~%Gk z+r2I{EfuGNdYI4vfo1))YTXe!J=xp+p z+fmjW&ac~Zdsj@otrys6^N4HtliN#bLf-nB81bF$70;WzZQ+dM&Z{TZ$xoTkxH4~5 zOSxoJ{mrT0A9mEw6w4Kk`>}*0B5kkw;s38L1RJN!f24n9rAUD5b@u%oGtSsJ&aC@$ zzIUt0vQ2XHf~z}KMtP! z?Yj2*tEYNbs;+1Cx;!r_5%ijO-|E?uC&Akvz2N?AVxQKYsd&i8LHOKpmZ~UYg@!*9 z7(P7U5&o}hEjIq}7hYvYnnPV?^Oz4x1Tz)Z&{T3z3Y zS*r1xuSta4m0d5VF7jhA54O#3yRz4C%CR@g*?%yJ3cu8_v$-DS#O3YjFg4YqCn^A^usGvmZO{>5?f%O>-?J7tzK%w4EBam{pbZCC4~FHZ6&RC*qo_PO_u{jFs& zE%U$C-+dy{e{6v`i{$6Y@7ogAS6#U1dEw-?MN%3}|j&v}ol+2nMp?C1d3qAjD%_nhm@^UghL46=;e^CvOIct>4*?4&RGy{y#|N19la@9~r- zMRVS~(8X2|=NB1QueqxN`c*Sy3iv9Pl8{(J0Z8mZj&oL6-#1~gH)n^OuwUQqe`$IOn zx9jMfJ9hT>i}y!uJnIvl9R0^pT`&Kl{mxV8sZHvuE()g{vM#&7=E3ey34bQMuALk* zQ}(mKOzrzS-9_Zr6y4|J=;n%7d-Gr8yB(ifM&9CZ7pJQqQZD{eF+V=@UGy!fdz~5{ zE+-h&)_>#N(s_1?yHx#}BR^#f9JJp({`4~NX~)c`+pRe?Z~XB}W@!T^wb( zi1qZJAC<3!rcQj|zgFSL9EHuo3@Hn0Rahg8jn8qM`M6nBsQag`$I^#|MpC7G{t>Q+ zHGFtR$eWg{R zyl-aI?@gFk@Y8)S>qDJiR(!!Wa$LtgzFw1RFM82b-2UmYtI|6p7B)UHJSAIJ`dIwU z9R9npr3*gpc(kJ~{?hFliwo7Q2W2w#YB|5ol>3@-Jd)wNr28H7756-J|Lwl3x$d># z4DTBgPUJtb7uO68Pt8jyE-6Y)%;ieWOUX~l;xbY&P?#*Omc(qJXE3=zO`ONj(!kWv m%+SQt)WCT1VzpQ%6U)g=>YEu&CT~;s - - + + - - -

-
-

- - The - LinnSequencer + + +

+
+

+ + The + LinnSequencer - - 32 - Track - MIDI - Sequence - Recorder + + 32 + Track + MIDI + Sequence + Recorder

-
-

- - The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is +

+

+ + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

-

- - extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: +

+ + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

-

- - ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST +

+ + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - - FORWARD, - REWIND, - and - LOCATE - controls. + + FORWARD, + REWIND, + and + LOCATE + controls.

-
-

- - e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may +

+

+ + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

-
-

- - synthesizers! +

+

+ + synthesizers!

-
-

- - ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes +

+

+ + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

-
-

- - per - disk! +

+

+ + per + disk!

-
-

- - ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. +

+

+ + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

-
-

- - rhythmic - value. +

+

+ + rhythmic + value.

-
-

- - ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. +

+

+ + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

-
-

- - ¢ - Optional - SMPTE - time - code - synchronization. +

+

+ + ¢ + Optional + SMPTE + time + code + synchronization.

-
-

- - © - Optional - remote - control. +

+

+ + © + Optional + remote + control.

-
-

- - Recording - a - Sequence +

+

+ + Recording + a + Sequence

-

- - To - record - a - sequence, - simply - press - RECORD - and - PLAY, +

+ + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

-
-

- - corrected! - (Timing - correction - may - be - adjusted - or - defeated). +

+

+ + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

-
-

- - Any - additional - notes - played - will - be - added - into - the - track +

+

+ + Any + additional + notes + played + will + be + added + into + the + track - - — - existing - notes - are - not - erased - while - recording! + + — + existing + notes + are + not + erased + while + recording!

-

- - FAST - FORWARD, - REWIND, - and - LOCATE - controls +

+ + FAST + FORWARD, + REWIND, + and + LOCATE + controls - - may - be - used - at - any - time - to - quickly - access - any - location - in + + may + be + used + at + any + time + to + quickly + access + any + location + in - - your - sequence - for - spot-recording. - To - overdub - a - new - part, + + your + sequence + for + spot-recording. + To + overdub + a + new + part, - - select - a - different - track - and - start - recording—while - you + + select + a + different + track + and + start + recording—while + you - - record, - the - first - track - will - play - in - perfect - sync - (unless - you + + record, + the + first + track + will + play + in + perfect + sync + (unless + you - - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - - including - pitch - bend, - modulation, - velocity, - aftertouch, + + including + pitch + bend, + modulation, + velocity, + aftertouch, - - sustain - pedal, - and - program - changes! + + sustain + pedal, + and + program + changes!

-
-

- - Editing +

+

+ + Editing

-

- - To - erase - a - wrong - note, - simply - hold - ERASE - and - press +

+ + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - - when - played - back, - it - will - be - gone. - Notes - may - also - be + + when + played + back, + it + will + be + gone. + Notes + may + also + be

-
-

- - added, - erased, - or - changed - using - the - SINGLE - STEP - func- +

+

+ + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

-
-

- - Additional - Features +

+

+ + Additional + Features

-
-

- - simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to +

+

+ + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - - find - the - desired - bar - number, - then - start - recording. + + find + the + desired + bar + number, + then + start + recording.

-

- - The - INSERT/COPY - function - allows - you - to - move - bars +

+ + The + INSERT/COPY + function + allows + you + to + move + bars - - from - one - location - to - another—in - the - same - sequence - or - a + + from + one + location + to + another—in + the + same + sequence + or + a - - different - one. - For - example, - you - might - insert - a - copy - of - the + + different + one. + For + example, + you + might + insert + a + copy + of + the - - first - verse - between - the - second - chorus - and - the - bridge. + + first + verse + between + the + second + chorus + and + the + bridge. - - DELETE - BARS - operates - the - same - way - to - remove + + DELETE + BARS + operates + the + same + way + to + remove - - unwanted - sections, + + unwanted + sections,

-
-

- - Creating - a - Song +

+

+ + Creating + a + Song

-

- - One - way - to - create - a - song - is - to - record - each - track - all - the +

+ + One + way + to + create + a + song + is + to + record + each + track + all + the - - way - through - (up - to - 999 - bars). - Another - way - is - to - record + + way + through + (up + to + 999 + bars). + Another + way + is + to + record - - each - basic - section - (verse, - chorus, - etc.) - in - individual + + each + basic + section + (verse, + chorus, + etc.) + in + individual - - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - - them - together. - CREATE - SONG - will - then - automatically + + them + together. + CREATE + SONG + will + then + automatically - - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

-
-

- - Composition - Without - Compromise +

+

+ + Composition + Without + Compromise

-

- - The - technology - you - use - should - never - be - so - complex - that +

+ + The + technology + you + use + should + never + be + so + complex + that - - it - interferes - with - the - creative - process. - That’s - precisely - why + + it + interferes + with + the + creative + process. + That’s + precisely + why - - the - LinnSequencer - is - designed - to - let - you - compose, - record + + the + LinnSequencer + is + designed + to + let + you + compose, + record - - and - edit - while - devoting - your - undivided - attention - to - your + + and + edit + while + devoting + your + undivided + attention + to + your - - music. - See - your - Linn - dealer - today - for - a - demonstration! + + music. + See + your + Linn + dealer + today + for + a + demonstration!

-
-

- - * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the +

+

+ + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

-
-

- - HELP - button - displays - additional - explanations. +

+

+ + HELP + button + displays + additional + explanations.

-
-

- - * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. +

+

+ + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

-
-

- - ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. +

+

+ + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

-
-

- - ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. +

+

+ + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

-
-

- - © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. +

+

+ + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

-
-

- - © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. +

+

+ + ® + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

-
-

- - (even - drop - frame!) +

+

+ + (even + drop + frame!)

-
-

- - ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes +

+

+ + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

-
-

- - on - the - TAP - TEMPO - button. +

+

+ + on + the + TAP + TEMPO + button.

-
-

- - ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. +

+

+ + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

-
-

- - linn - - - Linn - Electronics, - Inc. +

+

+ + nn

-
-

- - 18720 - Oxnard - Street, - Tarzana, - CA - 91356 +

+

+ + Linn + Electronics, + Inc. - - (818) - 708-8131 - TELEX - #298949 - LINN - UR + + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 + + + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin index d3c2e860..686fd1ac 100644 --- a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin @@ -103,7 +103,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE. © Will sync to standard LinnDrum or Linn 9000 sync tone. -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +® Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. * TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, (even drop frame!) @@ -115,9 +115,9 @@ on the TAP TEMPO button. ¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. ¢ Any TIME SIGNATURE may be used, and may be changed within a song. -linn -Linn Electronics, Inc. +nn +Linn Electronics, Inc. 18720 Oxnard Street, Tarzana, CA 91356 (818) 708-8131 TELEX #298949 LINN UR \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin index cb807c801d7651276e88678ddee0521150aa1cad..d8df92f633d9e488eed9062da9460747c099fe92 100644 GIT binary patch delta 8670 zcmaDIzCL2ZdJbk&W7EkG8I|hS#GYKIy0iE0_3%6LCo;}5U&v5+y|>I5|^N^hiRfcF3a9>w6gKFw=a3OKi+WmbgloKyZ-;}w~zb(`SyGJ`pxU- z|Ns4K8Q;<`uV>d~{wXhc_jHQYn##X_9)Fep|7+R4mD4Y-pZ~vV>6c$W>(}q!`!86t zK3z3}VR8Lw7HwI0qk5C0VTZ^~4Q>;LDyub;Vi@3fNW{kA-j|L;~E*SI-V z@RY%=2L^pTNe|Bcx+cEn;M=k{$IllYdD8OR-dS=>PPh5XjxRmM){hEW1lILOEIw)&)2guwZHT17i+!UkDrTYu^ah++W2#Y zwTa!VA1As)er{a2Qp5j|cb0d9an|Y0#leNfw?Dmb*=zUU@yXA&f8Q9@Te=4hAfdDE?Pk!d|J94i?_{KBC;ygI_ZY*MnCa-YP1SVdm_ zxY_+3tFF9F=Ie=4;OpsB$UE`gVA@TYHgAi0ow>!#m*uV`Z#jJNeEx>3H=YUH-eYv5 zxIUTZ^4i2pb`C2JuDks{=6TMARK8S$-0tVut68ou5B*wuZc^{Bycy;vV>LtW&n^f! z{#|vA-mYnD=dM><`PR!?TD;?6?3(w3S^CfNc zub$HmWl(*adOI;fs_Im*!c*IKQP%}ZPrQGK<4@tEtb`SD*{-dNYL+~d2~T^s*AHR|!)py+$$!;K9by|^-s!h^v}8j663bl^msJP`L>DUhxlG)?zJ+t2 zxbE#e-k0OQ%7w)=`C6TAlsBq3cbZzL@j2H`BYnT++Jq%0d^>J@3=BFGw1Go$Qu3l} zhke&y*9kMVXJLA`C&lE+Y{O5Mb!Xq7IlIE~YR%FHfknL5Q_qxyCb{`uJo34XZ~fG? zw*NirR-_qCoGKo3)Q6ShkU+tmy>sst3l_S`ua3w~XZ`uH?UPQUNn(OVB z`50a`7p!HT6J~3>$Y|ewf9HU2HJl1JK8w$I`7)AG?dAFBN|IA%S6r9U`FBjP{&Clm zrA_9=cdcLMcio9P*SX-ON?H8`vHR-#vVO1jf3~0@t%^1D?DLtMH_yx!cD{A_=2YkH zPc;2m*x&jor!xhXFS{kFxj<&+vC!4|Tw)jJ*M|iCDG=KDqkhr+yhlZOkJJ)_?yC2ezdI4*oOvEove#!jxMTv=D$ zc3r&5_0sMA$(q)u3uGN-zF*e8;TZ1!^&QXm%+0aq17`N04L?;i!9x35L)6M`?@#DS z)?Io$Rj4PUe%*bBnpKD08ICwi;BD5?7V=dLi@r2n$|S`m`0c_lyVvW&MCJ9fM1{78 zWo;JgdSEnJQS*7a)xLvU?abDGO`AJ8b=CG`SGex9s!q6_J9mZSlTE+;_Z`xFE@~qr zCHT_uTR!U>I&8?)mM80c2o7(Y9fbJ&b0!$z#=s*uPc_4oxU8{_#O35Co$tC%Mdt|+p< z`9-?qg1!4@7Da5ciS}G@=K|OCz-^aK{(jLLTa=Kj`Q4;unaLj(0**~s&dM?oj<&8sJtz|c(rf4o^+@Bl!hbe-m-_FK zhH#6<3wgV%QuB5cs4V@S<2qwei^>J-kXiNSy6?YmGahS^-PKmz_Iy|LiXO${r|Qb# z39WC|s9JKIIvupGjOWKy_Sx-g*t;hyg-^CnJlkb)#xVHEDWT5AoKx2BEfJ2MKEsT2 zipQhp1#T~=9R3qp5vQ*hS0KlxG7ys*L#cf?d`4os3(e;zq^V&@Ubo9kGv zO0F&W+4CvM*OH?ug+08=kJa^f4pX@L{TUA{EN$j?HVLd_;x8=tbYP;%zNG0b*4DEG z6GRPPmVLO`_+9bbrD(IYztjbHDdWXj?_SWvh7RfqsZ$9r>u&U_v#?;SRTX_z!Pn#em`fLXK z1D5%Dj?zi9l=jv&XFf?=a0HyTt9D_-ti^H>o1?KZGL}hp40liW|Nwl zX*1S-ZJuhl`o{JH0^BFEq@NVIX*+5r=!b1t;{8Kid4>x0qXH$CrF z>`w2WSNKEr@U<(;*`%7TE|{QUvEbe7-Wj*d+Oj#rxwfkwynZ;>Wp5F4dHuBboCW(I zTyo^Ax@J2i;b&X(3zPF5DKEIcPi>#QYNnRTLWZA8A+8=Z>rZNZ&PZh4x+3D42IHi> zr&E}BiXV7=#&qtHo(-8_IhEZmJlSlTxNoX|!I~fEq<5PwDvK&A%g(Kk5>(ckRsK6* zamt2B-ZV|AO3SopUfH|1jUU`(-V<3}zj#KYBvbF2%S&}j3K!DTHQYJ-Z@l4OHvg}q?v>&b=Qc5|o44kP z`l1WE7j@>juYbMB)bq%$i*usvOEog|b}c=(HiJLrG@kYY-_&9Er?MN+~wsU@Yy_6mvZ zn5R(E(m4C|k3*vK^X_onm3lWZP=?V^!MQE zy|!t4Zu7o3+2S#=-s#t2gI0%k*21YbvUW@EZ2Ix7HJ+~waZh3)5v)}M=t zULk9==lAo%hs(Ve>u_<|Kb&v0w=ndVWi-Frq{GFcx7(Xb{!MFqcl(xqy)^&4`_(S< zbPAnS4~5+Rf8qR;nOC**TQx)wsSWnPI$Rr>TBhV_!Y@271_6c zxxT0leA>e68OmiZf6L~Yi=YFWVR`w<2fc2Vi|#JE#xAUJ@I!N9%3h^&0$*a!I|WZ{ zeBCrh6s3%idQJxrz^1sX@2NTYx`)@G)uOk zqCU(1$nmcI$7PmPvd@$-y?6QiNr#|_>Pt^Gt<9^ic&{=z?6|x_yGK?f_SX-CUzS#x zh2kpa>OqfQv2@+tw)O7Syo0T4&v{2i@JK`L)3-o(Mdek)1fzaVuxQ7XBl=C%>&%QZ0MrvHF(o z+S-2$rdmvV^mSRih{k>Sm}80x#+N>>-TH>v_ci++>v+bOuQx7oUB($1smKv|Wx>Mo znC&S?l>(O>&-ker^e#t$+hLuI9p7ey7`;uI*NpPdP1SKbeeRLaHS>z82VW*0cqZJ; zBJ`~zUv-JZ!EJ7gQP&kJ!@@3Id)2+pE6)AOV%fPXdRMC4ODde+cBg)_m7PGwbB762 zgpSFrRE_kD67FBMIfiZ7Mvr3$c+PlU>$#&7qI{HfMxy+s#y+>C@BDhI0xr$>udP|Z zx3upl^M+-v2h)T5&fa}dyjq+iG-;XA=Avj$x!NNSt~mNi7T1IoT20<^G4MFMPfy?M z3t3mbJpBDBs8J|tM zU{Y6cX74(6RYBGhs!9xf8V5ApzujK_U{%cOpwNwL&n9n_Vs%>jCR^lHR`GQE(>fhb zHmvyfLTTgHX!B#YS1dSbULbuYaO0!dC-`l@f9#D@Q>(Do(`B8SHY1|V=b_NsJ0V8( z2N;%Bzdw=ZAn;aEWcAJ)`=uL>UV0bovme=j-mu1Pyj-h_Ki9A8US9B+H#JOK+|fxE)~yocD`UehgI8@ukh=D0bJzdOC1WftN0;&Z)r(*y-R&tnIt$96ma6MvQ0I?qA% z`=t5oFPSC;^Pc_WWwF{@&m)CB>Zl86XO^?$Q)Q-vok!jr-dEr#FP6(Z>B0N7@&{jP zg`WHl^7wGjeojvP#XGX{MTcYGW#8~xAGwP!=C#=G+h1}oHgUK)quP~ClHF(qG* zup~(Cepa`2r)kjldrqHt8Sn7A-?*Fm=wY7l!Ih`i%zqJXIR8-nswYbKPL!8@-=-#6 zX1cfPp5>Pb67$6OrYO9=m)@~|;*`THoo!F%IK|2wKdfZkd_lT`X_utOGUb2nU14QO zU-a&IzB_Ux+j-;1IZKwL=BZrH?oLr$qjMsjiDS=I);!x?&a(fsyWo9}E5)Xr^C?8!di%~IXD zp?GiWig`LN&%Zv?Jt|r*ATP4_OkPNH;@b3Q={s*E9O2&5(~$F9I+dm3e{|KJxpU*Z z+wU0`c~`NXDq31xKI7Dfcec^2y!v0dwsV}0sXp;RNR>JC!TH96Hx$`-J%93b_IAnD z#s`#mY!d2C&iB5Tt}NcZ5_$Y1tffA0_a+`NBLmrd-HDh_X5>E4ZSwjBBR5 z2FKQdb&C&GD^?!Md^YiC%;LMUNzw17-`Qq#`%cmk)sQ!qtB!2CJZ;U~q@JaT<tTV(abkQf5F@1i%h?iZenB2+P=-)*>QjB^y%!InC>av@!$+`)lJK|S+AkC z|MR5f?C(x^JlK3&-ge!_$)c8JyWehS-6ow=fB(drw5trd_Q!f7CVk%G^QQ7={H^zQ zJ-x3>@7un7-vqxJ1+_P)K2Fi}?}<;0Z4X>C-=Nu3!Aj@fqpMb*)8_q%GjMVGyXezn zep$(PPcIg!d9!R0aqRuapfUABiSnJy1qb%B)z{o(Wclo0vZeNO+>6BitFswX7VE!X z9Y2G|wclt;*se!j{LgYFUmsI;RlijKc-}Ghe+@}18a<7tZe*@k(&v1G zc(o-zw~UJ&D_8lfkkozpGbDDL%b)a3|5kyk$L7N8VhkHP(iOQAW7ZeQgdJGBJt3}T z;^|d)eLmIO?DkHGGyX9nr*`pw@9D}*TrJ-+Deb9!xa50kN>h*Bj@8GsyKnq;f0Fbu zUTx+XlTYE|uP5zxxOjVt!*A!0EK!r03MXsd`0e92OUb{=W`4};hKnX0AFpRGaNjy% zOIedU_vHtFA8p+7{jYVpaKXQ_C99|ZIinYu8!7qfvA5wZzlZgy=f8wU{kx)C#G8<9 zVZ(l}{IXn);O5^;zdU2I-E%R{^TMQz>o;QBFR|N)oi4lj=BK#nj0x|aK9{-4uh*EZ zxR+Th+h%Ip^r-0R+jp|LshwTzFn_bB$;w8jEhX-yjdHzfJ9jT$p!ZbL>WWLHO4gs6 z=4+w$lNUz**3d7wV0|h*seYTrLgyQEe|a2#VP26JR}rGd|uswDX%rtT9Miz4#%x20e1Y_rdOvOej$p}~^*qWX%*wYdw6Zlpa|xmjF0 zH%wP<>(}*SN;=;&D+OmP=B@qXUF4|yUGayz_%mOd;O^^pf0|@GOTHtio3yT8X!~oH z^1PW#=S_7Esw9NJzjA4|UYch2ZtrK+ zuj5lS8c*MUz?rGIVCB^*i}lK^^{2mlbGlw@xuKBw_8|A}Da%gmDTum!m{+%Mr`Ve} z=8H^r3cX9*dv%3}ko~9fLwnY)Ji}9~#8*9A@}0AH)e?KF^}|k`7%|o=Fu8o{_WKpO%I9Yc+D<LxGgxUbivXIVV&R?YcQ|A@bo&uvZflhpsQI~T}x zz1{rP%yY$C-+3{s)v|=k?Z2M#^xN~0`+Y@>zR;8~)ycE1{63vlQp2WJ^Pj&rOZ;jvhbfg1H=FSdt<$)J<;H?x%kDHE$4aFrdJxp|CN23zH)MIDE#_5E>!nZ zvbCef!j22VF*+H_rgjmAb8R;JUHbZ>j@?Np=u+Agi8%~UkIMP;ISIAB_m!(S&c^@J z|h?H`qS-dUjZI;gtWzVF(60nWuT5zL|UKi|%Cw2O-{ zv0bzClXuhSs#Mx+XzLNrBxhy`ErB#)buo~IR{bL*EQ;>FR~0(J#JpeS$*0p~n0LB( zPMP|^W#ewP{L8ZHo3EcgUt}VBUb%b1Y?b{LGuu9YFnfMX^SKuH|1_Oj`^z*>-be~C z-z0V7{4=BcsdW(-LzwFoX8uiSdOthzL*&T|mjAhS>|FoEyL)cG;gq!4c_-i9e5dX2 zywhfKZ9y(mHfPq}16o;^N{zK5XBPiZQhl|~K2H6(PO#9$kUe%`eRCIE%u4z2c>V@? ztsryb_pc6ZZ;9McX7F{Put2(&v;T}XKL5XY{l5hhXL*I@JotRt_3GUexq8`0e&Jhp zZDHIg?(Hwc6yG&@d(jE=C(EDhiOefjXTCRo<%FpR{ylTpw4rt7pQ}8)LDMuDc%9d= zaoEkXJRe&eTgs^0*Pv79z$0B$^s*r{^0T{elfVtH85PRk+in%lZ7vacyNuC9QCV3i z&~nn=$i1hw&aFQ5<<|? z=WXmP>pmCobOvv#J|=fBV7-%I{?6j4bd|p=GCrk5?pw;zzWGV5+7)ZDrdRb0d+&ZK z{Ho6P#Xc=H#p~sxY~zixrTgL>E`Q;Q_%(N};D-0}o~Jfc9$b=P{$tbiD=oM7p6xVe zb$?c$TQ&dnW?AiGCatVh+luT&_Jyyt$+?)8U0b@-I4Jj2c17IWj}w+`Y+sY5$Ntx4 zt;@Xn!cRNIXWu?7xUTIS&)j=*yvJ_-*?CJS>7lguEfK|BufFhz&vUn{tuw8y(oZ{( zdYyaAhcjjictydvF!YOCF}JI(;XA0yN3%+L*u$~)X9r#)0WcHHn-XSLhh zmF%IXUzZ-ruh;ZwdnD1L$$#**>A@dNOEhkAsFdaCZnMrh@!xcbdtG35qUAz~9rHA9 zi=48bsph@njEHuS+56L3aj(7`#nu14TdDo-&SAzUpZ?rnTXgxbNwrnxl0(lQ&pvOp z;byUstm^qWlh;*svL@YF9Hzfmb^YWmON1_(-eEmjwsE=jIi(w)Oi}~(X0GT=|L3_= z?^|Sss}5hi^;9oN4}ub+SU zDf#N##ftb_3IC^tus>w$Kh9gjWw-K*l(yVeiCeyjHA)w?rs*`qmrO`b4)Z;kJzG#! z(|Mb0p&JWhtip<--Sy{^ZY&Jwv3na5qE}|=efVVk#>w7uQgWU7%VVd`O_ylwQo_s}{+v454U%oq-x1P*byy0-Pplm_3Y-YHiPVJ%%yxzQ# z^EGTE^>xISGre9CTHnxm?^9cUhslQUcRCFQ+a@P(wC@(SGguJ0L&@M(>x;s#ZT!Dx z&vBE#7aQc4F0sMG!`CR<{-j{M&mF=4XSH*-|7Vo;eKle7Q_XT_Lp{UE@mk_MMg|6E qMn;BahL)!0lbf_+m@NzqCO_2L#AI$dd8)QAr-dn(s;aBM8y5i4Pr0@L delta 8619 zcmZ1<@j86NdJbk2GxNz08I|hS#GYIyV_kH2|N0g30W&`{NU$x@zS~;Pw#9hQri;7d zZg|)4wLIcswEEL2Ugrt7)*DUsJXY}N$b|oYPh7S2`Mp2o`ThStzaQWK_v_>D@&7*l zu9v%i=CZ`^A8GPp|K~bauq4mBXIuCG???ao-)Z}1wr}pgUtb+Dd;Y!l_5iK{r~vm@B01+^Gvs<|M|DR-Spv2`QH7!@^8zx{{3nC-{t;#W3}7g z)X&xT|8M(W&;I?->b2+E(!c)ucy?xbUVZ=Vi^Gu8T)Ym&7FHKAG|cnEJbRXFMP9FO-_Nx?c~5J&g*xUx(hJ(bbpPb6RreCsd=mawv*4A9 z<&njOj_W7QYhS(Q?f%1BUp1fkb|~eh{9ak?xA$TNv&Hm}W%Irly;IyHTfffemb%WS zEv6#-Zd>$ftUIb8IylHY zqpY>Qsd+wbt^YXd%}q~b_&2=`x};lpe{tdb((T72Q*_xS-#xbbVfXgV@!GbXI)U?R zTVFQ^ZGFhd)m2^U+O4tbMESz+m9_Pa7g#fFEPBK*Xj!gvp88Wj=u%Jcoa;ABqCNltRA+5AF1`i|B$f$)|#lm7U! zmsPDUQcOIq67%}@9SKW)sL1WTH8|^LSRf07k#JVawi^+xz1g>4En&FzY|oKvWfqq% z`b=-HDT_}{mlU-F)(w!tQzK``AqNIKuD z)OSZqq1yZnzGB~`SWk+7QO{g-YuUq-9KqcmUQVBI-+w}`e$FbFZS4uFrVaNRbIs*g za^7-;d0kf1wlTP~!Qg=5qtrzgXoheq_pRrSh~1y z-{O4!{PTJBH)cvWFc|9ieGxt3wpdj1%3UC#cIu)Ba7QMir;o+cm8+Ora30+^8vfqtk9|J7fv_4;`-;x z$=gxWXLZ)f?)d%fh0=Ya*Hfznbf<*fS$fs(EJOGE%U4W~FPL*7e2YWkOPSp>3@+6N z7IX+~VNLy@6Jxj1d1cU{W0OyuaQSKCcFY^r~F)~J$GmDwpKbLxzR8oDw)zn&*2 z{+*=q)pFumJ-466kKzS1R$VwaLHf;G zELuEstCGRT`hZ{usdvnR?GCa{;+g+!POJZITHgB9pr1=<^<1t?6Q3BWMZ z$(+{bPq$W6ft-?t$giB4 ze|j9}u9&J5@@&5DvE1fMbLXbSS+$!kW2@J0oMm90|G`GxaN608DW^W3D897+2jixg zRDsVwo~^3>^(=b7ujUK!tb>cildBFZ?Nl|fV3@42QuI^eB@^z1M+cWoFW`9Q_^X58 zU8(t<@rvku{a-^*Uh6QEJzl@3vDop^s#jC^1k*Vuo}M^AQA_Y=cU#UQwJTd5FX5OM z_C~>2VDFubME4In52xmo@E*CL#+BxA{QGIe?JumRRi4oN+1y)xe|zvZ5sNl^i_`~E z6aG(_?Gw|$n7gHE;s@3VKbAySr}=X|p8L7;h@0Q4<&j@!ZH<+FeawbgN^#Hh>Gj+E zC6rg`tNw<_vYyh@t3zYT7PN%sey^jdAk(Agg7cZe&X@K@@#1zl2k64AFN zYp362P@c5xUBj#`rcXAg?K5=qY~_?XB7JYu!Y6Tid_|6HCm%RFx9XwaUuCnGAM+~8 zr>e5OkGopd-!6XOSH{taTBCElL+$MQw&?e)u|(fV7}INEeyZ}1a5_2#=}jFA4#KUT4~ z4ykNe-&!AT80Pf*%4O3@X08sa?tEMJGJb+z_O7MBl3(UjD~iZ|mb(?U_?-;j+LK*^ zNmg^dpMR-mGX1p#%eJ>Vd5RX{M;@@d*B|v+FOq)9y>kX^hsAfs6W5{=y}eM#k`7T3*SXw z`1tgG>8o6oZkELQ$hJ+9qJ4L>ZqeT0y73C0~9`{y3GVK4-U@UtYk4CC|5r>uo$YX;oC;!_{81mPejvxf!YLdHQrr zoc^a(WjRl_|I#tOePxPHR&cXNkMX;JO{_=R!r~UpnbBvkanB-W=R#LeK9>T)&jm3p zJqNOrBwsBp@VskkCAg*IgTcacw_M+)HEnwTpe7~C`1P54(|#I#+Y`GwHEvt|>g5dn zA^LMnmo%Jx$T8z0$K$$S@$8UmziVx81o2;NRoHr9&oe!i!mvvwEix{P_>`5p6zV?g ze!4M}p{ev#QmXF>H!JfwqKP?z3EyO1sTCSW`tIMq)p|}{zj<%7mrG#8rYSpWxb8T; z%5_vRNbxwYablWq-|UP0`7vRudiz=z)jQ6WdZzw#wM-yW;=JwW6Qh08>h_$u-dZs2 zZNhV%rE3e89d!%#xfCK)d2h)vrUQ&W+bsXbe2J^m@t$#U<|Pdw)@-w7leX?XW$eZn z6eyebea)_ko*8jJA`e#IU2uwZy+LhD!P}*VkN$0~=;1hKf3|7*(mmeqnoWa0=}0`< z92;3%zrXuKUfsM-fe4?q3oI-(x5(YuynVvD*l!!wt)Bg~i49yL~Fec7ry9&pNXU(LBah^>y!mYU^H(+4@}QTsGg0 zQ>7OU3N zM>EzcOgnUpyVEEpa;3BW{!Ja~H}3ZxE@G%sp0AYjRBVaFXTSTA-vuJ?hb@?$et+$) zqqkR!vP9ld;#uyqn`?W6Sn!AzrxBs zTe5l%!#d90*Ub9O$!s512l}`0wp=;2T1-T(ty|~5oOSnJ zAx0tjavtHQ+k}oOmG%A#PxEZhVEtB-*`}yxXKB}@a4&e}Otl$oQSoQ$_iO)Jv}jh@ zv?kB2tf&{$=2aOzP*2YY+g^OhQ_1ZAs(l|$bX%@dnG`nF%tkWr`@!a)s&bds6i(9n z)%QKWKFmzh`1ibJkL1%k3q!uoc*8OK=!GpvGb3ML(nO_pv=^f-U>j)g7zo`pZ7L^#ZT*?w&A z(ptfk`dVVe_t=tdsf7#FiZ67eihoY}WYNAbv-#tiA7|8;Jj z30}N`%U!v(muRca{Ws;2*1QuR#OAIF)!pl!7SJ{=M)l3-rq#`2vKm)f*X+A{)x%W8 zb;E)O`}c~KaJ<(ukEpLwjy>lcTmOys;Oc4GLVT-s)OYxQR#&jQsV8{)bkUYb1;y-5 z|L=#eq>EO!9#Bek_wW90Fl!NKZ79kQanP= zyQF_rcQyX36Y{wh=Wg3C0XW8Wx4Xf z%*#{+$XmkHreeg|8b>aM(zT3S&MRxpE{CVg8CMB2Km4#{P5so>{C3`RH zKc8+;-J!2l7%ldHCS2M1(*`gJ{fb*eA%mF*ajb&4r zAFTbVk@|Y^gesTpyCIqC!4+R9CwkLY<9LOz zaN`W8<~48oLZn}1A95DBbl}p@IMcfqICEwNzS=Y0o+ZR%(tQKxN(dpAtqv`y^B?)iuK)<~|2W7<6VR`%S*Pb=lhExZlu^A`$FFJGj2 zG`dc?&00OvY~7TOo3Rb`KV z>M>!Lw2tI?<6FlTa^-1Xxw$|ph5P@8kPp6zb=&_P#usBRZF{p0nOadHQoXpCc=muPtxkylDUYQOW}KRiCGylx-JL zEtIO0cSe>bZgwzZTpF*-!Zytl(*73eU$efCA&Wb=)~4fykpH)L^*al5t|ccvco6#JT%tvPLZkHc z4SxzGmo2$6SoWoL+z~gvx~}1yLxe(B!q$y`RUx-H6?#^Mbsx}oe0Tfx!pZR`=DiCIK0f0` zQx$*ovbleK*LE$dKb)kQy=@_H(G#(U|E_y^itK;0b7_<(N7LtnC&iXeVB$A6xP zoQ&A&nWgfqSFAh^PRhP$E5*mFziIKD)iUu{yZN;zq)o3l|6Fd_rC7CS;jl+3?U zB*hV^y6Ho2_7lzxe2WcVzFube;LeTTMt1K`l+IrC@}X4Mv1O+cBt1{ppR;W@@OQ1h zv*}|O6VGN9sYR0l*2OcWExzR8>w8jt&DpgJmRDZfa`1yo*P8_yI;&o&F+7+f`21G; z^os|M?vdK@K*GyyZtAlgE_^$8UoC&RQn)%y{EqQ{4(mHnKZ-LO5{-AwoN`%DLtM+k zReiGCbv}+->u(RNJxv?CoGQZnmU*o-De%b0xA%zEm$__m}o3KN=W!lo#heA%q z|C-s`qoq{4i}a4goOkM2pXV>1r~h+yxDih)r-Ec^Z~gZ5Dwc1Y+-G-}Ong^odx7t| z+sug^QBz~KEt@Bs>=`#Fo&81BdgHc>H!Ch(-j&w7aLJ)kz8M+@?$`cU?N05t)jjf3 zVxAFmfZF48r-fP#+!`kEc{_hy_da-E(t{}n=9b03nZMul{oNd$iJ2XS)0w3gUeBu0 zpMCxH$}-_ zJ$0q0&!s2a4lq8V(vbc(PMvx2;wkl=0%sPuUU~6&xdl^?@n!3cUp7BrmwOcI;dc3~ zf^worbAxfR=1za+iSPK^UX~W``)_*Tvz_EI^FP8f-B2 ziTZalpTAf+>HA)}%L#MrRuwhYR~$MSv`F17jDO4FeN7Fq`Ru}b{AMjx6!;RjuGKDN zq2kN$O`@!;X38p@5PB#3VR~|7`u>3QRR@Y?Yj%VbW#kmU6_}m7*4XyPrfWTKJ8I=B z(gN@84*Gm%xvNBp0RQJ)zxl^D6lGof?IiXp`_3+xdViY<=b}Pi{N$0|tajD2RL^UD zRzmloszZTq{~3K`Ws>pIn9D8^cz5dyFJ;|NaxMl*Y{B`Nkqp;7^7@_qBTTES<9|O7 zP|5lGn18{-TPZ&ZJg*k1C*{c%E7l%ta_uYrGrhO+`fIDZyFL`JNPHruyCt>Bz5Z&& zk!EkEv|_*9)Ag&52WT#4StPjq$0ym&fL|Am6&_i-qjYUGo5T9hvUbKj!P^ox9+^~7 zd)eUpQ;%KeWch^;9@uz3Y2v5ZvODhl*iy5ASM{^N+ib-{jTRT;BBrnZ;Hc9u{jX8_ zbbe(<&Bgh7Cm!7{mvO%Cfg@iSi}N*3m9ay&SvXx|OPK33-2&*v@>ckBOGc3Jdr$Gq=nuc|*Pm2Y4V zcoA~SIoEQ(fro+BQ{@5np#ev99$(^-~!)I&+eSk-nV`8k#n!@;zHTGPH%dc$rpXs zTyXpQIkM~2U-BQ`>+Rnm`R&@?Q)>i{uCe(*To&LxT>6WjoF!u|&LX`X&uSc=|D6(}c(^ZXqou|i`?+6D z7I;@qIk9j5UfbSxzq}T%mAHHL@}8n9@sOJ*SRXvq@Bibw;{6#9)4vlHLxVrofBbxW zc{YoaYjywKzT-cX!;I#I)~Eb>7r?6Jvg*$bUiZ}cj}yhO8h=Vk5bk|YYrZ=3v(R%x z-|VFCv9CqH{(?tI3g=lZ)(A;x&O;OTT{z40$UjfET&WdTNAN(L; zyUy*3(Tm8SJ++hP9+cBKdP&Yc{PhIQ+g1^!rE_Z5i_foFm~kg)^E1m8Q#y+O?lBTD z5L5iTyEC-o>BhaWo3~$|X|wqA90N81&j(BYKQ7zDT~|-1YfUcyubg_oBa3rG;ZyyUvnrF5m6wFAj*i@WJl?v5`IlnUo_iWT z(lZ$4tGgb2-%%a(tx+gs^R4st=5lKz1+A|J&1^K9xcU6`M@fe)XFZLW()*=aq4me@ z$%4)cd7H0(z1iYuX5q!TeBBMd`jDIOj^vbnG>SdySHb$6tRo~cV7WdL>z2AjB zQ2`G>TdgvYJDx zcXZ!xh(Dol`}*yvQ^>h_^)5v4;0nW6bB->y z^}hel_a)yBOQ#RPY^#FU-l;tg@i5u(UL!Z{Huu>)hRYq96QaH)+IPPWnH`X)w13C4 zOFzPF>2RJ99~SjB5QghS&|Oc+=T- zd|%9!y6cv#u#SK3JBJd-)2U3)cG~HhuLvl&(4E{6yh_wCV%dX6i{B*@zl!H+KCHYU z%ACLG)a;I;y;s6Q+!yPLi>*$Kd2@-;)$zkJ=NU`m9C@_DV*E%@JI2=dXtdse zS?oW`%TX7X=a~sE_hbx?oT$v=6?%G-T zNPEwMHUAk(rhi=eib0mm`Uc-I&u#DS7p2wOh^&hEzUMd}>yPdG`S=dDY)$k|wmoiN zf1X>nO8Tl%(YE7S zmnT~$vsrKLkxppHfBWdoM7wAGC)rF6>#fvS4NooatX*KE7paHKUqKkwhO=3AGYv=~3mjMlW<(wHVd$%PdOwjxMFhp_SU z$EVA`wMCe|>DeDsH|y@%JrAaHrm9)+D!gP$NUOH2pPJ^>@pk6Cx7W|De)5bl>W^Mm z#$Ul7jdM2}ik~>};gFiwO_To&(^#&yPuA2bS2NHv(Bx83(049MO)SaG&vQvENmbBr xu`)6+v@|d^G&3|YH8n7vd`v5b+1$ivva0qbCIj=yH?@5^4b8b!RbBnvxB#5N&XfQE diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin index 0971ea05..37d78ff1 100644 --- a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin @@ -4,1059 +4,1089 @@ - - + + - - -
-
-

- - The - LinnSequencer - - - 32 - Track - MIDI - Sequence - Recorder + + +

+
+

+ + 2A + NNI‘I + 6F6867# + XATALL + IE18-80L + (818)

-
-

- - The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is +

+

+ + 9SEI6 + VO + “BUBZIRY, + “J0aNS + PIPUXO + OZLEI + + + “Uy + ‘soTUOMOI,q + UUrT + +

+
+
+

+ + uut] + +

+
+
+

+ + “‘SUOS + B + UIJIM + pasueyo + oq + ABU + pue + ‘posn + oq + AWW + AYN + IVNOIS + AWLL + AUV + + + “parlsop + Jr + SUOTIISUBI} + YIOOUIS + YIM + “BoueNbas + eB + OJUI + pourtueIZOId + 9q + ABU + SFONWHO + OdINAL + e + +

+
+
+

+ + ‘uonng + OdNAL + dV + L + 9) + uO + +

+
+
+

+ + sojou + Jayienb + Suiddy} + Aq + 10 + ‘syUSTIOIOUI + oINUTIAI-J8g-Jesg + & + JO + sys} + UL + ofquisn(pe + ‘ATTeouIAUINU + paiajus + oq + ABU + OdINALL + e + +

+
+
+

+ + (jouer + doup + u3a9) + +

+
+
+

+ + “puooes + Jed + souely + O€ + 10 + “SZ + “pz + 18 + [LVAG-MAd-SHN + VU + 10 + ALOANIWAAd-SLVAd + U! + patyoeds + aq + kewl + OAL + « + + + ‘uoTe1odo + [SVx + JO} + Aj[eusoyUT + JoyndUIOd + 11g + 9] + 98108 + ZHI + 8g + ‘poeds-ysry + Bann + soz] + e + +

+
+
+

+ + "9U0} + DUAS + 0006 + UUL] + Jo + wNIqUUr] + prepue}s + 0} + OUAS + [ITAA + © + +

+
+
+

+ + “ONYBA + 9}OU + poloapes + Aue + Je + sas—nd + jndyno + 07 + pewureigold + 3q + ACW + SL + Ad + LNO + YADONAL + OML + +

+
+
+

+ + "ALVOOT + 10 + GOLS/AV + 1d + ‘LWddad + “ASV + +

+
+
+

+ + SUIpNpoUr + ‘suOTIOUN] + posn + A[UOUILUOS + 94] + JO + AUBUT + [O1]UOD + AJ9]OWIAI + 0} + PousIsse + oq + ACUI + ST + AdNI + HOLIMSLOO + OME + « + + + “SUIPIONAI + I[IYM + P2sesd + JOU + Iv + $3}OU + BUTISIXO—ZUIPIOIA + SATON.ASOP-UON + +

+
+
+

+ + ‘suoneurldxa + peuoyippe + sdeydsip + uowng + g1TqH + +

+
+
+

+ + oy] + ‘pepsau + JI + ‘suoneiodo + [ye + yYsnosy] + NOA + sapins + ApIespo + Avfdsip + QO] + Joey + Z7¢ + 9y3—uoeIodo + Urea] + 0} + Ased + ‘aus + « + +

+
+
+

+ + jUorel]suowtap + & + IO} + Aepol + Jayeap + uur’] + INOA + dag + ‘dISHUL + + + INOA + 0} + UONUS}]¥ + PaplAIPUN + INOA + SUTJOASp + ITY + ps + pue + + + p1osai + ‘asoduod + no + Jay + 0} + pausisap + st + 1s0uenbesuur’] + oy) + + + Aum + Aposiooid + $,Jeu], + ‘SS9d0Id + SATTBS1D + OY} + YIM + SOIOJIOIUT + + + yey) + xo]dwWI0d + Os + dq + JOA9U + P[NoUsS + osn + NOA + AZopOuYdE} + oy + +

+
+
+

+ + ISTUMOIAUIO?) + NOAA + UOHISOdWIO) + +

+
+
+

+ + "NOSpr] + B + Oy + ‘AONUTJUT + yada + 0} + seq + Maz + Se] + BY] + Jas + UdAd + + + uvd + NOA + ‘palisap + JJ + ‘souanbes + Mou + ¥B + OVUT + sjied + ou] + [Te + Adoo + + + ATesrewO + Ne + WI) + [IM + ONOS + ALVAAO + JeyIe80} + wey} + + + ,deyd,, + 0} + UOTOUNJ + ONOS + ALVA + ou] + asn + usy] + ‘saouanbes + + + JENPIAIpUt + UI + (“949 + ‘snJOYD + ‘aS1OA) + UOTIDIS + JIseq + Yes + + + Pl0da1 + OF + ST + ABM + JOuIOUY + “(812g + 666 + 01 + dn) + ysnory) + ABM

-

- - extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: +

+ + dU} + [fe + YORI] + YORs + p10991 + 0} + ST + SUOS + B + 9789I9 + 0} + ABM + SUG, + +

+
+
+

+ + SUOS + & + SUTVAID + +

+
+
+

+ + *suoT}oes + poJUBMUN

-

- - ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST - - - FORWARD, - REWIND, - and - LOCATE - controls. - -

-
-
-

- - e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may - - - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic - -

-
-
-

- - synthesizers! - -

-
-
-

- - ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes - -

-
-
-

- - per - disk! - -

-
-
-

- - ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. - - - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. - - - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected - -

-
-
-

- - rhythmic - value. - -

-
-
-

- - ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. - -

-
-
-

- - ¢ - Optional - SMPTE - time - code - synchronization. - -

-
-
-

- - © - Optional - remote - control. - -

-
-
-

- - Recording - a - Sequence +

+ + SAOUIOI + 0} + ABM + SWS + dU} + SoyeIodo + SUV + ALATAaG

-

- - To - record - a - sequence, - simply - press - RECORD - and - PLAY, - - - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s - - - click - track. - When - the - sequence - loops - back - around - to - bar - 1, - - - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be - -

-
-
-

- - corrected! - (Timing - correction - may - be - adjusted - or - defeated). - -

-
-
-

- - Any - additional - notes - played - will - be - added - into - the - track - - - — - existing - notes - are - not - erased - while - recording! +

+ + “OBPLIq + dy} + PUB + SNIOY + PUOdAS + dT]] + Ud9MIAQ + SIDA + ISI

-

- - FAST - FORWARD, - REWIND, - and - LOCATE - controls - - - may - be - used - at - any - time - to - quickly - access - any - location - in - - - your - sequence - for - spot-recording. - To - overdub - a - new - part, - - - select - a - different - track - and - start - recording—while - you - - - record, - the - first - track - will - play - in - perfect - sync - (unless - you - - - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 - - - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded - - - including - pitch - bend, - modulation, - velocity, - aftertouch, - - - sustain - pedal, - and - program - changes! - -

-
-
-

- - Editing +

+ + ay) + Jo + Adoo + B + JJasuT + WYSE + NOAA + ‘afdwexs + 10.f + ‘UO + JUSIN]JIP

-

- - To - erase - a - wrong - note, - simply - hold - ERASE - and - press +

+ + B + IO + aouaNbas + sues + OY} + UI—JOY + OUP + 0} + UOTIEIO] + 9UO + WOT] - - the - note - to - be - erased - just - before - it - plays - in - the - sequence— - - - when - played - back, - it - will - be - gone. - Notes - may - also - be - -

-
-
-

- - added, - erased, - or - changed - using - the - SINGLE - STEP - func- - - - tion. - To - overdub - notes - at - specific - points - within - a - sequence, - -

-
-
-

- - Additional - Features - -

-
-
-

- - simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to - - - find - the - desired - bar - number, - then - start - recording. + + $1Bq + JAOUI + OF + NOA + sMOTIe + WOTIOUNS + AdOO/IMASNI + OULL

-

- - The - INSERT/COPY - function - allows - you - to - move - bars - - - from - one - location - to - another—in - the - same - sequence - or - a - - - different - one. - For - example, - you - might - insert - a - copy - of - the - - - first - verse - between - the - second - chorus - and - the - bridge. - - - DELETE - BARS - operates - the - same - way - to - remove - - - unwanted - sections, - -

-
-
-

- - Creating - a - Song +

+ + ‘SUIPIONAI + JIVIS + Udy) + “OQuINU + eq + porisop + ay} + puy

-

- - One - way - to - create - a - song - is - to - record - each - track - all - the - - - way - through - (up - to - 999 - bars). - Another - way - is - to - record - - - each - basic - section - (verse, - chorus, - etc.) - in - individual - - - sequences, - then - use - the - CREATE - SONG - function - to - “chain” - - - them - together. - CREATE - SONG - will - then - automatically - - - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can - - - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. +

+ + 0} + CNIMAY + 10 + ‘CYVM + Od + LSWA + “AEVOOT + esn + Apduns

-
-

- - Composition - Without - Compromise +

+

+ + sainjeay + [PUOHIPPY + +

+
+
+

+ + ‘gouanbas + & + UTYIIM + s]UTOd + a1y1dads + 3¥ + $9100 + QnPIOAO + OL + "UOT} + + + -ouns + dALLS + ATONIS + 24) + Suisn + pasueyo + Jo + ‘pasesa + ‘pappe + + + aq + osye + ABUT + S9]ON + ‘U0 + 9q + ]IIM + 1 + “yoeq + podeyd + uayM + + + —aouanbas + oy] + ul + skeyd + 71 + a10J9q + Isnf + posers + oq + 0} + d]0U + ayy + + + ssaid + pue + ASvwug + ploy + Aydunis + ‘jou + Suomm + & + aseso + OL + +

+
+
+

+ + sunipa + +

+
+
+

+ + jsesdueyo + ureisoid + pue + ‘fepod + ureysns + + + ‘yonoplalje + ‘AWOOTOA + ‘UOTyeTNpow + ‘pusg + youd + Surpnyour + + + pep10del + are + $199JJ2 + TCTIN + [WV + iPeqqnpseao + aq + Aeur + syoen + + + Ze + 07 + dn + ‘Kem + sie + Uy + *(foeI} + JOyOUR + OJOS + 10 + ALLAN + + + NOA + ssofum) + duAS + yOaysod + ul + Avy + [[IM + Yow] + ISI + 93 + “prooar + + + NOA + 3[IYM—SUIPIOIA + LIBIS + PU + YORI) + TUdIOTJIP + B + JOaTas + + + *y1ed + MOU + B + QNPIsA0 + OL, + “SuIps0daJ-jods + 10} + aouanbes + mno0k + + + UI + UOHBIO] + Aue + ssad0e + ATYOIND + 0} + owt} + Aue + ye + pasn + aq + AvUE + + + SJONUOD + FLIVOOT + pur + ‘ANIMA + ‘CYVMaYOd + LSVd + + + {SUIPIOSAI + {IY + posesa + JOU + se + So]OU + SuTsTXO— + + + yous} + 3U} + OUT + poppe + aq + JIM + poteyd + sajou + yeuonippe + Auy + +

+
+
+

+ + *(povesjap + 10 + poysn{pe + oq + ABW + UOTIIII0D + BUTUTT]) + j{paqoeLI09 + +

+
+
+

+ + 2q + ][IM + S1OLIe + Sur + [fe + ATUO—patey]d + nod + Jey + Jedy + ]],NOA

-

- - The - technology - you - use - should - never - be - so - complex - that +

+ + ‘] + req + 0] + punose + yoeq + sdoo] + sduanbas + ay] + Udy + AA + “YOu + Yor - - it - interferes - with - the - creative - process. - That’s - precisely - why +

+ +

+ + §,sa0uaNbas + at} + O] + SUIT) + UI + preogday + [IW] + INO + Avy + usy3 - - the - LinnSequencer - is - designed - to - let - you - compose, - record - - - and - edit - while - devoting - your - undivided - attention - to - your - - - music. - See - your - Linn - dealer - today - for - a - demonstration! + + AV'1d + pue + (YOON + ssoid + Ayduus + ‘aousnbes + & + p1o09es + OF,

-
-

- - * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the +

+

+ + g0uaNbas + & + SUIP10I0y]

-
-

- - HELP - button - displays - additional - explanations. +

+

+ + ‘JONWOD + s}JouNaI + TeuONdGO + e

-
-

- - * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. - - - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including +

+

+ + "UOTJEZIUOIYUAS + OPOS + UIT} + FLAWS + [euondo + e

-
-

- - ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. +

+

+ + ‘sou + .sulddoys, + noyyM + sayelodo + pue + yoegdvyd + ZuLINp + S¥IOM + NOLLOANNYOO + ONIWILL + e

-
-

- - ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. +

+

+ + ‘onqea + ory + AY

-
-

- - © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. +

+

+ + pojoojes-oid + & + ye + sajou + pyoy + Aue + syeadas + ATTeONewWO + Ne + UOTOUNS + [WAdAY + OAISNOX + e + + + ‘LSVJ + SUnIpS + soyeu + UOTOUN + ASV + UA + OUlN-[eal + SAISNIOXY + e + + + ‘Koy + B + JO + YONO} + 941 + 12 + CASOdSNVALL + 0g + ABU + Syde] + [Te + 10 + 9UC + e

-
-

- - © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. - - - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, +

+

+ + i + ASIP + Jed

-
-

- - (even - drop - frame!) +

+

+ + S9}0U + OOO‘OTT + JOA + SpfOy + puv + SpUOdeS + UT + SBUOS + Xa[AUIOD + So10}S + DALIP + YSIP + , + 74 + € + ISCJ-CNIN

-
-

- - ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes +

+

+ + jSIOZISOUJUAS

-
-

- - on - the - TAP - TEMPO - button. +

+

+ + stuoydAjod + of + 0} + dn + skeyd + A[snoourynuls + ‘spouueYd + [IW + 9T + JO + duo + 0} + pousisse + oq + + + ABUL + YORI] + YOR + ‘syous) + oruoydAjod + ‘snoouelnurs + 7¢ + SuTeJUOS + ssouUaNbas + QO] + OY} + JO + YORA + e

-
-

- - ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. +

+

+ + ‘SJONUOS + ATWOOT + pur + ‘GNIMAY + ‘GaVM + OA - - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + + LSVd + ‘GYOOde + AOLS + ‘AV + Td + YIM + Jopsocas + ade} + Yowsj-N[NU + O} + eps + st + UOTLISdO + @ + + + LOPNOUT + SaINjeoy + s[quyIeUlss + AUB + S.JJ + ‘OSN + pue + UIes] + 0} + o[duns + A[suIzeUe + JOA + ‘PnJsomod + APOUIOITXO + + + St + 1] + “UeIOIsNUL + feUOIssajoid + oY} + 10 + JOO} + soUBULIOJIJAd + pue + UOTIsOduIOS + 11e-dY1-JO-9}e)s + B + SI + IONUANbDaguUT] + ay

-
-

- - linn +

+

+ + JOps1odady + soUINbIS + [GTI + YVAL + ZE - - Linn - Electronics, - Inc. - -

-
-
-

- - 18720 - Oxnard - Street, - Tarzana, - CA - 91356 - - - (818) - 708-8131 - TELEX - #298949 - LINN - UR + + Jgouanbaguury + oy

diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin index d3c2e860..d80b111f 100644 --- a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin @@ -1,123 +1,128 @@ -The LinnSequencer -32 Track MIDI Sequence Recorder +2A NNI‘I 6F6867# XATALL IE18-80L (818) -The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is +9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI +“Uy ‘soTUOMOI,q UUrT -extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: +uut] -¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST -FORWARD, REWIND, and LOCATE controls. +“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV +“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e -e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may -be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic +‘uonng OdNAL dV L 9) uO -synthesizers! +sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e -¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes +(jouer doup u3a9) -per disk! +“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « +‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e -¢ One or all tracks may be TRANSPOSED at the touch of a key. -e Exclusive real-time ERASE function makes editing FAST. -* Exclusive REPEAT function automatically repeats any held notes at a pre-selected +"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © -rhythmic value. +“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML -¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. +"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV -¢ Optional SMPTE time code synchronization. +SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « +“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON -© Optional remote control. +‘suoneurldxa peuoyippe sdeydsip uowng g1TqH -Recording a Sequence +oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « -To record a sequence, simply press RECORD and PLAY, -then play your MIDI keyboard in time to the Sequencer’s -click track. When the sequence loops back around to bar 1, -you’ ll hear what you played—only all timing errors will be +jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL +INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue +p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) +Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT +yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy -corrected! (Timing correction may be adjusted or defeated). +ISTUMOIAUIO?) NOAA UOHISOdWIO) -Any additional notes played will be added into the track -— existing notes are not erased while recording! +"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd +uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo +ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} +,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes +JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes +Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM -FAST FORWARD, REWIND, and LOCATE controls -may be used at any time to quickly access any location in -your sequence for spot-recording. To overdub a new part, -select a different track and start recording—while you -record, the first track will play in perfect sync (unless you -MUTE it, or SOLO another track). In this way, up to 32 -tracks may be overdubbed! All MIDI effects are recorded -including pitch bend, modulation, velocity, aftertouch, -sustain pedal, and program changes! +dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, -Editing +SUOS & SUTVAID -To erase a wrong note, simply hold ERASE and press -the note to be erased just before it plays in the sequence— -when played back, it will be gone. Notes may also be +*suoT}oes poJUBMUN -added, erased, or changed using the SINGLE STEP func- -tion. To overdub notes at specific points within a sequence, +SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG -Additional Features +“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI -simply use LOCATE, FAST FORWARD, or REWIND to -find the desired bar number, then start recording. +ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP -The INSERT/COPY function allows you to move bars -from one location to another—in the same sequence or a -different one. For example, you might insert a copy of the -first verse between the second chorus and the bridge. -DELETE BARS operates the same way to remove -unwanted sections, +B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] +$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL -Creating a Song +‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy -One way to create a song is to record each track all the -way through (up to 999 bars). Another way is to record -each basic section (verse, chorus, etc.) in individual -sequences, then use the CREATE SONG function to “chain” -them together. CREATE SONG will then automatically -copy all the parts into a new sequence. If desired, you can -even set the last few bars to repeat infinitely, for a fadeout. +0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns -Composition Without Compromise +sainjeay [PUOHIPPY -The technology you use should never be so complex that -it interferes with the creative process. That’s precisely why -the LinnSequencer is designed to let you compose, record -and edit while devoting your undivided attention to your -music. See your Linn dealer today for a demonstration! +‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} +-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe +aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM +—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy +ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL -* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the +sunipa -HELP button displays additional explanations. +jsesdueyo ureisoid pue ‘fepod ureysns +‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour +pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen +Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN +NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar +NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas +*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k +UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE +SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd +{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— +yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy -* Non-destructive recording—existing notes are not erased while recording. -¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including +*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 -ERASE, REPEAT, PLAY/STOP, or LOCATE. +2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA -¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. +‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor -© Will sync to standard LinnDrum or Linn 9000 sync tone. +§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 +AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. -* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, +g0uaNbas & SUIP10I0y] -(even drop frame!) +‘JONWOD s}JouNaI TeuONdGO e -¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes +"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e -on the TAP TEMPO button. +‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e -¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. -¢ Any TIME SIGNATURE may be used, and may be changed within a song. +‘onqea ory AY -linn -Linn Electronics, Inc. +pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e +‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e +‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e -18720 Oxnard Street, Tarzana, CA 91356 -(818) 708-8131 TELEX #298949 LINN UR +i ASIP Jed + +S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN + +jSIOZISOUJUAS + +stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq +ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e + +‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA +LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ +LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO +St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay + +JOps1odady soUINbIS [GTI YVAL ZE +Jgouanbaguury oy \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin index 75cefdccafdac074ddc0199e567fd57fe6e3fe0f..9e0083188715b4a4128af76c618f96f90e2b1c01 100644 GIT binary patch delta 8978 zcmewvc|U5ydJbj_W6Q}88I|hS#GdX`u`JqiKKu%QiOziQ7t9^iIg4)} zezjtK{3OTav#lkk-c(Ku2)^5UribH_{Ih3n=j->VzP{)4SAY8J@AZHGzP5>+7pGe|$0j|LgttUp>Ek`q1+qn&)n8{@b>FN5PYy(+<1-`w$X8 zbKU)aGW&~bm$#Q!m)|JadwW&|OZmC)$3O4dDY5SU!@s{Ce!sQ2M|=La4|X4TuS@f6 zj5`-My-fbG*qOiXD*HBBA3y!_?TmJzyGz<#=kpl8!HO@=TCY& zan7F!$F~NZjJqvYcsRx;K-kBgq|YTZ{`@pzp~;U3XxOZP6h|I%8Ycm3Df z{WD(o{yUL7`T?K_I$~AQF8xi+ZrigUn|k|42J5y- zz74#5VU+#1MAOTW;(C%uI(LD?}^D9(=r{??znDV z;$IVDo&0>Vbl=NoeaZXk%F1(cr>)nWUj8Zjykti%uh7j?6*rhqsz>j-Q7z9e{9Tl~^2{UcF8b8nG?4ozkRq64$|N8=u)cyEwt?T|9f=j{+uN+Q@G}{+s zexY}_-R)nm4&1w*B$0LdCvN;mA76Ax|rl;OO~pz)myw4_I$rk zI@|g}{h@1nP8ucsHrUw`Xf($?^6&1{lhNg#{&%Wxw6482XR+{ZIX9`dKi5QVIp?RC z$kwOjtZFpP|CrLACyDnK=_j4s-+OoAyYjs$tL7)VDF5Ho5L`HSF@H zkeQE_TzkGGEWUFz{pInq4|dC&3hQ1!xAbBB?j6C6{gVxs=ASkA@NIG_&&Rfg^N9?P zw&_m1miIhVNWb*>!nFO$ru#(2mn{78ny*Uco50I`KfXVD&}hDV9Us%AFGsiCx1VHp zuCM=x#DtGBW|sf%ZQd1LUp7l7(rZi4_KUYJefsfmp`w9%O19s|-?dNoM%8w$3T!e@ zJlNX1c$cfkqmRM6zIA@9d{%deIV0UPK0syV?8SR$XwCXIS!uTB+7~Xt^R;%db{q(6 zi??2LK5p9l#iuw0Wd3Ox-Ar8Rqp#{9L$)`1$F7Um$P~FUFqJECb(yJ!KZsyxlA=qWOd!U(iqUN=gZPJE9y>^*sR&QD=>QMzw8Mo zJ|CSV^;;m!edDPvao?bG8r#|aorwzkXxp~Et0OJVrMMy7zqEdHW!wB+Ru?sr4WwuP zXJB{~BC(dC`?}PkuNsdyt|sqObO;mt`>CW;d}D=+uo@<)3cVuCshcE?v|YPku~l3 zf4`j%!3#&J{-_&Y1CXMt!~O(^bNLQ%dY~*ZqH@GLnL@~OG z@ZDPY^!MfmGQPd1PFi@@PFX!Y+*s0Voy`4uM$^8I*D0&he3mO6C{Nfcrgx}Q%I2NC z{e<1p2fPj34hn?UZ`8GB)_& zKL4)k|LZwpEE;!zpM5VNVgISCY736M+~Bi))te(${uwe%xyz@2chmLhjmZ8p|A@HT zlX<)M>21y`H;S%)bhh%@RUOM6e(}5lx8~bwcV$nW_1HTzvM^KdW|4i)iT=HZO&;D| z_u}8O-%}3@UK7}L;l~9=?@Nn9Yv#OFDAp^S(jc&8S!Sl$llsFF3`*+9IbT%-+*tPE zNa~ik3`O!fqRgh&j^9J)dT(6VFJHT~%C&HZPPO)n3s;VcecW>*!T9&V4MwaVE?;7P z$`jNp_Pgk9mT$A;?fdno=hz=^PpSypz|7OXXXBi0>o~W+KXl@}wFv8;%83{FW2Z>P z|1H1&nfut0<7XdFu+fB#+V3VmuG$}O%+b%W{-@o_Z%u}ArtOb} z&dv)w{>_`k!sLPf>gSUK_@=DXU3skLRE6wM=9O$~`KkdK&$;D>6S`E|fJX4*S(~2&(GO=G5-j?FRck0oU zi_^D${Fd-c%1LGGx2bbp&zUY1dN{_ySnhRj@o9m|J?m=QMa>cr)1MvNkF3(5qed0!tFYU8b6_4cG|#otFq{w>s4wnEf}KfUayXsY1e z2RS>^r(|9BbDMMfmdBYlCl@aB@Z5Fo(5ZF&KcX8JXU=<4^gGtNA(A0_A=~}4sehgX z?YjNJvvrI1DuuAijrZsIlcZJ5y(EfW(SMBrMmBDEfM3kS^y{g%6zOA&nV~P|X z-|wLP7RryTmG@`tkaHICu>7=L-j6w-DS*E&`ioV+|ITF(>b2RT#aYGJ_Z{-MqIhR! z=@O;>oT~-7{@vo$t0v@BOMi9N*uF$5_)tjC+Kp``(h@$QlM}@xru8O2-1TE&Qt!sR z*Ui?OUi`jss`qj(drw2Pq0-tsGyM}Rb3O_hPq&M>`+K8}+_%;vn#TX@#I=+z*Gu0} zmA;lcd-s_K?$;;3((0+t@E7d-MlYxt!+@vkg#M|c@7YnMx&?tix|uV&8GH%#V_CyThatYu?T3|YIv ztm%++Xu?t6Zy~$PE`0G})YG&vt-kbayWzVBN7=ZAn@YM1vvh^P zzg-P6IvbO%r_^!@PU9*zaZ6cgeQz;``f?60TP>dM&w?&z-lzUnzfc}m7$3l|_;z3F zodtOhed^fH+NvAfJ=S}r(KNa9QL3!%Z?{yx3q_Y=E?&Op^Wtfw#oQNvwhH`CU+%T& zkp)}u6=9_l&svLCek_PT)uL40&Z}OpBo-mF(LrIcR#V+ln|dYz$)8D|%6@&i&9rs0 zuVQQR!)or?am%k9w+)Jln^iZja`p2|l_&l?zMm+nWg~f_DU>0-p>6Bh^P+vGtkd)+ zd+2DaV*0Ury3xWH*L^;DedYfAD*tRsRaK40a?A5qO^+|nic?slST-mBBF~qcUyI}t znCclBTX%$+hDk(8?Y*$%aEA3O5rsY{k>15ecrV>v-uJ9BVTPaq<5aFE?Xqhw%RXrE z($)I+cO&oCbc?WW$z6v$<}!37v!C(XcuxKg<8GTK-X}YDoLa+tN^x#0OX#GvZC^x| zMB4FaSGjIVyXDZl=;*T>s^WR=Qyil6(%;$({hPS8J}BIITeINC(|s-+dH<@DuAlwn z>Y3{FH}c?;W|o!Sc^98Nk$wLop=+@atAT)LT<5H?D@^R1)!aS-KLz{^&u+?CmDg|8 zrnIE8toGK{mth(vc@@?UVvI94+-8}sFUSG<>$e`Z`ub)t0sJgjUMbYu; z1KxF)CYHHROnPjbGxKgrd{6#aXJ`8+i<3O%BHGPbrP(X@8qR38nZX?RtT8#uvg&4D zX;apVJyKy;ioZU6ws7qPq4|3jZ7vSE_Uh}J2)8Ikwx-J9hcm7i6(<=C$~AN7)&b(A-|(P770r(1bp0Y;3G4;Ie} zu$p}Np)uzN>Ff9I+ZY@Ci7jv_*cHCb>gdU|2}Hde_Q=l zdsU;FZT*tA!_Z2mlJp00AH zzM5_6(ifR2?~H#Z`|J5XS@QVbzkR=pe%;BoSbL+T~CR5_<3r{&73ImTP|X*{gF8@8tO*OtO-zfb<_6L{(QV5Oa! z8-K&9|F@SdF4yVRp51)GN#yv+b)A1du)MV?akAp6uYP)J!qsm_>iXuKezc_T?U%a+ z%v!vz6MWuRZ>uU_CN$mME9*p8*~=|km3ssKyi5t!t=%ak;_^_?_7CT7bm|-d6xQGODpBTsTU0_T^AL;T7Is1rF_q&#ev%{ zdAn|!T5VNQ#F>?IzPFJIld=mn#Tv8kHNnwjgxOkdbp z-#pRD_;>Yce)qJiDxc;^EV*T)bvpl+&3e%s{S6y6?oKgtkrjxF59DSju)g|W!J2KX z8q$_iSv6-*^_%l4GV4%Hfn1bI{jSdAFFTgq-?PJR>D|kf8Oyx()oio!&ied+Rf*Z! zt99)wW+@zamLBMv*Q>=|Gfc#G zB{ynK682N&b3MRbe?y6ZB&pbQj$vdXc zw)ZsJU0C-|ym~$&PjL72cUExNMJv&d+hQ zS6rn%$8yn%dJ!(>BBnOG*bjG0OgL`$+UM&0x!a>GQ2w}Q!QDF+hn~xA^n+%_`0Tc`*~Vv6)W1GfGh?iO()UKb>Gsoe(X3u< zU+=unG0Qejy>;*9lgyM$=C7|kRf`c@Q)~QWR%?}g_2=#nDT|X12x*$^o5Wc$q5oq?=X+ZTqt)x*<|KX~VhG+Z4j<^*R@>7L=EkZv5ov_u_D0$Ewfg8LscL zRFHBPb}c+(!yBOY{GN5ZO?{UF+s>X<7o=Ht9aNRxn3DSG&5nx3UC$DN$=KX;==k?Z>uQain+Sh9&WLkF8+~fAL=V=E7cgKiL{4Nv19eU@`(}rZR z?HaFHJY*e|`wpKoZdx>-J#63k$FF%hpRKHCxSBZsbE<^wrKl&pUb&*pn>K&5$UFAx zaC!$L%MM}BhjKQao+mqm_MPZc=!-g8zkDs<6kE199~1el%Q`1~Pco1{%*B$C7{GpH zhGs+CX3iAR*`fX2E2SUAU0#vPDfg7o_8DhzlXCp;x9d4?RqWI-FUWUXSuv@iJLvky zX~nOm)W7TP3l5*mC}iR&5?l2)cFry4pj~ycLe2(TWTZWhFB9q6yFqlHmB3UxHz$qg zEkVyN3QnE7`~lnRmQoQLWwAG1TaRsB%wZ8$Ecv^q#$k5Fo{0|@{Bh&?Yfv-gZr91! z&oho)dX~QM;J(UH|k^gIYI6E3Y*IzMm1(0TLUh8O$(ZI zr+4|RL_66ajw_YohgwQvO&=t$nv`-#=wgp(#pyc>`Cgssjm^6FdhUnP%u4FKaM0HQ8pKEq#*6-(YD@s}f zXJ_31JmF1;0((ocNRa0$9+tN&BFp7oIrYk&daUvJ&8-*wdXvQ5%$A+3U-MH{?)BtH z#SgnO4(!u0Zku=_qOsn<;obLt!S73!>pt_V*VmXQvLS4`PuKm%ulp{^?VH{Fd}j8R zUnjJuK7XvTEN|vH#d5#c^rG(~8c`ZSTxo}rH$F7%OnRrmn`4r zd-;CIueCP=znop6a^!pg#v7I&HISFdur+Uqkd*T>gn$#Rk4Z7ZIx@UPyw^lR)Y zlaJGEOefUeOFyWY6tebnrO4v5+o#IAa+W#2wp+|3zUu#ftAebVDQxULn+)`=Ecfuu zc~q~jr1gAJwaVQvxvvU*voF=QdlyYI$vYZ#GyBH=$qYX?hU=HUNcSu#c3XR+)Vgf1 z@C&VduYNp@C>1S#@uls<`TGL@j5$~2-`Vi&rFq^Ls}s6`fj$vh=G+hWGr#E;NxI#a zXZB+K%c5DQSATnW{ZfIRJCnQO*9A9^Z&aVvp;^jvyFPPEllSkP(QN-;efQ42l6plT z*f>-<|9s|@HeOM$50|aQ%#}DZDwBTa{!>ex!I7)r6SPKaEmNRT$g2IH=l*zJegE_F zerru8vzKBYZ<_n=N}n1(O_#Z_ChN|glG0@xW*w^iTQqZx#U`VZOo0>Eu?O9dxb=BG zyXX43a>;8l)+(~pU&vr@J6aIes_&-}Dqr>NP(b^#R6TJs=5tS%2EW<*dZxur%VfI~ z+ahGsw=)}Dw$T3;a>D#Qzoy^6>h)Z~Z#VwU_+&l{UQ=J4uX^O%vF%cGEn{CUn_XVc^(Dst-s_7T!8hu4AJ3}DaI4;1 zuk3ZGaYIU<)79+@`Z-lI%3M4n{g>p*J&qAr>Y871GB5R%;)jV7T4Uo^30m|n3z-vb z>}@o|=g3Zu@N12X_peUKJDIq-@+wdFF;o3T#@A}Sb!;n@d?zGkn8X>ToZgW4z~I_N zSLwX-Yq!eI(-PYf#V`AOLD;&N^&YRJC8HNB=BGtFAE>*vsrgv(k;(f5{${mu)jhnj zOmbhgl2WgXi_E3i=cPOf!n;;a;{Va78MH%E+VRN+ZrxO6=?Ml|Jqia_J!xOu$K5*R z%;&?^2fuU5*PTfXSajOVk}tLHXJ~l3>8Zb&b`}wm2HRIQ6`pcu6Nzei{cWR9y-t1k zwT4LfYx(_FuFuy1nX8VR| zpA_%O?LSg<_q^=5m0D|eIW|`Cp8V3pCC>jpOHX^r<}SPM@1x$!jniLkRBOxrAIq%u zQct|~!`th3OgE?=GZB_Ip7o??({2U*-if+>OY7Y(v-A9abSeJ+^B0qto|=4F;#{`H zd5(MY7vbfLnI2r@wPY%-duGn=FjJMu;mPEFzTN%8J-nim4*X6`dR=B#pnPy1v;E6z zw`oV41s6Zxc(AGbXvvi8hnIagv1iH~mENPg{l{b;FI@U@YvbnipH`lZk(;IYa)x`# zD=m#n6>`fyxYhSAm^pRAKef&NGoRn6ep9BlVZxhBAM|S{rDh%5X!-TeYc^;961|xF zw&%>Uj!!R}A9hRQr^m8qT^Hs_Y}5av#%6dWA+NM8nccmu_Ln*X1H=FSQ)4}g9vkrd z@v;*T_bi;)=jr>W>F59Gwk4bk7y}B=o^8J|aR+0L$*x%MV_{cv>(@OjnR2(8-~P#) zi|l&M7t*;_Tsfb6J&3V)<$@H>P!r~@lbPLgDq?KZR4zW||JP^OxU%_Y6vGOCUVQ<# zJ?psqbhQ_md1pt z7uIVRB;~2>xmgvk!peWSnDo2)>SG7bPB)o&+-^tP?+uIWx392`XV7~SDz|Q4)RMC8 ziS~BYUN)sRaaZ$L79KQ9TlhJOLz8j!svHvut)Dfm4}LMsI^q>$Av2%h`jPsMvWD=} z`RU2V!pHvJioBI^<?g4a82iidn!Md$95CG=roR4f zrZ~@;is}~Y+q?SDEAKhN8duuBfAfFKjJ1nAb&GDtHmDW{tt>J%QCBNX*fo)28Df6K@OcKwNb+wL|G#|_Ob}nVb{oTh4t?FNtX#RHDV=E*f?*F-) z`Oy<`8^%<&B@=mt9nKuBfoI$n>)0 zxE?N1YqikkZvHyEZd7CEaN(yvcX$~!U#au4NUSB6{D7fak%T&*wWwA?bc)z*r`Q`q2ORgr{ z@vTvXQx<(c^Yi>S3&WE;9t$jrOOl^|+Ft$9lB;p8UXFn$5@yJ=?4Q-+*OcGizw7GU z*Ry9wf4#wap+8_#rKs+s;yG!F(J^yd4_58Uu;@55b@8re8*iSr?A#WysWlfP{m?^PGBf*&e4`+@{P~Ndxg~X`)ul24=Hstp1s(JKh7DLS|`N9KprwU9u zJ4wo=)jY6o-wN6CfMkA)A98N*-p-iXeSZ40ew}ZxRSQ>Lvkc$G@niFJ1x=$bx5dPS zPtQvC+8uf2*uA%XJtn&^e+WwNV_)Ln}k~enQIi2cDX!EsVL&t}l5lOX*|o z(;7t<&EiK!fa>9;;?LSD!NHSo)#Q&+f;ioex`?EMeg-+ro4&S8VH|x`xh0MpcX=b06@Xx?{ZZ zh{?$}b04IzTJN+wRbW`qvPHM3=uFENpFIvik42_$acMhxdhdZNNn%g$BrMcFx^VKm zV}UcdOZWfKt9W|a=d0@3B*QPaxr}=EhkNDCI2#Z?#q7;tVctJ%qKuJ2k8WOVN_(cm z)U)dN-dAVm>&G}4H3m+WnA7*A!)*3%*YCa0#1$oOe7yA9^iT0QXZLw!2jBc-eq^)S zdGcqia%Mw4!^yeY;ygwM24+S^hGvGArWTVYXvZ)c8k$Uguf37k(A;wJVjUkYBMVC| KRaIAiH!c9{$&oVv delta 8893 zcmcZ~^)qtAdJblDWAn)m8I|gn#O~}<*cW<|^um9b?|K8c%mAmi#`|~v|x$^ht`}zNWed|9~k?vE! z;s2L!`T6z#cK$oOHvaed`aO}^d5!j>`=0w39*cYX{ZAYBcAxtOAI0vdhWmXj&iwQ3 z`k%SIg>`k;chC0yy!+he?@7D&t=)h4hpgRg{@UaBPVI@`xLt!==d1cP=^KT{-%azg z&c*M0qx*6CT~m9X_+`s~-~K*7H@W+@w6yS>tjyi@QROjaXRd#BvA=fZ+B4zY8M|j@ z+ilsOr+LyS>Gp-c| z{=P9g<@iB2D-rShnt7p2O*TO$pWV{#?3pjDdwpv3lC?)`Ir?`V^1sD>>(=(~GhQ>@ zFSedsHYYVYWw&?zGW{D`7yej;?|U-2RCcGGcNNdfm(s!iW%fOLzfbe4vC31$4R=KA zKg-D(o}0aGFOPo9f_tBJZ`bT*@@VuA=J^m%JjXsaHC|uu<)X}0vtHRP2~9M9ce>({ zTFjRvMHS0VKC3W_bSn5f|MkHMFK%ibeYJ5nlW!nbJbR*tdHKO)-<$QB()Z=`zb?*d zF9^tc{GP(RFF&rC zv0FU5=jEOgGj5+xoLKlX=62Ao<$I4LRORm9^{>AB!l#YDDvT5B)GfAD6^ov7IkfO; zf$@gqzmI*7Ox9SqUVc5-kG$O7?w9M=WnEkxEH}zoOfDDq-8)?hKv4O`p$7tYz6U ztE}Gc?3oXzPHov6JWnWcOTpb!j^?w^Tb=&<)ydMcIoX=Gc7x*C+4cU0uKj|OES!p0 zKKEDQxZlF?f<;(bwEW?|eH&IUluO@Rmz8Id<8eYHSMAY_b6;%hJ@(#Jy)DR{z#jL0 zSD5?*X}4=q4*%OO2}(@JzGQ0jSoF{HGC!WcHS7Im)%Pr}*d&vy9EV{epwikcfOU3)0$?R!i}3$bYH0=#IYo^i9GuYJ7Fd zOcN*XFRy>)$S_&Cp1Vuywa27O-?rSHY5yei@4?p>n!nwD^W~+diC)q|ouW5!w&`AP zIiCD!TQB=z*X@-WW^+RJSIwR<=T_~Ze{a=Y4fD6;_Jr*_$iw>dci^HE)uo-Aocnff z{lb3rpZ$Uc(N!tGVy^lu6>&dYW+w4-`?j@GcVAr!a=)}Xb8{H0#iISe^}8G-Z|csH z(u_!5ePic_M^k3AB=||nygx3c&0N}Y{;a}v*B zCtXs)vVL~oEvbTMI&JQKMZ@@#KxQ<}sH7KXa*_*mJP>u06x?>u1-8z29pe7eBA{LfB&I zO>5rPCtPa&GAFOJkSB9*aMR`1t&){O#?LE%ZdhYrs9rr~qsgbOx6;iwi)3aitx*0M zZxz{aMKmwrirl+TQ~8XUEIQda-1oZN)Xfq1=U%r>%fxR>?)~`^uD@Je^qcjt4v$8(evh-jA_k->J0Cxjy>FQEP)-(ajDy{8<*sGZ`C? zzA!%J^YyaY+i#!wely!W-M;S(f2_y4%kHzHR^PBa+-PmyuiRHF<~TW3&%{Sz<^}{pPF_Wj@ig=z3v2r^32BWfrB= zN*V>Gr|DZT=uY8g(8tJzZ@spu4AgX3FG~HhD>}`?sxM>2$9??4rc~({WYy z%es}FlmZQ=C$6tkQu?F6aQk(Z7`vK0h1SDc71Y8x7Fn$>V6c*%S>x;xnrr&rnWG_U z`HGu~B_4mKSo1SX*~BMkCGfi@RXy4JjLPGZ2XSoA`pt!>&Q%E!?d+~x;4f-WVwuZ2 z??Cu#tw~}Jvp;N%FLjDEZ{#+vmyug?^zC*Frjp%9n&&e!3&~FSP(82Hb++%;0PlH6 zgfc3RwwQ>^X)oij7Tk2BO8>y3lU3H?n-~S1wHTz1-kDS-lXOP(vFEif(rc{qVtXUY zHZR`sCab!6#^%<~e;FoCZMgVcv0k9Td}YOb-3x5N64M?Y=04Zx;2+o{dGte`V?}+a z-J&!izyr5?Ekvqud&($;Ua!UV{3F)NPY_-pK6H5Eu(f!oA;xN~~O)?%XLf8C% ze)!yQqN2K-M|Hl6OG8wm_i~9u5Az+1R83^AT#N7Cs$6lf&q*gpLuymZqQ}kZ(|=?~ zOsVYExY)3EIfGt1Z}gq~?<(^OHO=_{-`l>VzS`!@i<+#|C0CXhZFBWuS@PS#i0gsS z)UTTg&b^uty{Sg#)gRe!&u?B~Ok|0$>{z>N&yv`U1zYue_U~{Ko_geyd5grFgtlU> zzvcE98pR&{J*g+=;3XN86npQ~^`cF6@Amx`JAHe4R@w!>sNa7t?fn{|nYLef_c2lV zT+QHbf(^&(pSM`();zkBFX&c#c9F|YF~RGycFM(^W;>NGteJVmx!89@cql|K&}Lnrn90CO}YMQR6S78O;K4q zSL@Wa^B!$yQ%XzD?A?^uoAX{!<*xh8^LKOo);7BHn}qXkmD(_4A@}^BCLXsdkL$b& z|9bF+5F4NQnz{w|tG}w+)PL6Q*7>-0?Yol_y4y~!o5|`~EW7v4M9KGtwO{UXz5B** z%$0fTo!*WwZ{{;UTfp_TKxcx+yXZ^X|}f7 zh&o zb+Aiv412Uec!PAnj7trNMW+kRIG*C%b)wYYEN$oa*S899hiGmI$XV}Fe9EQlUD-Or za|#Zs^?O)fa~aw{t~dXte1MQ;)m{%@zn`u2=3 z4hPgrr^kgomEJYK=xhm-#g*0?+fCfpPykC=+zFY?F%m`@W6!-N#-&+;{Wn zrWhITERCsW*8OcM(W`BUNaOe>>oxh^MH%nR%|c%T=Qd5zu;5k+`PKE_ZI$RvcQ2bg zO}#6gHicv=D!Hva>gTdRThXmdftyFF$}?j1*}lCioFzTpX0Ca-iO=h6u$G=fCc8>x zUQ|WOL^%s1gNDl|Iv?*9sL@H;ThGL{;!=e1v68pW))rwm?tNZZQJ@pM>ACx9&V0ue zt2vfC)j9?Rzgnus(bS|9|FqvM?N&+4CE*K2YMZNk-+W&pKT#m(zwfPLhTVTp_-y&a zFje3ni;zlxggImXt&HoxK807$TYo!lvgx-NA7<~us)Zt7*W7e)Nbps>z_@CKiw$eN z{H!T0Ce!ct_Wf;rnAs&-^;)pKtZd>8hn*HlN78xbr1Q*C{9Eysr~lNI6N*j83$A-4 z1w@^le)+SbwXDtK?GH}%9Xy=3@$=Hv8tG3?3LZFO`)H>P=Dg2!^@smUoY%G_(SGNyVnmt$BawsmIdEiKl7glyTRwmWDXc>LJ!I*Jg)UfNl(?`Q7iY&Fd}X4F(vzk=)GGeh$OTOQ@? ze*d~g*vj73p@Q?T!Ln&yjo!1&>yGK}pTfT8*sJQfAFsH03pVdJ5OiHFw7g{g)2Vw* zt~c;LcYb~^cVSty-=kX_HNUO>e(e>Hw0yPn@_^{b-{o4%3xvzh6mb{MO}|&1J0XZS zQQ@G=ee0XG-(q>EO87VJ8g&~~m) zY=OaB=iQHWYqslLYn{k_K%iPEaNVQ}l?|KwH)#vMoG042#M$>sP-!g}gtmpl5Q_#J~pgK&P^%lAd6DhuLfC9UH2on>O4a;IMS z)2g-eR2X+nY;szhXW|lYB>EWZ8qph5#2%Y%(r;XI@xaaJ9qIK;Zr*8Z(cU3oQ`G(R zaQ2p>cPef7@?X3Ns{Njz!7p=7+pb=5`L7E86Gu89r5AqOF})>gXV4d`KtumezG4w8 zotHG8`1yLt(*p|AzJ5_|J2&fa#e}B^0~QxY3e;ycNPIiB`sMCV3-fBz|4jXTNAkb? z;z{!>Y*(fkdn>bTe6*$WWk5mrMPUxbypv&FTz8rFY}1;cxBF~CVD3g`zkl6-8ZJ(4 z^JH^g@r3WbM77fH+i&KkFKKbOt=nwyPrS5&yZrQFaf1!Dlg@QLDw@bSS*v3~HuJLZ zez7FSoS4>`^(L!)V+!xebn0HTT&^a&VF&->Q+0fAme#ULmRQywYnuO#*E;&MozxBE%Odq?jVgsD5V1hubn&^f(C>vGQt z$F9Jf#UG}*oz*M6_$MN-ztLRv=KiM^JC1lepRxJbXY=`~)AV|WABlNOx!zQlTEAs^ z_J8|-5hrE66P&(B((*Q5@v z%5lx-XO_O!D6={DpT{pvbFIernxC5MW0x_qD$mRdRgqDwTxQnCb)|FmQKg3)Cbu^4 zmXdmDnVeQ~UsEjP*6f2%>SwmElAE9}^Y8)FMWHJ<)%-Zw9w^j{K6lgVUd2()ZI-<9 zOWU~&w+Rxnt<$IMKHnw$>XQDt2-_3lGF!ElwOo>K_GIRAZwn6$Y4~BBJne}2x3>%Y z84TKZoAV~@SUyw6)jhPzr*JlAG#xzpwbT zcB|Dc+4xr{UPnCtw!~Y?hGW_Tx#L!?rsf;JO<1vI!~RS*_G*V2LX35%-Et${>R);u zsgs*EIh5yD=>~<4z{ZBecLJ-HGhNwru93kdS)=H#-Rh@Ghngn`?iP*y^gZWo@3Nfr z^FBV9q_tawZ^Fl4C*{1qe>^QzBGIuR_)+~!gWe+=#%KP1dsS)a)YTBZ=~n5>eGVnQ zX}hi`+?`(RKFeJ1>oT8d&Megpma9(b_{GUqcr(th|Hk)s(S=A}wHYQo<_zAB{at}` z6ME!CwM@(#Ui@6PScp~KZ^7$OO|_iqrD3;Umc32caeqp%s({+GMQ#fMLszfzJ>;`Z z?>JXT{bZg6**D$`D~C<{HYGP@QdOF1BfquWtjWJ`slGPN%bcCZ7a6wc75~MCvyz1ZK#k!iWjgQ|odJxt(Vf}WUwvt!R*hKf}pJ&;8ZcF%=<&*wx zRi62}VE&cDe^u8Oq%Cx_<9ho}&eG(SI1_J-MFUe6@7?!Pa?C5gTUhhmJU4M2o93Ro(v2ESU)`QM z+G<=3Dt7Bj-~Quh?%6#yrsvP_wrdv{>*xJ@!WjBo+{9^OvqJpr1hLl#eLn`cYV3U& zl(6_mzS2UC!yY>8pVw>pJveFlfz#ON+09_4nmdXwe$FT{*sbls94)YGp~b{`_xk3W z^v|eH*Lk5TQ_5}bb0P2or(14^f7Bnp4*hlW*b{;dsocmpo*?CX^w7=+{}$frnCyGw z(wna}Ey1lfJ=VW(4CK@9HZ0^0TPEz_`}*_JMdHsD?YEk^Z`>?gAEjSje=z6xwC0kO zgU^<&yg6Izne7zIbe)n1aTU$7yj;PBwmVK;behDQY=4&7@&3g3%Y0l}jc-T?FWDk> zyhQKruiWbUOac=3Y>w^goyVlJwz2B%o(1dtIFi=!tf~(w3)nX8qPY3QH$1i}*&-or zQA?K|R$N%)uzTLAOZ7AMZsGszc}lF_ebbVgD;}+_Ii<8F()_EX?e|O7hEr`$SNkvd z#3S?S>V|Gd|Jvo2JwH=BYqSj=R^K@xe%Hxv_p14Ymsr=$R(KoA%jXcCwPktH)aHta zHBEsBVlVB|*Gh1)S!yd(;~=nTb}`cwu|pz~BFcN#Ii6277VFdsGt_yr%<`;bn()g7 zOY2{z3b{;KIQ#gn=}Tutls{Phc*AjLuTtF;{l?Dj&u6ADWVx-9aAcF;MAkLGQ*=WU zo(D5afCC0@x~mB~?maxeSE^qGe`+>dx_zTwQYepS7ZBUyUx`JWZx#^8 zEwy*gS?pV|No4h(|FfnnQhRk^iOBYle}S*(&zq98VDo`h=buY1OnYVUSWt7;Ejd@F zY}E-|(zZ;`f*srDvwsZ!HFe^bg_TPSecMewc%5)Q_R@US>YnGn^f*lF=UMNbGk?Qr zpToJzN7q(b2GyL=_wD}6G>vD6s_i1yBiG-adywk9=)Ka@$+l;hg4d|Edj0vq=`y!# z{h7S{xF=gBw;sLR;poa;yQcq5{Z~2enTHlGUOLU$a__N4tuNBb5C53O_Uq-EADU-* z9YuYVAKjl^HSJNM2480B69N1C4-aves{v1e19_;)f#?E#Gl3+qaCler34AW}j1s{ub<9Vfiga_W~d1=jEY7 zHlj~FFFn8VA!E9s%`bt(Qy(VnYMpL+$}9<0=cq18ml#j)s- z-uudZB3ZkX@>8<=N@pl~tk`X^gKOdM+8Kw1)^0eG(pz6HIcH^1(_Z%Pu2;5L@SQ6& z(Ok$mYgLQ-$|N=4Zx`N_PSo8H>~`5a%Sg*FAodww)-q1M%hT28*5A{<+jgcn>bU5Y z?A?3Lh_)6+uK#Q~A(dtK-5MMAiU`w&^&2dfSzkXQwB+A{IR$6+UwsU8kd-?%Pa)yC z%;$wCqn^Dx5iEPDy}tLxCAJ>5I|0d)^-RSi_D_<1*{(P5LG||C^JC9ui}N$h_#=Jv zi#ju>t*xiet%Zx*4=J8|>o2o#&E1cNKJ`51rObO(T_)@C?wM9GZBur`%oV&op&Tjssu5a$!>u8QZ+GMCNUwWlm`C+FmPW}8D%uUzkL z{d3JJ>TXKF_L+~S8=T1w+`W*Y%-_G^t?TApdmkUGy86re%)vADr($BBoLwWNTv5u{ zuA1eta018vyAM8QsZAG~VDai90l|oA15*!91buYV>i2 z&~ww748Q*C{W$xK*XBY^&E4*aK};VezA1UtY^l?rcEU?jYt`yG`Vpe(&Y$=j&b`!( zTIDj)?Lxv=8|m_rsq5o4M0TAoOx!ayIOTlV-EGkq8Bf-zInDDlsPYfLWc+Ap`iygv zPuKHgzbM%3ef0BbuZI(#me|ROOXn}sFg&z<#>e@~zb`oz^mKDF1Dxg zKS};aB{v>!dAsytw9V1U`S%m#_wfduS9H9$^O&b=d%fiHkK%$CcI^p&&(^HAa!=#! zuz+Jtd!8t3zG~h&#b~ODL-VtDReI7Ty3@Y}_Q`_GC7{_m3YALB8))p3J{+3!i2&bytu2?c$#tg|nHR zpGN)IDp|cUFkyn9Po}Y4*1X9)yqmLRXWenFvtNDq$uZacvp+5mD*tqQr@C}*WU|)% zn1$H~br~%9WK&Fh4@^Juab@B<`D6W0YPuc0mv|kU&o2Evc(!xwwiORUR?WY@C`HD0 z0k@%rUA@oL>IwF@EPsR@S$I$2pmV*{Q=Y!}M#~LrGSx&~=U;v8b2WR8#_|)&EK9l< zh&SemAGiCZH}%aWwZ|Xq85kJ;|L=|Uo}{hFV|{T6$658gXRdg^t6#!yvQ5=FX!C4W zwU=zkDUsXS`xWXj%LDYn0DOLs16TO&|$-Sl_bE#^Iwnxq08 z>)ANFOHMpXkh(ZG=62S+ZJ7bGOE30B=mq?l8*^dm=6!|%dw#|||8Qwa)ehIReN{a{ zN1~tJUePR;SJr-gtB2v*A6sQPMLg`@Eq1&{No zZ674$6iRMMYizsuJcGqrxJg52-PLn}tUuTv^Dxzf&U2s56s+LP9e>vC*ekeU{fd3fjSI>&cME<{TBGq|-UqHtP6px| zm?zrpdm_y~hxwbxg3ZR0*WL79aclEE|Fm^Gz1w78+A|fLYF_AgH@rT%=-fx?Ew5RM zi?3cZHp+c$z`9~bl%%tooqDbfNGx+OUR?O5Y_T^}JF~ZPIdAptJeiblT1Jz2 zu0P~U>Jl^AuE4md-b;PqC(emwJ8T!s)OdB=HN^P9Rqc%WFMA^ISA7UedeYsn=)C&T zplLe#+a{_1Tp+%XlYPZ)%Yv;Px`Gkb!5`dOBJ&p8`ZrzoXVl$(ed3>(tFmdkuOFyC z{@{^9rnykVl0yOqcjPSh)BL{R%=T^72Na%rOz6n`F!$yX#@QiBqFb)#g{^1_YqBac zmk(dm>l-`y_lp_knr8#H-Zejw&m=E>(Q029|9$5}65-3k9Lg7@aESf$NKi2`dUh)G z&*}cN87IBUV$ARylfcWW(umkkX{(!WYWTJW@TxKGV+>o)uCJ$eo$1OPwhA5QzUF%= zbfJ{kGO;w)RnxS$?^fP>ej2w0+qToK&pD5Gw|JOmHR!a~1 - - + + - - -
-
-

- - The - LinnSequencer - - - 32 - Track - MIDI - Sequence - Recorder + + +

+
+

+ + 2A + NNI‘I + 6F6867# + XATALL + IE18-80L + (818)

-
-

- - The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is +

+

+ + 9SEI6 + VO + “BUBZIRY, + “J0aNS + PIPUXO + OZLEI + + + “Uy + ‘soTUOMOI,q + UUrT + +

+
+
+

+ + uu + +

+
+
+

+ + “‘SUOS + B + UIJIM + pasueyo + oq + ABU + pue + ‘posn + oq + AWW + AYN + IVNOIS + AWLL + AUV + + + “parlsop + Jr + SUOTIISUBI} + YIOOUIS + YIM + “BoueNbas + eB + OJUI + pourtueIZOId + 9q + ABU + SFONWHO + OdINAL + e + +

+
+
+

+ + ‘uonng + OdNAL + dV + L + 9) + uO + +

+
+
+

+ + sojou + Jayienb + Suiddy} + Aq + 10 + ‘syUSTIOIOUI + oINUTIAI-J8g-Jesg + & + JO + sys} + UL + ofquisn(pe + ‘ATTeouIAUINU + paiajus + oq + ABU + OdINALL + e + +

+
+
+

+ + (jouer + doup + u3a9) + +

+
+
+

+ + “puooes + Jed + souely + O€ + 10 + “SZ + “pz + 18 + [LVAG-MAd-SHN + VU + 10 + ALOANIWAAd-SLVAd + U! + patyoeds + aq + kewl + OAL + « + + + ‘uoTe1odo + [SVx + JO} + Aj[eusoyUT + JoyndUIOd + 11g + 9] + 98108 + ZHI + 8g + ‘poeds-ysry + Bann + soz] + e + +

+
+
+

+ + "9U0} + DUAS + 0006 + UUL] + Jo + wNIqUUr] + prepue}s + 0} + OUAS + [ITAA + © + +

+
+
+

+ + “ONYBA + 9}OU + poloapes + Aue + Je + sas—nd + jndyno + 07 + pewureigold + 3q + ACW + SL + Ad + LNO + YADONAL + OML + +

+
+
+

+ + "ALVOOT + 10 + GOLS/AV + 1d + ‘LWddad + “ASV + +

+
+
+

+ + SUIpNpoUr + ‘suOTIOUN] + posn + A[UOUILUOS + 94] + JO + AUBUT + [O1]UOD + AJ9]OWIAI + 0} + PousIsse + oq + ACUI + ST + AdNI + HOLIMSLOO + OME + « + + + “SUIPIONAI + I[IYM + P2sesd + JOU + Iv + $3}OU + BUTISIXO—ZUIPIOIA + SATON.ASOP-UON + +

+
+
+

+ + ‘suoneurldxa + peuoyippe + sdeydsip + uowng + g1TqH + +

+
+
+

+ + oy] + ‘pepsau + JI + ‘suoneiodo + [ye + yYsnosy] + NOA + sapins + ApIespo + Avfdsip + QO] + Joey + Z7¢ + 9y3—uoeIodo + Urea] + 0} + Ased + ‘aus + « + +

+
+
+

+ + jUorel]suowtap + & + IO} + Aepol + Jayeap + uur’] + INOA + dag + ‘dISHUL + + + INOA + 0} + UONUS}]¥ + PaplAIPUN + INOA + SUTJOASp + ITY + ps + pue + + + p1osai + ‘asoduod + no + Jay + 0} + pausisap + st + 1s0uenbesuur’] + oy) + + + Aum + Aposiooid + $,Jeu], + ‘SS9d0Id + SATTBS1D + OY} + YIM + SOIOJIOIUT + + + yey) + xo]dwWI0d + Os + dq + JOA9U + P[NoUsS + osn + NOA + AZopOuYdE} + oy + +

+
+
+

+ + ISTUMOIAUIO?) + NOAA + UOHISOdWIO) + +

+
+
+

+ + "NOSpr] + B + Oy + ‘AONUTJUT + yada + 0} + seq + Maz + Se] + BY] + Jas + UdAd + + + uvd + NOA + ‘palisap + JJ + ‘souanbes + Mou + ¥B + OVUT + sjied + ou] + [Te + Adoo + + + ATesrewO + Ne + WI) + [IM + ONOS + ALVAAO + JeyIe80} + wey} + + + ,deyd,, + 0} + UOTOUNJ + ONOS + ALVA + ou] + asn + usy] + ‘saouanbes + + + JENPIAIpUt + UI + (“949 + ‘snJOYD + ‘aS1OA) + UOTIDIS + JIseq + Yes + + + Pl0da1 + OF + ST + ABM + JOuIOUY + “(812g + 666 + 01 + dn) + ysnory) + ABM

-

- - extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: +

+ + dU} + [fe + YORI] + YORs + p10991 + 0} + ST + SUOS + B + 9789I9 + 0} + ABM + SUG, + +

+
+
+

+ + SUOS + & + SUTVAID + +

+
+
+

+ + *suoT}oes + poJUBMUN

-

- - ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST - - - FORWARD, - REWIND, - and - LOCATE - controls. - -

-
-
-

- - e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may - - - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic - -

-
-
-

- - synthesizers! - -

-
-
-

- - ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes - -

-
-
-

- - per - disk! - -

-
-
-

- - ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. - - - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. - - - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected - -

-
-
-

- - rhythmic - value. - -

-
-
-

- - ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. - -

-
-
-

- - ¢ - Optional - SMPTE - time - code - synchronization. - -

-
-
-

- - © - Optional - remote - control. - -

-
-
-

- - Recording - a - Sequence +

+ + SAOUIOI + 0} + ABM + SWS + dU} + SoyeIodo + SUV + ALATAaG

-

- - To - record - a - sequence, - simply - press - RECORD - and - PLAY, - - - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s - - - click - track. - When - the - sequence - loops - back - around - to - bar - 1, - - - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be - -

-
-
-

- - corrected! - (Timing - correction - may - be - adjusted - or - defeated). - -

-
-
-

- - Any - additional - notes - played - will - be - added - into - the - track - - - — - existing - notes - are - not - erased - while - recording! +

+ + “OBPLIq + dy} + PUB + SNIOY + PUOdAS + dT]] + Ud9MIAQ + SIDA + ISI

-

- - FAST - FORWARD, - REWIND, - and - LOCATE - controls - - - may - be - used - at - any - time - to - quickly - access - any - location - in - - - your - sequence - for - spot-recording. - To - overdub - a - new - part, - - - select - a - different - track - and - start - recording—while - you - - - record, - the - first - track - will - play - in - perfect - sync - (unless - you - - - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 - - - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded - - - including - pitch - bend, - modulation, - velocity, - aftertouch, - - - sustain - pedal, - and - program - changes! - -

-
-
-

- - Editing +

+ + ay) + Jo + Adoo + B + JJasuT + WYSE + NOAA + ‘afdwexs + 10.f + ‘UO + JUSIN]JIP

-

- - To - erase - a - wrong - note, - simply - hold - ERASE - and - press +

+ + B + IO + aouaNbas + sues + OY} + UI—JOY + OUP + 0} + UOTIEIO] + 9UO + WOT] - - the - note - to - be - erased - just - before - it - plays - in - the - sequence— - - - when - played - back, - it - will - be - gone. - Notes - may - also - be - -

-
-
-

- - added, - erased, - or - changed - using - the - SINGLE - STEP - func- - - - tion. - To - overdub - notes - at - specific - points - within - a - sequence, - -

-
-
-

- - Additional - Features - -

-
-
-

- - simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to - - - find - the - desired - bar - number, - then - start - recording. + + $1Bq + JAOUI + OF + NOA + sMOTIe + WOTIOUNS + AdOO/IMASNI + OULL

-

- - The - INSERT/COPY - function - allows - you - to - move - bars - - - from - one - location - to - another—in - the - same - sequence - or - a - - - different - one. - For - example, - you - might - insert - a - copy - of - the - - - first - verse - between - the - second - chorus - and - the - bridge. - - - DELETE - BARS - operates - the - same - way - to - remove - - - unwanted - sections, - -

-
-
-

- - Creating - a - Song +

+ + ‘SUIPIONAI + JIVIS + Udy) + “OQuINU + eq + porisop + ay} + puy

-

- - One - way - to - create - a - song - is - to - record - each - track - all - the - - - way - through - (up - to - 999 - bars). - Another - way - is - to - record - - - each - basic - section - (verse, - chorus, - etc.) - in - individual - - - sequences, - then - use - the - CREATE - SONG - function - to - “chain” - - - them - together. - CREATE - SONG - will - then - automatically - - - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can - - - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. +

+ + 0} + CNIMAY + 10 + ‘CYVM + Od + LSWA + “AEVOOT + esn + Apduns

-
-

- - Composition - Without - Compromise - -

- -

- - The - technology - you - use - should - never - be - so - complex - that - - - it - interferes - with - the - creative - process. - That’s - precisely - why - - - the - LinnSequencer - is - designed - to - let - you - compose, - record - - - and - edit - while - devoting - your - undivided - attention - to - your - - - music. - See - your - Linn - dealer - today - for - a - demonstration! +

+

+ + sainjeay + [PUOHIPPY

-
-

- - * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the +

+

+ + ‘gouanbas + & + UTYIIM + s]UTOd + a1y1dads + 3¥ + $9100 + QnPIOAO + OL + "UOT} + + + -ouns + dALLS + ATONIS + 24) + Suisn + pasueyo + Jo + ‘pasesa + ‘pappe + + + aq + osye + ABUT + S9]ON + ‘U0 + 9q + ]IIM + 1 + “yoeq + podeyd + uayM + + + —aouanbas + oy] + ul + skeyd + 71 + a10J9q + Isnf + posers + oq + 0} + d]0U + ayy + + + ssaid + pue + ASvwug + ploy + Aydunis + ‘jou + Suomm + & + aseso + OL

-
-

- - HELP - button - displays - additional - explanations. +

+

+ + sunipa

-
-

- - * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. +

+

+ + jsesdueyo + ureisoid + pue + ‘fepod + ureysns - - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + + ‘yonoplalje + ‘AWOOTOA + ‘UOTyeTNpow + ‘pusg + youd + Surpnyour + + + pep10del + are + $199JJ2 + TCTIN + [WV + iPeqqnpseao + aq + Aeur + syoen + + + Ze + 07 + dn + ‘Kem + sie + Uy + *(foeI} + JOyOUR + OJOS + 10 + ALLAN + + + NOA + ssofum) + duAS + yOaysod + ul + Avy + [[IM + Yow] + ISI + 93 + “prooar + + + NOA + 3[IYM—SUIPIOIA + LIBIS + PU + YORI) + TUdIOTJIP + B + JOaTas + + + *y1ed + MOU + B + QNPIsA0 + OL, + “SuIps0daJ-jods + 10} + aouanbes + mno0k + + + UI + UOHBIO] + Aue + ssad0e + ATYOIND + 0} + owt} + Aue + ye + pasn + aq + AvUE + + + SJONUOD + FLIVOOT + pur + ‘ANIMA + ‘CYVMaYOd + LSVd + + + {SUIPIOSAI + {IY + posesa + JOU + se + So]OU + SuTsTXO— + + + yous} + 3U} + OUT + poppe + aq + JIM + poteyd + sajou + yeuonippe + Auy + + + *(povesjap + 10 + poysn{pe + oq + ABW + UOTIIII0D + BUTUTT]) + j{paqoeLI09 + + + 2q + ][IM + S1OLIe + Sur + [fe + ATUO—patey]d + nod + Jey + Jedy + ]],NOA + + + ‘] + req + 0] + punose + yoeq + sdoo] + sduanbas + ay] + Udy + AA + “YOu + Yor + + + §,sa0uaNbas + at} + O] + SUIT) + UI + preogday + [IW] + INO + Avy + usy3 + + + AV'1d + pue + (YOON + ssoid + Ayduus + ‘aousnbes + & + p1o09es + OF,

-
-

- - ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. +

+

+ + g0uaNbas + & + SUIP10I0y]

-
-

- - ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. +

+

+ + ‘JONWOD + s}JouNaI + TeuONdGO + e

-
-

- - © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. +

+

+ + "UOTJEZIUOIYUAS + OPOS + UIT} + FLAWS + [euondo + e

-
-

- - © - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. - - - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, +

+

+ + ‘sou + .sulddoys, + noyyM + sayelodo + pue + yoegdvyd + ZuLINp + S¥IOM + NOLLOANNYOO + ONIWILL + e

-
-

- - (even - drop - frame!) +

+

+ + ‘onqea + ory + AY

-
-

- - ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes +

+

+ + pojoojes-oid + & + ye + sajou + pyoy + Aue + syeadas + ATTeONewWO + Ne + UOTOUNS + [WAdAY + OAISNOX + e + + + ‘LSVJ + SUnIpS + soyeu + UOTOUN + ASV + UA + OUlN-[eal + SAISNIOXY + e + + + ‘Koy + B + JO + YONO} + 941 + 12 + CASOdSNVALL + 0g + ABU + Syde] + [Te + 10 + 9UC + e

-
-

- - on - the - TAP - TEMPO - button. +

+

+ + i + ASIP + Jed

-
-

- - ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. - - - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. +

+

+ + S9}0U + OOO‘OTT + JOA + SpfOy + puv + SpUOdeS + UT + SBUOS + Xa[AUIOD + So10}S + DALIP + YSIP + , + 74 + € + ISCJ-CNIN

-
-

- - linn - - - Linn - Electronics, - Inc. +

+

+ + jSIOZISOUJUAS

-
-

- - 18720 - Oxnard - Street, - Tarzana, - CA - 91356 +

+

+ + stuoydAjod + of + 0} + dn + skeyd + A[snoourynuls + ‘spouueYd + [IW + 9T + JO + duo + 0} + pousisse + oq - - (818) - 708-8131 - TELEX - #298949 - LINN - UR + + ABUL + YORI] + YOR + ‘syous) + oruoydAjod + ‘snoouelnurs + 7¢ + SuTeJUOS + ssouUaNbas + QO] + OY} + JO + YORA + e + +

+
+
+

+ + ‘SJONUOS + ATWOOT + pur + ‘GNIMAY + ‘GaVM + OA + + + LSVd + ‘GYOOde + AOLS + ‘AV + Td + YIM + Jopsocas + ade} + Yowsj-N[NU + O} + eps + st + UOTLISdO + @ + + + LOPNOUT + SaINjeoy + s[quyIeUlss + AUB + S.JJ + ‘OSN + pue + UIes] + 0} + o[duns + A[suIzeUe + JOA + ‘PnJsomod + APOUIOITXO + + + St + 1] + “UeIOIsNUL + feUOIssajoid + oY} + 10 + JOO} + soUBULIOJIJAd + pue + UOTIsOduIOS + 11e-dY1-JO-9}e)s + B + SI + IONUANbDaguUT] + ay + +

+
+
+

+ + JOps1odady + soUINbIS + [GTI + YVAL + ZE + + + Jgouanbaguury + oy

diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin index d3c2e860..137fef56 100644 --- a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin @@ -1,123 +1,124 @@ -The LinnSequencer -32 Track MIDI Sequence Recorder +2A NNI‘I 6F6867# XATALL IE18-80L (818) -The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is +9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI +“Uy ‘soTUOMOI,q UUrT -extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: +uu -¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST -FORWARD, REWIND, and LOCATE controls. +“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV +“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e -e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may -be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic +‘uonng OdNAL dV L 9) uO -synthesizers! +sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e -¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes +(jouer doup u3a9) -per disk! +“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « +‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e -¢ One or all tracks may be TRANSPOSED at the touch of a key. -e Exclusive real-time ERASE function makes editing FAST. -* Exclusive REPEAT function automatically repeats any held notes at a pre-selected +"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © -rhythmic value. +“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML -¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. +"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV -¢ Optional SMPTE time code synchronization. +SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « +“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON -© Optional remote control. +‘suoneurldxa peuoyippe sdeydsip uowng g1TqH -Recording a Sequence +oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « -To record a sequence, simply press RECORD and PLAY, -then play your MIDI keyboard in time to the Sequencer’s -click track. When the sequence loops back around to bar 1, -you’ ll hear what you played—only all timing errors will be +jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL +INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue +p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) +Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT +yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy -corrected! (Timing correction may be adjusted or defeated). +ISTUMOIAUIO?) NOAA UOHISOdWIO) -Any additional notes played will be added into the track -— existing notes are not erased while recording! +"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd +uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo +ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} +,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes +JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes +Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM -FAST FORWARD, REWIND, and LOCATE controls -may be used at any time to quickly access any location in -your sequence for spot-recording. To overdub a new part, -select a different track and start recording—while you -record, the first track will play in perfect sync (unless you -MUTE it, or SOLO another track). In this way, up to 32 -tracks may be overdubbed! All MIDI effects are recorded -including pitch bend, modulation, velocity, aftertouch, -sustain pedal, and program changes! +dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, -Editing +SUOS & SUTVAID -To erase a wrong note, simply hold ERASE and press -the note to be erased just before it plays in the sequence— -when played back, it will be gone. Notes may also be +*suoT}oes poJUBMUN -added, erased, or changed using the SINGLE STEP func- -tion. To overdub notes at specific points within a sequence, +SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG -Additional Features +“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI -simply use LOCATE, FAST FORWARD, or REWIND to -find the desired bar number, then start recording. +ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP -The INSERT/COPY function allows you to move bars -from one location to another—in the same sequence or a -different one. For example, you might insert a copy of the -first verse between the second chorus and the bridge. -DELETE BARS operates the same way to remove -unwanted sections, +B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] +$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL -Creating a Song +‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy -One way to create a song is to record each track all the -way through (up to 999 bars). Another way is to record -each basic section (verse, chorus, etc.) in individual -sequences, then use the CREATE SONG function to “chain” -them together. CREATE SONG will then automatically -copy all the parts into a new sequence. If desired, you can -even set the last few bars to repeat infinitely, for a fadeout. +0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns -Composition Without Compromise +sainjeay [PUOHIPPY -The technology you use should never be so complex that -it interferes with the creative process. That’s precisely why -the LinnSequencer is designed to let you compose, record -and edit while devoting your undivided attention to your -music. See your Linn dealer today for a demonstration! +‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} +-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe +aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM +—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy +ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL -* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the +sunipa -HELP button displays additional explanations. +jsesdueyo ureisoid pue ‘fepod ureysns +‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour +pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen +Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN +NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar +NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas +*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k +UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE +SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd +{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— +yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy +*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 +2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA +‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor +§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 +AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, -* Non-destructive recording—existing notes are not erased while recording. -¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including +g0uaNbas & SUIP10I0y] -ERASE, REPEAT, PLAY/STOP, or LOCATE. +‘JONWOD s}JouNaI TeuONdGO e -¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. +"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e -© Will sync to standard LinnDrum or Linn 9000 sync tone. +‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e -© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. -* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, +‘onqea ory AY -(even drop frame!) +pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e +‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e +‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e -¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes +i ASIP Jed -on the TAP TEMPO button. +S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN -¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. -¢ Any TIME SIGNATURE may be used, and may be changed within a song. +jSIOZISOUJUAS -linn -Linn Electronics, Inc. +stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq +ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e -18720 Oxnard Street, Tarzana, CA 91356 -(818) 708-8131 TELEX #298949 LINN UR +‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA +LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ +LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO +St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay + +JOps1odady soUINbIS [GTI YVAL ZE +Jgouanbaguury oy \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin index 415d519d70f90431e8b334ff0521a9751a1cdd58..5c1ac320d98c4e47d50d19b57d4ce6dd21b32076 100644 GIT binary patch delta 9946 zcmZ3H*qO9pJqNS7x%uRWj7s%$Vo&zzSm)i{zy5^$iVe>g4S1fcO?)`FakufFO&52^ zZJGYBDqG>G9^nc=Gm9te*b^>-{)`Zk5#1mY^XnG z@i~9{|6PCoeV;8GKI8WL|2wy;KRkZO{7*=F&ddEbRrO*jV~cirnESt-ocZq?Yu)p* zrRm|-9SO|1NC(w(7%^B>nfyygPUej(oGQmVdi1f#can zT_M|>qCYI+&fWjM{{8)S$^MsOxl=F4{w_(M_hetd-$MneUr+sw zoEe{TnKgLk?fUqo7na}dD5{?R#rSNNPsnfUnuGZ>mYn(S!g1Q==;4Q|zYpKtbC+9s z<4>)JVm8|ar|+J4Wt&*tU9;mqi}&1Z-`TVA^Slq`tG1q<;GZkmeednc+!qJL%l(|< z18ORF?pWq~qvocA^djlYzdUv^No5HH8*q5{MSM}I`B^@xd zl_)&(`MF6__U1o1bET))MNc!|cGf>dd+wtTs`pdgB)zTO<8vVUu-US+|2L)PzE7CD zN9o-1k8|JL7TtP#U2fgtzRcb;`Ss`T?s>CFH|E!pzqkHde^=2BGWq=equZ=kwQ;-b zqSjro&-hdkmpED1%#mUD+Bf@m|9pQY*K$Up@br(5Pagd|LHr?Ka#ejs_z7LJy$XGE ztuIvFu*r43ThY8t@E7y%CE4xv&)C;&*UvB7vOS~nvO%xer-vT4Ys0&4i_M>J!nw#t zp#I*j^MzODNtt;|r%OD}$=BR2&bMj3A79P^(LeS#QXb7SnX=k9V0U&wmjBm`s$0Ri zS3=LezR52#E9AD1y}P=OvcnD8kLzde?@Lqc?YR5>49~4V)~zjOUws+c!junW{Eg_* z5_?>!;F){f`K#--Y*Q8Cjy1npw;AEtkK=@pxK=HTU)(OeZ1_SUkMX zEL4%Y;pD=TF5Vq;XUE=(v%h27Vy3+*I{3(|{+ZvizRB7JHQ6Vgn%cAYZU2NLua}!# z=e%CY$j%~s=Ui0J!I%ZQefQ4I5dJz}GX$u+mLTwhm^by%J^7%rl0pS>Xz7;P`{ww zL}0Dop@{nj&p+$hX0apa(4Ci)zV3`q_+AbYNks$BsKV&8v8`pBFHBBwU0bGpGJ^eiLhj=j-OSsijh91I z!+YA&L+9~ze!syv@9p+P&L5MX7~d(fyy;!U^Drr7vDfKLqh*t}9Y5K6T`VJF*Ogew zs<%14DXOfciJE)V9S!?mJWq%=R$Jr$Q2DvERKy9_GhWknXSpt`|CSz5^S+z2KTzjw z$I_`&nbv%{!?Ax)kLLI5A2JJG9^`FbxA{=pQ>Es65-R%}S6tjW`I&%N#&n&d;(NC& zkWWyxSh~mTtNNd9Q{+!fHAtBo>iBlanJ^lZ|f`6@(4d*?0FuWpS`%WKlDVRaP*nd2hF8_mmFnSqqnu? zka}p_kE0UREZy_@y&@aGMdy7xcxgrB>!!Y=a*OM~b3fu(IrGue8yjA%N-XH+?#_BV zuZm^s{LG|UmDLiOdgrtsq@Dc5dx&Sjn~LviJhnQY>B~Qwee|B|xx3XH6#27{A8s#S ze&nC$M!k0JH|t8Zbv87`P7+)D=h>>bnVWuwyLjlN=z4Ef-2V26fS2O@#;+13&tI(8 z2;I&dIdATYM|J1^Ji4PJ)Vb$wz4WE+2M+Db=VD2I z!Zu_23A@iC7mrO`y(7YZt7385ugY2tBLmh4D<^Y#YAC%5I%_gn|I%vrU5eA3-Mkih z=7~vdO5$(iKRHz>dJpgM{i`qJ_9O%xU%c7!%C3&el%C@M9qpwr`n0C77uRqd$XqeM zDl#)YA$Y1^z-PWW&Z(@5)9%(MTeW?NzG1lW9M@&<_nwTab?v9B%IViTee=B>a8S2j zaxPcO&R0{K?-VAUX7*!yY?suKo403gM^L(-=vHp8En(LdyZqazcJ4rSZT2y1ofVhA z^`2UkkiIzC`HYkJ@vXc?4Hs<|7By6Aa--D>N(A*8$S>6N(Y@(edjzut^UAU zr_bWyVuGt4Ev?EbEq^95ap$|_CF^809WQNv!(wy)qFd6kSY6HuQ|I0M`}7~jyb|B( zbF%8S0t99zNE;ZsN!jJ8$NsO7oZD}AonLQ*?%b4>%Pw!7Q`dIF@Qs{0ON8P+ad$uI z6BYiSe{cA+Q$|_6M$tuW|ynv ztqH&SWVXy*2G99N;^KBjOtWK_oH$R$^;u_~maP-=>1n%U&L?RY@$cj5%sG7(ul%OFu4n#r zX))I==7{?J8Vjr!+xE1`C)rO92xFPJ_}cpQnO~E9MQ^;Fp4iF8?7hlJ&L`pJDuv1! z$3H$Xn{FN2%E!?@a1yT4eHI=i%1b(Hdeht4le4trk2O zw(ivmxsaz}?YkeppSSS>!!O-);uSM&&QIfGHPg1U|E$&=SD${gpV@5k(QkQt&iWV6 zD*m?J>pU;?$C^D3{`&KB3zBB?wVjvW9AaN{N}~8&f~$YF%!c4D_5Pi&7vbU*&ubJaV+MWgBOQgIPTAZ(6@+qRq+3>$$ z(|3_iB^ND|R^B)uax_y`+0Q-IZ_SCNo-3wbT@bKqo6ZtP3&Eb(c0cl)o9a8e%ozEy zj`&@3QV5Bd`|x+5nMh{!jR1=siJO(vIy?`pQYaG?5^0N6Sf%3@VE@lGNIFO&=%nMS ze37P1WA{p%1`Mp`^Vm5AIXm$vg za$icjN6%5yEOY*?*$1Q;Ud5S;oqgxx(bIhAPuXuU7&Pkyi3sagN3a~!gFm2(rsO;{@h=-zYcH~Z*>Ttz7Ph6zkGi;9QY6afU zc^>tvb&|{{@=B@ZhPlr!N^S|bT9VT8?U(u?#hEFkin(&U%7u>t9!~l#r?=-?`pbuHW!#o~)wXmft)#ZiWjkcy*FhyPi8SY|()Y^+BsH+49`WuBsQ< zP{3X2xjtW_{OLW8DeJAC`UdUrQe>0JILF4nMf*wxTUt%olRat^4VJ%bG~rslEo;`A zd75{a5{hf$U%z?!uV&+>fVh=Q7p3>Mtk9qMzj^PDHz5HEGpgl2NPatzVP2b*y3nfo zm~P3tGnTi^WmE3gKU&zh*65JyoXq0G30J+2_e@RI*~*YRv2mL~!{T zhRswpTO<7YR|?;n36bKnUW7haRb6B<`}(qi0$rVtEZWNtF|6R4zF9TvoOrp%E}ioo zI_IpLIiq_W6;iDZOd{9h@;-J-07&r*3HL{ zsO+8A5U+ZEgJ{?EO_vllZ#38P;GY}tGV?_Ko7OW|n;FkJ2ZVLA&s@JKYv-becQjWO z1+Dv1u};tX%VH+3+BojjOJt(&%uFp=@M+SCoC~PwIJqS zLH%=Hxd6dQPfsjbHhJZPck7LJ?prXA&-{7g^NDT0V)%Fi<`nNaedvnDBE>nnY`-6yi#Z^OLq+&w^*!JS8 zcFQVJ>*lf(S*(vOJQu88FgK^#`&#Pms}Y9m6Zk^Hgz|Lu#WkuF=k=!+PYZ3^xhmRE zPdVa5jBE1pzK-ad=H1V~p1k2T?fEr{ThC8SlDWjV%kEQ&0^?S>{e63Q_by!Frl9;- zad(W)5)TcXdyMUC>yLPFNQ%B=cAe&3Rc}x^W!+-OyG!1Dyd{^wn7u9dk@sEZjYp`L^}_hyE!qEMQnMv^e6H9RPl=%*^)Ou z98cIhZ{iU>gNt(8C#`Z%Ey(uoKc|sBccz`4cCyumeW5yM_PrA7Z~tLFMQ6#k>h+(# zZZs;mT5hxW_LX}dVm^sHO{)!(yKf-oxZ;WNw<9})lRm8Zb-Vw1&`gu_#+A=LmwvI3 zd%Ac{{`z@7TI&jIlH(sOu)18l*iX3r)PkJqGn~=6c^poPv$Bokc3lzZJNAySCe@l<ma}&9n}?T&8jHP4AJ)%STTeZ(Sh#`$R&%MWa_ue^7ZCOUM^>wRajN7pE%fJ=I(z zE}9}4U($WnDe`80)ZEApmF#s_?zBzK*KLda^~r!+ZpZaKlcYEK*_KV2#C;^ba`85G zYp<13Gpi2PaQp9@y51vf!c(`ay3@G}HaA)-EA;63fBmyGL?coCZ`hiO-KS6LUf$R- zrB6P2@d^23E6yqZUb1WE_I4YKP5pTXWU8J|tp51uqGy!P)bsoctmf8doUnXkW**7b zRL)l@l3X+I;o73oymJnwt2Zpy{jYu}M=Uw_g?mZC=jb_6LBDP-Uso|Zr&xmH%~wU` z)lEmVKkeE4uw=y(t=*kJ7$r*eANybASeUf2Y!7e1hD?_QerzXB1?zlV!jijxb;aLv znbqF{p8uJCn8~_+t;@yBhb?;5^zAxt>ixP=?sUp(X|hqU{_KuDI_VpKSjv^!E$qpExUaL$ zc9x*>Ha&i(%4n1A21ef#B?{WEe5u=@a!r{dto-Dgy3na2zJWh3@U>1^>eUm_8?>al zzUEpgWb}{22DD%rid%tH`pf0p3jp!Tlz@%QIQMY785US zbC;dG-t}Y7q(*^npFAQ(KB;f+OBC4Zt;BFGuI$x|o&K9w8@nG%z1dt^96B{ajQM(F zlBs9m+=dT+&#hl^u9_qe*Y#IlIraXTCezNX&W`nIo90<=b58j^QTn3VX{+R`k&F5J zvkVr0Et74D%v0n~xPEDZqKT5kd$*VEr(Z?H9QQQSdbF?MZeq#motJ;II4R73_v&tQ zWs|yx!us1n3Cq>_(&94S{C^>kQ8}Z`aP_n^$)`16zZMlJmebImlP}eECurTH-)a1k zr#1_nPf$6ss@^(J^NQmu3Cx zSEj$yc52i5vxM*3M=q8-VcR<=xqm2bm%QCwJei;Cc;vhJ8VBQl+sS`CbYRhj-yO|H zPnc3I?0p#d4ET5&IqX~#;*556Mt^JSyzJJ=)E>DxT{p+2Y*Bq31Gi+{oXE}q_EV`m zdsq`ZCMbtA6nRb0xiioD($_UwU6Yqa+1iS%6jJ%b@z5w?(be}o7B8QtD9KH1`dGB6 zPe3%O-|g0(>wg=zEH8Z1et4bn^^&{0wx9IJNYIKFfE)f4kw?EsPs#1CI7@d1}Iaz5n88n~eWOY#*0R z6j$7R zUP-F)HFK%8hEhU$Db=PTQC4<5<*WQt?H8(Z-g~}tO_FL*4u9ii=iet5JQtD^d2u&% zg_uz6nffx8i?20Kh(^Xcy0mu{<|w;8s($f%zlQRijt_NiKXRvTxH*&S#gDHuxu(`~ zAARv)y5+sel&J&)>J`n&ikjF;Mdt)2+#+mC-elUBUg)Uq!xH-QBMFw%*pG*3sTN_Cv8B`*L;d zJ`TQ@6JEz>n6DJREhl@r^6iK|0-t{zuwL0mDMr7sZTWH zl*5`IpX371*iM-?S+%-8etWy`TQ#dA-&4;T{k;BIVo%9t zFP?*pS1k@j-Q#oqb8Lh3q~$h=vD=w=PhBtFe{k~U2PfWyiY*d)^z!(kRo53rnayjG zthp`e))!^AP3>? zj3Q}!Zq=^!n`*E^K6&1M^?i$1&3mNoxQ}Fo(?eldi@85%MNYh8VeAp9==;%A`BT=%+`KsbhJ1~zcYoA8V_L87dcG$9 z{nIs((=uJG6?sDP%9nh7aAl!D=<)+%ZV%t=E^`gbJh1aH7f+II!32ipeLceGW*yc{ z%VXdVQ%tDu`y2jwwqvAM_m7DA#s{9SFz;V|ng7eNR=*!h<8*HypZ>=`WZf~$=wNp=iZ@V8FeM7YQ`bO=7-%dSG^;S*1q`YS0hRf1R zbwwE>e(V*^>u&yiLd)vwf^BcD_`kFtvM7Jmy7lME%a>eaU#_h0&~7Ny`}8cJ&Gvhg z^Q7(ay$jFl70y{R{at_$%OUR%b0TYRDkphL-j@0C-IDih-75*zkd~|$d=081Z`c=} zes|$to8wjKStr_8{S?@sxM|*LiReu#&jVAFqc>@GyjQWka%j)&C!J3ndZKHla_-jk zSa&GN+|}lJ?POy$n}3RLF4Wf*rp>YsNu7FbnA;e`iJol<&#G+lDRftljgOW?@v=kwhTRwE zJz_f|xA9*@oX#;fCBwd-4+>UK-)YKt&z65#>LI?1^PaOWbvhLG^ZAuPE9I>w`%N+! z51TW-@AuGY|bW4b*8gxiF(y$?N{{f}0QARC!dE z%sT01Bp^I%f!BCuv1B|m=>_v<|VrHhP%<6W~0Z-!n-y_bIFzJc*0lSb)d^*pWrIklX>mWcninG(Lo zz3If#W{sj}Y+sDSZOkt8MwC2OThH4*m8q-plBkcy*Q%=DwYGO(KIOi$!#Q<=n@UX$ z&qTcgebRk`mJGaUsZ}SE(p&EH_e2OCnUTga`FvV?>@#bV-TqAf9*GJ4KT|h@Q}|D! zwu75E&)k(C>z8+>sMbg9+3=0CucKK0N9+65$qS4g+leix$qchvx$E32;hpz<6f5N) z@9SG6pLtkyRaxUCsSkO7=hTUXsZ7 zE8&^A;B?4M$E#hp1gCNNF09AYxmSWbJLVlD$EF&dDQ;wi#V;P zUv|HG!d7-WdAr+0#_8H_7hmr(YfQ@D&Acz?)O$_g74PLGHYK*KdC{pQ+Vn`CS6yP; zN`I{sKNGv0y$WTeV&}Rp&tEmU{l)wGC5BtwE0&)6`Elm2MSOS8*hqcNW0v{F*wC-L z%uz#m#mOw+*#|j0^e*n5Y#PPB<3zqDr<;;^?{YU8rMq0GLnRvK><#kN45S+%WY{&X{+8P$9A?UWCvNZFgr*?m(czpRjBpPz4E#yKuN z@%n|5>3$7c3_DdbJGK;@J5=&NZT5GW1veA2KRCRV68D-gVa4rMaZe7GOOK|Dlze_S z;Wx(uCzb&FzSn-`!6gnqO?J8~XK1o&vn+V<$*5`W)m=IPiW9DCJod7XtM9*cQ;m0W ze!j;+|0s_6n=G$J%SmVNJ~?G}M3ZM=%*up+i*Nt054rzZI(WN(MBXOTDV}aVv0H4E zwO;*I%gX(i}+=2;&FLmBe)=CZoYt+)Qs-bck+y}GlW-pt!3laaWOy5UQ+z< zqEFP$N%PLGG^*dc#Y;*@vVtN3NR zX2-p+Ny^1+T#pT(r|b!N995>VgRN5H&nvD#@rh#kcfJ;B2&UZmRKK`#YeW8uU3LqF zymvnGG12^Eo)m*V^CDX`o%oUe=I+}iqXiV((u%QU-oxgAMPuXp0MudIgW|K zKeoh9U=dPS7R9W7xOMey_8U3c6C_&_KDz{e-nW3Qe;=z!d-g{D=YcG?9A_WCQarDr z@6vau>Yv1&Y28yzoiunvrzGyKk6p4u{aZ>`%)C?fIp5glyUve%U_4uk{an9?|G^gC z&RGgqtoF)nz46-O{D=5EZ*~iwKCvfnW7tRcdfz27-_rcAq{Yd&Oh9m1z3s1q@Mj*_I;jUSHPU5$)x{+5Vw@Z|>hiJCwfG zKh5Xbe*1;;f9>@bQ&;J)NXh+?(x16)=Y`(^*LTjaFgNw(to~q7>DITpn(28Wv=%oa%Lv}cFBwSTe4Jb`PyvlwI>*& zlcyFkzw&TqEKu>f+a~*bOQPCji`$QnoOmkwZ}N@$(4C$&Ut4!wV?Q>l;OooqhwI+l z)!!R*U+TEsM;>i=&djx^TwKa`P4jSyI=GER;pp7L)<%f7gNKhbM7unpk}xGnsvt{-QM7yMgAu8ZLtC!ut;wGM5!c+*i0!aOmCH8=uuX zb_7*!R$crlJ@Zd}S)u4+5jWpGa@#gyKv-aY&uVRfxLeG}C;V~Ro&Gkd@Tg(^Q{|nvzx~}| zRZ)~Omr+?-h2gfe#$$z7{%dX4zRNmov8O#J!b{5C_-Lf|b_bg~stg`!S)ccMt1&r$ zEAM#N$@RPKMww!&)k~8JlMS5ICm5WVa6$adBF<{j^=%K~erdGsM0riycsuo+e6-)Nr zY%AU~)8%V^ruwx*OBo&tiC#2oPu=)dNyllyp?B%m+Bl>+70)~|O<7@kaoVlvPh(!J zssD0y-kjR;8{z&c@Q-AKLpg+?sy%)y`1;X}>G) zar(NLbNBb&sa?N)+oQ)fro8K^O#SmFcJ2Kb#^#e7!ylNuioQ_xfQw&Ue&){kv((hN z>vL5j6(2G-y-917_;;yRe>3a6+izqubmaLLm7grAzm^&`@3|8HR(mI|42Mal)NG>` zEqr(WIqUoSoxujrRCirm_%6+VufU3bzK%I%-SHDew(4Jd$Ti`Da8i2i%E6MnX-P5LhF?VljOZe@q|xfG4Lb5B}rEj!ru zpZ#Fxm-^Jal;X+D^orFC^$az+6cqHGi&7IyGV}9X5=&AQG+eBV42%p6%#4f-%?vF~ mEhpRSN3$3im>Nx5` zolpKheQF=~@6YQW?Zz6<5fQ{F$VthKLO{?~f%S^0l`^(DLFzxS_Ue)CwpU;0Pw<_~`6 zrQhT8{slfcz3^iD&(nWO>TCb(oOvrIZqxg+xZiW?uZJa^dL$9pR~=GZvhl-|*7oc# zH#XXOU2J?}nL2-$go(AOeH7o9l^Yp~osMmK>a#!oyX>bwpMS_csZ@?%c+~ye(YS}D z=lwobXBz9DHjs?Gn8@rYuW7w&;%)PI?W2cz|4ytdDwMmV-ut2Vd6WA0r{U|88}`47 znYweqU!xP1(@)jMe|UQFNnup(uaky-R`ZtD5YFi6(*nGWXn&qUHVtj;*feA}D@?NRG=vd{2Oe_^e;;Pc`@ z34hM+u*-)J_N?vqx7Ob!{P;)3{Au+s4traKa9!;o(J$cWcNtnNRt`b*Xcvj!+2Uo0QYcp&Y^K0D#^xw*4g++LXelnULz_wRFzd84^qpTSYNt8Lmdi;VT9)@npd z-8ubp!2)BGn=cvfuU@xk&nC&Z+V>kvmp87jHP+pf?;U$tYR&e@cv)kfduqidud2^I z^LJe?D7*XjwqEHqq4jHTz5ZHyR?MyB`U(c=5XO^xUVhplxaa2be`eAaFSE8h%siku z&END*spY;Y>N%hN3fB3qzPx{ZhRi{s(=n617jWkUJ?)Z=6P#ehxkh6156A2IjXZx; z#G*P&%VNJ}%-cL=C(Eun>!AFeeRCqhj-NNEYFQQ1G_%%E;hF!*SHCN6)w8^=X~_3t zvs~a>_##^D>2_KDaH;%nKmId2&v97(>G?4+)7g)%oxNM#_p5h$bb~SD470bCJjM@*i(|7yC70&&(NT zFBDiTnrN2dA$E_!xb9Ma140Q$1v4YMD}YFMt1~%*K>krgQet9i3`T9+cqstzu zA4^DQKmYZYbWxMx&%^BhvW#cP9R55(JNm-mx-vbVpHJ4F+a47@!%i)~L85u%ywr`C z-)~;qYB4{tPH|dtLubP(7b&SHEK}z)ZRCA%IWTqJN}l@Nlben8zpy>ap3%y{8N~cD zbpQ4Sk0ygpoNUJHE_27Q%Y57v%`bSfx-Mw{vmGz@MX_AkJwu{aOL)ijfS4QF?AB$% zzS&~tapxx7*>fpPWc!nse1(ZmmV|8!WKC1eR&bfuQZ+wu_1yPwRXG$J^kpWb$FCJh z6HoXY^k~b+EpFY6^^>kyTsJ@UqfI;f{BN$}?aY_{_?@!n>tWXxR$D51>>c;Zl)r9$ z0SA4Y7A!qoZFwbN_oLlO8T*e+d!lH(G^e;(FqVhi-gUjq5nBm6L*qA1FD2$rn^2Jc z&^+UOJeQPr6wDbB)>hUs-nT z7b~9psS6HCICOKhcB)Eo&*QJzI-6G97Ma#=d% zTu!Xalsfm~^UjOMf4!A-d>mpuQ}n%#+SVnr!r1uIb>thLz9@6n`<85zpLD!Zh4UVx zTTSSDt$AS<&yZSZ*_`XD)!p=?*{ehz2{TbOq&OKgMKM)Q-H7QQSVVts{JTf$k;~H8 zo9bI8Fh5j$7_1cGx8vR;vG2c?U)J2dFv+sL>3+9w(fe0zyRBYX&P%*-o+(J~L5j;{ z8@J+XI+`i_~7TXSU8@-#ruxOrEX?N6y;-?sIf=_EOEzhBT%+g0~B3U~+PmsuA%UfuC<9p7Q$7yVnR5)@<$ zx>sJ+J^tVOmG%ae&v~2g>=&O=r~WzTb@kh_dWCx)Hb&FW*W^V89(?O|=h9CW#Xt?) zi)WU#eC1T@Y`NfKwqWw93%~Y?)hzD$^DCv#T4BvDrMtFx*8g+NB*q4r&` zOf%e{Fy_DQm;SE1!{E9lqvH43S%;Qv$hTcLMa1a9G;<~oW7Elt%EFKRU-Ktruh{38 z+_lOJx#sN-EfKlxH%F{obM5W$Km|?KvkM-$zKgmWD8eXp#(ag7zlK=mt@NUaxA>Kl zB7>$x_Uw4bz<<4Ur725EuzN$zwED@SY@3r-EvU56==zsxQ@QU*Yo(G(z@$a{Uis8S zzE=D#l;Iq~R3vz2NwI+0WP!aQ-L?+?dd|%b9fy7GfB2o#aE_g!QzpQAXDPi*VBymwt3e} z%(-C9^nKf;ZJe$eoRdt|#Y-dK$CPSsb6%*J*}c90bFoOllFE-N+e`z$DbKM?@7?xe z^OuFCJ8rM2c+L6v(XwNkBAi-d6|dQFS$*<9)SmR-BPONZ z+TENi5>UNhQqRtByjowb?b)$xX8*!AmCtz=XX>^3>UA0lW4xZMaq5=+pZa>$qb1GT zpPDRv#W&sMe(avag9~o=&ANBMexL5$OY2i}(_0_!4&0eOt$Mb9#)+#^ua5{_)`?SH zrh0So9iNf)WZwOsA30(&60ZszKbp>`Aii#O&c8owPh0xxz4yMlUA!_|W!~dc zKlZ*>vGVlOFO1GU{fD^gTAaXBox%SrLQue8I?Y3DN7vX1Xyer2-q z1c%eRRV8Y#uItQ*pKWkqs?D><0YA6y33vO-aVkyez()0|jl6%~csMXCWqD8gt+>-5 ze662a#JBY96Dtj>j8@HCG&QmQKt`{`{PJ79p^~kt%+oY^wB&lzgpTYkSrL&Y?DbB2 z+S#1YZ?(HUnx3ln)n@3Pmj8cmUzo&djyL=&LK~Xb`v^~ao#zjVD zFrSo_?4P}|;U&Y=vqvO!<*VlTCq?a&`nIY&Z14BAd5_f^y_FkhCv7_VKW0nW!5wE^ zB`#MOFXdbI%jDC{lC53VoF*RvIIpvpRq}_u^!%mNY^Sl|c2g0X;`!A3Rc`UG^nUsN z*>38RaiPJe$-pXL=c=w*Ul{Lew(WhjF5_+eubn>E56k?F>h`&k-|$9@`BK*RJ%R_i zB9d+xpO|DE+T*)7xpnOaj^wG!xX-A#SZw*qdRctNqI|Qe;2m@C`bypwzFZ{!*YNd5 z2K&8ncT(*2`hF>_EAT#G{-UGzxSpbx#=*PIS6-R!4)&^1Sh%(HRN_VnYlDw5%5|k$ zN2W|zSy}(3({9sNo_jv*jT>8)!E2zzqTDwTj;S`VB$#ee7M%>3a`yOqWam4qyn7sJCH|Y+(a|Djg zH)Gniw5v~AZoQr4;lqoqPTSYtiTV)6bAHz*lRJS+oo;#5H}}3xnY*ek*3M4y)5DXm z-1E4~3OGN>*T-Vo{0r&z zTp#yG_uY~UcfA$I_f<8|-qU9DuG4-}&n`?d(@D@e|MMDS<&t>DQkFS4Bh*U5AK&m} z{nvQ?ll_O7;`C*Oo43{1l-=W8G_g$2{Nz;UXU{bm*H38(*O>F}=bcw`g9Gp6UzdF3 zpRxPHp`!S@Z^7=Sw#_b7xZjVp9q*s(DwE8`=m$~_9-R&33)$#NE z-#aD06TG{vYaajQWX~rvwKnH%ys=|@#GiWQZ{Ax!d#rix75H9!)4gL0bnQ9ruTo&T z9UI!-9=s`A*KxW0|CIH02h}^-e_fihHT;yo_U(7)%-g-;&{FQ#C(@TJKiB1vJ~e5k zu4_?YgJRL=Xj_}tA0t>i_HE^V>zF*h{&S9K(vMsrnd*6uyK)8Rri1mE1gYd6Ezdv4^ntSwC-YvCRZ}s`sZ_C)YY7Waw-9PRRUH4zS z=)s(D!?RIwZ8%4q$thxVq0UlihEy_&a3bLl5Zc^-}*GC_}o=5 z>9g>Ao4}zT$CR4aS?`AJN)600*zU>^m>PFb<}ZEIorP>(Wdm`XVy!l z#T#d&9MnDFt5VMSH0jZzdWVgwqKj^xU(?ZDFSYCN+|^dCH@pr`J#L<9xZ=!w7p{j) zd^UO}ys0t@1(!c-B`nh3X#e@;W;YSRT4wE{<2s9`EVzHQZ^^OqGgF+uuehr%`n<*Y zhsY}thZwE5CC}V$>s|fvwV^Xu;=HNbOR1Xh6?0|R)Sa#Vdb;?Np-MYJ^Tv(ALRpC)Lv{V0du z`uQu2n1o)qR3>dO(JNgi5V!YAkPpl8sIK~5tMf1GC%WxZ@#g*#!OQ-*M(AzF2fUO+R$bdR9!|{{ue%JtYmZSGX?%A6x8h5x0#nf>91 z9?#A&)(3}8FgoBE?%`>{5c}_CkKXGxwQt)4o*t^)oVRe6!fw|UyS@KM|I4=*`yt6< zl%ui9;lQ2a%3D?xPkXoGNndc-i7g{qO%@_xaKlpHj=# z&6S+@mPz~}=R}D+?Y&L?nz6DcVtMv6{rR8ExFSSb_OH!aKeiQqY+-B8iXEsDeZS1R zae{T1p3h3o^HGm?dWSrW?-yP*DR^gNV12qq)GW&&9t^ck!q@S)Y^m#I$k`=Yu(_Hi7+h#dabA zZ5KF%&A!)`fAQY_a;|N~;}(%K2G>^qZol*JorHn$rspg9rpO;I+{bsvC8mDr@gGd) z6`!sKf3^$Xc1G=nZ^p8^XTIjSKX_&+JXo;gmrXlQU$@mMhjev8(YK;O$J&=J+aT?+ zLQeIgS?R>J^0#tNT$7UYtE*oiEE??7^{4J}xyIq7dE%B;ih`ra$6>zyL@2X=EQT$kCe@nEV#d(t#Xo{n0MmEw}E$j z-_Ch_YWX|mc$;s#WZTc(Epy&|)m5E+>DK8X;vrIXg?%FR4EYhJP2V)G-s19G(QNUZ z*UNYF|AHT9X0JgrBFcj2LfhMl>!w9!=etk*?)>-t`0~=Ith{jcix=aK=bxPI zyr@O_@TujR+1DplZLuhS6(M>mHF%rkoE9a$*`N3lJtgF79PUOQ+Fp`s@xZ|7ojv1hsU<~CkV^6tM>h7^|QWL>ea->{#x$Biyb<9 zyEiPm=(^HEkJ;E-?rDp$n|Pd52;&M>C59(6%viIc^_~1P_!S~cb{#og|HUIh`CGz* z>iMbj1ozi-&1@_wuDo$9iErZK7hyHF2X~ZAOW(`l{j%)P)8H#!I;D1xVm)F1#_VZcp_?)WO+n3c`9Fl(5VngtflP1p}tnw`?s|vjvK4s5| zmHd5s#DC69tnOf0>UvP*#`T+roc^%;wCehb{SdkBw&a@f-6>u(*Z*nIB-gVhsOrKC5znH`__$tF^xQQ*v#qS8yOre5At zH0#?>{9TZj_Q8JjgNk$g^X4>7JHoc>wYFL9@3(oT%cqEMTmAa{n=Px2gB-1nSj2i+ zq3JI+ zmUG)E|FVwRtuoCj$Fs38y-0&qd)69F$FuyR!D_A(>#g?Mf3060_>k;_K#!SQ z7AT}j=-o}MEnXe8Me6&O?M{X@Z^i9HBD#7tKX-8`zuLXEX66*17ssZRwf*!sqPS>6 z5@*-c)3-NkcP%vAw_>8VP|=rpDO%e;T?!An)NxsvZ&IXK2E!v}PnZ8s4qBQuU%Dsa zxKV4`ZR=&E012*TD#}^;ldZ?tN0VYGfQmLeaH0q>(|J1m$bVZbSE7R z{;hb;vj6hK!xzmp*Vz8+{Kdl7w9hTX@OxjPmeE^KS=VGnu#4r@uVS&c0M$ zD6k{r&D+0UCoJ#goBLRq|DomHU#ef$Tv#%ctH`7DK%`6DZsujj%6m-J4Xrir9xXU0e9+Icxp~hmuF@xNCszHnn;$4KqayN^cC@~l z_&Q$M=f6u7%4S@jb!V?uTYcEuM*%h*%4{!{=cudguV1OPAXez$9>*nTmvJr-j!QI} zG2x-gUtf3OjR^tm?Ot2f-?of3yz{o_?N9xyJKkT~>3l}#u7}AcrblOe11=WIG}uo$ zQ$6YO!|Gzaxw5|1x2x=%cI{i=ea`KXWsB^d^EwZH9QBpAox>@&*yjzaS!M1fyQ6bH zxy)VDGlO@{rNjU4)(0wFRCj*hb^HR0<{<;^$!Bl5J* zR^8bqGBfx{zU`U|p9<_|+;UaE;3OQESCMEv?OdPP+y+SInVg`o;Ac1En~{H~*Co5Gy*SGw<9Tz_i&+x*)Tf82T-y>-Q3UdPbCC(W)qTfUq9@JW2Q z`iuIvb*-DElO-Sf=FeB&BFffeJzx34?x5b3t+J6;)4drFe3b0@7SVM|fA^x+_{j-E zezvVkXT_f9P=1{&&X~p5a=}jc%AEgSTD=m#vF`fIG_{z?EvMbx@@V@VC4K$!r-4F} zwUzrFFPqKXoPD@ZD=*uXJ;>pGO=bV|{$=&^Ed{meb6yADne6>#P5GB&3m%B-Owyd` z?5A#i$kz70&sM3($O8{QaqW5;^FJ=e&!W%Z%Hi-S|1VCjD{bg_czC5mv;SfbGyX|y z=X@=lefY4#B@v&ibEnQx71b>jKJ|2iWRdPCw)V$95m$eOe-=91oql}#8(I0Ik91hi z$*wz)csW~q=9I(r=4O|wFDzN`v9VHJX5S2-LsOLJ1{^kI&4>~_(7x}@!=Swnn-sn4 zUtYZ;HNV2S!gg7X8q>pqBR}+3yvsS(PB@oT`6Tb}l`bbmQJ(`KrIqwaTm6!aKP2mxQg=tM9M3wvRD)Yg=D3 z#Xk5$s^=EYzmfjFx0I%r^m_Q-|Co3=XZ5Fix+}@9_9V^_ZuddczL$Wx_mnQ`^n0KMZY=K3+rzicYn#{ z*6x07_Hw5+ySnqX1^Lpo2WD>5sgIjzrem^wNxqs^+m4BMPajRPb8Ywk-KXTsySb!F zTHsmb@89?9e*TnWvF!YEK-3}Y^-jA*9{EqiE;1Yc()=?``FYQ|c~2M4y_@gjKIwPo zodx$Mi%#tI=6&|``1bAw?hmj3GcYjx|KA(yS@hU|=g+B{gqTSl6F*&Csq(-6Qo+GX z^}MG}-YF``>11)#>CHPWy{^|cbiOdL z2W|brq>B0(R5CdAg#y*hr61=^$u)e$y+>fZ^FyAmWf$GFxBSZD=n3z7cJQ3|(ctU9 zH=io>E_ePJ_{r&!YRk;K6Q!mcH5S|(qsHL!dHau$7@qGBpZ#>n{aE-tSHa?9?!yxu z+smZ&A25j*{H(t${CD1;rOFJg!TTHg=Ph)$NYd;pew#Ms*%n9N8h7DFIKB7U0S35v4?5RLdA#jQ)8_@tci`@s~Nw}v>-2? z>Eky2`qvvf^?y|E{WkS3d+?w5`lj#eZr_npD*xK>Rq@vEv};dSJDgf~&(!(Bn?UP1 z*O+bJ90}NUR%$CpDwkmHlg!t`_Qw{)C~r{zU2p&1W!0(f6~8BVUF>>b9>Uqmm-{46 z=vhR@y`0f>LGQBPKP4$^!ABw)W4Io>rT$2o!@hQ zZE(u(*qu;v>5wpUyyySdi@45IGg+zzPMK8BZD+XRYdZIQY(LY-P4%1>*7xq+5W|| z1QepVb{XwizvO246ov>kU$Fc4ArQ<%V`mgD=M?U%Zr3qcE<@1mKV7dL&TI5dUWwjS|mVUqHdcBgn zm@j#yMRM}ahWmW=8Ygo1t~uV&p3?<|~Q>l?DY?eG+w+r-5B^?*X`WJN~3~1K?^$0t-j-;J!96KTg#2RHoD(xy8Qdz49x=$ z+c#glvXq(Yy%pyisc!)j@yRdY;j+w)_2eC8S8#qcn6+# z3X;6u`S{x8dnxIQ#Fb}CxW~_Zl#?L1(qPHLtqn?hI4Uyq&Aj#IBzk^o6T81BB|ZIJ z$gGr)!E?&FHD?B~=LRf4oa4Gsd{KSCp>W5W=GnXwmE6x_G*1_+yM|3WpliCwocUy~ zn|ocEcS1u;HqiL9=}Sgd8HB{Q-WHJ-iTF0_4Zn8Up5 zJ8p?KhAd`S`fmPHwhb{{3Sy^h>dnte>p6;To!{v5_g&6B2IaXK55C(pi=JA2#4_*V zYPs9m-Y?T8noaL}pt#F-Igf+v{Cz=x^L01(p0_hxGr!8O|L?oTv%Fdt3c|XZ9i^u< z@m49WW2tT_sJGS5^!i#bZ-rpDhfYTfQ*gde*X!xqUR$bZ7RzqU;_v8>Y&prlsPFpB z2-!73XY_<(CI6iXRIn{N`ccSeulgR-zOP@`RThh0vhzOk<-g*>#I{zwJM8M3!LlE2 zJiNj0x%@73$}DLa51|(`x2{++^Y{IEKgH+lwljNr{hrgGZ_LxBX1-e$9{i@kzTIS= zNLObyKpO-xM - - + + - - -
+ + +

- + The LinnSequencer - + 32 Track MIDI @@ -27,27 +27,27 @@

- + The - LinnSequencer - is + LinnSequencer + is a state-of-the-art composition and - performance - tool - for + performance + tool + for the - professional - musician. - It + professional + musician. + It is

- + extremely powerful, yet @@ -66,7 +66,7 @@

- + ¢ Operation is @@ -81,18 +81,18 @@ RECORD, FAST - + FORWARD, REWIND, and - LOCATE - controls. + LOCATE + controls.

- + e Each of @@ -108,7 +108,7 @@ track may - + be assigned to @@ -135,7 +135,7 @@

- + ¢ Ultra-fast 3%” @@ -147,8 +147,8 @@ in seconds and - holds - over + holds + over 110,000 notes @@ -164,23 +164,23 @@

- + ¢ - One - or - all + One + or + all tracks may be - TRANSPOSED - at + TRANSPOSED + at the touch - of - a + of + a key. - + e Exclusive real-time @@ -190,7 +190,7 @@ editing FAST. - + * Exclusive REPEAT @@ -209,18 +209,18 @@

- rhythmic + rhythmic value.

- + ¢ TIMING - CORRECTION - works + CORRECTION + works during playback and @@ -233,11 +233,11 @@

- + ¢ Optional - SMPTE - time + SMPTE + time code synchronization. @@ -246,7 +246,7 @@

- © + © Optional remote control. @@ -278,9 +278,9 @@ then play your - MIDI - keyboard - in + MIDI + keyboard + in time to the @@ -294,14 +294,14 @@ sequence loops back - around - to - bar - 1, + around + to + bar + 1, - you’ - ll + you’ + ll hear what you @@ -316,7 +316,7 @@

- + corrected! (Timing correction @@ -343,8 +343,8 @@ track - — - existing + — + existing notes are not @@ -358,10 +358,10 @@ FAST FORWARD, - REWIND, - and - LOCATE - controls + REWIND, + and + LOCATE + controls may @@ -385,8 +385,8 @@ To overdub a - new - part, + new + part, select @@ -400,9 +400,9 @@ record, - the - first - track + the + first + track will play in @@ -412,12 +412,12 @@ you - MUTE - it, + MUTE + it, or SOLO - another - track). + another + track). In this way, @@ -431,8 +431,8 @@ be overdubbed! All - MIDI - effects + MIDI + effects are recorded @@ -440,8 +440,8 @@ including pitch bend, - modulation, - velocity, + modulation, + velocity, aftertouch, @@ -465,13 +465,13 @@ To erase a - wrong - note, + wrong + note, simply hold - ERASE - and - press + ERASE + and + press the @@ -480,16 +480,16 @@ be erased just - before - it + before + it plays in the sequence— - when - played + when + played back, it will @@ -512,19 +512,19 @@ using the SINGLE - STEP - func- + STEP + func- tion. To - overdub - notes - at + overdub + notes + at specific points - within - a + within + a sequence,

@@ -544,10 +544,10 @@ use LOCATE, FAST - FORWARD, - or - REWIND - to + FORWARD, + or + REWIND + to find @@ -564,8 +564,8 @@

The - INSERT/COPY - function + INSERT/COPY + function allows you to @@ -580,8 +580,8 @@ another—in the same - sequence - or + sequence + or a @@ -603,15 +603,15 @@ between the second - chorus - and - the + chorus + and + the bridge. DELETE - BARS - operates + BARS + operates the same way @@ -619,8 +619,8 @@ remove - unwanted - sections, + unwanted + sections,

@@ -636,12 +636,12 @@

One - way - to + way + to create a - song - is + song + is to record each @@ -657,8 +657,8 @@ 999 bars). Another - way - is + way + is to record @@ -667,8 +667,8 @@ basic section (verse, - chorus, - etc.) + chorus, + etc.) in individual @@ -678,23 +678,23 @@ use the CREATE - SONG - function + SONG + function to - “chain” + “chain” them together. CREATE - SONG - will + SONG + will then automatically - copy - all + copy + all the parts into @@ -714,8 +714,8 @@ few bars to - repeat - infinitely, + repeat + infinitely, for a fadeout. @@ -757,8 +757,8 @@ the - LinnSequencer - is + LinnSequencer + is designed to let @@ -767,8 +767,8 @@ record - and - edit + and + edit while devoting your @@ -792,7 +792,7 @@

- + * Simple, easy @@ -806,8 +806,8 @@ clearly guides you - through - all + through + all operations. If needed, @@ -818,8 +818,8 @@

- HELP - button + HELP + button displays additional explanations. @@ -828,7 +828,7 @@

- + * Non-destructive recording—existing @@ -839,14 +839,14 @@ while recording. - + ¢ Two FOOTSWITCH - INPUTS - may - be - assigned + INPUTS + may + be + assigned to remotely control @@ -873,12 +873,12 @@

- - ¢ - Iwo + + ¢ + Iwo TRIGGER - OUTPUTS - may + OUTPUTS + may be programmed to @@ -887,7 +887,7 @@ at any selected - note + note value.

@@ -904,22 +904,22 @@ or Linn 9000 - sync - tone. + sync + tone.

- - © - Utilizes - ultra + + © + Utilizes + ultra high-speed, 8 MHz - 80186 - 16 + 80186 + 16 bit computer internally @@ -927,17 +927,17 @@ FAST operation. - + * - TEMPO - may - be - specified + TEMPO + may + be + specified in BEATS-PER-MINUTE or - FRAMES-PER-BEAT - at + FRAMES-PER-BEAT + at 24, 25, or @@ -959,19 +959,19 @@

- - ¢ - TEMPO - may - be + + ¢ + TEMPO + may + be entered numerically, - adjustable - in + adjustable + in tenths of - a - Beat-Per-Minute + a + Beat-Per-Minute increments, or by @@ -987,36 +987,36 @@ on the TAP - TEMPO - button. + TEMPO + button.

- + ¢ TEMPO - CHANGES - may + CHANGES + may be - programmed - into + programmed + into a sequence, with - smooth - transitions + smooth + transitions if desired. - + ¢ Any TIME - SIGNATURE - may - be + SIGNATURE + may + be used, and may diff --git a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 2a214e8cafb950d4b622494c24a8885436197f1d..13d33a38a99252af3e8e00cd3186c174fc1d3e6a 100644 GIT binary patch delta 7637 zcmZn)=nmMhfrH81Z1O`!rTR0mr`M@i=H2}s{)9iIXUPR+774ah@7G3Eq%SCI3pYzM z@BVlx-hNV$2*0Ll%9FQphcjaCyjs=Me&qT?#g%`4D@T1dc)EYe{Qdu5=l9qDfBQXt z|J(lf|DQL!UH|5Adr{($l z+xsuxx8CtSzd!%lud4U?=W;GyXTPxe=C7=iIb}9k`rB`OESkRUUzqzHAG;$;ubLOV z<^LzUt}Cs0o|y3VgvyWqHvgG&wBrAtv)8ZaW=*euQd_L{X>y!S`5ndIXJsCX$-R5< zEA(sE{J(quW^JzfGiyoJly$eQL$;6oXHb#q6eSpL*MpAWBmKmB*jyTqRJ_UbWT(*Lii z{%5;yOMdYXMs`$1~=)U>O=f%dy-X7XFbKjc7dB5Ks-E?c;!^^q1 zm~3rxu5C))s{ZZmT5hRRA33*Q$`bpvxclLU=G(jfy|I@M`j>KD@Sd>TG5N2vm-fqj zIyX72b>l)``=^s`3GM2+^{MP$$gMNmY-%nhv@d1fxqQC=T$%C(_iLN))bl?)wkzq^ z+3F3y)|fofh$`@J?PmU5y=L0qwO)C3dw*pe|NO0!=kMnKdfYxQQud^t{dikZbm1vw z&sLAWYVVxywS3`piJE^|?`zPKl!lvmp4nPAPnUmS%lvw4jbyv_qiJfl^i+TU44Cz( zPxp(rv&b@ct<(D1$J~P}^E(_$!j1Ck6MOmAT=exzOYO}x>sXlJs5#|rme8qP53l*R z79Gnf)wK4THTCne7O~5}JNA~=No0KV6=dqapceJ>{a>@Zjo0Vp+PW_MpVIlnGx)*n z?(6q0+eFjf%n@!g{BzMg{Or3a-``(K_?B_{arePD(PaW3bD5Rv+G~Hlb5S_)MB{PW z!&mjoKTQ=8n0BwyLP-9gvtL{6m0WfCwCP-T>*_x|`>fZxl+};x?2K0{EZnNMZ0oI+ z6gll@=*y8@+^0U_(DA5y3MM?|JRhbU{Bb$fA!v^B!uF(S#u8)49-Adr#$TKqr$?LL zczAks+nSiDn*mi42o$FeMkz0G@0@pDE^uQ%u1QS)=h{ohmHasD`Rb;Cl@ zg;pJ68~3Nx?+AD+^s#@{t!=v|R^?W5muA){dHgk7z1-b?&RkuE?=$Z0jg4)5)&F(( zWIdUJx0fDTT0fX|w>y{RU(Mv>=lA(+OG`5Yqnm@(GRjN zeU>;n_xmTMr?$7wi@0%B7(`ml*1pmEZAtwp060;{uP!W3>RfO6sA`(tvOByFmI;NFLRi%K>gN}vDq zYqhc~$3+hD3#>LG)4!!ZJ#3?1c=T!guh@^w`8N)?tP7Tqnl#0>W}~XqhUBI*=-E{@hUal4UEp#(H)G5G?BlCn1WYIhd==5_>e{}ja<>Y{iHnWX8bWluXnecWLcZy(`s^vh z1tm`&f25ge`=qYx;Nstzzx`agCr^@_F7hLPsbTqq`k-dxMp4FyZzp!92o$W)eYsq+ zC#EX%%r>UC6W*R$!KENzcTy*;DMFi@_fpiqOZ#dYdC!KaKNdSyHfgHZ`Tfis-%f9M z&Jlhl(SKc+>24?M2Nyb|C%epBUJ!M`ox}XXIVGkvNu6mNESHXHe0j!IbtCENhl{g3 zwzwR#6P_{6bLPByskgS4!ZmkpY+c|m>FShe((l#w928n&e7*Pv=cd4bc@1(EYosF& zbIjZq#Zbazw@8CRq|q>b>#tzqsa@F z@gHi4h)%lKsVTTUvFJur<@{+ktj?`V-Q?-5!FSH{ot9%#iPs+m@5uU3*9>0&ypfkE z7bBxFZE0CWg+x^bznQic$Bwg|ju$rRI%+P;P~B2+|N17~jn28xYHtedP3mTEoz1iB zV9Dw&U7r(df}S+pd!U`9S#Z=t*t4;>aEJ1)t9rEy3xX%+YOOrd8qkrr)Yodkv>$HK zizd$$^^PvG;$Wp*ZUUMo(yFvJ)V#dzx&X`5qw_%`+y?o)cWqyP=x- zj=BOD_pb6yPi}2G=dbYk(y2FG4YwP8-P43_o+>;%|NQKY0yB?VPYU^d+=%I0@T(p>Bve?UhRHTn!oj;f^l#9759nv!WR}c?hZ@5Bl>@R?{Q<-g)_{KhWEF` z?6A^XBkSt!#<)l&_DLFl-OO)J$u2Rne3++FlG_~+ zSFLrri6{F-0jJaPe<6RQtkRy%*t{yFv&_|cO+h{L6uz!qJ6spN_$oT*pIGBf9yi7_ z44iodSLU{6yg9IHQi;Eet9{ryzi#FKJ0B&iTd~GMHuysK{Gj~`4|sbXFJOJxbz=W! z!BoM0$Kvm_hdn)g^-&|crMBhTn*Qf4=jC57m}aqj;#}43w@bAy&wBOydQ{)RV@nnb z`mRd!5=fnC&HZNK^iHkU>%xy6c)~IJ%8S#bou?$y-)}w7u=%L#!MJ7h6IZUO zEtxUzdJsEPb^6PhHD3ymVtqwhUt9sh(oc&LwY^9l^(t0KdB)#>H zIC!Oy|FX{uhX(Z&4wJ4657Mjy?QD`aGWw)leZrubqVjkv_gQJ3`VDJ3yO%OA+xKuo z;^ITBYi(R)yz71{C8qm5Il#aFoqzO}q>zutrzO^AAAR_Z>DY>A`mb6pKR7BE!>@C5 z$!@inmenPF*}FgNh`Jc7CBw4s`!1)5#3tEB-KdAQ zY0nqky2PItSGQ^1Lcz<{n@c<2O?1gJS=uZ)qiLz|p+y%oyu?=Io|}8e_`CM%r80ky zE2XsvGBiz%t4T6Edn{)6(J2CR&u|G_qtyZ&JPF zniV0j3LV8BA&XWVR@v;xHJ63a&EbwhP@W5$_3r0qqMr5j1*|K}-`;$5qG{jh=@Az# ztK~W~(rPxOOj`KyYnHU5`ky=d!|y+~o*eCM$E2}`<3gD6yP2=oL{2k#v22xZo=wK0 z^z*9u+0z$KtNHd|;h#wfO}{ohxtSqsB3wC5sa}M!*JM`XCDZ3eW|qHwe@1^#)gC{#=Vy245#U~Qi*B;^6pdG4!vAGV|Ge~UGDD{4<2Xqe(JG(d1<+M>id(f zT}*E$eaUW@6kS)$cgrxPKK{|E2bWa3Y6B#*rIJJWwb-^5T@3jc=b1l6`m@X{kI79j zhvdS=if`Sm@2F0nx+X^Zr*J^Xia%@Q1(>=^izFVGDKAK$lj*IL+H+=R{5B?~%%=kE zdp#Dt>Jq%V?+xpn<1^S69Sg5**m1mz&Lr#q%P;86V8>4$Ed<(QCe>CDWs z<<-Q6g8i$K=9}-G@*=x`P0IBqR!P%c7@DdZY9;1Ibn>~ z3g&p;nz?nw%@3z16rJoaoK!#O>Iws$fZzW$51NVWSs=c~{n9u0!<)TTmdU@eRDE&G z-0)yA!>8D7+NY*1zxZx_`yGY&GlhrCt5#(6>?zT;I{v9b{CSw}ZdavAcjF987~Uzn)w9}q z?l9@y(v`6*WYZJJpPr_7W!CVPzVUTzxy_pT!Rqq#JEvTk>#k<<&1lIAUAEzX$~%ce z$0gh^*U2Y6*7$kYuWdywZ&%!aQyz$|BCjGlw&*)?7Su@JVJ_p1*)_c=*wli_crP%Zqdr zDHQ#7WArY#`ud6>+XJ!UUMo4@T($Fa8iU(T#rm!*l=862yewrMbzx2*-^>cpyPPMc zu1%i5#J;|1TSv^|iGr&xhD?z?w47mqbT$^>Lxvu(%?VW9@tto!h z;jL5mrX*TixVL75xyRe;#8)4l)Gl;=Wt!i~e4=V`xWbn(*Y9yYf0(2_ggV%F)r4|5 z6w7-yREb$;+r%la2`#LjzGkD16U#A0zk`1~!;grrpZH$g>UY3KtKVu&hg2soeDN$y zEPrEa#YdeF4>M;*39O4>BgC-VdTO8jlf!*GldL;a3nrOrZ(gPO$$zE2!uRP1=Cbyl z{P{=7BSz1%BCY>E^CF8^cC!-7-@a>5%Dg=J@5&xIW*=UTNhS8z+}#|hCXGxdep&-q-;?)W7;d#h2-Rskm6!@65-5@la2 zhIJbYI@t>+y54m%S|#=M?(!nGovImkl#MrEUwzEuMdB`Bcd-X+pZs7`{n>hH+N)RL zZ~iqblv|zGyjoSnr)9xA-*&;nB`4&p>!li7FD-s{!#G^0Deu`ayQiFX_g~1D@J!iv z!+ysXF4N70yRR>w@O05~p5#BfpJwU*YVnz*YGSybvE>c-?X->jpLfxzM#52S#uetU43V+HD$xH zi{%xJ^%~KSr@p=V=|*R#^NtM2cWC?3b__rFfXDj^Rk>Q*t}|I)OdPEL@_#q;_%h_6QmC9~`D@hh(i@ES}=)Zufydf6IGWPqA)|&n;$ZvcG2Xhj^3Ex^CF9?1)s0tlzrw6t9;rR`K=(N4+`lu`e;rqI*(S+0%D1>JJu~ z@7erq+x2}rSF~N^xpd@^!TAc=DQ?|crp9cYcj@xOJyRoA{ZGpCW2$a4`BJrbYw04> z2*0(DOw*gRLri3Elue9&s~Ug0P%A^Gei7Tw!;1{>_$$|+5!)~K(|Vrsvjgoqt9ice zIJT|#wNvK@%hQ`<3N~opz1~ydZaiuJos(MvodRd-?h&+~b4+#h-;Iwm*zekG)ZhO1 zo`GLn_MW`&k;RIeJT}<0@$W8wXC5c}dhc$DfSr;$v+dJN{zmz7ER&V|8BobuS_y=^Zc*m_&6 zV(q!b$3KR>&wctl?zMbs`a`95(zWew= zKVQRc-`_v)?Cd_08@1KEVA(fWul4|wtUuhFKJwnk=;qk?c9E>LQ}HeJ-v7)WEq9+h z;LSd@BiLWSw0`Q;#LN3P1=}2%o#gIWWZ$#quZeCzx_a6B-WU!hH<9zpx9mE>P-CZ8 zYSName3`}aANA_>ycWAZ=DmNl%KuAKbIF`dkGC`bS;#z3N@Sab;UZbpz|CJPMPQDL@Z@it?%`jW&g+S=K}Y`Z@b(W=ZMy~+@7QQCXwkO z-&bzl-S3kQZ8p9taO|+;+{B>rz(WgnB`|CbKR&%@Rojam3(1`)^^M#&NLwblFaOZ@ zk9C&9r&*Sf%MR+Mug$IfB%^To^rB0vCo}K)YwCQlB!9L-&XrS8=_mamTnSuBA1@HG1yKp2gm2rjiSKck|!3^vlZI zc4}oYzu_0br^~Z=Cw=3-kYxB!tm-bO<^EuhG5<&N=ld>WHM4!Vy4HPKkm~j0Og*XU$Mo+jCM|x^|I_)to%oGU z>n{H(6zdY&=%lv%ehknvVuAiaX=-kWsf=|JQ}Y6s<$%nr-?Ch+M7v$Q^^|4WR((NUE zYP(*vTedS=*9QwQoZvq8>e{baGoI?SXSqh#zMuEx+mTM2&xx_Oq*Kx|-p!eK>^UE| z%F>$SS|6joeDC>^-*u|s#gj*`MOJpYZE60<`TqOzgYnBtB%hky`#ssTsfaaD)46x2 zVO!9lmBqjBT(9R|7b0WFd$z@Ii*aYZt>x+?$=xYe>KZr6Uw!%~uFdH89m5W_kg3)N zuPlD{ay_|m-A;OMS3%?L$@*?$A1k=7)Tpf3P}KN>W8#I3ubaanoU?mBoj!W)eK4=q zm#+?$R_!|bo_a8P?>#EsJL%R;mBY)!)@bcHpnadSk0mDU(7Qaw`loExER5Cl&dFjC z=lhLI8ji#XXfFP$+7(yXq<-*(#o~@BC6gyb{Ay#%n49bDcFHoMJJ;{u1HMLuTjw92 zYzbRZ)sP-N?V04qj-t&LlM)>ztTnbykhEq zsmqw|@Q=Ua8~Lj=%pywu{nY-&GjuP$wV%Z@_qFrZ>)+%V7#RNl-y7>aZH@uYo#!Gu z9e&+xSQy!UhO?;c&VLTc-MXv47`C192{Z0)W81#ge`Uy~pZE3cF0#!2$d>%|O00<7 zF5!-uGb{X*zo>8?TiX@(VUrCf%eH$n-kdx=`&WHeoW&9r_H#Y2dY46Tzv`USvh%ZV zTNbk{`|p2#k-`Rk%Uadm-#a*E&itZklad2nx$3s3Tw`vnGTqrJa$x6@sax)|T`m_6 zPMj>0U4P3~Xh}@O}a4le0@*vhWA@nIH))YzT2bN^XA&RwFQ53OisW4mOAInERiqMW_~|7?IG{v181)- z&1jA|8p%4*t0KkfUPH;x1qw2;T-!A7+bblu9ErTN$>L4?kFd=g36-}BT$~&N!>abh zaWhv(FFdnd+dg~E|6~`|MMd=mx6ZDK%~^OQR){S^{fw2TsbB05--XS_jvxJcT|ak~ zPq4c!F)drbyx`LBwd=b+lrzpV^bDECTwbg(vtqLQ^n(tZTlwbYcOJ04+5YC3{t|<; zuVSWzR%#{rZ_mHGGS^ntKQnS_@6v5{X&g;rM+3Ako<6uRowKb^;n|{{Om#Dl*O&M| zQr!34{Dg8`;-wdX`}ck~dA0JVW<>t8Sqd8xmZoeezjlo^(|+}z6@AY>E`FZNou-32?}LkIV4 ze=3+Z^VtuEQ>{@CdfKxBdqg(nzKsa;z8zyyQNHgd%ZojS&pSUmtr&mH=+@2|)rJmX zw#r4#PIK#1O>GSgEiQ|>)*Y?bVR5+Emx(1d`d=&8Q`g#=!UgJVmQgX$`oTXKO6OSf z{QDsH_eA3Dqf39Tbe{A4cA4wfvM;a9uW#zCoj&1uxa8p(u6BkVD=SQR74tpL>ge_= zi~RItymfryv~Sv8vU|8?KK(a|3%K&te@3N_pyqGcQ{RQe>i;~8(Mp|W#an-+;{BBG zM{dU6P*eg~GQ9CjpgSwLY2SQb zjb1axO>5=&yPk5^ilsW9+ioX3Y0I+f4u8BvCTMSF+wnT^lAi5{vy*!luitQSMb+}j zn=ZPu?3YW7cUxXHS#x1V{U2GQTi(mP4h5CydH*^2x!^~2^@nTQ->XHFFmnx+@a-DX8Heuf6oTQ(>nf^-ryzCd9m=U18qZXO)ic z?rf74l{AyKU38$j;n|koNe?*17e&1}J4fQe^6ATT7#qAMPk1R=byV#RTgZm`gTlvK zb31d7&04P3nltsFH>=vOkEcF7Wju2-F7o28O}h-IY@L2&_EdR=E8cuNj>fp#Z8LBF z_2mhRr%L0(9iONCOsiu#|0)lRy;4xhCpOg~hSlrHo$?8XdOKxLoc1`p z>@T=Ti2wJky1_Lva+5yYk6T@;Kl%UkZ;#*q|MU9w{eM3`Uzh*) z_4WSvZU3}?7hThP{mn~pPu<_IU;XQUCf3cEZoEFezB2N;`1fC5?bp}W|J`h8STC_T z@2CHt>DTM`)ct*}y(2aLvu)kq>!+hOOPsgomH(Mu{qJ+>pU!_NQ%eHu3x0{@(s1m%G07`_HGT%-^+iq^C<<|17)Zw$QDrZ(A(h|NAG~llgHA zw_W~2t$zLQ-t**SPkl2Bym7Qz@MzZU{7;km;-8cskzV@ef=J(6A78h3pC{bjCHH^M zkJ#Jwo7;VY>lds&bNj8I*oAKiy59|R?Q5zVFR%pv`z^kAOFYkhpJ|x~re`)SdVXum zk#Bcw7hSxfestSs?Tmx!cUZS-o|wDl=2ht{p>^Bs!_fA4dr zR*7F;;4+bQX8o%DDciPBhuyzJIet z$g(dX()p8am15{c^}T;r-aS-%LN{ag_7&yT`P18W7QW59{LACp2jL!@uiAWTba$KW zt>T<{j6I|2_Ou=6!Xv&;3`>8KUjKWO)CU21-;%QtHxI~2%(&`hX1u{=yZI)jOLI78 zCC&aE{WXhUxA0gQU-s=ARa3$)2}P9Yb+D!V%+Ih}q55T3{kG~IehU&G|IS%$7B(p; zWBs!edrokuT*;Xol=n?p-uL7z_OJG?FaKtB|NU>_yv@H? zCS`XocQ8IGX4vF#X46adD51`b9g_q0^!_Spk;^$eWOWFq(Jm}hV{Atlnl@IlHH`g4$mbFIsQ-5>*^_|RVAGd1k z`je;ByJ6eWtIM1A^N_W_0~FCT_=dcTOOG=sZS|v`SI<)c|(F|K(B;d{af#oa)serHoyL& z6K(smGpCzx&*z!5*6dkswnm(PlD1h{oTUAGXO(9fF$bc{_AuQ0l6R#*I(_%TH{puc zSgZw({4Saj)^^!_+LHtB%^4NK77`L(#@}TG|5jz$e10C*+z{WnJ-*nkV(!CVr`Tl7 zeXFMKTgj8Dcyi%)(+Sg4xP9skS0sGwOVW0jrSyIelLC)(Zg8&tszdwCp1Uv5`ugvU zuG0UUV6IX&o^`EfIZMyQeY?MT|IE$jBGg}}o4&jknV$M;PmI!Z7Ec594fd=ZS93Ri zS3D%6v9P+})|C7d1&u@}Z>>XplXI_MsbO_aUVO`P(Jmd&fJKj7V*5^foBYjkQ~mz4 zaYvNXxLM>bHhbu8d}^clYX0@ssor5*j!*sjDlOzqqE^#_nTH-6(3|-t{tWBU3r6#| z_%?YQ%ih?{x5SL|Yi;_3Ypfqu@d=Bboi+d1j}FImT`kPB9o{p2oSHm~L6(VO?z|1R z>Pxnt-??idk9YnP^S2y_DwWk{E@o=Dlr@=Gw0;M_6MKU|gVKqO<_{f?^*E%w}CT^#4Zsryy%1y}t<eKId zuJ&Fz?bYDBdf(~!zUAdUnsN3qeO@<}Ieag?l-V!*BF68{!60Yl3lkRERzPwF;bv6zu{z&E*R-BSMgU^8ZkEsykSWv)zn zdTN#4k`lhX7Z+i_Qs8WgueOj&iY<_WB;sWt@m%(Z#p&S ztdiP-Gq+3*Z<=s}o9za70^fe!e~*?=F7qi}cx2LK@rAc5O5z^G7S*}(89E0%c8luY zV^drlr_{G7hemhAf=yvr-{^G5%KnL+vXl5f||2xD$<+aq;<%?7V8eDza9 zb9uspUV0iaB^rKPB$#QI>iVp*z|HC3+sqcn32WDGRfbZ_GSZB-n&_rdr2jt8u(-)b#b^-D;bpItF*%Ss7#olkeG z)4xauT}-IREiTELk;=U4*Or3|zsov)T2w3h*mqLOfsEYUx%)L*6124SN|~C^s@}e+ zmpD~l>|OnFuUxgBNn!dbUKU}qe#o+Lr!2o!ZWa@GB`CI$rR@6j@~e@ayYpL>`x?Kuoqu578B0!$2~(y%y0^e7K)Y6>>(Tc|&dqXNYAuXRUe);44Ieurhf)+evA zPvGQOz9qA>mWMx#?Vxnd*(sgctxPv;gE*&(ybZX1UyWNfcaur$O0N|Drfj|`@e_nD ztv^@pUUoG5`l;skIlQj}XYFpVyJn@iCApb%D!YfH%&PeYIhhl#-doS1Azoh`RJof$ z%reFD%$lqj_qHwl%zb~u9mUNo|E@HfrS%^E=DU6CjCwWAe9Hy5m!E$x$+_S8V||X| zp#YKhOW3MqLp?Sg_g-(`+9P=>XTi*}Ooxy;@AN`+n^?1KUsg|uxc1xl#EiSgw|%wF zZVcdL)?^BGee?E#_`8!EC3DVMC#a@Y*B|)hTsHIKUc+)TowBX}oO@li?XFlcd3(4v z#|=HkUyVy03!aGxeqSxd!|}sP!q5Aw>Wb%Y&uN|SyIaV)#=^SnYEa+dEJM@qAl($N z>RqyvKYTAMd7YuZ{C+9ZgP+TiB^h}+Ioa*=&gk`=GCL*})UUz)?X}BCf9t~w76*H| zFJ-S^6jsQYF!f!|3)UH?VTp#SJN{iS{~O{sVdh!O{V$6Sp03k6zxMtO5u+meNee9) z9)_O{2~g3U%%;h=uI8S1t3#=ca;k@;G;%7T502vPP!EPYf*39I`a!NTkF|)-gc*j@Kw6L@IHT2-m7`R zR&(KX?o4*(hg#0nCIv-pZ;_SB`x5y`BiG@5I+tLTc!;HocuMP)I~yi$J2mmwS?9-3 z=kE~AyAaFj7`s7r!8fkdmqz;%`6^45N=_~|IK_M9&i}@hnLmg1jW<~ZK<6u)4h}J zRe8|7TGpqg*Jo|GZl>0GxH9&}$oN4!0bbeTnH-Y83@NyCHw0MC7471tR3e9%9yvjgnuz-7sD!c zTQTPI%zwkW)>|!}Kl9L?sSk^l*7wLgdD1JQWny)Crx4HCH5@C~w}mfxHvjv^un8w- zl$-bT`lU}(lQq!Uo3miq=Yk`=EB+N{ZV6*{7w9Y%C_J!hMU-*x{7Yh!o8O)8m?PVG zrvB6y%?6WOye=zqKc2d@W7hSld*5zcJm;sCkuLMxWZ7SxD<{6Mo9oJvZ!+^pGQW^w z3R?+>j&^L*`_|g?P5QM&BC{m<#2Sf}Gj}$Y2*}Q_=PB}2SoO$1i0S5|Rfdfx|L>V8 zX`58CRfu`Fq1W1Vr=}(|T-bYf(;|blYV*pzF8sg$m9Ah(`rJiv=e(9(oxaqqGPf}% z@cyF$w^fQ~E+3k6V9)N5Kb}XbnztwkrU$Thhv;nf==FB~{UW$%kAp5mJRvuc6M zJ{w+s$-mjhD7Ck~?w$|FrunBPdhR~@d5WyWxl>Y0leP!#H9K|X*e%nZg58s>=kf2! z$eQoJBH+!VTgpd0{mXoc4;qVy7m2NoJ#ybOO@4XUg?ynM(m8u4%S11q_i)N%)<>Td z_r1?xBnSVGP zd3xXqckrhRcWS0b6#A!lZ%|t^xgT+_)>ORV>65HJssPL>NqWkHmBkMX8?PQxi z)w`uhxmodF{BdFTV!w$ev*wkQng&kST)Fh{-l=N$Yv=D_Rg!IdztY<>za!%+=ch2) z#W_9U(#g9YAKJ1(k-wnoU+Q#4Lof4J6L>Z2ti_az)cJ$_{Yvv}q_Q?1Dy(>VI5VR4 zoY1Vw8@uaN(?s%$yZ3P4sBk-bEGt9o)ite+g&XQu$3=2oySQonrYN&1y+JWyy%BeN z_DCgPX?6Jd@UYQ?CAIH#9^WdN_&RxGhbZtNtBbm1OW5(>a zGaWZ3J#diwUO#>79(~_Mcb91%%gFZ z`SV#(x>dQkb4!1Jo#`hoYHBCzvvp;2TLwo{p27;@g%&1(ew8Z!!m^p6-XlH<*Hbbc-^;#c#^?6t?_w^3U}>J@yhR~87YVYPp` zyu_d&%x>Qf3B~S^ZuLf%D8B8VSKg6%uyW?*Fy#{lzbqDZ|2g#jxMqSaYhR7uL3@ii zwyFo2CrrG4*-mWoyACz=mik})i;sS531XKjvw5v>)+)PAR{Z{S-jschFMPi%`A?ze zV6WD@S2*{Y|qn z4vnq1LzahGw61G9sn)-H)AZodFXsizEFB~zd$gzKls;9^UGK4D-ioJd?)0*(PTDOH zzxrBFFuV4R2LD?(+ALcHUn(rtzI=IhTCtT!B=1Cs&s)J|{6^j_oZ;2HLmQ*dPF%GncCq%9U&|$U z4bB`scxmmV^HwMG*rjLhnUnOPUOVUg`&nP^ta(>p-7{-p$)(0O+xwJPI$V5oWyve8 zInT=(1Y)Mwr>scedTe&yBz|*Y?CQ5`bY`>F7n)5enttt@`#h0EwVh{E%6_jC`=Huo zye;&aoYkqe)GAleWB#5IPXs^I>}Yp!dj3H-Bf`{U?$(z#k85m9T{g{jThENC`gW>Pry@4+sM|G8F+XMZgy(OmiGkWa>? z+mrPZA327ejtIVBcblV6ctZWPqe|>)Ib|~qR$tJ)Ir-utb61-swJmJNg{o_QZs;=S zS$uYfaBSku<#!V@1hX2r9?zOHm-Dyy)Q;W8QVR74IQ27HANz$$%t*BUAlEj(ez)+e z?R-}X&Tettm@cIA{F$=CpSS)u9!S33eND|HO5K(t_QL#mV!TB=dV?hLPxx3EMgCDa zwlscXGRLEm2OGaXU3G1v*R>xLxL5r(`g%dFZK?R*y|b;7w-gAv_7zw-b)OR z$$b0w*Pci@jtYzZ%E_muD8EwQD`OM)!PVi`i&Y%o=HLCY{yE39KK|XM&$A9Jx#k}y zVG`V^A5zc0^SsxR8=T8ls-0t9!1>uSCp^7=PMxlYmc`A-57W>3t@hH{Z?@(}`%I1A zDe8|sZiL#}o5=6XZIoux-BH9Fx;HBB^Mx`NVeMEZhjT|~|Ljnx$zJHKRBbu$YvaMx zlP6-nU#R7lcize8^CwsNuk3v5ELnH=U;fdWe9l3trA1Hc*9w&+XclaEZ&lyG((pcS z1(!+d|BN$Rc8b3$2{%a3IxYIoGJ5})EAmSh&DgQQb7PoPVq%UGyVX+fBl43q=eW#} zkq)X^)YP%?Ve3^-tM?^-hdiy83Us!=+``6JV;5Vu^Q8E_OIPIHeZRBRs`01yyh!^B z28)R{mcD7ejdgcEdej$uP2Adk`TCvfn@%UQ@0k4HGQ%pv|BrM;k8S#YKYgW5YjZ`! z8|kDs=gxi$HLkm&6Y-nXE~0eO8<$tkKd0SLS6`!Y)y8e_+O5q7Tk39H>xkOfhCndac^%=XyUIQ|E7*|D0q!8AD5IUv+n!JR>~i<@En|BD@W2e_vU0|L>~Qqc=HM z#?R+lvElzMl{(gk??NOk)9#kYMm^uEx>!5EVnf)Ey&vyR7MLEt#3w$&yzcHbeNKzl zg}3;xKUd{$%e?&afzPs?GkL%cYM^d_3iJ!0EfHft@pM@DPOdd zjcu9qJo@;YIW^D2WNH)VJxhOMpzAV&GvmML<-98{u^a7RyScnG@mHnO`5*o%5~m-yR9^be#K7?X|JGRVqNN5r zcbpWnG)!mFT70jW|#vD>3IgtY%T^!|wDUUsfp3DZ9Y z%WEvey}VgEczJK8>^(L^^ND7Ze;vzZy~t=kt55$k<`f0hM}<$$u$k35uU;VY%=vxM zO>e(0SUh>bqi6$t%VZZg-BX3_N8b%ocHM&k@}KF|~+C4u9qu zR6l;^BUdq1Tzkg-o+)hN8-thqTfH%%p0_z-Lc+P(AN}6=I&#MeV34O>ngdX)2F&-wrJ-b`lkp7*AI zgL(bW#^w`i?WSl|ERwEe^shMnZ?lj{jrmFLFDG7*KV4|& z&U>M!+PbzX#4aiC=W979E^GVZnGUKw{Z|Z5_;2N|-?heKdG5vEv02V*-HyB4++r!Z z^l?>dkb%Tp-&yTP54V1?mSJ5tfo1J=vstSqhWoL4$(EEQeMrACHMPF1NrJf{Q+~pK zvn9fFd~bK^=KYzrSmEsErLv!DocbaiR@Em>x2ikfyz63$*R@R+e?(@6|LV%P$p0nn zUsYDg-2JZWk4&jgb$Z{|9(`=F@;Axi=+E)zGM{#cnSK8DTtHo`_+WgL+XqYUG*yuk z4SziNH>9P_5u9P1VXPu{eY?Wcb;ljEuUOhNit;Y;*!rpPlM&Y${zcm@Eld}!bGxOM zxo%zGr~Z!XvC5f?%I%|E!&q$A-{jn;Je#+3MVN^}$~PvDLr?R+CX2N!e(~DDr?65( z wHE|w8O9N9wGeZ+o6C;z!i`8P8OpGQosc&L3FrU0l-G|fAoJ&>J)!&T^0HwmLIsgCw diff --git a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin deleted file mode 100644 index bd00d06d58aa38342d2146f3d8aabb668419456a..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 3610 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fti7^fuWIsp^>hExw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0B;o< AlmGw# diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin deleted file mode 100644 index 25fdded2..00000000 --- a/tests/cache/cmyk/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin +++ /dev/null @@ -1,13 +0,0 @@ -Portez ce vieux whisky au juge -blond qui fume sur son Ile -interieure, a cöte de l'alcöve -ovoide, oU les büches se -consument dans l'ätre, ce qui -lui permet de penser & la -caenogenese de |'etre dont il -est question dans la cause -ambigu& entendue a MoY, dans -un capharnaüm qui, pense-t-il, -diminue ca et la la qualite de son -ceuvre. - \ No newline at end of file diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin deleted file mode 100644 index f481ba25..00000000 --- a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ /dev/null @@ -1,54 +0,0 @@ - - - - - - - - - - -

-
-

- - Multicolor - Black - - - Pure - Black - (K - = - 100} - -

-
-
-

- - Pure - Magenta - -

-
-
-

- - Pure - Cyan - -

-
-
-

- - Pure - Yellow - -

-
-
- - diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin deleted file mode 100644 index 07716963..00000000 --- a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin +++ /dev/null @@ -1,9 +0,0 @@ -Multicolor Black -Pure Black (K = 100} - -Pure Magenta - -Pure Cyan - -Pure Yellow - \ No newline at end of file diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin deleted file mode 100644 index df52c271ea1ed1e337e04f2215fc00608a6c7305..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 3073 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%WewwaoBSZKA$S z+uXXjBJSVSIZrt6E!@*^Uc)WGPSHB$rrgA9XLxHe8W`Q=v!0*eYz}#{s&%#4-If_` zkshn^TieW)pNPBhertFa>~-jdK-0|i41vsReRWEWx+hoU7B1iF^CBuzH1w9OWfLd&zq!t8@OnvA4(!3G{L!@HR5VidnP?Voinw(mspb?T@YMoHAL5IXlaR|*T~2KQ_Rd5$$ce7iJ3X6MbKt? zaAs91C=?VF^n>#AOB6tn0M56dma76dPQcCe;*!Lo5^$gynVVQ}sj9mAyKw;k&Ho0s diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin deleted file mode 100644 index 07716963..00000000 --- a/tests/cache/cmyk/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin +++ /dev/null @@ -1,9 +0,0 @@ -Multicolor Black -Pure Black (K = 100} - -Pure Magenta - -Pure Cyan - -Pure Yellow - \ No newline at end of file diff --git a/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/francais/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 188f998a04be08ebf18edd8adfe6536fa3624162..462735e4ac3332c72e6dee3ff7d95a4f8e4cdd3b 100644 GIT binary patch delta 923 zcmbOwvr1;edJZO2qsbdNRO)T>?-=mheINdU-{mU1tEWSt!pYdSJk~v;bE8sDPw4ym zb!l$Uq7?QM4-Y&7D?h z93Lltr`h(*`kvn|u=3YUz;*DqF=_Bj!Mp@H*<>ooe zrS5_iW?wfJYBetnxLe|D{C;On#EHy`BTGYAxqD}|)y+0e`LWu|@vuPh`#Vz%cP>>- z-Y~Db-aC2n(&X&TTP?R_uGZvPQulDzwvDrGlV2Q}H^XFVeCJZvQ{VI6Ot|1Dp)Di6 zYKy}WN6j@$HX1xIyR%HhIi~P_a7p{Qy9xFWr&O3kGRzmXag8dTuvJetgEa?6izv?CMvO<|uu5v_`vLWlw?3rG~>3PDwOMf3BE! ztn#UM?B=wREJ=mC7rxsr?6r&h>!&W`6Wjb=qGHLuw5vXnKTIyVZPevx%22Q3I{k3Y zywxh1ZkrGEEjc+&Au%H0>H$^$jlbm<<~eb?`)RFzyCd@7%D{DZZ1bBO)?EAjp>TG# za@xtosc(M&^6~N7Xjot4!gsz{VYTY#d)Gr^j3*Z^YHdtp2+G&E`IITVec5@g7H77z zYf28D68rLR&!&Rwk)=t|J4=ppSbcDxk$p6oyfKFKcepZ>yoch%YH6NKk-hClln%6rsAf+a7!J8Y$_Q(ocz zxh2L*8yAQg#za245?Gq=Bm6b(<v-uD6)l$924SXRnFY(>hx7N>t@Z$J$E#x{!IAii#wmpIh_7{({leM<-Zi|YmDxn`F68qTJG<}`mBu| z8rcS0cE~ZuB`LUcp7b!>`aFC7rN_0KYFT&W1WRQWFZ)oWb2_)%y(&zPHD_vRLh~)I zg}I7PBjhd5_0DKt%yVYd7wckv2CeZF1`uQR5%?c=}B zdorniO>xniiQ>Fz+xDrqJ$Zg+A@j?f9SQZt{=J1<)~}vUIcMef@~_G9i0XQ-2~uow zs%J&7GqWx1jCXs}U;h8f<)bOy^IZcMH~f2l$KsdN{&^G4AC+aAooJ3-aHUY7MML#W z;+0kB+2g-$tCXIjxWIAe35k$J*%HAwmNv;gIu*Top7Vrx{;8K{sieAb@15IYc5LSv z*-Hjas`be;ojE@~zjtz-%wC-e#mu8K<*V+kH0)z_$hq3(qImhi{e$6YPX1HolmtJp zk$xcJEW#7SDqU-N_Jvgc$N3hnv57VP+BbjrwZ7APAQdLdIw#%m^v&C6%v`Q9=CZ1n zd5OuYFm_~ojQh4p9qH!zNTR{^^_CCu{hPFQxz0ARX!Jj`S#m-M zhq58_Bde;Xx948Z`Cb3hLrI1EYSX1-CW3}MN2j0oU?9@``?bT)wMu52S82rUIvAEJ z#u7GZ=KTjJxZc_QzM&^jq&&lY)AefEsMNDtL)G=0*4?~xzxKWKuFUuwHj}^1J)O(; za@m3tyvtprifb+_{(0)K*LLIl?{S5DQ)>TvZrjDW_}21s?fd6NI0RP2o^!jkVA-;F zDPDIb+Y7}%G5Xx*9QxbdvG<&L&;0)XyY3iFY5#Y6v3rb7>?O5IbGM7V;dj-ieeUbE zk^RnP`CjE<>F(d$UP@iNC#&+5GaKj`OrFmp&SPk4U}|V)Xkuz&WH9*5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fw_^nfr*Kwsim%gxw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0Gx{& Av;Y7A diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin deleted file mode 100644 index 25fdded2..00000000 --- a/tests/cache/graph_ocred/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin +++ /dev/null @@ -1,13 +0,0 @@ -Portez ce vieux whisky au juge -blond qui fume sur son Ile -interieure, a cöte de l'alcöve -ovoide, oU les büches se -consument dans l'ätre, ce qui -lui permet de penser & la -caenogenese de |'etre dont il -est question dans la cause -ambigu& entendue a MoY, dans -un capharnaüm qui, pense-t-il, -diminue ca et la la qualite de son -ceuvre. - \ No newline at end of file diff --git a/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index c741ec2f74ba81b1503fa37e08e5bf8abc94d3d3..ca4e132339ef0be1d16af952f22b7bae6d888a2e 100644 GIT binary patch delta 619 zcmaDW`A%}f1`bBE$s0LT>QDDwHQ>2>KKuuJ)|(riGMtJB&&*}DUh+;r`@)+vM(=%c zyM6O7pN@H^7~;D2(vu%QWUPO3MjERze{v5sSpM|$Z{?buKTYm)oY~(nU6ai;U|WQd z{h{6Mooe%Q4eS|$ogOlM&i}#nzOzV4Yvzhc`}1oz@JBvB6h zsnk9D@}KGDY93rkysr=1@%>&k+vs)9#_mULM`omi&#(U5Q#iNc z`>J2lU(VJl-1z^2qK}TqE5DdIy@&HBy-135)2#}4@>nyZuxIz{QzGS7TnQo1bVZ}? zEYr=dPOUFJwB_Bri011PBD>C%nb<-eT5 zIUDCs3!Ls8K6QahMEvDNGg(jnXG;DieR!qe(V6v?B>|hJJ6B5h{=RB_@ng~_?Tm?g zSDBv*_^5nIo+GR@dGZUcay3IeLrpFP1%2nD)Wnj^{5+S$l2io^7b_zJBLf37BO^mI jLrWvG$*tV6Ooj%NpKxzxG@U$y$CtyDOI6j?-;E0ZUhOM{ delta 618 zcmaDS`Brkn1`bBk$s0LT>Nn=yG2prTUi$}gnrG)Cl?56T)TS?tX3Te8E_TysN%DTX zB|3B8Uh#e3@JQjrg_HgL??YSQgor^<@SB^-`4+Rxu5v2p6N`F ztInxE!Vy}N*k306%JBYuRKp{6XSnjJX{WaeOq1Ao{M(Gkob|Gg_Sih%QNWRSgs;a` zXU5V+QvHoHi|daUdhOTmiTqu9t9`xlTEB<;^zF~HXI}5Qw0RwCWH{?qXOnG@zkY7} z`qJiqVnC+{)5S?AtotiIdc3K)sy)l@qKA8Z75hZb*!LMx>USNcu^N8J3a$`WUs24p zXh*V`bRX9>t0Mj*({@jIfA~vN>Wp(UKKYgWY=16(RL6@wQA;9o>Y}O3ryu9Y$l0x{ znsDbvk@v)(KeNPrWKM3K6Vco5$j^JkbVZBiTh$%byLs9^7U;N&W#6$i-MN+F>YoYT zFE_VHbJYthd-}HXsAaAZ*V)i>5{qM$H;D>O`?72*|Kol!9Sw1P<8^ClADHCJ{r(_1 zZ%OEk!(yM5=I)6yz0JZ9d}r^ikgPmQ*Q2`xHJ{39<$Urgx;VvGD(dvsUWfhvuC+bB zw%Tr*1b(#MH#l ga&jAYER%ud^a}0G3}LFaQ7m diff --git a/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/graph_ocred/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index e7472ec6..2e64e637 100644 --- a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -4,472 +4,464 @@ - - + + - - -
-
-

- - 4ist - ConGREss, - } - SENATE. - { - Ex. - Doc, + + +

+
+

+ + 41st + CONGRESS; + | + SENATE. - - 3d - Session. - No. - 25. + + 3d + Session. + }

-
-

- - MESSAGE +

+

+ + MESSAGE

-
-

- - OF - THE +

+

+ + OF + THE

-
-

- - PRESIDENT - OF - THE - UNITED - STATES, +

+

+ + PRESIDENT + OF + THE + UNITED + STATES,

-
-

- - COMMUNICATING +

+

+ + COMMUNICATING

-
-

- - A - copy - of - regulations - for - the - consular - courts - of - the - United - States - in - Japan, +

+

+ + A + copy + of + regulations + for + the + consular + courts + of + the + United + States + in + Japan, - - decreed - and - issued - by - the - minister - of - the - United - States - in - that - country. + + decreed + and + issued + by + the + minister + of + the + United + States + in + that + country.

-
-

- - JANUARY - 27, - 1871,—Read, - referred - to - the - Committee - on - Commerce, - and - ordered - to - be +

+

+ + January + 27, + 1871,—Read, + referred + to + the + Committee + on + Commerce, + and + ordered + to + be - - printed. + + printed.

-
-

- - To - the - Senate - and - House - of - Representatives - : +

+

+ + To + the + Senate + and + House + of + Representatives + :

-

- - I - transmit - herewith, - for - the - consideration - of - Congress, - a - report - from +

+ + I + transmit + herewith, + for + the + consideration + of + Congress, + a + report + from - - the - Secretary - of - State, - and - the - papers - which - accompanied - it, - concern- + + the + Secretary + of + State, + and + the + papers + which + accompanied + it, + concern- - - ing - regulations - for - the - consular - courts - of - the - United - States - in - Japan. + + ing + regulations + for + the + consular + courts + of + the + United + States + in + Japan.

-

- - U. - 8. - GRANT. +

+ + U. + 8. + GRANT.

-

- - ‘WASHINGTON, - January - 27, - 1871. +

+ + ‘WASHINGTON, + January + 27, + 1871.

-
-

- - DEPARTMENT - OF - STATE, +

+

+ + DEPARTMENT + OF + STATE, - - Washington, - January - 26, - 1870, + + . + Washington, + January + 26, + 1870,

-

- - The - Secretary - of - State - has - the - honor - to - submit - herewith, - for - revision +

+ + The + Secretary + of + State + has + the + honor + to + submit + herewith, + for + revision - - by - Congress, - in - conformity - with - the - provisions - of - section - 6 - of - the - act + + by + Congress, + in + conformity + with + the + provisions + of + section + 6 + of + the + act - - approved - 22d - of - June, - 1860, - a - copy - of - “regulations - for - the - consular + + approved + 22d + of + June, + 1860, + a + copy + of + “regulations + for + the + consular - - courts - of - the - United - States - in - Japan,” - decreed - and - issued - by - C. - BE. + + courts + of + the + United + States + in + Japan,” + decreed + and + issued + by + C. + E. - - De - Long, - the - minister - of - the - United - States - in - that - country, - in - Septem- + + De + Long, + the + miniater + of + the + United + States + in + that + country, + in + Septem- - - ber, - 1870; - and - also - the - papers - mentioned - in - the - subjoined - list, - which, + + ber, + 1870; + and + also + the + papers + mentioned + in + the + subjoined + list, + which, - - contain - suggestions - on - the - subject - thereof. + + contain + suggestions + on + the + subject + thereof.

-

- - A - copy - of - Article - XXVI - of - the - consular - regulations - is - also - submitted, +

+ + A + copy + of + Article + XXVI + of + the + consular + regulations + is + also + submitted, - - and - the - Secretary - of - State - respectfully - suggests, - for - the - consideration + + and + the + Secretary + of + State + respectfully + suggests, + for + the + consideration - - of - Congress, - the - propriety - of - limiting - the - power - of - ministers - to - make + + of + Congress, + the + propriety + of + limiting + the + power + of + ministers + to + make - - decrees - and - regulation, - in - the - sense - in - which - it - is - limited - by - paragraph + + decrees + and + regulation, + in + the + sense + in + which + it + is + limited + by + paragraph - - 431 - of - the - article - before - named—that - is, - “to - acts - necessary - to - organize + + 431 + of + the + article + before + named—that + is, + “to + acts + necessary + to + organize - - and - give - efficiency - to - the - courts - created - by - the - act.” + + and + give + efficiency + to + the + courts + created + by + the + act.”

-

- - Respectfully - submitted. +

+ + Respectfully + submitted. + .

-

- - HAMILTON - FISH. +

+ + HAMILTON + FISH.

-
-

- - The - PRESIDENT, +

+

+ + The + PRESIDENT,

-
-

- - List - of - accompanying - papers. +

+

+ + List + of + accompanying + papers.

-
-

- - 1, - Regulations - for - the - consular - courts - of - the - United - States - in - Japan. +

+

+ + 1, + Regulations + for + the + consular + courts + of + the + United + States + in + Japan. - - 2, - Mr. - Fish - to - Mr. - De - Long, - September - 10, - 1870, + + 2. + Mr. + Fish + to + Mr. + De + Long, + September + 10, + 1870. + .

- - + +

-
-

- - - -

-
-
-

- - +

+

+ +

diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin index a450f78e..a5ca4032 100644 --- a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin @@ -1,5 +1,5 @@ -4ist ConGREss, } SENATE. { Ex. Doc, -3d Session. No. 25. +41st CONGRESS; | SENATE. +3d Session. } MESSAGE @@ -12,7 +12,7 @@ COMMUNICATING A copy of regulations for the consular courts of the United States in Japan, decreed and issued by the minister of the United States in that country. -JANUARY 27, 1871,—Read, referred to the Committee on Commerce, and ordered to be +January 27, 1871,—Read, referred to the Committee on Commerce, and ordered to be printed. To the Senate and House of Representatives : @@ -26,13 +26,13 @@ U. 8. GRANT. ‘WASHINGTON, January 27, 1871. DEPARTMENT OF STATE, -Washington, January 26, 1870, +. Washington, January 26, 1870, The Secretary of State has the honor to submit herewith, for revision by Congress, in conformity with the provisions of section 6 of the act approved 22d of June, 1860, a copy of “regulations for the consular -courts of the United States in Japan,” decreed and issued by C. BE. -De Long, the minister of the United States in that country, in Septem- +courts of the United States in Japan,” decreed and issued by C. E. +De Long, the miniater of the United States in that country, in Septem- ber, 1870; and also the papers mentioned in the subjoined list, which, contain suggestions on the subject thereof. @@ -43,7 +43,7 @@ decrees and regulation, in the sense in which it is limited by paragraph 431 of the article before named—that is, “to acts necessary to organize and give efficiency to the courts created by the act.” -Respectfully submitted. +Respectfully submitted. . HAMILTON FISH. @@ -52,9 +52,7 @@ The PRESIDENT, List of accompanying papers. 1, Regulations for the consular courts of the United States in Japan. -2, Mr. Fish to Mr. De Long, September 10, 1870, - - +2. Mr. Fish to Mr. De Long, September 10, 1870. . diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 30f64b9bebd2a4df7a1b8f32e4df4fac39646850..556f0121c8e71ec43bb5f6ccc98907ba6183f104 100644 GIT binary patch delta 3308 zcmbQLcSUc*1`Z}m^T`hxmFl-fpX__GLvZi+@F)CJ&de1_;JFZc`k*;;sd3Gx1^LAr z`v3jn3`)saG)-swMURlMttyi??yC^|_5CpG=N12JxV$s(|NDCU@%#IAHFcl+&;Pf{ z`8~OIM`_{HhqW{`2$uuj=U8*!}_ytn_M7eQZ$IAMTWI+3qtI)!2ldq1*RJEe_tG;@9y!Lxaeuu1`}w)HudkJtNi z53PBgVYPX7rb+Bv-NeMnrM*GZg0?J7JIteY;G>gJQ%mgD%TK4yT{qKX=SGtqY!S;& zCvw)LX5aeAQy_x{B6VQG~vS*%Hf7L4&R8_X(C^CsTO8$jFI=2OEP@%D;Nt#cWCWr~{DL~onP?k+pp5&4;+`&@48 zrnPV< zmALwyh4Yp@-Ev{8!l$TXC4u$+Ld_rA%et3F_xnxaxt)0V#qTdF6LOSKZ+lo>zvpJ% zw%uw!S=PvW_B8(#(rdHP_4q>@A`iajT&*cR{Q?$-9K@OWs>K|8)0Aj`h5A4x;~1#g`Z&mlXUockKf7feHVGd`|BGf z*H{?uexvrRGkv4fBgHF>$4m0wUDh>ed3XCY?+pe=DW!#VeM~+qn?yt>ux@gb zVRpQ>{PEUZB0=pNgZku_25h`mDYIyUUceWR`p%ud1YMFG zq67-!quwsq9L~(!Hz#1nUtaN*7RN+7ZOvsKi_Q|QUVftDc!gg7ix4f_3)c8h_pw1(v{qYGeUU3Pp$xI2lRexRdRfBHhu7nJ2_09dOS-NHI7ynF|4#l*IdEm|2d9}LnO`h7y9969&MFcS zwcbHRvO{AbA4B+|&pRW^Rx+L6re!{NLEI(Bh_uH1bk-HCw$`7j^m8>?akeMX-_=ED z?&9FEvtKOU1YHrH8F@z_IqV?&UWTHTA~m~rANOwj5_I})#}}42^Ro_0Z8A<&d2xw{ zcae04%(F*D0)P3rbNYUyPjK~+YDt{eDPI`&?aX=qsWUenZd`Hu{KToOeOi+huGXw9 zdHv2km&R)bJ*g|H}Vgv$Qz3p;NTx zm|b9iNQ|q@<6UzWB*$Fckj&_6;hTM2Uc{()EDovPc-=?$ z?S)wyuMC$gbuZn>occgvp=wOY1g>w_K60$!PuM!6I`EOT;_-@A$D(U<_H$qV+Vk+M z^+MC%J?@1!HwFFLb478%8I8u+xy7x{Ycn%fPue-{JLjcijZT|VWnMh6))wzt8l~{5 z{CIQKDTRP9MWOFz1cY66U^{1YJ)zT2mdx#6m2nbl&i*yA`vOqU4p z?!5kg!4ZY2Gp=Zf{8&}P_a!5ZXX5%_OaW0Vm)=ip%`N?DaB^YahE-EH?dwOg^YWUK(?V-N{SZA+K{*@T3 z6+P$4=B$89|FW&DGncS*EBWkKxHf%C!!=&5>xOB+Pgz`?a5zbFeec$H$2aDwusHwI znkTS8$7TJYC_@36t9d+)EvLLTxaeNEFxOV$Yfa}r!BDMwr|9za(WRjmpLl1VTcwcM z8|)Jzyk2Z0TWmt^Yo3y=58jzLZGO_T{AjLZmDVIF*(a~p?9qB+^?S`kwaCL&O`&@j zgpTSvW}7n}+pQNlqc~9gs`U?rqBjg1H8%c9b>MeUG>#Ld-Y~Tbe`R{dAezv@rnn_vw2MZPA+4W`nOJdL;l3Ztsj`C zH?9fY@oDZdi90HEGt zbVkQ4D`i*c(~18TUH`Ite<4u1+V-aIf#-UsB&X~)tCcHPZ7*tGoARVBxPownnxZPzrTXXoR=@PEU$95f_*3_xW1q#|7pzGp2yn|C|4Eoa}chvGg_WjB@O_vi`f*BQ4N*_U~5Nz7Zd zS&>avD5@wkM)qR;hTMALD|Z{t?eEe#y`3k*XTzP8QyF#~23aS6J5E&#RhHeqRkdAp z&#JeZUZ}MDoJ+bpZ#HMful>8&Ln8P;>|p;AqP?rMp5;)VUPjEdt0I!zmnP)bGfpo( zDzx#SZl~sieGe{c-Zd)awz#vPeZ_|d981(Q-z~h{r*wS!b@fkE>UV6}_;jAMHbC`7C+JYE|A-zsz4B-Ac2c zW&F%LG=VKf;9Xq(LSOUe-}JdooLX`srO~0nJVAtQrvmf4%+%s?y@pfwtunUxRsEbI zt(<(fYf9&-MKc#wSbg6b8@ns%mc>FTqrUxDe*QgUTNI+4?5BD8`?q@6nL5kv_!WNK z+55{Wu=e%Q+dJppeJf^rQ0K{^>gO-TZY^%Sv#GTvw71L8VNRP2w_E+Q$jRwFmlIpX zwi#YN{OWIsipz5Ld+w7(R`EHlc{3$=N{`0$qiT&Z*~adc5xaZ6_N%XBo>IJIXa7n` zuddP)R+|GtpMIznGMbol^xEIhIg5o?l{>hdSIaaFHZ$7u+rmfL@UhfmyILK0=lzTC zoj-2EXM1>s<4X(4qt?%+T`URxSl|8do}9bHmVLoypSczE+}G|6oU>2++V)j)UXH~E zkxYpPK3%um$5VgTsbJj==C;QxY$h()RPixyL20Q)Bjb@v?IHJmm7H8XOXr|1>mm0Z zhIgVbB?5NJ{J0@EIqLntAA8p>-q_gDbA^jP;p@q{+aIrs)6zDTYPq{@$!@l@{Jv%7 z^-~kv_O(~dk34)X(_Oo4~`F10HU zr+o~SJ+XF9=ZyV5Y59k0yly@?@Nq}UZsUFCnGf$v-rF7iE@tZ@%^JoZmG43q9Lmx% z{QUjMi+f*ZJ2XZeX?@yZeYpBp>$AM3owF9S-~GJ3{J&tDsc`i0;r&Q1KH+&uUp1J#qQrda7y_*6fS5DutZdYo?rU$nIKUr=5*ZHF3 z&B~f@d`{V|FXHTTH=p}zc-v;a_UslV=3^In8shrbJua}9zZSlFd96k9$0fWQwyqMG z|1|2B=Eb}9J)d($uKByTZQt9^$Ho7~@r`0V)4qkr9!?ezEmt$tGt}f#P|$ZSN=+=u z%+GU4EJ;<+aIrEnFfuSOGcqzXGqf}@p1f8xmdV^`GLP6sW>aJH$p^%|xy($=xl~nM H{oS|#^Oke8 delta 3244 zcmcbjH&t)L1`Z};lgSSmmFl-fpYD_0$$R&=_AB;Xi?)Y27&P4NcP(SA5>J_$P!^|h zy5_HkR;P{5N;x@e1HE0j*QaXO@$q|JtrvTICFtLc2iHRCe_#Lp^nLuk+PaOWpVy1W z{`b^Ri}sCL^L2Hk-G2T0pMPEVt~jbcf4_dz>_2DE|JVH=@o&|XQmUQQ)*WK@4l@!fB$xO;PlN`-!I!>7568)`oP*%f9vhEx#L-U zH}r*N*Pqd-JG0gJQRO}z{ae37cRoEj{p{;Yq4nEidtHrVYYXq*dT6fNC+ZOH`Z9L9 z&ZN873?~1rt(QpMp4-2ru{CE~hihtW<$swu$;PYsZo1umB~#(P>fS^L(fr-lx?Wt> zsxsgGS-ktTPE@4rQq!dezfW;5Kc!yOIAvPvO@2MORnPp*XQ$bp|ChEXTK3(_(DEN^ z1mglXKEIY+pS%0&?#n;(j^`9kJK40g^}#NW+d9(@$g&^oRjdyd4Eucdrg>S}_0$a; z+mj5Zi@ytzZ{zrQC^das;I3Di$?j$s6K2ayKG~VDp1tD>DFblNhl{qh@gyxdRJ&8~Av70T{^l)w6&rt{3JnJ@3T{+g8~PGkT-dT)uH?eqx$J=ntBn759!YdtY5GYuNvLo(TYPU^{q?4T z-Ja(qwT-7quIB2Mbg)``&C`Ef=Fh#1TkaLlzJJ5%tm`7%Pc0mA=QEG3k^6J|8Eft> zdA_gulk?Ma*&mjFusvi^(%<8>}#38j%Ue-7aW{e_5OXUbjta% z#rvh%>mCFqIINYnUvQ-BQJGy_sSn3NDP=LeO&?i{UccEpBU$K4oc8r^M`ZOry%S?A z{<1(h^m6^J1w{*VH?l{&+onynQ~mhYv;WR^4j62^U7lh@r4za6)#$|=4f+p5}Y|El8TIlL+xvu_#R6n`Vy>vQeG zskQf(3EFn8cg@U+%&1-`zSKC=XxjWsFF5vH5maGH+7O*y&pTV8BxZl~L0*gC6_Qh5 zCm-x)Z_Q@)Yw>d^`uM{4%S{hdgz5MlncQ1{ zZN9Chkh5oS%u2_JvyO|#s`||n3A>g0TUB#6YkuA2JKNW_N;2OvR+RfuT4H?5VEx5c zkEZd5cbZxCuD4+JQ?9SsarwMQ;4Fz#cg=o$`ur(ZIQ95?R)OL!xzCTzI;i+?-_86g zHeYm4;W1scOTp&rOuXg7^ELn7--{HB{pz+n{li3K`LEgY6PGRHuoCfhc4R0KwtO7GA7f~}HcReeo%@FOhh$in z$gp_lADsG4W~yP|?g@)b(>KO)$a?3T>11N)HG4fb@#lW^tJ7~a+jRb5*d zj@bHJ^Hpq{{z&5wXSHPJY6fcqucx!>y+mGo%Zbkph+XwO zc~oY<$=Ss3^R#-TZWbSlF<$q3j+ODxeWz!u z(mn^Z7rToHEAKid#a61c_~gIMfe)40YSR82e7~6fxt`OeAZoeFT88@f+;XS;T~<0R z6+Cg6@r1F|S=MDt0v&hn3LFjeewuml*8Gzp&%XM9D9F})G=EVbYgWCeOXJbs2Q9s_ z|GFPoyNbj1(2+S))Fmu!0w1J$HnKhocFVic$@yNQQf9X=lj*^U$N9ckc%S@t;k3;I zgLapCbE}j0m#opQ=QtVB{rJtxhaXsSSRdC)P)5Y_It~hM<<7xfX({}5{ zgmo$!7oYs^WSze)_6Y~`&;Qrej+;$xZL2;StdsFQ^qAposkVPoZw347g#uE6O#lD1kb z_`0%|si$+x)0-|v?%J(}ljOf^98F!5?i;rElbZciOFxrW?yPSPUYqyv&iy^@HgkBV zOnOwmj<=!mrVQ&9(FU7xREJ!T~pfqRlsH6F ztLe>X<+#7uv-=HK*;2OyGm1W}XP>foUBc_TcNPfG^L8njFTi4=vR5Zaz3}hFyzdNo zm)~b@$rfF~vPf_F;blLXbrIQ2pGGuf!g zM>nzN`|0tdd{aDjtiCH`*3`)Jo}Kl_>fg@XpsIb}cGJw=2L;x~Nngkf+4r{BX`j^v z)xrxZB8NLJ+_TLT7tnM6^;>HpUw-@US?#lDwJRMe7CUjjO*(%2`^R_oT${vowQRC@ zl);xd8J%aM;>9%+szjf@3k!Yuhxdt~x?FqS;Vs3b`F(Q^=j7&`dM(;+CZ{QIZ2r?$ zCG#Ei-#T{fNjlc(PO#*6$vjh+YCE1bXV9g?Xt@m*ek?FvCXPWgl97!C4f-P}@P zrgUAw|GnSb{o5Q?KKQwH<>TAEXD26pRGGi2^yzA5`lQu zUggj0W=%Nx^W)X2{SOzIS3k~i$Hs!$K zTczQTs{F;@F6jIG+E#c)z|B)vcm1hK*>rTD+VldpjOIY`)o*N8Flnz%*1DA^XmshR z)Uj*aVN9F8nHJrg=O^})IYH#xx5@sZ=bc vhL#4VhGvE)rY45wli!HOG8tG-_7>a7Y+`OQ`M;Pqm#Kvbm#V6(zZ(|-VK7X1 diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin deleted file mode 100644 index 7d8ac39f02fe6b90b1dcdac74080f96a3944ada3..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 3610 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zre|QF0HMJ3uycN1Norn6v4SB;J_uw)5VRyLMyduuH6g6Dj8@POaB)-64@s>kQPB6y zO-xU9g3_K)+8LZ96(ALuf{l%Wewwa(VJuxFP{~@MC0wv!UUPJ_dbD73euLu zA7hqsumAJsU(&uiwt3|TbN2JTv2W}=m}EF%Zf{z(p?uxF-PP%z&cQnCHx@FV@BU z3}P=I=Bugse0X_)ef^itJIuF&`h2|>US~{i+sA*M_heH4n&P526UBMcw(V1Id-D9u zLgtq{I}(chdkeX&Up<|2&dTrQUz6hz)%9Exq}b$C&x&4WW?R@9@Ajs@{Qs59M^n7# zy9O?9`1k&f#V@J-^Cp@PumkgX#lV>_}etdrK_%6PQ`jpS`iOxSS}_Fgv+=y7GDnn?&PQuFEN&_q}cx zhkV;Jt<>9U)$1)E;`=vg?Q)%MV$tY-X0zmk5DsNS=0{dlPjAn?p7Z;shms2S)uv0w zOau*ij!r-E!9b+>_iKlpYn9A4uhNLybucVdj3sQ+%=-^caJ{qpeM3*6NO^|)rt8(R zQK@IQhN|l~t-E>Ye(ihdU77JWY$ku1dpeix<+24Qc$d3K71vx={PWadukFV9-{T7R zrqurT+_sB#@vY_O+V{_ka0sl3J$GxtvSsg5yzWf47m9yk^tsJB^tZiZ?>Y0H`ThTQ z-7%Qb{_pf+_ZXYlOKO$oZWnvQ@2XGx+}CR(`<=`3y~@GT-M_iLl)83flrKg&$`??9 zpzocT0&2>G%Ag={F4uR?FU>1aFhr^!4N*H50Y&*KrOBy93K}7)#l@*biOD4jCVB>X z2H<9qb5UwyNoIbYOJYf?f`*Hgk%6J5fti7^fuV_!k+H6Uxw?UYI#^{$W=T$}f{mLi zsEvatJSvJ()3^*242`%zjUWhCFf%nZHdRPdfQuQLDuAUF@?c`dhM-~xB#00*Ffg+~ zS7%^kWPl-NVrq^dW^RI^&d|`r7+syAv8f@3n589#ULzv|3^8Lvb0mM16eVWnq!vN@ zZ^4;Wsi06$P|y#`&o5B`MFO}$2X$!_z;OcZ#}$_(7L|Yl&DhAoluK3B)!&T^0C$!e Ang9R* diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin deleted file mode 100644 index 61f78d82..00000000 --- a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin deleted file mode 100644 index 25fdded2..00000000 --- a/tests/cache/lichtenstein/__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin +++ /dev/null @@ -1,13 +0,0 @@ -Portez ce vieux whisky au juge -blond qui fume sur son Ile -interieure, a cöte de l'alcöve -ovoide, oU les büches se -consument dans l'ätre, ce qui -lui permet de penser & la -caenogenese de |'etre dont il -est question dans la cause -ambigu& entendue a MoY, dans -un capharnaüm qui, pense-t-il, -diminue ca et la la qualite de son -ceuvre. - \ No newline at end of file diff --git a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 99418cee..8bd5f1a7 100644 --- a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -4,16 +4,16 @@ - - + + - - -
-
-

- - + + +

+
+

+ +

diff --git a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index db5de01cc9913b1d766118de8c1e1927393b3c85..06ef4e3c8670c8abbbdc66cc115e7f63dc363f57 100644 GIT binary patch delta 143 zcmZ1@wp47xdXC8t7?u6BLkzYV1R5|nNz9$ITHwS}kwd9#1cH_%aELWcP~6G&D{~jy zE;g=ZZ4M_*Se=u-ZH0@4TPJ9>X0S>#Gb=M&w??!I=(Flw7tmucnk7^)S)H?-*-+1L y@ delta 170 zcmZ1~wnl8jdJabO$qyNoLQOjrxf&b<7=G_dvMbnCy}Zbgr|GC#j_~9J&29 zgySYFaF(kX=ox5oDJbYW7o{eaWaj6&B$lKqXt-Dz85mj`m>QZHnwXjxnoZuz8Ovm5 aFj - - + + - - -
+ + +

@@ -19,21 +19,21 @@

- + THEY TIP-TOED ALONG.

-
-

- - ee +

+

+ + ee . - - Se + + Se We went tip-toeing @@ -89,8 +89,8 @@ root and made - a - noise. + a + noise. We @@ -114,8 +114,8 @@ in the kitchen - door - ; + door + ; we @@ -238,7 +238,7 @@

- + dasn’t scratch it; @@ -256,7 +256,7 @@ right be- - + tween my shoulders. @@ -275,14 +275,14 @@

- + noticed that thing plenty of - times - since. + times + since.

@@ -303,7 +303,7 @@

- + funeral, or trying diff --git a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 1744d114..25cb8633 100644 --- a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1,2 +1,2 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica Detected 60 diacritics diff --git a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index ce465f7c8cd1d76dd861f5572245b43432f3f007..8f5f36cacbe771655aa1dfdf89c3124e359537ad 100644 GIT binary patch delta 2555 zcmdn0@ls>MdJZNN)5#AQmFuTQ?d((8Dfsod_9ymPk7^kXF}+%qmAsermT=13q`jsp z!SVYY)lOQRShMQjf`*xl=PmA43!eEer}x|Wu>AZjr*{7R`h9=j{kodE{{_dt|L42> z@3c&qws8NOP`iEqt^R#{-u|!Qf8Wa=|2`Y+I@f#r`~UuW`TtjLz54b0>5uat|GuvO z`1s>q9kI<_|2*QRdW3Q3>Pm-hDqi_u>o>_d#(!}qHZ9ZO&aVB>Xx!6P!^Fe$mFN4v z_v?!p-=AdiyIy-v)F)ua()QOsqUV2Yh;NGQ7dJQ3yZ$t$Nb92Y&MNiQn=cujmrb&$ zFFqO*@*@4C)Y%hewst+4Y`;5BJ+~^ca5UfTTEC&`m~E4AbxYZ%&-(<;IX!!d-KN$} z3whtN%(ph9rC2q~>-BQY;DDnGbzg7bxaVUh5q_b|BPDPb73v(PI(xmGS{zVhvdM*F9~Uw)EvP4>^scYjPya&FFfyyVHQ zWz6{&QtHd=g_Y-==6Id@P=4~%ug@30Do9#nWfRG>zDl8Ly3q2}&1!C&{@R%)*?rGD z8&#>x_wB0kp)av}Z(Lj|>2XnT`JP>(OD34wgv&kuRI!Q8Ew^Mtmr7Wv`;yaEHixCR zt@RYUwb4EyEhlnXW!zTn>7Q+ubFMLSjoYiyY`G-!0Y`W5>znn`+vmt1(>UdxWbIz` zVs-mFvwodld>@#;FP~TaLF7_h@}*tLa~ssYs{3DQbc;M4JZ0v>S;8^=<&MQ}9@(Fj z#1c+;SRcN8uI%*8y43c|H_yebzQC*!ojQkMV^+h3k4xq~xLEsQCTqIIt|vQfW#krD zDik>1s{0k8A@!W`Ota_lPm$Z6=N9NCV`RdYZ+6FZZMSmx&eM_~+h;~!yzH>IVd_br zKQ^K-lfGN>F4oB9*?HyzgA=#;*EcCu1=mk*Ena~^wyTzIgcjpN1M?+Vj2&Uqbt z);ia4s>7|yP5;)O*LdBt@Ph(F?Y#?bih&ySy&N0-8+T=@#2mVRk=^F$=Z-3aG^LLM zYz`GE6Fax?tnt2EboPasPgeuSD$~d{b_E4KD#aOxDyKA9X#TtR(`U zOYVlP$~o<58FT9Y>iJLf%GCdz{KeWQ<&$_K?Rf0Xch5S##brv3A2Usfa&hQ+7@}Ay z5_j(B^!f%XRl_xN>iGN}Z}d#^t`?1no3u=oFZqN8Bfr6>u+_<3zeV&W1u-9S_Ihs^ zeUvdtWqTG&S6bzh)CX*@*FBuw^7158gJY)RWDCV94X4gmoty7`hyT8D981Yerg-hx zBc*v7>eIF~h|J6Kyu!-N;47}bX;lu-%yzDge;+(^%IL4>esHd^>*RFJ#O2#Nw@OQecYlB1oc^4H^Ta|Qt-$)g2^%&% z6IK3rJz$%74sYPYl^3}8>Zq@@W$u`sApS*U!?W}E^irG`f1TuFyR7_#$Zz9IyrHKr zZ=W;w)`OfNnr~YT?oJbAT zxODBa3a3y1)8;=}qu?vN?Yc-ndFKM-1ICyA%AZ`anm2LjXAZ7C4m(QfXH5|8I?X0ueTKUHJsW^Pj4y;yMR-SZ`a(d@}u#kXGXy7DYz(u$jfoIhIEuYD-I)-0tx zO>|Okj(YSpImWjS1<&rD6f!4QKS64LeL#rw-HS_Nw}j6NIlZ9Zf>pP`n|1D8r&hN( zl``ubVEH;*>Q!%-K+&~#v-;0>ZC>Ur6`K%xV)wqDd96OzBc|qCq-}Yh7u#s4J^7q` zZdIRBNm%^$y_QNJ7T(O5=0E3Lh`eCHJj1S>^uoFMJH5r%wl2+8{;nGs**H~g%dcr$ ze|glGPxv(Jjbv+VbEdvE&#NteH|l*~$a!JUmId=7j|Sg(_9NJ{d#{K`_~y(RS0#7% zEiP7mv|ip&YTve9Pal6-q2sFT@qcUe>>mFsZa;ijuMl*P19 z$<5IGap#M0E3?4o7m^)Z<+&05tD?0Rt?!=1=6q9cm2&;u3Z)Af-(-%g-1*LF_m*>J zFR$#Yyg5tp+1n?zpCsH)src^R6_woV8O8SM$V!LijNH%5n7(CHdX|WP*jioSnRvbF z&y~$>OUoRm85J&0p4P)B%vx#tVNn^E_pi+yhn3XcJ5+2Exv15XcXIXeiZ>Cfcxon< zOo@+BuKNAqhSkG`ruA)4HZQKJ`XCWnz2MTCpgncciskzqEH_7Wa|`;<)Yzxx#hJf0 za+m0~CF;j#?Jx@R+}t7Vuw~_@%CZL^?W;IB6)T0$>MuHbcE$JU3!mx96ZF?tkZt%0C^G6>+fhpR-2%pv{eh%0Jw}DGq<tJgYUPJ z^$NXJN8(w&6+i6n4S&~@c>5#U53c%Kjl!$mozdRH{;%Pm!@TUWvfBcROMXrBD!akC zrRsCjhSe7zzLS#P`lhosL*Ua|!TeI?AiMv+4!_avuKYbuA;ymN*U1j?Lq&H|ehcmV z*n8D9_1nHggZnHB{lB;;=}KIgo!*?$QWbSY<>wx!l6s+s6T9CpxK~xQy>7nuU%|v{ zjPnZhrhcR1Z#mK(p#d9_JiujlnbFQJo%4?C~=U)&|-Jp8j!*c)n{QRn}c=7d5XxUh`OZAJ>{smB}ri&sx;C2>DD*Ysqu#yLQ8+`}vjq z&ysTYnD+(fTK9ajx;LkM+0Vzivm<^dzsR~hov(hI#QPhs6sDcancKgoOzY81gKKi< zO+IW*WuIo`&-F~aWJdp%6kfHTXZoA#jeqR+sqddZzqqqZZ}RCFwQnACXInkiT|epI z`=5sQr{p(OPktVLgu8<|o5@a~$6~r|xa{+lIc{^<8|)>YPZO>=A$(xB|L{gnDF6g`bK&|;^?YzRh2c+vCADE|- zIp;I)y;G%42cEN@eY@rD7IBMJp4P5%hqebQBo*?`dMy8Dk;KdT(86;HmmXLytng$# zFQRf-#`@!}!!N4c(#~$*aY8*@E={3ecgm$j zv3^^YImJz@_~me;!QdD#W82HsqK%CmF71=Gavq&I@oe7l6&5R$pL~y$UgYLBEw6Nql2%y?Aq zQplO=sqP#bDR-7Xi zzx3Z0Et5zILHiqL=ggV=&paf*U{}BdiL83#xC{112jX{L zcYNcW@}j<_T(C_i`9VX7Ufms$DSEYgUtB7RSZ#3G&sB+)t?1vQia5K#4w(=6lcg3W z{g^p_rbm(p-*%m9!zCeo4FUnb!j?ZWIrslT)>;NWO^-*+Q?jle)l&}Lxsp3Y-jrig z(!}q^z0o(OI@wsfo8{S4+n;xS>Ah9%J5Ls>q%Sz6);X`e-s6K7x8U(}H!Dn~nw2U= zPTYTgZy^Jh8&4ziz5U|pUDCblzASq;$weWMFK@NV348qwmtylQ4Oi+$RZZ|+>v8A$ ziZI0&$6il5rQ%`ixax!@>#t5F?njeCR;&|y?9s8eoKKQzkLTNlS)rfur4|ZZ)sR{( zTK{c!xU2d{uh$mumhG)yqax2Uqv@XKZ?7P>>uUs-GIi%FPTblMnq;`FG*D>ew(Ope z&-E{bHoxZIXKCN&+v8w<%w)mg?qdf%E>xy%`VgMJ^Vamv*yC#)rrqMy`srN4a9+Ee zQS$1p5XN#lxp^OcvCp@&m}@upd3?%j4;BT_?0Ny#&YX*1c85N&4djfuP=6~s@uZZd z|2bu?2*p!om+~X0I7fHHubQ-Hy_5I7mo9}Xb6>qU5FT=U&zafwCTG0vo$&MX+q|u5 z+FJe0+fob(^TM?})b`2gX84+U$zEMm8sNUFaIMJ9+3YJGy^GRe(|!~Y&^gO1)h|$N zS`gRX8P{%Y*tS`km$CK2@obTN(Jtfq*{5z^=s&=(tCxIwcU-E-|AY7M$i`ZxEzUZ+ zM9sB6wPucntzX;2CM~wW&`;@3=aYj!?U0J^n$5BB=#B?Rrt7_g9L) zdpT>4@TsbYtTT6-NLgQefBI+78l9pe6U#PF$j&W)GGW?w^_%PT1NX?5Sm=6QvNMg( za;%r))KChZe7*FUXw%JJ8IEhsKisC)S}T`k9$XW4&8SiB<_DSJ6Lu#KhUD4woOisD zdt>>=?KT}f-Sf6q1wEF!@AM}i(%EpyvWl(!T@U;oCQSZtXGP7iCHr3m=ijN?u>5IW zm8FKZWSR5p)X@9swdv99yLLUS#exh+wyd1~i!^49W%nwv9z zncnYA+SdHj?`3$(-t(6^g^3#Dl*-vFXS8eaP&pUOc`l;Sht1q>SLS!YEn>UBoDH#ag zf3b3ckGprJz_B?#9_yY4m8^Esy}thX9F=#PuU9c+X6G?x!Gjx}uxj}=TlCg1M-UeU2G_G5*#-y`k`=T4fh zsyS>~q;uhnzl3{*^**QP*}i8U?_2x+-0k_*>sXuK&03qM|K;tZ3jcRf#XDSTA8!p` z6?BH-@2wRpW*wfUSZnU5-Pl<|5i!d?EYH4^~KXQxhvIY)D?N@zFH7%Ur;J& zeP_w`YKw3=t*fr_+j-@?t?!inSj&I6yt&-)vEk)K83B3cpGIW - - - - - - - - - -

-
-

- - YOOOxXYOO0O - pixels - at - GOO - DPI - - - oO] - megapixels - -

-
-
- - diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin deleted file mode 100644 index 16b617e5..00000000 --- a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stderr.bin +++ /dev/null @@ -1 +0,0 @@ -Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/stdout.bin deleted file mode 100644 index e69de29b..00000000 diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin deleted file mode 100644 index 21e1e995..00000000 --- a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin +++ /dev/null @@ -1,3 +0,0 @@ -YOOOxXYOO0O pixels at GOO DPI -oO] megapixels - \ No newline at end of file diff --git a/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin deleted file mode 100644 index 60fa7ef5fe5bf7055bb05ebfcb497b07ef0255f5..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 2998 zcmY!laB5EVh|Uk-6gd+IW;dOF|Pz9 zQmhcIU;;8qAr>U*nwOlPl9`vTpzoQInpcupQmN|!5)Up-DgnC#a2F#(FTnZqdAC#J&SzJ<7si5JQlAn|c6AmsdD9B08P0cG& z0J-1J4&-8xTS4Jblvz-cU!-6Tvli}MXHOT98j!DzKpH>{1$~eZi02ZnU}|81SX_iu5Pz^j9Sf#Ed+xq9A`XwQEJfz-Sd zsNbOJ6O>9IL8G7#4H#onV+A`qE~q+vXHOT;yvz~>{eYtU8V^c1_lNO zb_zCbZfWnMY#e*Gbn$<z`($f zlb@W(zylIs+`z!Vz?Pd>QNX~$z{$pE6+7`WhSK@3nBg7tuX%FMuEs3@q&!te*;Z)OH9 z2C#b>SQ*3^V18j`;ARkGU}j)s;e?8ZF{CmSF%&aoGUPMlF(@z?GUzcFgryc0XXfWA z80s0obb@RIVX%!L1`K1AaF!@FFeI&l>S|CH2}lIxQ)s2FprG%Ynv$6a$}-WQ3}a+y zW}s(apa7-8HL-JkUP)?RNwI<DqpdXT2QKF#l znVXoN>I9`dp|mqNPbxsFFa;YM1^qN#16>0JQ0pNGR2VCO2t=y_Y&k}rGBPqET%*+n z?Dk_Z8qWOWnT|d7xm)|iKPxR}x+`ndO#oMs{ zzc&C}3qkl&2nOEaCS>5#Vl9_?Mh09mo-e8mXYz3!q z;HsY!&Tosnt5PhJnEznf%0EZb`pqA0>Ew+#GsR^iNBDMI53PUl>?Ujc~+0D5~_mGgCm#EKvFg0*Al8bAD-FiGm?g z*=2~@=nE*yPbp1KEmF`3Ni8l;ElNx-Q83Xn&@%v6S(PscBpW3Wi2ppb83t70gUc zjZGEO6yRcprV3ywg*=#;k%^@Rx|o518K#(#kpYG}6H{{xF>@0Pb%ur}#^~w{jZF

- - + + - - -

+ + +

- Replacement - of + Replacement + of "creationism" with "intelligent @@ -83,9 +83,9 @@ 100 - - - Cc - 80 + + Cc + 80 > @@ -108,7 +108,7 @@ design" - = + = and "design proponent" @@ -121,13 +121,13 @@ 20 - - - —@— - —@® + + —@— + —@® - + 0 - e— + e— T T T @@ -135,51 +135,51 @@ w ° - - gp) - ee) - 0 - oN - g\ + + gp) + ee) + 0 + oN + g\ gD) op) - - oO - NC) - NC) - LN - eo + + oO + NC) + NC) + LN + eo N N - - S - os + + S + os o* - vs - ws - os + vs + ws + os os - - cs) - Re - ss + + cs) + Re + ss & - ow - x + ow + x & s? - + ge - ee - Oo - ss - Ss - Ne - qs + ee + Oo + ss + Ss + Ne + qs G @@ -190,26 +190,26 @@ S - Ros + Ros % - se - se - oe - AN + se + se + oe + AN - - Ss - 3s - S - Ss - Ss - Ss - Ss + + Ss + 3s + S + Ss + Ss + Ss + Ss - ow? - \O - Xo) + ow? + \O + Xo) R g g diff --git a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin index a59aaf6198443e101ffc76be5decccf72bbc6fcc..fcabb2c4b9cd85684b32a055aaa416e1fe19143e 100644 GIT binary patch delta 1623 zcmZowJgm53JqNR?q1ohzj7s%uBX;)5m~g}&KumRLB$Y<>NiCoHFr+wafaU;Crvzr*dn?|D!8 zZzwD}A?A!1#->i0g-&`TTci;XO++SS3Zm<8a zb8r0D|8H!6#c#9U_Itgwf+bhh%YUhppByN!eYkAB!HGq;4r|S7>b8_!)NwT=M{h=^ zv2amv@!I*yq2WGt1&lFyRmwX~#OC+Bo%7P`u;jg`CpYTUJ&{ZM_;l&sH{XA#@LF0u z6PqC;S$53Xp>BE5+822F4oNSnE&*z()!kkZhTyOMRqTrufBF!aB~Wbn<3#ro+EowWTidF;ratdZ~2Uw_{VpQ-zr?Pgg!4<8`f8 zJ+fzo@bnCkEBuZt`4p_}4c?Rl-PX8jutlcsfiIWtFHYuixfaj#>KK8*w+OC9c=mQ_L69a`I9wW2QliphNO)h6RPEt+^-v zjhT|3Cp&r57eCuoC(qYztN#^tE;?HBn^B)-WV7%V&P6T9lDkcrJ{52NaOr$h^}+2E zJVkTnB%O>DS-<96*wHr8*tvSA{49?h{Gak}a>Hl#C-V>RCLf$#$QH12UBu`0%lG`> zQ6QXHHt)&~qlje(%OoG{s%$lIv2a^Hdz$6iw4|B07Nn|3iHStk^gQx7;iuAAUv$X- zRw~nOUAv4!h9XQp{!dp;dGddO(BsbOK0Mui*?+G*?&{9@q`J(bRA_;lWYzZiC^h!i z>x8}bxqj_G+O{d0C3f}F=LXm04%j>kEWcR$kNa}cy%&goXy`^+DO|4-na=X|vIiVL&7n4V-f3YnkWgQJ7 zOK&dKZLGhvr^0?ZkNr8R#|mC|3wTyvOPF0K$~!qM`)1^peA`L>d~5c-`}VT* z?tAmI+>*n_iJ&KJ{$ zmv2omKBW;4KMFEaXR9fnwL^sQk0sQ%Qd-+FIUY_&rp*~K|$ZSC^fMpGe6HIu_RSN y!^O(Tz{tSB%*e>l%+S)vaI!3aER&_-#V?S$(8fh zrmIale0Yk6kazwPU^TA|sMjug}!_z1M=nviwR39J81dvYgbIUJ7+O{G0K}y;$?z z|3Vg>vx@RJD_?hJ7jmIB3A;B@i(iePC9w@&x;<@gnu(;$LwWc zJWLN8)vd@j#V7lViJqd1>JLxvf-YK7I?*3iN!*#{$ zqUn5>bNt?y3OSm2ik|ZvJk~OExxjUXTgO&2EZ_Q9tFgj-m9XC7OWKFmMmP%`Ub|$; zQH{m5%+k*%eVR4J@fdG_ql2Q4=JlU?!rNUQ?d??kHf!lw6YCrG6^5xV_pjae!cn4W zPhH+NSNCNTJvW~1Sew^Bqj33D4{MKG3l8$+rWU4NeCu#U@_vPJ@E=XRYxmh#a(W3> zyJ}xGoNQ2JH{D>`M*S^%zc1LV^X?0F+iEob=^Ks6{0~3R3v7EfW$P~Of_#I-8xwE8 z-p3Qz`EavDgLv2D!%kJkD{|{6Uafrb-OX5V*Ea2v>-!FwpQxCU<<%W{GH%)YH#bW7 zE{eGuTc@-IZ%VG2E;MWEC!hBni@pD>@t!DQ6qdchJI!UzYcG#ie;w@;r|k4|b)O&P zYWQNo<%fLl=kso6nKD=MiP{C&@V~x$}_s18vzPH}S=S?m0}}O!dxO zS#v|DcbfXdRpuvNm0HoJdF@Z{olS0MTk3C#FY64w!nWFYLf6uil4~1I?@OEDv^eWp z^|6idJZmSP@oWe>u9>)LZg~0bDM|XlS#x`iy4@|EWB9l%a`Qd!6{5C_jQ_2Po}qry zYx?IvcJ0oNc)xXgiNbs3zkQt$tLu07^=q{Q^~>@m@0;)8dU!8imA0Tr(~YcIo3-l_ zyrysSeKtcY&F}zw{)A<@>R0kSOum}zm02a;^|6jC&T#j=t!9&M{b~{NWpLP3+>@}V zTX)F{(}h<+L4v*D>v9qMBs4kzCpy!KkdZ^UM>sR10y*+zZ-rZ&BCu zd+DE4*e};7pD{SC_k4HzGV?{##df8KzMfnf-FP^7Hs{l&hNhtcH|zrDo!wd};h(V8 z-6Oj3lwnBHV~K^1i@8@#FlhbI>$1XO_T8PS2J?1w*FM&pICtZjoy!7uNMv0%$nj0y zyX-CF#zU%U^ZGsCvTy#tP{2FU$@_%=$(GkKiPt7hQR`jlxv2i(#KsK32) zBbPhKqF6kjxbj7onwDd=Z9#x_@j+hRTaGh!=Uv~}&bX|bSNlyAZ^gY%&S%HCTee(g=)ObRoU>IoTSaG^Zk^88yXADho75SX4W34qetvsy)W`qhK<&|8b}8%bEIl_( zc*}L2PYFpGT^}vBXdN~Ws87v#*(;;?VZynps1D&(%v_td%iT&%Kl|guv$p#l{}S8_ zmc(4pvrvwne|S?#-ZAT3H>D{%KGx6Gp6Hzs=hNzZTWFi9Kyvuw@)a=_!4pDH-?^U_ zvGlC_dX{6W&#amn-oA3du@J+f<+9m~vmYOLZ?Vo*M2kCg>9h*5Ij7D(dmEZ;ZgtrD zkkapB1-`sT;@YC$3Qz0`@wu~ey@B&r>mP?~CQmqG{_Aam@a3J?-_5%4ckzKe>_7SW z^2=o=@8c_1Gte{8gzEXmBzb4e^oRnTy;GBPl - - + + - - -

+ + +

- Replacement - of + Replacement + of "creationism" with "intelligent @@ -120,9 +120,9 @@ 20 - - + 0 - oe + oe I T T @@ -130,68 +130,68 @@ T © - + 3) - ©) + ©) Ay Ay Ay 9 o>) - - ee - ow - oe - oe - Cs + + ee + ow + oe + oe + Cs Cs - eS + eS - - RQ + + RQ Q R R - XR + XR Q R - + & - es - o + es + o a al & - eo + eo - + 3 e - oe + oe we a? - i) - as? + i) + as? - - 3 + + 3 4? 3 & - oe + oe & & - - oe - ww + + oe + ww 6 e Qe Qe - Qe + Qe

diff --git a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin index 603059cf45eae5ab7b28a9617bf1e03c6f9723db..2f7ca9b94c53c3c7f4c15e5a55159d2c4317f58c 100644 GIT binary patch delta 1368 zcmbOve@cGC1`Z}O^T`hxmFm|e!50wiujX$jbANqU&qJaeeXYC&;MfVztRhF zjml*#Ym}z1e>Yt}`o@#o(&_qqOusm8#{Mg=-x2TZA>rhwI7MOp{HKhSkN)Y^+kZbV z{?op-;bHUd%dvIE4Hu8=Ow+61ePmzPPUrZ#sp`^^F7tP3q{p8YaJ}O@^ak&So~*} z);`uxf`u2izbakeaVoF-ip-=*-Bxy%Uze$eo@Q-gQ+wTcz{7@f-yZdKEEk{u_m*_{ zH1p}=irMd$teW8UEqg`&i!Ps8nO9A(>a8+js%h7nAQ5<%>sFRF57QoHCoz?YKf-Ef zaxjL3PON`4%dg{7OQ6XyN#=vOAtBvU{?E#_ows>;&*zUDWR9opH{}eQ|LBC%l9Q*V zIcQt?I8MkvTJb&S32Q(5W)3BF!+o3y-48D2GNsOG%CKB81U`wAa!i#@#T-Dln{a?@O9 zeRVh((_4LQ{o=`6`M)hTs{eGUVS^su+6n7xJ1w{}9>;T^SSlJIan9_jpN7$iDsVWpU@^Z}~b6r%kdiY`$F)r8xClb@MGI<_p*MJ~cTc zy)*Km#37^Gmitw$9_gH%o?#?j{>pNr-P`zj^P)}1!HI9?*T=%JUY*H4qr4?=w+82F9OOsYkuXj8(dsn2Hy0KZ>7LM)R zyDu5-4WGPl+vbIFOP;zGuj4m;y(9Fj%+D7MJkK|&O*azjc6)tM7d0XgbcWcX%oyYsbF3jFMHTd7G%g33M_*2hrQ<|)_ zEcDu)nfn>HpOR!r%}XgRnLLTNNX<~sP?JkRLEpJ3HL)Z!KhGtxBvnDf#mdOQ$iTqN r$jH#l(9+0YvKC(qv#E*ElKS_SW1`Z}e!^saBmFm}q+x9;(<9Yi%`~`o6pyn%X4yHG6&xyQb@)2L+wc)Pl zbff)tOLQWyM3h@Mvv42g=Kt$qJE{MRxv)mWnPVU2CvInay}!QBbpPI(oqH>af8Tbf zWog@_cp|CjcUIl68+VTRckir~C^+cW_3T^!&%1x##5g=pD$ujsxBnw+&CkD)^)>n5 zx&Qm^pJ9LIZ8$T-HTQ2tzfU_bNC(*+cr%ua@Rd;-_dlXebdu22f+ie4D`I(rUGbIVjuG|xxbLS0bPL^Vd`ELEc5xOng%T9jYVo~NfZN*%* z=#J)tHx#nA{QcV9_3LSowJJx<7Cq_nXFlZCt7yKrbUGlo?zbXep%w41B=0$s_NLFC zzgtK&Y`>M-b7Sv$W`5IMf;x3yzU!5D3s6yS$WWD4X3`VtJaB1oI_H7ZPxTCp1-AMJe*Fx?%Fw^o_-l ze)s;L^;~DJpV(q7t=jlNMgGpU#VOK?=hj<&)e1Yl^&9i`;GktC3-Z^n&e{HusZpop z;=+G_b$Gtd`z-R|*}+9OcQ_PsRoNH3+_I2av~9CaowWGHxU+9%8Gjh08p!S`3T8fk z$>S9FDQS=9O9gM)IoRb>PjvP7nq6a_zbn&fmmf|NNAMUJf->@rk-3#HZ zHe1vto#ky>UA80pklvwNcM~r}&ARGS;Wq#2&j2gGFiq2RPp`O*kw4ofN_Eo?uV zbkDtOySwWek5gvbh1HVei?4iHv*x(Q)~i98j4wGO&2RtvcC2;Y=anUP{tjENTs}KX z{2kkqOZDV@n2lhxLUEMFPAae-L7=gJw|?Y8dJGt!8=7y3luK$CpQi_j;64uwSw@~&=sBvvr$ z$$a(lsW3Rb)XdLqX?Tx7*+1J0FU$45?f7H6PKJkh(>}$7GiLg_QcSMPdFta%FKCZZ zt!uX6et56r&IgAs#dU@{&l!qm-1@pT!}zM@iRcFo$`iYdW3zZGuQ`dW3wx{S@B3an z(D?nYf=O}pffKo{b6+yds1Dr|mnA<{86-`NG;Uo!}JPH zP^jPL`uk(&ZZFKM6jV$Kqlq>^ZbXCG@}a zwfu?IosLf?IoKGBrt&`Rz5mRqdh??Y>(bqGoYo)v{g2Ux@5}Qe-vcYpr0(~zpQ2Ev z8Dg#Sd#(w`lJooyrz}=au=~#C7GvV`l;??dbjr+4PV-*URUV8=Bo15P5cGNFDy4(QuA@ev8^*}`2KNQ?7yZl z`6zEWvw@z$WP3hw9z#n5Q$sUD6H^mov&l((G0Y|wW|I%{ZDckzHJn_^@6BaqX2_+g J>gw;t1pv&oVa5Oe diff --git a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin index edb5eee7..384e7c3c 100644 --- a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin @@ -4,12 +4,12 @@ - - + + - - -
+ + +

@@ -25,16 +25,16 @@ there - were - a - king - with - a - large + were + a + king + with + a + large jaw - and - a - queen + and + a + queen with @@ -382,8 +382,8 @@ rain to do - honour - to + honour + to a dirty @@ -490,8 +490,8 @@ rough outhouses of - some - tillers + some + tillers of the heavy diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin index 1c10f1167bb81026a7e62cc79ed87ff9ee3b69ff..e43434bc93231f0ad62682c58e899536f7599996 100644 GIT binary patch delta 5999 zcmccabkKRjdJbj_GqcGL8I|g%#-3hx*Q#jlZ|!gV5eqf!ToznBu(kiZ!e)kftddV3 zoqM;y_{??lm1~pMYKf`JnlP2$xm7jcYsA(p(ZB1%qyC=0`v1wNf1m6B*ZsY|U4H-m z-1YbWe~;?l9~GY+^)39?VWDr=-|vtA|M7bH_r3pWEB^j_J@wS1IjC z>Hlit@6DS~_werHM~8kIA20WN{xtgU{s{g5n(-O;&(FW|f7XLmyQ_Eq?vgnjqTgG| z8U0|omHp=()%)k%TcTt5>*1|8KLsC^C&t?BohW?2E~b87!OU-}^Yf=n-eGd&^Et_B ztN;H0`tmmaj-%VJm+ybJwC?-&(&b<89lxHtzG6?&mh$4epC5(m>mK;kd%jk*`0x6Y z{yGclepSAX5jRlT_lEs_&2O8!*||5*&7V?R`0}3A-)nDG--}hnu- z@iV`^+R>V$;H~UeKgYiQ@b&K--ySO5$sfH{|D(~C%9@?g^Y;AOd8In$+6nGU^JMR@ zpS{TFbDH4StL57|+V{QvzR$<8cKxl$T{>;4XERH#e2NR)SNzrIL&t{;TV-7*?DP-6 ztj^t0{dITA8aQV>xgBo~27! zZfv@I#clGpZZ7lHyB7Q3b-i{p^=q8H|Lu5756;EheLd^`Sw&u~yUN)lyI@_KR-Da- zyEfCmYs+nqxUb*Xxc_$2+4?x!kHXg#3a-_6dEfaaz4@!efg@gjY|2mUKinO$Us))` z_`9vp6;|KyHtxov*l#t{6~niE-ji~Cx_iXp$NmpJUkh1$`N<>yK0)97-sL$u)!yEV zmS2DJ@|*kpQ02#C%?&_Lv>`D8yZf1+w{(b!Ws^X=q zHs(IMseXa)?8R@}1kzm&+}P{MJ#p*Voj11rJ9MQWq3^_b<=Nur*Y!oH#a9crWJNyw z`;=SGFL?W%dWRk_Hl354^_!C$d~Fx)O}^u48pd)R^9wuw=Ek|}PH&l%9WJf+ zKU`5BD)Y(e_&*id&<_{b^ca8A>D`J67} zj-2S@UhGq?vQG;AdOl4l*5!@O6&_R8r@Cjtm!<7EQto@ML;9%4<9!DG7vC{Ne=k&; zdNz^*(Ip<>Wi6Q${qLS-L>U177bGqOyN_OG}|clm%SikLeU!5L;Lve-8}nS zT>Z9kox_G#AJ$Dhm>F@eZyske&&R&Y?e*tO98wdsKEKh76J6r!Ch|~6{yBH*SKpBP zFJ5Jq=0<96TDR(+RpRkLg`8vQTSeE;)KE)4dB1SYzy8;H`BlfR?EBubs(aH;sqoDA zt0g(=sBmSPIZ1d2&)OwNJ#;%(yS-wTi^;A6Gw1 zJ-a9xtDtw{rSSEFwE6XoXL=4Se`9*;-oY9-Nz-s~j*^MnW@OCZk8Au_p2itHeb*ex zEPHJk7RJP=y9LtE1o=3>#f`zb^SKir*p`^8VmRTno_O}%P7v+&>HsMeO48_ACf#bue!y7f$vv~itS>>0W1sfDh& z#)R&b$GD^!U;ORaBjbxx6n-Bt#`Wy@*Y?{?k^wGx_3dEbBuH-FIqS>8Xp&exJCKnP+~1 z|Ha_U0Rl}yIpYIQHl*%D-(Y7KZ^ znp!UL`}kB;;EgR?5_7-7g1>kFLhwl0$8{g*c_hxPVre;c_^JDyZrbgxP^4`7tJ_$}(>K{K{42Ii96EYA+; zv9j9P{ViD_ykm8m!QK!RUPHN3#p??nH~cR=q;t)q zi}U^?KXdB|uT1isx_w=AkDT{NbW4t8sJuQW=7K6`z2nPATvy~aZha8a^Jh*@A#+%H z==|5+v9e-QO_@VEWeu8l-BuNNGeJdL_{qI@!V7A?ojA4OX#Tv_ZK)dB%B_rxwEV){ zC-rh()sOrP1l^mxz&T}^f4YkCH<8{3|8p-soc8ySS>0+@FVwbsiQ+?@J2y+t zwy^2UZaA=aCcBOCu{GyTT=~%Y&F$DCPo49&GFKPR^@?zqdws?dMT1j9lbl^!c{%ml zRgA7lSsSqIn*WngRHfoxs=nUgU#BOy?ucT3a_D0DUEbvizxMk0m@3K0F!roy-P~0( zacXtos+KdR=dKtANX;~qDt=tQq_lTW;9-d`H|#{dJ$!S|g5#9Pv0u~v6!avS{Va^1 zuybuZ1Lp&?((7dj+53+CbkVpcZP6_LWl!N=*9!d$D;TzZy|+bamz4}t_sSywuL>(; zZ$xZRdN^n5nwj4&N`Grzm$}?egg2!_wOV)AZP!gsr3(T~FFjVTmoV!J*t5fHhI0K% z7Qg1T;`7!sl^l^uI;V{%KYv?P}tYhLQrMTF4YCmZ+wR_ZF@A_OoBD(VL8P6y0 zT$7tGm)rWkxM}`$d5yPIj@sYWrGdI%|AcyUIT_VoS8%S(jhU2EUG+opLej_X%ARFi zJ+_6xXM40BT0>#jEuCwNm*aslX+xmaQqicWP^CX~L8jkD9+E$?SgNx$;BURJ%u2GPC1; zhkQtTl{3G$Kl8EKCcTY<=F_d(^p0Ivb4MlVn?`V;(d-8mO?R*x>wJA2MXqV#E zH80GoH|~D(^!d(EuZ2brj|F`4{g(JuGeA=5nCvq_)54|enhY+A%)0Sn?Yd)6W3QJn z%YDDiX|d)~`1h}qy}I@W?0oZjv)<2bp@;7{o@F*%X|q7*Mb!BN6BJMMKUlu@#`Yq= zs|8+%%+{18HraT zF1&46wQu2W4eiaV=OxC@D81XRU{N9dVyE;OU-S6xDY;+YSgfgeF1a}BT7y+b;qg8V ztp?rC2BJ=hR};TAY=MVcwZ47CPm{!$%ICeL64pY?Jzwz3X*a!N;HVVsGAF zeEcPBQQWkv&n=H=FZ$NI=U4I%G2xxjAr`Y&RDM!yY^>DUzHfQ3|J#z&mXl|cI`LPF zXRh!&b3ow3)T*6l-*I}YEt+#iaptuPEEd{~ll26Z73~uwHRtnspGw-A>$z&dh7E=e zXV@2;PR^Vv`R-Lr`pVQ(D>5YJH5~i3=hwA*k;4U=TN=Bx673Vy*}lkH8?1Yvll{*7 z#UtyUjFN8d$t6dodQW=UF~M}=@n*3NyanZ|u7AIT9@-Q4BhY5yIjs$|=U6n)sHwFu z5>9D4%=Dz4OX)(}MByiC!px_{0$#Q0FhApxH8lqj4J>5(MRr+OOKx?S(>+dYL zwd!N>;+-(xq^i8KJZfpm4Lcsst9SH%Ni|OreipuT*~-oSyT7tbTO%u^s`~lTqIA`+ z7s;JQoV>-w?Ju7M^OYLae>K-u3eU~caOf88T;*3&c&2^^|J;fR*()qVnyzm4V-jiB z%@JR`OYQd-w#Q$8NEY0ez3R>LVa@F**>jr}?}-2HxzROwiQ6-CsikiwbMgnz7kpsT ze8=dHtzo{)33Um+O!bog4$?1W&&NwSOIVw$U+awe7qsv3bRIPm@$AF?tFjmTQ{JrU zv-JGMhZ{BHwl1h%Tz^pOP@>Rz8H>CGkJD#8KbcIlJgaiJ*`rRp_`|dp4|IA9&OTYm zz$&Hph)1G$*|r4^>aWv&xF#FLT-_v-d1+aL=achtsq8(Ow|eHTI?{jO$xkuoj|EpO zKITZwn-XB%a_~V||L=eU!86LQsQ~nxgx3K#oEWO>X#_Q`EGmn zjEnQc)N%ufA5%24?(clkwLC!X8i#s)-{U{CbY}~kz8i47c4DZQ`dsI`>x;f;iRu|_ z?P6Y3yDZ_r3mKNy%D}cuGvj2r8=6-e#=ha0e0!VhWZvk#>ra1Py`YMNX|+=40*zqq zm)kb?Z>m`vY+Jl`)r5fUCtas79@w+6p7lYN(mnIe-&K=TazmcAO#9%m@`&ET){Fjq zd|s!c-h{Tsc~01TUd89`mDLAo`K849m6qQd45gMP|pc z1Yf+i@K(U958g@kD~dT>{`Ib{3wyN4XrqKc_1$w`T_@xpToGGy-{t*c4W9EYo(07} z&iU2fO)1-5qsn<}=l{h~yZ8-ux1VS{SK58kapL!k2sxF=g|j|avF?csyLF|maL>*> zizA2UaZ0eyOkeNfvw~+zWWo%yam@>eHn%iejbY~!$6Zu~dXI^S12cKO$UDaHY-E|z~5UV7^Au~&i|O?z5PPjR2G zK2bSw=Pqldf4hPwIL$3Ec$K`j-Sj}fVULs3H>WN%Vw-wX;AqXY&cKFQPP6138Fm#~?GaFT zO}gafjfcKxKGdhmMXmlcQIa`wj$H}M#tW(C%NE4N$XKnuexO>;%+JWS-a0O}%lj^O z?p>E@?A0+re;T>B-Trp*ru6I(c~`k$x2s#qlg>C@E3moqHlg&8?%{(EonIe$c6a}p zWm)qJ6ormvN}fCL+F|n@-p$sMn--J^KYNkK-{Mp8?V=T%p0V`$aK_5pDQl%`mP?LyYm+@hd)`!RUp?~ z&enQ7c|xXU)cu5KVe^#rVt&M33S-LM@%>?M+hfgHb%%Z*5EScQ&>%7Iz`?cY^LQV> zytnM@vZ=!UzM5{%3bKN>m*~bbknwTDJh?z6@>9K$R(&lSE?tkYhIE$&)XvWbw zAsqWx2A_L&g7a9}T!|Cai}JF(GD3_ltPx$fJ9}zZy;8F4;^${3CfNy@NWNrqaA?XY zofmxJv{u@cjoTI61$_ACC|@zJPw;!pD!Xh)+rG4Bl|}Olmi=OAJA2me#9p1bpN%&8 zBr&DBsXv~xYI4twlm|TiiN<{o4p~mq`zgbx{7T)<=H0ne8*$aBou9SeRm}5OE^*ne z&ivYrq11STxIw+;=H0R(((ivYI+>K}Mzqd6rr^Gt%|qQU?CJ}l#Y@7L3+Mj}&b!0* zGO6~=%<#uSM_9X#x9k(vIQa3;itRn0Y`F|3T>f<7@<)f?E&BT(1z4r#zrGOs@h?MX z)0CR{BEJh?D=tZjE#4V-x~O5#i-N-z8k-&3xAaK23hX-Hd*yuhtNKg)_a-R@KFoAA zzZ|#xQd4?)5O=;m|D7LROZ(V==U-#|ppodo|w_jd5a{L=BKq`yvI3v*|SMxLD=IX_qR~;6L>SHTHZT4&TA2k0xx8$@!y;z7iXS0k7i}f|FuDOSdT$%hFgcDCXs{LQ*EXdFz zeW#yaY6lL$gy~uvy*ko(jR5(HaW&t`pF!=^lysXM&rfH&+{zk`qKXQ zV^m1z|A@)gLOBnVtX04L-tTb86APJjjn5~2S>a{8qh$u$^n*!jd$kjDmM)C7U6oQF zBL7yvcSFe{IiFoJeG7xDc(b%7GqQa3w@B3XlUsV}f2EAo{)xY4-CC|Nb%j(m*9?n- zuqNlu0Pm$_>}|q9CnG>*Khl@+-LF*5|ml|xOMS%k0xENY$k>)jNg~9 zdyqMQM&9OqA#*ZUvD~{BHa*`9KhKaQ&JOQ zda*ZaMckKpM{jOY5ED-4{QkqgyXtzH+r*ry^Cvh(>RF+C?3e*4gNq0l!OzpXM4(|&;Q0&CM@yXFnebI_{(^kk8 zEn-WlxwLC?mC@!`pZ~0Pl$?AtZOf$2eP{lyZ9f0mRP$Cuu%Vc*V5wBCL`duV?@M$4 zeEofUKjZK8;)=;Z^5v?AdWM=@3JUtpMX8A;nfZAxi6yBD8ZK5w21W)3W=2MaW`>qV kli$h5G8>v0PWDsS$ZTX_HknD$o6FeHj7wG3)!&T^0L(R?jsO4v delta 5891 zcmX@;eBEiodJZOYv&jz_mFwrmo?iFVqUg@|@Hz4sGnf1jN|1Q5D!TmP+{RmyaW`k& z)p@gY|Ggzg&h+>MS*7Zp5 z2hBO{{@Zw){pVetZ<(R@ z9{*WX7rd*;X3w_xBNnmYOm`*gg#U`YP`wlO@Z0mH^)tg3=cVp(iH?_l8k;OtuCqbw z@BQ1s-{l`auix{JXJ5|FegD2po%SW7_Eg1Lll&_i^9|kiuD|@W&wQ<1bV2FH^?A3S zdIwFcJ!_v~|NHudS#r?}t>3*p{Caa@>#m)r^c!BQ=s)<`r#tc`>rVR%;R%g&VT9s!drNsugY+HYQ@Hmz%{l9?^~RxugKq8S$jNM zK=Asqggx^QTI9*)7RZ-NI4-{bvgUcR#b>*BvyPcBd!{H?S3ir-{l#mk_kFtA+@A|p zPcZN7PQ3p9Q}GS`%2+1tuAIMTwJs;gub(=vcSo*Io5MxbG=o+Aoy9 z?pyV~PO&*6={A23oq4(Mjg|R|071*Nxd%C}tq(|Q&{>;PxBN+Qi>33K%37PHm0k_m zCs`i-kSltw;W#yB*$bBvz5nJF)y@`s0t+IW{_+d>{QvQ_+G#0ot;u%(+@nT|E*T3v z+c0$z)4C%YS)PS#zk9qrA#Cd7sqFi2v+gk2-sTd>SI<+`G^fep%-+Y`d~XuXU2lG} zc(g0!;VF;TdsJgqH3{rJcJ4K2@37ML0s0?zU|ta z6DCjE+vOW@JN$34-bvM!0?`-Zw$xVyM<%}c>9lzNzrQt;lk8e1*k2R5S6aN_tIKBI zNl*MW_Rh|KdHVJy=TqHZ_dfE|ZkiyrV1{@?H|J!zgkar&Z_Rrqp6YM9xcBv@3!a=H)Rw=Dh->lZ!Dq<9Q^@pK*Z5Kz!5#<6=U9Y$D4;+Hu z*GES@QWZ2i(IHmS)RY_%;T^{AZ@+C(Meuxq!#%#pQj%H)`T%>))BPhqHFYzGMAsWfNzGI~y`j zc_ozpG4f2v+gzEHUw%s_3yYr5(m0{rmwV^`^x)9?e?M)!*Q9C~+G*rV%z3v~%ZO2L z#`(nppO-rGN;_@7D4kH{u6611-g4f&LtKGVf}g(DYmKfBGI6{lBFp;b$;v)E+lMb4 zU#M-`dLwr0N0AKyPIpo_in{LE6B*>NYWMn9;VqljXPb#K9b9z#uqdnNj`@?C7TuU6 zz{=}$)b@PY(PQ-toOUb3{kAk7(XBc8{nLHltufJ;l2v-GW(4m@j}c>Xy0>4Ig?STC z!i$LuGX8%)eqttv*iHBIhbB$xzn%D0ab?VX_xIwzzw3k-3%=$%c0<4?=CFtFRP)tlrJ3Bk`RjHaYjjuQhwby^T5aD2=)btTqu^c6)>*qB2o*9u z-hEh`;iu}m6|bFAq^e(>UUufI(DAQXo{8xTKE-B!WDM6j+r(rVvvYN?xJ|<87A;-P z-3|8~!%i>Op5e;A{9vvNi-XI?&J)MLV@#uM8F}cJO|>cD>iDoY}KpbJcg( z%f5aTzQ|u)zOY-_ay^stflH~rDUIvT>+QMKRI;7;Cl;J$tES zW2pWX_mBri+x|B1pYLIL_e0!U-O#Y;iK3=^`B z|64UrJ*P>Oz2vcVXp;G}>W%fe^_LERv&_ug`k%~~+yj@C?#W!<_C0Y_)z_8A z$Q3*O=Hk;yPnwdn&im{)F-~Pyz8ZDq@ceD9zpYBcOE+t!b0)q$_Gobudx&lqdZ}9Hr_WLT8pi_P%?GzX*V-@B za{OiJeSuSZ&%TyCJSnU6ZtK!m$$!G$PiyZ#ab4Zc>DQ@mePM-T{d;XVIQMzM~dleqyKZ)iEKPrE50|5JCV^@REbX^CFF+-l4HSaW1} z8!}w9B4<;!Q(T$4mO0Cpd2!)A}b$k@n<7eWsSUnvdndh&PK z#t*{I*^URNE4>L>%d>KY*5`{4h0i>{ku%Z$Q6rbu#n)=p+HK43F-+aO=S9%*`W=TS zJe>b!M`C3-e&qO`;{p#}9P3l^|DU(k5O{NeQZTCOV@`<|5LpX5?s zn*<|M=VID}$%IT)g_ST_(SoQ&G!VTDM8++m=89nH{@xv*&WhZ}rp zrcUL@*BoVbn60qIO;`Kv-lcm=3%i=$l)gB9=;kaBPt_mw>-45OEX>JTRh-~_Zd1Pe z&DwVd&)D9|*(9v|^S0dD6Nd7Y|9TcPOGsB5AFz2h`{IP5C33Z)D$-5mG=FR7py0OS9>0iJ0#>5zv-2oX=s|xU9R?&srCx@BQ=|Q1+3bPhDb9`XY{BFj&pFae=4N z-x5X|o*z|+lx=bVQctXRx?J6<<7%NvQde57-^ZC)cua>!eY%4MP&NgY< zWp<6p%Wg;oD6N0Nm9uujhMvYHKDW0vz38ayTICZG(30xC;*S4zmwj5z59^IKuM+-0 zk0Jk^!{v=)ZWB*4Tj{U*zeZ}8y3gXTi_R>%);Bw}wmqn_S&wo3-iu%FpLiBNqiyH* z75YnqWZw5&`yRPAb9H*DoIpg$)TtS{|0WLcpR6b+f zlHz>OVfB?atM;AhTykS!J97>%FGu>)S*bF`hD+3DP1?QdV)oj#hK#S|m%o-R2)ORu zuuuuKq`7sMA2hk(rCI)Ce)7fO>ug_^FPj^Euk_0tF~Lh&DapsGemJSJs^&7w zUsBFEQ$Hi5I$hvDi|rfhS9}*sUWu7SRu*$FnSFJ}*-XCF6Mh{NyRxQ=Qg)I;plauJO{>LpRnm9u8UC=j3>eee=(aj!lPWWysClE!sP8=J#dVsT=!S<;vDf zkU4kyJJ(~y84FcA?n>96{=*qN;m*yUH0u8|IP%wtEzenN zmVPkFX<^d!xK+JPD$ZN4?Asn7=XRxu^GTZT48B(h8G(my9^o?HWwAJ|xL#;Ms>JFQ z#mO%=9own;so)Juh^D()v`6a3N3YW=ZupiCU5lvo2nk^4`#M zBGXcjxRWJTUSaKXG>$%>{n*Hg!8t_VVYAzghbf-nzsm~umvc_uTjx31vF5dkvvz>h%+Wa0|3HQ%B_h#|;La*mPN)9wEb3HzZ+f^;j^XJm6(aWRJV2f_|Al9=O3H+ z<5R9}`DI^vWp+vRD~EN#cUn?D1kIT>;ZwlHR*DE-!ki3|6>86-5+%f8xA%VzjU}0l|Q3Piu?W&ti~VC*_{n(SJzncfu2w>x-ExQ`5rk{^NFHEve*RygX*k z*{7=d8X0}5jaEOkoo8QK-}{b7xVYxEw5YGg6RU8qrK$F;zNvGJH`&B0&bhOl|8xDb zo4c4Q8eYF(HV)r#MAf4C`LlGJ7w&qests2-PIwuo8&LXm{VgWt$d3gfhs_FH3hGwv zP>d9{J#(0O{W7*gkNF-p7`l3YSSk6aa@Ij_evxygH+Dp*Ts{0Oxq@eV4$rH1AxCa+ zGTip{(xYVU@R&<86nVnntLuNG8-`~i1_0PU@Df6qY z%@DYp&SoRWkXbEvq|5kpPvM2;YKcfSZ*KTo2zCO zRn1a7DcE&+qU-lNIo@{?pWcx>#Q9VBL-;S7Dc;Ftp3|0hh0O01=e0ln!fSe(zj~{8 z=)w=~F|rZ$RyW&1zPz8^!0Dc(7HTi@a?b0sp4opspOe0A}J7_*C29MtZ}uAFJ1Kd$!fjSXd{u@5_eDZ|hr1 zmBm~pN~~zuFY+P!$=}{P6OT@P5dFG&E5FRj0JCLM2M?K)XfC%65-`}iz`}y}XoY0V+q_AvU3_i7EZ%F2Qm^W^IPYs>^j<*3u968!YrJFHMl2`_tdOF>pHfT_&dAAx%0TIegkPR~mPE zUfEc?`=oNktR=HV-A``H3x2nKQlkA=OT&)4J2Fm7S_j#MEtC21=KAu-oDV!K>ep;K z=ld&8BJ;wcvqv`u*BlS?nX#oUJy$hTza{yT-P)-EbM||=hGl2nXOHZP4NnizZ`_c; zbxU1b`1+I^lKZYqj2AZJ>E6)zrEcF*SuwKyLx8WdMR!ai9c%FT0K*oKPsn9quEt~HKSvd=F9e^mvecR z_&t~Xl=S*RS^b))H7|_CjZ+$bCs}TeZd-r5IJ@`9g>`CQLe3>FYt3T1B(z9IL2O!x zOrOI^?fR{YppEjNkwNhy9&>Zx>G9 zBVVp+pl6`TrJ$hiT$Gwvl9`|9l30?epy6U=WMF7%U}|V)Xkuz&GC4vamf67Abn*^` Yjm(Ay29t9Xy}67G4Y*WQUH#p-0DwGQsQ>@~ diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin index a4d97447..ddb2a438 100644 --- a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin @@ -4,695 +4,695 @@ - - + + - - -

-
-

- - with - a - plain + + +

+
+

+ + with + a + plain face, - on - the + on + the throne of England; - - there - were + + there + were a - king - with - a - large + king + with + a + large jaw and - a - queen + a + queen - - with - a - fair - face, + + with + a + fair + face, on - the - throne - of - France. + the + throne + of + France. In both - - countries - it + + countries + it was - clearer + clearer than - crystal + crystal to - the - lords + the + lords - - of - the - State - preserves - of - loaves + + of + the + State + preserves + of + loaves and - fishes, + fishes, that - - things - in + + things + in general - were - settled + were + settled for - ever. + ever.

-
-

- - It - was - the - year - of +

+

+ + It + was + the + year + of Our - Lord - one - thousand + Lord + one + thousand - - seven - hundred - and - seventy-five. + + seven + hundred + and + seventy-five. Spiritual - reve- + reve- - - lations - were - conceded + + lations + were + conceded to - England - at - that + England + at + that - - favoured - period, - as + + favoured + period, + as at - this. - Mrs. + this. + Mrs. Southcott had - - recently - attained - her + + recently + attained + her five-and-twentieth blessed - - birthday, - of - whom - a - prophetic + + birthday, + of + whom + a + prophetic private in the Life - - Guards - had - heralded - the - sublime - appearance - by + + Guards + had + heralded + the + sublime + appearance + by - - announcing - that - arrangements - were + + announcing + that + arrangements + were made - for - the + for + the - - swallowing - up - of - London + + swallowing + up + of + London and Westminster. - - Even - the - Cock-lane - ghost - had - been + + Even + the + Cock-lane + ghost + had + been laid - only + only a - - round - dozen - of - years, - after - rapping + + round + dozen + of + years, + after + rapping out its - mes- + mes- - - sages, - as - the - spirits - of - this + + sages, + as + the + spirits + of + this very year - last + last past - - (supernaturally - deficient - in - originality) + + (supernaturally + deficient + in + originality) rapped - - out - theirs. - Mere - messages - in + + out + theirs. + Mere + messages + in the earthly - order + order of - events - had - lately - come - to + events + had + lately + come + to the English - Crown - and + Crown + and - - People, - from - a - congress - of + + People, + from + a + congress + of British - subjects + subjects in - - America: - which, - strange - to - relate, + + America: + which, + strange + to + relate, have - proved + proved - - more - important - to - the - human - race + + more + important + to + the + human + race than - any + any com- - - munications - yet - received - through + + munications + yet + received + through any of the - - chickens - of - the - Cock-lane - brood. + + chickens + of + the + Cock-lane + brood.

-
-

- - France, - less - favoured - on - the - whole - as +

+

+ + France, + less + favoured + on + the + whole + as to - mat- + mat- - - ters - spiritual - than - her + + ters + spiritual + than + her sister - of - the - shield - and - tri- + of + the + shield + and + tri- - - dent, - rolled - with - exceeding - smoothness + + dent, + rolled + with + exceeding + smoothness down - - hill, + + hill, making - paper - money - and - spending - it. + paper + money + and + spending + it. Under - - the - guidance - of + + the + guidance + of her - Christian - pastors, + Christian + pastors, she enter- - - tained - herself, - besides, - with + + tained + herself, + besides, + with such - humane + humane - - achievements - as - sentencing - a - youth - to + + achievements + as + sentencing + a + youth + to have - his + his

-
-

- - hands - cut +

+

+ + hands + cut off, - his - tongue + his + tongue torn - out - with - pincers, + out + with + pincers, - - and - his - body - burned - alive, - because - he - had - not + + and + his + body + burned + alive, + because + he + had + not - + kneeled - down - in - the + down + in + the rain - to + to do - honour - to - a - dirty + honour + to + a + dirty - + procession of - monks + monks which passed - within - his + within + his - - view, - at + + view, + at a - distance - of - some - fifty - or - sixty - yards. - It + distance + of + some + fifty + or + sixty + yards. + It - - is + + is likely enough - that, - rooted - in - the - woods - of + that, + rooted + in + the + woods + of France and Norway, - there - were - growing - trees, + there + were + growing + trees, - - when - that + + when + that sufferer - was - put - to - death, - already + was + put + to + death, + already - - marked - by - the + + marked + by + the Woodman, - Fate, - to - come - down + Fate, + to + come + down - + and be - sawn + sawn into - boards, - to - make - a + boards, + to + make + a certain - mov- + mov- - + able - framework - with - a + framework + with + a sack and - a - knife - in - it, - ter- + a + knife + in + it, + ter- - - rible + + rible in history. It - is - likely - enough - that - in - the + is + likely + enough + that + in + the - + rough - outhouses - of - some - tillers - of - the - heavy + outhouses + of + some + tillers + of + the + heavy - + lands - adjacent + adjacent to - Paris, - there - were - sheltered + Paris, + there + were + sheltered - + from the weather - that - very - day, - rude - carts, + that + very + day, + rude + carts, - - bespattered - with - rustic - mire, - snuffed - about - by + + bespattered + with + rustic + mire, + snuffed + about + by - + pigs, and - roosted + roosted in - by - poultry, - which - the + by + poultry, + which + the - - Farmer, - Death, + + Farmer, + Death, had already - set - apart - to - be - his + set + apart + to + be + his - - tumbrils + + tumbrils of - the - Revolution. - But - that - Woodman + the + Revolution. + But + that + Woodman and that - Farmer, - though - they - work - unceasingly, + Farmer, + though + they + work + unceasingly, - - work - silently, - and - no - one - heard - them - as - they + + work + silently, + and + no + one + heard + them + as + they - + went - about - with - muffled - tread: - the - rather, - foras- + about + with + muffled + tread: + the + rather, + foras- - + much - as - to - entertain - any - suspicion - that - they + as + to + entertain + any + suspicion + that + they - + were - awake, - was - to - be + awake, + was + to + be atheistical - and - traitorous. + and + traitorous.

-
-

- +

+

+ In England, - there - was - scarcely - an - amount - of + there + was + scarcely + an + amount + of - - order + + order and - protection - to - justify - much - national + protection + to + justify + much + national - + boasting. Daring - burglaries - by - armed - men, - and + burglaries + by + armed + men, + and - + highway - robberies, - took - place - in - the - capital + robberies, + took + place + in + the + capital - + itself - every - night; - families - were - publicly - cau- + every + night; + families + were + publicly + cau- - + tioned not to go - out - of - town - without - removing + out + of + town + without + removing - - their + + their furniture - to - upholsterers' - warehouses - for + to + upholsterers' + warehouses + for - - security; + + security; the highwayman - in - the - dark - was - a - City + in + the + dark + was + a + City - - tradesman - in + + tradesman + in the light, - and, - being - recognised - and + and, + being + recognised + and

diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin index 3dc65020979a474a3659ffd6bff360b9b33b2b0b..2e797b57553a3ed0daed6a9796578fc932e941b9 100644 GIT binary patch delta 5588 zcmexkx6EM!69>16o{6b~nW3?sx#479jzvr+rjs{vDA#X|J-zPfjZ&D>YsC@rp@7|A~=ai86$ZJN44Q*_0JSbATDfxa#TIeY1}<+n6cJAMw^zZz=y><2fUVq)c-dq+^)74?5ATz@pi&);wK<9qVMy^nrquca;R-X&sarDQwO~sp^-hO>NQ#i43+wY7!x1?{>C+9^w zP3zq?asBtSS!=$1yJA@6JEH?|qEqmGo^#^Ndd=Ov_z<`C*sXK670Qvzt%miOs&gz}7g#ujW~i zt#(Fa%(ty?r?$j=lX_aW*@F3`#_EZUwKdlB>yxWLcz=3%$2rDq_vWt`9GlkZW~Tj2 z>-;G4{N1#xjTwEaKfZ5j7N1rBqJy`)U;7ou%*M9{%$4B>^Mohe6Hm`r`FrE5+s*uC zoEN`dx7^!$ci%0&wAJR8yJ{~n9k}VTdxym~hU_y&o1d?`r&XNGN& zNguxJ5fNrm`t8rEVA1~dI=4NRyKemO^`7o-d;bf^bj*It3ugbw=KlWb?1+*w(ffN_ zFNew<-=J3~p8dyS{pnk4WP-iQ?{G(N%`V@|b>ZRz*_$t|(oMVT(~{rY?6Y6?-RgAt zn)jt!%4e-%dT#yqq4-bj{Xa6pbZ39eyq$E%YFmB$a*;U`k^&pH-Y^Mhi2nYqqVV3U z?#~LVJ}oauxHgf|<22v4XALr6BXX_pNtl@S@Os_J)7_#xsqd{wO62Nqa%qzDEZYS> zePc1QGF|zrT>DM(wPmJTf^XD+Jl4RrIXdi|b@oRoi4`ZFEvgr*vACa><8UGW4nvvS zN!{6*71`b!j^B`pI;wQHcjL`pIaeFbB}!ag<99<`Ap5FFdEVrnmdDE@-ey?rxqGbt zskg+Wo{O)9mX=jsnBsfm+x@yfZwzbyZTb_N$A5Ibr^kH7j1<0x=ztOiV=)n%fAwc{ zw{^#tewnsoZK(bSzg7PGb~U<0UE5F#X`&XD77fmT{`=FxQuf+0CBAc&}l)$g?-cRGz+` zb*OIVnWFwr5A)|IY-(ys{!|`%Tysw8M}}n!>)n{xE^0qJu%2~w#{J;EjC=nF&)TcQ z_3r76+!cvU)@N_56nt}PN5sT67j|wmy)~Ql&S&Y{y2=`@FZ3;*3OKvxu(4?u)GI zxas+y^TDk8?ybgCCF1X#WIbCjExl)Ngx~pny&IP9So!WVi}@!>v4`n7I|~y+TixF; z-Cvw^vf|Ahi%^L>bCb0C7laD?HTw0K{C5Am^JvCGRKi68ycCG6P6-gkZR@Oc?h^1n3hsqB*3 z_pGjdd#vd7VvR$oW=crPgo%b%6+_N)#WdcOJkY{slwZ~?{kO2LR^a@U)la)4ota;5 z4rQ+qo?RWXcvsJ?TBT#I^(;!aTF=Gpe>BHben#+m{>tR5?x|~wqJJ%|Fx_ut$#?1W zyBAeUw%2_O?taTwo%W+Ez^9lo)MHZm%6nIt1t-`}eEhxZ=BmdPe}4Nt_?U8G9ph^G zwm{XVXQd2h`G$2X=x|SJn)r1oEE}yKb^lY=|f`4nWbAH@`S@hzP`?0*6dJp+12{%#!J8PR+-Bx?%Ube||TQ+wnKXM4z z)O%)kuttoe`oCQ#eD}88tUvUr;d^rVgr!^4I!!8$y+0~dx~FwZRD%G^FP)&va+;?E zmiRsoJiX1l-2P$u)uoZ17f1T?+0C+_|WG0POXqt z42leE=ZCz$!Te77%>?(5H;a!f3c7Lgjr^;V!TKk6NpC(dbNWHiRg8(pr3&1%91py; zJInAqS2aDsPkw&b-7PEHnD|!z+VR_C^)2Dbx9^>F9p5aHU*K-d#gydGxxB&s-5*9T z*NxB5>=KFj6?V`tb5Xs}nzDkj;D8p!O9C!UIS934_pwp-)Ic=fCJE1xI z<$SoW-!_~*!6MuJjky0LkDD3IOAEv<_=wEv(7Zi!&b^bpvXc%cKl88J#C{}yj*h@` zb>&GRVM`WfRUUfo@Mro{HR%*lmjnGvN^+((@6?xbNJcwMxTIyyo_S3Ab?~}Z*0M)EB`)j~<&m%KyZqy3zQ(c(T&v0^2O8?k zP2#I6n=Zw0^8B7$^+|49zc{h{v^scp@8j6Lj!b16jdmVNDq!i@nJ@T7pkDE$o)14) zSGQq9ih%X|Oe;0DYJcUJ%hLnzNb1-3Mz6ilcK&l*UE|{j(?5a-?%nnLvGL$)tNBG1 zsXH8$H(G9TYn9ZQyj_z|X!Tl!iIHe+OCm;4sOnUsAO@yW2CYpDKjY@%9)FnMn ztt%Tpul=Zg=fG*1xP6|bo0)P2|4p-yTJ&=@kGs?htyx~4WnEdDH_g^KGvBBVoIEoBb zcb&Jk>7+&p=E4ZmNzboWo@mTCDx9}I{xi2*+%@Y}m7!~#Y?Sx|pEhlsIDO+4;aB-; zHY~MW?rV-4GhPi7JRuo4GdQNYdq+*%wU|>SZgZD?k?gLST+2`yJi|>O<%YK6H+QzK zO){UPl%37)8r3t!GTq#-$Mcn#*5!FKF4G z&ZM}eS5RO!&y)oFV!gw648s&`jiQWQwGwgHHWor%XLx~JNyL^qi(UcFIg<_ecS*?sr- z=)6r@;i_rQaj5f}#*0mN6d23vCmS~{`lDm1FEAlw)$1hg+kftscu%ufYrN&iEvwL9 z{8cu~EQNLQ#IFiRHFlcv{&3F+NdWMhA@SFR#hI%+}4tXZ*-ToO1p86Le6S^l7H^Y!n$8_cE%Ea*N` zVEjexv=OUH`8lyO8h7od&E!{Y(llR;cadA#Nysc`kSJw&i1gE;D&v{1~AH6G^fAm(@hBZQaYY!gu zP-!oFdQa8ly~)ki)ht~B+;^_(e_P@bHraQcveTh)X8JA?t4y{ zhbuX~zH05gW_NW1C;!iw@OrClLQ}8Y+FGvi!cXvPNcx;(UEAwjZ-=-3z3yATH01mV zPKGx#{jM9$$?IM0wYk8kchmb-Oi9`c?|e@co+=jcN$0Y}1shYX^-8v)$qZVLrfwIw z@ty16-3Ma#cB(yDk#6DIU1!A6CKbeVr$_O~oRt!l@|sfFc}LnLnRw-GTdJ$0WVVG~ zdYHG&bV}J>)fu7{2CG>*>es%!RU9|@$;Gat%VZAE%Dc8n;YQZ=d&`e~dVfb{9yh!9 zRrN~FX$En#-poo~o7i+vQRqR>?0^87<|S?2TimjiT2`36TJWG*Z(U{6ZUOG%(-E2P zww`o0+Z%uFam|5)b7$;wcDQN8ZGZa5(vlAwz1q^5H@_&^Emj^UrJijcxvf;KUXbnL zqpVL84P5daz?6zA@$Un?$b5rU8O?y!mzR0Zt9EB$Wg?E+qEz7^!$|q+hwd~kQ zA*GIYSF^6?_p=|;E3}!hLV#`OLQlQcM3wrYigbN0>yCZ_m93=C+u#y2man@o0U~A)$tMcS}})-X;atgRrh26N^MPqwZgWUlDON7N zYvJ&1ZOzH$>s~M$-OhT-DR#w1^1-JBuEi4N;f|kY{&{hrKHC1BpV`~4hhm3IT+MPe zIjvym&ritQ_-N5F-U(l7nM$4cUM0?*_d?y{(Wyzd!e)L{nDQxS!mSt$)&w7(n!_yl z8#(&w-R<8mn;mHxyN|DSn(=mtC(T$I-ch~hCljF-w zSmJIUp&S%>mSafab=sLk*$J@8HAqHMWg@3!8aN8)W}=T^zy zczP^I*HGv~PWHN4_Fl|SIr4g&9{#>~{KjF13I0D6WLBjz-OT#g5MBD_!0p0*?IT%Q z23pU%BVV5E`*_rR9!pBvbdiS7N(o=2g3|OQ_#`Uwrlv*H<3O zYfe82Z_s*hAminkBffUM%+72HyAH^k1g`m~m0Enj{N5|o?9S|?ykD$DBX2b(*+~Zd zEcmyKL(q2S*Uh(f>vV6>e)j8zi8=Gs`ph#HJI=1Sl5lm^%a3Xw&N$XQoV30|pUq_( zpQhi(`}~ZIO%vjG-V^lJ-RgPd@dB>(-xFHaPFE8-uJ%bH=}75=($mN1MoCTZ-KOZQ z!nHn2MDt-=Wa`YB8E$g=9PX#OlNW6d&TO7zbcGdNOHq$$Np3|h3^>272I%m&Q7rx-&gAX=ZS2{Rn zuv*s~IQKT-?B4Y>Kk>J84Gh)!>RzR+S>_~PHSI`(S^=@#MHG3&fzSfzL%~Yj#0i-A6@?Z>ov}w zpCxD9IvN{Lc8#2(`OWbMc^<32 z?~#5w|NY{*-RU2W??^F}c=gj(|9eXP8#O;qf$7tD^uB5eKbSQAc-~SSK}o-ZNx|it z-&|p0+atQBNNt)0-_-fXXRT4@5!f!j^-#bvWX>b!V;pI%LsXSr0!(66@V_=at!r_L|x zvaf%0x6wMQ7O`rH;PSg-8jd-CN~Zl~oN>#%e#SZH zqA6{Qdn?1A8ib_Szq|H7qe(tO|JZNM#}=Hw>gS$TRbFEH|3N|58`VvoSGZm)Jo;x| zrhGj=H7}*Oq$o9Ua=mOmv!R~hr==95L_VwlX0Cl|?WVlua! TtS0ZnWocl_rK;-c@5TiH@@U0J delta 5427 zcmZ4H@W*Zg69>1Eo`HdanIVXp%*(Ne$;e{zMh@lrwXtW{-P^%?_qpai@s8E@-ev(J z9`DY^mV_(pUbI?l^K94MZ!U*_6FuqdwP~(nAWze|ADdPG$;kzs+8>|FuOI$@`rqgC z|Nr^>b-Vuk`nt~b^Z!@={2c#tdiiUi&)IiR2g(2b`TMp1{@T*`*Yzv?|9-o5`O)<+ zzkc4oem}mxY@NvS`cHBHeZHQ5tG~W}Pu;(-qJJlC|Fv%Z|Ejl%oq3OsZ_e-kvEj|n z&l7WFi}*gdh09l4%l)k>{^yt{TT5-axQoNm&p}xp8Pqvakaox zb@>m^9!#5aE9sPw`Hg!!*FIkRB_oQ-cGqL>mnVf@JpHg*WSR@p&FOpC_ty9MINf|= zsQYZoHghq7b07b|4uAap+h_gt>kqzfzqIey+nB`KlLf}PGgrTOZPF9$Q@V0RuTkB+ zud~^Am7cz}WZ(L3XV+TwxpL8p6W^`5u70ewzu)fNPy77RXxDieN!H~-1?GLBukJ28 z`!)JE@A}fvNr#Shc^wJy+7YwD+j+Le@p|#={yXRY|HHs~yqMMZ;FFmTKb+aMK>Fm> zvmYPb4ao0vd0zG0Do*@G=tS}O-kjxH?c)<&Zk#14I{)DIQ&s#5 z2X5TEP`~K#$7RJEtBY^lxVNiIbKUN5CoX#&QQ_{+iodoav-#79QxiM|qgTjH|GMbj ztf!OquT8K%*}QD}3-(*zB<5^yT~t>(dCN+ES$+NEVoAThe)Qh7Q>8GYJL~%M#qx{f zj3(TEA%DdA*Mm)psw8ZMv^3~rse+j0W-0>G{SZue{ptpaH|0*vTuGIk_+F91DvF2a`L_SotY3Lj zHCO%K<{!rDsXIN7>ho-|n*9Bv_oRiLtM)E;<%|gQtU7L>9{25tvl*kT>3gxl`&XED zJ&E$U9mu+3$)Uc6d$~M1yY6T>)|~(QFH^DbL9_a`J7VW6S2=ALZI#%1jrFK-z0n#$ zj+&P~_fy_YId*|7yu9X3Vnc){tEu_hk6EE-->p_UY<;9ZXH(6@hOo^ScD+t`%^_>a z7Mp7NZdcVwqxLJ6o{svOdz5e7=0BY+q^G{+{ezq=#icK&@$PZco9>eEeL-C^WJj{R7;)ZOgu z25;F#tG9l+7r4h3A@;;6(FlhxroYNL#UdSB6pCaoNc`3Oo73*KbXn-ZStgEc zKb<#4+^qLJvU`msUkm$$U_gRa|NiUYzeKO3r zE9z-7^H7{lPlsFRW|3lJj^7t+@|btZ8l`Y(pIrDogzw>#`Y#>xgC*Q7cwY#)UyP18 zxHQL9l!YPx?OaJYyT1=uTrN8)iF#Vc$_8Cbx~KN?@-LZ!nBDL11~*=G)H>i(uFj+% zuvE+W(yPo}+A}^Y_LpCAb-a|ZHSx0NgVe$mT=Onn;}Xrkn8hpL$oFwQ=fgh3%Br9I z%e6jzPM-GRR3+1jcFFoxS;zr%|)#wZWmbfI3kz2J2_OAC?F#>pEkxZi8M| ztF^XeYyGBFl|OrqEXrNl;3%Y_$9z-VKJ>@T?ROYH$GezotnTY#-tKr~4%42b-C-;K(3Gycbh7ltanV@*ii{!THgzehW-R_YT|0w>eN`s4HONB?r3 zaa>MQrPX)a1U9wNwbWxUX_q|0`A_0{jUiEICq^V%A z@<#WX!fP6qJ^P!oMZd2;6vo_W@N7ejk$YeAdd@i;Zzlgt+a4_^x?cN-#jiHKj=2xx ze%xhdscvE4c%AX;`lHT@{mgs6e(tJ&xiVv&rO(mQwiSC1rd<#hEowNx?0xxrj@f3{*5MITAB;@1gl!FnrT?&PiR_8p3NECjvNNc~mUluR- zf_ej`oTN|3;v~|yIyEh1O|Mk^{oU~G+KE{YZzL<>$WKSX)JesD2v(stoo^QqQ6GB-BsF{7Q08x@!-AeFMr;#Jy8F!-0ZSBX(~S&pptcYUvMR8G6_Ac|qm z%w0Ji+g&+}{}nzHa|!!&W9>=jg4+k$s^q_1;+VWAd+Kv`pKJ1~K4+v<$! znqjq|+;M`K;(PH8OJ@}-JiUG7V&D$XUha&uYoDwM2{ey#oUyvQXGUsm(5+_Iv&Ws% zma1=>aMbYQ=a=;jGQZ@Ou2g!lbj<>RRZCYC-qU2+Wx7iw=&M%as*E*zxvJ|O6ixcI zg0h)X%$)fHda*;w8&ZE>JP{r%Tp_U=5#&XSh@ zxwKBjmcybvpr5&Ky+^HlY2=KWDH}HWRhR}@T~fPX{^@bw&Q}kV?_2AYEmi6g2uVcQ?MXVCv5q zr4i*1-)`>{eciAHJQ2Lqk{8GD&pUms-Z~46AkSNcV-I*yOe0nbn?#9HncMD(XKUTlm zCtmQUv4ClEfUAeiTW0kcA9*hSQ#p2y)A7pIc`e@S9oHmF6?mLF@W*}Lj(=P$`C}G} z3bnJH`p&#fD?PS9>Iv83!=aDo3V8lA^IXDP6~}mtX~&klubk)j{n<`vC<)x}kqccd zCAM;*S=;4Hjt`>lo>`T>%PBS_NlmkOWgnA5{p(E?+MmifsxKlDFO z5={EZC6rm)-Rf1yYW)0@Z=7OIsUts zZBARtrfu+Tm-fE4sSIL<@AdWDaivbk6L9|)PK3RQ<2|JZ{yt@ zjVnTbm3!tKs(kdsB5BHRVYjMqrp`HXnFlzdjkBG3Y9`${c4NhRwz&q9wu_IIF6-do zyua1ZW}eIoEfJM9VFK~>Li0|W$SNrGa=nNyidlB3{-eR!W!j!sLk~|Z=wUwlJ;mzv zw33!R^~vT3nX@nSrv>P$bxQB3S}yl=({VQcjCW=855%$e<~SBbxV~K3qE}^f-eKYt zQU1*-%xc|opQR5h{#`h?Hc0t`30F{z`%7Nu-6s!!zVGNU#VBN3<3qvq-1Q3-o!0!& zwv(PUtHe5AN6o^rbz_TzdG_sFFWrAm5N+p9Y>99|EMfib>UWXNeo|paoGjGbwhZi$8 zPZyRqUp#Tyf~GjPnGegKsIF7A5&v#g*gGqep|_tY^LT~hp%cf_7W*8RlzsWpFrwpI zPLj3R=SK=M^Hr;&-YNu7zF}O}6u-A^(yI?ASN`2$pqu5%vqi+>S!k?fR8@ZYp;@cq zmVKyCjh&<9V5!uzls~p<)|}waZLhX0lW3L~mOrAZ@t!H9*tYV+j{Cf#-wJK6&M9K@ z3rq5xk#^(szO8*q$z@@SIv%uDAK2{tIr(s%@=I+C%lRp8Yu-P3;$y9~D(v*%lIJDs z`m5J$_Vkhne5I!Ok)eOBtpAw`F|P}0pV_R%Z{ANZtABg)VBZE$-D^kNor+B)Q=24b zzqT#5T5a#nc3hoj{%akx`FzhZBsQGA!1Q@aoSCfK^4RV3wpTGv-7+(lRrtzGhT3(1 z8q#+@e}3+I;0!OJtQM!1jQ-6r7jmQmWrg#OHuHr){^P;3!9gpc@Y1Qv?z@i1rb+z0 zp|M53ax*02#1lG+Cd8->T-CAK=(xYvH zOmR6AzTC_|#b-8?k8f?ny>fN8(+*f8p^%87G{x@)&DrzxM*Wa zhmOs}WYIqL_v;%3=NReQI`#?7OE}yY9@w!ve%_>KXVZ&scQ&qAG%wz9MZLg+Q)iX= zi>hZP)NA{y$>+7P-|}JZiHYs0zw!9cV?EX14=4Vr6D$=H@(+GIA!FKs>dHR8fP;$< zt+sDIm@AYKwd&rtSM%?_+wkznm%CRg8<{Rhht1sJW0s&YZwqhe#n9U)*k`})pTMuT zC_mbDe&%;+mo0fYR*zy5be}ALks~gDblQEdBg`es?`T4{m z&A35{wj;+lQlej%G*M7F@H*1;SdoAvFeXrcv=ezXGceSN^pJ{Fg z{XAtqZ-%P>rp`O2w8dvWL=EI?GRIEoHFYX+1~b`=P&CA9>ZU`Z!IT zUHL*gYZKS{Z;$I=iu0-3Zmlt%eK!2t%ron`yfT}ub$b~3=NaV8%s6|uMq;P^QkH-x zCnp+b?6@C4o%PEJzL&K}r6X?nh%1}V5KNmhx7;!)=~ch=Dbhh2Ildz6EfCS-=Dnc2t!PFZ>g_9+V(t+`hpGKDo>Tl9B$tH zCHZ%eJ=?lW3v+fp||xf z-7_Wm4X!cN1g|=jnfKPQW}9LztE>+k1v+ej;y2!Uw^k$F!^IMBg@QC?fe<9d`hZL@bo1Lw3&uXLWcIHsEz87z^JeGlYD{9@ ze&2G@f)A5y{r_B?a(r#A$Q-syw_DVgaNc`7|NULD-pQ->Z)aHUx~kVM>z_u`|2}8l zJ~>*WAXJdJQl27Iv#uD`>9uXMD`F;hj2D-X`THvnby^ zAM-ARH+HD2f3kkBeXh4kj$bdh%E)!bLKP{AYIFB9e&tg%!;@#1*@~Bk6oj(a{?&7u zzl5u-{_NRnvRpj1b6&hVSRc&%vLGO2=2NTLQ?6@GnfPR}#!Uy4)swpoIgO@`l(EA};H@+2y~|97bE>nHtOLBGovG&C3OP*2QR@6&j1 zb#CEOr)Og9pHAmxD0r1IamZ&(Z3!1)K5 - - + + - - -
+ + +

@@ -18,9 +18,9 @@ } SENATE. - + 3d - Session. + Session.

@@ -57,10 +57,10 @@ copy of regulations - for - the - consular - courts + for + the + consular + courts of the United @@ -74,12 +74,12 @@ issued by the - minister - of + minister + of the United - States - in + States + in that country. @@ -88,18 +88,18 @@

- Janvary - 27, + Janvary + 27, 1871—Read, - referred - to + referred + to the Committee on - Commerce, - and - ordered - to + Commerce, + and + ordered + to be @@ -144,8 +144,8 @@ the papers which - accompanied - it, + accompanied + it, concern- @@ -204,8 +204,8 @@ honor to submit - herewith, - for + herewith, + for revision @@ -248,13 +248,13 @@ decreed and issued - by - C. + by + C. E. - De - Long, + De + Long, the minister of @@ -274,11 +274,11 @@ also the papers - mentioned - in + mentioned + in the - subjoined - list, + subjoined + list, which @@ -296,8 +296,8 @@ of Art XVI - of - the + of + the consist regulations so @@ -306,8 +306,8 @@ and the - Secretary - of + Secretary + of r i ly @@ -317,8 +317,8 @@ consideration - of - ministers + of + ministers to make @@ -326,11 +326,11 @@ ion, in the - sense - in - which - it - is + sense + in + which + it + is limited by paragraph @@ -352,9 +352,9 @@ organize - and - give - efficiency + and + give + efficiency to the courts @@ -363,9 +363,9 @@ the act.” - - Respectful - submitted. + + Respectful + submitted. HAMILTON @@ -394,8 +394,8 @@ Regulations for the - consular - courts + consular + courts of the United @@ -411,8 +411,8 @@ Mr. De Long, - September - 10, + September + 10, 1870,

diff --git a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index cdcfe62c..46c06168 100644 --- a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1,2 +1,2 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica Detected 180 diacritics diff --git a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 8ed0fa16996f97f73d94aab6524cf2921c4fbe99..80c3c42595d0f3ffd5441a067e6349709385e549 100644 GIT binary patch delta 3096 zcmeyM)26#&69ZHqO7&}_Pp^~N$$R&=_AmA|7n`dX9T|SdRZom<*v-Ak>%mAn0<}!LvyxUyL|nh z%~$K||NQ>=vA>;v$GPM6A8W&0Xa5pskZ+m)<7eai6{hc;nvdVV`*CA_&csdKySlhf zwSK(zvt$zP{GuJ;yr>eeKu}sF4E%lRy z<~p$-oO7+?q>bPC?lb+7Hu2N*9(kLBABOvX9zR~6R5qPS?pAbK+49qIJ|(AiCNfq` zPt$&`{cO$i>ThfuKUQDuwBEK)kwx@|;i`Tn`*n7Gv)6ojyy57Qm&dYsQ@-?Hf3r2< zW6STbA9pt{nl^*MME*d*lIMCG&dJQ)b*tt>?q|nT-t*h%-`*Ul-En4p%MzOjEH%b$ z3iErt{FoNiH9FRBQ2M)aug-$ZnkRX;8aA!5T*Jsa!SkifJ^4F~4leC|Gi~`3v-fa? z<=IHj{Bjdh`IgIvMGh`Qq^c%OMN%Y2zQU)1n|`OC)z)s=~Nmpq-2-6-30in+M- zZ^@o&8!H)^+*XkZM+i<4oA&P%s%Jocl6;6hxQ_#nA3{4^MX?{JML-EYZ2q~>^2UJ>)ULF)AN&-IgR6wOA%<$usGv+dg#5$oOyCs1~_6Cr&Z$I=^x?N5m2~^9x&Sa$rd6vsisNK@;yAbxn(V&6jQlx2{#uWKl&Nso z8Qq$ikOh}NPjQ-gbGfR@>t8xO*RNcf!y3UTu{NTY_wSo~5fj^5CK{~W^+Yb-ca1gE zQ;E~-w%+UO5MKDVOj5ML^Yq#)$x6FEpNqX+Sh>Kew=6nN-YK;i>(+{ibHj?w>cUcS%dQb~&y;DJ;D?xlmhk z(r>lB{A_#k9|wgbs$@OjJhdc`k1x(^hV4O1$%V=RC%zQkQ1v`;{DhWg)Fsh}&Qsp$ ze$0N+5zbn_Hdm9K_1m1US2Z4N+?lvCc{a;zZ=M%^-?!dVna^}wqflTvTln@8&Piu_ z*mqr=`ceD)x+SllKUl^+Ve^t3U#oSExtgYZ^0{Dk_t-V|gk6$5_^VWmjjlhmzwY;@ zWY$@4)oQ23n(sDLpU|;<$I-c=_{MgF-&-6XOlH3&fA9O@>EHV6m)}3vxsvI#gt}OC zl~3CJ8*vk*+b&P*hZ!!D`>LYrv(IkjSjl?OFLIy6QJd{^Wd)8@ZSFQ)wfgFL z)oE!ZiEi01X5F?^I#k%MakHU*-Tf_j(Nn%|DsFUpy32Hdo#@W`_V(omci1vtx+Zg}sdcM$f!FM*y~Qnw>rFH()@4QA zh-a%*x+pxMg*`H|ZOfrUZ}ZfTsnkqZS)h_I%USwS^ruSCmF*8w*O(ONT$+#_6_WNy zgJ)yNHD3Rpvbp(}l4f_=*H6mjtUkyQSbBezK%YU)jD3pLQ>EvMH5^`Npj%Y5v3`YT za^j?4T|5zMlV*$F^H({rZr4%I^tdml9qaehtvh%2Q|rO!rLC4KG$ov_>bk5y}Nq!m2GVQ zl+A0xW_r(BU+~Q*_m;|&H>=Y1J0%jNO&5K$id=6j=qR#4m^H9DT(ECV(1{DnPnvK2 z9CftDd6`$_OyR z>gunYaG+kF|LbPC?itK3q7i|DEYGg?J#d>bsUl(1r42zDf_5FpGJeNw@L9P>`2LD% zbJy+`$er|J63?Moi{5@Qc(Zi#*ApUBCRtCLV(C+{x@11An z%vk-7XRgpopQy}oMfr-oTMs;a(fO)$j?RVU&rMiPWb)SX@A65aLT^nZ zn`T!+X131u=(A7R>TkF)pITd89x&}aS6R!y-M1?Re?K{>Srq@WJ>U2A4TsM) zQEnYuPu;2&i(Ho4^y6T5^{nH~v8?LOd*7!ltc@03dLVV?>HB8xJ_<~^?=tRu=Y6?v zF>_VZ=@6qc6Hg}_Kdq>na>IMmGwYxOdzGY@UA<-U;MHf=Fy?SOr|bIfg6>(@Z&|ha z!{yC;oPS<3n|j`C+4r{rYiH^FP-SwP_(Z@MRe58NX%4o~HknP{w&cB~r zdP>2*%IQl^bzNED&p-D*dQ0+ezsUclPXC0*w=>>O{uh5=+4g1;|Kh37w(ppKHM1=C znA^rJC(=yrP4*J{SzMWO%}{#xgPrNQ|J>LfUOIVe#?~JvgJSDnE$Mp~;nJ4BH&MOp z8Z+09`l}PoqqoKFDbM}2DdU=7>lc;3B~R}xxca6zCP2MGtV|~B(IK%|7iXnosSBp* zx!zR|Pl%rJDK6-qmY|!pVcHUHr@8+R6{f!1qWVf|!gj44yb(usxa7~XWm&ON;pFki zv*nKUc}W76i)0N?xVl@G*C%X^{mkRErSjj;roTK#@7|7<(eHNLH2GEMiBnk{FK;)k z&DHz*N+Q&8b03^CM_4WF+9&5{N>!% zr@h;53;*46@7DV{XXY$bD>gRW8^`ucb*J0EVjt?i|Ye=573koCFvVVUP28~gdv@Aqw2eE7L|PJDmy zqpCYsYnL|Ne!EuTu|$81^rDkbpXRTb_Vv=W`Vzf|!jLF~N{Enlr@n?53XHvi? ztz*UKbY;KJIdd|i=5JNd$<6DZ8=YC@c>jRgl}GKI>(gG(Y(2D8!u#cogj18v1?P2d zyb>h&^z39um+Rb0&peljzi|^RHfpd7pR+N|e?wc!Nh8mlPgYvkd#}v2GW@V|y@Xyc zKcmsSa~4camygZcAD@&rSNX`_V*hl`C0{(P=V+dFGM4i#xWN`J+VlM8bI%RB*HhQM zyObRIMEqrP<$L`lEfKeltx9UOXME1NWZ~p&k#aRdJwr_{1qFTQqSVBa%=|o;#FA76 y4HqjT10w?iGb1BIGeb)g!^s??u}o%WlhZ{vF&P_9mKO8jG%@5-Rdw}u;{pKnuN3kC delta 3074 zcmZqE{h+g969)Q!+y5s2UVZJ{VRkWJKD&S0-_EbU z{rvtmfy9aD>+_FQr0!YTZfo*5t7r{>((-TQ$t z{Fd0cl_JWvfzwxdzl&~&IVfMh{r%gs`M1kw3)d8WEw!0_^E}(5TX(J(Z3sJJY-Kg*Rj4*DHqI~`L0=8K-&Nw4X8ZS|hISCd2E{w&zI ztN+8*S7&%D*6@8>K6SOOl3c*zi@%t6fBSSc?3j3kX2!Ou_BBT(Qhm(r=Qz%|)4XcW z`)^z;vS0IP{dTz_5pZKc(WmE3;xz&6qD-HKzf8PTd*t!IHkk?cQ^6@$cm4lS1D7b}P3Kjfh=-N$zW5*rj)sD^zCRih8hdm%Q!G z!^z5ProH-Z{^N?kw@cN+ay(ONbf#=RwUb*mdGaYgJ(jlgQ}@Epm4=JR`e%luTsxdB zx10acLp4t2`09>7t$l|IkL=QvRs1=5ddt2d>t&$_=X1X|dKsJf?q2=F342}@)N5|| zGVxAO>%pz*zV+=beP<77Y8{^v;@i@Bl&wr}ndkq-YghKauQscCt!eO!NAtobksayn zLLo*u`ZjIJdCGmRGwoKLxMc9olYi3DxkZN)4T_f)e!P*qeSbnE%Yyx>JAdYEbeOHa z?F#qQlTRF^Z+vV}Tr>CBR)y1_66!beWZIqCdi=)HV|un~3y#f+`_1hYc;IMyB=72n zb^%IUrzLJTU)dg^ZGLo}*cTyL|K^pyqP0-w%{9Ie+HK)fU>All! z?Mq#{MM-pef4b}x_Xdk<*IT0beD&`YT(v`(x zB^lMG*1WeqseQIt=JM0k)}QkPOO`sQXTIuio>EtEJ;T&oEY!*F{*t*CW_O%lotnwC zu;sJM1A~>48_x#1R~?%v>Xm7)x&Cvflk!|+)rrO);uHSFl-(|${Ec1PDY5(Iw&Iun zYCR5Ke(pK<{K>Pa_Zk)(vGY6KdUhsK?e5J#Z$s)?7i^yzV4lRbFvnn3_JJ+MNop#p zjGk#%uY7E+{%Y;nkyfe2V;p3tdnV*;fXAU1N+LqDgxNb!-s#Va%s;Tn%r{lnSfX>~ zktddl$##ono>ZhLF40Nh6ZDC!%Jb6Mm_9LlgP0MoUqBY)yR_pPwkK2Mmz(&yYd9zP zX7IZz=Udco)q83xX*n-u|1w>=Uheah!g7<(jUVz#JcL`UzEA&l z*}hTg>M$DIemR0^Hs?=rIDXJ=NsDc ze%T?CV5hZ{;lMeCAZ10bHB~I@PeyOPlAP5&{n?aC2J=$k`1(ClZW;2|dGhSmKXSV1 zU|VL6nc){J6$fSpCiBw)2PbWl-o%vl&T>NStmi(qEJ-I!EV$oH*)n&}kIMV0ca5f= zJE^p0Z}GOZaV_5`FI#=%v-ox1>x-J>&7!xz^H35O)#2{)elzDJ)4dy~9cHPVlWI7b zad+WjxmL@9?C;ZL+?q~qt$(M*cGEy8%dJk6u_kHTx{RxSw|}$zocA&7#M02T4#C)k zJ}=)Xq^$|wW3v2M#tyf6B5oId`rZ60Be$YVZEBb7mkTwTI+qlF1)5EAmHBA>?S|xo zV9Un$=DO|Q`K(UZ7|&Q^?(}V?QSffI9P5wS2fYl#JUmn*x9q7&bCvm6C-bTPL1yBn zHKr?;b@k>gJ<_Fm{kub&k5=i6Fwr)-_r6=#)om=Pk_jTEaPhiOl&4YwPMGHVAZAEG)dGkd>68*0W1I+5c!oi*1v< zl5E3Lt0_O3!p^!V&nSu&V!IUgfXiy%v|6j3D-xT!o`q#ddCG0q$y=Tmnzet+i$eQz zB4-sl487&!zOhVTG@9L`BDZjLp7L772IY{O)mNO@Bhum*PMR*@gIxg`z60vwtopc z!hY$ukj3Pa=Q+}s9rivp^O^Gn$-+gCge=o*>3TUa#X@8MZ0{5v#|H6R@g+wc*PpMpoZ+#0$J5s1i|$NJ^gZq{LEV(yb1~22 zS6hUm%AT0pv@2g>EdKS*ciN753a5hKUyOA+>vd`IO3$53CyKp&xRWEk@Il^e{WHE* z@5`@$*F7Q{EWGUAmxB7dre&J{778#wSt!jhYvp6pOX9EpEy`f160%ji^!f?pYwo*8M|11nt$oPevf1O`y{35^RvDCk5jD+i5U$*o@tN<@@<>S` zzxF8~46cel`^<4WZvt1C>49TzO`1(_?Pp1(s-D{Q^2TdX?(65|Uv1s$x3YfsE8mm- zf84Z7Ce0FW-F$T31cu+E;baPy#0#)?LoyeQeI_r%P26eJ;0b zbWXVF%DpFE-E)S=1TO2@jq_@SYfB>X4&B(W?ZjPIH8(5Q@`-vIUP?w!u(xht<`#d{ zcieS){XxSoCf7gslzdCn+@&({!ltmcm#Zsp`!DOVs`F)>YkdBDjar^i_E*)%a`S&~ zP_Rz9w~XuD@f$2zZ1Y8PxsPovIeGb%#JTg%mhD!}mGMTs(a(i-9@_YxSoS<6M9I?d zjSWwUn~xWhhs8g?GofDt{=dEPBJBB(t>3w(@#=j)eQ;g9)6tD@8_zcPW?z4|m(gYA zUI#CZ8H}2$VoUzs@?JR6>r9YrXnsg?;Tg7lj~{7i7wlhni>rH6M@G2;hvBcsMHe@D zDm+rkJg}np{LPi6*P2WBhV5brf0ZmW|6G#AI|$gA}y`DdnyO&9M^bL%_%e6{s z>(`q()IL5OEW$P0A=yYIsiE1hB-h0W@?>!a~ z36A<(s&S*QBgcc zuM%|j^BsB4V(osCJ#I$NCXxDGYYIxQi)RV%e=$32-d|RO(;8-fw#-`lf!*Kay2Rv0 zk#aQyJp)ZH1qFTQqSVBa%=|o;#FA764HqjT14By#Q$sUD6H^l-qselju}mf=lWRpc WF&S7+HWc&WG_>SWRdw}u;{pH)3i1X3 diff --git a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index cdcfe62c..46c06168 100644 --- a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1,2 +1,2 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica Detected 180 diacritics diff --git a/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index c786f4e6750037403708462c03825a2c4a7eb5f9..450cbc3c9656e4e9555fb99440195118c17b17a2 100644 GIT binary patch delta 8940 zcmcZ=vMg!=2fvY-k&&K-nS!a2iK(8s;beY}MNH-vlOHfD*KdhE-KV-U@6P$~EBs$J z-s3-Da3T2W-Vf&--t(-_xpO!6hIjqmmE{wH=1pGyUZQ!{JonUPL5of#7_9j7o?p19 zQbpcgU`1-y4 z|1aC~U(NpI@$>QZ`*uN9Po~t{)%|^ce1H9)%|Z7xMTP!WEbaLzXHURr%lqo!HEeT5CTa?t8ezVfyBW)74q-&i|nvlU`Gkk$$^$)-u(s&G#O6*S)B> zos(U^Znaa5#pK_ocej6h{WN~>n?0Su)5H?*9C}^KetErC_@6h;<;z$8`j{OZ_4=^# zWtSUM7DvR)e)Y&L>dy0Q>CZ9}dzb(27Q5#1_e1ikxwR2n+XC~XbpLGXQMJ=^ zl3#j@oH`p4uXp?Y@vkR;-Be4xd%X)|?ojUdxA4WnT!u!q!kHPh8S`I$ z3cDP)??K$9Pd!&p*Svh26T`e>@xPn(b`KXVIeF%{5Bv8MiM&kmHud%^cD3JrpC09V zc)nt9-@BQ=rd-IYXUGqV+qf?%e#fs}WoJLnxc_U{{G54TZ{2<(EV3sp)_~t!vi_T; zvHgtomrqNzKHa5rhE@0b)Txu#U+%2&)H-)=b-Q)%mL8+6?MEI;?)x?W%Pv9L$$fcm z*$WqE&Yw8j=Z(41{5Nh-)Q;Whe5&Wa#{9r_zGs5R-={}AGjD#mareIFdiKuc<{p(d zBNzQhk-2^8>!Xdg?NS34Y25fO`tvKtyzCiYKkcitZTl8}CVR)}GpNB6n}jjJ5p-Kkf_@(LNTsB}0;l z*?6m=aQ!#wnTPL4PnFNK6790RS;%f!zwFA_aPw^K^@5h0B0oGg`+V28u8b{5BMSaT@g?rL(J|$^!l`YZ zd5ih$`;W-4y*_7c?Ud{XaEYo8bmtft(dLe$LL{xUWy)WA^Xh+Be@V`xM$lkDJyvZJ6k+|CZl@Rd>b~DsrpI}_-n!?Eh#EMGLo?dR2&hR*RWC`}*~p zn*3S^mo#a|Ojh2p=x+VOOwLoTEBo|Rw((7Kb7Nj@Z&Y@*yC!yzPoHPYKbeVJ|hk&FHg6I+ifDZi8MYbdab%rrQ( z#_;$8hiN9oIy#;m8>aL{3V)EuXZ1V!{RO|_HWm@bioNsw&#&J->F>Igc6;5MFW4N= z-sBX0KI!eW{c0P&7`m!2&(J72m!Q&6Ke4`;J9o$ZNsH%K3Nse*Ocj1zzWL!h2X}{0 zjDcab5>;Csh`XpJoS3(9?N{s9P77<3xQepP{{H)BKgsH5=z_f3DLr$$S$=#XW zcevciX2z8x-&vQxN_iE|%+V#c&iAnV>xT<%<{VnYbnT7xuDBzYOgx&^nA42zc&a8c z<;2&!9m!Cr=5VNrne&7?!1Ks&khv5xs$oGt&BTOA$iHuT?*f< zEhkE`v#p%8HG-eLpzL%t&%-@4+anhUo42ie9y@X88#$rf$x7cfn{FFy+hJ?2)9)s> z-f;d~wvsP)D;G>TBF{9VHo@wj!K`y8cLTPpPf)m!S3iZzEoRTgMmzuhgXLXQ!X1~a z=QglVndTjEZcduoL>8l0FIF7++c-%g=u(T|U6Z2q4;$@&Ip0mZdOmvZTZz|AEB##G zZ3}p`I-&H*Z5<|au`|N!*8Zz{EWTyQ3%#%J?=p*=+V$qZ{jUtEo{1*!M9g3Rlisti zlI6*HmE#_DJO8WQtY22%)crGL4nL1nnb}g&&g%clOCC9v>|txw`ptAPp-I|Hs)GNh zK$_B?ZSP`*=FCccIW6{TjQ`g=XTR}Gy4Wu#;jnY@)(

DiZ8Y6pJyYggxYaBxm=y zfa&F%yG(oU3G1-TOggxJ=I##V<65~-&hL}|H1ATNk!PKI>}Dr(VVkG*{43YI_LF?M zQZhevgF)AZysoc#YNh4eXF8Q`1j7O=XbspcbS9cG2T-p36p?=0i1L+T^cLvpe zExJE5`L$%VxZ?KRDtp++$5VXyq)VUcqdW8G8LugJFXs8je~7wwdi* zwdY~&2UfdFXDa2oh%#KyJ+ivQ_vR_{wy1m%9diF?l!3jllIgSk z>r|wFReR4o)$8jz9CM)0dov}1N`Kup)X}Y;Ws|#C6;G2(gBfo6^;a4%`TdFBv`1Q$uX8Zcz zsV`)YIQ&mjFzw{6ldPBC-*Qee?@rsh$&&GwM&1q6v~RoE1p?e%>h&KS+45-9#=S?B z8yHwC+K+T)3QIS3&NVT6Y?Cpmwxr?8L$$=4mln+YvF>hrO!&-yaY4(j1gZP3oc#F4 z(+JKThxAeng+GMudw$dSz22Sll=mSa5_LyPy)qf=!<1fk`+V5*`h3}OQ<3Sj@-2RB z@ha=TlH{((>D9!-m*QmXCF!2u#bc(MC$>jS<>|vof~PnmCR_hmpY)V-HIFmT@n5I6 zeSLAr^24cUDZAK=iOtg!Zp(2m=)9K@<8S%w!}Cnu6?|Fe60|SOl`JqiJR@F(SeL-}ZX{^N7XxIHpY>|z1A=^_zZfL#U^};DE@W2L{)sMQ@i!Cv$ zn|D1&>8*z2cG-`-*K)cRH#^<(u$a4tt@wWsr}>^mW-lggoUmEug5^xFiE~y}Z@=F5 ztNS+V~FJ9A3)t>dCSmuGKp4g}CNaTsRUFa=nU0+2H+&Z5+y%zAdYHHapvB-c%Qn zCEvC?NwJDX%z1E%IplxFHvjcr^EU`zs_|q$cJSUMmj~6HpEJ^a_{UEYw^3gicw^p) z_ev=eG2si|YSnbVbVxAb_|cjDS!wAQg4 zjha=^(HQb~SHJ?E53AcgSV~ANxtjQmNx&wC}w{E=jYrU?}A?N**KImHMxr%JrS7N;>!?GfR zn>#H1sk(yRoEJIrt9f2DX-LiscNwq8*JGdX zE*mFhvCD@OD~$x>3Kgb)`udW$;7j7pwX@dl_FjJLQf$@lgVBF1TJ()yJa7I`F`G|Z zEKW-`BXD_o{h_tGyDv%oiC-ywXixvAI|eSwzH*#t*Sp>-9lR-7u%iCbn#wO{Cr#D+ zCK~mP{o7QLVjGSSfjtIUj2k=k{^c#3ESQ%Xbll57{Y;eD)30|o%2=+OFfCliC$;L} z%f)=ZrT9pqmUy;jUb((}^YGAE;VS$h70CZqeG5VepT9n<`foHHepcK-P5xw$9Q zC+nKd`po~EY!~c!{>HaY?4(-MiU*ZJqEBBXbgnXKyYV==Z>w!ri`LS!hxykYPBe6! z=euFLsE+#!8MfnID-;%p%zhUX(>;0BPba}IZ-1T>JtM$*Y^t89^EACkEr$EQo9aWR zx-H!ovQs2vXWq#`%Z=`xS^v||DGWa}ILG*jq`>V&d zvMlYG^!?JU_v|_u9U6c2Zdg9j*O_uJW7W0Pb?Sz(=j;1!d=#ae?3!FX+AXEugmChsJoc+TIY1HwunSP)s&!TN4c5Sdvi__T47e6eSq`p>BWay z(r)B^5A=F(AIAQ={-h?C!NFUBsok2-POz*^wlVlNecG)Fmx3J8wKmU(7K6y}DXKSgqE3A})%`d$MvFnCepdDCxpyD^_wsxD zZqe^Ry@OUaOc!zo{`74NvCsZ%`__0}pIBnZl!NDVN*}66PO{V7xX|TpleAW1YR29T z+fu3?6+MjaS|Qx=J}Ul;`T^Go7mls^Z@BVV$D%uCM!sp&&-IA$-PJS*x%*~inc?Ya zmrOMyPV&l@UHz_g^gz47cg~vnXVYAMUD(OEJ~l<=klex0$iMDZUtb4A z?OyrYd&!nuQD51G=00yXI6Ih1&)H=EX8OdCI}0UT`>+0TTQ;ftLp!^LntsymY7MJ( zCy)5fWPECPI{8EOsh$(&+cQ>08m3R1>z%E^^J(X*Id8-E-O;Z!YfPV)>83S#j;i&s zZH1bDl{pJTcZao^@7;7xai@ZglKC#XTjm|=4zlH5o^Ls2-_}2EH}~%i|8Sx;=hND+ zyX*zFnr`CT8n(nId-mk8j|)<_WEd*=p?!yI%7;sHGb^RGT~)i3Gj9qvWB*c*bKevEb(Zd1KUZwOjqVfaQ!5kv>#G=L zt$qDXF31(|nAfRI67RXVqHc}>=cgI(wa#z_@i=H+6X2T5GSzm6%2Dr9fyZZTI*)yy zoN;bz+X)fJP4Xvd3ZGB$x_mhH?XBRICR#$86U$GF%)9jb$)(aC?k_&ypTMz#xqQRa z%@T&`-%Sz~{w7+?U6*1Ly;998LQ;)Y?m1h%!^t?6T@kX&I{rQRve?#hrqv_$JJl)2 z-Ii3CA3Le@`b9?6^i}7+h};XjF2=lk%dq91ZTB0KyhCliz1c=T zub;45aBJR!x18n-mR<)eUbU`z-Bou(+vhr4#MIMSCgK+x{C4eNE8F)x^r@W3%%JZ_ z-aISk`o6uM)%{OBAg=cp6foeO6? z-+8Zz;ZqlU_t8&v%?V9IzWvdgbuUd7*zHu-ov`k5k=vIqZ)5A<2bk!}MO&Vm*uU-1 zMBeMY#@iC@#m*VmYR;3>+1j{Wxlnk7b(e2PYt+L)IY|=VSBc~;P&lkiA+`f{TpSsAdQN!3y?nO%}&vkXaeut{) z&cIXVysP&+OK#csYf*%K{RhqHt1C{w*!Cm7`bgHVr!`g|eolX|RXiffxa8EEE4?@F z*Ia%YqP1&o#x<_A{kq(Rl_G1`oZfzkX~os6p~br;rn@)CwS1Aa)iPQB-GzNZFvrOg zsrx#=${w`a{aEw#gInI+`^5GHL}?%LJM;IB+SPmC*&ME@u6s25(A#rC1!vh$)c=1~ z`EHMvHQz7iUv;tEOl{^>_l>G137Stikg)W)M2PgIIX5^}_XgEi-`5sl1fOqHL ze<7zE9xl&N`1a`Jj-GtW7x{|cvo7rJwXjc7-thQ9)nS+H@=v!~GlG=Uj$E6j?mYLR zPvW+>iskZ~>N+p(yz;>By8MlIvD@;m6rFll&ue3Fc3i}=uEo$ z=CUhxoBlU1oAPPqF#*#pt4t-W4CKz~h`js0;r>ywv6S9{HFJf*us ztY3ZCwI|{l=KY8I`%4~fUfVb;Jt_bB#q1AJ54KNy9{V+sdwsc0PS4kCp90EqLsozE zf5GAZXwt^CD3eS1#wR`)xSYEE%m4S68~e9TKl3O=b5`W}#RmWDx!Bj7iLT0;*MBD? z`}xm~KmQJXoD(JTvvBIIV&C+*Ik$K7)|D&oOPsUvin2T}H}}u1YNl-W-Xx`n;5iJl zoHrcrcVrcMIX6}Gm*ec~{MzTw2w96srpl>*&G?bZs+U?Y{U$H-My@>^{;%Wno^Jnk z@Ne$_cQ5(Qm%7x+eCewDl2u=PfOCS+JDcf=0+ZW+y>zgSP^lJ6(|cw5R9K)*`Vr@b zqMzQg_*g8DdCQ#8JJ!~dP`N&_eRraZOIRWExs{(TRUf`H@tgkX>Wp=>RlX{4x%WCv zy25H0aeueYxjV^8fuFB6zTR`))45@X_Z>%J--CNNp2Y`VD&OJ1ror<~_QNIp_0#`` zwDvf^D_)iN)HB9sQfQwISCeAZ?1Imor*D|)8+1>6D$(hgB(>CDbiGH|V!`KP;WJmh z3R=EdUV(qQ{sXh;I~%feg1Qc?7cSf2y({eBw*s5;zdSAf?}jd};pb1P+mtoowrFXC z(1SOj?Ylor-OiV)$+_ad%Kf$H5^vs*JIz;Lb=s@5uXE>bQNzN|DVmDVI)fxV&h67SEg4vD;rXH^zNo{C-l{m+{*Ir{+(_r4FJIWiqM- z{BiB_cSWaN*(&pB;n62KZzn4>mH(XhsMk01^+&!p+ZOx0pZ@lm_}+t#CfgG$LRTqy!*f2Db>9;zPznL`d4+jQ`VnTQ4=q2^%xdm?>Uk#lgd+P--|mS8DyEZHM6T>oI`d*iL(aa7>K*Ry_pFRl`0cn{Qg{B3 zzcO(PRg*6r5ft=u|NS_V zAD-{ZpMTBvMDQY+Px8OnHZ*JenN+%QjU}suV&{f+``@)({?2z*b?W0}?#E~AZ#vG) zU$p(w(^`w8VQrgx6(pa}`_Fee>c!GKp@$CmMfI7-g`KqtxBa4d{|+BF@1l=CzCAQu zq?)PoyTM@kRQ7Gzbq#cbDg)U9nhTAbVhE`;lhXfrM=-C&WDztytnXv z{=UMGqNlz;`eOB{ck{I6`l{V-NmC5E78|zJ$0wg--IrNeUC3kY$NT!%CB?PbGGD4Y znWK*V>3>qWc=AV`5YGqvMhP7^F8o*Ol|2%uyPi+4QY+`d-&8SyUBB%a7#RNlKO5^k z?TMPeo!@%jgqc*fe3?CKXYc#{rV4g~oE#zZTFozSez7Vna6*ckn@ir6nP=wOziBe5 z&e40Y`*{5c?dMXe2d*5Ins2JHzvc9%2Zy&g_GMiBc5O4q0qbh<67Q|$)8?FTt)KI^ zr{H+1rBScc^VGAGz1`n$y35;gY9>qU4HM0y(>t2e{c9P0-aSyd*_gE3GVI@fk2kKy zKfW~0Yzbc4wod$t%!HepV!ox%GMRigSv1*Fp*v$;8k zdXuWzY_q$|)>?4;t&oTkd95MH=kMsZ^v$twnaNF|ypOdzYtwz)WqKd||N8OR-OZ{~ zKc`N8`{5OH+xmxUrt2=H=$wpQbmVGT_Zj8yeLlThi<(LY0t*$P` zukXE*$m?s?Emv(kx^l^p`u3CgpG1Y19I1&fW{q0&gKf+1)Tx$_{>!-jncTH)&4ljP ze;z8&TEAk=1$nWO-K`($a$g+Wo;4xl#f0#`x9>g)T;1ytpH_N@pON=}y7o!Gr~4-L z&KLLX_}$d^UjF^H$)Y>FmkA!3%)VPN@YgM$*d^DRRtnpPFo?W1l}Nl@j?6!qLU)1(@-G!@@hr}|&s>v<>l z=OU(`S0DNo-D%r6C9fn`WyXTOv*`yK+dhBzF?p6$`eIvwn!RD?e*Iwex}|b2QS>(7 zs}T7~g$BPi_vKo5Mj3^?){Vb>{G$I!wfW zyCr4obh!SqW3m3a0=@-@6RL7cZ|iHn*w3+SA@6bq&-zUlPKu;j&0kP3=h&6GC3(}Y zSUguLy=|4NVN|%WKJv}G<>DgU4cjD|;vV}v@sSC8IO{_9PKSS+P3+Pt|0D!d-*Q}J zv@OB$*d(W%Cr{rvR34aeqFK2{?!;Z|=W%x*t9IPldH9A%L%qt+$kL3OVOEnfeRPZ3y~z^ZBT@1G&Zt(vt8%yh639=!J6OPD)wnLh%Y*q-h4(}K_Z!iWYYKAuC%q(z>XclWLTIZzV9mf}L?r`)R-@Ejz)v`~@BhRGI+Ie@2 zr@n+Hr}HN6^PCYMbIa7`>^)KZc4hmAj%6=j@0mQczP`nIqW)Swz2!eGsg=F~OP28;*xx{eAtQnUp%ip&8NKhWS4^ zH6)W`3J|lA9herNhg`OSn4fVD%MmkEIT-RZ@ujKuYvXZSb+BQp8KH?Db z=9>Lux$(6Bt3TIA*}Y9PVwk&r)3TcnmR$b%v|G`~0%LlHz7{^Xqc%sT{eq?^x z{kQ+#hWU5xNv%4du=QEZyh=w)+5T&xldE>f>$A+4y*Bfx_Gh2vTH%^Ir#tv}&-j}^ ze-p#W2_0X2mD^>CCqDmiY3GyDw?F08To*(b#!fo_)@5Z?-1kK>Jv{Y`STg4DEz6p` zfGfmNIoTsqX!n-#@~&Np%11u(%m1jI@zdB@ZEab3{R7|cji=}KwY_wB_C#fV(POi< zTp|a9ihoxNd=O)x?X?xsQzu?wTA0jyzP58Oj5Xg@&JR1#S8b=xjgN{5(@?S zryPG_ULViD;r@SC+ik-3lS4GhnGN*}C!f<0=P@!cFf%eTG&3?VG@JZUBbvp~z|?$l cfaV4kLqj8@$t+sl+=fQRMqH|@uKsRZ057?I=>Px# delta 8237 zcmZ1$bt_~62fv|(fq|ZZfr7Can4Zkfv548g%w+OIMy2{Qu_xEbcoyyX9)5*?gUD|t z4(18(ZRJXY56oULxnS)WUf2&C6A1B5DlH<6rHU+a~cj zQUU%a|MzP|R)<@xu|n1A;8 z_VM-k_N`ANeJcL^nt!~0zwN)9GbDOEzq}Fo|LMo~-Ykw~@ym@KWWz6YG^>Q~o3PLD6W(Ej=B>w<}TE$O^p&QE!K`6`3> z&3B7qQ|gwL)b$&l$zJm`^7Z}tPBYOPE5G~iTJ!PBwi<7l?s{*#>G^~L>%Uum?7isMxJpGTIA^=~Q9Tw;mPf*J&5KfNYoCey zJo@dW)!vDOI@v5!u8?Ed@r9zpmP*3Azq1(5% zUCd9vy=eaPo%QpR?>{m6O-U={S8Y1^tM1D4I1 zm%kl!*(7#4s^FwwD=5@`z2?%{@wiVd1`8VVd1j;4Kpt) zu`aypGL32CZ{|A|sz%SxroFS8G*9)gbJqVNi@;qx({ik>u1$Ksp#J~Q4Nt7j=AAwN zW@%Jje9F?F42+tIdr$d#{Z8LAS9^(dIiJ{%i|5y`{ur=l%JMlWLYJx*R&DvV@}rkb zn}e%>!lt#(;lWjItB)R+T>a+O755<~#rWUfO)>cIvK(a%EH5 zDfcJ1DPELa|MMKX(}ol&jmPu&`u{B7G%@F2|1pUk_Pet`Cayhq&qv)uJ^W78jP3Px zA543KyC2N=u(&fbRE{ zPhA)Nk`ZOZqr9@(m3{I-c`>mJ?(gmbDh+m*n#~IYGbV0aBx$_D-0Z@Bj|o>Vc<=uE zjpc`!M6S1(4Uu%F0)h z?Q^W#zTfcKzd46*bI-WQHnql1r}lTz;l=U4PAk?oUFCT4nBj$NPn=ti!wjtk-&=W) z)9qDnpRc?j@>)<~;@sE=i}eMr&5xa68*w4~ash`^+2skIc?VdIPR-q*a;D*~vW9o$ zo0yaD`dlL%PnRrOqU^PpC8JGPkv;WP&7$SoFQ0zNUw_Sau@{5&Q-$2qqKo9F&GgG^ zI;yb6GydSVzx8=*pEJwsIG*Ou+|k9cys339YgARlET_b^5m%D7E8K16b5wobVK_^9 z0`s%O4JWSrd(ZmdLED?Kg{-Y9eEy#kbf)rbGe1A!Y7o!!^{mSso;Y5QS-rdBrk0HA z0>vcxvWKF+p8ebxr>x+@NdMlJw3@t@zIT}qnx;2e$}-*)T=l4Ky2AV#ky{p>KRQcHmMPphXYHgPu&KCV z>&fT0b&k&4$dWWM>45yim%4&W*6nTK^>(a^>wI{hs z{DhoYKJT`BxdsLS4|H>V)3oc^6#sDS+wcEp``x8`A7?ae?NYfJ&v)>9`SuS+DVbf& zGZ`!sls6P|bCwjmkZwJ4pl!xkn*}dRMgHA9@4Ccs`m2EHOpC(|%r*TPwF~>5AAY|5 z>Q=eXi!8kah7+^aNQkmEy=cv_6;2m?ReDi&?}Uz@QGa%Sa@acCszgV}Q}X_dZS}?R zYfH)tPbFP(X>d#|Z`fun=d?ZJbXiKRx9goxjE<@tn>d46jtH*dmVd>|S+Kcu;-i#I z8Dn8Zoifp+`^&DrSpK#}{9SV*n*;B`;wxgm59ICp#ofH?>eg-V7$<(dYnj=)a(B9o zQ5bjFMw3>1;m5mYy*_EZ>cz)-&nBg^ujIT{pMTP$w$NTX=E|8(7kqB#&tvpw+B74@ zLj7e@zeeiAGa07O4>-0IPmM8-`63ySk-=hr!{MJ;@(tTyWKGatVPHOl)c-=EX0OYdo#zUt-01`EQK``#b2 zIFYZWuNH0NbWH7ik))2mZU&y)Z&ky27vy}HXqy))Ygn_zsKSHbIB1-J+FL|cOT^FSP>VO*SGs`>*>Gycb#Xe7s;Atx$v!*&vjRBl@klg-30g9?G@K|tn@qkWm7I*;kKVQ4j0l^!f5o%5aGf23T+NSf-RKQCDzIa{`{zdu3olCCJ>z_{eIWDt0^QFZ+UZc&kADh`koWJxaZG%Z>oc(1n{f>FI zPfE<*``Z>A5Q%B_jja0Q8$RblT(rRt&Nnv#{xeM6{WVObXabi@Re{Y$!NhpyMXXil z&v*D_n+a?FpRlQibGh5LMpnmve-m~XOiq^04+<4wGfa*OIxKSB*WY(r{k7k-XG~i& z<9N$;$+LPhG}SjP-*Q>2pS`u`@RJR}^Sm)L`rx&y~+|&$Y zShu+A{`)Bedt7k4Vv4~b&Ymj|mhe^k8DO1q~_HVnVnA}zNP*ix` z?G(j+K7Y+FodX|~J-9l~Cq(w{Z^)=;Q7+lH!tsBZS3>tC{+ly*KhHn0!#uv`c2rAd z&u7EfFOQg*Udi5+-QEy3U&Cv{f62ZnDeI3_fBDgE%rRTY`mt&3rHW7X-{;Jjo{%{6 zsHMb=1lLE`l3yf*mrcJHA|n57)wQJ`XVs-k?z_k)b!4W-&7XVqTF+P}^l4~lvp@8m zbfo@5(PEu-1)U;(7S(rVtvzsJDf9CZ*W=E|IL}(YoOt^B&G$2(yRP?Q4>={o|3*dc zux#M!!1DJE}(bs>xn*=G%6KkMsRAokaKVGQsPc3aY!3_&o&oE7qDnSbk!` zh5No%AB+A}HouZ&?VP6lW7CBmK}f4o(8M$}+tb zSk}Br{5+#e|3%@&4poWW9Jfw%DD{cv8~J1$igHFRv`NQI9|PUvSqq z^WLLck3*WLtmHSYf3?Z%F8is1N881YE9Gx=s*&IqJ-LQYF;&Smalc}_nd#}0Nt|3h z6C^whj5SJvH!PoI#h4_OuX)Dp``i1wu4H{*@Fjhf#R5B>Ygv|YYZ9H!_@*z-xh%kQ zhwav~-ZcSR`HpHFXgK7f4I44IBT-CO=6h+{S(jXC|R}5Tiiu^o<)^q zZBA#uVJcS6vAnS?>?!XO`vjYqmnWXQDbbp@syx&9XUhS|}Ilgg$-%brZ)k~rokL5&mOZ|=0FMQ^8 zVzZ*Kgo$O#g~Hvwsl402X1w~4>Jl*ZsrBMs7YpC}*Gu`>-^Lw@jFdRHbsI<7k`w+J&)6;Ulp0{@aCB)W13RQ;q4dSDhICTf!?n3A3X27qu66NN|33 zj$UaXvOhlTY~rh3f>tMWS4>O2Ds$kKo^!#qY5(e@{u}l&b{+7vIj|*}^NR_u;HOJk zI~)z3Yd+d_-*03>hptrE=OOV z?oIx=j;G!|=E2z}>*W1!T>X8%sxM62JU_Kw{D!}-WIE%Qoh^5`>-Mbh73{uZtG38u zmd;{5)<=FF7wRXgman;XAWY=8!@(D#@1Jb+3v_afT${+&JKybTn}eL!_pmR)S>j5v zF4bn$s}pTsJPHbPNGdU0x#84%wg)Bx%ai-sv!`gPZ@U;bWe!)oSy`~lOm?--W4S8E zyK8@3+xNistBjqFqGLX{%*@y`D`n4`->C@`)HCZ(be_TSzelUSq*AloTE_D0k+lz` zndTZRap!Fb*PFw3D)h)~_1DryO;=lAh3R`P@;UT&>94T99|Bh#B;46-e`qtMf0Dht zB6vy1?Ml|VlRMO7+e;Sk9aO)!^y!tAui9+ZYCj5@DleQlskI}}XRgAQ48i9KSvxxX ze#~rgJtMSFhgCzBSEhznRbdom|@OqX7^Uo;gN}%8k@E1 za3K34_0A2e%hpGxIhp4j&An{-;#PEV4x7!iM8>H-k?0LJRq>Rr>(Kb?^A!F@-=lZF8KNw3a_|D7W8v0EsDs`FA2A;msOAZSc zA1P5ZY>;P**{Pnd(rj+kaN}EIb@M5nLQAoF{@Y!j6;?Gr@miN~?OOf3C(>>Wvk&SO zoZS6D;-t)rSuAfJJaBt)UG%@TcVcq-!IFP_zAp0*tqt}2!Twx8?B<)homwi(7jyAk z`%vs&zbt)Xu)^KXCswSxuw6-`xXtRA~*Jft0<#aLbbN|(O`()l%+erQM+1wpD zDi*eDeSLQ|ep+=)GbAH(Qx}Kz>HYZ%HR`Wtah;lU|J^^2m8%}<`CaZ1+H@*7-tpRO zo8y0N8n~Y6ope($J@NE`fY?m)2h3?DZrkqZ<(T_z?g*||Q4IF~^wM-&jHPK+quYzs zd0H&%wikcOKjNyJ-`rbx{fq}U-^6>x%Nuv~wS{};7@lZUG0ECpH!XJsgTVF;e{~Ml zFumfRYxHPZgR>%cz~Wm+@6SDbp;uS(j7-=fHN%bRb{2Dt{+{r%o@uz#X<7Fi>+KRw zT(XhflVjtS?{8arvA$=?i6;sMLQ7|Qp7V>$+RI`nyz^wxs?BHOlaqzSwC7a3Un-+D z<&?;lZ%3M@RxLiU^yqZ4MR(4&_iTD7m##O{Tl9RtAIyY@)=fh(cX9j1m!JB;1awm$KySZ8}ha({nZv+DcGzh6vTkC{9_AmkOaWwYE) z?P=N<8LzX6I(F>yd5~z5QRp}4m12E#y4l2gUoRQn@y@L(UTC;TIdDh)!*$m@FR=C| zn0{jV5wRwr`hqjt*&Eeb?KQo7MN@nJZu!VK=ccwq|b$MK)y~o%+VzZsi zrD;(qYHGzh&AGQb-{)xkzP&3}aYJBh*7i*TQg8N6{*$HoJE;1Lk#R$`+pnZO$pONb z>nC(3c0N+O=sRyV&t|Q!oVJBmRvv9?ZDjcW~3>%%?%)6z}@mi&yHY}WwMmnk%#uohNf&p#TX692 z)Wdg*jCaoNO1xJuo!flxy-95Mo!_8(HkSEZ2Si<&L!BK4OL$f&wg+cb(n}tpUz?SJH$Jm-nsQ z@BAe;)q>%ou~Gtq`2RDK`TGk?cm22${%>=_9nP>Y?V5`{ZXuF9ei7F^?-g)8uB<on`zF2fug=dr znQip{$Q9F1v(6@dzbUc2WZIUh^;U^{7jPLn)ZeyVGU0G$RZGJJ*?mcd_r0nG4t@8y zTk%5jx-K{G%&v1B8_X8V@gHB)p?ksfsMg8@(p&hx?25}WSTgfMidC|GkB>>N5~HxF zId{Fnfsn7ql+GXe{?&5JCjH4xaf>B`jlF|gz8=0&UwN+QY<{C0+jb+)pm$Yo5OZrrLr6{g%s3Hw2S+;* z`85k4t(xqo){*bv^7@2tM5xClzIjU>KD;X_`noM~w$W*BuI^3Y60fb_8t?om-D4}Z zBzgaZ)kSF&eXrP>=|tT4J*j+6=ar6EPkJtEpDSD>VY6awfBoj>V=HcF2)C~Ay>?`t zf!;NNKRwffckfjzJyKh3c+*8ha&1J~8ufp7cM6^P-mx$*CuGfwy&dhBq?JFr&Y9km zBf%0l!&Pxok;$ZF+qJc}-e^~{#0nVX{&Q;KAa=1z6K6uF>4|B|<}U+GMn zkb^}ohMNQ>b@qC2?tQc5puqpRQwnpP&6a*T{IctlYoOZ7XOlMNUb|d(l!I^azLG$O z(hVNk3zqOdecF>Ix@2dlr*hVhYq2|KYZTlJ+$-9BSmvC*X?FjKEsMOe4;0we>#KRb z-gW(O{zL_tOG_2QAM8}LEfG~0nVGWW+YjG;?cDNzvgvO<*mA#W*L-fjydlV2LsHu$iur-_ zyBLN;^@>yLrao)lKf!-;|KlVJmm}gQPZo$s>(s?e^sZprIrZp=sX^-)R_8QCGKJ;* zGI={)RXqE7+;aUsrJghA6Q90V{cNW0&quEoR;qn@7<0n0hjC7rl0vBI>eD}4427P~ zKF-!EDA^mbOZdn|##L)l_G#?<_GLK>?;X=Jn>ng>#}9t1f0qCDMb4psKZ`D3nqOW2 zWQT#Xk>X_oc1OPOi+QEm7rC<5i|;qxl(WihUgYHMb%ooOKb<{^20Od3ZvFZEk&y{NCi)_qVSfTX=EmqE}nb9BK(P(2v?1wI?RjkwtvshT2%` z-758QuR}5%JD&V|y7%6?3C9w1a~Cw9Diirx_wV0r&p(+R*2+uIN0rU2{~vWTQL?l6 z^XX?ScD)C7y)oD!SStD8hK%bM(Xir4{@lU$C9XYPle1XVKDgR&4foCy|Lfxq?@Bp( zbZ=s8S5N)sPv&0>L@K1-?q;8>q*a&lL44<&?!8(<&v&`aU43V})`JVDA6Lu%nsz?Z7nEiT6Jo=3HP_-Ic+|Fwc6?h#kK`x z{A!+kV2SDVtH5*Txya4~OAP9^CC)r` z`p$n3Mrp_G0ym6jznQe^`Nm$y`O~hf-L$_i?o5>RdWJ&pc@pX;zXVE_>D{vs&w77a zEW05jF=DmK3h}Pl6X%p0pHq6q-Nlk0x29y*rPCeuNA>+@3g%p|s-2!{xh|LES>5gx zQ8KrF*Vm~@{oNB4_vhHQO+0&*!UR5F)n&S6BKhOF&+PK#i^*NW?J))~53^mVmg{|K z^J#AS+EUK*lO>E=d`gog52fvki!rR9+S-?T=)(kkwuf>Gp?zPzoSP~#JN@SVoVl}j z?U;&pXEC36Qj(YOM{JFmX1-}kVS(@EeJ`sXym&h&QYxX|`|0|HAByhDPBfbL)^=A> zPUx#O7OQ92C49duRmYKQ-23p;>4g0TC;t@9-e<^XC;FB1PPw_-tgS(OX*NIFO!{Sn z#iyE_XML2F^!BWU8jpgPX4hqvCoNTnCVZXqa%1J3hEm5QXF(mmBnOAW+lr^oYD_%K z=X61zr~iKG>VEdgC+hbJItH3`EtWRjkaa@1yzF!4NB5F^O}F?(^Vb(lUAJl0^|co# zy!_^w_R+H~$#wQjZ6;31rIjw*0{nzr-d?pjRuvV|SJ%3~hqY@>eca;@LX*4Y!q;ZM z)|qwXa*+jVdR^{8-5*<~d~`d@H}&(8`)#hlpH^gR%&*e?Kdqp9#^(yrdXJmpy^{44 z_VUPW5jxV=r}RGE`K;TjBeOR@P!D^dRu;SZO!(w%*?T(^_T3b*ySn7+cH>>U7Jgt% zbB>yD&p$M8%GWK;JXe`@ANFS0DgKqM_;W|V_GCqYmwTc7iyrnjXYQ5G=3{?v`P%Y@ zMNekfg|+JFgMR+k_;Kx{$gf(PZKg}sMqJf6!nbS1`=ZYm z4_|OTE&p93Z9-6meV36z|Fd<|O#eO<)$My_@4RjA=k2#&Y@eLy^2u|}&Voai*8MHG z)#h?gL2+B6iRevDHsM2uJ~=B(KDTsldN5_(o3nXMDXiBD^krx763uCs+5CEW!<}7a zZ-OV*cdlH%cIhYX5|{HfIXa?~X8(9&A1Sf-Was^pg6GOK#AcmS49K{{5;j-q^?Ko7 zMRRY)c%9+0KGOAm z62pIrS$UqEHeu^jUq7zf221|4{Hi_Kx61yQ+m|rcsEqE$R;+frg zGKU_9Y<``oA-3kWfV|f>t%u!Vs&Av7OV+;aw&ma47vXoKyy%*aWY_0chnJt9&3}EpyT=+1}bZ-_dl1*|qK7zM}E-JayW_ z-=EsAwsD0M`{v{MvPxH;+p@a+3%PDs`R3Niq^n1T=P@icyJ01K{Un2>Xo2ItCzYpd zp2%+ea;}zRZ|3@jw7e=)!HX-^3lrO0B0UN!rDfi(aBpMz^M|`}?}2N+^%LZ{e{1I5 zKFo4TMR>_eCC$x_{1cpt^1uCA6)W{58O zp`X)kE)bn`S$CnDyMcBFLvkGV!V72XZym|*?Qha7lk`@(usO&&drbpX^r{ zR&*(PXYX}$E)-K|EswZ$zd*9x zL|b+DSCiMox0d7zRdbZPZ&>GfkJEkK-zd9k9rGUXq=4glJmq47FaFI`@VdHdeaZb* z&Gm;@h2GowpW)!r`RtPww9C~D^b9n)6cqHGi&7IyGV}9X5=&AQG+eBV3=Ay|ObyKp qO-xOV%qQ>Fj%G11HZhzmqqBj - - + + - - -

+ + +
diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 7db2edeeaefde567820b92a060331ae9caf5351c..979d31171a64014eab3ac1aaf63f864f4f72288a 100644 GIT binary patch delta 51 zcmaDS`c8C10;igxo}ng}f`YztQEFmIW`3SaVo9okhKrSvfsuiMnURs9nW3er`Q|px G3`PKwCJtu+ delta 51 zcmaDS`c8C10;igRo`EKpf`YztQEFmIW`3SaVo9okhKrSvfuW^=siB#niK&T^@#Z$p G3`PKwAr55# diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin +++ b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 714b7115..f5eed6db 100644 --- a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -4,12 +4,12 @@ - - + + - - -
+ + +

@@ -25,26 +25,26 @@

-
+

The - LinnSequencer - is + LinnSequencer + is a state-of-the-art composition and - performance - tool + performance + tool for the professional - musician. - It - is + musician. + It + is - + extremely powerful, yet @@ -53,17 +53,17 @@ to learn and - use. - It + use. + It ’s many remarkable features - include: + include:

-
+

* @@ -80,18 +80,18 @@ RECORD, FAST - - FORWARD, - REWIND, - and + + FORWARD, + REWIND, + and LOCATE - controls, + controls,

-
+

- + © Each of @@ -101,27 +101,27 @@ contains 32 simultaneous, - polyphonic - tracks. + polyphonic + tracks. Each track - may + may - + be assigned to - one - of + one + of 16 MIDI channels. - Simultaneously + Simultaneously plays up to 16 - polyphonic + polyphonic synthesizers! @@ -157,15 +157,15 @@

* - One - or - all + One + or + all tracks may be - TRANSPOSED - at - the + TRANSPOSED + at + the touch of a @@ -188,20 +188,20 @@ * Exclusive - REPEAT - function + REPEAT + function automatically repeats any held - notes - at + notes + at a pre-selected - rhythmic - value. + rhythmic + value.

@@ -220,13 +220,13 @@ ‘chopping’ notes, - + ¢ Optional - SMPTE - time + SMPTE + time code - synchronization. + synchronization. * @@ -246,10 +246,10 @@ use LOCATE, FAST - FORWARD, - or - REWIND - to + FORWARD, + or + REWIND + to To @@ -262,13 +262,13 @@ and PL AY, - _ + _ find the desired bar - number, - then + number, + then Start recording. @@ -277,12 +277,12 @@ play your MIDI - keyboard - in + keyboard + in time to the - Sequencer’s + Sequencer’s The INSERT/COPY function @@ -294,7 +294,7 @@

-
+

click @@ -304,28 +304,28 @@ sequence loops back - around - to - bar - 1, + around + to + bar + 1, from - one - location + one + location to another—in the same sequence - or - a + or + a you'll hear what you - played—only - all + played—only + all timing errors will @@ -335,11 +335,11 @@ For example, you - might - insert + might + insert a - copy - of + copy + of the @@ -351,20 +351,20 @@ adjusted or defeated), - _ + _ first verse between the second - chorus - and - the + chorus + and + the bridge. - Any - additional + Any + additional notes played will @@ -373,39 +373,39 @@ into the track - DELETE - BARS - operates - the + DELETE + BARS + operates + the same - way - to + way + to remove - + —existing notes are not erased while - recording! - unwanted - sections, + recording! + unwanted + sections,

-

- - FAST - FORWARD, - REWIND, - and +

+ + FAST + FORWARD, + REWIND, + and LOCATE - controls + controls . - + may be used @@ -416,11 +416,11 @@ quickly access any - location - in + location + in Creating a - Song + Song

@@ -428,26 +428,26 @@

your - sequence - for + sequence + for spot-recording. To overdub a new - part, - One - way - to + part, + One + way + to create a - song - is + song + is to record - each - track - all + each + track + all the @@ -472,10 +472,10 @@ record - record, - the - first - track + record, + the + first + track will play in @@ -487,18 +487,18 @@ basic section (verse, - chorus, - etc.) + chorus, + etc.) in individual

-
+

- MUTE - it, + MUTE + it, or SOLO another @@ -519,27 +519,27 @@ to “chain” - + tracks may be overdubbed! All - MIDI - effects + MIDI + effects are recorded them together. CREATE - SONG - will + SONG + will then - automatically + automatically

-
+

including @@ -548,15 +548,15 @@ modulation, velocity, aftertouch, - Copy - all + Copy + all the parts into a new - sequence. - If + sequence. + If desir ed, you @@ -570,22 +570,22 @@ changes! even set - the - last + the + last few bars to - repeat - infinitely, + repeat + infinitely, for a fadeout. - - Editing + + Editing Composition Without - Compromise + Compromise

@@ -594,13 +594,13 @@ To erase a - wrong - note, + wrong + note, simply hold ERASE and - press + press The technology you @@ -612,7 +612,7 @@ complex that - + the note to @@ -624,7 +624,7 @@ plays in the - sequence— + sequence— it interferes with @@ -633,13 +633,13 @@ process. That’s precisely - why + why when played - back, - it + back, + it will be gone. @@ -648,8 +648,8 @@ also be the - LinnSequencer - is + LinnSequencer + is designed to let @@ -670,14 +670,14 @@ the SINGLE STEP - func- + func- and edit while devoting your - undivided - attention + undivided + attention to your @@ -692,7 +692,7 @@ within a Sequence, - — + — music. See your @@ -713,7 +713,7 @@

-
+

* @@ -729,19 +729,19 @@ clearly guides you - through - all - operations, - If + through + all + operations, + If needed, the - - HELP - button + + HELP + button displays additional - explanations, + explanations,

@@ -760,26 +760,26 @@

-
-

- +

+

+ * Two FOOTSWITCH INPUTS - may - be - assigned + may + be + assigned to remotely control - many - of + many + of the commonly used functions, - including + including ERASE, @@ -791,19 +791,19 @@

-
+

* Two TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses + OUTPUTS + may + be + programmed + to + output + pulses at any selected @@ -813,8 +813,8 @@ © Will - sync - to + sync + to standard LinnDrum or @@ -823,21 +823,21 @@ sync tone, - + * Utilizes ultra high-speed, 8 - MHz - 80186 + MHz + 80186 16 bit computer internally for FAST - operation. + operation.

@@ -850,10 +850,10 @@ be specified in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at 24, 25, or @@ -874,9 +874,9 @@ * TEMPO - may - be - entered + may + be + entered numerically, adjustable in @@ -895,40 +895,40 @@ on the TAP - TEMPO - button. + TEMPO + button.

-
-

- +

+

+ * TEMPO CHANGES may be - programmed - into + programmed + into a sequence, with - smooth - transitions + smooth + transitions if - desired. + desired.

- + « Any TIME SIGNATURE - may - be + may + be used, and may @@ -936,7 +936,7 @@ changed within a - song. + song. linn diff --git a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin index 61f78d82..16b617e5 100644 --- a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin +++ b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/stderr.bin @@ -1 +1 @@ -Tesseract Open Source OCR Engine v4.0.0 with Leptonica +Tesseract Open Source OCR Engine v4.1.1 with Leptonica diff --git a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index ed122d0b931947adc2d4ce1249b1160119808143..11a1c3114fabfddd0850dc781980fc90d00b3409 100644 GIT binary patch delta 10125 zcmZolK9Ia&JqMGe+2jX|%JpkvPxqGkU9Mm{JqsRFFryu>t-`Cf&l&dB`oVfr0>%h;ViHu)~3cfaTN^JgCpeQ?xn&d!9DKUzZ8zq(Y@)UIv2|J1>*#b)s-@ept zlVP~myt#F`?eh~>AO3uJ`_oU``VRX~*Z=R|KL5|t)$hN%GOo^+tvLVYX8P%=ha#Lz zrl;TTTW(`NO+UT9PN-_hm(&y1^S^ zDlBWYq2;^L;}3r-W*d~f3co*5bwYvtg>z;5lk{g#uGqS*`p-|XdmCFm)(HIBvTc!V z@f)Lkmqq63m+9;F8phP!;rpU@t8}|TShxNBAh}OBo`~p6-(=g+zOKH%J+1E7bI0pH zn))BLeN*V?Wfau>#Cn^%*Ew7N^>ww&&$dl}>fI-1mUYLdNi;_L->W98_YV7yms!?@ z9Bgvqt>m4)`SlW~&>*Hm;-@|?o}JOm%VQU3s%Fv_?t1gyv`%lGzmhZ09KQa8SAW*d zg_>99*&X z!IGHCb$=7YROa7fn?L;*)2yvJ?^e0KDBV4;7%_ zUvJL;?62tZujR+GE?$|x{e_D}*n>7#nOV%c6LR-&;_*JII-%Pj_p$urxOW8|3lGfv z{`;m{pk=J3$8KkLp<{hJQ{|u4K4^XY;o*io?^I(Z)ul(BJik!raP$xP^|9++;zU)i zHeY0Bv?P~w=6E+7~WC1>&Aiee)8WY z{C;OsJ9*9INn7vFe|_iLJ-2`@AJ@!a)NWO`VtX`!ZQhxN6MJs!6y5&!G49En_kprE z&I>)iRU2@ATlV=Kt9H3~-*Ejb`ormn(Kce(r(eUjHEGv1k5>0rpZhRbSm(5#bc8_hMoiHTi%Bf9^3gv*)7`@;01=-d#J$&PLv~)kitz+|EP5lih zHB0_xEcb{wADwz^o~!yD88;sH_Rd4~habho%niFe>7nHjTPC-T@RO7KC;>rBaq|Xw~w-I$w=>pQz?S9uJKol%=_(s)BQqEN+QXYzd+$)ZNZyC{Mmh;0!MXfJWfAiQ zrcBiF`IO0Q&;Llwfz5!4d2)vPwXS@hKg+^$6FvLtc`qOQ>ny9G7Pc~WzW0@uP=${z zfA!nFllRx;s>@{F_Vu3qe*P(E=CD6&{#;)C$)`%|*P$=hb6K<|u~)|y1*=GNhc9_n zVYJ7_d#1@!h07<;{P&stC*s2DT&45h_OKt~b(1R)NU(qNjJ;O0YL!8S0>jj$U;BKL zPH=5)2~1k0ccp$;LuG-nq3BH2?0F35S?%>V&Dr=iWIx+k24nTahvk_+FRYz>wB(&Ocfypp&RQRO4gE`A`K@p}`fTTn#r-U<1y-*V0#8SEYQ?x+ zbQAa)Ty!+*RD1E+pkEd@)E}tNwRk4|nuV#cz1uwD;S>*;F3Znuv^wY)riS8%B-UEJ-cbUv+dK?gQlN@yiV62EtIogx^CyI0~?xFebqgFrhcx# zUHO=O3Rm*wHY|@jm?YY|_2QpTEZ4tmy`^oy^HlTYvA4!cy)8wuVv*I~_ z$02doq$mkb9?uSsJ=5zxo?z(a5_r1KU88(9@8K^QYq*7-JW@Wd6a1_a^6UbWiBrz2 zcD16X3v3kEf8p7F^s~jmk{tPy8$QThu$-D-fBeq}yB)^`olU=3)q8dKY&@Es9lkE> za{RREJ`b0jn{eh%kL0cV&}SM45(7*HV!8|eA7Jc1a9(6)<@v(QKPL4h`ubo?;oD;2xnO$$+my0*pmld?$>%1j$ABhp6YEA|MARJ zXOp1U9jEJ+m)$%aeA>F#WE$V4xf8kV)^6uDTC_km``Mw*+dJQsl*Zgq;%;a+`zEkA zuaT7_&*xRlsgCFq+BGw6(>PaVCZ2w~l(B(X+x7VtmgY%rm+ba0oyAbPs5N%;hEpY6 z&J~MZ`B_fX4ouy;kM&Z#&i$G1MT9=2CY=nL_xtJ32bGuO|JJu<%Wf!{{Cs1Y*v0Kr z%vm|4*$S^uscq>oQd+2W>|z32K(oR8h+Zz=$cDM^iX02}%v&Y&p<;c*(c&+0tUEl) zr;4u)V)S8Is^k5DvpQCKhGv_>UfF3eFF*OHR;KGjKH3y^In1Pdz4*tlOMFjSn~XaY zldBrv+b=!UII(TV)uvPRiKq7c^wEhj+^jXheR@dP@0JcH#pm0SQul4ovYj*2G`9B; z1Gng$le>bMIpXime6#Y~zTdu_)mds?3U9;~_5AWVeQ|eU74LeZpeYhxT4$uaQ+>t9 zXV5Ag%5zq``{){;|0fo2aXG$n%JKyZTF(}mn(IUaSx>rtEcvSS|4$Q-X!F+J3gJ4U zxq)5hhw9R<-44cG=7kT`_1w>MDQv%1cBRQmZ_AD)QLmImZr5f1Xg;Xmop*QB8oy~8 z3A|-50v%QhoSU~k`8}Jx!|g>=Zfrl1Xtw*9!h%m6>f7e-;M?^kZe<#yg2BfaId#De z7cTik%@@sFdAn*Xv-frA%% zlb1_3+Rj{fuG=LzFwa2h(+>$vwm&|JAK$L_nmL8ZN2zqTUk=f?$UHmTD%3~n^1s$>ZnZZ|v}QFiV6pPE+oYoYBdFSmt8{u6Fc zS!9x%2yA+^Q<0UX!ltjO|w6bFS~auF>?8bK!QS zxvbnb<%HW$b+L#W&)5`Je9r&NSGH6B4Q^@;77AR8yl%~6nIss>ev0+nmYecH>{_bz zllD4Q#H|&WHZwd>%jopX^{372+j=IyN|<|cUhn;lznacCD<9sn-C$Syf;mwP3aj+? z6ef0D?K!u%#@^q3nQ?RKlCOOgv1e}O2K31v-gI1iM2Qs ze46yxM=WLXbEd+$pO#2Hcgpro&NB(%9jI@Trw+!XY-o# z-&DMKol!)iTx!eow$n`8rtQ_zZhp9ZuI7}LlUC)Zr`Nph;W(0W)71Pnv($&mR&!5* z^y9ieCu}P)5ZAboTjH%=zGvPvS?T#x%hq4*nlG|)dum(v-0USeac{S281}JdHd(wE zwD!5)`$@9??oG3me!UlF<~r}&V4*fyb@{%PkHkK5pHXnHI6!6>$`b)$1J_X*K%X3m_z^9R|1m_zA1AUsyG?$bC5i{sXp=5y~U}gCvJWqz2sDc z=k4Ir!Kr>*4gS5_x21?k>@_%^V3Sc%(vyWovtE1 z-!DgawsdcWT*8`3?{^yd863Z8|DL_AC~T%>%-#N}?`*_`H%`01*QLJrs^GTFM?wtl zuMRMB+B%sko;{xTS7vE!z5UkH++s~9UNr>DzKJVuP@QsF@v&B{a9hUAOm_jcS(i_4 z+hgEp=BDH+wWe>CsHtGyftrbxyoNLHvpxUZP|UnvYE8y2x$~Mg{fu6&>XI>8tm72P zZln46^Dmb3-EV&VnDy>@QDe>g|A(rZo(k=>3fgJ4L|Z<_Yw?4hZjvm*!k6nCH_vl1 zFSIb^ot7-Me2SiUS?{GQ({>vP-eAbSyU0x2{`;JchKVbpMXEZSyOad-c4YV~`HTOq zbuc{pJJ_XBHBfkZUW4lSZ)vBunX(z|ylZ(Yv7+mK?c*?s7n-+H*Wdf#;B(~ChP9rm z)AN3-JI&}^Db|(D)|&d-O?5?R$D@S|^{*#(%o2O%atAvI|VJ&TAFixFgr*@AD{5beh)2Yi^5V zre*P^F5JB(F#F_T-NFnFul%{f5lajnedsXZhz)z@)~}E>@9T?j%|#WTLhiF)PEvB^ zVHB-@`}u(7)|+e7I4*Xdn|~~Di*Y5V=q80lO&3*sp2SUw+-U2wsWj&N!ACtjC$}6? z&+p(kE$8AyeOOd)qfqH0^8DUQA>M5__wcu+ z-PhpSWpPj8i@Aeu?cycpxL202XK7vf>wkWhOS+RhU&)kfyQH7J5%AURn_+bF?U$UZ z%NTEZ%ShV0e=L0U(OV>5a`g=^GktZf6MvT~1<98-9AEQd?!y!l*_%&g?v^!ta#a)J zNI7ls{P(^$#y_5h7}i&)S4ybQl+3=%^DIEF@9DMgRh7A0qb~n8y>)u$8ifw?MQcN+ zTf}5?pWDn(<&pU6?7?gEc3S>-n6|Z@wd}*mE0(XmXgp*8U925?tMbtiW8+iuy9FyY zbh5Wzv1$sqqb2KMbL7&x5A$_i%7q4BT5Z(5sBzb~Z~GWb%B`KH74@^*>+1sjRG)Uv ze)%uXO#F-CiUmcdT@#}pDyc@!bNH0zJH?UrPu1nkotC0IRg={UJ2%VlnMm6IekAj} z>V=u6(hN0q)`D$E9Cj?Se5AAU$?1=Z%KN8p-txDR0Jfi>3^!fsBWlNG*QwFyZCX6pO;#}xkV*=C$5v8p!`zNPJJoU z^;~AzFb1CADK0l2-D_1@&1TM4k>c!X{YqM9H5%2< zM5d(X`MNJzl{@irLz_kCt>Unt{+my|&K4{A9DefU-V?jy@q66-HG`K%vCQ7LPCn3O zVPerqZ?z6JpBFZ|#@TCsH?SIRca76rc;);OmbXjS)_+^++g+LTipkVu-{x3hEsNj# zmOB4V*j1Wy^k(e@|4o}DzrO2B@;9F2*zUHtL0nO;-m%Kb>RQ%1-Z@>LOqlO^vb>30 zs@VShjmz__4c)pN8?6rCu33~W-X#0<@cVgzuMI?Z>+uwET<=`Et;5V|;a8axi+9HT zIla+O{kBTTWv1=*IY&}ttd)bim)s9oY1gih`+%X^VjAcCXao7c+t&@Ow%$wsXTGaq z|GiB1j%8~mU9!-2eKX1Xy{h+*uF&iz~M%<^@H2P18LR zJBQ8E%PN<-j<30*`qljjej9SCr&V7Q>6qb@5PMzZ9`~Ap1L>@in)O*{-2I&r-%Bt4 zvF+Z%;8(Y%YU?PRzRmTDaqsoj0#h6obt*C42};_~eZH{mX@W~oU5C7t*p|*TkxO!m z?Y}!LE6hIei`|vyUe5e?Htecf74?J&Eiz%4^=WLU504SI)a?*L_!ib!@Qn>DU_BGymf@ z&Tk(b=BS?I+wsc!&7*mHd5Ye(rk@otzj@lB*Y<&}^F7v>@5?`JEVR1%!|ZZhS^VI=rFtQYk0h^v(7V+Vc~@R$85Pu7&P$$<8up%bTF&A6 z>Yvx9-aVSpzxb&1Ef(o2sUIIY(j$d;T|Z_oxmeLI&CWp0P}1y>QtTf{=%Vs;&Te#%z6H}-}TJLoTwEJmuIBv*X~zJ z-8=c%+a(`jUPV+-wm#(fi1m^|`dc0mb2;sW@{{W|BDCi({nOXASUR>h%C!6E{qJ4# zq9)&aRrk}abY9g0w{p9T1@m-%9|_RYJ$PvTBR}_Twf802Kgun>w@uD)PJ7rJx0T2L z=`U;AV>zQ=)^1u%@@qqzG`llSTE9(>>Uln~T+7#b@e+;xk{X|9yJ3<)fXn(1XsHn)$0vS)W|9Yjd=^uG`-azY|l-{bo;j z^XEg0m&9YQ4QaNq5Bf^1MSjXIny@sGVfp5?nd)bkW-7~RCVkACyfu2-$+Np2n%B5ljboRoB1Xn5nVn+JJMko|H~H)F5^OR2rk zeC}D-Ex9iw$(Fghxt$O&es!zfG^O#}*-6c9k*;rxxi(e2H)Z*%Z}R=@^9~o*$m_Zn zO()Hnyg*9-rp5N`r%$Gx*RELif7|pP!?L^59o1V-2AooOo_u<>+}tBibFZxAt<@KJ z?=$b-#O-F2zjvR>=U#8TdN< zKF4qQ=lr9qc#7SIhADr=a#l?Hd%)muuH>D{0v_A7nbmIMmaab2Ph75jt(~(W$fRvH z`>V@O7rQ^U4$d!_cEq7-vFDBMm#kH(d-8MqIXSRE7 z-a6-EGbj7Z!Yg+S-!&GtT>r(^^W4e#u*9v^%Jplvdll#$JDRdm*<;om7u7kF&mPV= zdf@Wh%DRB1$G*H!@j5j{VDH@96QX{bbL*_0@qWp&xvRQf%o0wSJ25!((d5Vm-fyD+ z-kNQ4{QjmjOv3QT+(#<&`DQd+&{rE`7@yue@bkf4^Yk^rlrNIv=@8rWM5osG9#++j%si>f0mji6Sqjt2q7M%=Rit`)bXVZ>?4L z+1X4q!k2VznIwOww9cN>%G}tqqWnhf_1_U~rE50zK1+}@mA31te{Joq-+giNuHJTS z@sGsrjt3ocdbP;3ulY3XLYQM0~eDHQI=b}IdmszW1D&uV=}WnDHAe$rMSwxd5~#v2!J4)f*B@e5yPr9dhJ2k1D3kriK)}~3H|u@chVxMGyAz7M@k7l7vAt_YrTe% zt@ll7DaXUk%cn@nT-iD4?OHDpo7Eq-XeVF%@VA4%V$rvCN`^DJ-#rj`{^QD7+4VPr z^K-fRK2Nzn<@>zPOCm#+Stqi_zLNjN7yRad(M(3$HPY#kV)rAvUdLRWAU$>AuD`Qp zOcT`PIlf1;6%A^^@`OIE`M9Hdo5IZ z)|t9qSpUqt{{GRAXUnb9ORQKzgUaHA?#UFg-s{mg{lx9WN_pe*t0KA9wNWA9q2(r(94necNLAIHvT>@)T?D%+#{@y#Z)Uw7NZo-5fp zYsXzEwLe=stFu4WYq7{5n@N)TryK*aR*`y;DKe8&B zd<<74-d%3TCe|(GDE!aQ{4l%wf_HXCccj;8M{kttE`PDoOGf&^cBN3stTQosk^(|J z9}{*=T&-5RB zS6{Hd*%`Hxxw|aoo^NZiSiNQVJe4Q^85kJ;|KA#$JL#?lkIhBDo5v)6r<+ZdNqt}6 z;_7rrm(fgr>u+_>rO)<$Wan6U^_AXtPUVlb?_Zt0yX@5}ugC{1E?-rQ1b6eCIdm{O ztv!8F>s^K?`zm@r>)dx~ojX})Yd~{zS%~zV6+Ks_O>e839(4R2C3aYaOXbm)diM;$ z-+s;tZb#;xNWcDdeylr#(W0-~-<~*cnf~u}o6VUS)@*W%7hUx)*}-|RzG%^8iRZU6 z)|GP3>pkGPFSzT1OS;>)gB7V!d{^wU&)Q61|8(j^XMwGTW%Jigyd)vR>BeT8-}3m| z`yG0_S!Ishee~qTnN-%W*K1q1sjad($|-xHzC>7Z!lfT?O!<8O#auoh@-jec;Xap{ z1}~T9=*ZkV>lA%CDCJ6IyvP}+iOlT@0!;R=*!Bu-X!1xbkF@XvHZJPYp|u!#Z`s- z_$CTk@-MI79C!W7rNkXwb2FnCm1Rg@mDHA(IW{R$`@xQ$?DaB&e0Luiw|1`B8oa*A zWtx|K*OY@Bo>tuRQr~uBfz$dq6H}kv)to1P>cIW?ci+vFD5`T&RZ__Nv87}0Po;+^ z6HW`a$Sv5$}5YkQpxVcN~vcsXuz-PzqCM|iUPytdWbKIux=+~=rwgnjYW zWTzU>%U?LWD(3ucGXIckzHHv6Q(V$Ydq1AqxlZhk^!1?pfR_2sFE9VDwd&9#rykPLF4X^iq+2kSi^btpDf_tx_cz?zaz~lfO}t2Ds%m3qxzU|loRw}?j4P_x zEmt$nntMIRWRi_DN5G^V_qFeotP|b&X2tqDtEc+O`JP<2elzPi=|o=9JyU#lKdLwU zQ2o(sO6iQtyXNgl&hIw<$tEMF-T$n!L!j@Om2_9q(SvcvSJtkRd%pJ1tl(ymr>~@L ztxbNWz9zG94KK^I3mH+$rIU>_a=4E0u|8cMt9f0pG2Tzof@@m7_x<-54l+vL%7{r_ zw5G28Mn_1wfU(T9;)ta!E|pdm4#9$#?oZlNzoXE}l`TwmMa7f@Q{wDT80)&kw=KKm zap%S3KszP#C*4t=cW&LVxGM6X>MX;$t3JQ>xj&dQLDwwlGxr{rS$^NU4%{gVjF0=Y z-}Ig5e#JX|PWq?coV=#A^LL-}Zv3Z&LskK?sGvDqJ4?nR}pr+y3 z!K~|()sCF5zqILpNaf_@*`bT3$*<=w z+o7&2M~(-4l&`Ssr+ItjpE98pKh*bv29ntagp^HGCVQ zkBId?3fRHt_fD;D9_N=QIvsyndM7dI)xVo$!)2=z`C-xXH|Np~>Mr=tu-h{yt$@q3 zBI3N6hoq!Pk(ei2>+9=LK3Sg(#UIXccf8H5=b(LtImNE@fT)ZCb^! zsOx;^kpl})&*c{7h>4o6QNQWa-WzuA{M|Ro*DXD>#zrRU;hX#ay?!$tJ-g3)*8C5Z zp3GwF%8v#{|H_NwUj0Y34*Ssk6b>4@ZNi8WH5Oq-ruRmwI>U#Wi6<=i5--k)oipJiPgn7;9F z@6M2WukXHCt<@OFC2>*yChsA)M4k+%^E)4WPdYPw_bj0|%a?whHUEUmlO1i-Y`1*a zqVh0aN?z|&?4Mf05T~Qn{_*Qn%HPJB2A5|D)tfRl^;bpYeXF}SU(~|Mys+k)TeEfl z4$X}>=7*k`Et_av<~-}InL4LHg-fJOB6V5{8eW1;^^k zdGdItlq!3FS!cF+baE<5BpuC?Ya;K7Uv9q0{lsaeS8-YItninYrj)NYpOfL+6VR{9$-} zsxm9bD(#@tm&JWYpMO`CSFX=E$<8jWT^E_8vh#MJ{O(k< z1A7fR|44j4rMl&lKmVZvY>RdKF2=Pss2^*|K6vxhnsd?lua9kZPJCP_Ibp)CvU@ep z%CpTMlt15W^#r4@iS3<`McLI7`>-fm;T}uhzz;5uzug^ z&bv>fE;JrEKF3;S%0$z*_BDs>|C!C!_fQiRBaZ zm1`%~?^vF^BSbl*Xv5?#=BH}^MP1~lEs37>WWLaYWWN7B4L9OB`b#gA8>{I{mdtK` zx*&X|(-w)S$ir=3vyG?5?4Dd^U-^RL<#DC_t{ti?%Xt%5DCM(io_{WvHS;p#>=M2` z>@kz;p6bz-osjnz{=U2L4dqWQeEx~r%899OXQ?JFn$ym=}c@$}NA-e;m!C#$0y z8&2!zOysXSEY50Zyv{ssjrNQjRWZ{gd#-ExrcP4bwen+V+I5q@8=QKxCusjnob&1X zyC-X2db9kU>b>C1eN&-V0ll5NsV`<~rvEuuAtAf7DB!@qf8quuYj#Yo(Jxms)HBrN zQc%!$E=o--$;{7lNi0cK&~ULbGB7eQFf%eTG&8g`GMg-E5W{R_Y&yBZU?a1!fx%=Q RLmw^^Ljx{VRabvEE&%Wukahq7 delta 10073 zcmX?*+?2dwJqNRerN!ijj7s%OVo$D%u`Ig#J^Ts(9+T&c20Sa4EN?Am+hV+@XmR=0 z4E?`f`xGRfaK_!MS!u+x^;Bpj`y|6T7KV4~?IxA-JO7`a{r~s=|22P(-~adHtNH&Q zU+wunui0v6cjj`%ul0}Xf5uxrpMP+PJKsym+PXi#K7Z8z|HDB3*~2%FuiMLATb5qW z`|#uIdino9e}pF|8K{Z=68``8V|;)7|9>3sX7((7{`c4K!?S*D`Mmt0_1`s@=AZwk zUbXMcjGUMLZ~ou;H~UYudEU?M$J@6)jLdM9KWFn&^!M49@mDL${lhDF9(tNCclz|- z?eAvJ4bGn{YkREk_+LZ4o~iF5Z%r*>-gl{e`RPpE`bX@47rM3_sPA7rsrr}5?a-rV zTW@cg_q}N4V?9$F-=Z4v*G+d+XMTP2;@H%htLHa1ZOFg&eEnt3-|c_&|F3>Ef7&-a z>CZkV4=Y};JN7a1z5k7A%eQZxUvT!k!0G@0JCyGp@4iy;`OAi!eTm;befZ#0)+N5z zZjSG!$L>F0_PpP(`LBNFWufan{O1YS4ZX@uOQ%jX*fy_)S?-RU zSbKnqzy$dMd6Df7Rzk=3>SWu#NvXa(|MB_qz*Qd4?e1+^l78~F?eovqH0xO-Lsp)b z+8(W4m~{4BvTw)C}(3L-+cbgU&#KAE!l!X!p+l`AJ%wxygu<(d{*)mmcP+_vB8Sz~f%x>CH6Xt^S(O#N)kl-h^y8KjWiYtKW_T->w%nH=Me@ zc5!D?xbBPk+n#|l^Aor~?PJ?is`%C2JUZNS!J44UWot7omJ3LPzgYfQcUG_7m#Aa8 zhqTxG{9DxaFWLT_@YQAaD>Z*!@xH_%d@Ohu|H8?|PnYI}8S8n}IIdcf@F7a6Qc&fr zmDTAx_4=>O?q3Z46M5&wal1qL%4^r2*lC>{_~gXAQ>S-m&OfOcJ3;kKaN(P+Ir~?2 zl&%&%qkd}Xn;Y_v{qn4r2h7|0W%o^so;LSJ{|#Rgb0!GuiidvL;#hm}OIpq3ODO_h zb7pOO?|h;?@%#O_-!8p1TC%Pm`==Z>X8+Q;jdA{?gw2`3_4|FL6Jq9^GdR4ub^h#7 zvH8V?jML`4*unl&j4dzy{`J(GJTv(a8r`#c#b|w1`?$WSX-oK#UjWN+i z{KfGn(!WaAJw8>bo1dV^Eh^a(mANOe`a_Y)z8mR@wOYuwoC@4PdX zll#2m`9G!3C1*=Rzuo;X`Td56n`3Q8ZZ=N6j za{F65XEK@#ynM3WVBZY=vhZ_$3!WG4w4GSL?N{Q>=U*EQR?VMwjd4}@M)^h&bNM1> z%?n+fPmU%=KT>1q+@^NsuZid3%bMn;fBv&3WFDU3tF$)}+cS@5Oi+Iyout6xR0FDdN~XEbg7xH)W|f6d;uO=2vII4)h= zyw37+*iHu>S&@V_%Ul;sH)GkM9~r>h_RvG8;?peC?WwmKC-JVxv@e^GcyrnN-7FVB z3e8B}oZ4#ru_5hIzE=C8PnKo}QlidWSsQFJdm4v~qlL0jz5DL#q>{3Khvr<2UQrY5E0!Rv}M7I_p{X%eutW*oKAY& zP&6aw;C5Eo+ljZX$7MQie)9Bg^S3zyEqg7No-upNk(Tb@U2)xchp?m8ktUNOmBwR! zZ*`VDOlslITy=Tbtp0}jiwBpy_DGsDTZ}t4T7akLSeKLW!P7FEs=j{yR#xNblrwYQ zXRb@T&(?C@E$p>B;p^&{zrpvSYKf^dF3W@)o2_k@2g`6V~E>BYhCZy9&6_5Qrg#$Dg}_*!6F zggTpJ!mTAK4+Or)e&9$DO38c?D0g_-`Hy=x$Yy+w6-m5Yy>Zqvxt00*PTO)ziErL( z{7~(9BE!t77Cb$NJ3>0&T(8=_(DRVaG5ZsZ<|lV++xBN)> zSm$becVG_O<44LrHwXZ<7XJ&}BRAq$H&)lUCo{jgxV?@c+u2Yd!u zqB*{CYPupP0-dn=o`#gD;2Uk7Vfnb!seqklf-WYj|$rCC(?>Nv;Q`T6^4iX@0Zl z!ET+SY&}2zoQMy(!nts}!Q2afPCU|RdU?RYi%;{Gb^Wae<;k&p4H@pr%d>Cf{8{O% zestHvrMXH56JEzp?wGpzL!a#H)+z(%YTXZtEhi40xiTSp(~JIlmd7PL79YB#@xG*N z+1`yR-bdn2_S#5!ZAcagOe;L}B>Zw?g6MMNofkFbw;YYj5o%1DzW(K+RD+WufG_VnkSmx za-heR=glGRqi)MT{7|U&GPp2_tNvo=w->YDZ!oK|`mfiZop?;sRv^T)J;Pf2)F!^R zrVnM4FVt@H)zZqS`{i!_z3Wq|l6b>VUNl3_d=4L$4=q&R&^fe2$+>-LEq5%(1YCr#V-OuSkj5 z`9$=}%+OL%5#GvkQxdwm5;khTezYsK{hN-k%B%Vr6WDV8gc)^iD4en6)Uls$LsbG^ zJ5MvHnOYe2G01Dv0*?;Oy9KRh_uY*1J9xo_*+pZo$f*^#npVtkE8;@mtFz{N?nUp#zn_|VXhYO=Io(%2 z$%i9cCb|o}?Wx}_v&gSOZ)Hz%_oD~xYpdRDWoo|p@!yU>2lpTQ7lmfbyjxl&b7A}A z)06+WgKs5wbXJL2Km1kg5|04YVVtFFY%HKuKWoPWvU<(yRh} ztiPrtDn*~;J8R;$SN&4MGiE1^`p(1Kc5ANYWi`xrG4t;f(Z}mozW$-JQ8m3r=#AXA zJ$L&qF1Z(Cw%*`pN$|cyvu+Amo$I|C@;ol9|NI2;2gCCzwcbuJ%esa<56MZvCLYTe%Iv)@`pvhig| zsH=rW%6l^`x@$Kt%&1p-%>y;3_wOZCobn|!UB2BcTR-od-&9V%tZX5tdYzT)oXS%C zrtCR&srhe7%pDW%=`Yj-QjH~c-H=;TqV+8DUVT7&qQ>Xl4#}*+kvFy-y7M|L&62B2 zI_5*RrE#GHW8Sf%cfn8ccCIM8c)j#=yk~^rQkkY73IRr&LU*jLPu^m2p^mGHcenA9 z0G%^`IG!KBR#*7wl;cB_$Lfj+drpA)(+u4&uapexwI)1i2KI<;J1B;lIn9}&vcyL^@OraPtYvI|X4=DK8g`fK{%if=s1)*fp**yG=A+Np9a zNPpdn)PkL>RJLgBK9uoPGPbyB+i!27mpcmmj>g=$c}Mp73YL0*E^hg^+tzQEYrc0X z+4b_dgQh1eRh(RY1n}AIl+M%M$Uc4Zg1{H4hPACKyDsJ_NOmV0K6q}Gaee*SMG0=| z5wn~HoSV)$Zn`S;Vd8ezX4RZ|b=Pl*sahW`$~T`a^giyaiDuNs*2o(sGp`CNN_-RM zD>?AyrT)xjkCzD^8zgFP*I#pc$k1Q#;**Bn0)ZVH-YyYQ&vg}wkUjn6kbw9Q` zw%A|VDs;yH*p1A;(rt7cZ%IV|CSD z%`}xqCT#YWRU6l6DHWtD+_e^cvSUetqg-tL0;J;Euuh(WK z&Ff9mdQq^``_Lu9^5!V5>8f)&xz4-JsTa94iDi1?wKt6`COB>BNjW;v0p;>dka|`rdW+&lU^A|Bqbw z6n+?_uHE$2>#FXApga7<+I6|=$KRili1YGvs|ufY;J18fjq&94#E7FiK1yEQUZra! zXc#-O-ivQlh3y;8XU7^2c~1K2rx#nAvOZ$RNjsj>*PrA=X5L%<=ktn#Z{14LPxS|+ zFh6@FUUzR_c2+ebE{w|2vb+A{rZ?2NjByQc80^eosa+tF`D7aqF_@&)jlW(%j=I5D4>CQJT zg|~@{)=h0t)t;lA@M7A>+Ehcg3_H*A%fcVuACWKOO|EHKQd#VB`WAoe^3u1lo9w5b znEJxu$)Zc2y`%UvCtMS};yR^d&zExmam9!|*RN0X z73Vp-VxwGB#f0E*7wU!WUyEJ+IwRF(?n&>`Telv5*HBOTc~GX{w{pC8?a~`}teDje zIX|X(_N2dgSQA%xhH34NkRA6|v-e4raI95)WU{t_Np{ZPRoubGoLmy@viI^Pe1Ec^ zJ7(+TUDZ3(j!T?)dFQ@mCO{{j2xg6cG!zppuKix8m<=!u*x*bT{eDB-JU7=oT?CcLp4N4BmT^4Lt3NySj zfk9jJ{_Ptr^Z#{y?%K?@C8I8XlIP@W^`_D1wmhHdW_Xe>XnFWt+akA#FV=0DekRC0 z`A}#`q4FHBQroN1h&Fu{%1qP2EExn`PPyxxMbM zoin}{Z2s)FwYf{Er|D7PnO7x&o*{EZ)=U0bd+gj+i9^NZ%dABWc=@XO z?!uhH%{c4ptbCU7q~_r;)|A!b#CWMq@o)%9A>D=zlRg#118 ze#61Yu4dEs<@W`7>N!R4?fb%|YiK$1&lE9^S62=!U2AmMYiq&QT`j-7e_wq(CEDTM zG0_EQfufC>-LMwpt=U_NYPfs@<{a{J*nXCVjf) z_+#q&PwX?QW^Qy5Sd*7?+ovhtMrp?UDA7k^YA&#Xt9- z+ftbweauX_rRK82#-PhDi>0$p--=~VU-B=0zfjN}n{_KQ9UamOmF|8QRX#j5t0Q4{g8h)4Nk7_WjHBs;I1#cQf-ZcdAwh&HOO?S(w~LgL5;S zHmqG2|0ddq-FuhU&7LV)vlPn3h?6eDnL)KVu?hEPEyNN%F?(Luxkp;#y9Rr*TF^&${|-LeSw?x8B8g)W!ZV zc)yzi{2dopaOYe%Tt5!*#&* z*RQ=B>JOV`9Xq~h>(g?(qoScce;04RtDTiN=jDrx%~tLq=66;v+vXx$x%$VUEsp-1 zSvjIZ9%&fAZaHps!;)*mwJ`hIgKXR99o$mH^3u@zndlU0zL&)-XY*e2Ua@*-la}!l zR^5u3{-4=JZ^>ppJzP6|`>vZe_PWGa7a2X=tp6+Z?xRrGExUUk##lC0Ga!PobB zt<3jw#VG99wYP`7U+I z=G}gwvXm`9)J0F{a8Bi}#V^^d&E$laUl5G?%XjQtSL(ugebsp7WhZjXBId1dB+`?|HokKUMBFZ*_8m14`a?zk26D=Z$CUTaWk-MaEvqXP9rLzOIc|MQN*z z@VxYvvsK#<|K6*Wbo*9lMUH@RW3*(UqRu1!2$ra){k*xk7m{bK)6e_dqr%n2F#(AZ9cMk87sZ=(?Hyn5&)=<9Kex}}{7=vPzL(cOomQ+6TWo*2Y!SksjzfbVAg-Y?^>moA$*~+LI z%&lzxW9+t0oY6{*mxoKSZTcCV*NS|X?Fu(DDNB}4{eH2NIbJ$-#}B*UosN4-Zpp+p z7u3J4vf3kWYnt~d$O)b{Mxld-mG}0}+2Rl+ir8!5^Gg|SJDqVb8d+SZ&kpW!>d9M3N&Ld= zfsgE0HtA+rol~87JmjUstbW7kYzB*^<&N^Zew-*dka~wtk`R_W~w){#sBGnwbnk*lN4DNJy_m9Gr;|kOjWLv zcMq?al>0evx#xHNxq5beY+LqRK!M%Dw9d4w$@)-j!?D=c(dt+4YOfU&+nhE(ky9vF z?6aB7?SjC=X`e0rXKp`V{6g*z^@neDORt&561rHL zVX2$xyGNH!?F)6#T|HYWoB64Z%%|j+hfb%xxl+eHS+3;&nt5lo$Cb4xK92pKHLb#! zbM?Vhi;mrW(YJ4Pf5+Mx)x8I=A1}SF?H71%#RbXUqdmT5GjCL>#~yOs`Y=d(4pW8t zbHjxxZLzA8CwsS~6^MDhVcPY&z9I7UpGguQbEo{<^`PR-g5D~l%X%iJ-(I#_i|zJ_ zX!+tGn%wZ>N5F<%CAo{etGps7=X`j}bZ_&PMtPH%-yJ@F-c?^2yPBJCMNgG#>gwI{ zYsK4W=lusw&rbT>%NDe&>g&9pw*T^STLLv_)b`&N`M>k%uEYOJTOUctam4RC5PqqC zFNe4Pp%p^@%b$C;wBPS(Ol(~|f4jYYSnQ#=hn6dStsP{f%rh$HG1(|hi+mtmZTfBI zN_pD@$LH>Tt2oEBhA02+vV!f7oX=-WcmB$zrIj4S%ozWMiU0BbHE!*Tsvp&D^evup z|Hql`V>^$Zms-PQZhX9Jig0MjX4l;gcdrz^)T@7eYxCR5JZE;r3V&=|RepFw{x+7# za*JClcn)4GsSo7(J5%tB#bmpNNj)!_XS{r6P`1PBcznjbYU zmY=^Rc4_sd+{eopN^{OV?q1ibQyIO|caPc^{ocnjTIODiUX^YrSIIrq=jh61<|`A` z?=LT09&${!z23%W$3F|VQw+N%C}u2AO8l8?!j^FP_NmXI-8SdUU!J+zD16;|+s@^d z&-^QonjU=G8~ebq;cujyszjeK9PT|mxF>jb%^b%Dk7M^+4 z%{M(Ej48tRp-`y!p9&wZlF<2+?}S|X-~H&7TgR)z`?XJ9U`pZSI?HsdDAOlr&(4@H zndd^hX77(&?jreL=>PervhL~!K1mgI{xw< z-^-)BkKPujc-22=Pu>}q_DiddeK(%*`+3l%32d9sZ$2e!aA%3|!^_O)zOs8oS6x$X zxgE#E691m7)$41@_VPnMh99PPYpyveb1P|9jvKpk2m8E_QJiulv~Ois^x=GttM^W0G9wSxa&mAQ0$P33|q}{wB9{lE+NLlRlqW?RC)^Mlnim%`F>T9`%yX9ZLiAi-{k=xuV ze^qBke0vh|rdxqOd#c#qjVtv(g=811R}{JpDp)r1=cvx+w;*`9r`@Y_vbcKZC6(Z>{=mi^lhwPlSV ztMCeEujBpw`>nseeWhZ0YLRaJDzAcGq4SeBSLyS)By_nhF*w~OtRc83_r z@`);2`JG}|%*=U8J5(p(YLly1-9_04m*$_-s^>jvGjU~nRQS9Vhi{y?weIqQpXLYt zam%byUmDCF{B8H5_H=kF>s)@8k!bzkoM$HTYX>YvQC{^4OKTD{k+fwA|hN|DG_(=K6y8hbh6E)YA)h+jl))aM5|smpek6@7!js_qqFJ{;Ia@6r9VX_Ws17uBq8-oNvunnylKoJ9gRX- zSZDF9%#i2Jaz~FZ>ROHo-cnOGor<&6{Oj0zd{yxcR*tWc3B1wkO)VBaO`E~HB{x<| z{GesJknj!X<|CG-{nbt;u`hO1$9tcBt$I9szU|?=$-a7v&gT7)`Wk?2mJ9OeA0Ob?FD(umYE zSNbe6L3h&Pm_L_`u6*bbOlhq-({u6)_xuOOpKREIS1jaG{XRA3_NGT~)lWQVy_Ixr zdDZlpCUW;x78`H7^0O?>$#2PxhN*u%bRXp;&E#R;s*(8Uc(d9q4-q*J9g&Nd?4HJ+ z*1DkZ&9uIsed3>O+deWIcynDYUn{pe{BqK%eJ}23{I8JGS$?ugTe1U1c|$+t8qJZ+&I)Vk^rhPfd;-f3)DQ zTizN$y;&8FJ1k!ZWXP!RpJFVK&eBCl|Of*9R6q2x8uQg5Rkl;Ln{+Ew)E?R>}%_s~=o@g=X;AqJ6ZJB95bCgYnn#(@?A?=ex0;EFaQ0L;SGl5LIcS^3F{vQ z+se z{3B;@r}*;A+nr&lk4?39-dZVrnK{Wdc0P|${uP;OQGVTD6YeH+E|R-bq3y6XyWyEm z@8?Yuqo(Z6cs0r7xS?mR%$JT^+wB{lP1sqlY-X_eW!A}Mn`LLMFn5disVnQTrML9+ z&ceARE(}LwY+P6QF;Dp?drz&)%k9CANpA%MEVKRi+;g%Xym<5NQLq2TWS(iy8U@O7 zbAIkIbeg}(eo1TI9j}+?S59r2HQWCGw%UzrxB1lS7H;!saI4Pgw~j9>U^L3~5lUY5 z!J&7?uKJUFmwMt8OR~$Z)n!~449)-TJ|p+ep+yslj~jYkpXOj8bLKMVzWmNf8R6K!uO@xyi<#|Ce$KT+|=r`THcDhLVv zX#XhclY5;faQ6+dcHXBt|3j~Tx_&2|H&b`6(2IZ%cP4(W7vW>$N%x$PV4kEcx$=Bj z)v*=n7nkqqZ$82qcyH^bBgQ-BWKKQXy?wbMN9JUs0;!34PVot^rYiBAh3i+3_U*Eo6iOaF`OeC4;v_lDE8jh~7|c^hI(CtN>$VB5`W7d}tZ*-#!K>TaC* z#xjUse&*?Wo9|0KxT;hin#(NG>u`RbjJ?sdwTHtu&(2)>YGVGTJ2~2qm!w+mFbi3A z!1Pn_%qPC_9~4U@pE;LWFd9a?h5WK#cQ-4uYuCgBPckR&oV``GF>&1{jk2Gw{vHtV z37l%Tz~m2umq7PzVKXNNt=C%VH#V_f-k|Gt`-?q0&kBLmyp-aSqSVA(uE|IAbJYy= z3^chE6!e{oQWHxu^YdI1OHvgyT&#=?3@r^z4b2QqOihf Date: Mon, 22 Jun 2020 16:37:51 -0700 Subject: [PATCH 559/880] v10.2.0 release notes Closes #582, #584, #545 --- docs/release_notes.rst | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 40b71d41..d7e3dde4 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,21 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.2.0 +======= + +- Update Docker image to use Ubuntu 20.04. +- Fixed issue PDF/A acquires title "Untitled" after conversion. (#582) +- Fixed a problem where, when using ``--pdf-renderer hocr``, some text would + be missing from the output when using a more recent version of Tesseract. + Tesseract began adding more detailed markup about the semantics of text + that our HOCR transform did not recognize, so it ignored them. This option is + not the default. If necessary ``--redo-ocr`` also redoing OCR to fix such issues. +- Fixed an error in Python 3.9 beta, due to removal of deprecated + ``Element.getchildren()``. (#584) +- Implemented support using the API with ``BytesIO`` and other file stream objects. + (#545) + v10.1.1 ======= @@ -37,7 +52,7 @@ v10.1.0 v10.0.1 ======= -- Fix regression when ``-l lang1+lang2`` is used from command line. +- Fixed regression when ``-l lang1+lang2`` is used from command line. v10.0.0 ======= From eb5a211e728c8c959726fde091d75470d8f4f176 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 16:59:59 -0700 Subject: [PATCH 560/880] New hocrtransform test isn't platform stable - mark runslow --- tests/test_hocrtransform.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index f4b0a1a1..ea993ec6 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -70,6 +70,7 @@ def test_mono_image(blank_hocr, outdir): check_pdf(str(outdir / 'mono.pdf')) +@pytest.runslow def test_hocrtransform_matches_sandwich(resources, outdir): check_ocrmypdf(resources / 'ccitt.pdf', outdir / 'hocr.pdf', '--pdf-renderer=hocr') check_ocrmypdf( From 66337813e63a37b2a6b6c9eff9d5025e5ef752e0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 22 Jun 2020 23:32:09 -0700 Subject: [PATCH 561/880] Spell runslow correctly --- tests/test_hocrtransform.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index ea993ec6..5eea4407 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -17,7 +17,6 @@ import re from io import StringIO -from pathlib import Path import pytest from pdfminer.converter import TextConverter @@ -70,7 +69,7 @@ def test_mono_image(blank_hocr, outdir): check_pdf(str(outdir / 'mono.pdf')) -@pytest.runslow +@pytest.mark.slow def test_hocrtransform_matches_sandwich(resources, outdir): check_ocrmypdf(resources / 'ccitt.pdf', outdir / 'hocr.pdf', '--pdf-renderer=hocr') check_ocrmypdf( From 01cae7a584b3cd2bc55ef61fb1114866bf04175f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 23 Jun 2020 02:08:24 -0700 Subject: [PATCH 562/880] docs: Update Fedora versions --- docs/installation.rst | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 570045e4..e372d38d 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -87,11 +87,11 @@ For full details on version availability for your platform, check the Fedora 29 or newer ------------------ -.. |fedora-29| image:: https://repology.org/badge/version-for-repo/fedora_29/ocrmypdf.svg - :alt: Fedora 29 +.. |fedora-31| image:: https://repology.org/badge/version-for-repo/fedora_31/ocrmypdf.svg + :alt: Fedora 31 -.. |fedora-30| image:: https://repology.org/badge/version-for-repo/fedora_30/ocrmypdf.svg - :alt: Fedora 30 +.. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg + :alt: Fedora 32 .. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg :alt: Fedore Rawhide @@ -101,7 +101,7 @@ Fedora 29 or newer +-----------------------------------------------+ | |latest| | +-----------------------------------------------+ -| |fedora-29| |fedora-30| |fedora-rawhide| | +| |fedora-31| |fedora-32| |fedora-rawhide| | +-----------------------------------------------+ Users of Fedora 29 or later may simply From 580f2ebb4b663d850222bb14ed898f7a78777fb9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Jun 2020 00:06:58 -0700 Subject: [PATCH 563/880] Python 3.9beta is now known to work (Fedora) --- setup.py | 1 + 1 file changed, 1 insertion(+) diff --git a/setup.py b/setup.py index 5fd923f4..bd95ed99 100644 --- a/setup.py +++ b/setup.py @@ -54,6 +54,7 @@ setup( "Programming Language :: Python :: 3.6", "Programming Language :: Python :: 3.7", "Programming Language :: Python :: 3.8", + "Programming Language :: Python :: 3.9", "Development Status :: 5 - Production/Stable", "Environment :: Console", "Intended Audience :: End Users/Desktop", From a92dde058ae3b4e1dc9fc87bacf43db545ffcd8b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Jun 2020 22:47:44 -0700 Subject: [PATCH 564/880] docs: promote one liner installs, reorg Windows --- docs/installation.rst | 62 +++++++++++++++++++++++-------------------- 1 file changed, 33 insertions(+), 29 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index e372d38d..4da28cd4 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -8,20 +8,24 @@ Installing OCRmyPDF |latest| The easiest way to install OCRmyPDF is to follow the steps for your operating -system/platform, although sometimes this version may be out of date. This -installation guide provides information allowing you to compare the current -version to the one provided by your platform. +system/platform. This version may be out of date, however. -If you want to use the latest version of OCRmyPDF and all of its optional -dependencies, the easiest way to get that is install the Homebrew package. Homebrew -is best known as a macOS package manger, but also works for -`Linux and Windows Subsystem for Linux `__. -After Homebrew is installed, simply run ``brew install ocrmypdf``. +These platforms have one-liner installs: -You can also use the more detailed procedures here to manually install OCRmyPDF -from source or with the ``pip`` package manager for binary wheels. The reason -for these varied steps is that OCRmyPDF requires third-party executables that are -not part of Python. ++-----------------------------+-------------------------------+ +| Debian, Ubuntu | ``apt install ocrmypdf`` | ++-----------------------------+-------------------------------+ +| Windows Subsystem for Linux | ``apt install ocrmypdf`` | ++-----------------------------+-------------------------------+ +| Fedora | ``dnf install ocrmypdf`` | ++-----------------------------+-------------------------------+ +| macOS | ``brew install ocrmypdf`` | ++-----------------------------+-------------------------------+ +| FreeBSD | ``pkg install py37-ocrmypdf`` | ++-----------------------------+-------------------------------+ + +More detailed procedures are outlined below. If you want to do a manual +install, or install a more recent version than your platform provides, read on. .. contents:: Platform-specific steps :depth: 2 @@ -162,7 +166,7 @@ To install for the current user only: pip3 install --user ocrmypdf Ubuntu 18.04 LTS -------------------------------------------------- +---------------- Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but it is quite old now. To install a more recent version, uninstall the old version @@ -535,12 +539,12 @@ See `OCRmyPDF Docker Image `__ for more information. Installing on Windows ===================== -.. warning:: +Native Windows +-------------- - Native Windows support is new. Consider it "beta" software. Some - functionality is missing or may be more difficult to enable. If you need a - production-ready solution, use Windows Subsystem for Linux or a Docker - image. +.. note:: + + It is easier to install OCRmyPDF on Windows Subsystem for Linux. .. note:: @@ -548,7 +552,7 @@ Installing on Windows You must install the following for Windows: -* Python 3.7 (64-bit) +* Python 3.7 (64-bit) or later * Tesseract 4.0 or later * Ghostscript 9.50 or later @@ -577,8 +581,8 @@ You may then use pip to install ocrmypdf: * ``pip install ocrmypdf`` -Installing on Windows Subsystem for Linux -========================================= +Windows Subsystem for Linux +--------------------------- #. Install Ubuntu 18.04 for Windows Subsystem for Linux, if not already installed. #. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 18.04 `. @@ -597,14 +601,8 @@ Then confirm that the expected version from PyPI (|latest|) is installed: You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing ``wsl``, and call it from Windows programs or batch files. -Docker -^^^^^^ - -You can also :ref:`Install the Docker ` container on Windows. Ensure that -your command prompt can run the docker "hello world" container. - -Installing on Cygwin64 under Windows -==================================== +Cygwin64 +-------- First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``:: @@ -650,6 +648,12 @@ The optional dependency "unpaper" that is currently not available under Cygwin. Without it, certain options such as ``--clean`` will produce an error message. However, the OCR-to-text-layer functionality is available. +Docker +------ + +You can also :ref:`Install the Docker ` container on Windows. Ensure that +your command prompt can run the docker "hello world" container. + Installing with Python pip ========================== From 638d68aa8a2e0636c9b57158450a798e22d56503 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Jun 2020 22:49:34 -0700 Subject: [PATCH 565/880] docs: move Windows ahead of FreeBSD --- docs/installation.rst | 50 +++++++++++++++++++++---------------------- 1 file changed, 25 insertions(+), 25 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4da28cd4..3457f756 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -511,31 +511,6 @@ The command line program should now be available: ocrmypdf --help -Installing on FreeBSD -===================== - -.. image:: https://repology.org/badge/version-for-repo/freebsd/python:ocrmypdf.svg - :alt: FreeBSD - :target: https://repology.org/project/python:ocrmypdf/versions - -FreeBSD 11.3, 12.0, 12.1-RELEASE and 13.0-CURRENT are supported. Other -versions likely work but have not been tested. - -.. code-block:: bash - - pkg install py37-ocrmypdf - -To install a more recent version, you could attempt to first install the system -version with ``pkg``, then use ``pip install --user ocrmypdf``. - -Installing the Docker image -=========================== - -For some users, installing the Docker image will be easier than -installing all of OCRmyPDF's dependencies. - -See `OCRmyPDF Docker Image `__ for more information. - Installing on Windows ===================== @@ -654,6 +629,31 @@ Docker You can also :ref:`Install the Docker ` container on Windows. Ensure that your command prompt can run the docker "hello world" container. +Installing on FreeBSD +===================== + +.. image:: https://repology.org/badge/version-for-repo/freebsd/python:ocrmypdf.svg + :alt: FreeBSD + :target: https://repology.org/project/python:ocrmypdf/versions + +FreeBSD 11.3, 12.0, 12.1-RELEASE and 13.0-CURRENT are supported. Other +versions likely work but have not been tested. + +.. code-block:: bash + + pkg install py37-ocrmypdf + +To install a more recent version, you could attempt to first install the system +version with ``pkg``, then use ``pip install --user ocrmypdf``. + +Installing the Docker image +=========================== + +For some users, installing the Docker image will be easier than +installing all of OCRmyPDF's dependencies. + +See `OCRmyPDF Docker Image `__ for more information. + Installing with Python pip ========================== From 7630c93e5bbe0a6dc388a1164f3d0358820b39b0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Jun 2020 23:27:42 -0700 Subject: [PATCH 566/880] install: drop Ubuntu 14.04 steps Bit rot must have set in. --- docs/installation.rst | 62 ------------------------------------------- 1 file changed, 62 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 3457f756..4195d51b 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -269,68 +269,6 @@ environment variable contains ``$HOME/.local/bin``. To add JBIG2 encoding, see :ref:`jbig2`. -Ubuntu 14.04 LTS ----------------- - -Installing on Ubuntu 14.04 LTS (trusty) is more difficult than some -other options, because of its age. Several backports are required. For -explanations of some steps of this procedure, see the similar steps for -Ubuntu 16.04. - -Install system dependencies: - -.. code-block:: bash - - sudo apt-get update - sudo apt-get install \ - software-properties-common python-software-properties \ - zlib1g-dev \ - libexempi3 \ - libjpeg-dev \ - libffi-dev \ - pngquant \ - qpdf - -We will need backports of Ghostscript 9.16, libav-11 (for unpaper 6.1), -Tesseract 4.00 (alpha), and Python 3.6. This will replace Ghostscript -and Tesseract 3.x on your system. Python 3.6 will be installed alongside -the system Python 3.4. - -If you prefer to not modify your system in this matter, consider using a -Docker container. - -.. code-block:: bash - - sudo add-apt-repository ppa:vshn/ghostscript -y - sudo add-apt-repository ppa:heyarje/libav-11 -y - sudo add-apt-repository ppa:alex-p/tesseract-ocr -y - sudo add-apt-repository ppa:jonathonf/python-3.6 -y - - sudo apt-get update - - sudo apt-get install \ - python3.6-dev \ - ghostscript \ - tesseract-ocr \ - tesseract-ocr-eng \ - libavformat56 libavcodec56 libavutil54 \ - wget - -Now we need to install ``pip`` and let it install ocrmypdf: - -.. code-block:: bash - - curl https://bootstrap.pypa.io/ez_setup.py -o - | python3.6 && python3.6 -m easy_install pip - pip3.6 install ocrmypdf - -The optional dependency ``unpaper`` is only available at 0.4.2 in Ubuntu 14.04, -and no backports are available. Previously the author maintained a backported -.deb package for unpaper 6.1, but since Ubuntu 14.04 is now end of life, this is -not supported. As such, ``unpaper`` is not available on Ubuntu 14.04 or must by -compiled by hand. - -To add JBIG2 encoding, see :ref:`jbig2`. - Arch Linux (AUR) ---------------- From f15d9049ebf333778c6649f5a7ffa7903a23b37a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Jun 2020 23:28:26 -0700 Subject: [PATCH 567/880] install: add Mageia Closes #586. Thanks to @yannick56 --- docs/installation.rst | 41 ++++++++++++++++++++++++++++++++++++++++- 1 file changed, 40 insertions(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4195d51b..9a7541ea 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -21,6 +21,8 @@ These platforms have one-liner installs: +-----------------------------+-------------------------------+ | macOS | ``brew install ocrmypdf`` | +-----------------------------+-------------------------------+ +| LinuxBrew | ``brew install ocrmypdf`` | ++-----------------------------+-------------------------------+ | FreeBSD | ``pkg install py37-ocrmypdf`` | +-----------------------------+-------------------------------+ @@ -62,7 +64,7 @@ Debian and Ubuntu 18.04 or newer | |ubu-1804| |ubu-2004| | +-----------------------------------------------+ -Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users +Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users of Windows Subsystem for Linux, may simply .. code-block:: bash @@ -349,6 +351,43 @@ To install OCRmyPDF for Alpine Linux: apk add ocrmypdf +Mageia 7 +-------- + +Install the following dependencies: + +.. code-block:: bash + + # As root user + urpmi.update -a + urpmi \ + ghostscript \ + icc-profiles-openicc \ + jbig2dec \ + lib64leptonica5 \ + pngquant \ + python3-pip \ + python3-cffi \ + python3-distutils-extra \ + python3-pkg-resources \ + python3-reportlab \ + qpdf \ + tesseract \ + tesseract-osd \ + tesseract-eng \ + tesseract-fra + +To install ocrmypdf for the system: + + # As root user + pip3 install ocrmypdf + ldconfig + +Or, to install for the current user only: + + export PATH=$HOME/.local/bin:$PATH + pip3 install --user ocrmypdf + Other Linux packages -------------------- From e5b6fe131796c63c9956ffb14f8aaab72d0c1152 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 29 Jun 2020 01:45:12 -0700 Subject: [PATCH 568/880] pyproject.toml: weird line wrapping? --- pyproject.toml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 9cfd37c6..a28f55c0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,8 +10,7 @@ build-backend = "setuptools.build_meta" [tool.black] line-length = 88 -target-version = ["py36", -"py37", "py38"] +target-version = ["py36", "py37", "py38"] skip-string-normalization = true include = '\.pyi?$' exclude = ''' From bbd174071d07a6a6f2f4349d5834979ad674b9d3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 29 Jun 2020 01:45:27 -0700 Subject: [PATCH 569/880] readme: markdown cleanup --- README.md | 36 +++++++++++------------------------- 1 file changed, 11 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index 1c38188c..bb26967c 100644 --- a/README.md +++ b/README.md @@ -3,15 +3,10 @@ [![Build Status][azure]](https://dev.azure.com/jim0585/ocrmypdf/_build/latest?definitionId=2&branchName=master) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] [azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master - [travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status" - [pypi]: https://img.shields.io/pypi/v/ocrmypdf.svg "PyPI version" - [homebrew]: https://img.shields.io/homebrew/v/ocrmypdf.svg "Homebrew version" - [docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD" - [pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "Supported Python versions" OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched or copy-pasted. @@ -30,8 +25,7 @@ ocrmypdf # it's a scriptable command line program [See the release notes for details on the latest changes](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html). -Main features -------------- +## Main features - Generates a searchable [PDF/A](https://en.wikipedia.org/?title=PDF/A) file from a regular PDF - Places OCR text accurately below the image to ease copy / paste @@ -47,8 +41,7 @@ Main features For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/en/latest/). -Motivation ----------- +## Motivation I searched the web for a free command line tool to OCR PDF files: I found many, but none of them were really satisfying: @@ -62,8 +55,7 @@ I searched the web for a free command line tool to OCR PDF files: I found many, ...so I decided to develop my own tool. -Installation ------------- +## Installation Linux, Windows, macOS and FreeBSD are supported. Docker images are also available. @@ -87,8 +79,7 @@ brew install ocrmypdf For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps. -Languages ---------- +## Languages OCRmyPDF uses Tesseract for OCR, and relies on its language packs. For Linux users, you can often find packages that provide language packs: @@ -105,8 +96,7 @@ pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English a You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested. -Documentation and support -------------------------- +## Documentation and support Once OCRmyPDF is installed, the built-in help which explains the command syntax and options can be accessed via: @@ -118,27 +108,24 @@ Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/e Please report issues on our [GitHub issues](https://github.com/jbarlow83/OCRmyPDF/issues) page, and follow the issue template for quick response. -Requirements ------------- +## Requirements In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. OCRmyPDF is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD. -Press & Media -------------- +## Press & Media - [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a) - [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c) - [c't 1-2014, page 59](https://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't - [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670) - [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html) +- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/) -Business enquiries ------------------- +## Business enquiries OCRmyPDF would not be the software that it is today without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system. -License -------- +## License The OCRmyPDF software is licensed under the GNU GPLv3. Certain files are covered by other licenses, as noted in their source files. @@ -146,7 +133,6 @@ The license for each test file varies, and is noted in tests/resources/README.rs OCRmyPDF versions prior to 6.0 were distributed under the MIT License. -Disclaimer ----------- +## Disclaimer The software is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. From b939584c7a915668ce143bf7ece6ad5f8dcb8230 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 29 Jun 2020 01:45:45 -0700 Subject: [PATCH 570/880] quality: fixing typing issues --- src/ocrmypdf/quality.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/quality.py b/src/ocrmypdf/quality.py index 293173bf..bfb95702 100644 --- a/src/ocrmypdf/quality.py +++ b/src/ocrmypdf/quality.py @@ -15,23 +15,23 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +"""Utilities to measure OCR quality""" + + import re from typing import Iterable -"""Utilities to measure OCR quality""" - class OcrQualityDictionary: """Manages a dictionary for simple OCR quality checks.""" - def __init__(self, *, wordlist: Iterable[str] = []): + def __init__(self, *, wordlist: Iterable[str]): """Construct a dictionary from a list of words. Words for which capitalization is important should be capitalized in the dictionary. Words that contain spaces or other punctuation will never match. """ - self.dictionary = set() - self.dictionary.update(w for w in wordlist) + self.dictionary = set(wordlist) def measure_words_matched(self, ocr_text: str) -> float: """Check how many unique words in the OCR text match a dictionary. From 86875997b8eda4aab064bef6f9b9bc397b4b2274 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 29 Jun 2020 02:17:14 -0700 Subject: [PATCH 571/880] Fix more mypy errors --- src/ocrmypdf/_concurrent.py | 2 +- src/ocrmypdf/_exec/ghostscript.py | 18 +++++++++++------- src/ocrmypdf/_exec/tesseract.py | 12 ++++++------ src/ocrmypdf/api.py | 6 +++--- src/ocrmypdf/helpers.py | 1 + 5 files changed, 22 insertions(+), 17 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 6d608eb6..f954d6d8 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -84,7 +84,7 @@ def exec_progress_pool( task_arguments: Optional[Iterable] = None, task_finished: Optional[Callable] = None, ): - log_queue = multiprocessing.Queue(-1) + log_queue: multiprocessing.Queue = multiprocessing.Queue(-1) listener = threading.Thread(target=log_listener, args=(log_queue,)) if use_threads: diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 0fb65b1b..44347a42 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -25,6 +25,7 @@ from os import fspath from pathlib import Path from shutil import which from subprocess import PIPE, CalledProcessError +from typing import Optional, cast from PIL import Image @@ -34,12 +35,12 @@ from ocrmypdf.subprocess import get_version, run log = logging.getLogger(__name__) -GS = 'gs' +_gswin = None if os.name == 'nt': - GS = which('gswin64c') - if not GS: - GS = which('gswin32c') - if not GS: + _gswin = which('gswin64c') + if not _gswin: + _gswin = which('gswin32c') + if not _gswin: raise MissingDependencyError( """ --------------------------------------------------------------------- @@ -52,7 +53,10 @@ if os.name == 'nt': --------------------------------------------------------------------- """ ) - GS = Path(GS).stem + _gswin = Path(_gswin).stem + +GS = _gswin if _gswin else 'gs' +del _gswin def version(): @@ -77,7 +81,7 @@ def jpeg_passthrough_available() -> bool: def _gs_error_reported(stream) -> bool: - return re.search(r'error', stream, flags=re.IGNORECASE) + return True if re.search(r'error', stream, flags=re.IGNORECASE) else False def rasterize_pdf( diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index bfa6305d..85dc5040 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -123,7 +123,7 @@ def get_languages(): return set(lang.strip() for lang in rest) -def tess_base_args(langs: List[str], engine_mode) -> List[str]: +def tess_base_args(langs: List[str], engine_mode: int) -> List[str]: args = ['tesseract'] if langs: args.extend(['-l', '+'.join(langs)]) @@ -132,7 +132,7 @@ def tess_base_args(langs: List[str], engine_mode) -> List[str]: return args -def get_orientation(input_file: Path, engine_mode, timeout: float): +def get_orientation(input_file: Path, engine_mode: int, timeout: float): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', @@ -229,9 +229,9 @@ def generate_hocr( input_file: Path, output_hocr: Path, output_text: Path, - languages: list, - engine_mode, - tessconfig: list, + languages: List[str], + engine_mode: int, + tessconfig: List[str], timeout: float, pagesegmode: int, user_words, @@ -290,7 +290,7 @@ def generate_pdf( output_pdf: Path, output_text: Path, languages: List[str], - engine_mode, + engine_mode: int, tessconfig: List[str], timeout: float, pagesegmode: int, diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 4fac0159..f9db6a6f 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -18,7 +18,6 @@ import logging import os import sys -from argparse import ArgumentParser from enum import IntEnum from pathlib import Path from typing import BinaryIO, Iterable, Union @@ -27,7 +26,8 @@ from ocrmypdf._logging import PageNumberFilter, TqdmConsole from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_options -from ocrmypdf.cli import get_parser +from ocrmypdf.cli import ArgumentParser, get_parser +from ocrmypdf.helpers import is_iterable_notstr try: import coloredlogs @@ -153,7 +153,7 @@ def create_options( cmdline.append(f"--{cmd_style_arg}") continue - if isinstance(val, Iterable) and not isinstance(val, str): + if is_iterable_notstr(val): for elem in val: cmdline.append(f"--{cmd_style_arg}") cmdline.append(elem) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index aca40b6e..c457ab8b 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -113,6 +113,7 @@ def samefile(f1: os.PathLike, f2: os.PathLike): def is_iterable_notstr(thing: Any) -> bool: + """Is this is an iterable type, other than a string?""" return isinstance(thing, Iterable) and not isinstance(thing, str) From 86a73191b01cb9b2a9da67dea30b63574d16945d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jun 2020 04:17:30 -0700 Subject: [PATCH 572/880] Plugin manager: accept Path(plugin) --- src/ocrmypdf/_plugin_manager.py | 8 +++++--- tests/test_main.py | 2 +- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index ba867525..13802f12 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -21,7 +21,7 @@ import importlib.util import sys from functools import partial from pathlib import Path -from typing import Callable, List, Tuple +from typing import Callable, List, Tuple, Union import pluggy @@ -64,7 +64,9 @@ class OcrmypdfPluginManager(pluggy.PluginManager): ) -def _setup_plugins(pm: pluggy.PluginManager, plugins: List[str], builtins: bool = True): +def _setup_plugins( + pm: pluggy.PluginManager, plugins: List[Union[str, Path]], builtins: bool = True +): pm.add_hookspecs(pluginspec) if builtins: @@ -75,7 +77,7 @@ def _setup_plugins(pm: pluggy.PluginManager, plugins: List[str], builtins: bool else: all_plugins = plugins for name in all_plugins: - if name.endswith('.py'): + if isinstance(name, Path) or name.endswith('.py'): # Import by filename module_name = Path(name).stem spec = importlib.util.spec_from_file_location(module_name, name) diff --git a/tests/test_main.py b/tests/test_main.py index 712753fe..65c33e39 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -256,7 +256,7 @@ def test_missing_docinfo(resources, outpdf): 'eng', '--skip-text', '--plugin', - 'tests/plugins/tesseract_noop.py', + Path('tests/plugins/tesseract_noop.py'), ) assert result == ExitCode.ok From 62924ee28044443721cd46b34d7c64227323a370 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jun 2020 04:20:14 -0700 Subject: [PATCH 573/880] Improve API documentation --- docs/api.rst | 4 --- docs/apiref.rst | 43 ++++++++++++++++++++++++++++++++ docs/conf.py | 10 +++++--- docs/index.rst | 1 + docs/plugins.rst | 40 +++++++++++++++++++++++++++++- src/ocrmypdf/helpers.py | 8 ++++++ src/ocrmypdf/hocrtransform.py | 30 +++++++++++++++++------ src/ocrmypdf/pdfa.py | 46 ++++++++++++++++------------------- src/ocrmypdf/pluginspec.py | 38 ++++++++++++++++++++++++----- src/ocrmypdf/subprocess.py | 30 +++++++++++++---------- 10 files changed, 190 insertions(+), 60 deletions(-) create mode 100644 docs/apiref.rst diff --git a/docs/api.rst b/docs/api.rst index ce6583fc..735354f6 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -106,7 +106,3 @@ Reference :undoc-members: .. autofunction:: ocrmypdf.configure_logging - -.. automodule:: ocrmypdf.exceptions - :members: - :undoc-members: diff --git a/docs/apiref.rst b/docs/apiref.rst new file mode 100644 index 00000000..caeb277a --- /dev/null +++ b/docs/apiref.rst @@ -0,0 +1,43 @@ +============= +API Reference +============= + +This page summarizes the rest of the public API. Generally speaking this +should mainly of interest to plugin developers. + +ocrmypdf.exceptions +=================== + +.. automodule:: ocrmypdf.exceptions + :members: + :undoc-members: + +ocrmypdf.helpers +================ + +.. automodule:: ocrmypdf.helpers + :members: + +ocrmypdf.hocrtransform +====================== + +.. automodule:: ocrmypdf.hocrtransform + :members: + +ocrmypdf.pdfa +============= + +.. automodule:: ocrmypdf.pdfa + :members: + +ocrmypdf.quality +================ + +.. automodule:: ocrmypdf.quality + :members: + +ocrmypdf.subprocess +=================== + +.. automodule:: ocrmypdf.subprocess + :members: diff --git a/docs/conf.py b/docs/conf.py index b78cc050..4a6dc0ac 100755 --- a/docs/conf.py +++ b/docs/conf.py @@ -1,5 +1,4 @@ #!/usr/bin/env python3 -# -*- coding: utf-8 -*- # # ocrmypdf documentation build configuration file, created by # sphinx-quickstart on Sun Sep 4 14:29:43 2016. @@ -21,6 +20,8 @@ # import sys # sys.path.insert(0, os.path.abspath('.')) +"""isort:skip_file""" + # -- General configuration ------------------------------------------------ # If your documentation needs a minimal Sphinx version, state it here. @@ -32,6 +33,8 @@ # ones. extensions = ['sphinx.ext.napoleon'] +napoleon_use_rtype = False + # Add any paths that contain templates here, relative to this directory. templates_path = ['_templates'] @@ -51,7 +54,7 @@ master_doc = 'index' # General information about the project. project = 'ocrmypdf' copyright = ( - '2019, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.' + '2020, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.' ) author = 'James R. Barlow' @@ -90,6 +93,7 @@ from pkg_resources import get_distribution, DistributionNotFound release = get_distribution('ocrmypdf').version version = '.'.join(release.split('.')[:2]) + # The language for content autogenerated by Sphinx. Refer to documentation # for a list of supported languages. # @@ -174,7 +178,7 @@ html_theme_options = {'display_version': False} # The name of an image file (relative to this directory) to place at the top # of the sidebar. # -# html_logo = None +# html_logo = "images/logo.svg" # looks bad # The name of an image file (relative to this directory) to use as a favicon of # the docs. This file should be a Windows icon file (.ico) being 16x16 or 32x32 diff --git a/docs/index.rst b/docs/index.rst index 2591b1b9..217f5411 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -36,6 +36,7 @@ image processing and OCR to existing PDFs. api plugins + apiref contributing Indices and tables diff --git a/docs/plugins.rst b/docs/plugins.rst index 9dd145c2..d55a01f8 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -76,5 +76,43 @@ A plugin may provide the following hooks. Hooks should be decorated with The following is a complete list of hooks that may be installed and when they are called. -.. automodule:: ocrmypdf.pluginspec +Custom command line arguments +----------------------------- + +.. autofunction:: ocrmypdf.pluginspec.add_options + +.. autofunction:: ocrmypdf.pluginspec.check_options + +Applying special behavior before processing +------------------------------------------- + +.. autofunction:: ocrmypdf.pluginspec.validate + +PDF page to image +----------------- + +.. autofunction:: ocrmypdf.pluginspec.rasterize_pdf_page + +Modifying intermediate images +----------------------------- + +.. autofunction:: ocrmypdf.pluginspec.filter_ocr_image + +.. autofunction:: ocrmypdf.pluginspec.filter_page_image + +OCR engine +---------- + +.. autofunction:: ocrmypdf.pluginspec.get_ocr_engine + +.. autoclass:: ocrmypdf.pluginspec.OcrEngine :members: + + .. automethod:: __str__ + +.. autoclass:: ocrmypdf.pluginspec.OrientationConfidence + +PDF/A production +---------------- + +.. autofunction:: ocrmypdf.pluginspec.generate_pdfa diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index c457ab8b..d2523496 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -35,6 +35,8 @@ log = logging.getLogger(__name__) class Resolution(namedtuple('Resolution', ('x', 'y'))): + """The number of pixels per inch in each 2D direction.""" + __slots__ = () def round(self, ndigits: int): @@ -128,6 +130,7 @@ def page_number(input_file: os.PathLike) -> int: def available_cpu_count() -> int: + """Returns number of CPUs in the system.""" try: return multiprocessing.cpu_count() except NotImplementedError: @@ -174,6 +177,10 @@ def is_file_writable(test_file: os.PathLike) -> bool: def check_pdf(input_file: Path) -> bool: + """Check if a PDF complies with the PDF specification. + + Checks for proper formatting and proper linearization. + """ pdf = None try: pdf = pikepdf.open(input_file) @@ -211,6 +218,7 @@ T = TypeVar('T') def clamp(n: T, smallest: T, largest: T) -> T: + """Clamps the value of n to between smallest and largest.""" return max(smallest, min(n, largest)) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 606a2850..7e2ef142 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -35,7 +35,7 @@ from collections import namedtuple from itertools import chain from math import atan, cos, sin from pathlib import Path -from typing import Union +from typing import Optional, Tuple, Union from xml.etree import ElementTree from reportlab.lib.colors import black, cyan, magenta, red @@ -133,7 +133,7 @@ class HocrTransform: return out @classmethod - def baseline(cls, element): + def baseline(cls, element) -> Tuple[float, float]: """ Returns a tuple containing the baseline slope and intercept. """ @@ -143,7 +143,7 @@ class HocrTransform: return float(matches.group(1)), int(matches.group(2)) return (0.0, 0.0) - def pt_from_pixel(self, pxl): + def pt_from_pixel(self, pxl) -> Rect: """ Returns the quantity in PDF units (pt) given quantity in pixels """ @@ -156,11 +156,11 @@ class HocrTransform: return xpath @classmethod - def replace_unsupported_chars(cls, s: str): + def replace_unsupported_chars(cls, s: str) -> str: """ Given an input string, returns the corresponding string that: - - is available in the helvetica facetype - - does not contain any ligature (to allow easy search in the PDF file) + * is available in the Helvetica facetype + * does not contain any ligature (to allow easy search in the PDF file) """ return s.translate(cls.ligatures) @@ -172,12 +172,12 @@ class HocrTransform: def to_pdf( self, out_filename: Path, - image_filename: Path = None, + image_filename: Optional[Path] = None, show_bounding_boxes: bool = False, fontname: str = "Helvetica", invisible_text: bool = False, interword_spaces: bool = False, - ): + ) -> None: """ Creates a PDF file with an image superimposed on top of the text. Text is positioned according to the bounding box of the lines in @@ -185,6 +185,20 @@ class HocrTransform: The image need not be identical to the image used to create the hOCR file. It can have a lower resolution, different color mode, etc. + + Arguments: + out_filename: Path of PDF to write. + image_filename: Image to use for this file. If omitted, the OCR text + is shown. + show_bounding_boxes: Show bounding boxes around various text regions, + for debugging. + fontname: Name of font to use. + invisible_text: If True, text is rendered invisible so that is + selectable but never drawn. If False, text is visible and may + be seen if the image is skipped or deleted in Acrobat. + interword_spaces: If True, insert spaces between words rather than + drawing each word without spaces. Generally this improves text + extraction. """ # create the PDF file # page size in points (1/72 in.) diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 3aeb7612..7fd42194 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -16,19 +16,7 @@ # along with OCRmyPDF. If not, see . """ -Generate a PDFMARK file for Ghostscript >= 9.14, for PDF/A conversion - -pdfmark is an extension to the Postscript language that describes some PDF -features like bookmarks and annotations. It was originally specified Adobe -Distiller, for Postscript to PDF conversion: -https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf - -Ghostscript uses pdfmark for PDF to PDF/A conversion as well. To use Ghostscript -to create a PDF/A, we need to create a pdfmark file with the necessary metadata. - -This takes care of the many version-specific bugs and pecularities in -Ghostscript's handling of pdfmark. - +Utilities for PDF/A production and confirmation with Ghostspcript. """ import base64 @@ -68,20 +56,28 @@ def """ -def generate_pdfa_ps(target_filename, icc='sRGB'): - """Create a Postscript pdfmark file for Ghostscript PDF/A conversion +def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'): + """Create a Postscript PDFMARK file for Ghostscript PDF/A conversion - A pdfmark file is a small Postscript program that provides some information - Ghostscript needs to perform PDF/A conversion. The only information we put - in specifies that we want the file to be a PDF/A, and we want to Ghostscript - to convert objects to the sRGB colorspace if it runs into any object that - it decides must be converted. + pdfmark is an extension to the Postscript language that describes some PDF + features like bookmarks and annotations. It was originally specified Adobe + Distiller, for Postscript to PDF conversion. - See the Adobe pdfmark Reference for details: - https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf + Ghostscript uses pdfmark for PDF to PDF/A conversion as well. To use Ghostscript + to create a PDF/A, we need to create a pdfmark file with the necessary metadata. - :param target_filename: filename to save - :param icc: ICC identifier such as 'sRGB' + This function takes care of the many version-specific bugs and pecularities in + Ghostscript's handling of pdfmark. + + The only information we put in specifies that we want the file to be a + PDF/A, and we want to Ghostscript to convert objects to the sRGB colorspace + if it runs into any object that it decides must be converted. + + Arguments: + target_filename: filename to save + icc: ICC identifier such as 'sRGB' + References: + Adobe PDFMARK Reference: https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf """ if icc == 'sRGB': icc_profile = SRGB_ICC_PROFILE @@ -102,7 +98,7 @@ def generate_pdfa_ps(target_filename, icc='sRGB'): return target_filename -def file_claims_pdfa(filename): +def file_claims_pdfa(filename: Path): """Determines if the file claims to be PDF/A compliant This only checks if the XMP metadata contains a PDF/A marker. It does not diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index a96178f9..9371aede 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -22,13 +22,13 @@ from pathlib import Path from typing import TYPE_CHECKING, AbstractSet, List, Optional import pluggy -from PIL import Image from ocrmypdf.helpers import Resolution if TYPE_CHECKING: from ocrmypdf._jobcontext import PageContext from ocrmypdf.pdfinfo import PdfInfo + from PIL import Image hookspec = pluggy.HookspecMarker('ocrmypdf') @@ -118,7 +118,7 @@ def rasterize_pdf_page( rotation: Cardinal angle, clockwise, to rotate page filter_vector: If True, remove vector graphics objects Returns: - output_file + Path: output_file if successful Note: This hook will be called from child processes. Modifying global state will not affect the main process or other child processes. @@ -126,7 +126,7 @@ def rasterize_pdf_page( @hookspec(firstresult=True) -def filter_ocr_image(page: 'PageContext', image: Image) -> Image: +def filter_ocr_image(page: 'PageContext', image: 'Image') -> 'Image': """Called to filter the image before it is sent to OCR. This is the image that OCR sees, not what the user sees when they view the @@ -159,20 +159,46 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) +"""Expresses an OCR engine's confidence in page rotation. + +Attributes: + angle (int): The clockwise angle (0, 90, 180, 270) that the page should be + rotated. 0 means no rotation. + confidence (float): How confident the OCR engine is that this the correct + rotation. 0 is not confident, 15 is very confident. Arbitrary units. +""" class OcrEngine(ABC): + """A class representing an OCR engine with capabilities similar to Tesseract OCR. + + This could be used to create a plugin for another OCR engine instead of + Tesseract OCR. + """ + @abstractstaticmethod def version() -> str: """Returns the version of the OCR engine.""" @abstractstaticmethod def creator_tag(options: Namespace) -> str: - """Returns the creator tag to identify this software's role in creating the PDF.""" + """Returns the creator tag to identify this software's role in creating the PDF. + + This tag will be inserted in the XMP metadata and DocumentInfo dictionary + as appropriate. Ideally you should include the name of the OCR engine and its + version. The text should not contain line breaks. This is to help developers + like yourself identify the software that produced this file. + + OCRmyPDF will always prepend its name to this value. + """ @abstractmethod def __str__(self): - """Returns name of OCR engine and version.""" + """Returns name of OCR engine and version. + + This is used when OCRmyPDF wants to mention the name of the OCR engine + to the user, usually in an error message. + """ @abstractstaticmethod def languages(options: Namespace) -> AbstractSet[str]: @@ -248,5 +274,5 @@ def generate_pdfa( pdfa_part: The desired PDF/A compliance level, such as ``'2B'``. Returns: - output_file: If successful, the hook should return ``output_file``. + Path: If successful, the hook should return ``output_file``. """ diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index 5cc540fa..9e4db351 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -36,18 +36,12 @@ log = logging.getLogger(__name__) def run(args, *, env=None, **kwargs): - """Wrapper around subprocess.run() - - The main purpose of this wrapper is to log subprocess output. - - Secondly we have to account for behavioral differences in Windows in particular. - Creating symbolic links in Windows requires administrator privileges and - may not work if for some reason we're using a FAT file system or the temporary - folder is on a different drive from the working folder. The test suite - works around this by creating shim Python scripts that perform the same function - as a symbolic link, but those shims require support on this side, to ensure - we call them with Python. + """Wrapper around :py:func:`subprocess.run` + The main purpose of this wrapper is to log subprocess output in an orderly + fashion that indentifies the responsible subprocess. An additional + task is that this function goes to greater lengths to find possible Windows + locations of our dependencies when they are not on the system PATH. """ if not env: env = os.environ @@ -106,8 +100,18 @@ def _fix_windows_args(program, args, env): @lru_cache(maxsize=None) -def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None): - """Get the version of the specified program""" +def get_version( + program: str, *, version_arg: str = '--version', regex=r'(\d+(\.\d+)*)', env=None +): + """Get the version of the specified program + + Arguments: + program: The program to version check. + version_arg: The argument needed to ask for its version, e.g. ``--version``. + regex: A regular expression to parse the program's output and obtain the + version. + env: Custom ``os.environ`` in which to run program. + """ args_prog = [program, version_arg] try: proc = run( From 378f54361929b0e3ff9390f9a2e45d57eb84e277 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jun 2020 00:41:36 -0700 Subject: [PATCH 574/880] TextPositionTracker: set boxes_flow=None We don't care about the order of lines in our analysis, and this is an expensive calculation in pdfminer. --- src/ocrmypdf/pdfinfo/layout.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 9eb8306a..98bd82e0 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -221,7 +221,7 @@ class TextPositionTracker(PDFLayoutAnalyzer): def get_page_analysis(infile, pageno, pscript5_mode): rman = pdfminer.pdfinterp.PDFResourceManager(caching=True) dev = TextPositionTracker( - rman, laparams=LAParams(all_texts=True, detect_vertical=True) + rman, laparams=LAParams(all_texts=True, detect_vertical=True, boxes_flow=None) ) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) From dc42beb6a87f863330138c8a4a957f77b52ab5e0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 30 Jun 2020 15:02:30 -0700 Subject: [PATCH 575/880] More typing improvements Typing fixes bugs. --- src/ocrmypdf/_plugin_manager.py | 14 +++++---- src/ocrmypdf/pdfa.py | 3 +- src/ocrmypdf/pdfinfo/info.py | 54 +++++++++++++++++---------------- 3 files changed, 38 insertions(+), 33 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 13802f12..cff09bbd 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -69,13 +69,15 @@ def _setup_plugins( ): pm.add_hookspecs(pluginspec) + all_plugins: List[Union[str, Path]] = [] if builtins: - all_plugins = [ - 'ocrmypdf.builtin_plugins.ghostscript', - 'ocrmypdf.builtin_plugins.tesseract_ocr', - ] + plugins - else: - all_plugins = plugins + all_plugins.extend( + [ + 'ocrmypdf.builtin_plugins.ghostscript', + 'ocrmypdf.builtin_plugins.tesseract_ocr', + ] + ) + all_plugins.extend(plugins) for name in all_plugins: if isinstance(name, Path) or name.endswith('.py'): # Import by filename diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 7fd42194..46334020 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -22,6 +22,7 @@ Utilities for PDF/A production and confirmation with Ghostspcript. import base64 from pathlib import Path from string import Template +from typing import Dict, Union import pikepdf import pkg_resources @@ -115,7 +116,7 @@ def file_claims_pdfa(filename: Path): } valid_part_conforms = {'1A', '1B', '2A', '2B', '2U', '3A', '3B', '3U'} conformance = f'PDF/A-{pdfmeta.pdfa_status}' - pdfa_dict = {} + pdfa_dict: Dict[str, Union[str, bool]] = {} if pdfmeta.pdfa_status in valid_part_conforms: pdfa_dict['pass'] = True pdfa_dict['output'] = 'pdfa' diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index b9aad6d3..8f15e6b9 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -25,7 +25,7 @@ from functools import partial from math import hypot, isclose from os import PathLike from pathlib import Path -from typing import Any, Dict, List +from typing import Any, Dict, List, Optional, Union from warnings import warn import pikepdf @@ -237,7 +237,7 @@ def _get_dpi(ctm_shorthand, image_size): width-axis vector v0 (1, 0), height-axis vector vh (0, 1) with the matrix, which gives the dimensions of the image in PDF units. From there we can compare to actual image dimensions. PDF uses - row vector * matrix_tranposed unlike the traditional + row vector * matrix_transposed unlike the traditional matrix * column vector. The offset, width and height vectors can be combined in a matrix and @@ -523,7 +523,7 @@ def _process_content_streams(*, pdf, container, shorthand=None): yield from _find_form_xobject_images(pdf, container, contentsinfo) -def _page_has_text(text_blocks, page_width, page_height): +def _page_has_text(text_blocks, page_width, page_height) -> bool: """Smarter text detection that ignores text in margins""" pw, ph = float(page_width), float(page_height) @@ -567,7 +567,7 @@ def simplify_textboxes(miner, textbox_getter): def _pdf_get_pageinfo( - pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis + pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool ): pageinfo: Dict[str, Any] = {} pageinfo['pageno'] = pageno @@ -696,41 +696,41 @@ class PageInfo: ) @property - def pageno(self): + def pageno(self) -> int: return self._pageno @property - def has_text(self): + def has_text(self) -> bool: return self._pageinfo['has_text'] @property - def has_corrupt_text(self): + def has_corrupt_text(self) -> bool: if not self._detailed_analysis: raise NotImplementedError('Did not do detailed analysis') return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) @property - def has_vector(self): + def has_vector(self) -> bool: return self._pageinfo['has_vector'] @property - def width_inches(self): + def width_inches(self) -> Decimal: return self._pageinfo['width_inches'] @property - def height_inches(self): + def height_inches(self) -> Decimal: return self._pageinfo['height_inches'] @property - def width_pixels(self): + def width_pixels(self) -> int: return int(round(float(self.width_inches) * self.dpi.x)) @property - def height_pixels(self): + def height_pixels(self) -> int: return int(round(float(self.height_inches) * self.dpi.y)) @property - def rotation(self): + def rotation(self) -> int: return self._pageinfo.get('rotate', None) @rotation.setter @@ -744,7 +744,9 @@ class PageInfo: def images(self): return self._pageinfo['images'] - def get_textareas(self, visible=None, corrupt=None): + def get_textareas( + self, visible: Optional[bool] = None, corrupt: Optional[bool] = None + ): def predicate(obj, want_visible, want_corrupt): result = True if want_visible is not None: @@ -767,15 +769,15 @@ class PageInfo: ) @property - def dpi(self): + def dpi(self) -> Resolution: return self._pageinfo.get('dpi', Resolution(0.0, 0.0)) @property - def userunit(self): + def userunit(self) -> Decimal: return self._pageinfo.get('userunit', None) @property - def min_version(self): + def min_version(self) -> str: if self.userunit is not None: return '1.6' else: @@ -795,9 +797,9 @@ class PdfInfo: def __init__( self, infile, - detailed_analysis=False, - progbar=False, - max_workers=None, + detailed_analysis: bool = False, + progbar: bool = False, + max_workers: int = None, check_pages=None, ): self._infile = infile @@ -828,29 +830,29 @@ class PdfInfo: return self._pages @property - def min_version(self): + def min_version(self) -> str: # The minimum PDF is the maximum version that any particular page needs return max(page.min_version for page in self.pages) @property - def has_userunit(self): + def has_userunit(self) -> bool: return any(page.userunit != 1.0 for page in self.pages) @property - def has_acroform(self): + def has_acroform(self) -> bool: return self._has_acroform @property - def filename(self): + def filename(self) -> Union[str, Path]: if not isinstance(self._infile, (str, Path)): raise NotImplementedError("can't get filename from stream") return self._infile @property - def needs_rendering(self): + def needs_rendering(self) -> bool: return self._needs_rendering - def __getitem__(self, item): + def __getitem__(self, item) -> PageInfo: return self._pages[item] def __len__(self): From 1722cb579ddb7dc7fda745c6d6349f2bb3e99898 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Jul 2020 03:26:57 -0700 Subject: [PATCH 576/880] v10.2.1 release notes --- docs/release_notes.rst | 9 +++++++++ requirements/main.txt | 2 +- 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index d7e3dde4..017c6364 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,15 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.2.1 +======= + +- Disabled calculation of text box order with pdfminer. We never needed this result + and it is expensive to calculate on files with complex pre-existing text. +- Fixed plugin manager to accept ``Path(plugin)`` as a path to a plugin. +- Fixed some typing errors. +- Documentation improvements. + v10.2.0 ======= diff --git a/requirements/main.txt b/requirements/main.txt index 74be53e4..5247e369 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ cffi == 1.14.0 coloredlogs == 14.0 # technically optional img2pdf == 0.3.6 pdfminer.six == 20200517 -pikepdf == 1.15.1 +pikepdf == 1.16.1 pluggy == 0.13.1 Pillow == 7.1.2 reportlab == 3.5.42 From 190294634c44f5fffbb73afac9d43337babf5410 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 3 Jul 2020 16:16:01 -0700 Subject: [PATCH 577/880] docs: edit plugins --- docs/plugins.rst | 13 +++++++++++-- src/ocrmypdf/cli.py | 19 +++++++++++-------- 2 files changed, 22 insertions(+), 10 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index d55a01f8..9e6b2269 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -15,7 +15,7 @@ Currently, it is possible to: OCRmyPDF plugins are based on the Python ``pluggy`` package and conform to its conventions. Note that: plugins installed with as setuptools entrypoints are not checked currently, because OCRmyPDF assumes you may not want to enable -plugins for all files. Also, plugins must be functions, not classes. +plugins for all files. How plugins are imported ======================== @@ -32,10 +32,12 @@ Script plugins ============== Script plugins may be called from the command line, by specifying the name of a file. +Script plugins may be convenient for informal or "one-off" plugins, when a certain +batch of files needs a special processing step for example. .. code-block:: bash - ocrmypdf --plugin example_plugin.py input.pdf output.pdf + ocrmypdf --plugin ocrmypdf_example_plugin.py input.pdf output.pdf Multiple plugins may be called by issuing the ``--plugin`` argument multiple times. @@ -44,6 +46,7 @@ Packaged plugins Installed plugins may be installed into the same virtual environment as OCRmyPDF is installed into. They may be invoked using Python standard module naming. +If you are intending to distribute a plugin, please package it. .. code-block:: bash @@ -59,6 +62,12 @@ as packaged plugins, and these modules should begin with the name ``ocrmypdf_`` similar to ``pytest`` packages such as ``pytest-cov`` (the package) and ``pytest_cov`` (the module). +.. note:: + + We strongly recommend plugin authors name their plugins with the prefix + ``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the + module), just like pytest plugins. + Plugin hooks ============ diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index fd76f7cc..ad867670 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -83,8 +83,8 @@ image, and then creates a PDF from the OCR information. """, epilog="""\ OCRmyPDF attempts to keep the output file at about the same size. If a file -contains losslessly compressed images, and output file will be losslessly -compressed as well. +contains losslessly compressed images, and images in the output file will be +losslessly compressed as well. PDF is a page description file that attempts to preserve a layout exactly. A PDF can contain vector objects (such as text or lines) and raster objects @@ -103,9 +103,8 @@ all objects on the page and produce an image-only PDF as output. If you are concerned about long-term archiving of PDFs, use the default option --output-type pdfa which converts the PDF to a standardized PDF/A-2b. This -converts images to sRGB colorspace, removes some features from the PDF such -as Javascript or forms. If you want to minimize the number of changes made to -your PDF, use --output-type pdf. +removes some features from the PDF such as Javascript or forms. If you want to +minimize the number of changes made to your PDF, use --output-type pdf. If OCRmyPDF is given an image file as input, it will attempt to convert the image to a PDF before processing. For more control over the conversion of @@ -113,8 +112,7 @@ images to PDF, use the Python package img2pdf or other image to PDF software. For example, this command uses img2pdf to convert all .png files beginning with the 'page' prefix to a PDF, fitting each image on A4-sized paper, and -sending the result to OCRmyPDF through a pipe. img2pdf is a dependency of -ocrmypdf so it is already installed. +sending the result to OCRmyPDF through a pipe. img2pdf --pagesize A4 page*.png | ocrmypdf - myfile.pdf @@ -462,7 +460,12 @@ Online documentation is located at: dest='plugins', action='append', default=[], - help="Name of plugin to import.", + help="Name of plugin to import. Argument may be issued multiple times to " + "import multiple plugins. Plugins may be specified as module names in " + "Python syntax, provided they are installed in the same Python (virtual) " + "environment as ocrmypdf; or you may give the path to the Python file that " + "contains the plugin. Plugins must conform to the specification in the " + "OCRmyPDF documentation.", ) debugging = parser.add_argument_group( From 60be64a5f10ffb1822d0a998b28dad005a6a1a2a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 4 Jul 2020 03:59:38 -0700 Subject: [PATCH 578/880] Fix debug.log missing pageno handler --- src/ocrmypdf/_sync.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index f7d594a6..c11b44fa 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -30,6 +30,7 @@ import PIL from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._graft import OcrGrafter from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files +from ocrmypdf._logging import PageNumberFilter from ocrmypdf._pipeline import ( convert_to_pdfa, copy_final, @@ -309,6 +310,7 @@ def configure_debug_logging(log_filename, prefix=''): '[%(asctime)s] - %(name)s - %(levelname)7s -%(pageno)s %(message)s' ) log_file_handler.setFormatter(formatter) + log_file_handler.addFilter(PageNumberFilter()) logging.getLogger(prefix).addHandler(log_file_handler) return log_file_handler From 26a415c5dd2226c79084e418ac336afa399723cf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 7 Jul 2020 21:26:57 -0700 Subject: [PATCH 579/880] docs: Note usage of OCR_JSON_SETTINGS for watcher --- docs/batch.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/batch.rst b/docs/batch.rst index 9d484790..99c3e2ae 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -127,7 +127,7 @@ Users may need to customize the script to meet their requirements. "OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)" "OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``" "OCR_DESKEW", "Apply deskew to crooked input PDFs" - "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``" + "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={"rotate_pages": true}'``. "OCR_POLL_NEW_FILE_SECONDS", "Polling interval" "OCR_LOGLEVEL", "Level of log messages to report" From 49734d5456af7f4647bd71dbadfd7712083446bc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 7 Jul 2020 21:52:11 -0700 Subject: [PATCH 580/880] optimize: fix incorrect to prevent re-optimizing JBIG2s --- src/ocrmypdf/optimize.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index d1a346ab..10c6037c 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -91,7 +91,7 @@ def extract_image_jbig2(*, pike, root, image, xref, options): if ( pim.bits_per_component == 1 - and filtdp != Name.JBIG2Decode + and filtdp[0] != Name.JBIG2Decode and jbig2enc.available() ): try: From b20a6e4c5d7b00f4e043e96e725ad98c759f2ca4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 7 Jul 2020 22:18:50 -0700 Subject: [PATCH 581/880] optimize: add type hints --- src/ocrmypdf/optimize.py | 68 +++++++++++++++++++++++++++------------- 1 file changed, 47 insertions(+), 21 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 10c6037c..f04fa79a 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -22,9 +22,10 @@ from collections import defaultdict from functools import partial from os import fspath from pathlib import Path +from typing import Any, Callable, Dict, Iterator, List, Optional, Sequence, Tuple, Union import pikepdf -from pikepdf import Dictionary, Name +from pikepdf import Dictionary, Name, Object, Pdf, PdfImage from PIL import Image from tqdm import tqdm @@ -41,30 +42,32 @@ DEFAULT_JPEG_QUALITY = 75 DEFAULT_PNG_QUALITY = 70 -def img_name(root, xref, ext): +def img_name(root: Path, xref: int, ext: str) -> str: return fspath(root / f'{xref:08d}{ext}') -def png_name(root, xref): +def png_name(root: Path, xref: int) -> str: return img_name(root, xref, '.png') -def jpg_name(root, xref): +def jpg_name(root: Path, xref: int) -> str: return img_name(root, xref, '.jpg') -def tif_name(root, xref): +def tif_name(root: Path, xref: int) -> str: return img_name(root, xref, '.tif') -def extract_image_filter(pike, root, image, xref): +def extract_image_filter( + pike: Pdf, root: Path, image: Object, xref: int +) -> Optional[Tuple[PdfImage, Tuple[Name, Any]]]: if image.Subtype != Name.Image: return None if image.Length < 100: log.debug("Skipping small image, xref %s", xref) return None - pim = pikepdf.PdfImage(image) + pim = PdfImage(image) if len(pim.filter_decodeparms) > 1: log.debug("Skipping multiply filtered, xref %s", xref) @@ -83,7 +86,9 @@ def extract_image_filter(pike, root, image, xref): return pim, filtdp -def extract_image_jbig2(*, pike, root, image, xref, options): +def extract_image_jbig2( + *, pike: pikepdf.Pdf, root: Path, image: Object, xref: int, options +) -> Optional[Tuple[int, str]]: result = extract_image_filter(pike, root, image, xref) if result is None: return None @@ -105,7 +110,9 @@ def extract_image_jbig2(*, pike, root, image, xref, options): return None -def extract_image_generic(*, pike, root, image, xref, options): +def extract_image_generic( + *, pike: Pdf, root: Path, image: PdfImage, xref: int, options +) -> Optional[Tuple[int, str]]: result = extract_image_filter(pike, root, image, xref) if result is None: return None @@ -174,7 +181,12 @@ def extract_image_generic(*, pike, root, image, xref, options): return None -def extract_images(pike, root, options, extract_fn): +def extract_images( + pike: Pdf, + root: Path, + options, + extract_fn: Callable[..., Optional[Tuple[int, str]]], +) -> Iterator[Tuple[int, int, str]]: """Extract image using extract_fn Enumerate images on each page, lookup their xref/ID number in the PDF. @@ -227,7 +239,9 @@ def extract_images(pike, root, options, extract_fn): yield pageno_for_xref[xref], xref, ext -def extract_images_generic(pike, root, options): +def extract_images_generic( + pike: Pdf, root: Path, options +) -> Tuple[List[int], List[int]]: """Extract any >=2bpp image we think we can improve""" jpegs = [] @@ -242,7 +256,9 @@ def extract_images_generic(pike, root, options): return jpegs, pngs -def extract_images_jbig2(pike, root, options): +def extract_images_jbig2( + pike: Pdf, root: Path, options +) -> Dict[int, List[Tuple[int, str]]]: """Extract any bitonal image that we think we can improve as JBIG2""" jbig2_groups = defaultdict(list) @@ -258,10 +274,12 @@ def extract_images_jbig2(pike, root, options): return jbig2_groups -def _produce_jbig2_images(jbig2_groups, root, options): +def _produce_jbig2_images( + jbig2_groups: Dict[int, List[Tuple[int, str]]], root: Path, options +) -> None: """Produce JBIG2 images from their groups""" - def jbig2_group_args(root, groups): + def jbig2_group_args(root: Path, groups: Dict[int, List[Tuple[int, str]]]): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' yield dict( @@ -270,7 +288,7 @@ def _produce_jbig2_images(jbig2_groups, root, options): out_prefix=prefix, ) - def jbig2_single_args(root, groups): + def jbig2_single_args(root, groups: Dict[int, List[Tuple[int, str]]]): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' # Second loop is to ensure multiple images per page are unpacked @@ -306,7 +324,9 @@ def _produce_jbig2_images(jbig2_groups, root, options): ) -def convert_to_jbig2(pike, jbig2_groups, root, options): +def convert_to_jbig2( + pike: Pdf, jbig2_groups: Dict[int, List[Tuple[int, str]]], root: Path, options +) -> None: """Convert images to JBIG2 and insert into PDF. When the JBIG2 page group size is > 1 we do several JBIG2 images at once @@ -344,7 +364,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, options): ) -def transcode_jpegs(pike, jpegs, root, options): +def transcode_jpegs(pike: Pdf, jpegs: Sequence[int], root: Path, options) -> None: for xref in tqdm( jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar ): @@ -367,7 +387,13 @@ def transcode_jpegs(pike, jpegs, root, options): im_obj.write(compdata.read(), filter=Name.DCTDecode) -def transcode_pngs(pike, images, image_name_fn, root, options): +def transcode_pngs( + pike: Pdf, + images: Sequence[int], + image_name_fn: Callable[[Path, int], str], + root: Path, + options, +) -> None: modified = set() if options.optimize >= 2: png_quality = ( @@ -431,7 +457,7 @@ def transcode_pngs(pike, images, image_name_fn, root, options): rewrite_png_as_g4(pike, im_obj, compdata) -def rewrite_png_as_g4(pike, im_obj, compdata): +def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: im_obj.BitsPerComponent = 1 im_obj.Width = compdata.w im_obj.Height = compdata.h @@ -451,7 +477,7 @@ def rewrite_png_as_g4(pike, im_obj, compdata): return -def rewrite_png(pike, im_obj, compdata): +def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # When a PNG is inserted into a PDF, we more or less copy the IDAT section from # the PDF and transfer the rest of the PNG headers to PDF image metadata. # One thing we have to do is tell the PDF reader whether a predictor was used @@ -504,7 +530,7 @@ def rewrite_png(pike, im_obj, compdata): im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) -def optimize(input_file, output_file, context, save_settings): +def optimize(input_file: Path, output_file: Path, context, save_settings) -> None: options = context.options if options.optimize == 0: safe_symlink(input_file, output_file) From 373f27832bbe2f6d1f81299891dfbc7f80c9fcb7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 7 Jul 2020 22:41:29 -0700 Subject: [PATCH 582/880] optimize: improve typing of xref_exts --- src/ocrmypdf/optimize.py | 74 +++++++++++++++++++++++----------------- 1 file changed, 43 insertions(+), 31 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index f04fa79a..245f505d 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -22,7 +22,19 @@ from collections import defaultdict from functools import partial from os import fspath from pathlib import Path -from typing import Any, Callable, Dict, Iterator, List, Optional, Sequence, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + Iterator, + List, + MutableSet, + NamedTuple, + Optional, + Sequence, + Tuple, + Union, +) import pikepdf from pikepdf import Dictionary, Name, Object, Pdf, PdfImage @@ -42,6 +54,11 @@ DEFAULT_JPEG_QUALITY = 75 DEFAULT_PNG_QUALITY = 70 +class XrefExt(NamedTuple): + xref: int + ext: str + + def img_name(root: Path, xref: int, ext: str) -> str: return fspath(root / f'{xref:08d}{ext}') @@ -60,7 +77,7 @@ def tif_name(root: Path, xref: int) -> str: def extract_image_filter( pike: Pdf, root: Path, image: Object, xref: int -) -> Optional[Tuple[PdfImage, Tuple[Name, Any]]]: +) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]: if image.Subtype != Name.Image: return None if image.Length < 100: @@ -88,7 +105,7 @@ def extract_image_filter( def extract_image_jbig2( *, pike: pikepdf.Pdf, root: Path, image: Object, xref: int, options -) -> Optional[Tuple[int, str]]: +) -> Optional[XrefExt]: result = extract_image_filter(pike, root, image, xref) if result is None: return None @@ -106,13 +123,13 @@ def extract_image_jbig2( imgname.rename(imgname.with_suffix(ext)) except pikepdf.UnsupportedImageTypeError: return None - return xref, ext + return XrefExt(xref, ext) return None def extract_image_generic( *, pike: Pdf, root: Path, image: PdfImage, xref: int, options -) -> Optional[Tuple[int, str]]: +) -> Optional[XrefExt]: result = extract_image_filter(pike, root, image, xref) if result is None: return None @@ -151,7 +168,7 @@ def extract_image_generic( imgname.rename(imgname.with_suffix(ext)) except pikepdf.UnsupportedImageTypeError: return None - return xref, ext + return XrefExt(xref, ext) elif ( pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES @@ -160,12 +177,12 @@ def extract_image_generic( # Try to improve on indexed images - these are far from low hanging # fruit in most cases pim.as_pil_image().save(png_name(root, xref)) - return xref, '.png' + return XrefExt(xref, '.png') elif not pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES: # An optimization opportunity here, not currently taken, is directly # generating a PNG from compressed data pim.as_pil_image().save(png_name(root, xref)) - return xref, '.png' + return XrefExt(xref, '.png') elif ( not pim.indexed and pim.colorspace == Name.ICCBased @@ -176,17 +193,14 @@ def extract_image_generic( # paying any attention to the ICC profile, provided we're not doing # lossy JBIG2 pim.as_pil_image().save(png_name(root, xref)) - return xref, '.png' + return XrefExt(xref, '.png') return None def extract_images( - pike: Pdf, - root: Path, - options, - extract_fn: Callable[..., Optional[Tuple[int, str]]], -) -> Iterator[Tuple[int, int, str]]: + pike: Pdf, root: Path, options, extract_fn: Callable[..., Optional[XrefExt]], +) -> Iterator[Tuple[int, XrefExt]]: """Extract image using extract_fn Enumerate images on each page, lookup their xref/ID number in the PDF. @@ -236,7 +250,7 @@ def extract_images( else: if result: _, ext = result - yield pageno_for_xref[xref], xref, ext + yield pageno_for_xref[xref], XrefExt(xref, ext) def extract_images_generic( @@ -246,25 +260,23 @@ def extract_images_generic( jpegs = [] pngs = [] - for _, xref, ext in extract_images(pike, root, options, extract_image_generic): - log.debug('xref = %s ext = %s', xref, ext) - if ext == '.png': - pngs.append(xref) - elif ext == '.jpg': - jpegs.append(xref) + for _, xref_ext in extract_images(pike, root, options, extract_image_generic): + log.debug('xref_ext = %s', xref_ext) + if xref_ext.ext == '.png': + pngs.append(xref_ext.xref) + elif xref_ext.ext == '.jpg': + jpegs.append(xref_ext.xref) log.debug("Optimizable images: JPEGs: %s PNGs: %s", len(jpegs), len(pngs)) return jpegs, pngs -def extract_images_jbig2( - pike: Pdf, root: Path, options -) -> Dict[int, List[Tuple[int, str]]]: +def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefExt]]: """Extract any bitonal image that we think we can improve as JBIG2""" jbig2_groups = defaultdict(list) - for pageno, xref, ext in extract_images(pike, root, options, extract_image_jbig2): + for pageno, xref_ext in extract_images(pike, root, options, extract_image_jbig2): group = pageno // options.jbig2_page_group_size - jbig2_groups[group].append((xref, ext)) + jbig2_groups[group].append(xref_ext) # Elide empty groups jbig2_groups = { @@ -275,11 +287,11 @@ def extract_images_jbig2( def _produce_jbig2_images( - jbig2_groups: Dict[int, List[Tuple[int, str]]], root: Path, options + jbig2_groups: Dict[int, List[XrefExt]], root: Path, options ) -> None: """Produce JBIG2 images from their groups""" - def jbig2_group_args(root: Path, groups: Dict[int, List[Tuple[int, str]]]): + def jbig2_group_args(root: Path, groups: Dict[int, List[XrefExt]]): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' yield dict( @@ -288,7 +300,7 @@ def _produce_jbig2_images( out_prefix=prefix, ) - def jbig2_single_args(root, groups: Dict[int, List[Tuple[int, str]]]): + def jbig2_single_args(root, groups: Dict[int, List[XrefExt]]): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' # Second loop is to ensure multiple images per page are unpacked @@ -325,7 +337,7 @@ def _produce_jbig2_images( def convert_to_jbig2( - pike: Pdf, jbig2_groups: Dict[int, List[Tuple[int, str]]], root: Path, options + pike: Pdf, jbig2_groups: Dict[int, List[XrefExt]], root: Path, options ) -> None: """Convert images to JBIG2 and insert into PDF. @@ -394,7 +406,7 @@ def transcode_pngs( root: Path, options, ) -> None: - modified = set() + modified: MutableSet[int] = set() if options.optimize >= 2: png_quality = ( max(10, options.png_quality - 10), From e33ba07aa4fc85ae120cef5e7a918eae72eb475b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 8 Jul 2020 23:45:53 -0700 Subject: [PATCH 583/880] Update pre-commit settings --- .pre-commit-config.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 544da6f1..05d84fd1 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,6 @@ repos: - repo: https://github.com/pre-commit/pre-commit-hooks - rev: v2.4.0 + rev: v3.1.0 hooks: - id: check-case-conflict - id: check-merge-conflict @@ -8,16 +8,16 @@ repos: - id: check-yaml - id: debug-statements - repo: https://github.com/asottile/seed-isort-config - rev: v1.9.3 + rev: v2.2.0 hooks: - id: seed-isort-config - repo: https://github.com/pre-commit/mirrors-isort - rev: v4.3.21 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases + rev: v5.0.5 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases hooks: - id: isort - repo: https://github.com/psf/black - rev: stable + rev: 19.10b0 hooks: - id: black - language_version: python3.7 + language_version: python3.8 exclude: ^src/ocrmypdf/lib/_leptonica.py From a510b21b20a24835947760271f15c00cb2a237fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 9 Jul 2020 02:20:01 -0700 Subject: [PATCH 584/880] optimize: add typing for Xref, remove fspath()'s --- src/ocrmypdf/_exec/pngquant.py | 3 ++ src/ocrmypdf/optimize.py | 63 ++++++++++++++++++---------------- 2 files changed, 37 insertions(+), 29 deletions(-) diff --git a/src/ocrmypdf/_exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py index 61f197fe..cb155aeb 100644 --- a/src/ocrmypdf/_exec/pngquant.py +++ b/src/ocrmypdf/_exec/pngquant.py @@ -17,6 +17,7 @@ """Interface to pngquant executable""" +from os import fspath from tempfile import NamedTemporaryFile from PIL import Image @@ -38,6 +39,8 @@ def available(): def quantize(input_file, output_file, quality_min, quality_max): + input_file = fspath(input_file) + output_file = fspath(output_file) if input_file.endswith('.jpg'): with Image.open(input_file) as im, NamedTemporaryFile(suffix='.png') as tmp: im.save(tmp) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 245f505d..2ff5989f 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -30,6 +30,7 @@ from typing import ( List, MutableSet, NamedTuple, + NewType, Optional, Sequence, Tuple, @@ -54,29 +55,32 @@ DEFAULT_JPEG_QUALITY = 75 DEFAULT_PNG_QUALITY = 70 +Xref = NewType('Xref', int) + + class XrefExt(NamedTuple): - xref: int + xref: Xref ext: str -def img_name(root: Path, xref: int, ext: str) -> str: - return fspath(root / f'{xref:08d}{ext}') +def img_name(root: Path, xref: Xref, ext: str) -> Path: + return root / f'{xref:08d}{ext}' -def png_name(root: Path, xref: int) -> str: +def png_name(root: Path, xref: Xref) -> Path: return img_name(root, xref, '.png') -def jpg_name(root: Path, xref: int) -> str: +def jpg_name(root: Path, xref: Xref) -> Path: return img_name(root, xref, '.jpg') -def tif_name(root: Path, xref: int) -> str: +def tif_name(root: Path, xref: Xref) -> Path: return img_name(root, xref, '.tif') def extract_image_filter( - pike: Pdf, root: Path, image: Object, xref: int + pike: Pdf, root: Path, image: Object, xref: Xref ) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]: if image.Subtype != Name.Image: return None @@ -104,7 +108,7 @@ def extract_image_filter( def extract_image_jbig2( - *, pike: pikepdf.Pdf, root: Path, image: Object, xref: int, options + *, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options ) -> Optional[XrefExt]: result = extract_image_filter(pike, root, image, xref) if result is None: @@ -117,7 +121,7 @@ def extract_image_jbig2( and jbig2enc.available() ): try: - imgname = Path(root / f'{xref:08d}') + imgname = root / f'{xref:08d}' with imgname.open('wb') as f: ext = pim.extract_to(stream=f) imgname.rename(imgname.with_suffix(ext)) @@ -128,7 +132,7 @@ def extract_image_jbig2( def extract_image_generic( - *, pike: Pdf, root: Path, image: PdfImage, xref: int, options + *, pike: Pdf, root: Path, image: PdfImage, xref: Xref, options ) -> Optional[XrefExt]: result = extract_image_filter(pike, root, image, xref) if result is None: @@ -162,7 +166,7 @@ def extract_image_generic( # with Image.open(stream) as im: # im.save(jpg_name(root, xref), icc_profile=iccbytes) try: - imgname = Path(root / f'{xref:08d}') + imgname = root / f'{xref:08d}' with imgname.open('wb') as f: ext = pim.extract_to(stream=f) imgname.rename(imgname.with_suffix(ext)) @@ -216,8 +220,8 @@ def extract_images( extension. extract_fn must also extract the file it finds interesting. """ - include_xrefs = set() - exclude_xrefs = set() + include_xrefs: MutableSet[Xref] = set() + exclude_xrefs: MutableSet[Xref] = set() pageno_for_xref = {} errors = 0 for pageno, page in enumerate(pike.pages): @@ -228,10 +232,10 @@ def extract_images( for _imname, image in dict(xobjs).items(): if image.objgen[1] != 0: continue # Ignore images in an incremental PDF - xref = image.objgen[0] + xref = Xref(image.objgen[0]) if hasattr(image, 'SMask'): # Ignore soft masks - smask_xref = image.SMask.objgen[0] + smask_xref = Xref(image.SMask.objgen[0]) exclude_xrefs.add(smask_xref) include_xrefs.add(xref) if xref not in pageno_for_xref: @@ -255,13 +259,13 @@ def extract_images( def extract_images_generic( pike: Pdf, root: Path, options -) -> Tuple[List[int], List[int]]: +) -> Tuple[List[Xref], List[Xref]]: """Extract any >=2bpp image we think we can improve""" jpegs = [] pngs = [] for _, xref_ext in extract_images(pike, root, options, extract_image_generic): - log.debug('xref_ext = %s', xref_ext) + log.debug('%s', xref_ext) if xref_ext.ext == '.png': pngs.append(xref_ext.xref) elif xref_ext.ext == '.jpg': @@ -376,19 +380,19 @@ def convert_to_jbig2( ) -def transcode_jpegs(pike: Pdf, jpegs: Sequence[int], root: Path, options) -> None: +def transcode_jpegs(pike: Pdf, jpegs: Sequence[Xref], root: Path, options) -> None: for xref in tqdm( jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar ): - in_jpg = Path(jpg_name(root, xref)) + in_jpg = jpg_name(root, xref) opt_jpg = in_jpg.with_suffix('.opt.jpg') # This produces a debug warning from PIL # DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute # 'close'. Seems to be mostly harmless # https://github.com/python-pillow/Pillow/issues/1144 - with Image.open(fspath(in_jpg)) as im: - im.save(fspath(opt_jpg), optimize=True, quality=options.jpeg_quality) + with Image.open(in_jpg) as im: + im.save(opt_jpg, optimize=True, quality=options.jpeg_quality) if opt_jpg.stat().st_size > in_jpg.stat().st_size: log.debug("xref %s, jpeg, made larger - skip", xref) @@ -401,12 +405,12 @@ def transcode_jpegs(pike: Pdf, jpegs: Sequence[int], root: Path, options) -> Non def transcode_pngs( pike: Pdf, - images: Sequence[int], - image_name_fn: Callable[[Path, int], str], + images: Sequence[Xref], + image_name_fn: Callable[[Path, Xref], Path], root: Path, options, ) -> None: - modified: MutableSet[int] = set() + modified: MutableSet[Xref] = set() if options.optimize >= 2: png_quality = ( max(10, options.png_quality - 10), @@ -556,7 +560,7 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1 with pikepdf.Pdf.open(input_file) as pike: - root = Path(output_file).parent / 'images' + root = output_file.parent / 'images' root.mkdir(exist_ok=True) jpegs, pngs = extract_images_generic(pike, root, options) @@ -569,12 +573,12 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non jbig2_groups = extract_images_jbig2(pike, root, options) convert_to_jbig2(pike, jbig2_groups, root, options) - target_file = Path(output_file).with_suffix('.opt.pdf') + target_file = output_file.with_suffix('.opt.pdf') pike.remove_unreferenced_resources() pike.save(target_file, **save_settings) - input_size = Path(input_file).stat().st_size - output_size = Path(target_file).stat().st_size + input_size = input_file.stat().st_size + output_size = target_file.stat().st_size if output_size == 0: raise OutputFileAccessError( f"Output file not created after optimizing. We probably ran " @@ -595,8 +599,8 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non def main(infile, outfile, level, jobs=1): - from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel from shutil import copy # pylint: disable=import-outside-toplevel + from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel class OptimizeOptions: """Emulate ocrmypdf's options""" @@ -614,6 +618,7 @@ def main(infile, outfile, level, jobs=1): self.quiet = True self.progress_bar = False + infile = Path(infile) options = OptimizeOptions( input_file=infile, jobs=jobs, From d2a9c413f812969651c02e9237917e458489f499 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 14 Jul 2020 01:25:16 -0700 Subject: [PATCH 585/880] docs: install notes for ARM64 --- docs/installation.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/installation.rst b/docs/installation.rst index 9a7541ea..10125a5d 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -665,6 +665,13 @@ instead use this for a system wide installation: pip3 install ocrmypdf +.. note:: + + AArch64 (ARM64) users: this process will be difficult because most + Python packages are not available as binary wheels for your platform. + You're probably better off using a platform install on Debian, Ubuntu, + or Fedora. + Requirements for pip and HEAD install ------------------------------------- From 1558e068f146cfd6c7fcf883deea090eea689eb4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 16 Jul 2020 00:01:59 -0700 Subject: [PATCH 586/880] docs: explain firstresult hook behavior --- docs/plugins.rst | 13 +++++++++++++ src/ocrmypdf/pluginspec.py | 29 +++++++++++++++++++++++++++-- 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index 9e6b2269..3fd62240 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -85,6 +85,19 @@ A plugin may provide the following hooks. Hooks should be decorated with The following is a complete list of hooks that may be installed and when they are called. +.. _firstresult: + +Note on firstresult hooks +^^^^^^^^^^^^^^^^^^^^^^^^^ + +If multiple plugins install implementations for this hook, they will be called in +the reverse of the order in which they are installed (i.e., last plugin wins). +When each hook implementation is called in order, the first implementation that +returns a value other than ``None`` will "win" and prevent execution of all other +hooks. As such, you cannot "chain" a series of plugin filters together in this +way. Instead, a single hook implementation should be responsible for any such +chaining operations. + Custom command line arguments ----------------------------- diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 9371aede..91540cdc 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -26,9 +26,10 @@ import pluggy from ocrmypdf.helpers import Resolution if TYPE_CHECKING: + from PIL import Image + from ocrmypdf._jobcontext import PageContext from ocrmypdf.pdfinfo import PdfInfo - from PIL import Image hookspec = pluggy.HookspecMarker('ocrmypdf') @@ -122,6 +123,8 @@ def rasterize_pdf_page( Note: This hook will be called from child processes. Modifying global state will not affect the main process or other child processes. + Note: + This is a :ref:`firstresult hook`. """ @@ -130,11 +133,25 @@ def filter_ocr_image(page: 'PageContext', image: 'Image') -> 'Image': """Called to filter the image before it is sent to OCR. This is the image that OCR sees, not what the user sees when they view the - PDF. + PDF. If ``redo_ocr`` is enabled, portions of the image will be masked so + they are not shown to OCR. The main use of this hook is expected to be hiding + content from OCR. + + The input image may be color, grayscale, or monochrome, and the + output image may differ. The pixel width and height of the + output image must be identical to the input image, or misalignment between + the OCR text layer and visual position of the text will occur. Likewise, + the output must be a faithful representation of the input, or alignment + errors may occurs. + + Tesseract OCR only deals with monochrome images, and internally converts + non-monochrome images to OCR. Note: This hook will be called from child processes. Modifying global state will not affect the main process or other child processes. + Note: + This is a :ref:`firstresult hook`. """ @@ -155,6 +172,8 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: Note: This hook will be called from child processes. Modifying global state will not affect the main process or other child processes. + Note: + This is a :ref:`firstresult hook`. """ @@ -239,6 +258,9 @@ def get_ocr_engine() -> OcrEngine: The OcrEngine may be instantiated multiple times, by both the main process and child process. As such, it must be obtain store any state in ``options`` or some common location. + + Note: + This is a :ref:`firstresult hook`. """ @@ -275,4 +297,7 @@ def generate_pdfa( Returns: Path: If successful, the hook should return ``output_file``. + + Note: + This is a :ref:`firstresult hook`. """ From ae68edefc5e0d28b1d03bad1865737167fa5b470 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 Jul 2020 01:50:35 -0700 Subject: [PATCH 587/880] pipelines: fix Python 3.7/3.8 on macOS --- azure-pipelines.yml | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 5f88b76b..3604ced3 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -152,9 +152,9 @@ stages: strategy: matrix: Python37: - python.version: "3.7" - # Python38: - # python.version: "3.8" + python.version: "" + Python38: + python.version: "python@3.8" steps: # https://github.com/actions/virtual-environments/issues/664 # - task: UsePythonVersion@0 @@ -163,6 +163,13 @@ stages: - bash: | brew update brew unlink python@2 + if [ "$(python.version)" != "" ]; then + brew upgrade $(python.version) + else + echo "Using Python `python3 --version`" + fi + displayName: "Update brew and Python" + - bash: | brew install \ exempi \ ghostscript \ @@ -170,7 +177,6 @@ stages: leptonica \ openjpeg \ pngquant \ - python \ tesseract \ unpaper displayName: "Install system packages" From 4ea9cffebd547b7b5a6c4aa530ee3d1fd419bf86 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 18 Jul 2020 00:57:11 -0700 Subject: [PATCH 588/880] Add locking to Leptonica error trap To protect another thread from interfering with our redirection of stderr. --- src/ocrmypdf/leptonica.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 6af38089..4aa9b18e 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -90,19 +90,21 @@ class _LeptonicaErrorTrap: """ + leptonica_lock = threading.Lock() + def __init__(self): self.tmpfile = None self.copy_of_stderr = -1 self.no_stderr = False def __enter__(self): - self.tmpfile = TemporaryFile() # Save the old stderr, and redirect stderr to temporary file - with suppress(AttributeError): - sys.stderr.flush() + self.leptonica_lock.acquire() try: + with suppress(AttributeError): + sys.stderr.flush() self.copy_of_stderr = os.dup(sys.stderr.fileno()) os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False) except AttributeError: @@ -114,7 +116,10 @@ class _LeptonicaErrorTrap: os.dup2(self.tmpfile.fileno(), 2, inheritable=False) except UnsupportedOperation: self.copy_of_stderr = None - return + except Exception: + self.leptonica_lock.release() + raise + return self def __exit__(self, exc_type, exc_value, traceback): # Restore old stderr @@ -131,6 +136,8 @@ class _LeptonicaErrorTrap: self.tmpfile.seek(0) # Cursor will be at end, so move back to beginning leptonica_output = self.tmpfile.read().decode(errors='replace') self.tmpfile.close() + self.leptonica_lock.release() + # If there are Python errors, record them if exc_type: logger.warning(leptonica_output) From 5cbbff8472a519687c12cdd0140132e7cdbd5f0f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 Jul 2020 01:50:09 -0700 Subject: [PATCH 589/880] For Leptonica 1.79+ use leptSetStderrHandler Lock free and considerably less dangerous to stderr messages. --- src/ocrmypdf/leptonica.py | 60 ++++++++++++++++++++++++++- src/ocrmypdf/lib/_leptonica.py | 10 ++--- src/ocrmypdf/lib/compile_leptonica.py | 2 + tests/test_lept.py | 24 +++++++---- 4 files changed, 81 insertions(+), 15 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 4aa9b18e..20394d3d 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -24,7 +24,9 @@ import argparse import logging import os import sys +import threading import warnings +from collections import deque from collections.abc import Sequence from contextlib import suppress from ctypes.util import find_library @@ -75,7 +77,7 @@ except ffi.error as e: ) from e -class _LeptonicaErrorTrap: +class _LeptonicaErrorTrap_Redirect: """ Context manager to trap errors reported by Leptonica. @@ -155,6 +157,62 @@ class _LeptonicaErrorTrap: return False +tls = threading.local() +tls.trap = None + + +@ffi.callback("void(char *)") +def _stderr_handler(cstr): + msg = ffi.string(cstr).decode(errors='replace') + if msg.startswith("Error"): + logger.error(msg) + elif msg.startswith("Warning"): + logger.warning(msg) + else: + logger.debug(msg) + if tls.trap is not None: + tls.trap.append(msg) + return + + +class _LeptonicaErrorTrap_Queue: + def __init__(self): + self.queue = deque() + + def __enter__(self): + self.queue.clear() + tls.trap = self.queue + + def __exit__(self, exc_type, exc_value, traceback): + tls.trap = None + output = ''.join(self.queue) + self.queue.clear() + + # If there are Python errors, record them + if exc_type: + logger.warning(output) + + if 'Error' in output: + if 'image file not found' in output: + raise FileNotFoundError() + if 'pixWrite: stream not opened' in output: + raise LeptonicaIOError() + if 'index not valid' in output: + raise IndexError() + raise LeptonicaError(output) + return False + + +try: + lept.leptSetStderrHandler(_stderr_handler) +except ffi.error: + # Pre-1.79 Leptonica does not have leptSetStderrHandler + _LeptonicaErrorTrap = _LeptonicaErrorTrap_Redirect +else: + # 1.79 have this new symbol + _LeptonicaErrorTrap = _LeptonicaErrorTrap_Queue + + class LeptonicaError(Exception): pass diff --git a/src/ocrmypdf/lib/_leptonica.py b/src/ocrmypdf/lib/_leptonica.py index 6d6e2e8b..996d5f8a 100644 --- a/src/ocrmypdf/lib/_leptonica.py +++ b/src/ocrmypdf/lib/_leptonica.py @@ -3,9 +3,9 @@ import _cffi_backend ffi = _cffi_backend.FFI('ocrmypdf.lib._leptonica', _version = 0x2601, - _types = b'\x00\x00\x01\x0D\x00\x01\x56\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x57\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x5B\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x5D\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x5C\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x58\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x18\x11\x00\x00\x05\x03\x00\x00\x11\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x61\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x64\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x76\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x78\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x11\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5F\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x5F\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x5F\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xA4\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x92\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x63\x0D\x00\x00\x4D\x11\x00\x00\x00\x0F\x00\x01\x63\x0D\x00\x00\x00\x0F\x00\x00\x2A\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x3A\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x59\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x18\x11\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x77\x03\x00\x00\x96\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x75\x03\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x25\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x11\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\xA4\x11\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x00\x4D\x03\x00\x00\x00\x0F\x00\x01\x7B\x0D\x00\x01\x7B\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x00\x0A\x09\x00\x01\x5A\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x06\x09\x00\x00\x07\x09\x00\x00\x04\x09\x00\x01\x60\x03\x00\x00\x08\x09\x00\x00\x09\x09\x00\x01\x63\x03\x00\x01\x64\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x2A\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x5E\x03\x00\x01\x73\x03\x00\x01\x74\x03\x00\x00\x05\x09\x00\x01\x76\x03\x00\x00\x04\x01\x00\x01\x78\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', - _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x3E\x23boxDestroy',0,b'\x00\x01\x41\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x01\x19\x23getImpliedFileFormat',0,b'\x00\x00\xBE\x23getLeptonicaVersion',0,b'\x00\x01\x44\x23l_CIDataDestroy',0,b'\x00\x01\x1C\x23l_generateCIDataForPdf',0,b'\x00\x01\x53\x23lept_free',0,b'\x00\x00\xC0\x23makePixelSumTab8',0,b'\x00\x00\x31\x23pixAnd',0,b'\x00\x00\x3E\x23pixBackgroundNorm',0,b'\x00\x00\x36\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x22\x23pixClipRectangle',0,b'\x00\x01\x01\x23pixColorFraction',0,b'\x00\x00\x85\x23pixColorMagnitude',0,b'\x00\x00\x1F\x23pixConvertRGBToLuminance',0,b'\x00\x00\x7C\x23pixConvertTo8',0,b'\x00\x00\xD3\x23pixCorrelationBinary',0,b'\x00\x00\xED\x23pixCountPixels',0,b'\x00\x00\x98\x23pixDeserializeFromMemory',0,b'\x00\x00\x7C\x23pixDeskew',0,b'\x00\x01\x47\x23pixDestroy',0,b'\x00\x00\x4A\x23pixDilate',0,b'\x00\x00\x1F\x23pixEndianByteSwapNew',0,b'\x00\x00\xD8\x23pixEqual',0,b'\x00\x00\x4A\x23pixErode',0,b'\x00\x00\x9C\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xE8\x23pixFindSkew',0,b'\x00\x00\x4F\x23pixGammaTRC',0,b'\x00\x00\x27\x23pixGenHalftoneMask',0,b'\x00\x00\xFA\x23pixGenerateCIData',0,b'\x00\x00\xDD\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x56\x23pixGlobalNormRGB',0,b'\x00\x00\x4A\x23pixHMT',0,b'\x00\x00\x2D\x23pixInvert',0,b'\x00\x00\x15\x23pixLocateBarcodes',0,b'\x00\x00\x80\x23pixMaskOverColorPixels',0,b'\x00\x00\x5E\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xF2\x23pixNumSignificantGrayColors',0,b'\x00\x01\x0A\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x6A\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\xA0\x23pixProcessBarcodes',0,b'\x00\x00\x91\x23pixRead',0,b'\x00\x00\xA7\x23pixReadBarcodes',0,b'\x00\x00\x94\x23pixReadMem',0,b'\x00\x00\x1B\x23pixReadStream',0,b'\x00\x00\x7C\x23pixRemoveColormap',0,b'\x00\x00\x80\x23pixRemoveColormapGeneral',0,b'\x00\x00\xCD\x23pixRenderBoxa',0,b'\x00\x00\x2D\x23pixRotate180',0,b'\x00\x00\x7C\x23pixRotateOrth',0,b'\x00\x00\x77\x23pixScale',0,b'\x00\x01\x14\x23pixSerializeToMemory',0,b'\x00\x00\x31\x23pixSubtract',0,b'\x00\x01\x22\x23pixWriteImpliedFormat',0,b'\x00\x01\x31\x23pixWriteMem',0,b'\x00\x01\x37\x23pixWriteMemJpeg',0,b'\x00\x01\x2B\x23pixWriteMemPng',0,b'\x00\x00\xC2\x23pixWriteStream',0,b'\x00\x00\xC7\x23pixWriteStreamJpeg',0,b'\x00\x01\x4A\x23pixaDestroy',0,b'\x00\x00\x10\x23pixaGetBox',0,b'\x00\x00\x8C\x23pixaGetPix',0,b'\x00\x01\x4D\x23sarrayDestroy',0,b'\x00\x00\xB4\x23selCreateBrick',0,b'\x00\x00\xAE\x23selCreateFromString',0,b'\x00\x01\x50\x23selDestroy',0,b'\x00\x00\xBB\x23selPrintToString',0,b'\x00\x01\x28\x23setMsgSeverity',0), - _struct_unions = ((b'\x00\x00\x01\x56\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x78\x11refcount'),(b'\x00\x00\x01\x57\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x78\x11refcount',b'\x00\x00\x25\x11box'),(b'\x00\x00\x01\x5A\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x75\x11datacomp',b'\x00\x00\x96\x11nbytescomp',b'\x00\x01\x63\x11data85',b'\x00\x00\x96\x11nbytes85',b'\x00\x01\x63\x11cmapdata85',b'\x00\x01\x63\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x96\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x5B\x00\x00\x00\x02Pix',b'\x00\x01\x78\x11w',b'\x00\x01\x78\x11h',b'\x00\x01\x78\x11d',b'\x00\x01\x78\x11spp',b'\x00\x01\x78\x11wpl',b'\x00\x01\x78\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x63\x11text',b'\x00\x01\x71\x11colormap',b'\x00\x01\x77\x11data'),(b'\x00\x00\x01\x5E\x00\x00\x00\x02PixColormap',b'\x00\x01\x54\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x74\x00\x00\x00\x10PixComp',),(b'\x00\x00\x01\x5C\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x78\x11refcount',b'\x00\x00\x18\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x5D\x00\x00\x00\x02PixaComp',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11offset',b'\x00\x01\x72\x11pixc',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x60\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x62\x11array'),(b'\x00\x00\x01\x61\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x6D\x11data',b'\x00\x01\x63\x11name'),(b'\x00\x00\x01\x58\x00\x00\x00\x10_IO_FILE',)), - _enums = (b'\x00\x00\x01\x66\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x67\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x68\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x69\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x6A\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x6B\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x6C\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), - _typenames = (b'\x00\x00\x01\x56BOX',b'\x00\x00\x01\x57BOXA',b'\x00\x00\x01\x58FILE',b'\x00\x00\x01\x5AL_COMP_DATA',b'\x00\x00\x01\x5BPIX',b'\x00\x00\x01\x5CPIXA',b'\x00\x00\x01\x5DPIXAC',b'\x00\x00\x01\x5EPIXCMAP',b'\x00\x00\x01\x60SARRAY',b'\x00\x00\x01\x61SEL',b'\x00\x00\x00\x3Al_float32',b'\x00\x00\x01\x65l_float64',b'\x00\x00\x01\x6Fl_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x6El_int64',b'\x00\x00\x01\x70l_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x7Al_uint16',b'\x00\x00\x01\x78l_uint32',b'\x00\x00\x01\x79l_uint64',b'\x00\x00\x01\x76l_uint8'), + _types = b'\x00\x00\x01\x0D\x00\x01\x5C\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x5D\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x61\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x63\x03\x00\x00\x00\x0F\x00\x00\x01\x0D\x00\x01\x62\x03\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x04\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x09\x03\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x5E\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x01\x11\x00\x00\x01\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x18\x11\x00\x00\x05\x03\x00\x00\x11\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x01\x67\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x6A\x03\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x7C\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x09\x0D\x00\x01\x7E\x03\x00\x00\x1C\x01\x00\x00\x00\x0F\x00\x00\x11\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x65\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x65\x03\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x65\x0D\x00\x00\x11\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xA4\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x92\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x4D\x0D\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x69\x0D\x00\x00\x4D\x11\x00\x00\x00\x0F\x00\x01\x69\x0D\x00\x00\x00\x0F\x00\x00\x2A\x0D\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x1C\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x04\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x3A\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x2A\x11\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x07\x01\x00\x00\x2A\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x01\x5F\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\xD6\x11\x00\x00\xD6\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x0D\x01\x00\x00\x18\x11\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x09\x11\x00\x01\x7D\x03\x00\x00\x96\x03\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x92\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x7B\x03\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x0D\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x00\x05\x0D\x00\x01\x2C\x11\x00\x01\x17\x11\x00\x00\x09\x11\x00\x00\x07\x01\x00\x00\x07\x01\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x25\x11\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x04\x03\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\xFF\x11\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x18\x11\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x11\x03\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\xA4\x11\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x4D\x03\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x00\x92\x11\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x01\x81\x03\x00\x00\x00\x0F\x00\x01\x81\x0D\x00\x01\x53\x03\x00\x00\x00\x0F\x00\x00\x00\x09\x00\x00\x01\x09\x00\x00\x0A\x09\x00\x01\x60\x03\x00\x00\x02\x09\x00\x00\x03\x09\x00\x00\x06\x09\x00\x00\x07\x09\x00\x00\x04\x09\x00\x01\x66\x03\x00\x00\x08\x09\x00\x00\x09\x09\x00\x01\x69\x03\x00\x01\x6A\x03\x00\x00\x02\x01\x00\x00\x0E\x01\x00\x00\x00\x0B\x00\x00\x01\x0B\x00\x00\x02\x0B\x00\x00\x03\x0B\x00\x00\x04\x0B\x00\x00\x05\x0B\x00\x00\x06\x0B\x00\x00\x2A\x03\x00\x00\x0B\x01\x00\x00\x05\x01\x00\x00\x03\x01\x00\x01\x64\x03\x00\x01\x79\x03\x00\x01\x7A\x03\x00\x00\x05\x09\x00\x01\x7C\x03\x00\x00\x04\x01\x00\x01\x7E\x03\x00\x00\x08\x01\x00\x00\x0C\x01\x00\x00\x06\x01\x00\x00\x00\x01', + _globals = (b'\xFF\xFF\xFF\x0BL_BF_ANY',1,b'\xFF\xFF\xFF\x0BL_BF_CODABAR',9,b'\xFF\xFF\xFF\x0BL_BF_CODE128',2,b'\xFF\xFF\xFF\x0BL_BF_CODE2OF5',5,b'\xFF\xFF\xFF\x0BL_BF_CODE39',7,b'\xFF\xFF\xFF\x0BL_BF_CODE93',8,b'\xFF\xFF\xFF\x0BL_BF_CODEI2OF5',6,b'\xFF\xFF\xFF\x0BL_BF_EAN13',4,b'\xFF\xFF\xFF\x0BL_BF_EAN8',3,b'\xFF\xFF\xFF\x0BL_BF_UNKNOWN',0,b'\xFF\xFF\xFF\x0BL_BF_UPCA',10,b'\xFF\xFF\xFF\x0BL_CLONE',2,b'\xFF\xFF\xFF\x0BL_COPY',1,b'\xFF\xFF\xFF\x0BL_COPY_CLONE',3,b'\xFF\xFF\xFF\x0BL_DEFAULT_ENCODE',0,b'\xFF\xFF\xFF\x0BL_FLATE_ENCODE',3,b'\xFF\xFF\xFF\x0BL_G4_ENCODE',2,b'\xFF\xFF\xFF\x0BL_INSERT',0,b'\xFF\xFF\xFF\x0BL_JP2K_ENCODE',4,b'\xFF\xFF\xFF\x0BL_JPEG_ENCODE',1,b'\xFF\xFF\xFF\x0BL_NOCOPY',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_ALL',1,b'\xFF\xFF\xFF\x0BL_SEVERITY_DEBUG',2,b'\xFF\xFF\xFF\x0BL_SEVERITY_ERROR',5,b'\xFF\xFF\xFF\x0BL_SEVERITY_EXTERNAL',0,b'\xFF\xFF\xFF\x0BL_SEVERITY_INFO',3,b'\xFF\xFF\xFF\x0BL_SEVERITY_NONE',6,b'\xFF\xFF\xFF\x0BL_SEVERITY_WARNING',4,b'\xFF\xFF\xFF\x0BL_USE_WIDTHS',1,b'\xFF\xFF\xFF\x0BL_USE_WINDOWS',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_BASED_ON_SRC',4,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_BINARY',0,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_FULL_COLOR',2,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_TO_GRAYSCALE',1,b'\xFF\xFF\xFF\x0BREMOVE_CMAP_WITH_ALPHA',3,b'\xFF\xFF\xFF\x0BSEL_DONT_CARE',0,b'\xFF\xFF\xFF\x0BSEL_HIT',1,b'\xFF\xFF\xFF\x0BSEL_MISS',2,b'\x00\x00\x00\x23boxClone',0,b'\x00\x01\x3E\x23boxDestroy',0,b'\x00\x01\x41\x23boxaDestroy',0,b'\x00\x00\x03\x23boxaGetBox',0,b'\x00\x01\x19\x23getImpliedFileFormat',0,b'\x00\x00\xBE\x23getLeptonicaVersion',0,b'\x00\x01\x44\x23l_CIDataDestroy',0,b'\x00\x01\x1C\x23l_generateCIDataForPdf',0,b'\x00\x01\x59\x23leptSetStderrHandler',0,b'\x00\x01\x56\x23lept_free',0,b'\x00\x00\xC0\x23makePixelSumTab8',0,b'\x00\x00\x31\x23pixAnd',0,b'\x00\x00\x3E\x23pixBackgroundNorm',0,b'\x00\x00\x36\x23pixCleanBackgroundToWhite',0,b'\x00\x00\x22\x23pixClipRectangle',0,b'\x00\x01\x01\x23pixColorFraction',0,b'\x00\x00\x85\x23pixColorMagnitude',0,b'\x00\x00\x1F\x23pixConvertRGBToLuminance',0,b'\x00\x00\x7C\x23pixConvertTo8',0,b'\x00\x00\xD3\x23pixCorrelationBinary',0,b'\x00\x00\xED\x23pixCountPixels',0,b'\x00\x00\x98\x23pixDeserializeFromMemory',0,b'\x00\x00\x7C\x23pixDeskew',0,b'\x00\x01\x47\x23pixDestroy',0,b'\x00\x00\x4A\x23pixDilate',0,b'\x00\x00\x1F\x23pixEndianByteSwapNew',0,b'\x00\x00\xD8\x23pixEqual',0,b'\x00\x00\x4A\x23pixErode',0,b'\x00\x00\x9C\x23pixExtractBarcodes',0,b'\x00\x00\x08\x23pixFindPageForeground',0,b'\x00\x00\xE8\x23pixFindSkew',0,b'\x00\x00\x4F\x23pixGammaTRC',0,b'\x00\x00\x27\x23pixGenHalftoneMask',0,b'\x00\x00\xFA\x23pixGenerateCIData',0,b'\x00\x00\xDD\x23pixGetAverageMaskedRGB',0,b'\x00\x00\x56\x23pixGlobalNormRGB',0,b'\x00\x00\x4A\x23pixHMT',0,b'\x00\x00\x2D\x23pixInvert',0,b'\x00\x00\x15\x23pixLocateBarcodes',0,b'\x00\x00\x80\x23pixMaskOverColorPixels',0,b'\x00\x00\x5E\x23pixMaskedThreshOnBackgroundNorm',0,b'\x00\x00\xF2\x23pixNumSignificantGrayColors',0,b'\x00\x01\x0A\x23pixOtsuAdaptiveThreshold',0,b'\x00\x00\x6A\x23pixOtsuThreshOnBackgroundNorm',0,b'\x00\x00\xA0\x23pixProcessBarcodes',0,b'\x00\x00\x91\x23pixRead',0,b'\x00\x00\xA7\x23pixReadBarcodes',0,b'\x00\x00\x94\x23pixReadMem',0,b'\x00\x00\x1B\x23pixReadStream',0,b'\x00\x00\x7C\x23pixRemoveColormap',0,b'\x00\x00\x80\x23pixRemoveColormapGeneral',0,b'\x00\x00\xCD\x23pixRenderBoxa',0,b'\x00\x00\x2D\x23pixRotate180',0,b'\x00\x00\x7C\x23pixRotateOrth',0,b'\x00\x00\x77\x23pixScale',0,b'\x00\x01\x14\x23pixSerializeToMemory',0,b'\x00\x00\x31\x23pixSubtract',0,b'\x00\x01\x22\x23pixWriteImpliedFormat',0,b'\x00\x01\x31\x23pixWriteMem',0,b'\x00\x01\x37\x23pixWriteMemJpeg',0,b'\x00\x01\x2B\x23pixWriteMemPng',0,b'\x00\x00\xC2\x23pixWriteStream',0,b'\x00\x00\xC7\x23pixWriteStreamJpeg',0,b'\x00\x01\x4A\x23pixaDestroy',0,b'\x00\x00\x10\x23pixaGetBox',0,b'\x00\x00\x8C\x23pixaGetPix',0,b'\x00\x01\x4D\x23sarrayDestroy',0,b'\x00\x00\xB4\x23selCreateBrick',0,b'\x00\x00\xAE\x23selCreateFromString',0,b'\x00\x01\x50\x23selDestroy',0,b'\x00\x00\xBB\x23selPrintToString',0,b'\x00\x01\x28\x23setMsgSeverity',0), + _struct_unions = ((b'\x00\x00\x01\x5C\x00\x00\x00\x02Box',b'\x00\x00\x05\x11x',b'\x00\x00\x05\x11y',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x01\x7E\x11refcount'),(b'\x00\x00\x01\x5D\x00\x00\x00\x02Boxa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x7E\x11refcount',b'\x00\x00\x25\x11box'),(b'\x00\x00\x01\x60\x00\x00\x00\x02L_Compressed_Data',b'\x00\x00\x05\x11type',b'\x00\x01\x7B\x11datacomp',b'\x00\x00\x96\x11nbytescomp',b'\x00\x01\x69\x11data85',b'\x00\x00\x96\x11nbytes85',b'\x00\x01\x69\x11cmapdata85',b'\x00\x01\x69\x11cmapdatahex',b'\x00\x00\x05\x11ncolors',b'\x00\x00\x05\x11w',b'\x00\x00\x05\x11h',b'\x00\x00\x05\x11bps',b'\x00\x00\x05\x11spp',b'\x00\x00\x05\x11minisblack',b'\x00\x00\x05\x11predictor',b'\x00\x00\x96\x11nbytes',b'\x00\x00\x05\x11res'),(b'\x00\x00\x01\x61\x00\x00\x00\x02Pix',b'\x00\x01\x7E\x11w',b'\x00\x01\x7E\x11h',b'\x00\x01\x7E\x11d',b'\x00\x01\x7E\x11spp',b'\x00\x01\x7E\x11wpl',b'\x00\x01\x7E\x11refcount',b'\x00\x00\x05\x11xres',b'\x00\x00\x05\x11yres',b'\x00\x00\x05\x11informat',b'\x00\x00\x05\x11special',b'\x00\x01\x69\x11text',b'\x00\x01\x77\x11colormap',b'\x00\x01\x7D\x11data'),(b'\x00\x00\x01\x64\x00\x00\x00\x02PixColormap',b'\x00\x01\x57\x11array',b'\x00\x00\x05\x11depth',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n'),(b'\x00\x00\x01\x7A\x00\x00\x00\x10PixComp',),(b'\x00\x00\x01\x62\x00\x00\x00\x02Pixa',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x01\x7E\x11refcount',b'\x00\x00\x18\x11pix',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x63\x00\x00\x00\x02PixaComp',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11offset',b'\x00\x01\x78\x11pixc',b'\x00\x00\x04\x11boxa'),(b'\x00\x00\x01\x66\x00\x00\x00\x02Sarray',b'\x00\x00\x05\x11nalloc',b'\x00\x00\x05\x11n',b'\x00\x00\x05\x11refcount',b'\x00\x01\x68\x11array'),(b'\x00\x00\x01\x67\x00\x00\x00\x02Sel',b'\x00\x00\x05\x11sy',b'\x00\x00\x05\x11sx',b'\x00\x00\x05\x11cy',b'\x00\x00\x05\x11cx',b'\x00\x01\x73\x11data',b'\x00\x01\x69\x11name'),(b'\x00\x00\x01\x5E\x00\x00\x00\x10_IO_FILE',)), + _enums = (b'\x00\x00\x01\x6C\x00\x00\x00\x16$1\x00L_DEFAULT_ENCODE,L_JPEG_ENCODE,L_G4_ENCODE,L_FLATE_ENCODE,L_JP2K_ENCODE',b'\x00\x00\x01\x6D\x00\x00\x00\x16$2\x00REMOVE_CMAP_TO_BINARY,REMOVE_CMAP_TO_GRAYSCALE,REMOVE_CMAP_TO_FULL_COLOR,REMOVE_CMAP_WITH_ALPHA,REMOVE_CMAP_BASED_ON_SRC',b'\x00\x00\x01\x6E\x00\x00\x00\x16$3\x00L_NOCOPY,L_INSERT,L_COPY,L_CLONE,L_COPY_CLONE',b'\x00\x00\x01\x6F\x00\x00\x00\x16$4\x00L_USE_WIDTHS,L_USE_WINDOWS',b'\x00\x00\x01\x70\x00\x00\x00\x16$5\x00L_BF_UNKNOWN,L_BF_ANY,L_BF_CODE128,L_BF_EAN8,L_BF_EAN13,L_BF_CODE2OF5,L_BF_CODEI2OF5,L_BF_CODE39,L_BF_CODE93,L_BF_CODABAR,L_BF_UPCA',b'\x00\x00\x01\x71\x00\x00\x00\x16$6\x00L_SEVERITY_EXTERNAL,L_SEVERITY_ALL,L_SEVERITY_DEBUG,L_SEVERITY_INFO,L_SEVERITY_WARNING,L_SEVERITY_ERROR,L_SEVERITY_NONE',b'\x00\x00\x01\x72\x00\x00\x00\x16$7\x00SEL_DONT_CARE,SEL_HIT,SEL_MISS'), + _typenames = (b'\x00\x00\x01\x5CBOX',b'\x00\x00\x01\x5DBOXA',b'\x00\x00\x01\x5EFILE',b'\x00\x00\x01\x60L_COMP_DATA',b'\x00\x00\x01\x61PIX',b'\x00\x00\x01\x62PIXA',b'\x00\x00\x01\x63PIXAC',b'\x00\x00\x01\x64PIXCMAP',b'\x00\x00\x01\x66SARRAY',b'\x00\x00\x01\x67SEL',b'\x00\x00\x00\x3Al_float32',b'\x00\x00\x01\x6Bl_float64',b'\x00\x00\x01\x75l_int16',b'\x00\x00\x00\x05l_int32',b'\x00\x00\x01\x74l_int64',b'\x00\x00\x01\x76l_int8',b'\x00\x00\x00\x05l_ok',b'\x00\x00\x01\x80l_uint16',b'\x00\x00\x01\x7El_uint32',b'\x00\x00\x01\x7Fl_uint64',b'\x00\x00\x01\x7Cl_uint8'), ) diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index d4ed8226..50ca52df 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -509,6 +509,8 @@ void selDestroy ( SEL **psel ); l_int32 setMsgSeverity(l_int32 newsev); +void +leptSetStderrHandler(void (*handler)(const char *)); """ ) diff --git a/tests/test_lept.py b/tests/test_lept.py index 8f7a12d0..bf3cac83 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -16,27 +16,25 @@ # along with OCRmyPDF. If not, see . -import os from os import fspath from pickle import dumps, loads -from unittest.mock import patch import pytest from PIL import Image, ImageChops -import ocrmypdf.leptonica as lept +from ocrmypdf import leptonica as lp def test_colormap_backgroundnorm(resources): # Issue #262 - unclear how to reproduce exactly, so just ensure leptonica # can handle that case - pix = lept.Pix.open(resources / 'baiona_colormapped.png') + pix = lp.Pix.open(resources / 'baiona_colormapped.png') pix.background_norm() @pytest.fixture def crom_pix(resources): - pix = lept.Pix.open(resources / 'crom.png') + pix = lp.Pix.open(resources / 'crom.png') im = Image.open(resources / 'crom.png') yield pix, im im.close() @@ -64,17 +62,17 @@ def test_pix_otsu(crom_pix): @pytest.mark.skipif( - lept.get_leptonica_version() < 'leptonica-1.76', + lp.get_leptonica_version() < 'leptonica-1.76', reason="needs new leptonica for API change", ) def test_crop(resources): - pix = lept.Pix.open(resources / 'linn.png') + pix = lp.Pix.open(resources / 'linn.png') foreground = pix.crop_to_foreground() assert foreground.width < pix.width def test_clean_bg(resources): - pix = lept.Pix.open(resources / 'congress.jpg') + pix = lp.Pix.open(resources / 'congress.jpg') imbg = pix.clean_background_to_white() @@ -96,4 +94,12 @@ def test_leptonica_compile(tmp_path): def test_file_not_found(): with pytest.raises(FileNotFoundError): - lept.Pix.open("does_not_exist1") + lp.Pix.open("does_not_exist1") + + +def test_error_trap(): + with pytest.raises(lp.LeptonicaError, match=r"Error in pixReadMem"): + with lp._LeptonicaErrorTrap(): + lp.Pix(lp.lept.pixReadMem(lp.ffi.NULL, 0)) + with lp._LeptonicaErrorTrap_Redirect(): + lp.Pix(lp.lept.pixReadMem(lp.ffi.NULL, 0)) From 4da33b8050c1b0dc217e157864742e4b2e927345 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 Jul 2020 03:43:48 -0700 Subject: [PATCH 590/880] Update debian/copyright from Debian, with fixes --- debian/copyright | 694 ++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 691 insertions(+), 3 deletions(-) diff --git a/debian/copyright b/debian/copyright index 046a20d7..b75ada41 100644 --- a/debian/copyright +++ b/debian/copyright @@ -2,7 +2,6 @@ Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/ Upstream-Name: OCRmyPDF Upstream-Contact: James R. Barlow Source: https://github.com/jbarlow83/OCRmyPDF -Files-Excluded: tests/resources/milk.pdf Files: * Copyright: @@ -10,10 +9,24 @@ Copyright: (C) 2013-2016, 2015-2017 2016, 2017, 2017-2018, 2018 James R. Barlow License: GPL-3+ +Files: misc/watcher.py +Copyright: + (C) 2019 Ian Alexander: https://github.com/ianalexander + (C) 2020 James R. Barlow +License: GPL-3+ + +Files: misc/webservice.py +Copyright: (C) 2019 James R. Barlow +License: AGPL-3+ + Files: docs tests/resources/* Copyright: (C) 2013-2018 James R. Barlow License: CC-BY-SA-4.0 +Files: docs/images/bitmap_vs_svg.svg +Copyright: (C) 2006 Yug +License: CC-BY-SA-2.5 + Files: src/ocrmypdf/hocrtransform.py Copyright: (C) 2010 Jonathan Brinley (C) 2013-14 Julien Pfefferkorn @@ -79,15 +92,20 @@ Copyright: held by the contributors to the Wikipedia article "Optical character (epson.pdf generated from Wikipedia article as of 2016-09-14) License: CC-BY-SA-3.0 -Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf tests/resources/3small.pdf +Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf Copyright: (C) 2005 Ellywa License: GFDL-1.2+ or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0 +Comment: + Obtained from: https://commons.wikimedia.org/wiki/File:Triumph.typewriter_text_Linzensoep.gif Files: tests/resources/overlay.pdf Copyright: (C) 2017 Max Anderson License: Expat -Files: tests/resources/baiona*.png tests/resources/3small.pdf +Files: + tests/resources/baiona*.png + tests/resources/baiona*.jpg + tests/resources/link.pdf Copyright: (C) 2014 Euskaldunaa License: CC-BY-SA-4.0 @@ -95,6 +113,13 @@ Files: tests/resources/vector.pdf Copyright: (C) 2018 Catscratch License: Expat +Files: tests/resources/3small.pdf +Copyright: (C) 2014 Euskaldunaa + (C) 2017 James R. Barlow + (C) 2005 Ellywa +License: CC-BY-SA-4.0 and (GFDL-1.2+ or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0) +Comment: concatenation of baiona_gray.png, crom.png and typewriter.png/2400dpi.pdf + Files: src/ocrmypdf/data/sRGB.icc Copyright: Kai-Uwe Behrmann Marti Maria @@ -124,6 +149,669 @@ License: GPL-3+ On Debian systems, the complete text of the GNU General Public License version 3 can be found in "/usr/share/common-licenses/GPL-3". +License: AGPL-3+ + GNU AFFERO GENERAL PUBLIC LICENSE + Version 3, 19 November 2007 + . + Copyright (C) 2007 Free Software Foundation, Inc. + Everyone is permitted to copy and distribute verbatim copies + of this license document, but changing it is not allowed. + . + Preamble + . + The GNU Affero General Public License is a free, copyleft license for + software and other kinds of works, specifically designed to ensure + cooperation with the community in the case of network server software. + . + The licenses for most software and other practical works are designed + to take away your freedom to share and change the works. By contrast, + our General Public Licenses are intended to guarantee your freedom to + share and change all versions of a program--to make sure it remains free + software for all its users. + . + When we speak of free software, we are referring to freedom, not + price. Our General Public Licenses are designed to make sure that you + have the freedom to distribute copies of free software (and charge for + them if you wish), that you receive source code or can get it if you + want it, that you can change the software or use pieces of it in new + free programs, and that you know you can do these things. + . + Developers that use our General Public Licenses protect your rights + with two steps: (1) assert copyright on the software, and (2) offer + you this License which gives you legal permission to copy, distribute + and/or modify the software. + . + A secondary benefit of defending all users' freedom is that + improvements made in alternate versions of the program, if they + receive widespread use, become available for other developers to + incorporate. Many developers of free software are heartened and + encouraged by the resulting cooperation. However, in the case of + software used on network servers, this result may fail to come about. + The GNU General Public License permits making a modified version and + letting the public access it on a server without ever releasing its + source code to the public. + . + The GNU Affero General Public License is designed specifically to + ensure that, in such cases, the modified source code becomes available + to the community. It requires the operator of a network server to + provide the source code of the modified version running there to the + users of that server. Therefore, public use of a modified version, on + a publicly accessible server, gives the public access to the source + code of the modified version. + . + An older license, called the Affero General Public License and + published by Affero, was designed to accomplish similar goals. This is + a different license, not a version of the Affero GPL, but Affero has + released a new version of the Affero GPL which permits relicensing under + this license. + . + The precise terms and conditions for copying, distribution and + modification follow. + . + TERMS AND CONDITIONS + . + 0. Definitions. + . + "This License" refers to version 3 of the GNU Affero General Public License. + . + "Copyright" also means copyright-like laws that apply to other kinds of + works, such as semiconductor masks. + . + "The Program" refers to any copyrightable work licensed under this + License. Each licensee is addressed as "you". "Licensees" and + "recipients" may be individuals or organizations. + . + To "modify" a work means to copy from or adapt all or part of the work + in a fashion requiring copyright permission, other than the making of an + exact copy. The resulting work is called a "modified version" of the + earlier work or a work "based on" the earlier work. + . + A "covered work" means either the unmodified Program or a work based + on the Program. + . + To "propagate" a work means to do anything with it that, without + permission, would make you directly or secondarily liable for + infringement under applicable copyright law, except executing it on a + computer or modifying a private copy. Propagation includes copying, + distribution (with or without modification), making available to the + public, and in some countries other activities as well. + . + To "convey" a work means any kind of propagation that enables other + parties to make or receive copies. Mere interaction with a user through + a computer network, with no transfer of a copy, is not conveying. + . + An interactive user interface displays "Appropriate Legal Notices" + to the extent that it includes a convenient and prominently visible + feature that (1) displays an appropriate copyright notice, and (2) + tells the user that there is no warranty for the work (except to the + extent that warranties are provided), that licensees may convey the + work under this License, and how to view a copy of this License. If + the interface presents a list of user commands or options, such as a + menu, a prominent item in the list meets this criterion. + . + 1. Source Code. + . + The "source code" for a work means the preferred form of the work + for making modifications to it. "Object code" means any non-source + form of a work. + . + A "Standard Interface" means an interface that either is an official + standard defined by a recognized standards body, or, in the case of + interfaces specified for a particular programming language, one that + is widely used among developers working in that language. + . + The "System Libraries" of an executable work include anything, other + than the work as a whole, that (a) is included in the normal form of + packaging a Major Component, but which is not part of that Major + Component, and (b) serves only to enable use of the work with that + Major Component, or to implement a Standard Interface for which an + implementation is available to the public in source code form. A + "Major Component", in this context, means a major essential component + (kernel, window system, and so on) of the specific operating system + (if any) on which the executable work runs, or a compiler used to + produce the work, or an object code interpreter used to run it. + . + The "Corresponding Source" for a work in object code form means all + the source code needed to generate, install, and (for an executable + work) run the object code and to modify the work, including scripts to + control those activities. However, it does not include the work's + System Libraries, or general-purpose tools or generally available free + programs which are used unmodified in performing those activities but + which are not part of the work. For example, Corresponding Source + includes interface definition files associated with source files for + the work, and the source code for shared libraries and dynamically + linked subprograms that the work is specifically designed to require, + such as by intimate data communication or control flow between those + subprograms and other parts of the work. + . + The Corresponding Source need not include anything that users + can regenerate automatically from other parts of the Corresponding + Source. + . + The Corresponding Source for a work in source code form is that + same work. + . + 2. Basic Permissions. + . + All rights granted under this License are granted for the term of + copyright on the Program, and are irrevocable provided the stated + conditions are met. This License explicitly affirms your unlimited + permission to run the unmodified Program. The output from running a + covered work is covered by this License only if the output, given its + content, constitutes a covered work. This License acknowledges your + rights of fair use or other equivalent, as provided by copyright law. + . + You may make, run and propagate covered works that you do not + convey, without conditions so long as your license otherwise remains + in force. You may convey covered works to others for the sole purpose + of having them make modifications exclusively for you, or provide you + with facilities for running those works, provided that you comply with + the terms of this License in conveying all material for which you do + not control copyright. Those thus making or running the covered works + for you must do so exclusively on your behalf, under your direction + and control, on terms that prohibit them from making any copies of + your copyrighted material outside their relationship with you. + . + Conveying under any other circumstances is permitted solely under + the conditions stated below. Sublicensing is not allowed; section 10 + makes it unnecessary. + . + 3. Protecting Users' Legal Rights From Anti-Circumvention Law. + . + No covered work shall be deemed part of an effective technological + measure under any applicable law fulfilling obligations under article + 11 of the WIPO copyright treaty adopted on 20 December 1996, or + similar laws prohibiting or restricting circumvention of such + measures. + . + When you convey a covered work, you waive any legal power to forbid + circumvention of technological measures to the extent such circumvention + is effected by exercising rights under this License with respect to + the covered work, and you disclaim any intention to limit operation or + modification of the work as a means of enforcing, against the work's + users, your or third parties' legal rights to forbid circumvention of + technological measures. + . + 4. Conveying Verbatim Copies. + . + You may convey verbatim copies of the Program's source code as you + receive it, in any medium, provided that you conspicuously and + appropriately publish on each copy an appropriate copyright notice; + keep intact all notices stating that this License and any + non-permissive terms added in accord with section 7 apply to the code; + keep intact all notices of the absence of any warranty; and give all + recipients a copy of this License along with the Program. + . + You may charge any price or no price for each copy that you convey, + and you may offer support or warranty protection for a fee. + . + 5. Conveying Modified Source Versions. + . + You may convey a work based on the Program, or the modifications to + produce it from the Program, in the form of source code under the + terms of section 4, provided that you also meet all of these conditions: + . + a) The work must carry prominent notices stating that you modified + it, and giving a relevant date. + . + b) The work must carry prominent notices stating that it is + released under this License and any conditions added under section + 7. This requirement modifies the requirement in section 4 to + "keep intact all notices". + . + c) You must license the entire work, as a whole, under this + License to anyone who comes into possession of a copy. This + License will therefore apply, along with any applicable section 7 + additional terms, to the whole of the work, and all its parts, + regardless of how they are packaged. This License gives no + permission to license the work in any other way, but it does not + invalidate such permission if you have separately received it. + . + d) If the work has interactive user interfaces, each must display + Appropriate Legal Notices; however, if the Program has interactive + interfaces that do not display Appropriate Legal Notices, your + work need not make them do so. + . + A compilation of a covered work with other separate and independent + works, which are not by their nature extensions of the covered work, + and which are not combined with it such as to form a larger program, + in or on a volume of a storage or distribution medium, is called an + "aggregate" if the compilation and its resulting copyright are not + used to limit the access or legal rights of the compilation's users + beyond what the individual works permit. Inclusion of a covered work + in an aggregate does not cause this License to apply to the other + parts of the aggregate. + . + 6. Conveying Non-Source Forms. + . + You may convey a covered work in object code form under the terms + of sections 4 and 5, provided that you also convey the + machine-readable Corresponding Source under the terms of this License, + in one of these ways: + . + a) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by the + Corresponding Source fixed on a durable physical medium + customarily used for software interchange. + . + b) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by a + written offer, valid for at least three years and valid for as + long as you offer spare parts or customer support for that product + model, to give anyone who possesses the object code either (1) a + copy of the Corresponding Source for all the software in the + product that is covered by this License, on a durable physical + medium customarily used for software interchange, for a price no + more than your reasonable cost of physically performing this + conveying of source, or (2) access to copy the + Corresponding Source from a network server at no charge. + . + c) Convey individual copies of the object code with a copy of the + written offer to provide the Corresponding Source. This + alternative is allowed only occasionally and noncommercially, and + only if you received the object code with such an offer, in accord + with subsection 6b. + . + d) Convey the object code by offering access from a designated + place (gratis or for a charge), and offer equivalent access to the + Corresponding Source in the same way through the same place at no + further charge. You need not require recipients to copy the + Corresponding Source along with the object code. If the place to + copy the object code is a network server, the Corresponding Source + may be on a different server (operated by you or a third party) + that supports equivalent copying facilities, provided you maintain + clear directions next to the object code saying where to find the + Corresponding Source. Regardless of what server hosts the + Corresponding Source, you remain obligated to ensure that it is + available for as long as needed to satisfy these requirements. + . + e) Convey the object code using peer-to-peer transmission, provided + you inform other peers where the object code and Corresponding + Source of the work are being offered to the general public at no + charge under subsection 6d. + . + A separable portion of the object code, whose source code is excluded + from the Corresponding Source as a System Library, need not be + included in conveying the object code work. + . + A "User Product" is either (1) a "consumer product", which means any + tangible personal property which is normally used for personal, family, + or household purposes, or (2) anything designed or sold for incorporation + into a dwelling. In determining whether a product is a consumer product, + doubtful cases shall be resolved in favor of coverage. For a particular + product received by a particular user, "normally used" refers to a + typical or common use of that class of product, regardless of the status + of the particular user or of the way in which the particular user + actually uses, or expects or is expected to use, the product. A product + is a consumer product regardless of whether the product has substantial + commercial, industrial or non-consumer uses, unless such uses represent + the only significant mode of use of the product. + . + "Installation Information" for a User Product means any methods, + procedures, authorization keys, or other information required to install + and execute modified versions of a covered work in that User Product from + a modified version of its Corresponding Source. The information must + suffice to ensure that the continued functioning of the modified object + code is in no case prevented or interfered with solely because + modification has been made. + . + If you convey an object code work under this section in, or with, or + specifically for use in, a User Product, and the conveying occurs as + part of a transaction in which the right of possession and use of the + User Product is transferred to the recipient in perpetuity or for a + fixed term (regardless of how the transaction is characterized), the + Corresponding Source conveyed under this section must be accompanied + by the Installation Information. But this requirement does not apply + if neither you nor any third party retains the ability to install + modified object code on the User Product (for example, the work has + been installed in ROM). + . + The requirement to provide Installation Information does not include a + requirement to continue to provide support service, warranty, or updates + for a work that has been modified or installed by the recipient, or for + the User Product in which it has been modified or installed. Access to a + network may be denied when the modification itself materially and + adversely affects the operation of the network or violates the rules and + protocols for communication across the network. + . + Corresponding Source conveyed, and Installation Information provided, + in accord with this section must be in a format that is publicly + documented (and with an implementation available to the public in + source code form), and must require no special password or key for + unpacking, reading or copying. + . + 7. Additional Terms. + . + "Additional permissions" are terms that supplement the terms of this + License by making exceptions from one or more of its conditions. + Additional permissions that are applicable to the entire Program shall + be treated as though they were included in this License, to the extent + that they are valid under applicable law. If additional permissions + apply only to part of the Program, that part may be used separately + under those permissions, but the entire Program remains governed by + this License without regard to the additional permissions. + . + When you convey a copy of a covered work, you may at your option + remove any additional permissions from that copy, or from any part of + it. (Additional permissions may be written to require their own + removal in certain cases when you modify the work.) You may place + additional permissions on material, added by you to a covered work, + for which you have or can give appropriate copyright permission. + . + Notwithstanding any other provision of this License, for material you + add to a covered work, you may (if authorized by the copyright holders of + that material) supplement the terms of this License with terms: + . + a) Disclaiming warranty or limiting liability differently from the + terms of sections 15 and 16 of this License; or + . + b) Requiring preservation of specified reasonable legal notices or + author attributions in that material or in the Appropriate Legal + Notices displayed by works containing it; or + . + c) Prohibiting misrepresentation of the origin of that material, or + requiring that modified versions of such material be marked in + reasonable ways as different from the original version; or + . + d) Limiting the use for publicity purposes of names of licensors or + authors of the material; or + . + e) Declining to grant rights under trademark law for use of some + trade names, trademarks, or service marks; or + . + f) Requiring indemnification of licensors and authors of that + material by anyone who conveys the material (or modified versions of + it) with contractual assumptions of liability to the recipient, for + any liability that these contractual assumptions directly impose on + those licensors and authors. + . + All other non-permissive additional terms are considered "further + restrictions" within the meaning of section 10. If the Program as you + received it, or any part of it, contains a notice stating that it is + governed by this License along with a term that is a further + restriction, you may remove that term. If a license document contains + a further restriction but permits relicensing or conveying under this + License, you may add to a covered work material governed by the terms + of that license document, provided that the further restriction does + not survive such relicensing or conveying. + . + If you add terms to a covered work in accord with this section, you + must place, in the relevant source files, a statement of the + additional terms that apply to those files, or a notice indicating + where to find the applicable terms. + . + Additional terms, permissive or non-permissive, may be stated in the + form of a separately written license, or stated as exceptions; + the above requirements apply either way. + . + 8. Termination. + . + You may not propagate or modify a covered work except as expressly + provided under this License. Any attempt otherwise to propagate or + modify it is void, and will automatically terminate your rights under + this License (including any patent licenses granted under the third + paragraph of section 11). + . + However, if you cease all violation of this License, then your + license from a particular copyright holder is reinstated (a) + provisionally, unless and until the copyright holder explicitly and + finally terminates your license, and (b) permanently, if the copyright + holder fails to notify you of the violation by some reasonable means + prior to 60 days after the cessation. + . + Moreover, your license from a particular copyright holder is + reinstated permanently if the copyright holder notifies you of the + violation by some reasonable means, this is the first time you have + received notice of violation of this License (for any work) from that + copyright holder, and you cure the violation prior to 30 days after + your receipt of the notice. + . + Termination of your rights under this section does not terminate the + licenses of parties who have received copies or rights from you under + this License. If your rights have been terminated and not permanently + reinstated, you do not qualify to receive new licenses for the same + material under section 10. + . + 9. Acceptance Not Required for Having Copies. + . + You are not required to accept this License in order to receive or + run a copy of the Program. Ancillary propagation of a covered work + occurring solely as a consequence of using peer-to-peer transmission + to receive a copy likewise does not require acceptance. However, + nothing other than this License grants you permission to propagate or + modify any covered work. These actions infringe copyright if you do + not accept this License. Therefore, by modifying or propagating a + covered work, you indicate your acceptance of this License to do so. + . + 10. Automatic Licensing of Downstream Recipients. + . + Each time you convey a covered work, the recipient automatically + receives a license from the original licensors, to run, modify and + propagate that work, subject to this License. You are not responsible + for enforcing compliance by third parties with this License. + . + An "entity transaction" is a transaction transferring control of an + organization, or substantially all assets of one, or subdividing an + organization, or merging organizations. If propagation of a covered + work results from an entity transaction, each party to that + transaction who receives a copy of the work also receives whatever + licenses to the work the party's predecessor in interest had or could + give under the previous paragraph, plus a right to possession of the + Corresponding Source of the work from the predecessor in interest, if + the predecessor has it or can get it with reasonable efforts. + . + You may not impose any further restrictions on the exercise of the + rights granted or affirmed under this License. For example, you may + not impose a license fee, royalty, or other charge for exercise of + rights granted under this License, and you may not initiate litigation + (including a cross-claim or counterclaim in a lawsuit) alleging that + any patent claim is infringed by making, using, selling, offering for + sale, or importing the Program or any portion of it. + . + 11. Patents. + . + A "contributor" is a copyright holder who authorizes use under this + License of the Program or a work on which the Program is based. The + work thus licensed is called the contributor's "contributor version". + . + A contributor's "essential patent claims" are all patent claims + owned or controlled by the contributor, whether already acquired or + hereafter acquired, that would be infringed by some manner, permitted + by this License, of making, using, or selling its contributor version, + but do not include claims that would be infringed only as a + consequence of further modification of the contributor version. For + purposes of this definition, "control" includes the right to grant + patent sublicenses in a manner consistent with the requirements of + this License. + . + Each contributor grants you a non-exclusive, worldwide, royalty-free + patent license under the contributor's essential patent claims, to + make, use, sell, offer for sale, import and otherwise run, modify and + propagate the contents of its contributor version. + . + In the following three paragraphs, a "patent license" is any express + agreement or commitment, however denominated, not to enforce a patent + (such as an express permission to practice a patent or covenant not to + sue for patent infringement). To "grant" such a patent license to a + party means to make such an agreement or commitment not to enforce a + patent against the party. + . + If you convey a covered work, knowingly relying on a patent license, + and the Corresponding Source of the work is not available for anyone + to copy, free of charge and under the terms of this License, through a + publicly available network server or other readily accessible means, + then you must either (1) cause the Corresponding Source to be so + available, or (2) arrange to deprive yourself of the benefit of the + patent license for this particular work, or (3) arrange, in a manner + consistent with the requirements of this License, to extend the patent + license to downstream recipients. "Knowingly relying" means you have + actual knowledge that, but for the patent license, your conveying the + covered work in a country, or your recipient's use of the covered work + in a country, would infringe one or more identifiable patents in that + country that you have reason to believe are valid. + . + If, pursuant to or in connection with a single transaction or + arrangement, you convey, or propagate by procuring conveyance of, a + covered work, and grant a patent license to some of the parties + receiving the covered work authorizing them to use, propagate, modify + or convey a specific copy of the covered work, then the patent license + you grant is automatically extended to all recipients of the covered + work and works based on it. + . + A patent license is "discriminatory" if it does not include within + the scope of its coverage, prohibits the exercise of, or is + conditioned on the non-exercise of one or more of the rights that are + specifically granted under this License. You may not convey a covered + work if you are a party to an arrangement with a third party that is + in the business of distributing software, under which you make payment + to the third party based on the extent of your activity of conveying + the work, and under which the third party grants, to any of the + parties who would receive the covered work from you, a discriminatory + patent license (a) in connection with copies of the covered work + conveyed by you (or copies made from those copies), or (b) primarily + for and in connection with specific products or compilations that + contain the covered work, unless you entered into that arrangement, + or that patent license was granted, prior to 28 March 2007. + . + Nothing in this License shall be construed as excluding or limiting + any implied license or other defenses to infringement that may + otherwise be available to you under applicable patent law. + . + 12. No Surrender of Others' Freedom. + . + If conditions are imposed on you (whether by court order, agreement or + otherwise) that contradict the conditions of this License, they do not + excuse you from the conditions of this License. If you cannot convey a + covered work so as to satisfy simultaneously your obligations under this + License and any other pertinent obligations, then as a consequence you may + not convey it at all. For example, if you agree to terms that obligate you + to collect a royalty for further conveying from those to whom you convey + the Program, the only way you could satisfy both those terms and this + License would be to refrain entirely from conveying the Program. + . + 13. Remote Network Interaction; Use with the GNU General Public License. + . + Notwithstanding any other provision of this License, if you modify the + Program, your modified version must prominently offer all users + interacting with it remotely through a computer network (if your version + supports such interaction) an opportunity to receive the Corresponding + Source of your version by providing access to the Corresponding Source + from a network server at no charge, through some standard or customary + means of facilitating copying of software. This Corresponding Source + shall include the Corresponding Source for any work covered by version 3 + of the GNU General Public License that is incorporated pursuant to the + following paragraph. + . + Notwithstanding any other provision of this License, you have + permission to link or combine any covered work with a work licensed + under version 3 of the GNU General Public License into a single + combined work, and to convey the resulting work. The terms of this + License will continue to apply to the part which is the covered work, + but the work with which it is combined will remain governed by version + 3 of the GNU General Public License. + . + 14. Revised Versions of this License. + . + The Free Software Foundation may publish revised and/or new versions of + the GNU Affero General Public License from time to time. Such new versions + will be similar in spirit to the present version, but may differ in detail to + address new problems or concerns. + . + Each version is given a distinguishing version number. If the + Program specifies that a certain numbered version of the GNU Affero General + Public License "or any later version" applies to it, you have the + option of following the terms and conditions either of that numbered + version or of any later version published by the Free Software + Foundation. If the Program does not specify a version number of the + GNU Affero General Public License, you may choose any version ever published + by the Free Software Foundation. + . + If the Program specifies that a proxy can decide which future + versions of the GNU Affero General Public License can be used, that proxy's + public statement of acceptance of a version permanently authorizes you + to choose that version for the Program. + . + Later license versions may give you additional or different + permissions. However, no additional obligations are imposed on any + author or copyright holder as a result of your choosing to follow a + later version. + . + 15. Disclaimer of Warranty. + . + THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY + APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT + HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY + OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, + THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR + PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM + IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF + ALL NECESSARY SERVICING, REPAIR OR CORRECTION. + . + 16. Limitation of Liability. + . + IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING + WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS + THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY + GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE + USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF + DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD + PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), + EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF + SUCH DAMAGES. + . + 17. Interpretation of Sections 15 and 16. + . + If the disclaimer of warranty and limitation of liability provided + above cannot be given local legal effect according to their terms, + reviewing courts shall apply local law that most closely approximates + an absolute waiver of all civil liability in connection with the + Program, unless a warranty or assumption of liability accompanies a + copy of the Program in return for a fee. + . + END OF TERMS AND CONDITIONS + . + How to Apply These Terms to Your New Programs + . + If you develop a new program, and you want it to be of the greatest + possible use to the public, the best way to achieve this is to make it + free software which everyone can redistribute and change under these terms. + . + To do so, attach the following notices to the program. It is safest + to attach them to the start of each source file to most effectively + state the exclusion of warranty; and each file should have at least + the "copyright" line and a pointer to where the full notice is found. + . + + Copyright (C) + . + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU Affero General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + . + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU Affero General Public License for more details. + . + You should have received a copy of the GNU Affero General Public License + along with this program. If not, see . + . + Also add information on how to contact you by electronic and paper mail. + . + If your software can interact with users remotely through a computer + network, you should also make sure that it provides a way for users to + get its source. For example, if your program is a web application, its + interface could display a "Source" link that leads users to an archive + of the code. There are many ways you could offer source, and different + solutions will be better for different programs; see section 13 for the + specific requirements. + . + You should also get your employer (if you work as a programmer) or school, + if any, to sign a "copyright disclaimer" for the program, if necessary. + For more information on this, and how to apply and follow the GNU AGPL, see + . + License: Expat Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the From d80d963cea02f565bfe85dc4782f4d210ae47984 Mon Sep 17 00:00:00 2001 From: fcatus <56323389+fcatus@users.noreply.github.com> Date: Mon, 20 Jul 2020 04:21:58 -0500 Subject: [PATCH 591/880] pdfinfo: Replace list comp with gen expr'n --- src/ocrmypdf/pdfinfo/info.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 8f15e6b9..645c6d65 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -91,7 +91,7 @@ UNIT_SQUARE = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0) def _is_unit_square(shorthand): values = map(float, shorthand) pairwise = zip(values, UNIT_SQUARE) - return all([isclose(a, b, rel_tol=1e-3) for a, b in pairwise]) + return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise) XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_depth']) From 44149ad3191c2494664153029e96a43c4fa62250 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 19 Jul 2020 21:54:51 -0700 Subject: [PATCH 592/880] Disable test_error_trap for Leptonica < 1.79 Old error trap seems unreliable in the first place so difficult to set up a test. --- src/ocrmypdf/leptonica.py | 2 ++ tests/test_lept.py | 6 ++++-- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 20394d3d..dd3a543b 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -921,6 +921,8 @@ def get_leptonica_version(): Caveat: Leptonica expects the caller to free this memory. We don't, since that would involve binding to libc to access libc.free(), a pointless effort to reclaim 100 bytes of memory. + + Reminder that this returns "leptonica-1.xx" or "leptonica-1.yy.0". """ return ffi.string(lept.getLeptonicaVersion()).decode() diff --git a/tests/test_lept.py b/tests/test_lept.py index bf3cac83..3dcf1fa7 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -97,9 +97,11 @@ def test_file_not_found(): lp.Pix.open("does_not_exist1") +@pytest.mark.skipif( + lp.get_leptonica_version() < 'leptonica-1.79.0', + reason="test not reliable on all platforms for old leptonica", +) def test_error_trap(): with pytest.raises(lp.LeptonicaError, match=r"Error in pixReadMem"): with lp._LeptonicaErrorTrap(): lp.Pix(lp.lept.pixReadMem(lp.ffi.NULL, 0)) - with lp._LeptonicaErrorTrap_Redirect(): - lp.Pix(lp.lept.pixReadMem(lp.ffi.NULL, 0)) From 5f45f77b4efb44167dbb45875396db1ea19daa00 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 21 Jul 2020 23:53:30 -0700 Subject: [PATCH 593/880] docs: plugins update --- docs/plugins.rst | 57 ++++++++++++++++++++++++++++++++++-------------- 1 file changed, 41 insertions(+), 16 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index 3fd62240..12abf0ab 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -2,6 +2,11 @@ Plugins ======= + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL + NOT", "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and + "OPTIONAL" in this document are to be interpreted as described in + RFC 2119. + You can use plugins to customize the behavior of OCRmyPDF at certain points of interest. @@ -11,23 +16,15 @@ Currently, it is possible to: - override the decision for whether or not to perform OCR on a particular file - modify the image is about to be sent for OCR - modify the page image before it is converted to PDF +- replace the Tesseract OCR with another OCR engine that has similar behavior +- replace Ghostscript with another PDF to image converter (rasterizer) or + PDF/A generator OCRmyPDF plugins are based on the Python ``pluggy`` package and conform to its conventions. Note that: plugins installed with as setuptools entrypoints are not checked currently, because OCRmyPDF assumes you may not want to enable plugins for all files. -How plugins are imported -======================== - -Plugins are imported on demand, by the OCRmyPDF worker process that needs to use -them. As such, plugins cannot share state with other plugins, cannot rely on -their module's or the interpreter's global state, and should expect asynchronous -copies of themselves to be running. Plugins can write intermediate files to the -folder specified in ``options.work_folder``. - -Plugins should work whether executed in threads or processes. - Script plugins ============== @@ -39,7 +36,7 @@ batch of files needs a special processing step for example. ocrmypdf --plugin ocrmypdf_example_plugin.py input.pdf output.pdf -Multiple plugins may be called by issuing the ``--plugin`` argument multiple times. +Multiple plugins may be installed by issuing the ``--plugin`` argument multiple times. Packaged plugins ================ @@ -68,10 +65,39 @@ similar to ``pytest`` packages such as ``pytest-cov`` (the package) and ``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the module), just like pytest plugins. +Plugin requirements +=================== + +OCRmyPDF generally uses multiple worker processes. When a new worker is started, +Python will import all plugins again, including all plugins that were imported earlier. +This means that the global state of a plugin in one worker will not be shared with +other workers. As such, plugin hook implementations should be stateless, relying +only on their inputs. Hook implementations may use their input parameters to +to obtain a reference to shared state prepared by another hook implementation. +Plugins must expect that other instances of the plugin will be running +simultaneously. + +The ``context`` object that is passed to many hooks can be used to share information +about a file being worked on. Plugins must write private, plugin-specific data to +a subfolder named ``{options.work_folder}/ocrmypdf-plugin-name``. Plugins MAY +read and write files in ``options.work_folder``, but should be aware that their +semantics are subject to change. + +OCRmyPDF will delete ``options.work_folder`` when it has finished OCRing +a file, unless invoked with ``--keep-temporary-files``. + +The documentation for some plugin hooks contain a detailed description of the +execution context in which they will be called. + +Plugins should be prepared to work whether executed in worker threads or worker +processes. Generally, OCRmyPDF uses processes, but has a semi-hidden threaded +argument that simplifies debugging. + + Plugin hooks ============ -A plugin may provide the following hooks. Hooks should be decorated with +A plugin may provide the following hooks. Hooks must be decorated with ``ocrmypdf.hookimpl``, for example: .. code-block:: python @@ -82,13 +108,12 @@ A plugin may provide the following hooks. Hooks should be decorated with def add_options(parser): pass -The following is a complete list of hooks that may be installed and when +The following is a complete list of hooks that are available, and when they are called. .. _firstresult: -Note on firstresult hooks -^^^^^^^^^^^^^^^^^^^^^^^^^ +**Note on firstresult hooks** If multiple plugins install implementations for this hook, they will be called in the reverse of the order in which they are installed (i.e., last plugin wins). From addc2cbad0081809de8f9835fa94c63274c6d8ed Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 8 Jul 2020 00:43:13 -0700 Subject: [PATCH 594/880] Enable pikepdf mmap and set up signal handlers --- src/ocrmypdf/__main__.py | 15 ++++++++++++++- src/ocrmypdf/_concurrent.py | 13 +++++++++++++ src/ocrmypdf/_sync.py | 7 +++++++ 3 files changed, 34 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 474c9678..9f15ba31 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -18,6 +18,7 @@ import logging import os +import signal import sys from multiprocessing import set_start_method @@ -26,11 +27,20 @@ from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_closed_streams, check_options from ocrmypdf.api import Verbosity, configure_logging -from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError +from ocrmypdf.exceptions import ( + BadArgsError, + ExitCode, + InputFileError, + MissingDependencyError, +) log = logging.getLogger('ocrmypdf') +def sigbus(*args): + raise InputFileError("Lost access to the input file") + + def run(args=None): _parser, options, plugin_manager = get_parser_options_plugins(args=args) @@ -62,6 +72,9 @@ def run(args=None): log.error(e) return ExitCode.missing_dependency + if hasattr(signal, 'SIGBUS'): + signal.signal(signal.SIGBUS, sigbus) + result = run_pipeline(options=options, plugin_manager=plugin_manager) return result diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index f954d6d8..fdd9319e 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -28,6 +28,8 @@ from typing import Callable, Iterable, Optional from tqdm import tqdm +from ocrmypdf.exceptions import InputFileError + def log_listener(queue): """Listen to the worker processes and forward the messages to logging @@ -53,12 +55,20 @@ def log_listener(queue): traceback.print_exc(file=sys.stderr) +def process_sigbus(*args): + raise InputFileError("A worker process lost access to an input file") + + def process_init(queue, user_init): """Initialize a process pool worker""" # Ignore SIGINT (our parent process will kill us gracefully) signal.signal(signal.SIGINT, signal.SIG_IGN) + # Install SIGBUS handler (so our parent process can abort somewhat gracefully) + if hasattr(signal, 'SIGBUS'): + signal.signal(signal.SIGBUS, process_sigbus) + # Reconfigure the root logger for this process to send all messages to a queue h = logging.handlers.QueueHandler(queue) root = logging.getLogger() @@ -70,6 +80,9 @@ def process_init(queue, user_init): def thread_init(_queue, user_init): + # As a thread, block SIGBUS so the main thread deals with it... + if hasattr(signal, 'SIGBUS'): + signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) if user_init: user_init() diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index c11b44fa..a3c64604 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -25,6 +25,7 @@ from pathlib import Path from tempfile import mkdtemp from typing import List, NamedTuple, Optional, Tuple +import pikepdf import PIL from ocrmypdf._concurrent import exec_progress_pool @@ -331,6 +332,12 @@ def run_pipeline(options, *, plugin_manager, api=False): ): debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") + try: + if pikepdf._qpdf.set_access_default_mmap(True): + log.debug("pikepdf mmap enabled") + except AttributeError: + log.debug("pikepdf mmap not available") + try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) From a672422b0b334e29b363dc06404c7fcbcab703c0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 Jul 2020 00:20:07 -0700 Subject: [PATCH 595/880] Enable pikepdf mmap in other contexts --- src/ocrmypdf/_sync.py | 14 ++++++++------ src/ocrmypdf/helpers.py | 8 ++++++++ src/ocrmypdf/pdfinfo/info.py | 3 ++- 3 files changed, 18 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a3c64604..2932367d 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -64,7 +64,12 @@ from ocrmypdf._validation import ( report_output_file_size, ) from ocrmypdf.exceptions import ExitCode, ExitCodeException -from ocrmypdf.helpers import available_cpu_count, check_pdf, samefile +from ocrmypdf.helpers import ( + available_cpu_count, + check_pdf, + pikepdf_enable_mmap, + samefile, +) from ocrmypdf.pdfa import file_claims_pdfa log = logging.getLogger(__name__) @@ -242,6 +247,7 @@ def worker_init(max_pixels: int): # the parent process, so ensure workers get it set. Not needed when running # threaded, but harmless to set again. PIL.Image.MAX_IMAGE_PIXELS = max_pixels + pikepdf_enable_mmap() def exec_concurrent(context: PdfContext): @@ -332,11 +338,7 @@ def run_pipeline(options, *, plugin_manager, api=False): ): debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") - try: - if pikepdf._qpdf.set_access_default_mmap(True): - log.debug("pikepdf mmap enabled") - except AttributeError: - log.debug("pikepdf mmap not available") + pikepdf_enable_mmap() try: check_requested_output_file(options) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index d2523496..d8e1f4a3 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -222,6 +222,14 @@ def clamp(n: T, smallest: T, largest: T) -> T: return max(smallest, min(n, largest)) +def pikepdf_enable_mmap(): + try: + if pikepdf._qpdf.set_access_default_mmap(True): + log.debug("pikepdf mmap enabled") + except AttributeError: + log.debug("pikepdf mmap not available") + + def deprecated(func): """Warn that function is deprecated""" diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 8f15e6b9..372cc468 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -33,7 +33,7 @@ from pikepdf import PdfMatrix from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf.exceptions import EncryptedPdfError -from ocrmypdf.helpers import Resolution, available_cpu_count +from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes logger = logging.getLogger() @@ -639,6 +639,7 @@ worker_pdf = None def _pdf_pageinfo_sync_init(infile): global worker_pdf # pylint: disable=global-statement + pikepdf_enable_mmap() worker_pdf = pikepdf.open(infile) From 4ce802fdb2c16aefe6e6bd4ace95103d0f94c284 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 22 Jul 2020 00:34:27 -0700 Subject: [PATCH 596/880] v10.3.0 release notes --- docs/release_notes.rst | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 017c6364..478e5cc5 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,21 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.3.0 +======= + +- Fixed an issue where we would consider images that were already JBIG2-encoded + for optimization, potentially producing a less optimized image than the original. + We do not believe this issue would ever cause an image to loss fidelity. +- Where available, pikepdf memory mapping is now used. This improves performance. +- When Leptonica 1.79+ is installed, use its new error handling API to avoid + a "messy" redirection of stderr which was necessary to capture its error + messages. +- For older versions of Leptonica, added a new thread level lock. This fixes a + possible race condition in handling error conditions in Leptonica (although + there is no evidence it ever caused issues in practice). +- Documentation improvements and more type hinting. + v10.2.1 ======= From d6128e69375a7c3506624cb2d6ac04c8ae5b2833 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Jul 2020 21:51:25 -0700 Subject: [PATCH 597/880] Fix support for older versions of pdfminer.six (boxes_flow error) --- src/ocrmypdf/pdfinfo/layout.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 98bd82e0..7cd0b57e 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -220,8 +220,16 @@ class TextPositionTracker(PDFLayoutAnalyzer): def get_page_analysis(infile, pageno, pscript5_mode): rman = pdfminer.pdfinterp.PDFResourceManager(caching=True) + if pdfminer.__version__ < '20200402': + # Workaround for https://github.com/pdfminer/pdfminer.six/issues/395 + disable_boxes_flow = 2 + else: + disable_boxes_flow = None dev = TextPositionTracker( - rman, laparams=LAParams(all_texts=True, detect_vertical=True, boxes_flow=None) + rman, + laparams=LAParams( + all_texts=True, detect_vertical=True, boxes_flow=disable_boxes_flow + ), ) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) From 436af550507c1ab58559b4d0383982b35f681f7b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Jul 2020 21:51:49 -0700 Subject: [PATCH 598/880] Approve pdfminer.six 20200720 --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index bd95ed99..d1f4ab1e 100644 --- a/setup.py +++ b/setup.py @@ -83,7 +83,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, <= 20200517', + 'pdfminer.six >= 20191110, <= 20200720', 'pikepdf >= 1.14.0, < 2', 'Pillow >= 7.0.0', 'pluggy >= 0.13.0', From 0287d9187452367ddcb783652614fe188729bbb5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 26 Jul 2020 21:53:08 -0700 Subject: [PATCH 599/880] v10.3.1 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 478e5cc5..66cf9945 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,12 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.3.1 +======= + +- Fixed a number of test suite failures with pdfminer.six older than veresion 20200420. +- Enabled support for pdfminer.six 20200720. + v10.3.0 ======= From 7263702de9dd294a5612108422d2398eaeb7aaef Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 29 Jul 2020 16:31:47 -0700 Subject: [PATCH 600/880] Remove gs.py (spoofers entirely removed) and update copyright --- debian/copyright | 2 +- tests/spoof/gs.py | 40 ---------------------------------------- 2 files changed, 1 insertion(+), 41 deletions(-) delete mode 100644 tests/spoof/gs.py diff --git a/debian/copyright b/debian/copyright index b75ada41..d82f721e 100644 --- a/debian/copyright +++ b/debian/copyright @@ -43,7 +43,7 @@ Copyright: (C) 2014 Armin Ronacher (C) 2017 James R. Barlow License: BSD-3-clause -Files: tests/spoof/* +Files: tests/plugins/* Copyright: (C) 2016, 2017, 2016-2018 James R. Barlow License: Expat diff --git a/tests/spoof/gs.py b/tests/spoof/gs.py deleted file mode 100644 index 978f346b..00000000 --- a/tests/spoof/gs.py +++ /dev/null @@ -1,40 +0,0 @@ -# © 2019 James R. Barlow: github.com/jbarlow83 -# -# Permission is hereby granted, free of charge, to any person obtaining a -# copy of this software and associated documentation files (the -# "Software"), to deal in the Software without restriction, including -# without limitation the rights to use, copy, modify, merge, publish, -# distribute, sublicense, and/or sell copies of the Software, and to -# permit persons to whom the Software is furnished to do so, subject to -# the following conditions: -# -# The above copyright notice and this permission notice shall be included -# in all copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS -# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF -# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. -# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, -# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE -# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - -"""Find Ghostscript executable""" - - -import os -import shutil - - -def real_ghostscript(argv): - if os.name != 'nt': - gs = shutil.which('gs') - gs_args = [gs] + argv[1:] - os.execv(gs_args[0], gs_args) - else: - gs = shutil.which('gswin64c') - if not gs: - gs = shutil.which('gswin32c') - os.execv(gs, argv[1:]) - - return # Not reachable From 4cc0dc6b4a6b7bb496eb63da27cb5f6157b0ce2b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Aug 2020 16:03:29 -0700 Subject: [PATCH 601/880] Additional size increase reasons --- src/ocrmypdf/_validation.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index f64ed497..06cb6d8b 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -416,6 +416,10 @@ def report_output_file_size(options, input_file, output_file): f"The optional dependency '{name}' was not found, so some image " f"optimizations could not be attempted." ) + if options.output_type.startswith('pdfa'): + reasons.append("PDF/A conversion was enabled. (Try `--output-type pdf`.)") + if options.plugins: + reasons.append("Plugins were used.") if reasons: explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n" From a29e4952fbcb8d6944b577b215c47c02b0f0c90d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Aug 2020 16:03:54 -0700 Subject: [PATCH 602/880] Document use of mmap --- docs/api.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/api.rst b/docs/api.rst index 735354f6..0c12932e 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -51,6 +51,10 @@ Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That way your application will survive and remain interactive even if OCRmyPDF does not. +Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal +handler (except on Windows), to raise an exception if access to a memory +mapped file fails. OCRmyPDF may use memory mapping. + .. warning:: On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected From e821ca46d5ebcbea72774620315848d098227ac8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 3 Aug 2020 16:04:16 -0700 Subject: [PATCH 603/880] Approve pdfminer.six 20200726 --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index d1f4ab1e..97531c90 100644 --- a/setup.py +++ b/setup.py @@ -83,7 +83,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, <= 20200720', + 'pdfminer.six >= 20191110, <= 20200726', 'pikepdf >= 1.14.0, < 2', 'Pillow >= 7.0.0', 'pluggy >= 0.13.0', From 1d91c09963de31cf06d5ec01178197817e47eed2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 4 Aug 2020 23:53:56 -0700 Subject: [PATCH 604/880] Clarify license status of misc/completion/* files These files were contributed when the project license was GPLv3. On discussion, all known authors of these files agreed to place them under MIT license. See https://github.com/jbarlow83/OCRmyPDF/issues/600 --- debian/copyright | 11 +++++++++++ misc/completion/ocrmypdf.bash | 21 +++++++++++++++++++++ misc/completion/ocrmypdf.fish | 20 ++++++++++++++++++++ 3 files changed, 52 insertions(+) diff --git a/debian/copyright b/debian/copyright index d82f721e..11e2ae4a 100644 --- a/debian/copyright +++ b/debian/copyright @@ -9,6 +9,17 @@ Copyright: (C) 2013-2016, 2015-2017 2016, 2017, 2017-2018, 2018 James R. Barlow License: GPL-3+ +Files: misc/completion/ocrmypdf.bash +Copyright: + (C) 2019 Frank Pille + (C) 2020 Alex Willner +License: Expat + +Files: misc/completion/ocrmypdf.fish +Copyright: + (C) 2020 James R. Barlow +License: Expat + Files: misc/watcher.py Copyright: (C) 2019 Ian Alexander: https://github.com/ianalexander diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash index 59bf0cc3..b652769e 100644 --- a/misc/completion/ocrmypdf.bash +++ b/misc/completion/ocrmypdf.bash @@ -1,5 +1,26 @@ # ocrmypdf completion -*- shell-script -*- +# Copyright 2019 Frank Pille +# Copyright 2020 Alex Willner +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + set -o errexit _ocrmypdf() diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index ce9fc9e3..fd3d41a8 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -1,3 +1,23 @@ +# Copyright 2020 James R. Barlow +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + complete -c ocrmypdf -x -n '__fish_is_first_arg' -l version complete -c ocrmypdf -x -n '__fish_is_first_arg' -s h -s "?" -l help From e824cdbc4ec1736a807a1104759849d07ea46d8b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 4 Aug 2020 23:57:41 -0700 Subject: [PATCH 605/880] Change license of misc/watcher.py to MIT The authors of this file all agreed to relicense it under the MIT license. https://github.com/jbarlow83/OCRmyPDF/issues/600 --- debian/copyright | 2 +- misc/watcher.py | 25 +++++++++++++++---------- 2 files changed, 16 insertions(+), 11 deletions(-) diff --git a/debian/copyright b/debian/copyright index 11e2ae4a..78a41b24 100644 --- a/debian/copyright +++ b/debian/copyright @@ -24,7 +24,7 @@ Files: misc/watcher.py Copyright: (C) 2019 Ian Alexander: https://github.com/ianalexander (C) 2020 James R. Barlow -License: GPL-3+ +License: Expat Files: misc/webservice.py Copyright: (C) 2019 James R. Barlow diff --git a/misc/watcher.py b/misc/watcher.py index 4b5e4675..d5d583f3 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -1,18 +1,23 @@ # Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander # Copyright (C) 2020 James R Barlow: https://github.com/jbarlow83 # -# This program is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: # -# This program is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. # -# You should have received a copy of the GNU General Public License -# along with this program. If not, see . +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. import json import logging From d39778ce3a37bd6cfecfa60e145020a83a54557b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 00:12:44 -0700 Subject: [PATCH 606/880] Clarify copyright status of misc/batch.py, synology.py At the time these files were contributed there was no discussion of the license that the authors wanted to use, but the project was MIT licensed at the time. As such, these files deemed to be MIT licensed. https://github.com/jbarlow83/OCRmyPDF/issues/600 --- debian/copyright | 15 +++++++++++++++ misc/batch.py | 20 +++++++++++++++++++- misc/synology.py | 20 +++++++++++++++++++- 3 files changed, 53 insertions(+), 2 deletions(-) diff --git a/debian/copyright b/debian/copyright index 78a41b24..60fd57aa 100644 --- a/debian/copyright +++ b/debian/copyright @@ -9,6 +9,11 @@ Copyright: (C) 2013-2016, 2015-2017 2016, 2017, 2017-2018, 2018 James R. Barlow License: GPL-3+ +Files: misc/* +Copyright: + (C) 2020 James R. Barlow +License: Expat + Files: misc/completion/ocrmypdf.bash Copyright: (C) 2019 Frank Pille @@ -20,6 +25,16 @@ Copyright: (C) 2020 James R. Barlow License: Expat +Files: misc/batch.py +Copyright: + (C) 2016 findingorder: https://github.com/findingorder +License: Expat + +Files: misc/synology.py +Copyright: + (C) github.com/Enantiomerie +License: Expat + Files: misc/watcher.py Copyright: (C) 2019 Ian Alexander: https://github.com/ianalexander diff --git a/misc/batch.py b/misc/batch.py index 2a565d8a..fcd0e5cf 100644 --- a/misc/batch.py +++ b/misc/batch.py @@ -1,5 +1,23 @@ #!/usr/bin/env python3 -# Original version by DeliciousPickle@github; modified +# Copyright 2016 findingorder: https://github.com/findingorder +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. # This script must be edited to meet your needs. diff --git a/misc/synology.py b/misc/synology.py index 1ead8dec..995ca8e9 100644 --- a/misc/synology.py +++ b/misc/synology.py @@ -1,5 +1,23 @@ #!/bin/env python3 -# Contributed by github.com/Enantiomerie +# Copyright 2017 github.com/Enantiomerie +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. # This script must be edited to meet your needs. From 12c567ee10317a0865afdf8f83af37a470681f89 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 00:14:17 -0700 Subject: [PATCH 607/880] Copyright cleanup: relicense example_plugin.py The author is relicensing this file to MIT. --- misc/example_plugin.py | 25 +++++++++++++++---------- 1 file changed, 15 insertions(+), 10 deletions(-) diff --git a/misc/example_plugin.py b/misc/example_plugin.py index d6c93363..69196f57 100644 --- a/misc/example_plugin.py +++ b/misc/example_plugin.py @@ -1,17 +1,22 @@ # © 2020 James R Barlow: https://github.com/jbarlow83 # -# This program is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: # -# This program is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. # -# You should have received a copy of the GNU General Public License -# along with this program. If not, see . +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. import logging From aa0ec40102d837ead2bb8f001ff7a5a47b9d8e6d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 00:44:42 -0700 Subject: [PATCH 608/880] Change license of all GPLv3 files to MPL-2.0 https://github.com/jbarlow83/OCRmyPDF/issues/600 --- LICENSE | 1047 ++++++----------- README.md | 12 +- debian/copyright | 14 +- docs/contributing.rst | 12 +- docs/docker.rst | 3 +- docs/pdfsecurity.rst | 2 +- docs/release_notes.rst | 12 +- setup.py | 20 +- src/ocrmypdf/__init__.py | 17 +- src/ocrmypdf/__main__.py | 18 +- src/ocrmypdf/_concurrent.py | 18 +- src/ocrmypdf/_exec/__init__.py | 18 +- src/ocrmypdf/_exec/ghostscript.py | 18 +- src/ocrmypdf/_exec/jbig2enc.py | 18 +- src/ocrmypdf/_exec/pngquant.py | 18 +- src/ocrmypdf/_exec/tesseract.py | 18 +- src/ocrmypdf/_exec/unpaper.py | 18 +- src/ocrmypdf/_graft.py | 18 +- src/ocrmypdf/_jobcontext.py | 18 +- src/ocrmypdf/_logging.py | 18 +- src/ocrmypdf/_pipeline.py | 18 +- src/ocrmypdf/_plugin_manager.py | 18 +- src/ocrmypdf/_sync.py | 18 +- src/ocrmypdf/_validation.py | 17 +- src/ocrmypdf/_version.py | 17 +- src/ocrmypdf/api.py | 18 +- src/ocrmypdf/builtin_plugins/__init__.py | 17 +- src/ocrmypdf/builtin_plugins/ghostscript.py | 18 +- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 18 +- src/ocrmypdf/cli.py | 18 +- src/ocrmypdf/exceptions.py | 17 +- src/ocrmypdf/helpers.py | 18 +- src/ocrmypdf/leptonica.py | 18 +- src/ocrmypdf/lib/__init__.py | 18 +- src/ocrmypdf/lib/compile_leptonica.py | 18 +- src/ocrmypdf/optimize.py | 18 +- src/ocrmypdf/pdfa.py | 18 +- src/ocrmypdf/pdfinfo/__init__.py | 18 +- src/ocrmypdf/pdfinfo/info.py | 18 +- src/ocrmypdf/pdfinfo/layout.py | 18 +- src/ocrmypdf/pluginspec.py | 18 +- src/ocrmypdf/quality.py | 18 +- src/ocrmypdf/subprocess.py | 18 +- tests/conftest.py | 18 +- tests/test_acroform.py | 18 +- tests/test_api.py | 18 +- tests/test_check_pdf.py | 18 +- tests/test_completion.py | 18 +- tests/test_ghostscript.py | 18 +- tests/test_graft.py | 18 +- tests/test_helpers.py | 20 +- tests/test_hocrtransform.py | 18 +- tests/test_image_input.py | 18 +- tests/test_lept.py | 17 +- tests/test_main.py | 18 +- tests/test_metadata.py | 17 +- tests/test_optimize.py | 18 +- tests/test_page_numbers.py | 18 +- tests/test_pdfinfo.py | 18 +- tests/test_preprocessing.py | 18 +- tests/test_quality.py | 20 +- tests/test_rotation.py | 18 +- tests/test_stdio.py | 18 +- tests/test_tesseract.py | 18 +- tests/test_unpaper.py | 18 +- tests/test_userunit.py | 18 +- tests/test_validation.py | 20 +- 67 files changed, 646 insertions(+), 1537 deletions(-) diff --git a/LICENSE b/LICENSE index f288702d..a612ad98 100644 --- a/LICENSE +++ b/LICENSE @@ -1,674 +1,373 @@ - GNU GENERAL PUBLIC LICENSE - Version 3, 29 June 2007 - - Copyright (C) 2007 Free Software Foundation, Inc. - Everyone is permitted to copy and distribute verbatim copies - of this license document, but changing it is not allowed. - - Preamble - - The GNU General Public License is a free, copyleft license for -software and other kinds of works. - - The licenses for most software and other practical works are designed -to take away your freedom to share and change the works. By contrast, -the GNU General Public License is intended to guarantee your freedom to -share and change all versions of a program--to make sure it remains free -software for all its users. We, the Free Software Foundation, use the -GNU General Public License for most of our software; it applies also to -any other work released this way by its authors. You can apply it to -your programs, too. - - When we speak of free software, we are referring to freedom, not -price. Our General Public Licenses are designed to make sure that you -have the freedom to distribute copies of free software (and charge for -them if you wish), that you receive source code or can get it if you -want it, that you can change the software or use pieces of it in new -free programs, and that you know you can do these things. - - To protect your rights, we need to prevent others from denying you -these rights or asking you to surrender the rights. Therefore, you have -certain responsibilities if you distribute copies of the software, or if -you modify it: responsibilities to respect the freedom of others. - - For example, if you distribute copies of such a program, whether -gratis or for a fee, you must pass on to the recipients the same -freedoms that you received. You must make sure that they, too, receive -or can get the source code. And you must show them these terms so they -know their rights. - - Developers that use the GNU GPL protect your rights with two steps: -(1) assert copyright on the software, and (2) offer you this License -giving you legal permission to copy, distribute and/or modify it. - - For the developers' and authors' protection, the GPL clearly explains -that there is no warranty for this free software. For both users' and -authors' sake, the GPL requires that modified versions be marked as -changed, so that their problems will not be attributed erroneously to -authors of previous versions. - - Some devices are designed to deny users access to install or run -modified versions of the software inside them, although the manufacturer -can do so. This is fundamentally incompatible with the aim of -protecting users' freedom to change the software. The systematic -pattern of such abuse occurs in the area of products for individuals to -use, which is precisely where it is most unacceptable. Therefore, we -have designed this version of the GPL to prohibit the practice for those -products. If such problems arise substantially in other domains, we -stand ready to extend this provision to those domains in future versions -of the GPL, as needed to protect the freedom of users. - - Finally, every program is threatened constantly by software patents. -States should not allow patents to restrict development and use of -software on general-purpose computers, but in those that do, we wish to -avoid the special danger that patents applied to a free program could -make it effectively proprietary. To prevent this, the GPL assures that -patents cannot be used to render the program non-free. - - The precise terms and conditions for copying, distribution and -modification follow. - - TERMS AND CONDITIONS - - 0. Definitions. - - "This License" refers to version 3 of the GNU General Public License. - - "Copyright" also means copyright-like laws that apply to other kinds of -works, such as semiconductor masks. - - "The Program" refers to any copyrightable work licensed under this -License. Each licensee is addressed as "you". "Licensees" and -"recipients" may be individuals or organizations. - - To "modify" a work means to copy from or adapt all or part of the work -in a fashion requiring copyright permission, other than the making of an -exact copy. The resulting work is called a "modified version" of the -earlier work or a work "based on" the earlier work. - - A "covered work" means either the unmodified Program or a work based -on the Program. - - To "propagate" a work means to do anything with it that, without -permission, would make you directly or secondarily liable for -infringement under applicable copyright law, except executing it on a -computer or modifying a private copy. Propagation includes copying, -distribution (with or without modification), making available to the -public, and in some countries other activities as well. - - To "convey" a work means any kind of propagation that enables other -parties to make or receive copies. Mere interaction with a user through -a computer network, with no transfer of a copy, is not conveying. - - An interactive user interface displays "Appropriate Legal Notices" -to the extent that it includes a convenient and prominently visible -feature that (1) displays an appropriate copyright notice, and (2) -tells the user that there is no warranty for the work (except to the -extent that warranties are provided), that licensees may convey the -work under this License, and how to view a copy of this License. If -the interface presents a list of user commands or options, such as a -menu, a prominent item in the list meets this criterion. - - 1. Source Code. - - The "source code" for a work means the preferred form of the work -for making modifications to it. "Object code" means any non-source -form of a work. - - A "Standard Interface" means an interface that either is an official -standard defined by a recognized standards body, or, in the case of -interfaces specified for a particular programming language, one that -is widely used among developers working in that language. - - The "System Libraries" of an executable work include anything, other -than the work as a whole, that (a) is included in the normal form of -packaging a Major Component, but which is not part of that Major -Component, and (b) serves only to enable use of the work with that -Major Component, or to implement a Standard Interface for which an -implementation is available to the public in source code form. A -"Major Component", in this context, means a major essential component -(kernel, window system, and so on) of the specific operating system -(if any) on which the executable work runs, or a compiler used to -produce the work, or an object code interpreter used to run it. - - The "Corresponding Source" for a work in object code form means all -the source code needed to generate, install, and (for an executable -work) run the object code and to modify the work, including scripts to -control those activities. However, it does not include the work's -System Libraries, or general-purpose tools or generally available free -programs which are used unmodified in performing those activities but -which are not part of the work. For example, Corresponding Source -includes interface definition files associated with source files for -the work, and the source code for shared libraries and dynamically -linked subprograms that the work is specifically designed to require, -such as by intimate data communication or control flow between those -subprograms and other parts of the work. - - The Corresponding Source need not include anything that users -can regenerate automatically from other parts of the Corresponding -Source. - - The Corresponding Source for a work in source code form is that -same work. - - 2. Basic Permissions. - - All rights granted under this License are granted for the term of -copyright on the Program, and are irrevocable provided the stated -conditions are met. This License explicitly affirms your unlimited -permission to run the unmodified Program. The output from running a -covered work is covered by this License only if the output, given its -content, constitutes a covered work. This License acknowledges your -rights of fair use or other equivalent, as provided by copyright law. - - You may make, run and propagate covered works that you do not -convey, without conditions so long as your license otherwise remains -in force. You may convey covered works to others for the sole purpose -of having them make modifications exclusively for you, or provide you -with facilities for running those works, provided that you comply with -the terms of this License in conveying all material for which you do -not control copyright. Those thus making or running the covered works -for you must do so exclusively on your behalf, under your direction -and control, on terms that prohibit them from making any copies of -your copyrighted material outside their relationship with you. - - Conveying under any other circumstances is permitted solely under -the conditions stated below. Sublicensing is not allowed; section 10 -makes it unnecessary. - - 3. Protecting Users' Legal Rights From Anti-Circumvention Law. - - No covered work shall be deemed part of an effective technological -measure under any applicable law fulfilling obligations under article -11 of the WIPO copyright treaty adopted on 20 December 1996, or -similar laws prohibiting or restricting circumvention of such -measures. - - When you convey a covered work, you waive any legal power to forbid -circumvention of technological measures to the extent such circumvention -is effected by exercising rights under this License with respect to -the covered work, and you disclaim any intention to limit operation or -modification of the work as a means of enforcing, against the work's -users, your or third parties' legal rights to forbid circumvention of -technological measures. - - 4. Conveying Verbatim Copies. - - You may convey verbatim copies of the Program's source code as you -receive it, in any medium, provided that you conspicuously and -appropriately publish on each copy an appropriate copyright notice; -keep intact all notices stating that this License and any -non-permissive terms added in accord with section 7 apply to the code; -keep intact all notices of the absence of any warranty; and give all -recipients a copy of this License along with the Program. - - You may charge any price or no price for each copy that you convey, -and you may offer support or warranty protection for a fee. - - 5. Conveying Modified Source Versions. - - You may convey a work based on the Program, or the modifications to -produce it from the Program, in the form of source code under the -terms of section 4, provided that you also meet all of these conditions: - - a) The work must carry prominent notices stating that you modified - it, and giving a relevant date. - - b) The work must carry prominent notices stating that it is - released under this License and any conditions added under section - 7. This requirement modifies the requirement in section 4 to - "keep intact all notices". - - c) You must license the entire work, as a whole, under this - License to anyone who comes into possession of a copy. This - License will therefore apply, along with any applicable section 7 - additional terms, to the whole of the work, and all its parts, - regardless of how they are packaged. This License gives no - permission to license the work in any other way, but it does not - invalidate such permission if you have separately received it. - - d) If the work has interactive user interfaces, each must display - Appropriate Legal Notices; however, if the Program has interactive - interfaces that do not display Appropriate Legal Notices, your - work need not make them do so. - - A compilation of a covered work with other separate and independent -works, which are not by their nature extensions of the covered work, -and which are not combined with it such as to form a larger program, -in or on a volume of a storage or distribution medium, is called an -"aggregate" if the compilation and its resulting copyright are not -used to limit the access or legal rights of the compilation's users -beyond what the individual works permit. Inclusion of a covered work -in an aggregate does not cause this License to apply to the other -parts of the aggregate. - - 6. Conveying Non-Source Forms. - - You may convey a covered work in object code form under the terms -of sections 4 and 5, provided that you also convey the -machine-readable Corresponding Source under the terms of this License, -in one of these ways: - - a) Convey the object code in, or embodied in, a physical product - (including a physical distribution medium), accompanied by the - Corresponding Source fixed on a durable physical medium - customarily used for software interchange. - - b) Convey the object code in, or embodied in, a physical product - (including a physical distribution medium), accompanied by a - written offer, valid for at least three years and valid for as - long as you offer spare parts or customer support for that product - model, to give anyone who possesses the object code either (1) a - copy of the Corresponding Source for all the software in the - product that is covered by this License, on a durable physical - medium customarily used for software interchange, for a price no - more than your reasonable cost of physically performing this - conveying of source, or (2) access to copy the - Corresponding Source from a network server at no charge. - - c) Convey individual copies of the object code with a copy of the - written offer to provide the Corresponding Source. This - alternative is allowed only occasionally and noncommercially, and - only if you received the object code with such an offer, in accord - with subsection 6b. - - d) Convey the object code by offering access from a designated - place (gratis or for a charge), and offer equivalent access to the - Corresponding Source in the same way through the same place at no - further charge. You need not require recipients to copy the - Corresponding Source along with the object code. If the place to - copy the object code is a network server, the Corresponding Source - may be on a different server (operated by you or a third party) - that supports equivalent copying facilities, provided you maintain - clear directions next to the object code saying where to find the - Corresponding Source. Regardless of what server hosts the - Corresponding Source, you remain obligated to ensure that it is - available for as long as needed to satisfy these requirements. - - e) Convey the object code using peer-to-peer transmission, provided - you inform other peers where the object code and Corresponding - Source of the work are being offered to the general public at no - charge under subsection 6d. - - A separable portion of the object code, whose source code is excluded -from the Corresponding Source as a System Library, need not be -included in conveying the object code work. - - A "User Product" is either (1) a "consumer product", which means any -tangible personal property which is normally used for personal, family, -or household purposes, or (2) anything designed or sold for incorporation -into a dwelling. In determining whether a product is a consumer product, -doubtful cases shall be resolved in favor of coverage. For a particular -product received by a particular user, "normally used" refers to a -typical or common use of that class of product, regardless of the status -of the particular user or of the way in which the particular user -actually uses, or expects or is expected to use, the product. A product -is a consumer product regardless of whether the product has substantial -commercial, industrial or non-consumer uses, unless such uses represent -the only significant mode of use of the product. - - "Installation Information" for a User Product means any methods, -procedures, authorization keys, or other information required to install -and execute modified versions of a covered work in that User Product from -a modified version of its Corresponding Source. The information must -suffice to ensure that the continued functioning of the modified object -code is in no case prevented or interfered with solely because -modification has been made. - - If you convey an object code work under this section in, or with, or -specifically for use in, a User Product, and the conveying occurs as -part of a transaction in which the right of possession and use of the -User Product is transferred to the recipient in perpetuity or for a -fixed term (regardless of how the transaction is characterized), the -Corresponding Source conveyed under this section must be accompanied -by the Installation Information. But this requirement does not apply -if neither you nor any third party retains the ability to install -modified object code on the User Product (for example, the work has -been installed in ROM). - - The requirement to provide Installation Information does not include a -requirement to continue to provide support service, warranty, or updates -for a work that has been modified or installed by the recipient, or for -the User Product in which it has been modified or installed. Access to a -network may be denied when the modification itself materially and -adversely affects the operation of the network or violates the rules and -protocols for communication across the network. - - Corresponding Source conveyed, and Installation Information provided, -in accord with this section must be in a format that is publicly -documented (and with an implementation available to the public in -source code form), and must require no special password or key for -unpacking, reading or copying. - - 7. Additional Terms. - - "Additional permissions" are terms that supplement the terms of this -License by making exceptions from one or more of its conditions. -Additional permissions that are applicable to the entire Program shall -be treated as though they were included in this License, to the extent -that they are valid under applicable law. If additional permissions -apply only to part of the Program, that part may be used separately -under those permissions, but the entire Program remains governed by -this License without regard to the additional permissions. - - When you convey a copy of a covered work, you may at your option -remove any additional permissions from that copy, or from any part of -it. (Additional permissions may be written to require their own -removal in certain cases when you modify the work.) You may place -additional permissions on material, added by you to a covered work, -for which you have or can give appropriate copyright permission. - - Notwithstanding any other provision of this License, for material you -add to a covered work, you may (if authorized by the copyright holders of -that material) supplement the terms of this License with terms: - - a) Disclaiming warranty or limiting liability differently from the - terms of sections 15 and 16 of this License; or - - b) Requiring preservation of specified reasonable legal notices or - author attributions in that material or in the Appropriate Legal - Notices displayed by works containing it; or - - c) Prohibiting misrepresentation of the origin of that material, or - requiring that modified versions of such material be marked in - reasonable ways as different from the original version; or - - d) Limiting the use for publicity purposes of names of licensors or - authors of the material; or - - e) Declining to grant rights under trademark law for use of some - trade names, trademarks, or service marks; or - - f) Requiring indemnification of licensors and authors of that - material by anyone who conveys the material (or modified versions of - it) with contractual assumptions of liability to the recipient, for - any liability that these contractual assumptions directly impose on - those licensors and authors. - - All other non-permissive additional terms are considered "further -restrictions" within the meaning of section 10. If the Program as you -received it, or any part of it, contains a notice stating that it is -governed by this License along with a term that is a further -restriction, you may remove that term. If a license document contains -a further restriction but permits relicensing or conveying under this -License, you may add to a covered work material governed by the terms -of that license document, provided that the further restriction does -not survive such relicensing or conveying. - - If you add terms to a covered work in accord with this section, you -must place, in the relevant source files, a statement of the -additional terms that apply to those files, or a notice indicating -where to find the applicable terms. - - Additional terms, permissive or non-permissive, may be stated in the -form of a separately written license, or stated as exceptions; -the above requirements apply either way. - - 8. Termination. - - You may not propagate or modify a covered work except as expressly -provided under this License. Any attempt otherwise to propagate or -modify it is void, and will automatically terminate your rights under -this License (including any patent licenses granted under the third -paragraph of section 11). - - However, if you cease all violation of this License, then your -license from a particular copyright holder is reinstated (a) -provisionally, unless and until the copyright holder explicitly and -finally terminates your license, and (b) permanently, if the copyright -holder fails to notify you of the violation by some reasonable means -prior to 60 days after the cessation. - - Moreover, your license from a particular copyright holder is -reinstated permanently if the copyright holder notifies you of the -violation by some reasonable means, this is the first time you have -received notice of violation of this License (for any work) from that -copyright holder, and you cure the violation prior to 30 days after -your receipt of the notice. - - Termination of your rights under this section does not terminate the -licenses of parties who have received copies or rights from you under -this License. If your rights have been terminated and not permanently -reinstated, you do not qualify to receive new licenses for the same -material under section 10. - - 9. Acceptance Not Required for Having Copies. - - You are not required to accept this License in order to receive or -run a copy of the Program. Ancillary propagation of a covered work -occurring solely as a consequence of using peer-to-peer transmission -to receive a copy likewise does not require acceptance. However, -nothing other than this License grants you permission to propagate or -modify any covered work. These actions infringe copyright if you do -not accept this License. Therefore, by modifying or propagating a -covered work, you indicate your acceptance of this License to do so. - - 10. Automatic Licensing of Downstream Recipients. - - Each time you convey a covered work, the recipient automatically -receives a license from the original licensors, to run, modify and -propagate that work, subject to this License. You are not responsible -for enforcing compliance by third parties with this License. - - An "entity transaction" is a transaction transferring control of an -organization, or substantially all assets of one, or subdividing an -organization, or merging organizations. If propagation of a covered -work results from an entity transaction, each party to that -transaction who receives a copy of the work also receives whatever -licenses to the work the party's predecessor in interest had or could -give under the previous paragraph, plus a right to possession of the -Corresponding Source of the work from the predecessor in interest, if -the predecessor has it or can get it with reasonable efforts. - - You may not impose any further restrictions on the exercise of the -rights granted or affirmed under this License. For example, you may -not impose a license fee, royalty, or other charge for exercise of -rights granted under this License, and you may not initiate litigation -(including a cross-claim or counterclaim in a lawsuit) alleging that -any patent claim is infringed by making, using, selling, offering for -sale, or importing the Program or any portion of it. - - 11. Patents. - - A "contributor" is a copyright holder who authorizes use under this -License of the Program or a work on which the Program is based. The -work thus licensed is called the contributor's "contributor version". - - A contributor's "essential patent claims" are all patent claims -owned or controlled by the contributor, whether already acquired or -hereafter acquired, that would be infringed by some manner, permitted -by this License, of making, using, or selling its contributor version, -but do not include claims that would be infringed only as a -consequence of further modification of the contributor version. For -purposes of this definition, "control" includes the right to grant -patent sublicenses in a manner consistent with the requirements of -this License. - - Each contributor grants you a non-exclusive, worldwide, royalty-free -patent license under the contributor's essential patent claims, to -make, use, sell, offer for sale, import and otherwise run, modify and -propagate the contents of its contributor version. - - In the following three paragraphs, a "patent license" is any express -agreement or commitment, however denominated, not to enforce a patent -(such as an express permission to practice a patent or covenant not to -sue for patent infringement). To "grant" such a patent license to a -party means to make such an agreement or commitment not to enforce a -patent against the party. - - If you convey a covered work, knowingly relying on a patent license, -and the Corresponding Source of the work is not available for anyone -to copy, free of charge and under the terms of this License, through a -publicly available network server or other readily accessible means, -then you must either (1) cause the Corresponding Source to be so -available, or (2) arrange to deprive yourself of the benefit of the -patent license for this particular work, or (3) arrange, in a manner -consistent with the requirements of this License, to extend the patent -license to downstream recipients. "Knowingly relying" means you have -actual knowledge that, but for the patent license, your conveying the -covered work in a country, or your recipient's use of the covered work -in a country, would infringe one or more identifiable patents in that -country that you have reason to believe are valid. - - If, pursuant to or in connection with a single transaction or -arrangement, you convey, or propagate by procuring conveyance of, a -covered work, and grant a patent license to some of the parties -receiving the covered work authorizing them to use, propagate, modify -or convey a specific copy of the covered work, then the patent license -you grant is automatically extended to all recipients of the covered -work and works based on it. - - A patent license is "discriminatory" if it does not include within -the scope of its coverage, prohibits the exercise of, or is -conditioned on the non-exercise of one or more of the rights that are -specifically granted under this License. You may not convey a covered -work if you are a party to an arrangement with a third party that is -in the business of distributing software, under which you make payment -to the third party based on the extent of your activity of conveying -the work, and under which the third party grants, to any of the -parties who would receive the covered work from you, a discriminatory -patent license (a) in connection with copies of the covered work -conveyed by you (or copies made from those copies), or (b) primarily -for and in connection with specific products or compilations that -contain the covered work, unless you entered into that arrangement, -or that patent license was granted, prior to 28 March 2007. - - Nothing in this License shall be construed as excluding or limiting -any implied license or other defenses to infringement that may -otherwise be available to you under applicable patent law. - - 12. No Surrender of Others' Freedom. - - If conditions are imposed on you (whether by court order, agreement or -otherwise) that contradict the conditions of this License, they do not -excuse you from the conditions of this License. If you cannot convey a -covered work so as to satisfy simultaneously your obligations under this -License and any other pertinent obligations, then as a consequence you may -not convey it at all. For example, if you agree to terms that obligate you -to collect a royalty for further conveying from those to whom you convey -the Program, the only way you could satisfy both those terms and this -License would be to refrain entirely from conveying the Program. - - 13. Use with the GNU Affero General Public License. - - Notwithstanding any other provision of this License, you have -permission to link or combine any covered work with a work licensed -under version 3 of the GNU Affero General Public License into a single -combined work, and to convey the resulting work. The terms of this -License will continue to apply to the part which is the covered work, -but the special requirements of the GNU Affero General Public License, -section 13, concerning interaction through a network will apply to the -combination as such. - - 14. Revised Versions of this License. - - The Free Software Foundation may publish revised and/or new versions of -the GNU General Public License from time to time. Such new versions will -be similar in spirit to the present version, but may differ in detail to -address new problems or concerns. - - Each version is given a distinguishing version number. If the -Program specifies that a certain numbered version of the GNU General -Public License "or any later version" applies to it, you have the -option of following the terms and conditions either of that numbered -version or of any later version published by the Free Software -Foundation. If the Program does not specify a version number of the -GNU General Public License, you may choose any version ever published -by the Free Software Foundation. - - If the Program specifies that a proxy can decide which future -versions of the GNU General Public License can be used, that proxy's -public statement of acceptance of a version permanently authorizes you -to choose that version for the Program. - - Later license versions may give you additional or different -permissions. However, no additional obligations are imposed on any -author or copyright holder as a result of your choosing to follow a -later version. - - 15. Disclaimer of Warranty. - - THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY -APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT -HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY -OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, -THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR -PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM -IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF -ALL NECESSARY SERVICING, REPAIR OR CORRECTION. - - 16. Limitation of Liability. - - IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING -WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS -THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY -GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE -USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF -DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD -PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), -EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF -SUCH DAMAGES. - - 17. Interpretation of Sections 15 and 16. - - If the disclaimer of warranty and limitation of liability provided -above cannot be given local legal effect according to their terms, -reviewing courts shall apply local law that most closely approximates -an absolute waiver of all civil liability in connection with the -Program, unless a warranty or assumption of liability accompanies a -copy of the Program in return for a fee. - - END OF TERMS AND CONDITIONS - - How to Apply These Terms to Your New Programs - - If you develop a new program, and you want it to be of the greatest -possible use to the public, the best way to achieve this is to make it -free software which everyone can redistribute and change under these terms. - - To do so, attach the following notices to the program. It is safest -to attach them to the start of each source file to most effectively -state the exclusion of warranty; and each file should have at least -the "copyright" line and a pointer to where the full notice is found. - - - Copyright (C) - - This program is free software: you can redistribute it and/or modify - it under the terms of the GNU General Public License as published by - the Free Software Foundation, either version 3 of the License, or - (at your option) any later version. - - This program is distributed in the hope that it will be useful, - but WITHOUT ANY WARRANTY; without even the implied warranty of - MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the - GNU General Public License for more details. - - You should have received a copy of the GNU General Public License - along with this program. If not, see . - -Also add information on how to contact you by electronic and paper mail. - - If the program does terminal interaction, make it output a short -notice like this when it starts in an interactive mode: - - Copyright (C) - This program comes with ABSOLUTELY NO WARRANTY; for details type `show w'. - This is free software, and you are welcome to redistribute it - under certain conditions; type `show c' for details. - -The hypothetical commands `show w' and `show c' should show the appropriate -parts of the General Public License. Of course, your program's commands -might be different; for a GUI interface, you would use an "about box". - - You should also get your employer (if you work as a programmer) or school, -if any, to sign a "copyright disclaimer" for the program, if necessary. -For more information on this, and how to apply and follow the GNU GPL, see -. - - The GNU General Public License does not permit incorporating your program -into proprietary programs. If your program is a subroutine library, you -may consider it more useful to permit linking proprietary applications with -the library. If this is what you want to do, use the GNU Lesser General -Public License instead of this License. But first, please read -. +Mozilla Public License Version 2.0 +================================== + +1. Definitions +-------------- + +1.1. "Contributor" + means each individual or legal entity that creates, contributes to + the creation of, or owns Covered Software. + +1.2. "Contributor Version" + means the combination of the Contributions of others (if any) used + by a Contributor and that particular Contributor's Contribution. + +1.3. "Contribution" + means Covered Software of a particular Contributor. + +1.4. "Covered Software" + means Source Code Form to which the initial Contributor has attached + the notice in Exhibit A, the Executable Form of such Source Code + Form, and Modifications of such Source Code Form, in each case + including portions thereof. + +1.5. "Incompatible With Secondary Licenses" + means + + (a) that the initial Contributor has attached the notice described + in Exhibit B to the Covered Software; or + + (b) that the Covered Software was made available under the terms of + version 1.1 or earlier of the License, but not also under the + terms of a Secondary License. + +1.6. "Executable Form" + means any form of the work other than Source Code Form. + +1.7. "Larger Work" + means a work that combines Covered Software with other material, in + a separate file or files, that is not Covered Software. + +1.8. "License" + means this document. + +1.9. "Licensable" + means having the right to grant, to the maximum extent possible, + whether at the time of the initial grant or subsequently, any and + all of the rights conveyed by this License. + +1.10. "Modifications" + means any of the following: + + (a) any file in Source Code Form that results from an addition to, + deletion from, or modification of the contents of Covered + Software; or + + (b) any new file in Source Code Form that contains any Covered + Software. + +1.11. "Patent Claims" of a Contributor + means any patent claim(s), including without limitation, method, + process, and apparatus claims, in any patent Licensable by such + Contributor that would be infringed, but for the grant of the + License, by the making, using, selling, offering for sale, having + made, import, or transfer of either its Contributions or its + Contributor Version. + +1.12. "Secondary License" + means either the GNU General Public License, Version 2.0, the GNU + Lesser General Public License, Version 2.1, the GNU Affero General + Public License, Version 3.0, or any later versions of those + licenses. + +1.13. "Source Code Form" + means the form of the work preferred for making modifications. + +1.14. "You" (or "Your") + means an individual or a legal entity exercising rights under this + License. For legal entities, "You" includes any entity that + controls, is controlled by, or is under common control with You. For + purposes of this definition, "control" means (a) the power, direct + or indirect, to cause the direction or management of such entity, + whether by contract or otherwise, or (b) ownership of more than + fifty percent (50%) of the outstanding shares or beneficial + ownership of such entity. + +2. License Grants and Conditions +-------------------------------- + +2.1. Grants + +Each Contributor hereby grants You a world-wide, royalty-free, +non-exclusive license: + +(a) under intellectual property rights (other than patent or trademark) + Licensable by such Contributor to use, reproduce, make available, + modify, display, perform, distribute, and otherwise exploit its + Contributions, either on an unmodified basis, with Modifications, or + as part of a Larger Work; and + +(b) under Patent Claims of such Contributor to make, use, sell, offer + for sale, have made, import, and otherwise transfer either its + Contributions or its Contributor Version. + +2.2. Effective Date + +The licenses granted in Section 2.1 with respect to any Contribution +become effective for each Contribution on the date the Contributor first +distributes such Contribution. + +2.3. Limitations on Grant Scope + +The licenses granted in this Section 2 are the only rights granted under +this License. No additional rights or licenses will be implied from the +distribution or licensing of Covered Software under this License. +Notwithstanding Section 2.1(b) above, no patent license is granted by a +Contributor: + +(a) for any code that a Contributor has removed from Covered Software; + or + +(b) for infringements caused by: (i) Your and any other third party's + modifications of Covered Software, or (ii) the combination of its + Contributions with other software (except as part of its Contributor + Version); or + +(c) under Patent Claims infringed by Covered Software in the absence of + its Contributions. + +This License does not grant any rights in the trademarks, service marks, +or logos of any Contributor (except as may be necessary to comply with +the notice requirements in Section 3.4). + +2.4. Subsequent Licenses + +No Contributor makes additional grants as a result of Your choice to +distribute the Covered Software under a subsequent version of this +License (see Section 10.2) or under the terms of a Secondary License (if +permitted under the terms of Section 3.3). + +2.5. Representation + +Each Contributor represents that the Contributor believes its +Contributions are its original creation(s) or it has sufficient rights +to grant the rights to its Contributions conveyed by this License. + +2.6. Fair Use + +This License is not intended to limit any rights You have under +applicable copyright doctrines of fair use, fair dealing, or other +equivalents. + +2.7. Conditions + +Sections 3.1, 3.2, 3.3, and 3.4 are conditions of the licenses granted +in Section 2.1. + +3. Responsibilities +------------------- + +3.1. Distribution of Source Form + +All distribution of Covered Software in Source Code Form, including any +Modifications that You create or to which You contribute, must be under +the terms of this License. You must inform recipients that the Source +Code Form of the Covered Software is governed by the terms of this +License, and how they can obtain a copy of this License. You may not +attempt to alter or restrict the recipients' rights in the Source Code +Form. + +3.2. Distribution of Executable Form + +If You distribute Covered Software in Executable Form then: + +(a) such Covered Software must also be made available in Source Code + Form, as described in Section 3.1, and You must inform recipients of + the Executable Form how they can obtain a copy of such Source Code + Form by reasonable means in a timely manner, at a charge no more + than the cost of distribution to the recipient; and + +(b) You may distribute such Executable Form under the terms of this + License, or sublicense it under different terms, provided that the + license for the Executable Form does not attempt to limit or alter + the recipients' rights in the Source Code Form under this License. + +3.3. Distribution of a Larger Work + +You may create and distribute a Larger Work under terms of Your choice, +provided that You also comply with the requirements of this License for +the Covered Software. If the Larger Work is a combination of Covered +Software with a work governed by one or more Secondary Licenses, and the +Covered Software is not Incompatible With Secondary Licenses, this +License permits You to additionally distribute such Covered Software +under the terms of such Secondary License(s), so that the recipient of +the Larger Work may, at their option, further distribute the Covered +Software under the terms of either this License or such Secondary +License(s). + +3.4. Notices + +You may not remove or alter the substance of any license notices +(including copyright notices, patent notices, disclaimers of warranty, +or limitations of liability) contained within the Source Code Form of +the Covered Software, except that You may alter any license notices to +the extent required to remedy known factual inaccuracies. + +3.5. Application of Additional Terms + +You may choose to offer, and to charge a fee for, warranty, support, +indemnity or liability obligations to one or more recipients of Covered +Software. However, You may do so only on Your own behalf, and not on +behalf of any Contributor. You must make it absolutely clear that any +such warranty, support, indemnity, or liability obligation is offered by +You alone, and You hereby agree to indemnify every Contributor for any +liability incurred by such Contributor as a result of warranty, support, +indemnity or liability terms You offer. You may include additional +disclaimers of warranty and limitations of liability specific to any +jurisdiction. + +4. Inability to Comply Due to Statute or Regulation +--------------------------------------------------- + +If it is impossible for You to comply with any of the terms of this +License with respect to some or all of the Covered Software due to +statute, judicial order, or regulation then You must: (a) comply with +the terms of this License to the maximum extent possible; and (b) +describe the limitations and the code they affect. Such description must +be placed in a text file included with all distributions of the Covered +Software under this License. Except to the extent prohibited by statute +or regulation, such description must be sufficiently detailed for a +recipient of ordinary skill to be able to understand it. + +5. Termination +-------------- + +5.1. The rights granted under this License will terminate automatically +if You fail to comply with any of its terms. However, if You become +compliant, then the rights granted under this License from a particular +Contributor are reinstated (a) provisionally, unless and until such +Contributor explicitly and finally terminates Your grants, and (b) on an +ongoing basis, if such Contributor fails to notify You of the +non-compliance by some reasonable means prior to 60 days after You have +come back into compliance. Moreover, Your grants from a particular +Contributor are reinstated on an ongoing basis if such Contributor +notifies You of the non-compliance by some reasonable means, this is the +first time You have received notice of non-compliance with this License +from such Contributor, and You become compliant prior to 30 days after +Your receipt of the notice. + +5.2. If You initiate litigation against any entity by asserting a patent +infringement claim (excluding declaratory judgment actions, +counter-claims, and cross-claims) alleging that a Contributor Version +directly or indirectly infringes any patent, then the rights granted to +You by any and all Contributors for the Covered Software under Section +2.1 of this License shall terminate. + +5.3. In the event of termination under Sections 5.1 or 5.2 above, all +end user license agreements (excluding distributors and resellers) which +have been validly granted by You or Your distributors under this License +prior to termination shall survive termination. + +************************************************************************ +* * +* 6. Disclaimer of Warranty * +* ------------------------- * +* * +* Covered Software is provided under this License on an "as is" * +* basis, without warranty of any kind, either expressed, implied, or * +* statutory, including, without limitation, warranties that the * +* Covered Software is free of defects, merchantable, fit for a * +* particular purpose or non-infringing. The entire risk as to the * +* quality and performance of the Covered Software is with You. * +* Should any Covered Software prove defective in any respect, You * +* (not any Contributor) assume the cost of any necessary servicing, * +* repair, or correction. This disclaimer of warranty constitutes an * +* essential part of this License. No use of any Covered Software is * +* authorized under this License except under this disclaimer. * +* * +************************************************************************ + +************************************************************************ +* * +* 7. Limitation of Liability * +* -------------------------- * +* * +* Under no circumstances and under no legal theory, whether tort * +* (including negligence), contract, or otherwise, shall any * +* Contributor, or anyone who distributes Covered Software as * +* permitted above, be liable to You for any direct, indirect, * +* special, incidental, or consequential damages of any character * +* including, without limitation, damages for lost profits, loss of * +* goodwill, work stoppage, computer failure or malfunction, or any * +* and all other commercial damages or losses, even if such party * +* shall have been informed of the possibility of such damages. This * +* limitation of liability shall not apply to liability for death or * +* personal injury resulting from such party's negligence to the * +* extent applicable law prohibits such limitation. Some * +* jurisdictions do not allow the exclusion or limitation of * +* incidental or consequential damages, so this exclusion and * +* limitation may not apply to You. * +* * +************************************************************************ + +8. Litigation +------------- + +Any litigation relating to this License may be brought only in the +courts of a jurisdiction where the defendant maintains its principal +place of business and such litigation shall be governed by laws of that +jurisdiction, without reference to its conflict-of-law provisions. +Nothing in this Section shall prevent a party's ability to bring +cross-claims or counter-claims. + +9. Miscellaneous +---------------- + +This License represents the complete agreement concerning the subject +matter hereof. If any provision of this License is held to be +unenforceable, such provision shall be reformed only to the extent +necessary to make it enforceable. Any law or regulation which provides +that the language of a contract shall be construed against the drafter +shall not be used to construe this License against a Contributor. + +10. Versions of the License +--------------------------- + +10.1. New Versions + +Mozilla Foundation is the license steward. Except as provided in Section +10.3, no one other than the license steward has the right to modify or +publish new versions of this License. Each version will be given a +distinguishing version number. + +10.2. Effect of New Versions + +You may distribute the Covered Software under the terms of the version +of the License under which You originally received the Covered Software, +or under the terms of any subsequent version published by the license +steward. + +10.3. Modified Versions + +If you create software not governed by this License, and you want to +create a new license for such software, you may create and use a +modified version of this License if you rename the license and remove +any references to the name of the license steward (except to note that +such modified license differs from this License). + +10.4. Distributing Source Code Form that is Incompatible With Secondary +Licenses + +If You choose to distribute Source Code Form that is Incompatible With +Secondary Licenses under the terms of this version of the License, the +notice described in Exhibit B of this License must be attached. + +Exhibit A - Source Code Form License Notice +------------------------------------------- + + This Source Code Form is subject to the terms of the Mozilla Public + License, v. 2.0. If a copy of the MPL was not distributed with this + file, You can obtain one at http://mozilla.org/MPL/2.0/. + +If it is not possible or desirable to put the notice in a particular +file, then You may include the notice in a location (such as a LICENSE +file in a relevant directory) where a recipient would be likely to look +for such a notice. + +You may add additional accurate notices of copyright ownership. + +Exhibit B - "Incompatible With Secondary Licenses" Notice +--------------------------------------------------------- + + This Source Code Form is "Incompatible With Secondary Licenses", as + defined by the Mozilla Public License, v. 2.0. diff --git a/README.md b/README.md index bb26967c..cb91148c 100644 --- a/README.md +++ b/README.md @@ -127,11 +127,15 @@ OCRmyPDF would not be the software that it is today without companies and users ## License -The OCRmyPDF software is licensed under the GNU GPLv3. Certain files are covered by other licenses, as noted in their source files. +The OCRmyPDF software is licensed under the Mozilla Public License 2.0 +(MPL-2.0). This license permits integration of OCRmyPDF with other code, +included commercial and closed source, but asks you to publish source-level +modifications you make to OCRmyPDF. -The license for each test file varies, and is noted in tests/resources/README.rst. The documentation is licensed under Creative Commons Attribution-ShareAlike 4.0 (CC-BY-SA 4.0). - -OCRmyPDF versions prior to 6.0 were distributed under the MIT License. +Some components of OCRmyPDF have other licenses, as noted in those files and the +``debian/copyright`` file. Most files in ``misc/`` use the MIT license, and the +documentation and test files are generally licensed under Creative Commons +ShareAlike 4.0 (CC-BY-SA 4.0). ## Disclaimer diff --git a/debian/copyright b/debian/copyright index 60fd57aa..4c0cf4f7 100644 --- a/debian/copyright +++ b/debian/copyright @@ -5,9 +5,10 @@ Source: https://github.com/jbarlow83/OCRmyPDF Files: * Copyright: - (C) 2013-2017 The OCRmyPDF Authors - (C) 2013-2016, 2015-2017 2016, 2017, 2017-2018, 2018 James R. Barlow -License: GPL-3+ + (C) 2013-2015 Julien Pfefferkorn + (C) 2015-2020 James R. Barlow + (C) 2019 Martin Wind +License: MPL-2.0 Files: misc/* Copyright: @@ -158,6 +159,13 @@ Files: debian/* Copyright: (C) 2016 Sean Whitton License: GPL-3+ +License: MPL-2.0 + This Source Code Form is subject to the terms of the Mozilla Public + License, v. 2.0. + . + On Debian systems the full text of the MPL-2.0 can be found in + /usr/share/common-licenses/MPL-2.0. + License: GPL-3+ This program is free software; you can redistribute it and/or modify it under the terms of the GNU General Public License as published by diff --git a/docs/contributing.rst b/docs/contributing.rst index d240bc8b..6c928933 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -32,7 +32,8 @@ If you are proposing a change that will require a new Python dependency, we prefer dependencies that are already packaged by Debian or Red Hat. This makes life much easier for our downstream package maintainers. -Python dependencies must also be GPLv3 compatible. +Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are likely +incompatible with the project's license, but LGPLv3 is compatible. New non-Python dependencies =========================== @@ -55,3 +56,12 @@ of that platform. Packager maintainers, please ensure that the command line completion scripts in ``misc/`` are installed. + +Copyright and license +===================== + +For contributions over 10 lines of code, please include your name to list of +copyright holders for that file. The core program is licensed under MPL-2.0, +test files and documentation under CC-BY-SA 4.0, and miscellaneous files under +MIT. Please contribute code only that you wrote and you have the permission to +contribute or license to us. diff --git a/docs/docker.rst b/docs/docker.rst index 95ad668c..c38a63b9 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -163,8 +163,7 @@ complete. This may entail setting a long timeout; this interface is more useful for internal HTTP API calls. Unlike the rest of OCRmyPDF, this web service is licensed under the -Affero GPLv3 (AGPLv3) since Ghostscript, a dependency of OCRmyPDF, is -also licensed in this way. +Affero GPLv3 (AGPLv3) since Ghostscript is also licensed in this way. In addition to the above, please read our :ref:`general remarks on using OCRmyPDF as a service `. diff --git a/docs/pdfsecurity.rst b/docs/pdfsecurity.rst index 36246960..04ad4e90 100644 --- a/docs/pdfsecurity.rst +++ b/docs/pdfsecurity.rst @@ -64,7 +64,7 @@ malicious user could upload a chosen PDF. In particular, it is not necessarily secure against PDF malware or PDFs that cause denial of service. OCRmyPDF relies on Ghostscript, and therefore, if deployed online one should be prepared to comply with Ghostscript's Affero GPL -license, OCRmyPDF's GPL license, and any other licenses. +license, and any other licenses. Setting aside these concerns, a side effect of OCRmyPDF is it may incidentally sanitize PDFs that contain certain types of malware. It diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 66cf9945..bcbfed2e 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,14 +12,10 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. -Note that it is licensed under GPLv3, so scripts that -``import ocrmypdf`` and are released publicly should probably also be -licensed under GPLv3. - v10.3.1 ======= -- Fixed a number of test suite failures with pdfminer.six older than veresion 20200420. +- Fixed a number of test suite failures with pdfminer.six older than veresion 20200402. - Enabled support for pdfminer.six 20200720. v10.3.0 @@ -362,7 +358,7 @@ v9.0.0 - Added a high level API for applications that want to integrate OCRmyPDF. Special thanks to Martin Wind (@mawi1988) whose made significant contributions - to this effort. OCRmyPDF is GPLv3-licensed. + to this effort. - Added progress bars for long-running steps. ■■■■■■■□□ - We now create linearized ("fast web view") PDFs by default. The new parameter ``--fast-web-view`` provides control over when this feature is applied. @@ -989,8 +985,8 @@ v6.1.0 v6.0.0 ====== -- The software license has been changed to GPLv3. Test resource files - and some individual sources may have other licenses. +- The software license has been changed to GPLv3 [it has since changed again]. + Test resource files and some individual sources may have other licenses. - OCRmyPDF now depends on `PyMuPDF `__. Including PyMuPDF is the primary reason for the change to GPLv3. diff --git a/setup.py b/setup.py index 97531c90..93990648 100644 --- a/setup.py +++ b/setup.py @@ -2,20 +2,10 @@ # -*- coding: utf-8 -*- # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from __future__ import print_function, unicode_literals @@ -60,7 +50,7 @@ setup( "Intended Audience :: End Users/Desktop", "Intended Audience :: Science/Research", "Intended Audience :: System Administrators", - "License :: OSI Approved :: GNU General Public License v3 (GPLv3)", + "License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)", "Operating System :: MacOS :: MacOS X", "Operating System :: Microsoft :: Windows :: Windows 10", "Operating System :: POSIX", diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index d64253b8..c7852fb6 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -1,19 +1,8 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. from pluggy import HookimplMarker as _HookimplMarker diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 9f15ba31..3c897b72 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -1,20 +1,10 @@ #!/usr/bin/env python3 # © 2015-19 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import os diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index fdd9319e..bd02f65b 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import logging.handlers diff --git a/src/ocrmypdf/_exec/__init__.py b/src/ocrmypdf/_exec/__init__.py index 8c6d0bb3..36dc7182 100644 --- a/src/ocrmypdf/_exec/__init__.py +++ b/src/ocrmypdf/_exec/__init__.py @@ -1,18 +1,8 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Manage third party executables""" diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 44347a42..0358dc78 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -1,19 +1,9 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Interface to Ghostscript executable""" diff --git a/src/ocrmypdf/_exec/jbig2enc.py b/src/ocrmypdf/_exec/jbig2enc.py index deced89a..9311ec3f 100644 --- a/src/ocrmypdf/_exec/jbig2enc.py +++ b/src/ocrmypdf/_exec/jbig2enc.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Interface to jbig2 executable""" diff --git a/src/ocrmypdf/_exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py index cb155aeb..b071eda1 100644 --- a/src/ocrmypdf/_exec/pngquant.py +++ b/src/ocrmypdf/_exec/pngquant.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Interface to pngquant executable""" diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 85dc5040..3be2d36e 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -1,19 +1,9 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Interface to Tesseract executable""" diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index b3154ef6..6b4383d1 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -1,19 +1,9 @@ # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + # unpaper documentation: # https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 14c09af1..85caeaba 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging from contextlib import suppress diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index ae7b5f52..f0767814 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import os import shutil diff --git a/src/ocrmypdf/_logging.py b/src/ocrmypdf/_logging.py index af28742d..e33616a4 100644 --- a/src/ocrmypdf/_logging.py +++ b/src/ocrmypdf/_logging.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import sys diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 36105fb7..2ebd9fea 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -1,19 +1,9 @@ # © 2016 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import os diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index cff09bbd..3e65d829 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import argparse import importlib diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 2932367d..4aabbfb1 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -1,19 +1,9 @@ # © 2016 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import logging.handlers diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 06cb6d8b..74c4d8d2 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -1,20 +1,9 @@ #!/usr/bin/env python3 # © 2015-17 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. import locale diff --git a/src/ocrmypdf/_version.py b/src/ocrmypdf/_version.py index 430f76e7..6751fede 100644 --- a/src/ocrmypdf/_version.py +++ b/src/ocrmypdf/_version.py @@ -1,19 +1,8 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. import pkg_resources diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index f9db6a6f..898333ea 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import os diff --git a/src/ocrmypdf/builtin_plugins/__init__.py b/src/ocrmypdf/builtin_plugins/__init__.py index 0ed32bc2..61f9f6d6 100644 --- a/src/ocrmypdf/builtin_plugins/__init__.py +++ b/src/ocrmypdf/builtin_plugins/__init__.py @@ -1,16 +1,5 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 63f79f5e..27f4a99b 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 867e232c..93570cc8 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import os diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index ad867670..df865aa9 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -1,19 +1,9 @@ # © 2015-19 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import argparse diff --git a/src/ocrmypdf/exceptions.py b/src/ocrmypdf/exceptions.py index a2df1963..5228b241 100644 --- a/src/ocrmypdf/exceptions.py +++ b/src/ocrmypdf/exceptions.py @@ -1,19 +1,8 @@ # © 2016 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. from enum import IntEnum diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index d8e1f4a3..76347440 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -1,19 +1,9 @@ # © 2016 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import multiprocessing diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index dd3a543b..808846df 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -3,20 +3,10 @@ # # © 2013-16: jbarlow83 from Github (https://github.com/jbarlow83) # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + # # Python FFI wrapper for Leptonica library diff --git a/src/ocrmypdf/lib/__init__.py b/src/ocrmypdf/lib/__init__.py index 06ca8523..45460814 100644 --- a/src/ocrmypdf/lib/__init__.py +++ b/src/ocrmypdf/lib/__init__.py @@ -1,18 +1,8 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Bindings to external libraries""" diff --git a/src/ocrmypdf/lib/compile_leptonica.py b/src/ocrmypdf/lib/compile_leptonica.py index 50ca52df..0e3afb56 100644 --- a/src/ocrmypdf/lib/compile_leptonica.py +++ b/src/ocrmypdf/lib/compile_leptonica.py @@ -1,20 +1,10 @@ #!/usr/bin/env python3 # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from pathlib import Path diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 2ff5989f..5df6d66d 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import sys diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 46334020..e87db425 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -1,19 +1,9 @@ # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """ Utilities for PDF/A production and confirmation with Ghostspcript. diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index 0e8b8750..2c9a1be9 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -1,19 +1,9 @@ #!/usr/bin/env python3 # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 0f0bf5b1..a2303077 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -1,20 +1,10 @@ #!/usr/bin/env python3 # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import re diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 7cd0b57e..3e0be611 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import re from math import copysign diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 91540cdc..bf2fd22d 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from abc import ABC, abstractmethod, abstractstaticmethod from argparse import ArgumentParser, Namespace diff --git a/src/ocrmypdf/quality.py b/src/ocrmypdf/quality.py index bfb95702..dab816b0 100644 --- a/src/ocrmypdf/quality.py +++ b/src/ocrmypdf/quality.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Utilities to measure OCR quality""" diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index 9e4db351..b8478fe4 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -1,19 +1,9 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Wrappers to manage subprocess calls""" diff --git a/tests/conftest.py b/tests/conftest.py index 7d495171..d5bda6cd 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,19 +1,9 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import os import platform diff --git a/tests/test_acroform.py b/tests/test_acroform.py index 4ab52406..f6932ebe 100644 --- a/tests/test_acroform.py +++ b/tests/test_acroform.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging diff --git a/tests/test_api.py b/tests/test_api.py index c71c81cd..78323ce0 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging from io import BytesIO, StringIO diff --git a/tests/test_check_pdf.py b/tests/test_check_pdf.py index b3516e90..2d192dac 100644 --- a/tests/test_check_pdf.py +++ b/tests/test_check_pdf.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import pytest diff --git a/tests/test_completion.py b/tests/test_completion.py index ccb024aa..fa7cde15 100644 --- a/tests/test_completion.py +++ b/tests/test_completion.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from subprocess import PIPE, run diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index af14f3b3..37a1a1e9 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging from decimal import Decimal diff --git a/tests/test_graft.py b/tests/test_graft.py index 1329e869..19fa3b9d 100644 --- a/tests/test_graft.py +++ b/tests/test_graft.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from unittest.mock import patch diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 397f068e..fe1e5513 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import multiprocessing @@ -22,7 +12,7 @@ from unittest.mock import MagicMock import pytest -import ocrmypdf.helpers as helpers +from ocrmypdf import helpers as helpers from ocrmypdf.subprocess import shim_paths_with_program_files diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index 5eea4407..f2f0c660 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -1,19 +1,9 @@ # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import re from io import StringIO diff --git a/tests/test_image_input.py b/tests/test_image_input.py index c3636952..3a872000 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from unittest.mock import patch diff --git a/tests/test_lept.py b/tests/test_lept.py index 3dcf1fa7..996aa0fe 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -1,19 +1,8 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. from os import fspath diff --git a/tests/test_main.py b/tests/test_main.py index 65c33e39..e30bd62c 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -1,19 +1,9 @@ # © 2015-19 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import os import shutil diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 1c851cef..1580ba79 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -1,19 +1,8 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. import datetime diff --git a/tests/test_optimize.py b/tests/test_optimize.py index b4fee041..f6c1f081 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from os import fspath from pathlib import Path diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index 733153fc..28f59416 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import pytest diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index a7937e4b..16ec12fd 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -1,19 +1,9 @@ # © 2015 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import pickle from math import isclose diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 7cabe827..2fc81c10 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from math import isclose diff --git a/tests/test_quality.py b/tests/test_quality.py index 99f15653..132eef3d 100644 --- a/tests/test_quality.py +++ b/tests/test_quality.py @@ -1,23 +1,13 @@ # © 2020 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import pytest -import ocrmypdf.quality as qual +from ocrmypdf import quality as qual def test_quality_measurement(): diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 2a8056ba..6c17bf8b 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -1,19 +1,9 @@ # © 2018 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from io import BytesIO from os import fspath diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 39a3fade..cfd7f7b3 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import os import sys diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index 0db09110..57ece9af 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -1,19 +1,9 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging import os diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index ca54eb34..09fdcb8f 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -1,19 +1,9 @@ # © 2015-17 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from os import fspath from unittest.mock import patch diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 462c396b..6926d160 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -1,19 +1,9 @@ # © 2017 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + from math import isclose diff --git a/tests/test_validation.py b/tests/test_validation.py index bd9fe098..83aba683 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -1,19 +1,9 @@ # © 2019 James R. Barlow: github.com/jbarlow83 # -# This file is part of OCRmyPDF. -# -# OCRmyPDF is free software: you can redistribute it and/or modify -# it under the terms of the GNU General Public License as published by -# the Free Software Foundation, either version 3 of the License, or -# (at your option) any later version. -# -# OCRmyPDF is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with OCRmyPDF. If not, see . +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + import logging from unittest.mock import patch @@ -21,7 +11,7 @@ from unittest.mock import patch import pikepdf import pytest -import ocrmypdf._validation as vd +from ocrmypdf import _validation as vd from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.api import create_options from ocrmypdf.cli import get_parser From 8c90f7c972202cbd12dc03765626fd82731871ac Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 01:30:31 -0700 Subject: [PATCH 609/880] Replace GPLv3-derived PDF/A template with PostScript generator --- src/ocrmypdf/pdfa.py | 79 ++++++++++++++++++++++++++++---------------- 1 file changed, 50 insertions(+), 29 deletions(-) diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index e87db425..88bd1bad 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -1,4 +1,4 @@ -# © 2015 James R. Barlow: github.com/jbarlow83 +# © 2020 James R. Barlow: github.com/jbarlow83 # # This Source Code Form is subject to the terms of the Mozilla Public # License, v. 2.0. If a copy of the MPL was not distributed with this @@ -11,8 +11,7 @@ Utilities for PDF/A production and confirmation with Ghostspcript. import base64 from pathlib import Path -from string import Template -from typing import Dict, Union +from typing import Dict, Iterator, Union import pikepdf import pkg_resources @@ -22,29 +21,56 @@ ICC_PROFILE_RELPATH = 'data/sRGB.icc' SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPATH) -# This is a template written in PostScript which is needed to create PDF/A -# files, from the Ghostscript documentation. Lines beginning with % are -# comments. Python substitution variables have a '$' prefix. -pdfa_def_template = u"""%! -% Define an ICC profile : -/ICCProfile $icc_profile -def +def _postscript_objdef( + alias: str, + dictionary: Dict[str, str], + *, + stream_name: str = None, + stream_data: bytes = None, +) -> Iterator[str]: + assert (stream_name is None) == (stream_data is None) -[/_objdef {icc_PDFA} /type /stream /OBJ pdfmark -[{icc_PDFA} << /N 3 >> /PUT pdfmark -[{icc_PDFA} ICCProfile /PUT pdfmark + objtype = '/stream' if stream_name else '/dict' -% Define the output intent dictionary : + if stream_name: + a85_data = base64.a85encode(stream_data, adobe=True).decode('ascii') + yield f'{stream_name} ' + a85_data + yield 'def' -[/_objdef {OutputIntent_PDFA} /type /dict /OBJ pdfmark -[{OutputIntent_PDFA} << - /Type /OutputIntent % Must be so (the standard requires). - /S /GTS_PDFA1 % Must be so (the standard requires). - /DestOutputProfile {icc_PDFA} % Must be so (see above). - /OutputConditionIdentifier ($icc_identifier) ->> /PUT pdfmark -[{Catalog} <> /PUT pdfmark -""" + if alias != '{Catalog}': # Catalog needs no definition + yield f'[/_objdef {alias} /type {objtype} /OBJ pdfmark' + + yield f'[{alias} <<' + for key, val in dictionary.items(): + yield f' {key} {val}' + yield '>> /PUT pdfmark' + + if stream_name: + yield f'[{alias} {stream_name[1:]} /PUT pdfmark' + + +def _make_postscript(icc_name: str, icc_data: bytes, colors: int) -> Iterator[str]: + yield '%!' + yield from _postscript_objdef( + '{icc_PDFA}', # Not an f-string + {'/N': str(colors)}, + stream_name='/ICCProfile', + stream_data=icc_data, + ) + yield '' + yield from _postscript_objdef( + '{OutputIntent_PDFA}', + { + '/Type': '/OutputIntent', + '/S': '/GTS_PDFA1', + '/DestOutputProfile': '{icc_PDFA}', + '/OutputConditionIdentifier': f'({icc_name})', # Only f-string + }, + ) + yield '' + yield from _postscript_objdef( + '{Catalog}', {'/OutputIntents': '[ {OutputIntent_PDFA} ]'} + ) def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'): @@ -75,13 +101,8 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'): else: raise NotImplementedError("Only supporting sRGB") - # Read the ICC profile, encode as ASCII85 and convert to a string which we - # will insert in the .ps file bytes_icc_profile = Path(icc_profile).read_bytes() - icc_profile = base64.a85encode(bytes_icc_profile, adobe=True).decode('ascii') - - t = Template(pdfa_def_template) - ps = t.substitute(icc_profile=icc_profile, icc_identifier=icc) + ps = '\n'.join(_make_postscript(icc, bytes_icc_profile, 3)) # We should have encoded everything to pure ASCII by this point, and # to be safe, only allow ASCII in PostScript From bed74501fcbb74d3fe14bec2eec01000a44cdc8d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 01:35:26 -0700 Subject: [PATCH 610/880] Fix test breakage in validation Broken in commit 4cc0dc --- tests/test_validation.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index bd9fe098..754b5b66 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -21,7 +21,7 @@ from unittest.mock import patch import pikepdf import pytest -import ocrmypdf._validation as vd +from ocrmypdf import _validation as vd from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.api import create_options from ocrmypdf.cli import get_parser @@ -141,7 +141,7 @@ def test_report_file_size(tmp_path, caplog): pdf = pikepdf.new() pdf.save(in_) pdf.save(out) - opts = make_opts() + opts = make_opts(output_type='pdf') vd.report_output_file_size(opts, in_, out) assert caplog.text == '' caplog.clear() @@ -166,7 +166,7 @@ def test_report_file_size(tmp_path, caplog): assert 'optional dependency' in caplog.text caplog.clear() - opts = make_opts(in_, out, optimize=0) + opts = make_opts(in_, out, optimize=0, output_type='pdf') vd.report_output_file_size(opts, in_, out) assert 'disabled' in caplog.text caplog.clear() From 4fa28d7e74e6966382811da43ec7ab75102bb856 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 5 Aug 2020 01:36:45 -0700 Subject: [PATCH 611/880] v10.3.2 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 66cf9945..736e35d4 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,13 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.3.2 +======= + +- Fixed a case where we reported "no reason" for a file size increase, when we + could determine the reason. +- Enabled support for pdfminer.six 20200726. + v10.3.1 ======= From 9b641055e10f5bfd1938d200d68ea47337800987 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 7 Aug 2020 02:21:02 -0700 Subject: [PATCH 612/880] Fix KeyError: 'dpi' when using --threshold on image to PDF Fixes #607 --- src/ocrmypdf/_pipeline.py | 4 +++- tests/test_main.py | 14 ++++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 36105fb7..0a69d92b 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -530,7 +530,9 @@ def create_ocr_image(image: Path, page_context: PageContext): if options.threshold: pix = leptonica.Pix.frompil(im) pix = pix.masked_threshold_on_background_norm() - im = pix.topil() + im_pix = pix.topil() + im_pix.info['dpi'] = im.info['dpi'] + im = im_pix del draw diff --git a/tests/test_main.py b/tests/test_main.py index 65c33e39..b6754cd5 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -877,3 +877,17 @@ def test_image_dpi_not_image(caplog, resources, outpdf): 'tests/plugins/tesseract_noop.py', ) assert '--image-dpi is being ignored' in caplog.text + + +def test_image_dpi_threshold(resources, outpdf): + check_ocrmypdf( + resources / 'typewriter.png', + outpdf, + '--threshold', + '--image-dpi=170', + '--output-type=pdf', + '--optimize=0', + '--plugin', + 'tests/plugins/tesseract_noop.py', + ) + assert outpdf.exists() From 173ce2f215e9b12ce4ab7ec230321acb93853112 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 7 Aug 2020 02:23:21 -0700 Subject: [PATCH 613/880] v10.3.3 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 736e35d4..e8979637 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -16,6 +16,12 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and are released publicly should probably also be licensed under GPLv3. +v10.3.3 +======= + +- Fixed a "KeyError: 'dpi'" error message when using ``--threshold`` on an image. + (#607) + v10.3.2 ======= From 04fb1892b40e9c397dd2f8f07dc64aa3598e991e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Aug 2020 11:40:42 -0700 Subject: [PATCH 614/880] Don't ask for sample files anymore --- .github/issue_template.md | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/.github/issue_template.md b/.github/issue_template.md index f3142596..1487de51 100644 --- a/.github/issue_template.md +++ b/.github/issue_template.md @@ -15,11 +15,8 @@ Please check any or all that apply about the test file: - [ ] This is the input file - [ ] The file contains no personal or confidential information -- [ ] I am the copyright holder for this file -- [ ] I permit this file to be included in the OCRmyPDF test suite under the CC-BY-SA 4.0 license -- [ ] I am not the copyright holder, but this file is available under a free software license -Files that are not free for inclusion in this project are quite welcome, but we like to collect free files for our test suite when possible. Please do *not* submit files with confidential information. At your option you may encrypt files for OCRmyPDF's author only. +Please do *not* submit files with confidential information. At your option you may encrypt files for OCRmyPDF's author only. **Expected behavior** A clear and concise description of what you expected to happen. Include screenshots if applicable. @@ -27,7 +24,7 @@ A clear and concise description of what you expected to happen. Include screensh **System:** - OS: [e.g. Linux, macOS] -- OCRmyPDF Version: [e.g. v7.4.0] +- OCRmyPDF Version: [e.g. v10.3.0] **Additional context** Add any other context about the problem here. From 07ab98f5afbb10325885817b93876f9e59f55860 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Aug 2020 12:12:00 -0700 Subject: [PATCH 615/880] docs: mention that Ghostscript PDF/A can swallow hyperlinks Addresses #605 --- docs/introduction.rst | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/docs/introduction.rst b/docs/introduction.rst index 5fb5c594..360d57ec 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -186,6 +186,11 @@ Ghostscript also imposes some limitations: - Ghostscript's PDF/A conversion removes any XMP metadata that is not one of the standard XMP metadata namespaces for PDFs. In particular, PRISM Metdata is removed. +- Ghostscript's PDF/A conversion seems to remove or deactivate + hyperlinks and other active content. + +You can use ``--output-type pdf`` to disable PDF/A conversion and produce +a standard, non-archival PDF. Regarding OCRmyPDF itself: From 56184a762fdf4c9a70667151a0e67bf7304b720c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Aug 2020 12:12:37 -0700 Subject: [PATCH 616/880] Issue template:Give stronger hints about sample input files --- .github/issue_template.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/.github/issue_template.md b/.github/issue_template.md index 1487de51..4aedbdf1 100644 --- a/.github/issue_template.md +++ b/.github/issue_template.md @@ -9,15 +9,17 @@ ocrmypdf ...arguments... input.pdf output.pdf ``` **Example file** -Please include an example *input* PDF (or image). The input file is more helpful. +Please include an example *input* PDF (or image). You could also try to use of the files in ``tests/resources/`` to illustrate your issue. -Please check any or all that apply about the test file: +Please check any or all that apply about the example file: - [ ] This is the input file - [ ] The file contains no personal or confidential information Please do *not* submit files with confidential information. At your option you may encrypt files for OCRmyPDF's author only. +Issues submitted without an example input file are less likely to be resolved. The output file is generally not helpful. + **Expected behavior** A clear and concise description of what you expected to happen. Include screenshots if applicable. From cd35216f217caf317043ca6e6dc3556026467afd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 12 Aug 2020 13:10:08 -0700 Subject: [PATCH 617/880] setup: blacklist pdfminer.six 20200720 NotAllowedError is going to removed https://github.com/pdfminer/pdfminer.six/commit/99f0c09869370d74eb6a27234284b894c33414f4 --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 93990648..3622b52f 100644 --- a/setup.py +++ b/setup.py @@ -73,7 +73,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, <= 20200726', + 'pdfminer.six >= 20191110, != 20200720, <= 20200726', 'pikepdf >= 1.14.0, < 2', 'Pillow >= 7.0.0', 'pluggy >= 0.13.0', From caeba76a61f17e86add8fd63c7a7865abb105823 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 14 Aug 2020 01:34:04 -0700 Subject: [PATCH 618/880] Approve img2pdf 0.4 as it passes tests --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 3622b52f..3b63d44f 100644 --- a/setup.py +++ b/setup.py @@ -72,7 +72,7 @@ setup( install_requires=[ 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional - 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely + 'img2pdf >= 0.3.0, < 0.5', # pure Python, so track HEAD closely 'pdfminer.six >= 20191110, != 20200720, <= 20200726', 'pikepdf >= 1.14.0, < 2', 'Pillow >= 7.0.0', From fc523e837cabb8fe85f0dec7988678d56743f6e8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 17 Aug 2020 23:23:31 -0700 Subject: [PATCH 619/880] Clarify that the GPL-3 portion of pdfa.py was removed Removal was in 8c90f7c97. pdfa.py now has no special licensing and falls unders the "Files: *" clause of debian/copyright. --- debian/copyright | 5 ----- 1 file changed, 5 deletions(-) diff --git a/debian/copyright b/debian/copyright index 4c0cf4f7..a48ac1dc 100644 --- a/debian/copyright +++ b/debian/copyright @@ -60,11 +60,6 @@ Copyright: (C) 2010 Jonathan Brinley (C) 2015-16 James R. Barlow License: Expat -Files: src/ocrmypdf/pdfa.py -Copyright: (C) 2015 James R. Barlow - (C) 1986-2017 The authors of GhostScript -License: GPL-3+ - Files: src/ocrmypdf/_unicodefun.py Copyright: (C) 2014 Armin Ronacher (C) 2017 James R. Barlow From b51a5887e5e81accb2b51f4d99c68af91fdb7e79 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 17 Aug 2020 23:25:31 -0700 Subject: [PATCH 620/880] v11.0.1 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 25e177ec..3d57db29 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.0.1 +======= + +- Blacklist pdfminer.six 20200720, which has a regression fixed in 20200726. +- Approve img2pdf 0.4 as it passes tests. +- Clarify that the GPL-3 portion of pdfa.py was removed with the changes in v11.0.0; + the debian/copyright file did not properly annotate this change. + v11.0.0 ======= From 2ae028bf38cfd18b7112ea569e8bcceb0a52ceb7 Mon Sep 17 00:00:00 2001 From: jbarlow83 Date: Wed, 26 Aug 2020 17:03:09 -0700 Subject: [PATCH 621/880] Update issue templates --- .github/ISSUE_TEMPLATE/bug_report.md | 38 +++++++++--------- .github/ISSUE_TEMPLATE/feature_request.md | 5 ++- .github/ISSUE_TEMPLATE/general-issues.md | 28 +++++++++++++ .../problem-with-a-specific-input-file.md | 40 +++++++++++++++++++ 4 files changed, 90 insertions(+), 21 deletions(-) create mode 100644 .github/ISSUE_TEMPLATE/general-issues.md create mode 100644 .github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md index daf2e24d..dd84ea78 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -11,22 +11,11 @@ assignees: '' A clear and concise description of what the bug is. **To Reproduce** -What command line or API call were you trying to run? - -```bash -ocrmypdf ...arguments... input.pdf output.pdf -``` - -Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful. - -**Example file** -Include an input PDF or image that demonstrates your issue. - -Please provide an input file with no personal or confidential information. At your option you may `GPG-encrypt the file ` for OCRmyPDF's author only. - -Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue. - -(Exceptions: Issues with installation, command line argument parsing, test suite failures.Issues without example files usually cannot be resolved.) +Steps to reproduce the behavior: +1. Go to '...' +2. Click on '....' +3. Scroll down to '....' +4. See error **Expected behavior** A clear and concise description of what you expected to happen. @@ -34,7 +23,16 @@ A clear and concise description of what you expected to happen. **Screenshots** If applicable, add screenshots to help explain your problem. -**System** - - OS: [e.g. Linux, Windows, macOS] - - OCRmyPDF Version: ``ocrmypdf --version`` - - How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image? +**Desktop (please complete the following information):** + - OS: [e.g. iOS] + - Browser [e.g. chrome, safari] + - Version [e.g. 22] + +**Smartphone (please complete the following information):** + - Device: [e.g. iPhone6] + - OS: [e.g. iOS8.1] + - Browser [e.g. stock browser, safari] + - Version [e.g. 22] + +**Additional context** +Add any other context about the problem here. diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md index 59094e26..bbcbbe7d 100644 --- a/.github/ISSUE_TEMPLATE/feature_request.md +++ b/.github/ISSUE_TEMPLATE/feature_request.md @@ -2,7 +2,7 @@ name: Feature request about: Suggest an idea for this project title: '' -labels: enhancement +labels: '' assignees: '' --- @@ -13,5 +13,8 @@ A clear and concise description of what the problem is. Ex. I'm always frustrate **Describe the solution you'd like** A clear and concise description of what you want to happen. +**Describe alternatives you've considered** +A clear and concise description of any alternative solutions or features you've considered. + **Additional context** Add any other context or screenshots about the feature request here. diff --git a/.github/ISSUE_TEMPLATE/general-issues.md b/.github/ISSUE_TEMPLATE/general-issues.md new file mode 100644 index 00000000..1e58db9e --- /dev/null +++ b/.github/ISSUE_TEMPLATE/general-issues.md @@ -0,0 +1,28 @@ +--- +name: General issues +about: Installation, packages, dependencies, "nothing works", test suite failures... +title: '' +labels: '' +assignees: '' + +--- + +**Describe the bug** +What's the problem? + +**To Reproduce** +Steps to reproduce the behavior. + +**Expected behavior** +What did you expected to happen? + +**Screenshots** +If applicable, add screenshots to help explain your problem. + +**System (please complete the following information):** + - OS: + - Python version: + - OCRmyPDF version: + +**Additional context** +Add any other context about the problem here. diff --git a/.github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md b/.github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md new file mode 100644 index 00000000..595d9cee --- /dev/null +++ b/.github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md @@ -0,0 +1,40 @@ +--- +name: Problem with a specific input file +about: Something went wrong while trying to OCR a specific file +title: '' +labels: '' +assignees: '' + +--- + +**Describe the bug** +A clear and concise description of what the bug is. + +**To Reproduce** +What command line or API call were you trying to run? + +```bash +ocrmypdf ...arguments... input.pdf output.pdf +``` + +Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful. + +**Example file** +If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue. + +Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/jbarlow83/OCRmyPDF/wiki) for OCRmyPDF's author only. + +Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue. + +*(Issues without example files usually cannot be resolved. It's like reporting an issue against a web browser without providing a URL.)* + +**Expected behavior** +A clear and concise description of what you expected to happen. + +**Screenshots** +If applicable, add screenshots to help explain your problem. + +**System** + - OS: [e.g. Linux, Windows, macOS] + - OCRmyPDF Version: ``ocrmypdf --version`` + - How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image? From bcf5657e5c77225f5f1e1ed5672af7e669bd3477 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 26 Aug 2020 17:11:52 -0700 Subject: [PATCH 622/880] Reorganize issue templates --- ...{general-issues.md => 1-general-issues.md} | 0 ...> 2-problem-with-a-specific-input-file.md} | 0 ...eature_request.md => 3-feature_request.md} | 0 .github/ISSUE_TEMPLATE/bug_report.md | 38 ------------------- .github/issue_template.md | 32 ---------------- .gitignore | 1 + 6 files changed, 1 insertion(+), 70 deletions(-) rename .github/ISSUE_TEMPLATE/{general-issues.md => 1-general-issues.md} (100%) rename .github/ISSUE_TEMPLATE/{problem-with-a-specific-input-file.md => 2-problem-with-a-specific-input-file.md} (100%) rename .github/ISSUE_TEMPLATE/{feature_request.md => 3-feature_request.md} (100%) delete mode 100644 .github/ISSUE_TEMPLATE/bug_report.md delete mode 100644 .github/issue_template.md diff --git a/.github/ISSUE_TEMPLATE/general-issues.md b/.github/ISSUE_TEMPLATE/1-general-issues.md similarity index 100% rename from .github/ISSUE_TEMPLATE/general-issues.md rename to .github/ISSUE_TEMPLATE/1-general-issues.md diff --git a/.github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md b/.github/ISSUE_TEMPLATE/2-problem-with-a-specific-input-file.md similarity index 100% rename from .github/ISSUE_TEMPLATE/problem-with-a-specific-input-file.md rename to .github/ISSUE_TEMPLATE/2-problem-with-a-specific-input-file.md diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/3-feature_request.md similarity index 100% rename from .github/ISSUE_TEMPLATE/feature_request.md rename to .github/ISSUE_TEMPLATE/3-feature_request.md diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md deleted file mode 100644 index dd84ea78..00000000 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -name: Bug report -about: Create a report to help us improve -title: '' -labels: '' -assignees: '' - ---- - -**Describe the bug** -A clear and concise description of what the bug is. - -**To Reproduce** -Steps to reproduce the behavior: -1. Go to '...' -2. Click on '....' -3. Scroll down to '....' -4. See error - -**Expected behavior** -A clear and concise description of what you expected to happen. - -**Screenshots** -If applicable, add screenshots to help explain your problem. - -**Desktop (please complete the following information):** - - OS: [e.g. iOS] - - Browser [e.g. chrome, safari] - - Version [e.g. 22] - -**Smartphone (please complete the following information):** - - Device: [e.g. iPhone6] - - OS: [e.g. iOS8.1] - - Browser [e.g. stock browser, safari] - - Version [e.g. 22] - -**Additional context** -Add any other context about the problem here. diff --git a/.github/issue_template.md b/.github/issue_template.md deleted file mode 100644 index 4aedbdf1..00000000 --- a/.github/issue_template.md +++ /dev/null @@ -1,32 +0,0 @@ -**Describe the issue** -A clear and concise description of what the issue is. - -**To Reproduce** -What command line were you trying to run? - -```bash -ocrmypdf ...arguments... input.pdf output.pdf -``` - -**Example file** -Please include an example *input* PDF (or image). You could also try to use of the files in ``tests/resources/`` to illustrate your issue. - -Please check any or all that apply about the example file: - -- [ ] This is the input file -- [ ] The file contains no personal or confidential information - -Please do *not* submit files with confidential information. At your option you may encrypt files for OCRmyPDF's author only. - -Issues submitted without an example input file are less likely to be resolved. The output file is generally not helpful. - -**Expected behavior** -A clear and concise description of what you expected to happen. Include screenshots if applicable. - -**System:** - -- OS: [e.g. Linux, macOS] -- OCRmyPDF Version: [e.g. v10.3.0] - -**Additional context** -Add any other context about the problem here. diff --git a/.gitignore b/.gitignore index 90884590..7fc65e63 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ !.gitignore !.pre-commit-config.yaml !.readthedocs.yml +!.github/ # Dev scratch *.ipynb From 1f15ecbca54038470d3ff4887bfbe493da7a3db1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Sep 2020 02:34:10 -0700 Subject: [PATCH 623/880] Add "Postprocessing" message as a hint for long Ghostscript runs --- src/ocrmypdf/_sync.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 4aabbfb1..fc01ba65 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -288,6 +288,7 @@ def exec_concurrent(context: PdfContext): pdf = ocrgraft.finalize() # PDF/A and metadata + log.info("Postprocessing...") pdf = post_process(pdf, context) # Copy PDF file to destination From 31994258fb48dbbd689a7bbd2a839c479879577c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Sep 2020 02:35:16 -0700 Subject: [PATCH 624/880] metadata fixup: don't try to update original PDF's metadata with docinfo --- src/ocrmypdf/_pipeline.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e2fa7766..a37963e3 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -753,7 +753,9 @@ def metadata_fixup(working_file: Path, context: PdfContext): # Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1 # and the XMP Spec do not make this recommendation. if meta.get('dc:title') == 'Untitled': - with original.open_metadata() as original_meta: + with original.open_metadata( + set_pikepdf_as_editor=False, update_docinfo=False + ) as original_meta: if 'dc:title' not in original_meta: del meta['dc:title'] From fa06ea360001ee85da62938a2b832e572c99c8ba Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Sep 2020 02:38:57 -0700 Subject: [PATCH 625/880] v11.0.2 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 3d57db29..0ed04818 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.0.2 +======= + +- Fixed issue #612, TypeError exception. Fixed by eliminating unnecessary repair of + input PDF metadata in memory. + v11.0.1 ======= From 624df9bb23e15b2c60e2820099c65b634e4be854 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 14 Sep 2020 14:35:50 -0700 Subject: [PATCH 626/880] Extend example plugin with example of mono conversion --- misc/example_plugin.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/misc/example_plugin.py b/misc/example_plugin.py index 69196f57..960d0290 100644 --- a/misc/example_plugin.py +++ b/misc/example_plugin.py @@ -30,6 +30,7 @@ log = logging.getLogger(__name__) @hookimpl def add_options(parser): parser.add_argument('--grayscale-ocr', action='store_true') + parser.add_argument('--mono-page', action='store_true') @hookimpl @@ -52,7 +53,13 @@ def filter_ocr_image(page, image): @hookimpl def filter_page_image(page, image_filename): - output = image_filename.with_suffix('.jpg') - with Image.open(image_filename) as im: - im.save(output) - return output + if page.options.mono_page: + with Image.open(image_filename) as im: + im = im.convert('1') + im.save(image_filename) + return image_filename + else: + output = image_filename.with_suffix('.jpg') + with Image.open(image_filename) as im: + im.save(output) + return output From 8b5b02e0d8008267c9095bcdcd94e88892cf28f7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 14 Sep 2020 14:36:12 -0700 Subject: [PATCH 627/880] Expand documentation of filter_page_image --- src/ocrmypdf/pluginspec.py | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index bf2fd22d..469b183f 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -154,11 +154,29 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: produced for a given page, this function will not be called. This is not the image that will be shown to OCR. - ocrmypdf will create the PDF page based on the image format used. If you + If the function does not want to modify the image, it should return + ``image_filename``. The hook may overwrite ``image_filename`` with a new file. + + The output image should preserve the same physical unit dimensions, that is + (width * dpi_x, height * dpi_y). That is, if the image is resized, the DPI + must be adjusted by the reciprocal. If this is not preserved, the PDF page + will be resized and the OCR layer misaligned. OCRmyPDF does not nothing + to enforce these constraints; it is up to the plugin to do sensible things. + + OCRmyPDF will create the PDF page based on the image format used. If you convert the image to a JPEG, the output page will be created as a JPEG, etc. - Note that the ocrmypdf image optimization stage may ultimately chose a + If you change the colorspace, that change will be kept. Note that the + OCRmyPDF image optimization stage, if enabled, may ultimately chose a different format. + If the return value is a file that does not exist, ``FileNotFoundError`` + will occur. The return value should be a path to a file in the same folder + as ``image_filename``. + + Implementation detail: If the value returned is falsy, OCRmyPDF will ignore + the return value and assume the input file was unmodified. This is deprecated. + To leave the image unmodified, ``image_filename`` should be returned. + Note: This hook will be called from child processes. Modifying global state will not affect the main process or other child processes. From 6b994221c615dd928c43d923bd9a41d835b79a7c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 16 Sep 2020 23:44:18 -0700 Subject: [PATCH 628/880] Remove Python 3.7 from build since homebrew removed it --- azure-pipelines.yml | 13 ++----------- 1 file changed, 2 insertions(+), 11 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 3604ced3..f45d1075 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -149,12 +149,6 @@ stages: - job: "macOS_Mojave" pool: vmImage: "macos-10.14" - strategy: - matrix: - Python37: - python.version: "" - Python38: - python.version: "python@3.8" steps: # https://github.com/actions/virtual-environments/issues/664 # - task: UsePythonVersion@0 @@ -163,11 +157,8 @@ stages: - bash: | brew update brew unlink python@2 - if [ "$(python.version)" != "" ]; then - brew upgrade $(python.version) - else - echo "Using Python `python3 --version`" - fi + brew upgrade python + echo "Using Python `python3 --version`" displayName: "Update brew and Python" - bash: | brew install \ From b93cf51c0fa9b99bbbbc6203a255ca71c2b45cee Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 16 Sep 2020 23:48:55 -0700 Subject: [PATCH 629/880] Disable pikepdf mmap Infrequently we can reproduce this error: terminating with uncaught exception of type std::runtime_error: pybind11_object_dealloc(): Tried to deallocate unregistered instance! The error is probably related to pybind11 issue #2252 and a bunch of other related issues. Until that is resolved in pybind11 and pikepdf we will disable the pikepdf mmap interface. --- src/ocrmypdf/helpers.py | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 76347440..42d5e725 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -213,11 +213,15 @@ def clamp(n: T, smallest: T, largest: T) -> T: def pikepdf_enable_mmap(): - try: - if pikepdf._qpdf.set_access_default_mmap(True): - log.debug("pikepdf mmap enabled") - except AttributeError: - log.debug("pikepdf mmap not available") + # try: + # if pikepdf._qpdf.set_access_default_mmap(True): + # log.debug("pikepdf mmap enabled") + # except AttributeError: + # log.debug("pikepdf mmap not available") + # We found a race condition probably related to pybind issue #2252 that can + # cause a crash. For now, disable pikepdf mmap to be on the safe side. + log.debug("pikepdf mmap disabled") + return def deprecated(func): From 306a903854a614fd21ee9c7749e66142ff22303d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Sep 2020 01:20:02 -0700 Subject: [PATCH 630/880] Remove unused function log_page_orientations --- src/ocrmypdf/_validation.py | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 74c4d8d2..7bcd081d 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -308,17 +308,6 @@ def check_closed_streams(options): # pragma: no cover return True -def log_page_orientations(pdfinfo): - direction = {0: 'n', 90: 'e', 180: 's', 270: 'w'} - orientations = [] - for n, page in enumerate(pdfinfo): - angle = page.rotation or 0 - if angle != 0: - orientations.append('{0}{1}'.format(n + 1, direction.get(angle, ''))) - if orientations: - log.info('Page orientations detected: %s', ' '.join(orientations)) - - def create_input_file(options, work_folder: Path) -> Tuple[Path, str]: if options.input_file == '-': # stdin From 67553fc5c6b4e2f05cd7bb171da378c4e54dddbf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Sep 2020 01:20:50 -0700 Subject: [PATCH 631/880] Display page numbers in log messages when grafting --- src/ocrmypdf/_graft.py | 2 +- src/ocrmypdf/_sync.py | 22 +++++++++++++--------- 2 files changed, 14 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 85caeaba..59607348 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -130,7 +130,7 @@ class OcrGrafter: text_rotation = autorotate_correction text_misaligned = (text_rotation - content_rotation) % 360 log.debug( - f"Rotations for page {pageno}: [text, auto, misalign, content] = " + f"Rotations for page: [text, auto, misalign, content] = " f"{text_rotation}, {autorotate_correction}, " f"{text_misaligned}, {content_rotation}" ) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index fc01ba65..cf0bfd61 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -252,15 +252,19 @@ def exec_concurrent(context: PdfContext): ocrgraft = OcrGrafter(context) def update_page(result: PageResult, pbar): - sidecars[result.pageno] = result.text - pbar.update() - ocrgraft.graft_page( - pageno=result.pageno, - image=result.pdf_page_from_image, - textpdf=result.ocr, - autorotate_correction=result.orientation_correction, - ) - pbar.update() + try: + tls.pageno = result.pageno + 1 + sidecars[result.pageno] = result.text + pbar.update() + ocrgraft.graft_page( + pageno=result.pageno, + image=result.pdf_page_from_image, + textpdf=result.ocr, + autorotate_correction=result.orientation_correction, + ) + pbar.update() + finally: + tls.pageno = None exec_progress_pool( use_threads=context.options.use_threads, From 1327ab37d4c267ccdb313052040d5547e6485ead Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Sep 2020 02:57:00 -0700 Subject: [PATCH 632/880] Fix page rotation regression Fixes #634, #581 --- src/ocrmypdf/_graft.py | 46 ++++++++++++++++++++++++------------------ 1 file changed, 26 insertions(+), 20 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 59607348..b5c6928f 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -109,7 +109,6 @@ class OcrGrafter: if textpdf and not self.font: self.font, self.font_key = self._find_font(textpdf) - emplaced_page = False content_rotation = self.pdfinfo[pageno].rotation path_image = Path(image).resolve() if image else None if path_image is not None and path_image != self.path_base: @@ -123,17 +122,21 @@ class OcrGrafter: local_image_page = self.pdf_base.pages[-1] self.pdf_base.pages[pageno].emplace(local_image_page) del self.pdf_base.pages[-1] - emplaced_page = True + # The pdf_image_page will always be created with any /Rotate applied + # applied already + content_rotation = 0 - if emplaced_page: - content_rotation = autorotate_correction - text_rotation = autorotate_correction - text_misaligned = (text_rotation - content_rotation) % 360 - log.debug( - f"Rotations for page: [text, auto, misalign, content] = " - f"{text_rotation}, {autorotate_correction}, " - f"{text_misaligned}, {content_rotation}" - ) + if content_rotation != 0: + # Text can be misaligned on a /Rotate'd page. + # That is because we rasterize pages with /Rotate applied, + # so that the OCR image text is upright and comes back upright. + text_misaligned = (autorotate_correction - content_rotation) % 360 + log.debug( + f"Text rotation: (autorotate, content) -> text misalignment = " + f"({autorotate_correction}, {content_rotation}) -> {text_misaligned}" + ) + else: + text_misaligned = 0 if textpdf and self.font: # Graft the text layer onto this page, whether new or old @@ -143,15 +146,18 @@ class OcrGrafter: textpdf=textpdf, font=self.font, font_key=self.font_key, - rotation=text_misaligned, + text_rotation=text_misaligned, procset=self.procset, strip_old_text=strip_old, ) - # Correct the rotation if applicable - self.pdf_base.pages[pageno].Rotate = ( - content_rotation - autorotate_correction - ) % 360 + # Correct the page rotation + page_rotation = (content_rotation - autorotate_correction) % 360 + self.pdf_base.pages[pageno].Rotate = page_rotation + log.debug( + f"Page rotation: (content, auto) -> page = " + f"({content_rotation}, {autorotate_correction}) -> {page_rotation}" + ) if self.emplacements % MAX_REPLACE_PAGES == 0: self.save_and_reload() @@ -226,7 +232,7 @@ class OcrGrafter: font: pikepdf.Object, font_key: pikepdf.Object, procset: pikepdf.Object, - rotation: int, + text_rotation: int, strip_old_text: bool, ): """Insert the text layer from text page 0 on to pdf_base at page_num""" @@ -256,13 +262,13 @@ class OcrGrafter: corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1]) # -rotation because the input is a clockwise angle and this formula # uses CCW - rotation = -rotation % 360 - rotate = pikepdf.PdfMatrix().rotated(rotation) + text_rotation = -text_rotation % 360 + rotate = pikepdf.PdfMatrix().rotated(text_rotation) # Because of rounding of DPI, we might get a text layer that is not # identically sized to the target page. Scale to adjust. Normally this # is within 0.998. - if rotation in (90, 270): + if text_rotation in (90, 270): wt, ht = ht, wt scale_x = wp / wt scale_y = hp / ht From d464d3122e94205250de868e7e2f09f7cb7e9b9c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 16 Sep 2020 23:57:44 -0700 Subject: [PATCH 633/880] Use img2pdf to create optimized PNG images Fixes #629, #620 --- src/ocrmypdf/optimize.py | 58 +++++++++++++++++++++------------------- 1 file changed, 31 insertions(+), 27 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 5df6d66d..6c6b41db 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -10,6 +10,7 @@ import sys import tempfile from collections import defaultdict from functools import partial +from io import BytesIO from os import fspath from pathlib import Path from typing import ( @@ -27,6 +28,7 @@ from typing import ( Union, ) +import img2pdf import pikepdf from pikepdf import Dictionary, Name, Object, Pdf, PdfImage from PIL import Image @@ -37,7 +39,7 @@ from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._jobcontext import PdfContext from ocrmypdf.exceptions import OutputFileAccessError -from ocrmypdf.helpers import safe_symlink +from ocrmypdf.helpers import deprecated, safe_symlink log = logging.getLogger(__name__) @@ -393,6 +395,30 @@ def transcode_jpegs(pike: Pdf, jpegs: Sequence[Xref], root: Path, options) -> No im_obj.write(compdata.read(), filter=Name.DCTDecode) +def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool: + output = filename.with_suffix('.png.pdf') + with output.open('wb') as f: + img2pdf.convert(fspath(filename), outputstream=f) + + with pikepdf.open(output) as pdf_image: + foreign_image = next(pdf_image.pages[0].images.values()) + local_image = pike.copy_foreign(foreign_image) + + im_obj = pike.get_object(xref, 0) + im_obj.write( + local_image.read_raw_bytes(), + filter=local_image.Filter, + decode_parms=local_image.DecodeParms, + ) + + del_keys = set(im_obj.keys()) - set(local_image.keys()) + for key in local_image.keys(): + if key != Name.Length: + im_obj[key] = local_image[key] + for key in del_keys: + del im_obj[key] + + def transcode_pngs( pike: Pdf, images: Sequence[Xref], @@ -435,34 +461,11 @@ def transcode_pngs( ) for xref in modified: - im_obj = pike.get_object(xref, 0) - try: - pix = leptonica.Pix.open(png_name(root, xref)) - if pix.mode == '1': - compdata = pix.generate_pdf_ci_data(leptonica.lept.L_G4_ENCODE, 0) - else: - compdata = leptonica.CompressedData.open(png_name(root, xref)) - except leptonica.LeptonicaError as e: - # Most likely this means file not found, i.e. quantize did not - # produce an improved version - log.error(e) - continue - - # If re-coded image is larger don't use it - we test here because - # pngquant knows the size of the temporary output file but not the actual - # object in the PDF - if len(compdata) > int(im_obj.stream_dict.Length): - log.debug( - f"pngquant: pngquant did not improve over original image " - f"{len(compdata)} > {int(im_obj.stream_dict.Length)}" - ) - continue - if compdata.type == leptonica.lept.L_FLATE_ENCODE: - rewrite_png(pike, im_obj, compdata) - elif compdata.type == leptonica.lept.L_G4_ENCODE: - rewrite_png_as_g4(pike, im_obj, compdata) + filename = png_name(root, xref) + _transcode_png(pike, filename, xref) +@deprecated def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: im_obj.BitsPerComponent = 1 im_obj.Width = compdata.w @@ -483,6 +486,7 @@ def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: return +@deprecated def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # When a PNG is inserted into a PDF, we more or less copy the IDAT section from # the PDF and transfer the rest of the PNG headers to PDF image metadata. From 9a6cd95e5fe2826d40861229aaa0431b76e302e7 Mon Sep 17 00:00:00 2001 From: Suyash Behera Date: Thu, 17 Sep 2020 15:44:42 +0530 Subject: [PATCH 634/880] load zlib before liblept on windows (#633) fixes #631 --- src/ocrmypdf/leptonica.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 808846df..336c00ff 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -58,6 +58,24 @@ if not _libpath: --------------------------------------------------------------------- """ ) +if os.name == 'nt': + # On Windows, recent versions of libpng require zlib. We have to make sure + # the zlib version being loaded is the same one that libpng was built with. + # This tries to import zlib from Tesseract's installation folder, falling back + # to find_library() if liblept is being loaded from somewhere else. + # Loading zlib from other places could cause a version mismatch + _zlib_path = os.path.join(os.path.dirname(_libpath), 'zlib1.dll') + if not os.path.exists(_zlib_path): + _zlib_path = find_library('zlib') + try: + zlib = ffi.dlopen(_zlib_path) + except ffi.error as e: + raise MissingDependencyError( + """ + Could not load the zlib library. It could be that Tesseract is not installed properly, + we can't find the installation on your system PATH environment variable. + """ + ) from e try: lept = ffi.dlopen(_libpath) lept.setMsgSeverity(lept.L_SEVERITY_WARNING) From b170be120b7315b988527afb39163f8dad07dc01 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Sep 2020 03:21:06 -0700 Subject: [PATCH 635/880] v11.1.0 release notes --- docs/release_notes.rst | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 0ed04818..bfaf826f 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,21 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.1.0 +======= + +- Fixed page rotation issues: #634, #589. +- Fixed some cases where optimization created an invalid image such as a + 1-bit "RGB" iamge: #629, #620. +- Page numbers are now displayed in debug logs when pages are being grafted. +- ocrmypdf.optimize.rewrite_png and ocrmypdf.optimize.rewrite_png_as_g4 were + marked deprecated. Strictly speaking these should have been internal APIs, + but they were never hidden. +- As a precaution, pikepdf mmap-based file access has been disabled due to a + rare race condition that causes a crash when certain objects are deallocated. + The problem is likely in pikepdf's dependency pybind11. +- Extended the example plugin to demonstrate conversion to mono. + v11.0.2 ======= From a40361db3c2dce069e2aa8dfbd081a584147271e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Sep 2020 03:38:48 -0700 Subject: [PATCH 636/880] Remove unpaper from macOS build Homebrew seems to be having issues with its deps? --- azure-pipelines.yml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index f45d1075..5fbb2cf6 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -168,8 +168,7 @@ stages: leptonica \ openjpeg \ pngquant \ - tesseract \ - unpaper + tesseract displayName: "Install system packages" - bash: | pip3 install --upgrade pip From 29097837d614bf8d9d9346023bb9df1927638269 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 19 Sep 2020 00:49:36 -0700 Subject: [PATCH 637/880] Release notes typo --- docs/release_notes.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index bfaf826f..889d2e81 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -17,7 +17,7 @@ v11.1.0 - Fixed page rotation issues: #634, #589. - Fixed some cases where optimization created an invalid image such as a - 1-bit "RGB" iamge: #629, #620. + 1-bit "RGB" image: #629, #620. - Page numbers are now displayed in debug logs when pages are being grafted. - ocrmypdf.optimize.rewrite_png and ocrmypdf.optimize.rewrite_png_as_g4 were marked deprecated. Strictly speaking these should have been internal APIs, From bfe4a5b329b069f8e493e80c18b4091b2b65fba2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Sep 2020 14:32:10 -0700 Subject: [PATCH 638/880] Tidy a log message --- src/ocrmypdf/builtin_plugins/ghostscript.py | 8 +++----- tests/test_validation.py | 2 +- 2 files changed, 4 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 27f4a99b..de21fe30 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -39,13 +39,11 @@ def check_options(options): if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin: # https://bugs.ghostscript.com/show_bug.cgi?id=696874 # Ghostscript < 9.20 fails to encode multibyte characters properly - msg = ( - "The installed version of Ghostscript does not work correctly " - "with the OCR languages you specified. Use --output-type pdf or " + log.warning( + f"The installed version of Ghostscript ({gs_version}) does not work " + "correctly with the OCR languages you specified. Use --output-type pdf or " "upgrade to Ghostscript 9.20 or later to avoid this issue." ) - msg += f"Found Ghostscript {gs_version}" - log.warning(msg) if options.output_type == 'pdfa': options.output_type = 'pdfa-2' diff --git a/tests/test_validation.py b/tests/test_validation.py index e4445435..06481c05 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -55,7 +55,7 @@ def test_old_ghostscript(caplog): vd._check_options( *make_opts_pm(language='chi_sim', output_type='pdfa'), {'chi_sim'} ) - assert 'Ghostscript does not work correctly' in caplog.text + assert 'does not work correctly' in caplog.text with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'), patch( 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True From 28eec73eedd80be56adf23e666e5d99e25d4631e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Sep 2020 16:43:31 -0700 Subject: [PATCH 639/880] Tighten unpaper-args validation to exclude . and .. Just in case --- src/ocrmypdf/_exec/unpaper.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index 6b4383d1..7b797891 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -107,7 +107,7 @@ def run(input_file, output_file, dpi, mode_args): def validate_custom_args(args: str): unpaper_args = shlex.split(args) - if any('/' in arg for arg in unpaper_args): + if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args): raise ValueError('No filenames allowed in --unpaper-args') return unpaper_args From 3ef8872a1ea369a9c445477f6b8bc844e5142ca7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 25 Sep 2020 00:17:19 -0700 Subject: [PATCH 640/880] pngquant driver: refactor, use streams instead of temporary files --- src/ocrmypdf/_exec/pngquant.py | 48 ++++++++++++++++++---------------- src/ocrmypdf/optimize.py | 1 + 2 files changed, 26 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/_exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py index b071eda1..d7e345f2 100644 --- a/src/ocrmypdf/_exec/pngquant.py +++ b/src/ocrmypdf/_exec/pngquant.py @@ -7,7 +7,11 @@ """Interface to pngquant executable""" +from contextlib import contextmanager +from io import BytesIO from os import fspath +from pathlib import Path +from subprocess import PIPE from tempfile import NamedTemporaryFile from PIL import Image @@ -28,34 +32,32 @@ def available(): return True -def quantize(input_file, output_file, quality_min, quality_max): - input_file = fspath(input_file) - output_file = fspath(output_file) - if input_file.endswith('.jpg'): - with Image.open(input_file) as im, NamedTemporaryFile(suffix='.png') as tmp: - im.save(tmp) - args = [ - 'pngquant', - '--force', - '--skip-if-larger', - '--output', - output_file, - '--quality', - f'{quality_min}-{quality_max}', - '--', - tmp.name, - ] - run(args) +@contextmanager +def input_as_png(input_file: Path): + if not input_file.name.endswith('.png'): + with Image.open(input_file) as im: + bio = BytesIO() + im.save(bio, format='png') + bio.seek(0) + yield bio else: + with open(input_file, 'rb') as f: + yield f + + +def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int): + with input_as_png(input_file) as input_stream: args = [ 'pngquant', '--force', '--skip-if-larger', - '--output', - output_file, '--quality', f'{quality_min}-{quality_max}', - '--', - input_file, + '--', # pngquant: stop processing arguments + '-', # pngquant: stream input and output ] - run(args) + result = run(args, stdin=input_stream, stdout=PIPE, stderr=PIPE, check=False) + + if result.returncode == 0: + # input_file could be the same as output_file, so we defer the write + output_file.write_bytes(result.stdout) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 6c6b41db..3201f9d1 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -417,6 +417,7 @@ def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool: im_obj[key] = local_image[key] for key in del_keys: del im_obj[key] + return True def transcode_pngs( From 581c5020ab8fc6fe9d9225a8d201da28b1a5008c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 25 Sep 2020 00:28:38 -0700 Subject: [PATCH 641/880] v11.1.1 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 889d2e81..5cfdf233 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.1.1 +======= + +- We now avoid using named temporary files when using pngquant allowing containerized + pngquant installs to be used. +- Clarified an error message. +- Highest number of 1's in a release ever! + v11.1.0 ======= From 82b8b41e80916ae70b321d69807b397f3318cbd1 Mon Sep 17 00:00:00 2001 From: Jimit Dholakia Date: Sat, 26 Sep 2020 00:24:31 +0530 Subject: [PATCH 642/880] docs: Add 'unpaper' optional dependency for Ubuntu 18.04 (#639) --- docs/installation.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 10125a5d..b1c54559 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -190,7 +190,8 @@ of ocrmypdf, and install the following dependencies: python3-reportlab \ qpdf \ tesseract-ocr \ - zlib1g + zlib1g \ + unpaper We will need a newer version of ``pip`` then was available for Ubuntu 18.04: From 4eacb3454f142dc8b3f590d592140a0ed5e506f7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 29 Sep 2020 02:45:11 -0700 Subject: [PATCH 643/880] hOCR: write text in correct order Fixes #642 --- src/ocrmypdf/hocrtransform.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 7e2ef142..6631cf1e 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -167,7 +167,10 @@ class HocrTransform: def topdown_position(self, element): pxl_line_coords = self.element_coordinates(element) line_box = self.pt_from_pixel(pxl_line_coords) - return -line_box.y2 + # Coordinates here are still in the hocr coordinate system, so 0 on the y axis + # is the top of the page and increasing values of y will move towards the + # bottom of the page. + return line_box.y2 def to_pdf( self, From cccdc178c37223d6df5b30594c1ecdb4bc338e0e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 29 Sep 2020 02:46:18 -0700 Subject: [PATCH 644/880] v11.1.2 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 5cfdf233..b762490e 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.1.2 +======= + +- Fix hOCR renderer writing the text in roughly reverse order. This should not + affect reasonably smart PDF readers that properly locate the position of all + text, but may confuse those that rely on the order of objects in the content + stream. (#642) + v11.1.1 ======= From e0a522ad506b5252e395d864f9dcb64d4b89d302 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 5 Oct 2020 15:01:44 -0700 Subject: [PATCH 645/880] Document the example plugin --- misc/example_plugin.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/misc/example_plugin.py b/misc/example_plugin.py index 960d0290..cabb4ebe 100644 --- a/misc/example_plugin.py +++ b/misc/example_plugin.py @@ -18,6 +18,25 @@ # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. +""" +An example of an OCRmyPDF plugin. + +This plugin adds two new command line arguments + --grayscale-ocr: converts the image to grayscale before performing OCR on it + (This is occasionally useful for images whose color confounds OCR. It only + affects the image shown to OCR. The image is not saved.) + --mono-page: converts pages all pages in the output file to black and white + +To use this from the command line: + ocrmypdf --plugin path/to/example_plugin.py --mono-page input.pdf output.pdf + +To use this as an API: + import ocrmypdf + ocrmypdf.ocr('input.pdf', 'output.pdf', + plugins=['path/to/example_plugin.py'], mono_page=True + ) +""" + import logging from PIL import Image From 8b01ab8ad293b93bada51ba0b3bd5f5b2099d268 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 5 Oct 2020 15:02:34 -0700 Subject: [PATCH 646/880] Better type checking on ocrmypdf.ocr(plugins=...) --- src/ocrmypdf/api.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 898333ea..898d9e11 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -226,7 +226,7 @@ def ocr( # pylint: disable=unused-argument user_words: os.PathLike = None, user_patterns: os.PathLike = None, fast_web_view: float = None, - plugins: Iterable[str] = None, + plugins: Iterable[Union[str, Path]] = None, keep_temporary_files: bool = None, progress_bar: bool = None, **kwargs, @@ -280,6 +280,8 @@ def ocr( # pylint: disable=unused-argument """ if not plugins: plugins = [] + elif isinstance(plugins, (str, Path)): + plugins = [plugins] else: plugins = list(plugins) From 4e15eb8d14e9fca921129aa5cfa160a14d758656 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Oct 2020 03:20:54 -0700 Subject: [PATCH 647/880] Fix image optimization discarding image masks and soft masks associated with PNGs Fixes #648 --- src/ocrmypdf/optimize.py | 27 ++++++++++++++++++++++++--- 1 file changed, 24 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 3201f9d1..9677879b 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -77,23 +77,26 @@ def extract_image_filter( if image.Subtype != Name.Image: return None if image.Length < 100: - log.debug("Skipping small image, xref %s", xref) + log.debug(f"Skipping small image, xref {xref}") return None pim = PdfImage(image) if len(pim.filter_decodeparms) > 1: - log.debug("Skipping multiply filtered, xref %s", xref) + log.debug(f"Skipping multiply filtered image, xref {xref}") return None filtdp = pim.filter_decodeparms[0] if pim.bits_per_component > 8: + log.debug(f"Skipping wide gamut image, xref {xref}") return None # Don't mess with wide gamut images if filtdp[0] == Name.JPXDecode: + log.debug(f"Skipping JPEG2000 iamge, xref {xref}") return None # Don't do JPEG2000 if Name.Decode in image: + log.debug(f"Skipping image with Decode table, xref {xref}") return None # Don't mess with custom Decode tables return pim, filtdp @@ -229,7 +232,9 @@ def extract_images( # Ignore soft masks smask_xref = Xref(image.SMask.objgen[0]) exclude_xrefs.add(smask_xref) + log.debug(f"Skipping image {smask_xref} because it is an SMask") include_xrefs.add(xref) + log.debug(f"Treating {xref} as an optimization candidate") if xref not in pageno_for_xref: pageno_for_xref[xref] = pageno @@ -411,9 +416,25 @@ def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool: decode_parms=local_image.DecodeParms, ) + # Don't copy keys from the new image... del_keys = set(im_obj.keys()) - set(local_image.keys()) + # ...except for the keep_fields, which are essential to displaying + # the image correctly and preserving its metadata. (/Decode arrays + # and /SMaskInData are implicitly discarded prior to this point.) + keep_fields = { + '/ID', + '/Intent', + '/Interpolate', + '/Mask', + '/Metadata', + '/OC', + '/OPI', + '/SMask', + '/StructParent', + } + del_keys -= keep_fields for key in local_image.keys(): - if key != Name.Length: + if key != Name.Length and str(key) not in keep_fields: im_obj[key] = local_image[key] for key in del_keys: del im_obj[key] From 07c6654057a9f05207158e74567d0343f28a3ef8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Oct 2020 03:22:48 -0700 Subject: [PATCH 648/880] v11.1.3 release notes --- docs/release_notes.rst | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index b762490e..7699b478 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,10 +12,16 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.1.3 +======= + +- Fixed an issue with optimizing PNG-type images that had soft masks or image masks. + This is a regression introduced in (or about) issue v11.1.0. + v11.1.2 ======= -- Fix hOCR renderer writing the text in roughly reverse order. This should not +- Fixed hOCR renderer writing the text in roughly reverse order. This should not affect reasonably smart PDF readers that properly locate the position of all text, but may confuse those that rely on the order of objects in the content stream. (#642) From 6eb393590b4124e7daab8391a4f7bacd27e75562 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Oct 2020 03:24:31 -0700 Subject: [PATCH 649/880] v11.2.0 release notes Change v11.1.3 to v11.2.0 since it contains functional changes. --- docs/release_notes.rst | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 7699b478..55d9bb30 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,11 +12,13 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. -v11.1.3 +v11.2.0 ======= - Fixed an issue with optimizing PNG-type images that had soft masks or image masks. - This is a regression introduced in (or about) issue v11.1.0. + This is a regression introduced in (or about) v11.1.0. +- Improved type checking of the ``plugins`` parameter for the ``ocrmypdf.ocr`` + API call. v11.1.2 ======= From 204c9d6ae19851c182e31da34e41f1bcd75ea6b3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Oct 2020 04:08:50 -0700 Subject: [PATCH 650/880] Fix inverted colors during JBIG2 optimization on paletted images Fixes #640 --- docs/release_notes.rst | 6 ++++++ src/ocrmypdf/optimize.py | 11 +++++++++++ 2 files changed, 17 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 55d9bb30..c697aaf4 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.2.1 +======= + +- Fixed an issue where optimization of a 1-bit image with a color palette or + associated ICC that was optimized to JBIG2 could have its colors inverted. + v11.2.0 ======= diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 9677879b..a9adba77 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -116,12 +116,23 @@ def extract_image_jbig2( and jbig2enc.available() ): try: + # Save any colorspace associated with the image, so that we + # will export a pure 1-bit PNG with no palette or ICC profile. + # Showing the palette or ICC to jbig2enc will cause it to perform + # colorspace transform to 1bpp, which will conflict the palette or + # ICC if it exists. + colorspace = pim.obj.ColorSpace + # Set to DeviceGray temporarily; we already in 1 bpc. + pim.obj.ColorSpace = pikepdf.Name.DeviceGray imgname = root / f'{xref:08d}' with imgname.open('wb') as f: ext = pim.extract_to(stream=f) imgname.rename(imgname.with_suffix(ext)) except pikepdf.UnsupportedImageTypeError: return None + finally: + # Restore image colorspace after temporarily setting it to DeviceGray + pim.obj.ColorSpace = colorspace return XrefExt(xref, ext) return None From 6be2242c215db0f6067d865fb7d30c3709c98d43 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Oct 2020 01:03:42 -0700 Subject: [PATCH 651/880] Describe "OCR" step as "Image processing" when --tesseract-timeout=0 Fixes #647 --- src/ocrmypdf/_sync.py | 20 ++++++++++--------- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 4 +--- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index cf0bfd61..22b9afa8 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -209,9 +209,10 @@ def exec_page_sync(page_context: PageContext): if options.pdf_renderer == 'hocr': (hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context) ocr_out = render_hocr_page(hocr_out, page_context) - - if options.pdf_renderer == 'sandwich': + elif options.pdf_renderer == 'sandwich': (ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context) + else: + raise NotImplementedError(f"pdf_renderer {options.pdf_renderer}") return PageResult( pageno=page_context.pageno, @@ -244,7 +245,8 @@ def exec_concurrent(context: PdfContext): """Execute the pipeline concurrently""" # Run exec_page_sync on every page context - max_workers = min(len(context.pdfinfo), context.options.jobs) + options = context.options + max_workers = min(len(context.pdfinfo), options.jobs) if max_workers > 1: log.info("Start processing %d pages concurrently", max_workers) @@ -267,14 +269,14 @@ def exec_concurrent(context: PdfContext): tls.pageno = None exec_progress_pool( - use_threads=context.options.use_threads, + use_threads=options.use_threads, max_workers=max_workers, tqdm_kwargs=dict( total=(2 * len(context.pdfinfo)), - desc='OCR', + desc='OCR' if options.tesseract_timeout > 0 else 'Image processing', unit='page', unit_scale=0.5, - disable=not context.options.progress_bar, + disable=not options.progress_bar, ), task_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS), task=exec_page_sync, @@ -283,10 +285,10 @@ def exec_concurrent(context: PdfContext): ) # Output sidecar text - if context.options.sidecar: + if options.sidecar: text = merge_sidecars(sidecars, context) # Copy text file to destination - copy_final(text, context.options.sidecar, context) + copy_final(text, options.sidecar, context) # Merge layers to one single pdf pdf = ocrgraft.finalize() @@ -296,7 +298,7 @@ def exec_concurrent(context: PdfContext): pdf = post_process(pdf, context) # Copy PDF file to destination - copy_final(pdf, context.options.output_file, context) + copy_final(pdf, options.output_file, context) class NeverRaise(Exception): diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 93570cc8..bbcb6720 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -121,9 +121,7 @@ def validate(pdfinfo, options): os.environ['OMP_THREAD_LIMIT'] = str(tess_threads) else: tess_threads = int(os.environ['OMP_THREAD_LIMIT']) - - if tess_threads > 1: - log.info("Using Tesseract OpenMP thread limit %d", tess_threads) + log.debug("Using Tesseract OpenMP thread limit %d", tess_threads) class TesseractOcrEngine(OcrEngine): From 10c8e4f8b405dcb84269cbe0af54dbf4fb6e610d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 20 Oct 2020 01:29:22 -0700 Subject: [PATCH 652/880] Only create debug.log when running from command line When used as a library ocrmypdf shouldn't make policy decisions, like where to put a log file. Unsurprisingly, creating it causes problems for library users because we deleted the temporary folder which held the log file and made no effort to move it to a new location. Also update the documentation to better described how an application should handle this. Closes #657 --- src/ocrmypdf/_sync.py | 7 +++++-- src/ocrmypdf/api.py | 32 ++++++++++++++++++++++++-------- 2 files changed, 29 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 22b9afa8..68c24da5 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -330,9 +330,12 @@ def run_pipeline(options, *, plugin_manager, api=False): work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf.")) debug_log_handler = None - if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get( - 'PYTEST_CURRENT_TEST', '' + if ( + (options.keep_temporary_files or options.verbose >= 1) + and not os.environ.get('PYTEST_CURRENT_TEST', '') + and not api ): + # Debug log for command line interface only with verbose output debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") pikepdf_enable_mmap() diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 898d9e11..f8809157 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -11,6 +11,7 @@ import sys from enum import IntEnum from pathlib import Path from typing import BinaryIO, Iterable, Union +from warnings import warn from ocrmypdf._logging import PageNumberFilter, TqdmConsole from ocrmypdf._plugin_manager import get_plugin_manager @@ -44,16 +45,28 @@ def configure_logging( ): """Set up logging. - Library users may wish to use this function if they want their log output to be - similar to ocrmypdf command line interface. If not used, the external application - should configure logging on its own. + Before calling :func:`ocrmypdf.ocr()`, you can use this function to + configure logging, if you want ocrmypdf's output to look like the ocrmypdf + command line interface. It will register log handlers, log filters, and + formatters, configure color logging to standard error, and adjust the log + levels of third party libraries. Details of this are fine-tuned and subject + to change. The ``verbosity`` argument is equivalent to the argument + ``--verbose`` and applies those settings. - ocrmypdf will perform all of its logging under the ``"ocrmypdf"`` logging namespace. - In addition, ocrmypdf imports pdfminer, which logs under ``"pdfminer"``. A library - user may wish to configure both; note that pdfminer is extremely chatty at the log - level ``logging.INFO``. + If this function is not called, ocrmypdf will not configure logging, and it + is up to the caller of ``ocrmypdf.ocr()`` to set up logging as it wishes using + the Python standard library's logging module. If this function is called, + the caller may of course make further adjustments to logging. - Library users may perform additional configuration afterwards. + Regardless of whether this function is called, ocrmypdf will perform all of + its logging under the ``"ocrmypdf"`` logging namespace. In addition, + ocrmypdf imports pdfminer, which logs under ``"pdfminer"``. A library user + may wish to configure both; note that pdfminer is extremely chatty at the + log level ``logging.INFO``. + + This function does not set up the ``debug.log`` log file that the command + line interface does at certain verbosity levels. Applications should configure + their own debug logging. Args: verbosity (Verbosity): Verbosity level. @@ -294,6 +307,9 @@ def ocr( # pylint: disable=unused-argument } create_options_kwargs.update(kwargs) + if 'verbose' in kwargs: + warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().") + options = create_options(**create_options_kwargs) check_options(options, _plugin_manager) return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True) From d1e0c81edab4744d975811c87413537e046de663 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 22 Oct 2020 00:38:16 -0700 Subject: [PATCH 653/880] Ensure worker_pdf is closed after gathering info in a thread This is hacky, uses global state, but it does improve the situation for now. --- src/ocrmypdf/pdfinfo/info.py | 32 +++++++++++++++++++++----------- 1 file changed, 21 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index a2303077..44638abb 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -630,6 +630,9 @@ worker_pdf = None def _pdf_pageinfo_sync_init(infile): global worker_pdf # pylint: disable=global-statement pikepdf_enable_mmap() + # If this function is called as a thread initializer, we need a messy hack + # to close worker_pdf. If called as a process, it will be released when the + # process is terminated. worker_pdf = pikepdf.open(infile) @@ -643,6 +646,7 @@ def _pdf_pageinfo_sync(args): def _pdf_pageinfo_concurrent( pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False ): + global worker_pdf # pylint: disable=global-statement pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -663,17 +667,23 @@ def _pdf_pageinfo_concurrent( # a separate process. use_threads = True - exec_progress_pool( - use_threads=use_threads, - max_workers=n_workers, - tqdm_kwargs=dict( - total=total, desc="Scanning contents", unit='page', disable=not progbar - ), - task_initializer=partial(_pdf_pageinfo_sync_init, infile), - task=_pdf_pageinfo_sync, - task_arguments=contexts, - task_finished=update_pageinfo, - ) + try: + exec_progress_pool( + use_threads=use_threads, + max_workers=n_workers, + tqdm_kwargs=dict( + total=total, desc="Scanning contents", unit='page', disable=not progbar + ), + task_initializer=partial(_pdf_pageinfo_sync_init, infile), + task=_pdf_pageinfo_sync, + task_arguments=contexts, + task_finished=update_pageinfo, + ) + finally: + if worker_pdf and use_threads: + assert n_workers == 1, "Should have only one worker when threaded" + # This is messy, but if we ran in thread, close worker_pdf + worker_pdf.close() return pages From 8c35d6e6e41600cd8852352853dafdb254fc78b5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 22 Oct 2020 02:20:06 -0700 Subject: [PATCH 654/880] Fix debug log messages being suppressed from child processes --- src/ocrmypdf/_concurrent.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index bd02f65b..22234580 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -49,7 +49,7 @@ def process_sigbus(*args): raise InputFileError("A worker process lost access to an input file") -def process_init(queue, user_init): +def process_init(queue, user_init, loglevel): """Initialize a process pool worker""" # Ignore SIGINT (our parent process will kill us gracefully) @@ -62,6 +62,7 @@ def process_init(queue, user_init): # Reconfigure the root logger for this process to send all messages to a queue h = logging.handlers.QueueHandler(queue) root = logging.getLogger() + root.setLevel(loglevel) root.handlers = [] root.addHandler(h) @@ -69,7 +70,7 @@ def process_init(queue, user_init): user_init() -def thread_init(_queue, user_init): +def thread_init(_queue, user_init, _loglevel): # As a thread, block SIGBUS so the main thread deals with it... if hasattr(signal, 'SIGBUS'): signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) @@ -102,7 +103,7 @@ def exec_progress_pool( pool = pool_class( processes=max_workers, initializer=initializer, - initargs=(log_queue, task_initializer), + initargs=(log_queue, task_initializer, logging.getLogger("").level), ) try: results = pool.imap_unordered(task, task_arguments) From b5ccbfdf25188e166f0aecae00045b327e341151 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 24 Oct 2020 02:35:18 -0700 Subject: [PATCH 655/880] Fix hookspec of rasterize_pdf_page to remove default parameters --- src/ocrmypdf/_exec/ghostscript.py | 4 ++-- src/ocrmypdf/_pipeline.py | 9 +++++++-- src/ocrmypdf/builtin_plugins/ghostscript.py | 6 +++--- src/ocrmypdf/pluginspec.py | 6 +++--- 4 files changed, 15 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 0358dc78..e2fa9726 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -81,8 +81,8 @@ def rasterize_pdf( raster_device: str, raster_dpi: Resolution, pageno: int = 1, - page_dpi: Resolution = None, - rotation: int = None, + page_dpi: Optional[Resolution] = None, + rotation: Optional[int] = None, filter_vector: bool = False, ): """Rasterize one page of a PDF at resolution raster_dpi in canvas units.""" diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index a37963e3..67bb26c8 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -332,8 +332,10 @@ def rasterize_preview(input_file: Path, page_context: PageContext): output_file=output_file, raster_device='jpeggray', raster_dpi=canvas_dpi, - page_dpi=page_dpi, pageno=page_context.pageinfo.pageno + 1, + page_dpi=page_dpi, + rotation=0, + filter_vector=False, ) return output_file @@ -433,7 +435,7 @@ def rasterize( device = colorspaces[device_idx] - log.debug(f"Rasterize with {device}") + log.debug(f"Rasterize with {device}, rotation {correction}") # Produce the page image with square resolution or else deskew and OCR # will not work properly. @@ -534,6 +536,9 @@ def create_ocr_image(image: Path, page_context: PageContext): # Pillow requires integer DPI dpi = tuple(round(coord) for coord in im.info['dpi']) + if page_context.pageinfo.rotation != 0: + log.info(f"Rotating {page_context.pageinfo.rotation}") + im = im.rotate(page_context.pageinfo.rotation) im.save(output_file, dpi=dpi) return output_file diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index de21fe30..3be822d0 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -61,9 +61,9 @@ def rasterize_pdf_page( raster_device, raster_dpi, pageno, - page_dpi=None, - rotation=None, - filter_vector=False, + page_dpi, + rotation, + filter_vector, ): ghostscript.rasterize_pdf( input_file, diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 469b183f..b34e467f 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -89,9 +89,9 @@ def rasterize_pdf_page( raster_device: str, raster_dpi: Resolution, pageno: int, - page_dpi: Optional[Resolution] = None, - rotation: Optional[int] = None, - filter_vector: bool = False, + page_dpi: Optional[Resolution], + rotation: Optional[int], + filter_vector: bool, ) -> Path: """Rasterize one page of a PDF at resolution raster_dpi in canvas units. From ca735278e02f06d8e5ea731924f20667bb1f0464 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 24 Oct 2020 02:35:41 -0700 Subject: [PATCH 656/880] setup: Version pluggy better --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 3b63d44f..c53a183e 100644 --- a/setup.py +++ b/setup.py @@ -76,7 +76,7 @@ setup( 'pdfminer.six >= 20191110, != 20200720, <= 20200726', 'pikepdf >= 1.14.0, < 2', 'Pillow >= 7.0.0', - 'pluggy >= 0.13.0', + 'pluggy >= 0.13.0, < 1.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling 'tqdm >= 4', ], From 5ba56adb5338db331d1356ddd69759a4e791871e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 24 Oct 2020 02:45:21 -0700 Subject: [PATCH 657/880] Fix page rotation issue (again) Commit 1327ab3 introduced a fix for a regression, which was reported in #581, #634. It appears that the actual cause of this issue was default parameters to rasterize_pdf_page in pluggy not working as expected, causing a default rotation=0 even when a rotation was needed. As such the OCR image was generated with the wrong orientation, causing the initial regression and fix in commit 1327ab3. Now that the real problem is identified, it's apparent that the logic prior to 1327ab3 was found and we can revert to 1327ab3 since it fixes all known cases including #658. This reverts 1327ab3 except for retaining improves to rotation output. --- src/ocrmypdf/_graft.py | 32 +++++++++++++++----------------- 1 file changed, 15 insertions(+), 17 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index b5c6928f..33d18e13 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -109,6 +109,7 @@ class OcrGrafter: if textpdf and not self.font: self.font, self.font_key = self._find_font(textpdf) + emplaced_page = False content_rotation = self.pdfinfo[pageno].rotation path_image = Path(image).resolve() if image else None if path_image is not None and path_image != self.path_base: @@ -122,24 +123,21 @@ class OcrGrafter: local_image_page = self.pdf_base.pages[-1] self.pdf_base.pages[pageno].emplace(local_image_page) del self.pdf_base.pages[-1] - # The pdf_image_page will always be created with any /Rotate applied - # applied already - content_rotation = 0 + emplaced_page = True - if content_rotation != 0: - # Text can be misaligned on a /Rotate'd page. - # That is because we rasterize pages with /Rotate applied, - # so that the OCR image text is upright and comes back upright. - text_misaligned = (autorotate_correction - content_rotation) % 360 - log.debug( - f"Text rotation: (autorotate, content) -> text misalignment = " - f"({autorotate_correction}, {content_rotation}) -> {text_misaligned}" - ) - else: - text_misaligned = 0 + # Calculate if the text is misaligned compared to the content + if emplaced_page: + content_rotation = autorotate_correction + text_rotation = autorotate_correction + text_misaligned = (text_rotation - content_rotation) % 360 + log.debug( + f"Text rotation: (text, autorotate, content) -> text misalignment = " + f"({text_rotation}, {autorotate_correction}, {content_rotation}) -> {text_misaligned}" + ) if textpdf and self.font: - # Graft the text layer onto this page, whether new or old + # Graft the text layer onto this page, whether new or old, possibly + # rotating the text layer by the amount is misaligned. strip_old = self.context.options.redo_ocr self._graft_text_layer( page_num=pageno + 1, @@ -151,14 +149,14 @@ class OcrGrafter: strip_old_text=strip_old, ) - # Correct the page rotation + # Correct the overall page rotation if needed, now that the text and content + # are aligned page_rotation = (content_rotation - autorotate_correction) % 360 self.pdf_base.pages[pageno].Rotate = page_rotation log.debug( f"Page rotation: (content, auto) -> page = " f"({content_rotation}, {autorotate_correction}) -> {page_rotation}" ) - if self.emplacements % MAX_REPLACE_PAGES == 0: self.save_and_reload() From e8285b1d10446cf026e08ca840536d1e698e155a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 24 Oct 2020 03:10:59 -0700 Subject: [PATCH 658/880] Add test to confirm rasterize_pdf_page rotates correct --- tests/test_rotation.py | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 6c17bf8b..e0494d6e 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -15,6 +15,7 @@ from PIL import Image from ocrmypdf import leptonica from ocrmypdf._exec import ghostscript, tesseract +from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import PdfInfo @@ -256,3 +257,33 @@ def test_tesseract_orientation(resources, tmp_path): tesseract.get_orientation( # Test results of this are unreliable tmp_path / '000001.png', engine_mode='3', timeout=10 ) + + +def test_rasterize_rotates(resources, tmp_path): + pm = get_plugin_manager(['tests/plugins/tesseract_rotate90.py']) + + img = tmp_path / 'img90.png' + pm.hook.rasterize_pdf_page( + input_file=resources / 'graph.pdf', + output_file=img, + raster_device='pngmono', + raster_dpi=Resolution(20, 20), + page_dpi=Resolution(20, 20), + pageno=1, + rotation=90, + filter_vector=False, + ) + assert Image.open(img).size == (123, 151), "Image not rotated" + + img = tmp_path / 'img180.png' + pm.hook.rasterize_pdf_page( + input_file=resources / 'graph.pdf', + output_file=img, + raster_device='pngmono', + raster_dpi=Resolution(20, 20), + page_dpi=Resolution(20, 20), + pageno=1, + rotation=180, + filter_vector=False, + ) + assert Image.open(img).size == (151, 123), "Image not rotated" From b0dcaa7512ede6dd1130fd88ba21d8acfb364b15 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 24 Oct 2020 03:19:32 -0700 Subject: [PATCH 659/880] v11.3.0 release notes --- docs/release_notes.rst | 19 +++++++++++++++++++ tests/test_rotation.py | 2 +- 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c697aaf4..57c012ef 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,25 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.3.0 +======= + +- The "OCR" step is describing as "Image processing" in the output messages when + OCR is disabled, to better explain the application's behavior. +- Debug logs are now only created when run as a command line, and not when OCR + is performed for an API call. It is the calling application's responsibility + to set up logging. +- For PDFs with a low number of pages, we gathered information about the input PDF + in a thread rather than process (when there are more pages). When run as a + thread, we did not close the file handle to the working PDF, leaking one file + handle per call of ``ocrmypdf.ocr``. +- Fixed an issue where debug messages send by child worker processes did not match + the log settings of parent process, causing messages to be dropped. This affected + macOS and Windows only where the parent process is not forked. +- Fixed the hookspec of rasterize_pdf_page to remove default parameters that + were not handled in an expected way by pluggy. +- Fixed another issue with automatic page rotation (#658) due to the issue above. + v11.2.1 ======= diff --git a/tests/test_rotation.py b/tests/test_rotation.py index e0494d6e..090d5b32 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -260,7 +260,7 @@ def test_tesseract_orientation(resources, tmp_path): def test_rasterize_rotates(resources, tmp_path): - pm = get_plugin_manager(['tests/plugins/tesseract_rotate90.py']) + pm = get_plugin_manager([]) img = tmp_path / 'img90.png' pm.hook.rasterize_pdf_page( From 2def7e3392874d5249823dc948eca266d5310fbf Mon Sep 17 00:00:00 2001 From: Edward Betts Date: Wed, 28 Oct 2020 06:09:14 +0000 Subject: [PATCH 660/880] Use % for percentage in string format (#643) --- src/ocrmypdf/optimize.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index a9adba77..16060ec6 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -613,7 +613,7 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non ) ratio = input_size / output_size savings = 1 - output_size / input_size - log.info(f"Optimize ratio: {ratio:.2f} savings: {(100 * savings):.1f}%") + log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}") if savings < 0: log.info("Image optimization did not improve the file - discarded") From 21b90d2d147fe570aab5416a92ed85d998ee0128 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Oct 2020 17:40:16 -0700 Subject: [PATCH 661/880] Endorse pikepdf 2.x --- setup.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/setup.py b/setup.py index c53a183e..0bed1e8d 100644 --- a/setup.py +++ b/setup.py @@ -63,7 +63,6 @@ setup( python_requires=' >= 3.6', setup_requires=[ # can be removed whenever we can drop pip 9 support 'cffi >= 1.9.1', # to build the leptonica module - 'pytest-runner', # to enable python setup.py test 'setuptools_scm', # so that version will work 'setuptools_scm_git_archive', # enable version from github tarballs ], @@ -74,7 +73,7 @@ setup( 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.5', # pure Python, so track HEAD closely 'pdfminer.six >= 20191110, != 20200720, <= 20200726', - 'pikepdf >= 1.14.0, < 2', + 'pikepdf >= 1.14.0, < 3', 'Pillow >= 7.0.0', 'pluggy >= 0.13.0, < 1.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From d55e673d9cbcdb714cbf385820d265df4aaaef79 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Oct 2020 17:40:58 -0700 Subject: [PATCH 662/880] Fix warning about --pdfa-image-compression argument at wrong times Closes #663 --- src/ocrmypdf/_validation.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 7bcd081d..9b82ebb4 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -213,12 +213,12 @@ def check_options_optimizing(options): def check_options_advanced(options): - if options.pdfa_image_compression != 'auto' and options.output_type.startswith( + if options.pdfa_image_compression != 'auto' and not options.output_type.startswith( 'pdfa' ): log.warning( - "--pdfa-image-compression argument has no effect when " - "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" + "--pdfa-image-compression argument only applies when " + "--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'" ) From 67f99c5bb730805b60bc5cc46af875af44b6926e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Oct 2020 17:43:22 -0700 Subject: [PATCH 663/880] Endorse pdfminer.six 20201018 --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 0bed1e8d..b941b26c 100644 --- a/setup.py +++ b/setup.py @@ -72,7 +72,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.5', # pure Python, so track HEAD closely - 'pdfminer.six >= 20191110, != 20200720, <= 20200726', + 'pdfminer.six >= 20191110, != 20200720, <= 20201018', 'pikepdf >= 1.14.0, < 3', 'Pillow >= 7.0.0', 'pluggy >= 0.13.0, < 1.0', From 709c65b41aa21ba6d00b4c4cdc0cadff7ffd4b0f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Oct 2020 23:11:11 -0700 Subject: [PATCH 664/880] v11.3.1 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 57c012ef..1eda9aca 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.3.1 +======= + +- Declare support for new versions: pdfminer.six 20201018 and pikepdf 2.x +- Fix warning related to ``--pdfa-image-compression`` that appears at the wrong + time. + v11.3.0 ======= From b21b048ec4f68067de41aa2aa266c57ef9c2fc53 Mon Sep 17 00:00:00 2001 From: Graham Miln Date: Fri, 30 Oct 2020 09:09:06 +0100 Subject: [PATCH 665/880] Add macOS brew language support (#615) Note `brew` command for installing additional languages on macOS. --- README.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/README.md b/README.md index cb91148c..c4c16856 100644 --- a/README.md +++ b/README.md @@ -92,6 +92,9 @@ apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified lan # Arch Linux users pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs + +# brew macOS users +brew install tesseract-lang ``` You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested. From a354663ee172706374979a5b8cfc57df9f66b4a5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 00:58:28 -0800 Subject: [PATCH 666/880] Fix typo in API documentation --- src/ocrmypdf/api.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index f8809157..a9e2d625 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -261,7 +261,7 @@ def ocr( # pylint: disable=unused-argument read. output_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is interpreted as file system path to the output file. If the object - appears to be a writable stream (with methods such as ``.read()`` and + appears to be a writable stream (with methods such as ``.write()`` and ``.seek()``), the output will be written to this stream. If ``output_file`` is ``"-"``, the output will be written to ``sys.stdout`` (provided that standard output does not seem to be a terminal device). From 664d0c7969d6b7e53fd554c7c342ee6a10547b43 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 00:59:00 -0800 Subject: [PATCH 667/880] Document configure_debug_logging --- src/ocrmypdf/_sync.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 68c24da5..2525fe05 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -307,7 +307,14 @@ class NeverRaise(Exception): pass # pylint: disable=unnecessary-pass -def configure_debug_logging(log_filename, prefix=''): +def configure_debug_logging(log_filename, prefix: str = ''): + """ + Create a debug log file at a specified location. + + Arguments: + log_filename: Where to the put the log file. + prefix: The logging domain prefix that should be sent to the log. + """ log_file_handler = logging.FileHandler(log_filename, delay=True) log_file_handler.setLevel(logging.DEBUG) formatter = logging.Formatter( From d57df2d980e4531b3f73d88839d22be28d6d3d61 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 01:00:59 -0800 Subject: [PATCH 668/880] subprocess: support programs that write their messages to stdout --- src/ocrmypdf/subprocess.py | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index b8478fe4..68f6557c 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -25,13 +25,21 @@ from ocrmypdf.exceptions import MissingDependencyError log = logging.getLogger(__name__) -def run(args, *, env=None, **kwargs): +def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs): """Wrapper around :py:func:`subprocess.run` The main purpose of this wrapper is to log subprocess output in an orderly fashion that indentifies the responsible subprocess. An additional task is that this function goes to greater lengths to find possible Windows locations of our dependencies when they are not on the system PATH. + + Arguments should be identical to ``subprocess.run``, except for following: + + Arguments: + logs_errors_to_stdout: If True, indicates that the process writes its error + messages to stdout rather than stderr, so stdout should be logged + if there is an error. If False, stderr is logged. Could be used with + stderr=STDOUT, stdout=PIPE for example. """ if not env: env = os.environ @@ -50,18 +58,22 @@ def run(args, *, env=None, **kwargs): kwargs['close_fds'] = False stderr = None + stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout' try: proc = subprocess_run(args, env=env, **kwargs) except CalledProcessError as e: - stderr = getattr(e, 'stderr', None) + stderr = getattr(e, stderr_name, None) raise else: - stderr = getattr(proc, 'stderr', None) + stderr = getattr(proc, stderr_name, None) finally: if process_log.isEnabledFor(logging.DEBUG) and stderr: with suppress(AttributeError, UnicodeDecodeError): stderr = stderr.decode('utf-8', 'replace') - process_log.debug("stderr = %s", stderr) + if logs_errors_to_stdout: + process_log.debug("stdout/stderr = %s", stderr) + else: + process_log.debug("stderr = %s", stderr) return proc From 6425977998f257b23bfe559efb7b19c3231ff854 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 01:05:56 -0800 Subject: [PATCH 669/880] unpaper: use pnm instead of png Some users reported problems with PNG recently; try PNM. Fixes #665 Fixes #667 --- src/ocrmypdf/_exec/unpaper.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index 7b797891..a490f985 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -55,21 +55,21 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: "Failed to convert image to a supported format." ) from e - if im_modified or input_file.suffix != '.png': - input_png = tmpdir / 'input.png' - im.save(input_png, format='PNG', compress_level=1) + if im_modified or input_file.suffix != '.pnm': + input_pnm = tmpdir / 'input.pnm' + im.save(input_pnm, format='PPM') else: # No changes, PNG input, just use the file we already have - input_png = input_file + input_pnm = input_file output_pnm = tmpdir / f'output{suffix}' - return input_png, output_pnm + return input_pnm, output_pnm def run(input_file, output_file, dpi, mode_args): args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args with TemporaryDirectory() as tmpdir: - input_png, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file) + input_pnm, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file) # To prevent any shenanigans from accepting arbitrary parameters in # --unpaper-args, we: @@ -78,7 +78,7 @@ def run(input_file, output_file, dpi, mode_args): # 3) append absolute paths for the input and output file # This should ensure that a user cannot clobber some other file with # their unpaper arguments (whether intentionally or otherwise) - args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)]) + args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)]) try: proc = external_run( args_unpaper, From e86be0031c6b41f38390dbdbdcfa78389561ef1e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 01:07:41 -0800 Subject: [PATCH 670/880] unpaper: fix process output handling With the ocrmypdf.subprocess wrapper, logging the output here is redundant and loses the page number context. --- src/ocrmypdf/_exec/unpaper.py | 41 +++++++++++++++-------------------- 1 file changed, 18 insertions(+), 23 deletions(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index a490f985..f61132e6 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -79,30 +79,25 @@ def run(input_file, output_file, dpi, mode_args): # This should ensure that a user cannot clobber some other file with # their unpaper arguments (whether intentionally or otherwise) args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)]) + external_run( + args_unpaper, + close_fds=True, + check=True, + universal_newlines=True, + stderr=STDOUT, # unpaper writes logging output to stdout and stderr + stdout=PIPE, # and cannot send file output to stdout + cwd=tmpdir, + logs_errors_to_stdout=True, + ) try: - proc = external_run( - args_unpaper, - check=True, - close_fds=True, - universal_newlines=True, - stderr=STDOUT, # unpaper writes logging output to stdout and stderr - cwd=tmpdir, # and cannot send file output to stdout - stdout=PIPE, - ) - except CalledProcessError as e: - log.debug(e.stderr) - raise e from e - else: - log.debug(proc.stderr) - try: - with Image.open(output_pnm) as imout: - imout.save(output_file, dpi=(dpi, dpi)) - except (FileNotFoundError, OSError): - raise SubprocessOutputError( - "unpaper: failed to produce the expected output file. " - + " Called with: " - + str(args_unpaper) - ) from None + with Image.open(output_pnm) as imout: + imout.save(output_file, dpi=(dpi, dpi)) + except (FileNotFoundError, OSError): + raise SubprocessOutputError( + "unpaper: failed to produce the expected output file. " + + " Called with: " + + str(args_unpaper) + ) from None def validate_custom_args(args: str): From 19bf3aeb00c41392853f71c05ff0297b446bb7ae Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 02:33:34 -0800 Subject: [PATCH 671/880] api: improve typing --- src/ocrmypdf/api.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index a9e2d625..017a8506 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -10,7 +10,7 @@ import os import sys from enum import IntEnum from pathlib import Path -from typing import BinaryIO, Iterable, Union +from typing import AnyStr, BinaryIO, Iterable, Optional, Union from warnings import warn from ocrmypdf._logging import PageNumberFilter, TqdmConsole @@ -26,7 +26,8 @@ except ModuleNotFoundError: coloredlogs = None -PathOrIO = Union[BinaryIO, os.PathLike, str, bytes] +StrPath = Union[os.PathLike, AnyStr] +PathOrIO = Union[BinaryIO, StrPath] class Verbosity(IntEnum): @@ -202,7 +203,7 @@ def ocr( # pylint: disable=unused-argument language: Iterable[str] = None, image_dpi: int = None, output_type=None, - sidecar: os.PathLike = None, + sidecar: Optional[StrPath] = None, jobs: int = None, use_threads: bool = None, title: str = None, @@ -239,7 +240,7 @@ def ocr( # pylint: disable=unused-argument user_words: os.PathLike = None, user_patterns: os.PathLike = None, fast_web_view: float = None, - plugins: Iterable[Union[str, Path]] = None, + plugins: Iterable[StrPath] = None, keep_temporary_files: bool = None, progress_bar: bool = None, **kwargs, From e5df98cbdfc55bc72b0b8b669940cbdf03ef8136 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 2 Nov 2020 02:43:32 -0800 Subject: [PATCH 672/880] v11.3.2 release notes --- docs/release_notes.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 1eda9aca..f80f318b 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,15 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.3.2 +======= + +- On some systems, unpaper seems to be unable to process the PNGs we offer it + as input. We now convert the input to PNM format, which unpaper always accepts. + Fixes #665 and #667. +- Debug and error messages from unpaper were being suppressed. +- Some documentation tweaks. + v11.3.1 ======= From dce206d3dc01c2a7f74912dd3ab317f51df46c77 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 00:20:25 -0800 Subject: [PATCH 673/880] Fix pre-commit for Py3.9 --- .pre-commit-config.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 05d84fd1..53d72257 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -19,5 +19,5 @@ repos: rev: 19.10b0 hooks: - id: black - language_version: python3.8 + language_version: python exclude: ^src/ocrmypdf/lib/_leptonica.py From 7f73a6ed1ee9b2a4c32854db8cd243709dbaec8c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 00:45:47 -0800 Subject: [PATCH 674/880] Some Python 3.9 fixes --- docs/release_notes.rst | 3 +++ setup.py | 3 ++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f80f318b..f9906f5f 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -15,6 +15,9 @@ wish to use some of its features for working with PDFs. v11.3.2 ======= +- Explicitly require pikepdf 2.0.0 or newer when running on Python 3.9. (There are + concerns about the stability of pybind11 2.5.x with Python 3.9, which is used in + pikepdf 1.x.) - On some systems, unpaper seems to be unable to process the PNGs we offer it as input. We now convert the input to PNM format, which unpaper always accepts. Fixes #665 and #667. diff --git a/setup.py b/setup.py index b941b26c..c4a1e576 100644 --- a/setup.py +++ b/setup.py @@ -73,7 +73,8 @@ setup( 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.5', # pure Python, so track HEAD closely 'pdfminer.six >= 20191110, != 20200720, <= 20201018', - 'pikepdf >= 1.14.0, < 3', + "pikepdf >= 1.14.0, < 3 ; python_version < '3.9'", + "pikepdf >= 2.0.0 ; python_version >= '3.9'", 'Pillow >= 7.0.0', 'pluggy >= 0.13.0, < 1.0', 'reportlab >= 3.3.0', # oldest released version with sane image handling From 54bbbfdeb3b6853f7aaa67082f73f6dbfcc9d00d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:08:14 -0800 Subject: [PATCH 675/880] Fix UnboundLocalError when considering ImageMasks for optimization Uncovered by test file in issue 667, although unrelated to that issue. --- src/ocrmypdf/optimize.py | 42 ++++++++++++++++++++++------------------ 1 file changed, 23 insertions(+), 19 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 16060ec6..cc9c8318 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -115,25 +115,29 @@ def extract_image_jbig2( and filtdp[0] != Name.JBIG2Decode and jbig2enc.available() ): - try: - # Save any colorspace associated with the image, so that we - # will export a pure 1-bit PNG with no palette or ICC profile. - # Showing the palette or ICC to jbig2enc will cause it to perform - # colorspace transform to 1bpp, which will conflict the palette or - # ICC if it exists. - colorspace = pim.obj.ColorSpace - # Set to DeviceGray temporarily; we already in 1 bpc. - pim.obj.ColorSpace = pikepdf.Name.DeviceGray - imgname = root / f'{xref:08d}' - with imgname.open('wb') as f: - ext = pim.extract_to(stream=f) - imgname.rename(imgname.with_suffix(ext)) - except pikepdf.UnsupportedImageTypeError: - return None - finally: - # Restore image colorspace after temporarily setting it to DeviceGray - pim.obj.ColorSpace = colorspace - return XrefExt(xref, ext) + # Save any colorspace associated with the image, so that we + # will export a pure 1-bit PNG with no palette or ICC profile. + # Showing the palette or ICC to jbig2enc will cause it to perform + # colorspace transform to 1bpp, which will conflict the palette or + # ICC if it exists. + colorspace = pim.obj.get(pikepdf.Name.ColorSpace, None) + if colorspace is not None or pim.image_mask: + try: + # Set to DeviceGray temporarily; we already in 1 bpc. + pim.obj.ColorSpace = pikepdf.Name.DeviceGray + imgname = root / f'{xref:08d}' + with imgname.open('wb') as f: + ext = pim.extract_to(stream=f) + imgname.rename(imgname.with_suffix(ext)) + except pikepdf.UnsupportedImageTypeError: + return None + finally: + # Restore image colorspace after temporarily setting it to DeviceGray + if colorspace is not None: + pim.obj.ColorSpace = colorspace + else: + del pim.obj.ColorSpace + return XrefExt(xref, ext) return None From ced7ad9164e195fedb7609b0771c2cf2eba5ac5e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:14:57 -0800 Subject: [PATCH 676/880] unpaper: round off DPI --- src/ocrmypdf/_exec/unpaper.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index f61132e6..d263bf9a 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -66,7 +66,7 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: def run(input_file, output_file, dpi, mode_args): - args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args + args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args with TemporaryDirectory() as tmpdir: input_pnm, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file) From 3707af3b74d64bb56eab5f5ac88655c1a259e429 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:30:31 -0800 Subject: [PATCH 677/880] Change pdf.root to pdf.Root --- src/ocrmypdf/pdfinfo/info.py | 8 ++++---- tests/test_metadata.py | 4 ++-- tests/test_validation.py | 4 ++-- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 44638abb..4c59f938 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -818,12 +818,12 @@ class PdfInfo: check_pages=check_pages, detailed_analysis=detailed_analysis, ) - self._needs_rendering = pdf.root.get('/NeedsRendering', False) + self._needs_rendering = pdf.Root.get('/NeedsRendering', False) self._has_acroform = False - if '/AcroForm' in pdf.root: - if len(pdf.root.AcroForm.get('/Fields', [])) > 0: + if '/AcroForm' in pdf.Root: + if len(pdf.Root.AcroForm.get('/Fields', [])) > 0: self._has_acroform = True - elif '/XFA' in pdf.root.AcroForm: + elif '/XFA' in pdf.Root.AcroForm: self._has_acroform = True @property diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 1580ba79..a41fcab1 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -302,8 +302,8 @@ def test_kodak_toc(resources, outpdf): p = pikepdf.open(outpdf) - if pikepdf.Name.First in p.root.Outlines: - assert isinstance(p.root.Outlines.First, pikepdf.Dictionary) + if pikepdf.Name.First in p.Root.Outlines: + assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary) def test_metadata_fixup_warning(resources, outdir, caplog): diff --git a/tests/test_validation.py b/tests/test_validation.py index 06481c05..ec8cf8dd 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -137,9 +137,9 @@ def test_report_file_size(tmp_path, caplog): caplog.clear() waste_of_space = b'Dummy' * 5000 - pdf.root.Dummy = waste_of_space + pdf.Root.Dummy = waste_of_space pdf.save(in_) - pdf.root.Dummy2 = waste_of_space + waste_of_space + pdf.Root.Dummy2 = waste_of_space + waste_of_space pdf.save(out) with patch('ocrmypdf._validation.jbig2enc.available', return_value=True), patch( From 36e9a54f02905788899916609fd6ccd488cc841e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:34:28 -0800 Subject: [PATCH 678/880] Remove extraneous page rotation This was added in commit b5ccbfd but seems to have been ill-advised. --- src/ocrmypdf/_pipeline.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 67bb26c8..aa55f5d4 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -536,9 +536,6 @@ def create_ocr_image(image: Path, page_context: PageContext): # Pillow requires integer DPI dpi = tuple(round(coord) for coord in im.info['dpi']) - if page_context.pageinfo.rotation != 0: - log.info(f"Rotating {page_context.pageinfo.rotation}") - im = im.rotate(page_context.pageinfo.rotation) im.save(output_file, dpi=dpi) return output_file From dd8a5a4c7230552219885955b2ca62d158b886f9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:44:35 -0800 Subject: [PATCH 679/880] Fix log domain names ocrmypdf.subprocess.subprocess.ghostscript -> ocrmypdf.subprocess.ghostscript --- src/ocrmypdf/subprocess.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index 68f6557c..fe8c9943 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -51,7 +51,7 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs): args = _fix_windows_args(program, args, env) log.debug("Running: %s", args) - process_log = log.getChild('subprocess.' + os.path.basename(program)) + process_log = log.getChild(os.path.basename(program)) if sys.version_info < (3, 7) and os.name == 'nt': # Can't use close_fds=True on Windows with Python 3.6 or older # https://bugs.python.org/issue19575, etc. From b913e5dfefa6e5be17857edd49abab6946de0bf9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 01:45:06 -0800 Subject: [PATCH 680/880] ghostscript: don't repeat log in debug Subprocess already does this for us. --- src/ocrmypdf/_exec/ghostscript.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index e2fa9726..bfc4f0e7 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -122,8 +122,6 @@ def rasterize_pdf( stderr = p.stderr.decode(errors='replace') if _gs_error_reported(stderr): log.error(stderr) - elif stderr: - log.debug(stderr) with Image.open(BytesIO(p.stdout)) as im: if rotation is not None: From d22a1b3367566258b0d4cb79d7075d9a9d7c59c2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 02:03:25 -0800 Subject: [PATCH 681/880] v11.3.2 release notes (2) Since we never tagged it, fix other things. --- docs/release_notes.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f9906f5f..3a65ffc9 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -18,9 +18,13 @@ v11.3.2 - Explicitly require pikepdf 2.0.0 or newer when running on Python 3.9. (There are concerns about the stability of pybind11 2.5.x with Python 3.9, which is used in pikepdf 1.x.) +- Fixed another issue related to page rotation. +- Fixed an issue where image marked as image masks were not properly considered + as optimization candidates. - On some systems, unpaper seems to be unable to process the PNGs we offer it as input. We now convert the input to PNM format, which unpaper always accepts. Fixes #665 and #667. +- DPI sent to unpaper is now rounded to a more reasonable number of decimal digits. - Debug and error messages from unpaper were being suppressed. - Some documentation tweaks. From 14a85f9473eff26abdea3b756c42846694d8d338 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 04:12:47 -0800 Subject: [PATCH 682/880] Fix pinned dependencies --- azure-pipelines.yml | 3 +-- requirements/main.txt | 14 +++++++------- 2 files changed, 8 insertions(+), 9 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 5fbb2cf6..4025bdc3 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -156,9 +156,8 @@ stages: # versionSpec: "$(python.version)" - bash: | brew update - brew unlink python@2 brew upgrade python - echo "Using Python `python3 --version`" + echo "Using `python3 --version`" displayName: "Update brew and Python" - bash: | brew install \ diff --git a/requirements/main.txt b/requirements/main.txt index 5247e369..fde1197e 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -1,12 +1,12 @@ # requirements.txt can be used to replicate the developer's build environment # setup.py lists a separate set of requirements that are looser to simplify # installation -cffi == 1.14.0 +cffi == 1.14.3 coloredlogs == 14.0 # technically optional -img2pdf == 0.3.6 -pdfminer.six == 20200517 -pikepdf == 1.16.1 +img2pdf == 0.4.0 +pdfminer.six == 20201018 +pikepdf == 2.0.0 pluggy == 0.13.1 -Pillow == 7.1.2 -reportlab == 3.5.42 -tqdm == 4.46.1 +Pillow == 8.0.1 +reportlab == 3.5.55 +tqdm == 4.51.0 From 13018d3d5c5dc049110b753c6bcc2b803e39d4f3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 04:15:14 -0800 Subject: [PATCH 683/880] ci: Extend test matrix to Python 3.9 --- azure-pipelines.yml | 82 ++++++++++++++++++++++++--------------------- 1 file changed, 43 insertions(+), 39 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index 4025bdc3..cd22364d 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -22,6 +22,8 @@ stages: python.version: "3.7" Python38: python.version: "3.8" + Python39: + python.version: "3.9" steps: - task: UsePythonVersion@0 inputs: @@ -59,45 +61,47 @@ stages: python.version: "3.7" Python38: python.version: "3.8" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" - - bash: | - sudo apt-get update - sudo apt-get install -y --no-install-recommends \ - python3-software-properties \ - curl \ - ghostscript \ - img2pdf \ - libexempi3 \ - libffi-dev \ - liblept5 \ - libsm6 libxext6 libxrender-dev \ - pngquant \ - poppler-utils \ - tesseract-ocr \ - tesseract-ocr-deu \ - tesseract-ocr-eng \ - unpaper \ - zlib1g - displayName: "Install system packages" - - bash: | - curl https://bootstrap.pypa.io/get-pip.py | python3 - pip3 install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - bash: | - tesseract --version - displayName: "Record versions" - - bash: | - # -n auto is slower on Linux and breaks on Python 3.8 - pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() + Python39: + python.version: "3.9" + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - bash: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + python3-software-properties \ + curl \ + ghostscript \ + img2pdf \ + libexempi3 \ + libffi-dev \ + liblept5 \ + libsm6 libxext6 libxrender-dev \ + pngquant \ + poppler-utils \ + tesseract-ocr \ + tesseract-ocr-deu \ + tesseract-ocr-eng \ + unpaper \ + zlib1g + displayName: "Install system packages" + - bash: | + curl https://bootstrap.pypa.io/get-pip.py | python3 + pip3 install -r requirements/main.txt -r requirements/test.txt . + displayName: "Install Python packages" + - bash: | + tesseract --version + displayName: "Record versions" + - bash: | + # -n auto is slower on Linux and breaks on Python 3.8 + pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() - job: "Ubuntu_1604" pool: vmImage: "ubuntu-16.04" From 6d5f8133e0fbc513b853eb65203ba98b1ac04f2f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 15:28:33 -0800 Subject: [PATCH 684/880] docs: show ifmain guard in example --- docs/api.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/api.rst b/docs/api.rst index 0c12932e..c0ff81d9 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -20,7 +20,8 @@ and largely have the same functions. import ocrmypdf - ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True) + if __name__ == '__main__': # To ensure correct behavior on Windows + ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True) With a few exceptions, all of the command line arguments are available and may be passed as equivalent keywords. From 5d1d1a712be22872adfee274de6832871a271195 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 17:09:58 -0800 Subject: [PATCH 685/880] docs: more details about macOS API changes Due to fork->spawn --- docs/api.rst | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index c0ff81d9..78428699 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -36,8 +36,9 @@ The :func:`ocrmypdf.ocr` function runs OCRmyPDF similar to command line execution. To do this, it will: - create a monitoring thread -- create worker processes (forking itself) -- manage the signal flags of worker processes +- create worker processes (on Linux, forking itself; on Windows and macOS, by + spawning) +- manage the signal flags of its worker processes - execute other subprocesses (forking and executing other programs) The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently @@ -48,9 +49,9 @@ There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker processes. -Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That +Creating a child process to call ``ocrmypdf.ocr()`` is suggested. That way your application will survive and remain interactive even if -OCRmyPDF does not. +OCRmyPDF fails for any reason. Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal handler (except on Windows), to raise an exception if access to a memory @@ -58,11 +59,10 @@ mapped file fails. OCRmyPDF may use memory mapping. .. warning:: - On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected - by an "ifmain" guard (``if __name__ == '__main__'``) or you must use - ``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one - of these steps, Windows process semantics will prevent OCRmyPDF from working - correctly. + On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be + protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do + not take at least one of these steps, process semantics will prevent + OCRmyPDF from working correctly. Logging ------- From 6d3f9ff15af22cc927f4eb9754bf6d3e52482f79 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 3 Nov 2020 17:10:52 -0800 Subject: [PATCH 686/880] api: rework ocr() slightly to simplify variable handling --- src/ocrmypdf/api.py | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 017a8506..d9f1c1a8 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -299,18 +299,18 @@ def ocr( # pylint: disable=unused-argument else: plugins = list(plugins) - parser = get_parser() - _plugin_manager = get_plugin_manager(plugins) - _plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member - - create_options_kwargs = { - k: v for k, v in locals().items() if not k.startswith('_') and k != 'kwargs' - } + # No new variable names should be assigned until these two steps are run + create_options_kwargs = {k: v for k, v in locals().items() if k != 'kwargs'} create_options_kwargs.update(kwargs) + parser = get_parser() + create_options_kwargs['parser'] = parser + plugin_manager = get_plugin_manager(plugins) + plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member + if 'verbose' in kwargs: warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().") options = create_options(**create_options_kwargs) - check_options(options, _plugin_manager) - return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True) + check_options(options, plugin_manager) + return run_pipeline(options=options, plugin_manager=plugin_manager, api=True) From b51abf22495aa0710f66b05d33b69958713b4022 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 4 Nov 2020 12:19:35 -0800 Subject: [PATCH 687/880] azure: Fix indentation mistake --- azure-pipelines.yml | 78 ++++++++++++++++++++++----------------------- 1 file changed, 39 insertions(+), 39 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index cd22364d..db05746c 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -63,45 +63,45 @@ stages: python.version: "3.8" Python39: python.version: "3.9" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" - - bash: | - sudo apt-get update - sudo apt-get install -y --no-install-recommends \ - python3-software-properties \ - curl \ - ghostscript \ - img2pdf \ - libexempi3 \ - libffi-dev \ - liblept5 \ - libsm6 libxext6 libxrender-dev \ - pngquant \ - poppler-utils \ - tesseract-ocr \ - tesseract-ocr-deu \ - tesseract-ocr-eng \ - unpaper \ - zlib1g - displayName: "Install system packages" - - bash: | - curl https://bootstrap.pypa.io/get-pip.py | python3 - pip3 install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - bash: | - tesseract --version - displayName: "Record versions" - - bash: | - # -n auto is slower on Linux and breaks on Python 3.8 - pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() + steps: + - task: UsePythonVersion@0 + inputs: + versionSpec: "$(python.version)" + - bash: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + python3-software-properties \ + curl \ + ghostscript \ + img2pdf \ + libexempi3 \ + libffi-dev \ + liblept5 \ + libsm6 libxext6 libxrender-dev \ + pngquant \ + poppler-utils \ + tesseract-ocr \ + tesseract-ocr-deu \ + tesseract-ocr-eng \ + unpaper \ + zlib1g + displayName: "Install system packages" + - bash: | + curl https://bootstrap.pypa.io/get-pip.py | python3 + pip3 install -r requirements/main.txt -r requirements/test.txt . + displayName: "Install Python packages" + - bash: | + tesseract --version + displayName: "Record versions" + - bash: | + # -n auto is slower on Linux and breaks on Python 3.8 + pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml + displayName: "Test" + - task: PublishTestResults@2 + inputs: + testResultsFiles: "test.xml" + testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" + condition: succeededOrFailed() - job: "Ubuntu_1604" pool: vmImage: "ubuntu-16.04" From 5a59e4d5432e40537846c3bfb1f81f8aed9e3598 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 7 Nov 2020 00:18:27 -0800 Subject: [PATCH 688/880] unpaper: don't use universal_newlines=True There's no specific reason to do this. We can log binary output equally well. --- src/ocrmypdf/_exec/unpaper.py | 1 - 1 file changed, 1 deletion(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index d263bf9a..e17ebb12 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -83,7 +83,6 @@ def run(input_file, output_file, dpi, mode_args): args_unpaper, close_fds=True, check=True, - universal_newlines=True, stderr=STDOUT, # unpaper writes logging output to stdout and stderr stdout=PIPE, # and cannot send file output to stdout cwd=tmpdir, From 895fddd85e449273709417a14677ec68d119f09a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 7 Nov 2020 00:48:08 -0800 Subject: [PATCH 689/880] Replace most uses of universal_newlines with text The parameters are equivalent but the latter is better named. Since Python 3.6 doesn't support text= we use our wrapper to add it in that place. This is for subprocess.run. --- src/ocrmypdf/_exec/tesseract.py | 4 +--- src/ocrmypdf/subprocess.py | 15 ++++++++++----- tests/conftest.py | 4 ++-- tests/test_main.py | 4 ++-- tests/test_rotation.py | 2 +- 5 files changed, 16 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 3be2d36e..5d85f838 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -99,9 +99,7 @@ def get_languages(): args_tess = ['tesseract', '--list-langs'] try: - proc = run( - args_tess, universal_newlines=True, stdout=PIPE, stderr=STDOUT, check=True - ) + proc = run(args_tess, text=True, stdout=PIPE, stderr=STDOUT, check=True) output = proc.stdout except CalledProcessError as e: raise MissingDependencyError(lang_error(e.output)) from e diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index fe8c9943..f218fccf 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -52,10 +52,15 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs): log.debug("Running: %s", args) process_log = log.getChild(os.path.basename(program)) - if sys.version_info < (3, 7) and os.name == 'nt': - # Can't use close_fds=True on Windows with Python 3.6 or older - # https://bugs.python.org/issue19575, etc. - kwargs['close_fds'] = False + if sys.version_info < (3, 7): + if os.name == 'nt': + # Can't use close_fds=True on Windows with Python 3.6 or older + # https://bugs.python.org/issue19575, etc. + kwargs['close_fds'] = False + if 'text' in kwargs: + # Convert run(...text=) to run(...universal_newlines=) for Python 3.6 + kwargs['universal_newlines'] = kwargs['text'] + del kwargs['text'] stderr = None stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout' @@ -119,7 +124,7 @@ def get_version( proc = run( args_prog, close_fds=True, - universal_newlines=True, + text=True, stdout=PIPE, stderr=STDOUT, check=True, diff --git a/tests/conftest.py b/tests/conftest.py index d5bda6cd..7619769f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -128,7 +128,7 @@ def run_ocrmypdf_api(input_file, output_file, *args): @pytest.helpers.register -def run_ocrmypdf(input_file, output_file, *args, universal_newlines=True): +def run_ocrmypdf(input_file, output_file, *args, text=True): "Run ocrmypdf and let caller deal with results" p_args = ( @@ -151,7 +151,7 @@ def run_ocrmypdf(input_file, output_file, *args, universal_newlines=True): p_args, stdout=PIPE, stderr=PIPE, - universal_newlines=universal_newlines, + universal_newlines=text, # When dropping support for Python 3.6 change to text= env=env, check=False, ) diff --git a/tests/test_main.py b/tests/test_main.py index 5a660e9c..47a9cb40 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -598,7 +598,7 @@ def test_compression_preserved(ocrmypdf_exec, resources, image, outpdf): stdout=PIPE, stderr=PIPE, stdin=input_stream, - universal_newlines=True, + universal_newlines=True, # When dropping support for Python 3.6 change to text= check=False, ) @@ -659,7 +659,7 @@ def test_compression_changed(ocrmypdf_exec, resources, image, compression, outpd stdout=PIPE, stderr=PIPE, stdin=input_stream, - universal_newlines=True, + universal_newlines=True, # When dropping support for Python 3.6 change to text= check=False, ) assert p.returncode == ExitCode.ok, p.stderr diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 090d5b32..bb699124 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -241,7 +241,7 @@ def test_rotate_page_level(image_angle, page_angle, resources, outdir): '--rotate-pages', '--rotate-pages-threshold', '0.001', - universal_newlines=False, + text=False, ) err = err.decode('utf-8', errors='replace') assert p.returncode == 0, err From 71f0e7f545f754cdc37c0551be703c98dbd05abf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 7 Nov 2020 00:53:33 -0800 Subject: [PATCH 690/880] v11.3.3 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 3a65ffc9..17ff8425 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.3.3 +======= + +- If unpaper outputs non-UTF-8 data, quietly fix this rather than choke on the + conversion. (Possibly addresses #671.) + v11.3.2 ======= From 4fc7d6d93e75eda67258b7a4f3f839050552e8bf Mon Sep 17 00:00:00 2001 From: pretentious7 Date: Mon, 9 Nov 2020 19:53:02 -0500 Subject: [PATCH 691/880] fix typo "charcter" -> "character" (#673) --- docs/index.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/index.rst b/docs/index.rst index 217f5411..91eac433 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -1,7 +1,7 @@ OCRmyPDF documentation ====================== -OCRmyPDF adds an optical charcter recognition (OCR) text layer to scanned PDF +OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF files, allowing them to be searched. PDF is the best format for storing and exchanging scanned documents. From 22cd9b236485e3004f40fdaaac81c919f5de8431 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 10 Nov 2020 04:07:49 -0800 Subject: [PATCH 692/880] docs: fix csv-table errors --- docs/batch.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/batch.rst b/docs/batch.rst index 99c3e2ae..63f6e41b 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -127,7 +127,7 @@ Users may need to customize the script to meet their requirements. "OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)" "OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``" "OCR_DESKEW", "Apply deskew to crooked input PDFs" - "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={"rotate_pages": true}'``. + "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``." "OCR_POLL_NEW_FILE_SECONDS", "Polling interval" "OCR_LOGLEVEL", "Level of log messages to report" From a03863a17d449eee7843ffbd77663213a98575aa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 10 Nov 2020 04:08:01 -0800 Subject: [PATCH 693/880] docs: fix link to docker image --- docs/docker.rst | 2 ++ docs/installation.rst | 4 ++-- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index c38a63b9..788efbcd 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -1,3 +1,5 @@ +.. _docker: + ===================== OCRmyPDF Docker image ===================== diff --git a/docs/installation.rst b/docs/installation.rst index b1c54559..84520f6e 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -604,7 +604,7 @@ However, the OCR-to-text-layer functionality is available. Docker ------ -You can also :ref:`Install the Docker ` container on Windows. Ensure that +You can also :ref:`Install the Docker ` container on Windows. Ensure that your command prompt can run the docker "hello world" container. Installing on FreeBSD @@ -630,7 +630,7 @@ Installing the Docker image For some users, installing the Docker image will be easier than installing all of OCRmyPDF's dependencies. -See `OCRmyPDF Docker Image `__ for more information. +See :ref:`docker` for more information. Installing with Python pip ========================== From 5c56f6120923547b35947601f085357499dc297c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 11 Nov 2020 02:59:37 -0800 Subject: [PATCH 694/880] unpaper: type hints --- src/ocrmypdf/_exec/unpaper.py | 24 +++++++++++++++++------- src/ocrmypdf/_plugin_manager.py | 2 +- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index e17ebb12..5798ed55 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -13,10 +13,11 @@ import logging import os import shlex +from decimal import Decimal from pathlib import Path -from subprocess import PIPE, STDOUT, CalledProcessError +from subprocess import PIPE, STDOUT from tempfile import TemporaryDirectory -from typing import Tuple +from typing import List, Optional, Tuple, Union from PIL import Image @@ -24,10 +25,12 @@ from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError from ocrmypdf.subprocess import get_version from ocrmypdf.subprocess import run as external_run +DecFloat = Union[Decimal, float] + log = logging.getLogger(__name__) -def version(): +def version() -> str: return get_version('unpaper') @@ -53,7 +56,7 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: except KeyError: raise MissingDependencyError( "Failed to convert image to a supported format." - ) from e + ) from None if im_modified or input_file.suffix != '.pnm': input_pnm = tmpdir / 'input.pnm' @@ -65,7 +68,9 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: return input_pnm, output_pnm -def run(input_file, output_file, dpi, mode_args): +def run( + input_file: Path, output_file: Path, dpi: DecFloat, mode_args: List[str] +) -> None: args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args with TemporaryDirectory() as tmpdir: @@ -99,14 +104,19 @@ def run(input_file, output_file, dpi, mode_args): ) from None -def validate_custom_args(args: str): +def validate_custom_args(args: str) -> List[str]: unpaper_args = shlex.split(args) if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args): raise ValueError('No filenames allowed in --unpaper-args') return unpaper_args -def clean(input_file, output_file, dpi, unpaper_args=None): +def clean( + input_file: Path, + output_file: Path, + dpi: DecFloat, + unpaper_args: Optional[List[str]] = None, +): default_args = [ '--layout', 'none', diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 3e65d829..e939e064 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -82,7 +82,7 @@ def _setup_plugins( pm.register(module) -def get_plugin_manager(plugins: List[str], builtins=True): +def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( project_name='ocrmypdf', setup_func=partial(_setup_plugins, plugins=plugins, builtins=builtins), From d0cdbd5e1c5cda68ba6dac11effc03449d00cb5a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 12 Nov 2020 02:29:47 -0800 Subject: [PATCH 695/880] watcher: include uppercase .PDF too --- misc/watcher.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/watcher.py b/misc/watcher.py index d5d583f3..c6d6163d 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -45,7 +45,7 @@ OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) USE_POLLING = bool(os.getenv('OCR_USE_POLLING', '')) LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() -PATTERNS = ['*.pdf'] +PATTERNS = ['*.pdf', '*.PDF'] log = logging.getLogger('ocrmypdf-watcher') From 1f598da3c168c6dcfd39318b1082ab883d23abbb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Nov 2020 11:34:17 -0800 Subject: [PATCH 696/880] ghostscript: better docs and comments --- src/ocrmypdf/_exec/ghostscript.py | 32 +++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index bfc4f0e7..0644e0d3 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -25,24 +25,24 @@ from ocrmypdf.subprocess import get_version, run log = logging.getLogger(__name__) +missing_gs_error = """ +--------------------------------------------------------------------- +This error normally occurs when ocrmypdf find can't Ghostscript. +Please ensure Ghostscript is installed and its location is added to +the system PATH environment variable. + +For details see: + https://ocrmypdf.readthedocs.io/en/latest/installation.html +--------------------------------------------------------------------- +""" + _gswin = None if os.name == 'nt': _gswin = which('gswin64c') if not _gswin: _gswin = which('gswin32c') if not _gswin: - raise MissingDependencyError( - """ - --------------------------------------------------------------------- - This error normally occurs when ocrmypdf can't Ghostscript. Please - ensure Ghostscript is installed and its location is added to the - system PATH environment variable. - - For details see: - https://ocrmypdf.readthedocs.io/en/latest/installation.html - --------------------------------------------------------------------- - """ - ) + raise MissingDependencyError(missing_gs_error) _gswin = Path(_gswin).stem GS = _gswin if _gswin else 'gs' @@ -146,6 +146,9 @@ def generate_pdfa( pdf_version: str = '1.5', pdfa_part: str = '2', ): + # Ghostscript's compression is all or nothing. We can either force all images + # to JPEG, force all to Flate/PNG, or let it decide how to encode the images. + # In most case it's best to let it decide. compression_args = [] if compression == 'jpeg': compression_args = [ @@ -173,8 +176,9 @@ def generate_pdfa( strategy = 'RGB' if version() >= '9.19' else '/RGB' if version() == '9.23': - # 9.23: new feature JPEG passthrough is broken in some cases, best to - # disable it always + # 9.23: added JPEG passthrough as a new feature, but with a bug that + # incorrectly formats some images. Fixed as of 9.24. So we disable this + # feature for 9.23. # https://bugs.ghostscript.com/show_bug.cgi?id=699216 compression_args.append('-dPassThroughJPEGImages=false') From d71e50e83d95892b111cded61eea380deb83901b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Nov 2020 11:52:17 -0800 Subject: [PATCH 697/880] Fix "readLinearizationData for file that is not linearized" pikepdf 2.1.0 throws wrong type of exception in this case, so special-case it. Closes #680 Closes #681 --- src/ocrmypdf/_sync.py | 7 +------ src/ocrmypdf/helpers.py | 10 ++++++++++ 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 2525fe05..a24c05e7 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -55,6 +55,7 @@ from ocrmypdf._validation import ( ) from ocrmypdf.exceptions import ExitCode, ExitCodeException from ocrmypdf.helpers import ( + NeverRaise, available_cpu_count, check_pdf, pikepdf_enable_mmap, @@ -301,12 +302,6 @@ def exec_concurrent(context: PdfContext): copy_final(pdf, options.output_file, context) -class NeverRaise(Exception): - """An exception that is never raised""" - - pass # pylint: disable=unnecessary-pass - - def configure_debug_logging(log_filename, prefix: str = ''): """ Create a debug log file at a specified location. diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 42d5e725..b26e4021 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -58,6 +58,10 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))): return f"Resolution({self.x}x{self.y} dpi)" +class NeverRaise(Exception): + """An exception that is never raised""" + + def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): """ Helper function: relinks soft symbolic link if necessary @@ -191,6 +195,12 @@ def check_pdf(input_file: Path) -> bool: pdf.check_linearization(sio) except RuntimeError: pass + except ( + getattr(pikepdf, 'ForeignObjectError') + if pikepdf.__version__ == '2.1.0' # This version may throw wrong exception + else NeverRaise + ): + pass else: linearize = sio.getvalue() if linearize: From 43f41863fa55a4708815552cbcd76e8bdad36983 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Nov 2020 11:54:07 -0800 Subject: [PATCH 698/880] check_pdf: document how we handle linearization --- src/ocrmypdf/helpers.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index b26e4021..3a2c9051 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -190,8 +190,10 @@ def check_pdf(input_file: Path) -> bool: log.warning(msg) sio = StringIO() - linearize = None + linearize_msgs = '' try: + # If linearization is missing entirely, we do not complain. We do + # complain if linearization is present but incorrect. pdf.check_linearization(sio) except RuntimeError: pass @@ -202,11 +204,11 @@ def check_pdf(input_file: Path) -> bool: ): pass else: - linearize = sio.getvalue() - if linearize: - log.warning(linearize) + linearize_msgs = sio.getvalue() + if linearize_msgs: + log.warning(linearize_msgs) - if not messages and not linearize: + if not messages and not linearize_msgs: return True return False finally: From a2bbbe2a26421fcf4c1326be2e084811fbfc55bd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Nov 2020 11:56:29 -0800 Subject: [PATCH 699/880] v11.3.4 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 17ff8425..8efb37c6 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.3.4 +======= + +- Fixed an error message 'called readLinearizationData for file that is not + linearized' that may occur when pikepdf 2.1.0 is used. (Upgrading to pikepdf + 2.1.1 also fixes the issue.) + v11.3.3 ======= From 8224d89bc6a225fcefae2f1f039d462a0fe37bd2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 18 Nov 2020 11:57:28 -0800 Subject: [PATCH 700/880] v11.3.4 release notes --- docs/release_notes.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 8efb37c6..4037fb60 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -18,6 +18,9 @@ v11.3.4 - Fixed an error message 'called readLinearizationData for file that is not linearized' that may occur when pikepdf 2.1.0 is used. (Upgrading to pikepdf 2.1.1 also fixes the issue.) +- File watcher now automatically includes ``.PDF`` in addition to ``.pdf`` to + better support case sensitive file systems. +- Some documentation and comment improvements. v11.3.3 ======= From 0cdb9bd04a5ba11bd6b2bfb3c08b695773450c17 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 23 Nov 2020 12:36:04 -0800 Subject: [PATCH 701/880] docs: remove description of how OMP_THREAD_LIMIT is managed --- docs/advanced.rst | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/docs/advanced.rst b/docs/advanced.rst index 9c30f1c5..4d72c776 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -125,8 +125,7 @@ include: .. envvar:: OMP_THREAD_LIMIT Controls the number of threads Tesseract will use. OCRmyPDF will - manage this environment if it is not already set. (Currently, it will - set it to 1 because this gives the best results in testing.) + manage this environment variable if it is not already set. For example, if you have a development build of Tesseract don't wish to use the system installation, you can launch OCRmyPDF as follows: From f0e7bea8ba7b579592342c6c95d0c28011650a86 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 27 Nov 2020 13:54:36 -0800 Subject: [PATCH 702/880] docs: remove redundant statement --- docs/api.rst | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index 78428699..a93c4a9c 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -20,7 +20,7 @@ and largely have the same functions. import ocrmypdf - if __name__ == '__main__': # To ensure correct behavior on Windows + if __name__ == '__main__': # To ensure correct behavior on Windows and macOS ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True) With a few exceptions, all of the command line arguments are available @@ -42,8 +42,7 @@ execution. To do this, it will: - execute other subprocesses (forking and executing other programs) The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently -privileged to perform these actions. If it is not, ``ocrmypdf()`` will -fail. +privileged to perform these actions. There is no currently no option to manage how jobs are scheduled other than the argument ``jobs=`` which will limit the number of worker From 80e957908a3ebb8f627a8a0e6949e801f348f83c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Nov 2020 14:25:46 -0800 Subject: [PATCH 703/880] tesseract: fix run call with logs_errors_to_stdout --- src/ocrmypdf/_exec/tesseract.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 5d85f838..171bf306 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -99,7 +99,14 @@ def get_languages(): args_tess = ['tesseract', '--list-langs'] try: - proc = run(args_tess, text=True, stdout=PIPE, stderr=STDOUT, check=True) + proc = run( + args_tess, + text=True, + stdout=PIPE, + stderr=STDOUT, + logs_errors_to_stdout=True, + check=True, + ) output = proc.stdout except CalledProcessError as e: raise MissingDependencyError(lang_error(e.output)) from e From b83d7f6d1aa9c9cc27697a1e05e7d9573aefda84 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Nov 2020 14:36:03 -0800 Subject: [PATCH 704/880] subprocess: refactor and add run_polling_stderr --- src/ocrmypdf/subprocess.py | 81 ++++++++++++++++++++++++++++---------- 1 file changed, 60 insertions(+), 21 deletions(-) diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess.py index f218fccf..1d98a747 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess.py @@ -17,7 +17,7 @@ from contextlib import suppress from distutils.version import LooseVersion from functools import lru_cache from pathlib import Path -from subprocess import PIPE, STDOUT, CalledProcessError +from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen from subprocess import run as subprocess_run from ocrmypdf.exceptions import MissingDependencyError @@ -41,26 +41,7 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs): if there is an error. If False, stderr is logged. Could be used with stderr=STDOUT, stdout=PIPE for example. """ - if not env: - env = os.environ - - # Search in spoof path if necessary - program = args[0] - - if os.name == 'nt': - args = _fix_windows_args(program, args, env) - - log.debug("Running: %s", args) - process_log = log.getChild(os.path.basename(program)) - if sys.version_info < (3, 7): - if os.name == 'nt': - # Can't use close_fds=True on Windows with Python 3.6 or older - # https://bugs.python.org/issue19575, etc. - kwargs['close_fds'] = False - if 'text' in kwargs: - # Convert run(...text=) to run(...universal_newlines=) for Python 3.6 - kwargs['universal_newlines'] = kwargs['text'] - del kwargs['text'] + args, env, process_log, _text = _fix_process_args(args, env, kwargs) stderr = None stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout' @@ -82,6 +63,64 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs): return proc +def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs): + """Run a process like ``ocrmypdf.subprocess.run``, and poll stderr. + + Every line of produced by stderr will be forwarded to the callback function. + The intended use is monitoring progress of subprocesses that output their + own progress indicators. In addition, each line will be logged if debug + logging is enabled. + + Requires stderr to be opened in text mode for ease of handling errors. In + addition the expected encoding= and errors= arguments should be set. Note + that if stdout is already set up, it need not be binary. + """ + args, env, process_log, text = _fix_process_args(args, env, kwargs) + assert text, "Must use text=True" + + proc = Popen(args, env=env, **kwargs) + + lines = [] + while proc.poll() is None: + for msg in iter(proc.stderr.readline, ''): + if process_log.isEnabledFor(logging.DEBUG): + process_log.debug(msg.strip()) + callback(msg) + lines.append(msg) + stderr = ''.join(lines) + + if check and proc.returncode != 0: + raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr) + return CompletedProcess(args, proc.returncode, None, stderr=stderr) + + +def _fix_process_args(args, env, kwargs): + assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines" + + if not env: + env = os.environ + + # Search in spoof path if necessary + program = args[0] + + if os.name == 'nt': + args = _fix_windows_args(program, args, env) + + log.debug("Running: %s", args) + process_log = log.getChild(os.path.basename(program)) + text = kwargs.get('text', False) + if sys.version_info < (3, 7): + if os.name == 'nt': + # Can't use close_fds=True on Windows with Python 3.6 or older + # https://bugs.python.org/issue19575, etc. + kwargs['close_fds'] = False + if 'text' in kwargs: + # Convert run(...text=) to run(...universal_newlines=) for Python 3.6 + kwargs['universal_newlines'] = kwargs['text'] + del kwargs['text'] + return args, env, process_log, text + + def _fix_windows_args(program, args, env): """Adjust our desired program and command line arguments for use on Windows""" From 7e1223c12c295152ddb245e8e51a870f1f5659f0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 29 Nov 2020 14:53:35 -0800 Subject: [PATCH 705/880] ghostscript: add output tracing --- src/ocrmypdf/_exec/ghostscript.py | 43 +++++++++++++++++++++++++----- tests/plugins/gs_pdfa_failure.py | 6 ++--- tests/plugins/gs_render_failure.py | 2 +- 3 files changed, 41 insertions(+), 10 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 0644e0d3..bbf13277 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -18,10 +18,11 @@ from subprocess import PIPE, CalledProcessError from typing import Optional, cast from PIL import Image +from tqdm import tqdm from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError from ocrmypdf.helpers import Resolution -from ocrmypdf.subprocess import get_version, run +from ocrmypdf.subprocess import get_version, run, run_polling_stderr log = logging.getLogger(__name__) @@ -139,6 +140,27 @@ def rasterize_pdf( im.save(fspath(output_file), dpi=page_dpi) +class GhostscriptFollower: + re_process = re.compile(r"Processing pages \d+ through (\d+).") + re_page = re.compile(r"Page (\d+)") + + def __init__(self): + self.count = 0 + self.tqdm = None + + def __call__(self, line): + if not self.tqdm: + m = self.re_process.match(line.strip()) + if m: + self.count = int(m.group(1)) + self.tqdm = tqdm(total=self.count, desc="Ghostscript", unit='page') + return + else: + m = self.re_page.match(line.strip()) + if m: + self.tqdm.update() + + def generate_pdfa( pdf_pages, output_file: os.PathLike, @@ -188,7 +210,6 @@ def generate_pdfa( args_gs = ( [ GS, - "-dQUIET", "-dBATCH", "-dNOPAUSE", "-dSAFER", @@ -208,16 +229,26 @@ def generate_pdfa( ] ) args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs + try: with Path(output_file).open('wb') as output: - p = run(args_gs, stdout=output, stderr=PIPE, check=True) + p = run_polling_stderr( + args_gs, + stdout=output, + stderr=PIPE, + check=True, + text=True, + encoding='utf-8', + errors='replace', + callback=GhostscriptFollower(), + ) except CalledProcessError as e: # Ghostscript does not change return code when it fails to create # PDF/A - check PDF/A status elsewhere - log.error(e.stderr.decode(errors='replace')) - raise SubprocessOutputError('Ghostscript PDF/A rendering failed') + log.error(e.stderr) + raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e else: - stderr = p.stderr.decode('utf-8', errors='replace') + stderr = p.stderr if _gs_error_reported(stderr): last_part = None repcount = 0 diff --git a/tests/plugins/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py index dcad94f6..8a694de9 100644 --- a/tests/plugins/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -23,7 +23,7 @@ from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.subprocess import run +from ocrmypdf.subprocess import run_polling_stderr def run_rig_args(args, **kwargs): @@ -33,13 +33,13 @@ def run_rig_args(args, **kwargs): new_args = [ arg for arg in args if not arg.startswith('-dPDFA') and not arg.endswith('.ps') ] - proc = run(new_args, **kwargs) + proc = run_polling_stderr(new_args, **kwargs) return proc @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf._exec.ghostscript.run', new=run_rig_args): + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=run_rig_args): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index e3cee162..2cad1f4e 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -34,7 +34,7 @@ def raise_gs_fail(*args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail): + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=raise_gs_fail): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, From ce0e0ecd4d8018d361d3eac33cab51e0c959cf7b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 1 Dec 2020 12:09:11 -0800 Subject: [PATCH 706/880] Decouple tqdm from progressbar setup --- src/ocrmypdf/_exec/ghostscript.py | 19 ++++++++++++------- src/ocrmypdf/_pipeline.py | 2 ++ src/ocrmypdf/builtin_plugins/ghostscript.py | 11 ++++++++++- src/ocrmypdf/pluginspec.py | 14 +++++++++++++- tests/plugins/gs_feature_elision.py | 1 + tests/plugins/gs_pdfa_failure.py | 1 + tests/plugins/gs_render_failure.py | 1 + 7 files changed, 40 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index bbf13277..d96cd3c3 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -18,7 +18,6 @@ from subprocess import PIPE, CalledProcessError from typing import Optional, cast from PIL import Image -from tqdm import tqdm from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError from ocrmypdf.helpers import Resolution @@ -144,21 +143,26 @@ class GhostscriptFollower: re_process = re.compile(r"Processing pages \d+ through (\d+).") re_page = re.compile(r"Page (\d+)") - def __init__(self): + def __init__(self, progressbar_class): self.count = 0 - self.tqdm = None + self.progressbar_class = progressbar_class + self.progressbar = None def __call__(self, line): - if not self.tqdm: + if not self.progressbar_class: + return + if not self.progressbar: m = self.re_process.match(line.strip()) if m: self.count = int(m.group(1)) - self.tqdm = tqdm(total=self.count, desc="Ghostscript", unit='page') + self.progressbar = self.progressbar_class( + total=self.count, desc="PDF/A conversion", unit='page' + ) return else: m = self.re_page.match(line.strip()) if m: - self.tqdm.update() + self.progressbar.update() def generate_pdfa( @@ -167,6 +171,7 @@ def generate_pdfa( compression: str, pdf_version: str = '1.5', pdfa_part: str = '2', + progressbar_class=None, ): # Ghostscript's compression is all or nothing. We can either force all images # to JPEG, force all to Flate/PNG, or let it decide how to encode the images. @@ -240,7 +245,7 @@ def generate_pdfa( text=True, encoding='utf-8', errors='replace', - callback=GhostscriptFollower(), + callback=GhostscriptFollower(progressbar_class), ) except CalledProcessError as e: # Ghostscript does not change return code when it fails to create diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index aa55f5d4..76223ac6 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -19,6 +19,7 @@ import img2pdf import pikepdf from pikepdf.models.metadata import encode_pdf_date from PIL import Image, ImageColor, ImageDraw +from tqdm import tqdm from ocrmypdf import leptonica from ocrmypdf._exec import unpaper @@ -709,6 +710,7 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext): output_file=output_file, compression=options.pdfa_image_compression, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 + progressbar_class=tqdm if options.progress_bar else None, ) return output_file diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 3be822d0..d1c6a628 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -79,12 +79,21 @@ def rasterize_pdf_page( @hookimpl -def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): +def generate_pdfa( + pdf_pages, + pdfmark, + output_file, + compression, + pdf_version, + pdfa_part, + progressbar_class, +): ghostscript.generate_pdfa( pdf_pages=[*pdf_pages, pdfmark], output_file=output_file, compression=compression, pdf_version=pdf_version, pdfa_part=pdfa_part, + progressbar_class=progressbar_class, ) return output_file diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index b34e467f..9aee6596 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -59,7 +59,7 @@ def check_options(options: Namespace) -> None: Note: This hook will be called from the main process, and may modify global state before child worker processes are forked. - """ + """ @hookspec @@ -280,6 +280,7 @@ def generate_pdfa( compression: str, pdf_version: str, pdfa_part: str, + progressbar_class, ) -> Path: """Generate a PDF/A. @@ -302,10 +303,21 @@ def generate_pdfa( At its own discretion, the PDF/A generator may raise the version, but should not lower it. pdfa_part: The desired PDF/A compliance level, such as ``'2B'``. + progressbar_class: The class of a progress bar with a tqdm-like API. An + instance of this class will be initialized when PDF/A conversion + begins, using + ``instance = progressbar_class(total: int, desc: str, unit:str)``, + defining the number of work units, a user-visible description, + and the name of the work units ("page"). Then ``instance.update()`` + will be called when a work unit is completed. If ``None``, no + progress information is reported. Returns: Path: If successful, the hook should return ``output_file``. Note: This is a :ref:`firstresult hook`. + + See also: + https://github.com/tqdm/tqdm """ diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index 419855cb..98e78086 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -45,5 +45,6 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf compression=compression, pdf_version=pdf_version, pdfa_part=pdfa_part, + progressbar_class=None, ) return output_file diff --git a/tests/plugins/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py index 8a694de9..8973dcaf 100644 --- a/tests/plugins/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -47,5 +47,6 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf compression=compression, pdf_version=pdf_version, pdfa_part=pdfa_part, + progressbar_class=None, ) return output_file diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index 2cad1f4e..dcbc6c5a 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -42,5 +42,6 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf compression=compression, pdf_version=pdf_version, pdfa_part=pdfa_part, + progressbar_class=None, ) return output_file From ed5e17d0a40c2429da9f694eb011adc41a9c56c6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 1 Dec 2020 13:31:26 -0800 Subject: [PATCH 707/880] completions: consider *.PDF and some images too --- misc/completion/ocrmypdf.fish | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index fd3d41a8..31e30b2a 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -135,4 +135,4 @@ complete -c ocrmypdf -r -l user-words -d "specify location of user words file" complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file" complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF" -complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)" +complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf; __fish_complete_suffix .PDF; __fish_complete_suffix .jpg; __fish_complete_suffix .png)" From 3cba50bfbd6f63b29ab9592f8da855586444598b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 2 Dec 2020 16:00:12 -0800 Subject: [PATCH 708/880] windows: look in registry for Tesseract and Ghostscript --- src/ocrmypdf/leptonica.py | 11 +- .../{subprocess.py => subprocess/__init__.py} | 56 +----- src/ocrmypdf/subprocess/_windows.py | 162 ++++++++++++++++++ tests/test_helpers.py | 14 +- 4 files changed, 179 insertions(+), 64 deletions(-) rename src/ocrmypdf/{subprocess.py => subprocess/__init__.py} (84%) create mode 100644 src/ocrmypdf/subprocess/_windows.py diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 336c00ff..af69129f 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -27,15 +27,16 @@ from tempfile import TemporaryFile from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.lib._leptonica import ffi -from ocrmypdf.subprocess import shim_paths_with_program_files # pylint: disable=protected-access logger = logging.getLogger(__name__) if os.name == 'nt': + from ocrmypdf.subprocess._windows import shim_env_path + libname = 'liblept-5' - os.environ['PATH'] = shim_paths_with_program_files() + os.environ['PATH'] = shim_env_path() else: libname = 'lept' _libpath = find_library(libname) @@ -58,9 +59,9 @@ if not _libpath: --------------------------------------------------------------------- """ ) -if os.name == 'nt': - # On Windows, recent versions of libpng require zlib. We have to make sure - # the zlib version being loaded is the same one that libpng was built with. +if os.name == 'nt': + # On Windows, recent versions of libpng require zlib. We have to make sure + # the zlib version being loaded is the same one that libpng was built with. # This tries to import zlib from Tesseract's installation folder, falling back # to find_library() if liblept is being loaded from somewhere else. # Loading zlib from other places could cause a version mismatch diff --git a/src/ocrmypdf/subprocess.py b/src/ocrmypdf/subprocess/__init__.py similarity index 84% rename from src/ocrmypdf/subprocess.py rename to src/ocrmypdf/subprocess/__init__.py index 1d98a747..e68f8b0c 100644 --- a/src/ocrmypdf/subprocess.py +++ b/src/ocrmypdf/subprocess/__init__.py @@ -10,7 +10,6 @@ import logging import os import re -import shutil import sys from collections.abc import Mapping from contextlib import suppress @@ -104,7 +103,9 @@ def _fix_process_args(args, env, kwargs): program = args[0] if os.name == 'nt': - args = _fix_windows_args(program, args, env) + from ocrmypdf.subprocess._windows import fix_windows_args + + args = fix_windows_args(program, args, env) log.debug("Running: %s", args) process_log = log.getChild(os.path.basename(program)) @@ -121,30 +122,6 @@ def _fix_process_args(args, env, kwargs): return args, env, process_log, text -def _fix_windows_args(program, args, env): - """Adjust our desired program and command line arguments for use on Windows""" - - if sys.version_info < (3, 8): - # bpo-33617 - Windows needs manual Path -> str conversion - args = [os.fspath(arg) for arg in args] - program = os.fspath(program) - - # If we are running a .py on Windows, ensure we call it with this Python - # (to support test suite shims) - if program.lower().endswith('.py'): - args = [sys.executable] + args - - paths = os.pathsep.join(os.get_exec_path(env)) - if not shutil.which(args[0], path=paths): - # If the program we want is not on the PATH, add some interesting - # locations in %PROGRAMFILES% to the PATH and try again - shimmed_path = shim_paths_with_program_files(env) - new_args0 = shutil.which(args[0], path=shimmed_path) - if new_args0: - args[0] = new_args0 - return args - - @lru_cache(maxsize=None) def get_version( program: str, *, version_arg: str = '--version', regex=r'(\d+(\.\d+)*)', env=None @@ -193,33 +170,6 @@ def get_version( return version -def shim_paths_with_program_files(env=None): - if not env: - env = os.environ - program_files = env.get('PROGRAMFILES', '') - if not program_files: - return env.get('PATH', '') - - def path_walker(): - for path in Path(program_files).iterdir(): - if not path.is_dir(): - continue - if path.name.lower() == 'tesseract-ocr': - yield path - elif path.name.lower() == 'gs': - yield from (p for p in path.glob('**/bin') if p.is_dir()) - - paths = sorted( - (p for p in path_walker()), key=lambda p: (p.name, p.parent.name), reverse=True - ) - paths.extend( - Path(str_path) - for str_path in os.get_exec_path(env) - if Path(str_path) not in set(paths) - ) - return os.pathsep.join(str(p) for p in paths) - - missing_program = ''' The program '{program}' could not be executed or was not found on your system PATH. diff --git a/src/ocrmypdf/subprocess/_windows.py b/src/ocrmypdf/subprocess/_windows.py new file mode 100644 index 00000000..aec082b7 --- /dev/null +++ b/src/ocrmypdf/subprocess/_windows.py @@ -0,0 +1,162 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +import logging +import os +import shutil +import sys +from distutils.version import LooseVersion +from itertools import chain, filterfalse +from pathlib import Path +from typing import Any, Callable, Iterator, Optional, Tuple, TypeVar, cast + +try: + import winreg +except ModuleNotFoundError as e: + raise ModuleNotFoundError("This module is for Windows only") from e + +log = logging.getLogger(__name__) + +T = TypeVar('T') + + +def registry_enum( + key: winreg.HKEYType, enum_fn: Callable[[winreg.HKEYType, int], T] +) -> Iterator[T]: + LIMIT = 999 + n = 0 + while n < LIMIT: + try: + yield enum_fn(key, n) + n += 1 + except OSError: + break + if n == LIMIT: + raise ValueError(f"Too many registry keys under {key}") + + +def registry_subkeys(key: winreg.HKEYType) -> Iterator[str]: + return registry_enum(key, winreg.EnumKey) + + +def registry_values(key: winreg.HKEYType) -> Iterator[Tuple[str, Any, int]]: + return registry_enum(key, winreg.EnumValue) + + +def registry_path_ghostscript(env=None) -> Iterator[Path]: + try: + with winreg.OpenKey( + winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript" + ) as k: + latest_gs = max(registry_subkeys(k), key=LooseVersion) + with winreg.OpenKey( + winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}" + ) as k: + _, gs_path, _ = next(registry_values(k)) + yield Path(gs_path) / 'bin' + except OSError as e: + log.warning(e) + + +def registry_path_tesseract(env=None) -> Iterator[Path]: + try: + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Tesseract-OCR") as k: + for subkey, val, _valtype in registry_values(k): + if subkey == 'InstallDir': + tesseract_path = Path(val) + yield tesseract_path + except OSError as e: + log.warning(e) + + +def program_files_paths(env=None) -> Iterator[Path]: + if not env: + env = os.environ + program_files = env.get('PROGRAMFILES', '') + + def path_walker() -> Iterator[Path]: + for path in Path(program_files).iterdir(): + if not path.is_dir(): + continue + if path.name.lower() == 'tesseract-ocr': + yield path + elif path.name.lower() == 'gs': + yield from (p for p in path.glob('**/bin') if p.is_dir()) + + return iter( + sorted( + (p for p in path_walker()), + key=lambda p: (p.name, p.parent.name), + reverse=True, + ) + ) + + +def paths_from_env(env=None) -> Iterator[Path]: + return (Path(p) for p in os.get_exec_path(env) if p) + + +def shim_path(new_paths: Callable[[Any], Iterator[Path]], env=None) -> str: + if not env: + env = os.environ + return os.pathsep.join(str(p) for p in new_paths(env) if p) + + +SHIMS = [ + paths_from_env, + registry_path_ghostscript, + registry_path_tesseract, + program_files_paths, +] + + +def fix_windows_args(program, args, env): + """Adjust our desired program and command line arguments for use on Windows""" + + if sys.version_info < (3, 8): + # bpo-33617 - Windows needs manual Path -> str conversion + args = [os.fspath(arg) for arg in args] + program = os.fspath(program) + + # If we are running a .py on Windows, ensure we call it with this Python + # (to support test suite shims) + if program.lower().endswith('.py'): + args = [sys.executable] + args + + # If the program we want is not on the PATH, check elsewhere + for shim in SHIMS: + shimmed_path = shim_path(shim, env) + new_args0 = shutil.which(args[0], path=shimmed_path) + if new_args0: + args[0] = new_args0 + break + + return args + + +def unique_everseen(iterable, key=None): + "List unique elements, preserving order. Remember all elements ever seen." + # unique_everseen('AAAABBBCCDAABBB') --> A B C D + # unique_everseen('ABBCcAD', str.lower) --> A B C D + seen = set() + seen_add = seen.add + if key is None: + key = lambda x: x + for element in iterable: + k = key(element) + if k not in seen: + seen_add(k) + yield element + + +def shim_env_path(env=None): + if env is None: + env = os.environ + + shim_paths = chain.from_iterable(shim(env) for shim in SHIMS) + return os.pathsep.join( + str(p) for p in unique_everseen(shim_paths, key=lambda p: str.casefold(str(p))) + ) diff --git a/tests/test_helpers.py b/tests/test_helpers.py index fe1e5513..6d5e358c 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -13,7 +13,6 @@ from unittest.mock import MagicMock import pytest from ocrmypdf import helpers as helpers -from ocrmypdf.subprocess import shim_paths_with_program_files class TestSafeSymlink: @@ -94,7 +93,10 @@ class TestFileIsWritable: assert not helpers.is_file_writable(pathmock) +@pytest.mark.skipif(os.name != 'nt', reason="Windows test") def test_shim_paths(tmp_path): + from ocrmypdf.subprocess._windows import shim_env_path + progfiles = tmp_path / 'Program Files' progfiles.mkdir() (progfiles / 'tesseract-ocr').mkdir() @@ -103,9 +105,9 @@ def test_shim_paths(tmp_path): syspath = tmp_path / 'bin' env = {'PROGRAMFILES': str(progfiles), 'PATH': str(syspath)} - result_str = shim_paths_with_program_files(env=env) + result_str = shim_env_path(env=env) results = result_str.split(os.pathsep) - assert results[0].endswith('tesseract-ocr') - assert results[1].endswith(os.path.join('gs', '9.52', 'bin')) - assert results[2].endswith(os.path.join('gs', '9.51', 'bin')) - assert results[3] == str(syspath) + assert results[0] == str(syspath), results + assert results[-3].endswith('tesseract-ocr'), results + assert results[-2].endswith(os.path.join('gs', '9.52', 'bin')), results + assert results[-1].endswith(os.path.join('gs', '9.51', 'bin')), results From a707c56fae2f86141a2b71d3d3d7f08fbfd75f52 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 3 Dec 2020 14:17:19 -0800 Subject: [PATCH 709/880] docs: improve windows instructions --- docs/installation.rst | 37 ++++++++++++++++++------------------- 1 file changed, 18 insertions(+), 19 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 84520f6e..cc4ce35f 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -495,10 +495,6 @@ Installing on Windows Native Windows -------------- -.. note:: - - It is easier to install OCRmyPDF on Windows Subsystem for Linux. - .. note:: Administrator privileges will be required for some of these steps. @@ -509,30 +505,33 @@ You must install the following for Windows: * Tesseract 4.0 or later * Ghostscript 9.50 or later -You can install these with the Chocolatey package manager: +Using the `Chocolatey `_ package manager, install the +following when running in an Administrator command prompt: * ``choco install python3`` * ``choco install --pre tesseract`` * ``choco install ghostscript`` +* ``choco install pngquant`` (optional) -Also consider adding: +The commands above will install Python 3.x (latest version), Tesseract, Ghostscript +and pngquant. Chocolatey may also need to install the Windows Visual C++ Runtime +DLLs or other Windows patches, and may require a reboot. -* ``choco install pngquant`` +You may then use ``pip`` to install ocrmypdf. (This can performed by a user or +Administrator.): -Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier -versions of Windows and 32-bit versions of these programs are not tested, and not -supported at this time. +* ``pip install ocrmypdf -OCRmyPDF will check for Tesseract-OCR and Ghostscript in your Program Files folder. -If they are in some other location, you may need to modify the ``PATH`` -environment variable so Tesseract, Ghostscript, and other any optional executables can -be found. You can enter it in the command line or -`follow these directions `_ -to make the change persistent and system-wide. +Chocolatey automatically selects appropriate versions of these applications. If you +are installing them manually, please install 64-bit versions of all applications for +64-bit Windows, or 32-bit versions of all applications for 32-bit Windows. Mixing +the "bitness" of these programs will lead to errors. -You may then use pip to install ocrmypdf: - -* ``pip install ocrmypdf`` +OCRmyPDF will check the Windows Registry and standard locations in your Program Files +for third party software it needs (specifically, Tesseract and Ghostscript). To +override the versions OCRmyPDF selects, you can modify the ``PATH`` environment +variable. `Follow these directions `_ +to change the PATH. Windows Subsystem for Linux --------------------------- From 4194430dc16b58b7c89570b5e2b5a282b3660667 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 4 Dec 2020 13:28:04 -0800 Subject: [PATCH 710/880] Begin next release notes --- docs/release_notes.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 4037fb60..6461d1f7 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,16 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.0 +======= + +- When looking for Tesseract and Ghostscript, we now check the Windows Registry. + This should help Windows users who have installed these programs to non-standard + locations. +- We now report on the progress of PDF/A conversion, since this operation is + sometimes slow. +- Improved command line completions. + v11.3.4 ======= From 68a57a7839f866a0c57ac763edfaa65ee7a4d14a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 4 Dec 2020 17:38:48 -0800 Subject: [PATCH 711/880] Add feature to generate hocr-pdf with visible debug text --- misc/completion/ocrmypdf.fish | 3 ++- src/ocrmypdf/_pipeline.py | 9 ++++++--- src/ocrmypdf/_sync.py | 2 +- src/ocrmypdf/_validation.py | 2 +- src/ocrmypdf/cli.py | 2 +- 5 files changed, 11 insertions(+), 7 deletions(-) diff --git a/misc/completion/ocrmypdf.fish b/misc/completion/ocrmypdf.fish index 31e30b2a..d085acdd 100644 --- a/misc/completion/ocrmypdf.fish +++ b/misc/completion/ocrmypdf.fish @@ -59,7 +59,8 @@ complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "se function __fish_ocrmypdf_pdf_renderer echo -e "auto\t"(_ "auto select PDF renderer") - echo -e "hocr\t"(_ "use hocr renderer") + echo -e "hocr\t"(_ "use hOCR renderer") + echo -e "hocrdebug\t"(_ "uses hOCR renderer in debug mode, showing recognized text") echo -e "sandwich\t"(_ "use sandwich renderer") end complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options" diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 76223ac6..e0989f30 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -602,14 +602,17 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext): def render_hocr_page(hocr: Path, page_context: PageContext): + options = page_context.options output_file = page_context.get_path('ocr_hocr.pdf') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) + dpi = get_page_square_dpi(page_context.pageinfo, options) + debug_mode = options.pdf_renderer == 'hocrdebug' + hocrtransform = HocrTransform(hocr, dpi.x) # square hocrtransform.to_pdf( output_file, image_filename=None, - show_bounding_boxes=False, - invisible_text=True, + show_bounding_boxes=False if not debug_mode else True, + invisible_text=True if not debug_mode else False, interword_spaces=True, ) return output_file diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a24c05e7..624ac7bf 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -207,7 +207,7 @@ def exec_page_sync(page_context: PageContext): visible_image_out, page_context ) - if options.pdf_renderer == 'hocr': + if options.pdf_renderer.startswith('hocr'): (hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context) ocr_out = render_hocr_page(hocr_out, page_context) elif options.pdf_renderer == 'sandwich': diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 9b82ebb4..9354a629 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -78,7 +78,7 @@ def check_options_languages(options, ocr_engine_languages): def check_options_output(options): is_latin = options.languages.issubset(HOCR_OK_LANGS) - if options.pdf_renderer == 'hocr' and not is_latin: + if options.pdf_renderer.startswith('hocr') and not is_latin: msg = ( "The 'hocr' PDF renderer is known to cause problems with one " "or more of the languages in your document. Use " diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index df865aa9..6d8d4296 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -407,7 +407,7 @@ Online documentation is located at: ) advanced.add_argument( '--pdf-renderer', - choices=['auto', 'hocr', 'sandwich'], + choices=['auto', 'hocr', 'sandwich', 'hocrdebug'], default='auto', help="Choose OCR PDF renderer - the default option is to let OCRmyPDF " "choose. See documentation for discussion.", From f11bb53e61d539c2427ba81a6a4df567c6a55aee Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 7 Dec 2020 21:36:30 -0800 Subject: [PATCH 712/880] Change prefix of temporary folders Shouldn't really use a name that suggests a connection to GitHub. --- docs/release_notes.rst | 4 ++++ src/ocrmypdf/_sync.py | 2 +- tests/plugins/tesseract_cache.py | 2 +- 3 files changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 6461d1f7..f3a63ab7 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -21,6 +21,10 @@ v11.4.0 - We now report on the progress of PDF/A conversion, since this operation is sometimes slow. - Improved command line completions. +- The prefix of the temporary folder OCRmyPDF creates has been changed from + ``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this + prefix may need to be adjusted. (This has always been an implementation detail so is + not considered part of the semantic versioning "contract".) v11.3.4 ======= diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 624ac7bf..a4839aea 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -330,7 +330,7 @@ def run_pipeline(options, *, plugin_manager, api=False): if not plugin_manager: plugin_manager = get_plugin_manager(options.plugins) - work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf.")) + work_folder = Path(mkdtemp(prefix="ocrmypdf.io.")) debug_log_handler = None if ( (options.keep_temporary_files or options.verbose >= 1) diff --git a/tests/plugins/tesseract_cache.py b/tests/plugins/tesseract_cache.py index 9d75f574..37c4a690 100644 --- a/tests/plugins/tesseract_cache.py +++ b/tests/plugins/tesseract_cache.py @@ -165,7 +165,7 @@ def cached_run(options, run_args, **run_kwargs): def clean_sys_argv(): for arg in run_args[1:]: - yield re.sub(r'.*/com.github.ocrmypdf[^/]+[/](.*)', r'$TMPDIR/\1', arg) + yield re.sub(r'.*/ocrmypdf[.]io[.][^/]+[/](.*)', r'$TMPDIR/\1', arg) manifest['args'] = list(clean_sys_argv()) with (Path(CACHE_ROOT) / 'manifest.jsonl').open('a') as f: From a5feef07d0821f6feea04ebf6c1cfcee5bc6a58b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:53:22 -0800 Subject: [PATCH 713/880] Declare ocrmypdf as typed --- src/ocrmypdf/py.typed | 1 + 1 file changed, 1 insertion(+) create mode 100644 src/ocrmypdf/py.typed diff --git a/src/ocrmypdf/py.typed b/src/ocrmypdf/py.typed new file mode 100644 index 00000000..0a417894 --- /dev/null +++ b/src/ocrmypdf/py.typed @@ -0,0 +1 @@ +# ocrmypdf is typed From 0b7e52fb5eab704f57dc6cc57f5a443c32a10d3e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:54:21 -0800 Subject: [PATCH 714/880] api: parse cmdline in more type friendly way --- src/ocrmypdf/api.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index d9f1c1a8..478005c4 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -174,14 +174,14 @@ def create_options( else: raise TypeError(f"{arg}: {val} ({type(val)})") - try: - cmdline.append(os.fspath(input_file)) - except TypeError: + if isinstance(input_file, BinaryIO): cmdline.append('stream://input_file') - try: - cmdline.append(os.fspath(output_file)) - except TypeError: + else: + cmdline.append(os.fspath(input_file)) + if isinstance(output_file, BinaryIO): cmdline.append('stream://output_file') + else: + cmdline.append(os.fspath(output_file)) parser._api_mode = True options = parser.parse_args(cmdline) From 156d5d9a9c2739a4c5b1837a99f7ec5efb5b818b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:55:06 -0800 Subject: [PATCH 715/880] cli: typing --- src/ocrmypdf/cli.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 6d8d4296..24d5e036 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -6,12 +6,15 @@ import argparse +from typing import Optional, Type, TypeVar from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME from ocrmypdf._version import __version__ as _VERSION +T = TypeVar('T') -def numeric(basetype, min_=None, max_=None): + +def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = None): """Validator for numeric params""" min_ = basetype(min_) if min_ is not None else None max_ = basetype(max_) if max_ is not None else None From 043258242cce31a9ab46a9ad987872b81347b571 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:56:18 -0800 Subject: [PATCH 716/880] hocrtransform: trivial typing --- src/ocrmypdf/hocrtransform.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 6631cf1e..e464f95b 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -42,6 +42,8 @@ from reportlab.lib.colors import black, cyan, magenta, red from reportlab.lib.units import inch from reportlab.pdfgen.canvas import Canvas +Element = ElementTree.Element + Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2']) @@ -105,7 +107,7 @@ class HocrTransform: else: return '' - def _get_element_text(self, element): + def _get_element_text(self, element: Element): """ Return the textual content of the element and its children """ @@ -119,7 +121,7 @@ class HocrTransform: return text @classmethod - def element_coordinates(cls, element) -> Rect: + def element_coordinates(cls, element: Element) -> Rect: """ Returns a tuple containing the coordinates of the bounding box around an element @@ -133,7 +135,7 @@ class HocrTransform: return out @classmethod - def baseline(cls, element) -> Tuple[float, float]: + def baseline(cls, element: Element) -> Tuple[float, float]: """ Returns a tuple containing the baseline slope and intercept. """ @@ -149,7 +151,7 @@ class HocrTransform: """ return Rect._make((c / self.dpi * inch) for c in pxl) - def _child_xpath(self, html_tag, html_class=None): + def _child_xpath(self, html_tag: str, html_class: Optional[str] = None) -> str: xpath = f".//{self.xmlns}{html_tag}" if html_class: xpath += f"[@class='{html_class}']" From 997bf7578d8a3bf2f61bc9ec8a3dd34ade669a8e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:57:10 -0800 Subject: [PATCH 717/880] hocrtransform: fix exception if no div ocr_page object --- src/ocrmypdf/hocrtransform.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index e464f95b..13e266cb 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -282,13 +282,15 @@ class HocrTransform: def _do_line( self, pdf: Canvas, - line, + line: Optional[Element], elemclass: str, fontname: str, invisible_text: bool, interword_spaces: bool, show_bounding_boxes: bool, ): + if not line: + return pxl_line_coords = self.element_coordinates(line) line_box = self.pt_from_pixel(pxl_line_coords) line_height = line_box.y2 - line_box.y1 From d2908640c61480011091ddd20e1245e72507f038 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:57:30 -0800 Subject: [PATCH 718/880] pdfa: help mypy figure out a type --- src/ocrmypdf/pdfa.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 88bd1bad..8313169a 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -33,6 +33,7 @@ def _postscript_objdef( objtype = '/stream' if stream_name else '/dict' if stream_name: + assert stream_data is not None a85_data = base64.a85encode(stream_data, adobe=True).decode('ascii') yield f'{stream_name} ' + a85_data yield 'def' From 5172dbde8d0a4140e694ee8d7b2d051fc60213b7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 8 Dec 2020 19:58:17 -0800 Subject: [PATCH 719/880] subprocess: use more mypy-friendly syntax --- src/ocrmypdf/subprocess/__init__.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/subprocess/__init__.py b/src/ocrmypdf/subprocess/__init__.py index e68f8b0c..f349bd65 100644 --- a/src/ocrmypdf/subprocess/__init__.py +++ b/src/ocrmypdf/subprocess/__init__.py @@ -159,13 +159,14 @@ def get_version( raise MissingDependencyError( f"Could not find program '{program}' on the PATH" ) from e - try: - version = re.match(regex, output.strip()).group(1) - except AttributeError as e: + + match = re.match(regex, output.strip()) + if not match: raise MissingDependencyError( f"The program '{program}' did not report its version. " f"Message was:\n{output}" ) + version = match.group(1) return version From b4c1f66bc1d9b39811644f33ce1e3da80c0d0a53 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 9 Dec 2020 12:44:03 -0800 Subject: [PATCH 720/880] typing: tidy up --- src/ocrmypdf/helpers.py | 6 ++---- src/ocrmypdf/optimize.py | 5 ----- 2 files changed, 2 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 3a2c9051..afd8eaa7 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -216,10 +216,7 @@ def check_pdf(input_file: Path) -> bool: pdf.close() -T = TypeVar('T') - - -def clamp(n: T, smallest: T, largest: T) -> T: +def clamp(n, smallest, largest): # mypy doesn't understand types for this """Clamps the value of n to between smallest and largest.""" return max(smallest, min(n, largest)) @@ -232,6 +229,7 @@ def pikepdf_enable_mmap(): # log.debug("pikepdf mmap not available") # We found a race condition probably related to pybind issue #2252 that can # cause a crash. For now, disable pikepdf mmap to be on the safe side. + # Fix is not in pybind11 2.6.0 log.debug("pikepdf mmap disabled") return diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index cc9c8318..9fd80db7 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -25,7 +25,6 @@ from typing import ( Optional, Sequence, Tuple, - Union, ) import img2pdf @@ -294,10 +293,6 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefE group = pageno // options.jbig2_page_group_size jbig2_groups[group].append(xref_ext) - # Elide empty groups - jbig2_groups = { - group: xrefs for group, xrefs in jbig2_groups.items() if len(xrefs) > 0 - } log.debug("Optimizable images: JBIG2 groups: %s", (len(jbig2_groups),)) return jbig2_groups From b8aa89e1ece7fbd58910e390057415dab96fea62 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 11 Dec 2020 14:09:33 -0800 Subject: [PATCH 721/880] Fix log message queue flooding on certain files Fixes #692 --- src/ocrmypdf/pdfinfo/info.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 4c59f938..c21aa746 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -627,9 +627,12 @@ def _pdf_get_pageinfo( worker_pdf = None -def _pdf_pageinfo_sync_init(infile): +def _pdf_pageinfo_sync_init(infile: Path, pdfminer_loglevel): global worker_pdf # pylint: disable=global-statement pikepdf_enable_mmap() + + logging.getLogger('pdfminer').setLevel(pdfminer_loglevel) + # If this function is called as a thread initializer, we need a messy hack # to close worker_pdf. If called as a process, it will be released when the # process is terminated. @@ -674,7 +677,9 @@ def _pdf_pageinfo_concurrent( tqdm_kwargs=dict( total=total, desc="Scanning contents", unit='page', disable=not progbar ), - task_initializer=partial(_pdf_pageinfo_sync_init, infile), + task_initializer=partial( + _pdf_pageinfo_sync_init, infile, logging.getLogger('pdfminer').level + ), task=_pdf_pageinfo_sync, task_arguments=contexts, task_finished=update_pageinfo, From 78b71618c1ac8244f2fd96bb37d185f32c69186f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 11 Dec 2020 14:19:06 -0800 Subject: [PATCH 722/880] Fix BufferedReader TypeError --- src/ocrmypdf/api.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 478005c4..6eaf5a4b 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -9,6 +9,7 @@ import logging import os import sys from enum import IntEnum +from io import IOBase from pathlib import Path from typing import AnyStr, BinaryIO, Iterable, Optional, Union from warnings import warn @@ -174,11 +175,11 @@ def create_options( else: raise TypeError(f"{arg}: {val} ({type(val)})") - if isinstance(input_file, BinaryIO): + if isinstance(input_file, (BinaryIO, IOBase)): cmdline.append('stream://input_file') else: cmdline.append(os.fspath(input_file)) - if isinstance(output_file, BinaryIO): + if isinstance(output_file, (BinaryIO, IOBase)): cmdline.append('stream://output_file') else: cmdline.append(os.fspath(output_file)) From 594ef83551be9233c6fe12a4b958ed793c7b51ba Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 11 Dec 2020 15:09:41 -0800 Subject: [PATCH 723/880] v11.4.0 release notes --- docs/release_notes.rst | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index f3a63ab7..e3eb3143 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -15,8 +15,9 @@ wish to use some of its features for working with PDFs. v11.4.0 ======= -- When looking for Tesseract and Ghostscript, we now check the Windows Registry. - This should help Windows users who have installed these programs to non-standard +- When looking for Tesseract and Ghostscript, we now check the Windows Registry to + see if their installers registered the location of their executables. This should + help Windows users who have installed these programs to non-standard locations. - We now report on the progress of PDF/A conversion, since this operation is sometimes slow. @@ -25,6 +26,10 @@ v11.4.0 ``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this prefix may need to be adjusted. (This has always been an implementation detail so is not considered part of the semantic versioning "contract".) +- Fixed issue #692, where a particular file with malformed fonts would flood an + internal message cue by generating so many debug messages. +- Fixed an exception on processing hOCR files with no page record. Tesseract + is not known to generate such files. v11.3.4 ======= From ad202693b3dcf905e180a665a54f349d00d8dfba Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 12 Dec 2020 16:27:38 -0800 Subject: [PATCH 724/880] v11.4.0 release notes - remove change not actually implemented Remove a change that was pushed back to a future release. --- docs/release_notes.rst | 4 ---- 1 file changed, 4 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index e3eb3143..093b0824 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -22,10 +22,6 @@ v11.4.0 - We now report on the progress of PDF/A conversion, since this operation is sometimes slow. - Improved command line completions. -- The prefix of the temporary folder OCRmyPDF creates has been changed from - ``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this - prefix may need to be adjusted. (This has always been an implementation detail so is - not considered part of the semantic versioning "contract".) - Fixed issue #692, where a particular file with malformed fonts would flood an internal message cue by generating so many debug messages. - Fixed an exception on processing hOCR files with no page record. Tesseract From 7fe2954edef586c93cb9a1b9aaef666b10d96d65 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 12 Dec 2020 16:49:04 -0800 Subject: [PATCH 725/880] Change wheel tag to py36, update package_data to include py.typed --- setup.cfg | 2 +- setup.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/setup.cfg b/setup.cfg index 487ed30d..603daa71 100644 --- a/setup.cfg +++ b/setup.cfg @@ -1,5 +1,5 @@ [bdist_wheel] -python-tag = py35 +python-tag = py36 [aliases] test=pytest diff --git a/setup.py b/setup.py index c4a1e576..4e1c2bb3 100644 --- a/setup.py +++ b/setup.py @@ -82,7 +82,7 @@ setup( ], tests_require=tests_require, entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']}, - package_data={'ocrmypdf': ['data/sRGB.icc']}, + package_data={'ocrmypdf': ['data/sRGB.icc', 'py.typed']}, include_package_data=True, zip_safe=False, project_urls={ From add64e4fa2e8887be42759b8117b0feec62aa320 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 00:46:20 -0800 Subject: [PATCH 726/880] docs: com.github.ocrmypdf -> ocrmypdf.io --- docs/advanced.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/advanced.rst b/docs/advanced.rst index 4d72c776..09a7376a 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -314,7 +314,7 @@ message is: .. code-block:: none Temporary working files retained at: - /tmp/com.github.ocrmypdf.u20wpz07 + /tmp/ocrmypdf.io.u20wpz07 The organization of this folder is an implementation detail and subject to change between releases. However the general organization is that From 0ba32b96b71bc683139b1c52ed75ec37fb36028e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 00:47:25 -0800 Subject: [PATCH 727/880] Revert "v11.4.0 release notes - remove change not actually implemented" This reverts commit ad202693b3dcf905e180a665a54f349d00d8dfba. Temporary folder prefix was actually changed in commit f11bb53e. --- docs/release_notes.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 093b0824..e3eb3143 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -22,6 +22,10 @@ v11.4.0 - We now report on the progress of PDF/A conversion, since this operation is sometimes slow. - Improved command line completions. +- The prefix of the temporary folder OCRmyPDF creates has been changed from + ``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this + prefix may need to be adjusted. (This has always been an implementation detail so is + not considered part of the semantic versioning "contract".) - Fixed issue #692, where a particular file with malformed fonts would flood an internal message cue by generating so many debug messages. - Fixed an exception on processing hOCR files with no page record. Tesseract From 3675ae918cf4e91b4fed9a237350c2c0f7a5d1aa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 01:22:14 -0800 Subject: [PATCH 728/880] Fix certain invalid page ranges causing exception Closes #686 --- src/ocrmypdf/_validation.py | 19 ++++++++++++++----- tests/test_page_numbers.py | 6 ++++++ 2 files changed, 20 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 9354a629..a6ee9a08 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -13,7 +13,7 @@ import sys import unicodedata from pathlib import Path from shutil import copyfileobj -from typing import Tuple +from typing import List, Set, Tuple, Union import pikepdf import PIL @@ -136,10 +136,10 @@ def check_options_preprocessing(options): raise BadArgsError(str(e)) -def _pages_from_ranges(ranges): +def _pages_from_ranges(ranges: str) -> Set[int]: if is_iterable_notstr(ranges): return set(ranges) - pages = [] + pages: List[int] = [] page_groups = ranges.replace(' ', '').split(',') for g in page_groups: if not g: @@ -150,9 +150,18 @@ def _pages_from_ranges(ranges): pages.append(int(g) - 1) else: try: - pages.extend(range(int(start) - 1, int(end))) + new_pages = list(range(int(start) - 1, int(end))) + if not new_pages: + raise BadArgsError(f"invalid page subrange '{start}-{end}'") + pages.extend(new_pages) except ValueError: - raise BadArgsError("invalid page range") + raise BadArgsError("invalid page range") from None + + if not pages: + raise BadArgsError( + f"The string of page ranges '{ranges}' did not contain any recognizable " + f"page ranges." + ) if not monotonic(pages): log.warning( diff --git a/tests/test_page_numbers.py b/tests/test_page_numbers.py index 28f59416..71f06a0d 100644 --- a/tests/test_page_numbers.py +++ b/tests/test_page_numbers.py @@ -28,6 +28,12 @@ from ocrmypdf.pdfinfo import PdfInfo ['1,3,-11', BadArgsError], ['1-,', BadArgsError], ['start-end', BadArgsError], + ['1-0', BadArgsError], + ['99-98', BadArgsError], + ['0-0', BadArgsError], + ['1-0,3-4', BadArgsError], + [',', BadArgsError], + ['', BadArgsError], ], ) def test_pages(pages, result): From ab1ff3331b3d9c42ec4b8920fedac5834f7e10aa Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 01:38:41 -0800 Subject: [PATCH 729/880] misc: synology fix Accept user-contributed fix. Not testable. Close #690. --- misc/synology.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/misc/synology.py b/misc/synology.py index 995ca8e9..6e294ce1 100644 --- a/misc/synology.py +++ b/misc/synology.py @@ -79,8 +79,10 @@ for dir_name, subdirs, file_list in os.walk(start_dir): stdout=output_file, stderr=subprocess.PIPE, check=False, + text=True, + errors='ignore', ) - logging.info(proc.stderr.read()) + logging.info(proc.stderr) os.chmod(full_path_ocr, 0o664) os.chmod(full_path, 0o664) full_path_ocr_archive = sys.argv[2] From 4b8ccbe8cb76480b03ab42b0c61814acd1c59a60 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 01:40:31 -0800 Subject: [PATCH 730/880] v11.4.1 release notes --- docs/release_notes.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index e3eb3143..8b543517 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,16 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.1 +======= + +- Fixed an issue where invalid pages ranges passed using the ``pages`` argument, + such as "1-0" would cause unhandled exceptions. +- Accepted a user-contributed to the Synology demo script in misc/synology.py. +- Clarified documentation about change of temporary file location ``ocrmypdf.io``. +- Fixed Python wheel tag which was incorrectly set to py35 even though we long + since dropped support for Python 3.5. + v11.4.0 ======= From bb258fc99c3d3676977d708dded660d3802bc283 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Dec 2020 01:47:53 -0800 Subject: [PATCH 731/880] pdfinfo: Refactor pageinfo dictionary into a class --- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/pdfinfo/info.py | 97 +++++++++++++++++++++--------------- tests/test_main.py | 4 +- 3 files changed, 60 insertions(+), 43 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e0989f30..a546e802 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -150,7 +150,7 @@ def get_pdfinfo( progbar=False, max_workers=None, check_pages=None, -): +) -> PdfInfo: try: return PdfInfo( input_file, diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index c21aa746..207518ae 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -556,12 +556,27 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) +class RawPageInfo: + def __init__(self, pageno): + self.pageno: int = pageno + self.images: List = [] + self.textboxes: List = [] + self.has_text: Optional[bool] = False + self.userunit: Decimal = Decimal(1.0) + self.width_inches: Optional[Decimal] = None + self.height_inches: Optional[Decimal] = None + self.rotate: int = 0 + self.has_vector = False + self.has_text = False + self.dpi: Optional[Decimal] = None + self.width_pixels: Optional[int] = None + self.height_pixels: Optional[int] = None + + def _pdf_get_pageinfo( pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool ): - pageinfo: Dict[str, Any] = {} - pageinfo['pageno'] = pageno - pageinfo['images'] = [] + pageinfo = RawPageInfo(pageno) page = pdf.pages[pageno] mediabox = [Decimal(d) for d in page.MediaBox.as_list()] @@ -573,53 +588,53 @@ def _pdf_get_pageinfo( if check_this_page and detailed_analysis: pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') miner = get_page_analysis(infile, pageno, pscript5_mode) - pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes)) - bboxes = (box.bbox for box in pageinfo['textboxes']) + pageinfo.textboxes = list(simplify_textboxes(miner, get_text_boxes)) + bboxes = (box.bbox for box in pageinfo.textboxes) - pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt) + pageinfo.has_text = _page_has_text(bboxes, width_pt, height_pt) else: - pageinfo['textboxes'] = [] - pageinfo['has_text'] = None # i.e. "no information" + pageinfo.textboxes = [] + pageinfo.has_text = None # i.e. "no information" userunit = page.get('/UserUnit', Decimal(1.0)) if not isinstance(userunit, Decimal): userunit = Decimal(userunit) - pageinfo['userunit'] = userunit - pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0) - pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0) + pageinfo.userunit = userunit + pageinfo.width_inches = width_pt * userunit / Decimal(72.0) + pageinfo.height_inches = height_pt * userunit / Decimal(72.0) try: - pageinfo['rotate'] = int(page['/Rotate']) + pageinfo.rotate = int(page['/Rotate']) except KeyError: - pageinfo['rotate'] = 0 + pageinfo.rotate = 0 userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) if check_this_page: - pageinfo['has_vector'] = False - pageinfo['has_text'] = False - pageinfo['images'] = [] + pageinfo.has_vector = False + pageinfo.has_text = False + pageinfo.images = [] for ci in _process_content_streams( pdf=pdf, container=page, shorthand=userunit_shorthand ): if isinstance(ci, VectorMarker): - pageinfo['has_vector'] = True + pageinfo.has_vector = True elif isinstance(ci, TextMarker): - pageinfo['has_text'] = True + pageinfo.has_text = True elif isinstance(ci, ImageInfo): - pageinfo['images'].append(ci) + pageinfo.images.append(ci) else: raise NotImplementedError() else: - pageinfo['has_vector'] = None # i.e. "no information" - pageinfo['has_text'] = None - pageinfo['images'] = None + pageinfo.has_vector = None # i.e. "no information" + pageinfo.has_text = None + pageinfo.images = None - if pageinfo['images']: - dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images']) - pageinfo['dpi'] = dpi - pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches']))) - pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches']))) + if pageinfo.images: + dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo.images) + pageinfo.dpi = dpi + pageinfo.width_pixels = int(round(dpi.x * float(pageinfo.width_inches))) + pageinfo.height_pixels = int(round(dpi.y * float(pageinfo.height_inches))) return pageinfo @@ -707,25 +722,25 @@ class PageInfo: @property def has_text(self) -> bool: - return self._pageinfo['has_text'] + return self._pageinfo.has_text @property def has_corrupt_text(self) -> bool: if not self._detailed_analysis: raise NotImplementedError('Did not do detailed analysis') - return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) + return any(tbox.is_corrupt for tbox in self._pageinfo.textboxes) @property def has_vector(self) -> bool: - return self._pageinfo['has_vector'] + return self._pageinfo.has_vector @property def width_inches(self) -> Decimal: - return self._pageinfo['width_inches'] + return self._pageinfo.width_inches @property def height_inches(self) -> Decimal: - return self._pageinfo['height_inches'] + return self._pageinfo.height_inches @property def width_pixels(self) -> int: @@ -737,18 +752,18 @@ class PageInfo: @property def rotation(self) -> int: - return self._pageinfo.get('rotate', None) + return self._pageinfo.rotate @rotation.setter def rotation(self, value): if value in (0, 90, 180, 270, 360, -90, -180, -270): - self._pageinfo['rotate'] = value + self._pageinfo.rotate = value else: raise ValueError("rotation must be a cardinal angle") @property def images(self): - return self._pageinfo['images'] + return self._pageinfo.images def get_textareas( self, visible: Optional[bool] = None, corrupt: Optional[bool] = None @@ -763,24 +778,26 @@ class PageInfo: result = False return result - if 'textboxes' not in self._pageinfo: + if not self._pageinfo.textboxes: if visible is not None and corrupt is not None: raise NotImplementedError('Incomplete information on textboxes') - return self._pageinfo['bboxes'] + return self._pageinfo.textboxes return ( obj.bbox - for obj in self._pageinfo['textboxes'] + for obj in self._pageinfo.textboxes if predicate(obj, visible, corrupt) ) @property def dpi(self) -> Resolution: - return self._pageinfo.get('dpi', Resolution(0.0, 0.0)) + if self._pageinfo.dpi is None: + return Resolution(0.0, 0.0) + return self._pageinfo.dpi @property def userunit(self) -> Decimal: - return self._pageinfo.get('userunit', None) + return self._pageinfo.userunit @property def min_version(self) -> str: diff --git a/tests/test_main.py b/tests/test_main.py index 47a9cb40..33d45fba 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -92,10 +92,10 @@ def test_skip_ocr(resources, outpdf): def test_redo_ocr(resources, outpdf): in_ = resources / 'graph_ocred.pdf' - before = PdfInfo(in_) + before = PdfInfo(in_, detailed_analysis=True) out = outpdf out = check_ocrmypdf(in_, out, '--redo-ocr') - after = PdfInfo(out) + after = PdfInfo(out, detailed_analysis=True) assert before[0].has_text and after[0].has_text assert ( before[0].get_textareas() != after[0].get_textareas() From 037b96ca16217bb91f30e0f838bdab3eee74c5d8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Dec 2020 02:57:44 -0800 Subject: [PATCH 732/880] pdfinfo: refactor to eliminate RawPageInfo --- src/ocrmypdf/pdfinfo/info.py | 181 +++++++++++++++-------------------- 1 file changed, 77 insertions(+), 104 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 207518ae..a57b35e9 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -556,89 +556,6 @@ def simplify_textboxes(miner, textbox_getter): yield TextboxInfo(box.bbox, visible, corrupt) -class RawPageInfo: - def __init__(self, pageno): - self.pageno: int = pageno - self.images: List = [] - self.textboxes: List = [] - self.has_text: Optional[bool] = False - self.userunit: Decimal = Decimal(1.0) - self.width_inches: Optional[Decimal] = None - self.height_inches: Optional[Decimal] = None - self.rotate: int = 0 - self.has_vector = False - self.has_text = False - self.dpi: Optional[Decimal] = None - self.width_pixels: Optional[int] = None - self.height_pixels: Optional[int] = None - - -def _pdf_get_pageinfo( - pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool -): - pageinfo = RawPageInfo(pageno) - - page = pdf.pages[pageno] - mediabox = [Decimal(d) for d in page.MediaBox.as_list()] - width_pt = mediabox[2] - mediabox[0] - height_pt = mediabox[3] - mediabox[1] - - check_this_page = pageno in check_pages - - if check_this_page and detailed_analysis: - pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') - miner = get_page_analysis(infile, pageno, pscript5_mode) - pageinfo.textboxes = list(simplify_textboxes(miner, get_text_boxes)) - bboxes = (box.bbox for box in pageinfo.textboxes) - - pageinfo.has_text = _page_has_text(bboxes, width_pt, height_pt) - else: - pageinfo.textboxes = [] - pageinfo.has_text = None # i.e. "no information" - - userunit = page.get('/UserUnit', Decimal(1.0)) - if not isinstance(userunit, Decimal): - userunit = Decimal(userunit) - pageinfo.userunit = userunit - pageinfo.width_inches = width_pt * userunit / Decimal(72.0) - pageinfo.height_inches = height_pt * userunit / Decimal(72.0) - - try: - pageinfo.rotate = int(page['/Rotate']) - except KeyError: - pageinfo.rotate = 0 - - userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) - - if check_this_page: - pageinfo.has_vector = False - pageinfo.has_text = False - pageinfo.images = [] - for ci in _process_content_streams( - pdf=pdf, container=page, shorthand=userunit_shorthand - ): - if isinstance(ci, VectorMarker): - pageinfo.has_vector = True - elif isinstance(ci, TextMarker): - pageinfo.has_text = True - elif isinstance(ci, ImageInfo): - pageinfo.images.append(ci) - else: - raise NotImplementedError() - else: - pageinfo.has_vector = None # i.e. "no information" - pageinfo.has_text = None - pageinfo.images = None - - if pageinfo.images: - dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo.images) - pageinfo.dpi = dpi - pageinfo.width_pixels = int(round(dpi.x * float(pageinfo.width_inches))) - pageinfo.height_pixels = int(round(dpi.y * float(pageinfo.height_inches))) - - return pageinfo - - worker_pdf = None @@ -712,9 +629,69 @@ class PageInfo: self._pageno = pageno self._infile = infile self._detailed_analysis = detailed_analysis - self._pageinfo = _pdf_get_pageinfo( - pdf, pageno, infile, check_pages, detailed_analysis - ) + self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis) + + def _gather_pageinfo( + self, pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool + ): + page = pdf.pages[pageno] + mediabox = [Decimal(d) for d in page.MediaBox.as_list()] + width_pt = mediabox[2] - mediabox[0] + height_pt = mediabox[3] - mediabox[1] + + check_this_page = pageno in check_pages + + if check_this_page and detailed_analysis: + pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5') + miner = get_page_analysis(infile, pageno, pscript5_mode) + self._textboxes = list(simplify_textboxes(miner, get_text_boxes)) + bboxes = (box.bbox for box in self._textboxes) + + self._has_text = _page_has_text(bboxes, width_pt, height_pt) + else: + self._textboxes = [] + self._has_text = None # i.e. "no information" + + userunit = page.get('/UserUnit', Decimal(1.0)) + if not isinstance(userunit, Decimal): + userunit = Decimal(userunit) + self._userunit = userunit + self._width_inches = width_pt * userunit / Decimal(72.0) + self._height_inches = height_pt * userunit / Decimal(72.0) + + try: + self._rotate = int(page['/Rotate']) + except KeyError: + self._rotate = 0 + + userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) + + if check_this_page: + self._has_vector = False + self._has_text = False + self._images = [] + for ci in _process_content_streams( + pdf=pdf, container=page, shorthand=userunit_shorthand + ): + if isinstance(ci, VectorMarker): + self._has_vector = True + elif isinstance(ci, TextMarker): + self._has_text = True + elif isinstance(ci, ImageInfo): + self._images.append(ci) + else: + raise NotImplementedError() + else: + self._has_vector = None # i.e. "no information" + self._has_text = None + self._images = None + + self._dpi = None + if self._images: + dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in self._images) + self._dpi = dpi + self._width_pixels = int(round(dpi.x * float(self._width_inches))) + self._height_pixels = int(round(dpi.y * float(self._height_inches))) @property def pageno(self) -> int: @@ -722,25 +699,25 @@ class PageInfo: @property def has_text(self) -> bool: - return self._pageinfo.has_text + return self._has_text @property def has_corrupt_text(self) -> bool: if not self._detailed_analysis: raise NotImplementedError('Did not do detailed analysis') - return any(tbox.is_corrupt for tbox in self._pageinfo.textboxes) + return any(tbox.is_corrupt for tbox in self._textboxes) @property def has_vector(self) -> bool: - return self._pageinfo.has_vector + return self._has_vector @property def width_inches(self) -> Decimal: - return self._pageinfo.width_inches + return self._width_inches @property def height_inches(self) -> Decimal: - return self._pageinfo.height_inches + return self._height_inches @property def width_pixels(self) -> int: @@ -752,18 +729,18 @@ class PageInfo: @property def rotation(self) -> int: - return self._pageinfo.rotate + return self._rotate @rotation.setter def rotation(self, value): if value in (0, 90, 180, 270, 360, -90, -180, -270): - self._pageinfo.rotate = value + self._rotate = value else: raise ValueError("rotation must be a cardinal angle") @property def images(self): - return self._pageinfo.images + return self._images def get_textareas( self, visible: Optional[bool] = None, corrupt: Optional[bool] = None @@ -778,26 +755,22 @@ class PageInfo: result = False return result - if not self._pageinfo.textboxes: + if not self._textboxes: if visible is not None and corrupt is not None: raise NotImplementedError('Incomplete information on textboxes') - return self._pageinfo.textboxes + return self._textboxes - return ( - obj.bbox - for obj in self._pageinfo.textboxes - if predicate(obj, visible, corrupt) - ) + return (obj.bbox for obj in self._textboxes if predicate(obj, visible, corrupt)) @property def dpi(self) -> Resolution: - if self._pageinfo.dpi is None: + if self._dpi is None: return Resolution(0.0, 0.0) - return self._pageinfo.dpi + return self._dpi @property def userunit(self) -> Decimal: - return self._pageinfo.userunit + return self._userunit @property def min_version(self) -> str: From 416df803d46e39235fd05d61ab0a4e34f57efd67 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 24 Dec 2020 22:39:00 -0800 Subject: [PATCH 733/880] pdfinfo: stricter typing --- src/ocrmypdf/pdfinfo/info.py | 56 ++++++++++++++++++++++++---------- src/ocrmypdf/pdfinfo/layout.py | 7 +++-- 2 files changed, 44 insertions(+), 19 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index a57b35e9..dc80fd12 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -15,11 +15,11 @@ from functools import partial from math import hypot, isclose from os import PathLike from pathlib import Path -from typing import Any, Dict, List, Optional, Union +from typing import Any, Container, Dict, Iterator, List, Optional, Tuple, Union from warnings import warn import pikepdf -from pikepdf import PdfMatrix +from pikepdf import Object, Pdf, PdfMatrix from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf.exceptions import EncryptedPdfError @@ -115,7 +115,7 @@ def _normalize_stack(graphobjs): yield (operands, operator) -def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): +def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE): """Interpret the PDF content stream. The stack represents the state of the PDF graphics stack. We are only @@ -204,7 +204,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): ) -def _get_dpi(ctm_shorthand, image_size): +def _get_dpi(ctm_shorthand, image_size) -> Resolution: """Given the transformation matrix and image size, find the image DPI. PDFs do not include image resolution information within image data. @@ -271,8 +271,14 @@ def _get_dpi(ctm_shorthand, image_size): class ImageInfo: DPI_PREC = Decimal('1.000') - def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None): - + def __init__( + self, + *, + name='', + pdfimage: Optional[Object] = None, + inline: Optional[Object] = None, + shorthand=None, + ): self._name = str(name) self._shorthand = shorthand @@ -282,6 +288,8 @@ class ImageInfo: elif pdfimage is not None: self._origin = 'xobject' pim = pikepdf.PdfImage(pdfimage) + else: + raise ValueError("Either pdfimage or inline must be set") self._width = pim.width self._height = pim.height @@ -371,7 +379,7 @@ class ImageInfo: ).format(**class_locals) -def _find_inline_images(contentsinfo): +def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]: "Find inline images in the contentstream" for n, inline in enumerate(contentsinfo.inline_images): @@ -380,7 +388,7 @@ def _find_inline_images(contentsinfo): ) -def _image_xobjects(container): +def _image_xobjects(container) -> Iterator[Tuple[Object, str]]: """Search for all XObject-based images in the container Usually the container is a page, but it could also be a Form XObject @@ -400,7 +408,7 @@ def _image_xobjects(container): return xobjs = resources['/XObject'].as_dict() for xobj in xobjs: - candidate = xobjs[xobj] + candidate: Object = xobjs[xobj] if not '/Subtype' in candidate: continue if candidate['/Subtype'] == '/Image': @@ -408,7 +416,9 @@ def _image_xobjects(container): yield (pdfimage, xobj) -def _find_regular_images(container, contentsinfo): +def _find_regular_images( + container: Object, contentsinfo: ContentsInfo +) -> Iterator[ImageInfo]: """Find images stored in the container's /Resources /XObject Usually the container is a page, but it could also be a Form XObject @@ -432,7 +442,7 @@ def _find_regular_images(container, contentsinfo): yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand) -def _find_form_xobject_images(pdf, container, contentsinfo): +def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo): """Find any images that are in Form XObjects in the container The container may be a page, or a parent Form XObject. @@ -464,7 +474,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo): ) -def _process_content_streams(*, pdf, container, shorthand=None): +def _process_content_streams( + *, pdf: Pdf, container: Object, shorthand=None +) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]: """Find all individual instances of images drawn in the container Usually the container is a page, but it may also be a Form XObject. @@ -526,7 +538,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool: margin_ratio * ph, # bottom (first quadrant: bottom < top) ) - def rects_intersect(a, b): + def rects_intersect(a, b) -> bool: """ Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3) https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other @@ -542,7 +554,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool: return has_text -def simplify_textboxes(miner, textbox_getter): +def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]: """Extract only limited content from text boxes We do this to save memory and ensure that our objects are pickleable. @@ -625,14 +637,26 @@ def _pdf_pageinfo_concurrent( class PageInfo: - def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False): + def __init__( + self, + pdf: Pdf, + pageno: int, + infile: PathLike, + check_pages: Container[int], + detailed_analysis: bool = False, + ): self._pageno = pageno self._infile = infile self._detailed_analysis = detailed_analysis self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis) def _gather_pageinfo( - self, pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool + self, + pdf: Pdf, + pageno: int, + infile: PathLike, + check_pages: Container[int], + detailed_analysis: bool, ): page = pdf.pages[pageno] mediabox = [Decimal(d) for d in page.MediaBox.as_list()] diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 3e0be611..36b15d5d 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -223,6 +223,7 @@ def get_page_analysis(infile, pageno, pscript5_mode): ) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) + patcher = None if pscript5_mode: patcher = patch.multiple( 'pdfminer.pdffont.PDFType3Font', @@ -237,10 +238,10 @@ def get_page_analysis(infile, pageno, pscript5_mode): with Path(infile).open('rb') as f: page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0) interp.process_page(next(page)) - except PDFTextExtractionNotAllowed: - raise EncryptedPdfError() + except PDFTextExtractionNotAllowed as e: + raise EncryptedPdfError() from e finally: - if pscript5_mode: + if patcher is not None: patcher.stop() return dev.get_result() From 91db94cf2ec166b7a30bdf01391f2ed5c771f84c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 27 Dec 2020 02:02:44 -0800 Subject: [PATCH 734/880] watcher: fix OCR_LOGLEVEL env var not processed Closes #702 --- misc/watcher.py | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/misc/watcher.py b/misc/watcher.py index c6d6163d..68437878 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -44,7 +44,7 @@ DESKEW = bool(os.getenv('OCR_DESKEW', '')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) USE_POLLING = bool(os.getenv('OCR_USE_POLLING', '')) -LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() +LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO') PATTERNS = ['*.pdf', '*.PDF'] log = logging.getLogger('ocrmypdf-watcher') @@ -117,7 +117,12 @@ class HandleObserverEvent(PatternMatchingEventHandler): def main(): ocrmypdf.configure_logging( - verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True + verbosity=( + ocrmypdf.Verbosity.default + if LOGLEVEL != 'DEBUG' + else ocrmypdf.Verbosity.debug + ), + manage_root_logger=True, ) log.setLevel(LOGLEVEL) log.info( @@ -135,7 +140,7 @@ def main(): f"ARGS: {OCR_JSON_SETTINGS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"USE_POLLING: {USE_POLLING}\n" - f"LOGLEVEL: {LOGLEVEL}\n" + f"LOGLEVEL: {LOGLEVEL}" ) if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: From b01d9e07e80435f755429c978018b850015ee60d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 27 Dec 2020 02:24:00 -0800 Subject: [PATCH 735/880] Deal with missing pthread_sigmask on Cygwin Closes #701 --- src/ocrmypdf/_concurrent.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 22234580..70ee03dd 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -12,6 +12,7 @@ import os import signal import sys import threading +from contextlib import suppress from multiprocessing import Pool as ProcessPool from multiprocessing.dummy import Pool as ThreadPool from typing import Callable, Iterable, Optional @@ -56,7 +57,7 @@ def process_init(queue, user_init, loglevel): signal.signal(signal.SIGINT, signal.SIG_IGN) # Install SIGBUS handler (so our parent process can abort somewhat gracefully) - if hasattr(signal, 'SIGBUS'): + with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS signal.signal(signal.SIGBUS, process_sigbus) # Reconfigure the root logger for this process to send all messages to a queue @@ -72,7 +73,8 @@ def process_init(queue, user_init, loglevel): def thread_init(_queue, user_init, _loglevel): # As a thread, block SIGBUS so the main thread deals with it... - if hasattr(signal, 'SIGBUS'): + with suppress(AttributeError): + # Windows and Cygwin do not have pthread_sigmask or SIGBUS signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) if user_init: user_init() From 607e2d7e8172da0534744ff89b2c7a930d9eb794 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 27 Dec 2020 03:29:35 -0800 Subject: [PATCH 736/880] v11.4.2 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 8b543517..7a8400a8 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.2 +======= + +- Fixed support for Cygwin, hopefully. +- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted. + v11.4.1 ======= From 81602cf420e8e79609097c39b62d85cdf5828604 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 27 Dec 2020 16:01:50 -0800 Subject: [PATCH 737/880] Fix test not patching properly after Ghostscript polling change --- tests/plugins/gs_feature_elision.py | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index 98e78086..fd85204a 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -19,25 +19,28 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -from unittest.mock import patch +from unittest.mock import Mock, patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript -from ocrmypdf.subprocess import run +from ocrmypdf.subprocess import run_polling_stderr elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 not permitted in PDF/A-2, overprint mode not set""" def run_append_stderr(*args, **kwargs): - proc = run(*args, **kwargs) - proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')]) + proc = run_polling_stderr(*args, **kwargs) + proc.stderr += '\n' + elision_warning + '\n' return proc @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr): + m = Mock() + m.side_effect = run_append_stderr + + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', m): ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, @@ -47,4 +50,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf pdfa_part=pdfa_part, progressbar_class=None, ) - return output_file + m.assert_called_once() + return output_file From 0ff0d2f8d16a38bb0216cc848cff7baed6f08994 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 27 Dec 2020 16:19:05 -0800 Subject: [PATCH 738/880] Remove PDF/A overprint debug message Since we currently log all of a process's output at debug it's redundant to log this separate message. --- src/ocrmypdf/_exec/ghostscript.py | 10 ++-------- 1 file changed, 2 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index d96cd3c3..2db45528 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -254,6 +254,8 @@ def generate_pdfa( raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e else: stderr = p.stderr + # If there is an error we log the whole stderr, except for filtering + # duplicates. if _gs_error_reported(stderr): last_part = None repcount = 0 @@ -266,11 +268,3 @@ def generate_pdfa( else: repcount += 1 last_part = part - elif 'overprint mode not set' in stderr: - # Unless someone is going to print PDF/A documents on a - # magical sRGB printer I can't see the removal of overprinting - # being a problem.... - log.debug( - "Ghostscript had to remove PDF 'overprinting' from the " - "input file to complete PDF/A conversion. " - ) From dc06990e5d67f4b89e4f342437384c3ad41925ae Mon Sep 17 00:00:00 2001 From: Tim Gates Date: Tue, 29 Dec 2020 10:28:34 +1100 Subject: [PATCH 739/880] docs: fix simple typo, instsalled -> installed (#704) There is a small typo in docs/installation.rst. Should read `installed` rather than `instsalled`. --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index cc4ce35f..6bec3e20 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -637,7 +637,7 @@ Installing with Python pip OCRmyPDF is delivered by PyPI because it is a convenient way to install the latest version. However, PyPI and ``pip`` cannot address the fact that ``ocrmypdf`` depends on certain non-Python system libraries and -programs being instsalled. +programs being installed. For best results, first install `your platform's version `__ of From babc76fa740a282240fada0620088c439a1cc12d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 28 Dec 2020 23:51:55 -0800 Subject: [PATCH 740/880] tests: assert that most patched functions are called We were not actually checking if functions we patched we called when expected. --- tests/plugins/gs_feature_elision.py | 10 ++++------ tests/plugins/gs_pdfa_failure.py | 4 +++- tests/plugins/gs_raster_failure.py | 4 +++- tests/plugins/gs_render_failure.py | 4 +++- tests/plugins/tesseract_badutf8.py | 13 +++++++++++-- tests/plugins/tesseract_big_image_error.py | 15 ++++++++++++--- tests/plugins/tesseract_crash.py | 15 ++++++++++++--- tests/test_helpers.py | 5 +++++ tests/test_image_input.py | 3 ++- tests/test_metadata.py | 2 +- tests/test_optimize.py | 4 +++- tests/test_unpaper.py | 10 ++++++---- tests/test_validation.py | 9 ++++++--- 13 files changed, 71 insertions(+), 27 deletions(-) diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index fd85204a..97a17ab3 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -19,7 +19,7 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -from unittest.mock import Mock, patch +from unittest.mock import patch from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript @@ -37,10 +37,8 @@ def run_append_stderr(*args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - m = Mock() - m.side_effect = run_append_stderr - - with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', m): + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock: + mock.side_effect = run_append_stderr ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, @@ -50,5 +48,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf pdfa_part=pdfa_part, progressbar_class=None, ) - m.assert_called_once() + mock.assert_called_once() return output_file diff --git a/tests/plugins/gs_pdfa_failure.py b/tests/plugins/gs_pdfa_failure.py index 8973dcaf..43f6df0f 100644 --- a/tests/plugins/gs_pdfa_failure.py +++ b/tests/plugins/gs_pdfa_failure.py @@ -39,7 +39,8 @@ def run_rig_args(args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=run_rig_args): + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock: + mock.side_effect = run_rig_args ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, @@ -49,4 +50,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf pdfa_part=pdfa_part, progressbar_class=None, ) + mock.assert_called() return output_file diff --git a/tests/plugins/gs_raster_failure.py b/tests/plugins/gs_raster_failure.py index fbf3d5cd..0d85b263 100644 --- a/tests/plugins/gs_raster_failure.py +++ b/tests/plugins/gs_raster_failure.py @@ -44,7 +44,8 @@ def rasterize_pdf_page( rotation=None, filter_vector=False, ) -> Path: - with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail): + with patch('ocrmypdf._exec.ghostscript.run') as mock: + mock.side_effect = raise_gs_fail ghostscript.rasterize_pdf_page( input_file=input_file, output_file=output_file, @@ -55,4 +56,5 @@ def rasterize_pdf_page( rotation=rotation, filter_vector=filter_vector, ) + mock.assert_called() return output_file diff --git a/tests/plugins/gs_render_failure.py b/tests/plugins/gs_render_failure.py index dcbc6c5a..3c88e71f 100644 --- a/tests/plugins/gs_render_failure.py +++ b/tests/plugins/gs_render_failure.py @@ -34,7 +34,8 @@ def raise_gs_fail(*args, **kwargs): @hookimpl def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part): - with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=raise_gs_fail): + with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock: + mock.side_effect = raise_gs_fail ghostscript.generate_pdfa( pdf_pages=pdf_pages, pdfmark=pdfmark, @@ -44,4 +45,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf pdfa_part=pdfa_part, progressbar_class=None, ) + mock.assert_called() return output_file diff --git a/tests/plugins/tesseract_badutf8.py b/tests/plugins/tesseract_badutf8.py index 3511938d..5a5e1e32 100644 --- a/tests/plugins/tesseract_badutf8.py +++ b/tests/plugins/tesseract_badutf8.py @@ -26,6 +26,7 @@ that is not UTF-8 compatible, so we are forced to check that we can convert it and present it to the user. """ +from contextlib import contextmanager from subprocess import CalledProcessError from unittest.mock import patch @@ -42,17 +43,25 @@ def bad_utf8(*args, **kwargs): ) +@contextmanager +def patch_tesseract_run(): + with patch('ocrmypdf._exec.tesseract.run') as mock: + mock.side_effect = bad_utf8 + yield + mock.assert_called() + + class BadUtf8OcrEngine(TesseractOcrEngine): @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8): + with patch_tesseract_run(): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8): + with patch_tesseract_run(): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/plugins/tesseract_big_image_error.py b/tests/plugins/tesseract_big_image_error.py index 04d0e0cd..e2382ac5 100644 --- a/tests/plugins/tesseract_big_image_error.py +++ b/tests/plugins/tesseract_big_image_error.py @@ -19,6 +19,7 @@ # TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +from contextlib import contextmanager from subprocess import CalledProcessError from unittest.mock import patch @@ -35,22 +36,30 @@ def raise_size_exception(*args, **kwargs): ) +@contextmanager +def patch_tesseract_run(): + with patch('ocrmypdf._exec.tesseract.run') as mock: + mock.side_effect = raise_size_exception + yield + mock.assert_called() + + class BigImageErrorOcrEngine(TesseractOcrEngine): @staticmethod def get_orientation(input_file, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): + with patch_tesseract_run(): return TesseractOcrEngine.get_orientation(input_file, options) @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): + with patch_tesseract_run(): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception): + with patch_tesseract_run(): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/plugins/tesseract_crash.py b/tests/plugins/tesseract_crash.py index c76bafd5..60ff1110 100755 --- a/tests/plugins/tesseract_crash.py +++ b/tests/plugins/tesseract_crash.py @@ -20,6 +20,7 @@ # SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. import signal +from contextlib import contextmanager from subprocess import CalledProcessError from unittest.mock import patch @@ -37,22 +38,30 @@ def raise_crash(*args, **kwargs): ) +@contextmanager +def patch_tesseract_run(): + with patch('ocrmypdf._exec.tesseract.run') as mock: + mock.side_effect = raise_crash + yield + mock.assert_called() + + class CrashOcrEngine(TesseractOcrEngine): @staticmethod def get_orientation(input_file, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): + with patch_tesseract_run(): return TesseractOcrEngine.get_orientation(input_file, options) @staticmethod def generate_hocr(input_file, output_hocr, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): + with patch_tesseract_run(): TesseractOcrEngine.generate_hocr( input_file, output_hocr, output_text, options ) @staticmethod def generate_pdf(input_file, output_pdf, output_text, options): - with patch('ocrmypdf._exec.tesseract.run', new=raise_crash): + with patch_tesseract_run(): TesseractOcrEngine.generate_pdf( input_file, output_pdf, output_text, options ) diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 6d5e358c..f0a5e4d0 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -36,12 +36,17 @@ class TestSafeSymlink: def test_no_cpu_count(monkeypatch): + invoked = False + def cpu_count_raises(): + nonlocal invoked + invoked = True raise NotImplementedError() monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises) with pytest.warns(expected_warning=UserWarning): assert helpers.available_cpu_count() == 1 + assert invoked, "Patched function called during test" def test_deprecated(): diff --git a/tests/test_image_input.py b/tests/test_image_input.py index 3a872000..5efad8c0 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -65,11 +65,12 @@ def test_cmyk_no_icc(caplog, resources, no_outpdf): def test_img2pdf_fails(resources, no_outpdf): with patch( 'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError() - ): + ) as mock: rc = run_ocrmypdf_api( resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200' ) assert rc == ocrmypdf.ExitCode.input_file + mock.assert_called() def test_jpeg_in_jpeg_out(resources, outpdf): diff --git a/tests/test_metadata.py b/tests/test_metadata.py index a41fcab1..16bdb16b 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -318,7 +318,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog): ) metadata_fixup(working_file=outdir / 'graph.pdf', context=context) for record in caplog.records: - assert record.levelname != 'WARNING' + assert record.levelname != 'WARNING', "Unexpected warning" # Now add some metadata that will not be copyable graph = pikepdf.open(outdir / 'graph.pdf') diff --git a/tests/test_optimize.py b/tests/test_optimize.py index f6c1f081..e04fc939 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -141,7 +141,8 @@ def test_multiple_pngs(resources, outdir): draw.rectangle((0, 0, im.width, im.height), fill=128) im.save(output_file) - with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant): + with patch('ocrmypdf.optimize.pngquant.quantize') as mock: + mock.side_effect = mockquant check_ocrmypdf( outdir / 'in.pdf', outdir / 'out.pdf', @@ -155,6 +156,7 @@ def test_multiple_pngs(resources, outdir): '--plugin', 'tests/plugins/tesseract_noop.py', ) + mock.assert_called() with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open( outdir / 'out.pdf' diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 09fdcb8f..eefe9852 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -28,11 +28,12 @@ def test_no_unpaper(resources, no_outpdf): output = fspath(no_outpdf) _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) - with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version: - mock_unpaper_version.side_effect = FileNotFoundError("unpaper") + with patch("ocrmypdf._exec.unpaper.version") as mock: + mock.side_effect = FileNotFoundError("unpaper") with pytest.raises(MissingDependencyError): check_options(options, pm) + mock.assert_called() def test_old_unpaper(resources, no_outpdf): @@ -40,11 +41,12 @@ def test_old_unpaper(resources, no_outpdf): output = fspath(no_outpdf) _parser, options, pm = get_parser_options_plugins(["--clean", input_, output]) - with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version: - mock_unpaper_version.return_value = '0.5' + with patch("ocrmypdf._exec.unpaper.version") as mock: + mock.return_value = '0.5' with pytest.raises(MissingDependencyError): check_options(options, pm) + mock.assert_called() @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") diff --git a/tests/test_validation.py b/tests/test_validation.py index ec8cf8dd..f7ce5286 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -188,18 +188,20 @@ def test_language_warning(caplog): caplog.set_level(logging.DEBUG) with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') - ): + ) as mock: vd.check_options_languages(opts, {'eng'}) assert opts.languages == {'eng'} assert '' in caplog.text + mock.assert_called_once() opts = make_opts(language=None) with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8') - ): + ) as mock: vd.check_options_languages(opts, {'eng'}) assert opts.languages == {'eng'} assert 'assuming --language' in caplog.text + mock.assert_called_once() def test_version_comparison(): @@ -265,7 +267,8 @@ def test_pagesegmode_warning(caplog): def test_two_languages(): - with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True): + with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock: vd._check_options( *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} ) + mock.assert_called() From 96d68c2413672dd6b37c2aef8b5365b75d76cd28 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 29 Dec 2020 01:47:32 -0800 Subject: [PATCH 741/880] pipeline: refactor metadata_fixup --- src/ocrmypdf/_pipeline.py | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index a546e802..ab38a76d 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -756,19 +756,17 @@ def metadata_fixup(working_file: Path, context: PdfContext): if 'xmp:CreateDate' not in meta: meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '') - # Ghostscript likes to set title to Untitled if omitted from input. - # Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1 - # and the XMP Spec do not make this recommendation. - if meta.get('dc:title') == 'Untitled': - with original.open_metadata( - set_pikepdf_as_editor=False, update_docinfo=False - ) as original_meta: - if 'dc:title' not in original_meta: + with original.open_metadata( + set_pikepdf_as_editor=False, update_docinfo=False, strict=False + ) as meta_original: + if meta.get('dc:title') == 'Untitled': + # Ghostscript likes to set title to Untitled if omitted from input. + # Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1 + # and the XMP Spec do not make this recommendation. + if 'dc:title' not in meta_original: del meta['dc:title'] - - meta_original = original.open_metadata() - missing = set(meta_original.keys()) - set(meta.keys()) - report_on_metadata(missing) + missing = set(meta_original.keys()) - set(meta.keys()) + report_on_metadata(missing) pdf.save( output_file, From 72fa347c38319613516dc0b6a2004fbfc0367b39 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 29 Dec 2020 01:47:52 -0800 Subject: [PATCH 742/880] tests: skip metadata test for two pikepdf versions that warn incorrectly --- tests/test_metadata.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 16bdb16b..e6496d36 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -306,6 +306,9 @@ def test_kodak_toc(resources, outpdf): assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary) +@pytest.mark.skipif( + pikepdf.__version__ in ('2.2.2', '2.2.3'), reason="Raises wrong warning" +) def test_metadata_fixup_warning(resources, outdir, caplog): options = get_parser().parse_args( args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf'] From b0afef09efe23124adbdef95e625739719286f14 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 29 Dec 2020 21:40:35 -0800 Subject: [PATCH 743/880] v11.4.3 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 7a8400a8..68008924 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.3 +======= + +- Removed a redundant debug message. +- Test suite now asserts that most patched functions are called when they should be. +- Test suite now skips a test that fails on two particular versions of piekpdf. + v11.4.2 ======= From 6ba4b7b3f3a2dbbec9f974df97f551d23fff025d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 30 Dec 2020 01:40:56 -0800 Subject: [PATCH 744/880] ci: temporarily disable pngquant on Windows Looks like a packaging error, choco complains of bad hashes. --- azure-pipelines.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index db05746c..cece2435 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -32,7 +32,7 @@ stages: choco install --yes --no-progress --pre tesseract choco install --yes --no-progress python3 choco install --yes --no-progress ghostscript - choco install --yes --no-progress pngquant + # choco install --yes --no-progress pngquant displayName: "Install system packages" - pwsh: | refreshenv From bd0f00586147795993629c8d362134213fae77ff Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 30 Dec 2020 01:58:57 -0800 Subject: [PATCH 745/880] tests: tag tests that need pngquant, jbig2enc --- tests/test_optimize.py | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index e04fc939..d5ffc5fd 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -21,7 +21,15 @@ from ocrmypdf.helpers import Resolution check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101 +needs_pngquant = pytest.mark.skipif( + not pngquant.available(), reason="pngquant not installed" +) +needs_jbig2enc = pytest.mark.skipif( + not jbig2enc.available(), reason="jbig2enc not installed" +) + +@needs_pngquant @pytest.mark.parametrize('pdf', ['multipage.pdf', 'palette.pdf']) def test_basic(resources, pdf, outpdf): infile = resources / pdf @@ -30,6 +38,7 @@ def test_basic(resources, pdf, outpdf): assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size +@needs_pngquant def test_mono_not_inverted(resources, outdir): infile = resources / '2400dpi.pdf' opt.main(infile, outdir / 'out.pdf', level=3) @@ -45,7 +54,7 @@ def test_mono_not_inverted(resources, outdir): assert im.getpixel((0, 0)) == 255, "Expected white background" -@pytest.mark.skipif(not pngquant.available(), reason='need pngquant') +@needs_pngquant def test_jpg_png_params(resources, outpdf): check_ocrmypdf( resources / 'crom.png', @@ -63,7 +72,7 @@ def test_jpg_png_params(resources, outpdf): ) -@pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc') +@needs_jbig2enc @pytest.mark.parametrize('lossy', [False, True]) def test_jbig2_lossy(lossy, resources, outpdf): args = [ @@ -95,10 +104,8 @@ def test_jbig2_lossy(lossy, resources, outpdf): assert len(pim.decode_parms) == 0 -@pytest.mark.skipif( - not jbig2enc.available() or not pngquant.available(), - reason='need jbig2enc and pngquant', -) +@needs_pngquant +@needs_jbig2enc def test_flate_to_jbig2(resources, outdir): # This test requires an image that pngquant is capable of converting to # to 1bpp - so use an existing 1bpp image, convert up, confirm it can @@ -126,6 +133,7 @@ def test_flate_to_jbig2(resources, outdir): assert pim.filters[0] == '/JBIG2Decode' +@needs_pngquant def test_multiple_pngs(resources, outdir): with Path.open(outdir / 'in.pdf', 'wb') as inpdf: img2pdf.convert( From df6e1062033003ca3dcf5463b95d04b74d00481b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 Jan 2021 00:44:46 -0800 Subject: [PATCH 746/880] concurrent: simplify results loop --- src/ocrmypdf/_concurrent.py | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 70ee03dd..177ffe0a 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -109,15 +109,11 @@ def exec_progress_pool( ) try: results = pool.imap_unordered(task, task_arguments) - while True: - try: - result = results.next() - if task_finished: - task_finished(result, pbar) - else: - pbar.update() - except StopIteration: - break + for result in results: + if task_finished: + task_finished(result, pbar) + else: + pbar.update() except KeyboardInterrupt: # Terminate pool so we exit instantly pool.terminate() From 1e80d412fa983e96d30128770fbaba366132cc58 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 Jan 2021 00:46:00 -0800 Subject: [PATCH 747/880] tesseract: fix typing of some optional arguments --- src/ocrmypdf/_exec/tesseract.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 171bf306..46e810c7 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -14,7 +14,7 @@ from collections import namedtuple from os import fspath from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired -from typing import List +from typing import List, Optional from PIL import Image @@ -118,7 +118,7 @@ def get_languages(): return set(lang.strip() for lang in rest) -def tess_base_args(langs: List[str], engine_mode: int) -> List[str]: +def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]: args = ['tesseract'] if langs: args.extend(['-l', '+'.join(langs)]) @@ -127,7 +127,7 @@ def tess_base_args(langs: List[str], engine_mode: int) -> List[str]: return args -def get_orientation(input_file: Path, engine_mode: int, timeout: float): +def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float): args_tesseract = tess_base_args(['osd'], engine_mode) + [ '--psm', '0', From 0b3a526049f10032a29939a82dcd5113cd895935 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 Jan 2021 01:11:32 -0800 Subject: [PATCH 748/880] Partial fix crash on 'userunit' None (#700) Our method of getting data from pdfminer would silently consume a StopIteration if pdfminer returned no processed pages, leading to odd error message. We improve an error from pdfminer properly, and returning a more descriptive error of our own. It would be possible for ocrmypdf to repair the file before sending it to pdfminer, but this seems to be rare enough that we won't do that yet. --- src/ocrmypdf/pdfinfo/info.py | 4 +++- src/ocrmypdf/pdfinfo/layout.py | 11 ++++++++--- tests/test_pdfinfo.py | 19 +++++++++++++++++++ 3 files changed, 30 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index dc80fd12..ae02fb98 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -22,7 +22,7 @@ import pikepdf from pikepdf import Object, Pdf, PdfMatrix from ocrmypdf._concurrent import exec_progress_pool -from ocrmypdf.exceptions import EncryptedPdfError +from ocrmypdf.exceptions import EncryptedPdfError, InputFileError from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes @@ -598,6 +598,8 @@ def _pdf_pageinfo_concurrent( def update_pageinfo(result, pbar): page = result + if not page: + raise InputFileError("Could read a page in the PDF") pages[page.pageno] = page pbar.update() diff --git a/src/ocrmypdf/pdfinfo/layout.py b/src/ocrmypdf/pdfinfo/layout.py index 36b15d5d..4159a1cb 100644 --- a/src/ocrmypdf/pdfinfo/layout.py +++ b/src/ocrmypdf/pdfinfo/layout.py @@ -21,7 +21,7 @@ from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined from pdfminer.pdfpage import PDFPage from pdfminer.utils import bbox2str, matrix2str -from ocrmypdf.exceptions import EncryptedPdfError +from ocrmypdf.exceptions import EncryptedPdfError, InputFileError STRIP_NAME = re.compile(r'[0-9]+') @@ -236,8 +236,13 @@ def get_page_analysis(infile, pageno, pscript5_mode): try: with Path(infile).open('rb') as f: - page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0) - interp.process_page(next(page)) + page_iter = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0) + page = next(page_iter, None) + if page is None: + raise InputFileError( + f"pdfminer could not process page {pageno} (counting from 0)." + ) + interp.process_page(page) except PDFTextExtractionNotAllowed as e: raise EncryptedPdfError() from e finally: diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 16ec12fd..4e6e53ce 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -15,7 +15,11 @@ from PIL import Image from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo +from ocrmypdf.exceptions import InputFileError from ocrmypdf.pdfinfo import Colorspace, Encoding +from ocrmypdf.pdfinfo.layout import PDFPage + +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api # pylint: disable=protected-access @@ -179,3 +183,18 @@ def test_stack_abuse(): with pytest.warns(None): with pytest.raises(RuntimeError): pdfinfo.info._interpret_contents(stream) + + +def test_pages_issue700(monkeypatch, resources): + def get_no_pages(*args, **kwargs): + return iter([]) + + monkeypatch.setattr(PDFPage, 'get_pages', get_no_pages) + + with pytest.raises(InputFileError, match="pdfminer"): + pdfinfo.PdfInfo( + resources / 'cardinal.pdf', + detailed_analysis=True, + progbar=False, + max_workers=1, + ) From df157552f3040773eaac84a2d6dee807e75a1b7e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 Jan 2021 01:37:09 -0800 Subject: [PATCH 749/880] Make ocrmypdf.ocr take a threading lock --- docs/api.rst | 5 +++++ src/ocrmypdf/api.py | 23 ++++++++++++++++------- 2 files changed, 21 insertions(+), 7 deletions(-) diff --git a/docs/api.rst b/docs/api.rst index a93c4a9c..3459e8b8 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -56,6 +56,11 @@ Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal handler (except on Windows), to raise an exception if access to a memory mapped file fails. OCRmyPDF may use memory mapping. +``ocrmypdf.ocr()`` will take a threading lock to prevent multiple runs of itself +in the same Python interpreter process. This is not thread-safe, because of how +OCRmyPDF's plugins and Python's library import system work. If you need to parallelize +OCRmyPDF, use processes. + .. warning:: On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 6eaf5a4b..9a37cad6 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -8,6 +8,7 @@ import logging import os import sys +import threading from enum import IntEnum from io import IOBase from pathlib import Path @@ -30,6 +31,8 @@ except ModuleNotFoundError: StrPath = Union[os.PathLike, AnyStr] PathOrIO = Union[BinaryIO, StrPath] +_api_lock = threading.Lock() + class Verbosity(IntEnum): """Verbosity level for configure_logging.""" @@ -306,12 +309,18 @@ def ocr( # pylint: disable=unused-argument parser = get_parser() create_options_kwargs['parser'] = parser - plugin_manager = get_plugin_manager(plugins) - plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member - if 'verbose' in kwargs: - warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().") + with _api_lock: + # We can't allow multiple ocrmypdf.ocr() threads to run in parallel, because + # they might install different plugins, and generally speaking we have areas + # of code that use global state. - options = create_options(**create_options_kwargs) - check_options(options, plugin_manager) - return run_pipeline(options=options, plugin_manager=plugin_manager, api=True) + plugin_manager = get_plugin_manager(plugins) + plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member + + if 'verbose' in kwargs: + warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().") + + options = create_options(**create_options_kwargs) + check_options(options, plugin_manager) + return run_pipeline(options=options, plugin_manager=plugin_manager, api=True) From 47ef1914d491f9a6652c554aa6c93099410f2439 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 1 Jan 2021 01:39:24 -0800 Subject: [PATCH 750/880] v11.4.4 release notes --- docs/release_notes.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 68008924..18e8de4c 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,15 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.4 +======= + +- Fixed ``AttributeError: 'NoneType' object has no attribute 'userunit'``, issue #700, + related to OCRmyPDF not properly forwarded an error message from pdfminer.six. +- Adjusted typing of some arguments. +- ``ocrmypdf.ocr`` now takes a ``threading.Lock`` for reasons outlined in the + documentation. + v11.4.3 ======= From 2846d46bb83ea846fb71dfafab46ef03067943ec Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 Jan 2021 03:58:18 -0800 Subject: [PATCH 751/880] Remove .coveragerc and fold into setup.cfg --- .coveragerc | 24 ------------------------ setup.cfg | 29 +++++++++++++++++++++++++++++ src/ocrmypdf/_sync.py | 2 ++ src/ocrmypdf/optimize.py | 4 ++-- tests/conftest.py | 9 --------- 5 files changed, 33 insertions(+), 35 deletions(-) delete mode 100644 .coveragerc diff --git a/.coveragerc b/.coveragerc deleted file mode 100644 index 42b8e78f..00000000 --- a/.coveragerc +++ /dev/null @@ -1,24 +0,0 @@ -[paths] -source = - src - */site-packages - -[run] -branch = true -parallel = true -concurrency = - thread - multiprocessing -source = - src/ocrmypdf - -[report] -exclude_lines = - pragma: no cover - def __repr__ - raise AssertionError - raise NotImplementedError - if 0: - if False: - if __name__ == .__main__.: - if TYPE_CHECKING: diff --git a/setup.cfg b/setup.cfg index 603daa71..44ed90c5 100644 --- a/setup.cfg +++ b/setup.cfg @@ -15,6 +15,8 @@ filterwarnings = ignore:.*XMLParser.*:DeprecationWarning markers = slow +addopts = + -n auto [isort] multi_line_output=3 @@ -27,3 +29,30 @@ known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_re [metadata] license_file = LICENSE + +[coverage:paths] +source = + src/ + +[coverage:run] +branch = true +parallel = true +concurrency = multiprocessing +source = + src/ocrmypdf + +[coverage:report] +# Regexes for lines to exclude from consideration +exclude_lines = + # Have to re-enable the standard pragma + pragma: no cover + + # Don't complain if tests don't hit defensive assertion code: + raise AssertionError + raise NotImplementedError + + # Don't complain if non-runnable code isn't run: + if 0: + if False: + if __name__ == .__main__.: + if TYPE_CHECKING: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index a4839aea..83de4150 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -338,6 +338,8 @@ def run_pipeline(options, *, plugin_manager, api=False): and not api ): # Debug log for command line interface only with verbose output + # See https://github.com/pytest-dev/pytest/issues/5502 for why we skip this + # when pytest is running debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") pikepdf_enable_mmap() diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 9fd80db7..c097e32e 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -498,7 +498,7 @@ def transcode_pngs( @deprecated -def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: +def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cover im_obj.BitsPerComponent = 1 im_obj.Width = compdata.w im_obj.Height = compdata.h @@ -519,7 +519,7 @@ def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: @deprecated -def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: +def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cover # When a PNG is inserted into a PDF, we more or less copy the IDAT section from # the PDF and transfer the rest of the PNG headers to PDF image metadata. # One thing we have to do is tell the PDF reader whether a predictor was used diff --git a/tests/conftest.py b/tests/conftest.py index 7619769f..70741548 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -137,16 +137,7 @@ def run_ocrmypdf(input_file, output_file, *args, text=True): + [str(input_file), str(output_file)] ) - # Tell subprocess where to find coverage.py configuration - # This has no unless except when coverage is running - # Details: https://coverage.readthedocs.io/en/coverage-5.0/subprocess.html - coverage_rc = Path(__file__).parent.parent / '.coveragerc' env = os.environ.copy() - if coverage_rc.exists(): - env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc) - elif not running_in_docker(): - assert False, "could not find .coveragerc" - p = run( p_args, stdout=PIPE, From 62e5edc72bfbff8640f1589355a89c1d30d46b7c Mon Sep 17 00:00:00 2001 From: Jonas Winkler <17569239+jonaswinkler@users.noreply.github.com> Date: Wed, 6 Jan 2021 12:59:28 +0100 Subject: [PATCH 752/880] fix unclosed file warnings. (#710) Co-authored-by: Jonas Winkler --- src/ocrmypdf/subprocess/__init__.py | 25 ++++++++++++------------- 1 file changed, 12 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/subprocess/__init__.py b/src/ocrmypdf/subprocess/__init__.py index f349bd65..2755337b 100644 --- a/src/ocrmypdf/subprocess/__init__.py +++ b/src/ocrmypdf/subprocess/__init__.py @@ -77,20 +77,19 @@ def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs): args, env, process_log, text = _fix_process_args(args, env, kwargs) assert text, "Must use text=True" - proc = Popen(args, env=env, **kwargs) + with Popen(args, env=env, **kwargs) as proc: + lines = [] + while proc.poll() is None: + for msg in iter(proc.stderr.readline, ''): + if process_log.isEnabledFor(logging.DEBUG): + process_log.debug(msg.strip()) + callback(msg) + lines.append(msg) + stderr = ''.join(lines) - lines = [] - while proc.poll() is None: - for msg in iter(proc.stderr.readline, ''): - if process_log.isEnabledFor(logging.DEBUG): - process_log.debug(msg.strip()) - callback(msg) - lines.append(msg) - stderr = ''.join(lines) - - if check and proc.returncode != 0: - raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr) - return CompletedProcess(args, proc.returncode, None, stderr=stderr) + if check and proc.returncode != 0: + raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr) + return CompletedProcess(args, proc.returncode, None, stderr=stderr) def _fix_process_args(args, env, kwargs): From d32324859ca54cc0ef225f3724ea3296438442c7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 6 Jan 2021 11:42:28 -0800 Subject: [PATCH 753/880] v11.4.5 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 18e8de4c..7260b903 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.4.5 +======= + +- Fixed an issue where files may not be closed when the API is used. +- Improved ``setup.cfg`` with better settings for test coverage. + v11.4.4 ======= From 6f4b38b103d8481986bf870257185f5875f775ab Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Jan 2021 00:41:03 -0800 Subject: [PATCH 754/880] ghostscript: tidy comments --- src/ocrmypdf/_exec/ghostscript.py | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 2db45528..3582a6eb 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -15,7 +15,7 @@ from os import fspath from pathlib import Path from shutil import which from subprocess import PIPE, CalledProcessError -from typing import Optional, cast +from typing import Optional from PIL import Image @@ -56,16 +56,14 @@ def version(): def jpeg_passthrough_available() -> bool: """Returns True if the installed version of Ghostscript supports JPEG passthru - Prior to 9.23, Ghostscript decode and re-encoded JPEGs internally. In 9.23 + Prior to 9.23, Ghostscript decoded and re-encoded JPEGs internally. In 9.23 it gained the ability to keep JPEGs unmodified. However, the 9.23 implementation was buggy and would deletes the last two bytes of images in some cases, as reported here. https://bugs.ghostscript.com/show_bug.cgi?id=699216 The issue was fixed for 9.24, hence that is the first version we consider - the feature available. (However, we don't use 9.24 at all, so the first - version that allows JPEG passthrough is 9.25. - + the feature available. (Ghostscript 9.24 has its own problems is blacklisted.) """ return version() >= '9.24' From f687180ecc3561ee129ee4991755f7d6b3bea02f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Jan 2021 15:04:52 -0800 Subject: [PATCH 755/880] tests: tidy pdfinfo --- tests/test_pdfinfo.py | 48 ++++++++++++++++++++++--------------------- 1 file changed, 25 insertions(+), 23 deletions(-) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 4e6e53ce..90e75ac3 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -4,14 +4,15 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. - import pickle +from io import BytesIO from math import isclose import img2pdf import pikepdf import pytest from PIL import Image +from reportlab.lib.units import inch from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo @@ -19,17 +20,15 @@ from ocrmypdf.exceptions import InputFileError from ocrmypdf.pdfinfo import Colorspace, Encoding from ocrmypdf.pdfinfo.layout import PDFPage -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api - # pylint: disable=protected-access def test_single_page_text(outdir): filename = outdir / 'text.pdf' - pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) + pdf = Canvas(str(filename), pagesize=(8 * inch, 6 * inch)) text = pdf.beginText() text.setFont('Helvetica', 12) - text.setTextOrigin(1 * 72, 3 * 72) + text.setTextOrigin(1 * inch, 3 * inch) text.textLine( "Methink'st thou art a general offence and every" " man should beat thee." ) @@ -46,25 +45,32 @@ def test_single_page_text(outdir): assert len(page.images) == 0 -def test_single_page_image(outdir): - filename = outdir / 'image-mono.pdf' - - im_tmp = outdir / 'tmp.png' +@pytest.fixture(scope='session') +def eight_by_eight(): im = Image.new('1', (8, 8), 0) for n in range(8): im.putpixel((n, n), 1) - im.save(str(im_tmp), format='PNG') + return im + + +def test_single_page_image(eight_by_eight, outpdf): + im = eight_by_eight + bio = BytesIO() + im.save(bio, format='PNG') + bio.seek(0) imgsize = ((img2pdf.ImgSize.dpi, 8), (img2pdf.ImgSize.dpi, 8)) layout_fun = img2pdf.get_layout_fun(None, imgsize, None, None, None) - im_bytes = im_tmp.read_bytes() - pdf_bytes = img2pdf.convert( - im_bytes, producer="img2pdf", with_pdfrw=False, layout_fun=layout_fun - ) - filename.write_bytes(pdf_bytes) - - info = pdfinfo.PdfInfo(filename) + with outpdf.open('wb') as f: + img2pdf.convert( + bio, + producer="img2pdf", + with_pdfrw=False, + layout_fun=layout_fun, + outputstream=f, + ) + info = pdfinfo.PdfInfo(outpdf) assert len(info) == 1 page = info[0] @@ -81,16 +87,12 @@ def test_single_page_image(outdir): assert isclose(pdfimage.dpi.y, 8) -def test_single_page_inline_image(outdir): +def test_single_page_inline_image(eight_by_eight, outdir): filename = outdir / 'image-mono-inline.pdf' pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) - im = Image.new('1', (8, 8), 0) - for n in range(8): - im.putpixel((n, n), 1) - # Draw image in a 72x72 pt or 1"x1" area - pdf.drawInlineImage(im, 0, 0, width=72, height=72) + pdf.drawInlineImage(eight_by_eight, 0, 0, width=72, height=72) pdf.showPage() pdf.save() From b267494e4a38e694178f8f49bba7d0a760a8a847 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 8 Jan 2021 15:10:43 -0800 Subject: [PATCH 756/880] Create raster PDF pages to match input page size Previously we produced a raster image, then multiplied image width by DPI to get the page size. However if there is rounding the page size may not match exactly. In this modified approach we constrain the page size to match. --- src/ocrmypdf/_pipeline.py | 9 +++++++-- tests/test_tesseract.py | 3 ++- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index ab38a76d..97d5b718 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -588,12 +588,17 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext): # except that the hocr renderer does not understand non-square DPI. The # sandwich renderer would be fine. output_file = page_context.get_path('visible.pdf') - dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - layout_fun = img2pdf.get_fixed_dpi_layout_fun(dpi) + + pageinfo = page_context.pageinfo + pagesize = 72.0 * float(pageinfo.width_inches), 72.0 * float(pageinfo.height_inches) + if pageinfo.rotation % 180 == 90: + pagesize = pagesize[1], pagesize[0] # This create a single page PDF with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf: log.debug('convert') + + layout_fun = img2pdf.get_layout_fun(pagesize) img2pdf.convert( imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf ) diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index 57ece9af..f7331e48 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -45,7 +45,8 @@ def test_skip_pages_does_not_replicate(resources, basename, outdir): assert len(page.images) == 1, "skipped page was replicated" for n, info_out_n in enumerate(info): - assert info_out_n.width_inches == info_in[n].width_inches + assert info_out_n.width_inches == info_in[n].width_inches, "output resized" + assert info_out_n.height_inches == info_in[n].height_inches, "output resized" def test_content_preservation(resources, outpdf): From 91aa175602e2b3a878dc4bc2bda8c8865e8a5e4b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Jan 2021 16:01:49 -0800 Subject: [PATCH 757/880] Consider text when determining page raster DPI Previously if we found vectors of any sort on a page, we would bump the DPI up to 400. We did nothing about pages with text. As a result, pages with a low image resolution and printable text would have the text downgraded to image resolution when --force-ocr was used. We don't try to determine if the text is visible or invisible OCR text, since that is a slower test. --redo-ocr would improve such cases anyway. --- src/ocrmypdf/_pipeline.py | 12 +++++--- tests/test_pipeline.py | 63 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 4 deletions(-) create mode 100644 tests/test_pipeline.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 97d5b718..6de1f2e9 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -206,17 +206,21 @@ def validate_pdfinfo_options(context: PdfContext): context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options) +def _vector_page_dpi(pageinfo): + return VECTOR_PAGE_DPI if pageinfo.has_vector or pageinfo.has_text else 0.0 + + def get_page_dpi(pageinfo, options): "Get the DPI when nonsquare DPI is tolerable" xres = max( pageinfo.dpi.x or VECTOR_PAGE_DPI, options.oversample or 0.0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + _vector_page_dpi(pageinfo), ) yres = max( pageinfo.dpi.y or VECTOR_PAGE_DPI, options.oversample or 0, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + _vector_page_dpi(pageinfo), ) return Resolution(float(xres), float(yres)) @@ -230,7 +234,7 @@ def get_page_square_dpi(pageinfo, options) -> Resolution: max( (xres * userunit) or VECTOR_PAGE_DPI, (yres * userunit) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + _vector_page_dpi(pageinfo), options.oversample or 0.0, ) ) @@ -243,7 +247,7 @@ def get_canvas_square_dpi(pageinfo, options) -> Resolution: max( (pageinfo.dpi.x) or VECTOR_PAGE_DPI, (pageinfo.dpi.y) or VECTOR_PAGE_DPI, - VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0, + _vector_page_dpi(pageinfo), options.oversample or 0.0, ) ) diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py new file mode 100644 index 00000000..64254e84 --- /dev/null +++ b/tests/test_pipeline.py @@ -0,0 +1,63 @@ +# © 2021 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +from unittest.mock import Mock + +import pytest +from PIL import Image +from reportlab.lib.units import inch +from reportlab.lib.utils import ImageReader +from reportlab.pdfgen.canvas import Canvas + +from ocrmypdf import _pipeline, pdfinfo +from ocrmypdf.helpers import Resolution + + +@pytest.fixture(scope='session') +def rgb_image(): + im = Image.new('RGB', (8, 8)) + im.putpixel((4, 4), (255, 0, 0)) + im.putpixel((5, 5), (0, 255, 0)) + im.putpixel((6, 6), (0, 0, 255)) + return ImageReader(im) + + +DUMMY_OVERSAMPLE_RESOLUTION = Resolution(42.0, 42.0) +VECTOR_RESOLUTION = Resolution(_pipeline.VECTOR_PAGE_DPI, _pipeline.VECTOR_PAGE_DPI) + + +@pytest.mark.parametrize( + 'image, text, vector, result', + [ + (False, False, False, VECTOR_RESOLUTION), + (False, True, False, VECTOR_RESOLUTION), + (True, False, False, DUMMY_OVERSAMPLE_RESOLUTION), + (True, True, False, VECTOR_RESOLUTION), + (False, False, True, VECTOR_RESOLUTION), + (False, True, True, VECTOR_RESOLUTION), + (True, False, True, VECTOR_RESOLUTION), + (True, True, True, VECTOR_RESOLUTION), + ], +) +def test_dpi_needed(image, text, vector, result, rgb_image, outdir): + + c = Canvas(str(outdir / 'dpi.pdf'), pagesize=(5 * inch, 5 * inch)) + if image: + c.drawImage(rgb_image, 1 * inch, 1 * inch, width=1 * inch, height=1 * inch) + if text: + c.drawString(1 * inch, 4 * inch, "Actual text") + if vector: + c.ellipse(3 * inch, 3 * inch, 4 * inch, 4 * inch) + c.showPage() + c.save() + + mock = Mock() + mock.oversample = DUMMY_OVERSAMPLE_RESOLUTION[0] + + pi = pdfinfo.PdfInfo(outdir / 'dpi.pdf') + + assert _pipeline.get_canvas_square_dpi(pi[0], mock) == result + assert _pipeline.get_page_square_dpi(pi[0], mock) == result From c7c447be66cb51b9a389901e66a56d738512fad8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Jan 2021 16:02:12 -0800 Subject: [PATCH 758/880] Add test for configure_debug_logging Since we can't directly test it --- src/ocrmypdf/_sync.py | 6 ++++-- tests/test_logging.py | 21 +++++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) create mode 100644 tests/test_logging.py diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 83de4150..af32ad8c 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -302,7 +302,7 @@ def exec_concurrent(context: PdfContext): copy_final(pdf, options.output_file, context) -def configure_debug_logging(log_filename, prefix: str = ''): +def configure_debug_logging(log_filename: Path, prefix: str = ''): """ Create a debug log file at a specified location. @@ -340,7 +340,9 @@ def run_pipeline(options, *, plugin_manager, api=False): # Debug log for command line interface only with verbose output # See https://github.com/pytest-dev/pytest/issues/5502 for why we skip this # when pytest is running - debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log") + debug_log_handler = configure_debug_logging( + Path(work_folder) / "debug.log" + ) # pragma: no cover pikepdf_enable_mmap() diff --git a/tests/test_logging.py b/tests/test_logging.py new file mode 100644 index 00000000..5fac42b4 --- /dev/null +++ b/tests/test_logging.py @@ -0,0 +1,21 @@ +# © 2021 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +import logging + +import pytest + +from ocrmypdf._sync import configure_debug_logging + + +def test_debug_logging(tmp_path): + # Just exercise the debug logger but don't validate it + # See https://github.com/pytest-dev/pytest/issues/5502 for pytest logging quirks + prefix = 'test_debug_logging' + log = logging.getLogger(prefix) + handler = configure_debug_logging(tmp_path, prefix) + log.info("test message") + log.removeHandler(handler) From ebacff1b3915435365b9391e768f4547b558081d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Jan 2021 16:41:57 -0800 Subject: [PATCH 759/880] tests: Fix debug logging test --- tests/test_logging.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_logging.py b/tests/test_logging.py index 5fac42b4..6bbd551f 100644 --- a/tests/test_logging.py +++ b/tests/test_logging.py @@ -16,6 +16,6 @@ def test_debug_logging(tmp_path): # See https://github.com/pytest-dev/pytest/issues/5502 for pytest logging quirks prefix = 'test_debug_logging' log = logging.getLogger(prefix) - handler = configure_debug_logging(tmp_path, prefix) + handler = configure_debug_logging(tmp_path / 'test.log', prefix) log.info("test message") log.removeHandler(handler) From 7a1cccbc4e098762fb0a6303cf5193b94a503985 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 3 Jan 2021 01:51:57 -0800 Subject: [PATCH 760/880] Fallback to LeptonicaErrorTrap_Redirect if ffi.callback fails Might fix issue #709, Apple silicon support. --- src/ocrmypdf/leptonica.py | 33 ++++++++++++++++++--------------- 1 file changed, 18 insertions(+), 15 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index af69129f..69f6e444 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -13,6 +13,7 @@ import argparse import logging import os +import platform import sys import threading import warnings @@ -170,20 +171,6 @@ tls = threading.local() tls.trap = None -@ffi.callback("void(char *)") -def _stderr_handler(cstr): - msg = ffi.string(cstr).decode(errors='replace') - if msg.startswith("Error"): - logger.error(msg) - elif msg.startswith("Warning"): - logger.warning(msg) - else: - logger.debug(msg) - if tls.trap is not None: - tls.trap.append(msg) - return - - class _LeptonicaErrorTrap_Queue: def __init__(self): self.queue = deque() @@ -213,9 +200,25 @@ class _LeptonicaErrorTrap_Queue: try: + + @ffi.callback("void(char *)") + def _stderr_handler(cstr): + msg = ffi.string(cstr).decode(errors='replace') + if msg.startswith("Error"): + logger.error(msg) + elif msg.startswith("Warning"): + logger.warning(msg) + else: + logger.debug(msg) + if tls.trap is not None: + tls.trap.append(msg) + return + lept.leptSetStderrHandler(_stderr_handler) -except ffi.error: +except (ffi.error, MemoryError): # Pre-1.79 Leptonica does not have leptSetStderrHandler + # And some platforms, notably Apple ARM 64, do not allow the write+execute + # memory needed to set up the callback function. _LeptonicaErrorTrap = _LeptonicaErrorTrap_Redirect else: # 1.79 have this new symbol From 1ebf3144afd0ba454551e043d71523fb7f46182f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 9 Jan 2021 16:46:15 -0800 Subject: [PATCH 761/880] v11.5.0 release notes --- docs/release_notes.rst | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 7260b903..173473d7 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,21 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.5.0 +======= + +- Fixed an issue where the output page size might differ by a fractional amount + due to rounding, when ``--force-ocr`` was used and the page contained objects + with multiple resolutions. +- When determining the resolution at which to rasterize a page, we now consider + printed text on the page as requiring a higher resolution. This fixes issues + with certain pages being rendered with unacceptably low resolution text, but + may increase output file sizes in some workflows where low resolution text + is acceptable. +- Added a workaround to fix an exception that occurs when trying to + ``import ocrmypdf.leptonica`` on Apple ARM silicon (or potentially, other + platforms that do not permit write+executable memory). + v11.4.5 ======= From ce66bcc9c84b224bb557a8265f6332b317771ab5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 00:40:47 -0800 Subject: [PATCH 762/880] github: Ask how ocrmypdf was installed --- .github/ISSUE_TEMPLATE/1-general-issues.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/.github/ISSUE_TEMPLATE/1-general-issues.md b/.github/ISSUE_TEMPLATE/1-general-issues.md index 1e58db9e..3aa3c2d6 100644 --- a/.github/ISSUE_TEMPLATE/1-general-issues.md +++ b/.github/ISSUE_TEMPLATE/1-general-issues.md @@ -21,8 +21,12 @@ If applicable, add screenshots to help explain your problem. **System (please complete the following information):** - OS: - - Python version: + - Python version: - OCRmyPDF version: +**Installation** +How did you install OCRmyPDF? Did you install it from your operating system's +package manager, or using pip? + **Additional context** Add any other context about the problem here. From 4879a1f0ded5e1e4cbeb98e6f6f1b4f4943ee4be Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 18 Jan 2021 13:27:31 -0800 Subject: [PATCH 763/880] docs: no MS Store Python --- docs/installation.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/installation.rst b/docs/installation.rst index 6bec3e20..4fc3738a 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -533,6 +533,12 @@ override the versions OCRmyPDF selects, you can modify the ``PATH`` environment variable. `Follow these directions `_ to change the PATH. +.. warning:: + + As of early 2021, users have reported problems with the Microsoft Store version of + Python affected most third party Python packages including OCRmyPDF. Please use + Python downloaded from Python.org or Chocolatey as recommended here. + Windows Subsystem for Linux --------------------------- From 1a982da442a91bea8a50b0d67b4de60820be1b9e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 01:53:36 -0800 Subject: [PATCH 764/880] tests: confirm that we produce pdf when optimization is off --- tests/test_optimize.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index d5ffc5fd..e49e4878 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -173,3 +173,15 @@ def test_multiple_pngs(resources, outdir): inim = next(iter(inpdf.pages[n].images.values())) outim = next(iter(outpdf.pages[n].images.values())) assert len(outim.read_raw_bytes()) < len(inim.read_raw_bytes()), n + + +def test_optimize_off(resources, outpdf): + check_ocrmypdf( + resources / 'trivial.pdf', + outpdf, + '--optimize=0', + '--output-type', + 'pdf', + '--plugin', + 'tests/plugins/tesseract_noop.py', + ) From 956310d1ecc69c9eca121560aa35533daf0c55d6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 02:04:47 -0800 Subject: [PATCH 765/880] Import PageContext, PdfContext since they are referenced in pluginspec --- src/ocrmypdf/__init__.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index c7852fb6..081324e9 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -8,6 +8,7 @@ from pluggy import HookimplMarker as _HookimplMarker from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo +from ocrmypdf._jobcontext import PageContext, PdfContext from ocrmypdf._version import PROGRAM_NAME, __version__ from ocrmypdf.api import Verbosity, configure_logging, ocr from ocrmypdf.exceptions import ( From 9ff627472b24dfab4ef33a4cc06faccc98298dc3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 21:12:28 -0800 Subject: [PATCH 766/880] Update pre-commit --- .pre-commit-config.yaml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 53d72257..3cd240e2 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,6 @@ repos: - repo: https://github.com/pre-commit/pre-commit-hooks - rev: v3.1.0 + rev: v3.4.0 hooks: - id: check-case-conflict - id: check-merge-conflict @@ -12,11 +12,11 @@ repos: hooks: - id: seed-isort-config - repo: https://github.com/pre-commit/mirrors-isort - rev: v5.0.5 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases + rev: v5.7.0 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases hooks: - id: isort - repo: https://github.com/psf/black - rev: 19.10b0 + rev: 20.8b1 hooks: - id: black language_version: python From 084610c242be607d5cf586146f0829c71b3a7951 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 02:16:50 -0800 Subject: [PATCH 767/880] Automate insertion of builtin modules --- src/ocrmypdf/_plugin_manager.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index e939e064..392f959f 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -8,6 +8,7 @@ import argparse import importlib import importlib.util +import pkgutil import sys from functools import partial from pathlib import Path @@ -15,6 +16,7 @@ from typing import Callable, List, Tuple, Union import pluggy +import ocrmypdf.builtin_plugins from ocrmypdf import pluginspec from ocrmypdf.cli import get_parser, plugins_only_parser @@ -62,12 +64,11 @@ def _setup_plugins( all_plugins: List[Union[str, Path]] = [] if builtins: all_plugins.extend( - [ - 'ocrmypdf.builtin_plugins.ghostscript', - 'ocrmypdf.builtin_plugins.tesseract_ocr', - ] + f'ocrmypdf.builtin_plugins.{module.name}' + for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__) ) all_plugins.extend(plugins) + for name in all_plugins: if isinstance(name, Path) or name.endswith('.py'): # Import by filename From f559316881befc67d389530599d0bbd10f31019e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 02:17:12 -0800 Subject: [PATCH 768/880] Insert setuptools plugins with ocrmypdf prefix --- src/ocrmypdf/_plugin_manager.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 392f959f..a8011243 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -82,6 +82,8 @@ def _setup_plugins( module = importlib.import_module(name) pm.register(module) + pm.load_setuptools_entrypoints('ocrmypdf') + def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( From ee23976858418a464956fcd48c719d18f36816d4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 02:21:54 -0800 Subject: [PATCH 769/880] Re-sequence plugin installation --- src/ocrmypdf/_plugin_manager.py | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index a8011243..7c4c2670 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -61,15 +61,18 @@ def _setup_plugins( ): pm.add_hookspecs(pluginspec) - all_plugins: List[Union[str, Path]] = [] + # 1. Register builtins if builtins: - all_plugins.extend( - f'ocrmypdf.builtin_plugins.{module.name}' - for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__) - ) - all_plugins.extend(plugins) + for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__): + name = f'ocrmypdf.builtin_plugins.{module.name}' + module = importlib.import_module(name) + pm.register(module) - for name in all_plugins: + # 2. Register setuptools plugins + pm.load_setuptools_entrypoints('ocrmypdf') + + # 3. Register plugins specified on command line + for name in plugins: if isinstance(name, Path) or name.endswith('.py'): # Import by filename module_name = Path(name).stem @@ -82,8 +85,6 @@ def _setup_plugins( module = importlib.import_module(name) pm.register(module) - pm.load_setuptools_entrypoints('ocrmypdf') - def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( From 504d5776d2118bfea6bddcc2b7407604378c43fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 14:10:59 -0800 Subject: [PATCH 770/880] Refactor plugin manager to eliminate callback --- src/ocrmypdf/_plugin_manager.py | 77 ++++++++++++++++----------------- 1 file changed, 38 insertions(+), 39 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 7c4c2670..510ae65b 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -32,64 +32,63 @@ class OcrmypdfPluginManager(pluggy.PluginManager): """ def __init__( - self, *args, setup_func: Callable[[pluggy.PluginManager], None], **kwargs + self, *args, plugins: List[Union[str, Path]], builtins: bool = True, **kwargs, ): - self._init_args = args - self._setup_func = setup_func - self._init_kwargs = kwargs + self.__init_args = args + self.__init_kwargs = kwargs + self.__plugins = plugins + self.__builtins = builtins super().__init__(*args, **kwargs) - setup_func(self) + self.setup_plugins() def __getstate__(self): state = dict( - _init_args=self._init_args, - _setup_func=self._setup_func, - _init_kwargs=self._init_kwargs, + init_args=self.__init_args, + plugins=self.__plugins, + builtins=self.__builtins, + init_kwargs=self.__init_kwargs, ) return state def __setstate__(self, state): self.__init__( - *state['_init_args'], - setup_func=state['_setup_func'], - **state['_init_kwargs'], + *state['init_args'], + plugins=state['plugins'], + builtins=state['builtins'], + **state['init_kwargs'], ) + def setup_plugins(self): + self.add_hookspecs(pluginspec) -def _setup_plugins( - pm: pluggy.PluginManager, plugins: List[Union[str, Path]], builtins: bool = True -): - pm.add_hookspecs(pluginspec) + # 1. Register builtins + if self.__builtins: + for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__): + name = f'ocrmypdf.builtin_plugins.{module.name}' + module = importlib.import_module(name) + self.register(module) - # 1. Register builtins - if builtins: - for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__): - name = f'ocrmypdf.builtin_plugins.{module.name}' - module = importlib.import_module(name) - pm.register(module) + # 2. Register setuptools plugins + self.load_setuptools_entrypoints('ocrmypdf') - # 2. Register setuptools plugins - pm.load_setuptools_entrypoints('ocrmypdf') - - # 3. Register plugins specified on command line - for name in plugins: - if isinstance(name, Path) or name.endswith('.py'): - # Import by filename - module_name = Path(name).stem - spec = importlib.util.spec_from_file_location(module_name, name) - module = importlib.util.module_from_spec(spec) - sys.modules[module_name] = module - spec.loader.exec_module(module) - else: - # Import by dotted module name - module = importlib.import_module(name) - pm.register(module) + # 3. Register plugins specified on command line + for name in self.__plugins: + if isinstance(name, Path) or name.endswith('.py'): + # Import by filename + module_name = Path(name).stem + spec = importlib.util.spec_from_file_location(module_name, name) + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + spec.loader.exec_module(module) + else: + # Import by dotted module name + module = importlib.import_module(name) + self.register(module) def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( - project_name='ocrmypdf', - setup_func=partial(_setup_plugins, plugins=plugins, builtins=builtins), + project_name='ocrmypdf', plugins=plugins, builtins=builtins, ) return pm From 34e564cd7de9847c63093b8d3601968341b22148 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 17 Dec 2020 01:57:05 -0800 Subject: [PATCH 771/880] Use queue.Queue instead of multiprocessing.Queue in threaded mode --- src/ocrmypdf/_concurrent.py | 39 +++++++++++++++++++++++++------------ 1 file changed, 27 insertions(+), 12 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 177ffe0a..f38150dd 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -9,20 +9,23 @@ import logging import logging.handlers import multiprocessing import os +import queue import signal import sys import threading from contextlib import suppress from multiprocessing import Pool as ProcessPool from multiprocessing.dummy import Pool as ThreadPool -from typing import Callable, Iterable, Optional +from typing import Callable, Iterable, Optional, Union from tqdm import tqdm from ocrmypdf.exceptions import InputFileError +Queue = Union[multiprocessing.Queue, queue.Queue] -def log_listener(queue): + +def log_listener(q: Queue): """Listen to the worker processes and forward the messages to logging For simplicity this is a thread rather than a process. Only one process @@ -34,7 +37,7 @@ def log_listener(queue): while True: try: - record = queue.get() + record = q.get() if record is None: break logger = logging.getLogger(record.name) @@ -50,7 +53,7 @@ def process_sigbus(*args): raise InputFileError("A worker process lost access to an input file") -def process_init(queue, user_init, loglevel): +def process_init(q: Queue, user_init: Callable[[], None], loglevel): """Initialize a process pool worker""" # Ignore SIGINT (our parent process will kill us gracefully) @@ -61,23 +64,24 @@ def process_init(queue, user_init, loglevel): signal.signal(signal.SIGBUS, process_sigbus) # Reconfigure the root logger for this process to send all messages to a queue - h = logging.handlers.QueueHandler(queue) + h = logging.handlers.QueueHandler(q) root = logging.getLogger() root.setLevel(loglevel) root.handlers = [] root.addHandler(h) - if user_init: - user_init() + user_init() + return -def thread_init(_queue, user_init, _loglevel): +def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel): # As a thread, block SIGBUS so the main thread deals with it... with suppress(AttributeError): # Windows and Cygwin do not have pthread_sigmask or SIGBUS signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) - if user_init: - user_init() + + user_init() + return def exec_progress_pool( @@ -90,15 +94,26 @@ def exec_progress_pool( task_arguments: Optional[Iterable] = None, task_finished: Optional[Callable] = None, ): - log_queue: multiprocessing.Queue = multiprocessing.Queue(-1) - listener = threading.Thread(target=log_listener, args=(log_queue,)) if use_threads: + log_queue = queue.Queue(-1) pool_class = ThreadPool initializer = thread_init else: + log_queue = multiprocessing.Queue(-1) pool_class = ProcessPool initializer = process_init + + if not task_initializer: + + def _noop(): + return + + task_initializer = _noop + + # Regardless of whether we use_threads for worker processes, the log_listener + # must be a thread + listener = threading.Thread(target=log_listener, args=(log_queue,)) listener.start() with tqdm(**tqdm_kwargs) as pbar: From 26b4d9bb4b4bf508ed8d22419eff8a8e64512c67 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 22 Dec 2020 00:44:59 -0800 Subject: [PATCH 772/880] Refactor concurrency so that it is pluggable However, this may not be the best idea because it involves global state that could be overridden by a parallel call to ocrmypdf.ocr. --- src/ocrmypdf/_concurrent.py | 152 +++--------------- src/ocrmypdf/_plugin_manager.py | 10 +- src/ocrmypdf/_sync.py | 6 +- src/ocrmypdf/builtin_plugins/concurrency.py | 164 ++++++++++++++++++++ src/ocrmypdf/pdfinfo/info.py | 2 +- src/ocrmypdf/pluginspec.py | 43 ++++- 6 files changed, 242 insertions(+), 135 deletions(-) create mode 100644 src/ocrmypdf/builtin_plugins/concurrency.py diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index f38150dd..8dc3a6c3 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -4,84 +4,20 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. - -import logging -import logging.handlers -import multiprocessing -import os -import queue -import signal -import sys -import threading -from contextlib import suppress -from multiprocessing import Pool as ProcessPool -from multiprocessing.dummy import Pool as ThreadPool -from typing import Callable, Iterable, Optional, Union - -from tqdm import tqdm - -from ocrmypdf.exceptions import InputFileError - -Queue = Union[multiprocessing.Queue, queue.Queue] +from typing import Callable, Iterable, Optional -def log_listener(q: Queue): - """Listen to the worker processes and forward the messages to logging - - For simplicity this is a thread rather than a process. Only one process - should actually write to sys.stderr or whatever we're using, so if this is - made into a process the main application needs to be directed to it. - - See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes - """ - - while True: - try: - record = q.get() - if record is None: - break - logger = logging.getLogger(record.name) - logger.handle(record) - except Exception: # pylint: disable=broad-except - import traceback # pylint: disable=import-outside-toplevel - - print("Logging problem", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - - -def process_sigbus(*args): - raise InputFileError("A worker process lost access to an input file") - - -def process_init(q: Queue, user_init: Callable[[], None], loglevel): - """Initialize a process pool worker""" - - # Ignore SIGINT (our parent process will kill us gracefully) - signal.signal(signal.SIGINT, signal.SIG_IGN) - - # Install SIGBUS handler (so our parent process can abort somewhat gracefully) - with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS - signal.signal(signal.SIGBUS, process_sigbus) - - # Reconfigure the root logger for this process to send all messages to a queue - h = logging.handlers.QueueHandler(q) - root = logging.getLogger() - root.setLevel(loglevel) - root.handlers = [] - root.addHandler(h) - - user_init() +def _task_noop(*_args, **_kwargs): return -def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel): - # As a thread, block SIGBUS so the main thread deals with it... - with suppress(AttributeError): - # Windows and Cygwin do not have pthread_sigmask or SIGBUS - signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) +def _model(**_kwargs): + raise RuntimeError("Parallel executor not set up") - user_init() - return + +def set_execution_model(model): + global _model + _model = model def exec_progress_pool( @@ -89,64 +25,22 @@ def exec_progress_pool( use_threads: bool, max_workers: int, tqdm_kwargs: dict, - task_initializer: Optional[Callable] = None, - task: Optional[Callable] = None, + worker_initializer: Optional[Callable] = None, + task: Callable, task_arguments: Optional[Iterable] = None, task_finished: Optional[Callable] = None, ): + if not worker_initializer: + worker_initializer = _task_noop + if not task_finished: + task_finished = _task_noop - if use_threads: - log_queue = queue.Queue(-1) - pool_class = ThreadPool - initializer = thread_init - else: - log_queue = multiprocessing.Queue(-1) - pool_class = ProcessPool - initializer = process_init - - if not task_initializer: - - def _noop(): - return - - task_initializer = _noop - - # Regardless of whether we use_threads for worker processes, the log_listener - # must be a thread - listener = threading.Thread(target=log_listener, args=(log_queue,)) - listener.start() - - with tqdm(**tqdm_kwargs) as pbar: - pool = pool_class( - processes=max_workers, - initializer=initializer, - initargs=(log_queue, task_initializer, logging.getLogger("").level), - ) - try: - results = pool.imap_unordered(task, task_arguments) - for result in results: - if task_finished: - task_finished(result, pbar) - else: - pbar.update() - except KeyboardInterrupt: - # Terminate pool so we exit instantly - pool.terminate() - # Don't try listener.join() here, will deadlock - raise - except Exception: - if not os.environ.get("PYTEST_CURRENT_TEST", ""): - # Unless inside pytest, exit immediately because no one wants - # to wait for child processes to finalize results that will be - # thrown away. Inside pytest, we want child processes to exit - # cleanly so that they output an error messages or coverage data - # we need from them. - pool.terminate() - raise - finally: - # Terminate log listener - log_queue.put_nowait(None) - pool.close() - pool.join() - - listener.join() + _model( + use_threads=use_threads, + max_workers=max_workers, + tqdm_kwargs=tqdm_kwargs, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + ) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 510ae65b..76e12bd7 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -32,7 +32,11 @@ class OcrmypdfPluginManager(pluggy.PluginManager): """ def __init__( - self, *args, plugins: List[Union[str, Path]], builtins: bool = True, **kwargs, + self, + *args, + plugins: List[Union[str, Path]], + builtins: bool = True, + **kwargs, ): self.__init_args = args self.__init_kwargs = kwargs @@ -88,7 +92,9 @@ class OcrmypdfPluginManager(pluggy.PluginManager): def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( - project_name='ocrmypdf', plugins=plugins, builtins=builtins, + project_name='ocrmypdf', + plugins=plugins, + builtins=builtins, ) return pm diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index af32ad8c..1d78f1d9 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -18,7 +18,7 @@ from typing import List, NamedTuple, Optional, Tuple import pikepdf import PIL -from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._concurrent import exec_progress_pool, set_execution_model from ocrmypdf._graft import OcrGrafter from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files from ocrmypdf._logging import PageNumberFilter @@ -279,7 +279,7 @@ def exec_concurrent(context: PdfContext): unit_scale=0.5, disable=not options.progress_bar, ), - task_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS), + worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS), task=exec_page_sync, task_arguments=context.get_page_contexts(), task_finished=update_page, @@ -346,6 +346,8 @@ def run_pipeline(options, *, plugin_manager, api=False): pikepdf_enable_mmap() + set_execution_model(plugin_manager.hook.get_parallel_executor()) + try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py new file mode 100644 index 00000000..6cf56e42 --- /dev/null +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -0,0 +1,164 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + + +import logging +import logging.handlers +import multiprocessing +import os +import queue +import signal +import sys +import threading +from contextlib import suppress +from multiprocessing import Pool as ProcessPool +from multiprocessing.dummy import Pool as ThreadPool +from typing import Callable, Iterable, Optional, Union + +from tqdm import tqdm + +from ocrmypdf import hookimpl +from ocrmypdf.exceptions import InputFileError + +Queue = Union[multiprocessing.Queue, queue.Queue] + + +def log_listener(q: Queue): + """Listen to the worker processes and forward the messages to logging + + For simplicity this is a thread rather than a process. Only one process + should actually write to sys.stderr or whatever we're using, so if this is + made into a process the main application needs to be directed to it. + + See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes + """ + + while True: + try: + record = q.get() + if record is None: + break + logger = logging.getLogger(record.name) + logger.handle(record) + except Exception: # pylint: disable=broad-except + import traceback # pylint: disable=import-outside-toplevel + + print("Logging problem", file=sys.stderr) + traceback.print_exc(file=sys.stderr) + + +def process_sigbus(*args): + raise InputFileError("A worker process lost access to an input file") + + +def process_init(q: Queue, user_init: Callable[[], None], loglevel): + """Initialize a process pool worker""" + + # Ignore SIGINT (our parent process will kill us gracefully) + signal.signal(signal.SIGINT, signal.SIG_IGN) + + # Install SIGBUS handler (so our parent process can abort somewhat gracefully) + with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS + # Windows and Cygwin do not have pthread_sigmask or SIGBUS + signal.signal(signal.SIGBUS, process_sigbus) + + # Reconfigure the root logger for this process to send all messages to a queue + h = logging.handlers.QueueHandler(q) + root = logging.getLogger() + root.setLevel(loglevel) + root.handlers = [] + root.addHandler(h) + + user_init() + return + + +def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel): + # As a thread, block SIGBUS so the main thread deals with it... + with suppress(AttributeError): + signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) + + user_init() + return + + +def exec_progress_pool( + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Optional[Iterable] = None, + task_finished: Callable, +): + + if use_threads: + log_queue = queue.Queue(-1) + pool_class = ThreadPool + initializer = thread_init + else: + log_queue = multiprocessing.Queue(-1) + pool_class = ProcessPool + initializer = process_init + + if not worker_initializer: + + def _noop(): + return + + worker_initializer = _noop + + # Regardless of whether we use_threads for worker processes, the log_listener + # must be a thread + listener = threading.Thread(target=log_listener, args=(log_queue,)) + listener.start() + + with tqdm(**tqdm_kwargs) as pbar: + pool = pool_class( + processes=max_workers, + initializer=initializer, + initargs=(log_queue, worker_initializer, logging.getLogger("").level), + ) + try: + results = pool.imap_unordered(task, task_arguments) + for result in results: + if task_finished: + task_finished(result, pbar) + else: + pbar.update() + except KeyboardInterrupt: + # Terminate pool so we exit instantly + pool.terminate() + # Don't try listener.join() here, will deadlock + raise + except Exception: + if not os.environ.get("PYTEST_CURRENT_TEST", ""): + # Unless inside pytest, exit immediately because no one wants + # to wait for child processes to finalize results that will be + # thrown away. Inside pytest, we want child processes to exit + # cleanly so that they output an error messages or coverage data + # we need from them. + pool.terminate() + raise + finally: + # Terminate log listener + log_queue.put_nowait(None) + pool.close() + pool.join() + + listener.join() + + +@hookimpl +def get_parallel_executor(): + return exec_progress_pool diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index ae02fb98..a11e64cf 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -623,7 +623,7 @@ def _pdf_pageinfo_concurrent( tqdm_kwargs=dict( total=total, desc="Scanning contents", unit='page', disable=not progbar ), - task_initializer=partial( + worker_initializer=partial( _pdf_pageinfo_sync_init, infile, logging.getLogger('pdfminer').level ), task=_pdf_pageinfo_sync, diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 9aee6596..538be7a1 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -9,7 +9,7 @@ from abc import ABC, abstractmethod, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple from pathlib import Path -from typing import TYPE_CHECKING, AbstractSet, List, Optional +from typing import TYPE_CHECKING, AbstractSet, Callable, Iterable, List, Optional import pluggy @@ -62,6 +62,47 @@ def check_options(options: Namespace) -> None: """ +class ParallelExecutor(ABC): + @abstractstaticmethod + def __call__( + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_finished: Callable, + task_arguments: Optional[Iterable] = None, + ): + """ + Args: + use_threads: If False, the workload is the sort that will benefit from + running in a multiprocessing context (for example, it uses Python + heavily, and parallelizing it with threads is not expected to be + performant). + max_workers: The maximum number of workers that should be run. + tdqm_kwargs: Arguments to set up the progress bar. + worker_initializer: Called when the worker is initialized, in the worker's + execution context. Must be possible to marshall to the worker. + task: Called when the worker starts a new task, in the worker's execution + context. Must be possible to marshallable to the worker. + task_finished: Called when a worker finishes a task, in the parent's + context. + task_arguments: An iterable that generates a group of parameters for each + task. This runs in the parent's context, but the parameters must be + marshallable to the worker. + """ + + +@hookspec(firstresult=True) +def get_parallel_executor() -> Callable: + """Called to perform parallel execution + + This may be used to replace OCRmyPDF's default parallel execution system + with a third party alternative. For example, you could make OCRmyPDF run in a + distributed environment. + """ + + @hookspec def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None: """Called to give a plugin an opportunity to review *options* and *pdfinfo*. From 6953f324653ea4b5903e6137a142578733df7c70 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 10 Jan 2021 14:20:28 -0800 Subject: [PATCH 773/880] pdfinfo: remove some messy concurrency handling We can cut down on the use of global variables and save opening an extra copy of the Pdf when threaded. --- src/ocrmypdf/_concurrent.py | 2 +- src/ocrmypdf/pdfinfo/info.py | 68 +++++++++++++++++++++--------------- 2 files changed, 40 insertions(+), 30 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 8dc3a6c3..87c9656c 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -11,7 +11,7 @@ def _task_noop(*_args, **_kwargs): return -def _model(**_kwargs): +def _model(**_kwargs) -> None: raise RuntimeError("Parallel executor not set up") diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index a11e64cf..1ee82cf9 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -6,6 +6,7 @@ # file, You can obtain one at http://mozilla.org/MPL/2.0/. +import atexit import logging import re from collections import defaultdict, namedtuple @@ -571,29 +572,33 @@ def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]: worker_pdf = None -def _pdf_pageinfo_sync_init(infile: Path, pdfminer_loglevel): +def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel): global worker_pdf # pylint: disable=global-statement pikepdf_enable_mmap() logging.getLogger('pdfminer').setLevel(pdfminer_loglevel) - # If this function is called as a thread initializer, we need a messy hack - # to close worker_pdf. If called as a process, it will be released when the - # process is terminated. - worker_pdf = pikepdf.open(infile) + # If the pdf is not opened, open a copy for our worker process to use + if pdf is None: + worker_pdf = pikepdf.open(infile) + + def on_process_close(): + worker_pdf.close() + + # Close when this process exits + atexit.register(on_process_close) def _pdf_pageinfo_sync(args): - global worker_pdf # pylint: disable=global-statement - pageno, infile, check_pages, detailed_analysis = args - page = PageInfo(worker_pdf, pageno, infile, check_pages, detailed_analysis) + pageno, thread_pdf, infile, check_pages, detailed_analysis = args + pdf = thread_pdf if thread_pdf is not None else worker_pdf + page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis) return page def _pdf_pageinfo_concurrent( pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False ): - global worker_pdf # pylint: disable=global-statement pages = [None] * len(pdf.pages) def update_pageinfo(result, pbar): @@ -607,7 +612,6 @@ def _pdf_pageinfo_concurrent( max_workers = available_cpu_count() total = len(pdf.pages) - contexts = ((n, infile, check_pages, detailed_analysis) for n in range(total)) use_threads = False # No performance gain if threaded due to GIL n_workers = min(1 + len(pages) // 4, max_workers) @@ -616,25 +620,31 @@ def _pdf_pageinfo_concurrent( # a separate process. use_threads = True - try: - exec_progress_pool( - use_threads=use_threads, - max_workers=n_workers, - tqdm_kwargs=dict( - total=total, desc="Scanning contents", unit='page', disable=not progbar - ), - worker_initializer=partial( - _pdf_pageinfo_sync_init, infile, logging.getLogger('pdfminer').level - ), - task=_pdf_pageinfo_sync, - task_arguments=contexts, - task_finished=update_pageinfo, - ) - finally: - if worker_pdf and use_threads: - assert n_workers == 1, "Should have only one worker when threaded" - # This is messy, but if we ran in thread, close worker_pdf - worker_pdf.close() + # If we use a thread, we can pass the already-open Pdf for them to use + # If we use processes, we pass a None which tells the init function to open its + # own + initial_pdf = pdf if use_threads else None + + contexts = ( + (n, initial_pdf, infile, check_pages, detailed_analysis) for n in range(total) + ) + assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable" + exec_progress_pool( + use_threads=use_threads, + max_workers=n_workers, + tqdm_kwargs=dict( + total=total, desc="Scanning contents", unit='page', disable=not progbar + ), + worker_initializer=partial( + _pdf_pageinfo_sync_init, + initial_pdf, + infile, + logging.getLogger('pdfminer').level, + ), + task=_pdf_pageinfo_sync, + task_arguments=contexts, + task_finished=update_pageinfo, + ) return pages From 173c0d12740cfc9a6731a427670e1a9590effa28 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 10 Jan 2021 14:22:17 -0800 Subject: [PATCH 774/880] concurrency: lock progress pool For API sanity and to communicate expectations. One progress pool at a time is plenty of complexity. --- src/ocrmypdf/builtin_plugins/concurrency.py | 32 +++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 6cf56e42..7d7fbbfc 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -21,7 +21,7 @@ import sys import threading from contextlib import suppress from multiprocessing import Pool as ProcessPool -from multiprocessing.dummy import Pool as ThreadPool +from multiprocessing.pool import ThreadPool from typing import Callable, Iterable, Optional, Union from tqdm import tqdm @@ -31,6 +31,8 @@ from ocrmypdf.exceptions import InputFileError Queue = Union[multiprocessing.Queue, queue.Queue] +pool_lock = threading.Lock() + def log_listener(q: Queue): """Listen to the worker processes and forward the messages to logging @@ -96,7 +98,7 @@ def exec_progress_pool( use_threads: bool, max_workers: int, tqdm_kwargs: dict, - worker_initializer: Callable, + worker_initializer: Optional[Callable], task: Callable, task_arguments: Optional[Iterable] = None, task_finished: Callable, @@ -118,6 +120,32 @@ def exec_progress_pool( worker_initializer = _noop + with pool_lock: + _exec_progress_pool( + max_workers=max_workers, + tqdm_kwargs=tqdm_kwargs, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + log_queue=log_queue, + pool_class=pool_class, + initializer=initializer, + ) + + +def _exec_progress_pool( + *, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Optional[Iterable] = None, + task_finished: Callable, + log_queue: Queue, + pool_class: Callable, + initializer: Callable, +): # Regardless of whether we use_threads for worker processes, the log_listener # must be a thread listener = threading.Thread(target=log_listener, args=(log_queue,)) From 7bccb8c74844af7b1b7f1376a2abc3bdce7bb466 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 19 Jan 2021 14:15:07 -0800 Subject: [PATCH 775/880] tests: fix concurrency --- tests/test_validation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_validation.py b/tests/test_validation.py index f7ce5286..deed9769 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -173,7 +173,7 @@ def test_false_action_store_true(): def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) plugin_manager = get_plugin_manager(opts.plugins) - with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: + with patch('ocrmypdf.builtin_plugins.concurrency.tqdm', autospec=True) as tqdmpatch: vd._check_options(opts, plugin_manager, set()) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) assert pdfinfo is not None From 5545bae76f986f8d965c0aa0c4787c71e77d8b30 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 16:19:58 -0800 Subject: [PATCH 776/880] lambda_plugin.py: doesn't work since entry point needs to be in package --- misc/lambda_plugin.py | 192 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 192 insertions(+) create mode 100644 misc/lambda_plugin.py diff --git a/misc/lambda_plugin.py b/misc/lambda_plugin.py new file mode 100644 index 00000000..b39270e4 --- /dev/null +++ b/misc/lambda_plugin.py @@ -0,0 +1,192 @@ +# © 2021 James R Barlow: https://github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +"""Alternate executor to support OCRmyPDF in AWS Lambda""" + + +import logging +import logging.handlers +import multiprocessing +import os +import queue +import signal +import sys +import threading +from contextlib import suppress +from itertools import islice, repeat, takewhile, zip_longest +from multiprocessing import Pipe, Process, process +from multiprocessing.connection import Connection, wait +from typing import Callable, Iterable, Optional, Union +from unittest.mock import Mock + +from ocrmypdf import hookimpl +from ocrmypdf.exceptions import InputFileError + +pool_lock = threading.Lock() + + +def split_every(n: int, iterable: Iterable): + iterator = iter(iterable) + return takewhile(bool, (list(islice(iterator, n)) for _ in repeat(None))) + + +def log_listener(q): + """Listen to the worker processes and forward the messages to logging + + For simplicity this is a thread rather than a process. Only one process + should actually write to sys.stderr or whatever we're using, so if this is + made into a process the main application needs to be directed to it. + + See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes + """ + + while True: + try: + record = q.get() + if record is None: + break + logger = logging.getLogger(record.name) + logger.handle(record) + except Exception: # pylint: disable=broad-except + import traceback # pylint: disable=import-outside-toplevel + + print("Logging problem", file=sys.stderr) + traceback.print_exc(file=sys.stderr) + + +def process_sigbus(*args): + raise InputFileError("A worker process lost access to an input file") + + +def process_loop( + conn: Connection, user_init: Callable[[], None], loglevel, task, task_args +): + """Initialize a process pool worker""" + + # Ignore SIGINT (our parent process will kill us gracefully) + signal.signal(signal.SIGINT, signal.SIG_IGN) + + # Install SIGBUS handler (so our parent process can abort somewhat gracefully) + with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS + # Windows and Cygwin do not have pthread_sigmask or SIGBUS + signal.signal(signal.SIGBUS, process_sigbus) + + # Reconfigure the root logger for this process to send all messages to a queue + # h = logging.handlers.QueueHandler(q) + # root = logging.getLogger() + # root.setLevel(loglevel) + # root.handlers = [] + # root.addHandler(h) + + user_init() + + for args in task_args: + try: + result = task(*args) + except Exception as e: + return # for now + else: + conn.send(result) + + conn.close() + return + + +def exec_progress_pool( + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Optional[Callable], + task: Callable, + task_arguments: Optional[Iterable] = None, + task_finished: Callable, +): + + if not worker_initializer: + + def _noop(): + return + + worker_initializer = _noop + + with pool_lock: + _exec_progress_pool( + max_workers=max_workers, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + ) + + +def _exec_progress_pool( + *, + max_workers: int, + worker_initializer: Callable, + task: Callable, + task_arguments: Optional[Iterable] = None, + task_finished: Callable, +): + pbar = Mock() + + task_arguments = list(task_arguments) + grouped_args = list(zip_longest(*list(split_every(max_workers, task_arguments)))) + + processes = [] + connections = [] + for n in range(max_workers): + parent_conn, child_conn = Pipe() + + worker_args = [args for args in grouped_args[n] if args is not None] + process = Process( + target=process_loop, + args=( + child_conn, + worker_initializer, + logging.getLogger("").level, + task, + worker_args, + ), + ) + process.daemon = True + processes.append(process) + connections.append(parent_conn) + + for process in processes: + process.start() + + while connections: + for r in wait(connections): + try: + msg = r.recv() + except EOFError: + connections.remove(r) + else: + if task_finished: + task_finished(msg, pbar) + + for process in processes: + process.join() + + +@hookimpl +def get_parallel_executor(): + return exec_progress_pool From c6a2716cdbe85ca94cc69216ba94b5b04f29cfcf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 16:21:07 -0800 Subject: [PATCH 777/880] Temporary move into package --- misc/lambda_plugin.py => src/ocrmypdf/_lambda_plugin.py | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename misc/lambda_plugin.py => src/ocrmypdf/_lambda_plugin.py (100%) diff --git a/misc/lambda_plugin.py b/src/ocrmypdf/_lambda_plugin.py similarity index 100% rename from misc/lambda_plugin.py rename to src/ocrmypdf/_lambda_plugin.py From 8d23d0b4414ac955483d433edc568c6b7b0c7804 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 19:18:40 -0800 Subject: [PATCH 778/880] Operational lambda executor --- src/ocrmypdf/_lambda_plugin.py | 73 ++++++++++++++++------------------ 1 file changed, 34 insertions(+), 39 deletions(-) diff --git a/src/ocrmypdf/_lambda_plugin.py b/src/ocrmypdf/_lambda_plugin.py index b39270e4..69b709ac 100644 --- a/src/ocrmypdf/_lambda_plugin.py +++ b/src/ocrmypdf/_lambda_plugin.py @@ -31,7 +31,7 @@ import sys import threading from contextlib import suppress from itertools import islice, repeat, takewhile, zip_longest -from multiprocessing import Pipe, Process, process +from multiprocessing import Pipe, Process from multiprocessing.connection import Connection, wait from typing import Callable, Iterable, Optional, Union from unittest.mock import Mock @@ -47,64 +47,48 @@ def split_every(n: int, iterable: Iterable): return takewhile(bool, (list(islice(iterator, n)) for _ in repeat(None))) -def log_listener(q): - """Listen to the worker processes and forward the messages to logging - - For simplicity this is a thread rather than a process. Only one process - should actually write to sys.stderr or whatever we're using, so if this is - made into a process the main application needs to be directed to it. - - See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes - """ - - while True: - try: - record = q.get() - if record is None: - break - logger = logging.getLogger(record.name) - logger.handle(record) - except Exception: # pylint: disable=broad-except - import traceback # pylint: disable=import-outside-toplevel - - print("Logging problem", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - - def process_sigbus(*args): raise InputFileError("A worker process lost access to an input file") +class ConnectionLogHandler(logging.handlers.QueueHandler): + def __init__(self, conn: Connection) -> None: + super().__init__(None) + self.conn = conn + + def enqueue(self, record): + self.conn.send(('log', record)) + + def process_loop( conn: Connection, user_init: Callable[[], None], loglevel, task, task_args ): """Initialize a process pool worker""" - # Ignore SIGINT (our parent process will kill us gracefully) - signal.signal(signal.SIGINT, signal.SIG_IGN) - # Install SIGBUS handler (so our parent process can abort somewhat gracefully) with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS # Windows and Cygwin do not have pthread_sigmask or SIGBUS signal.signal(signal.SIGBUS, process_sigbus) # Reconfigure the root logger for this process to send all messages to a queue - # h = logging.handlers.QueueHandler(q) - # root = logging.getLogger() - # root.setLevel(loglevel) - # root.handlers = [] - # root.addHandler(h) + h = ConnectionLogHandler(conn) + root = logging.getLogger() + root.setLevel(loglevel) + root.handlers = [] + root.addHandler(h) user_init() for args in task_args: try: - result = task(*args) + result = task(args) except Exception as e: - return # for now + conn.send(('exception', str(e))) + break else: - conn.send(result) + conn.send(('result', result)) + conn.send(('complete', None)) conn.close() return @@ -149,6 +133,8 @@ def _exec_progress_pool( task_arguments = list(task_arguments) grouped_args = list(zip_longest(*list(split_every(max_workers, task_arguments)))) + if not grouped_args: + return processes = [] connections = [] @@ -176,12 +162,21 @@ def _exec_progress_pool( while connections: for r in wait(connections): try: - msg = r.recv() + msg_type, msg = r.recv() except EOFError: connections.remove(r) else: - if task_finished: - task_finished(msg, pbar) + if msg_type == 'result': + if task_finished: + task_finished(msg, pbar) + elif msg_type == 'log': + record = msg + logger = logging.getLogger(record.name) + logger.handle(record) + elif msg_type == 'exception': + print(msg) + elif msg_type == 'complete': + connections.remove(r) for process in processes: process.join() From c395436ba30bc78d529c147bb75aaf18c741fe92 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 20:03:57 -0800 Subject: [PATCH 779/880] lambda: tidying, special casing use_threads --- src/ocrmypdf/_lambda_plugin.py | 104 ++++++++++++++++++--------------- 1 file changed, 56 insertions(+), 48 deletions(-) diff --git a/src/ocrmypdf/_lambda_plugin.py b/src/ocrmypdf/_lambda_plugin.py index 69b709ac..d6a4e4a3 100644 --- a/src/ocrmypdf/_lambda_plugin.py +++ b/src/ocrmypdf/_lambda_plugin.py @@ -23,17 +23,14 @@ import logging import logging.handlers -import multiprocessing -import os -import queue import signal -import sys import threading from contextlib import suppress +from enum import Enum, auto from itertools import islice, repeat, takewhile, zip_longest from multiprocessing import Pipe, Process from multiprocessing.connection import Connection, wait -from typing import Callable, Iterable, Optional, Union +from typing import Callable, Iterable, Optional from unittest.mock import Mock from ocrmypdf import hookimpl @@ -42,6 +39,12 @@ from ocrmypdf.exceptions import InputFileError pool_lock = threading.Lock() +class MessageType(Enum): + exception = auto() + result = auto() + complete = auto() + + def split_every(n: int, iterable: Iterable): iterator = iter(iterable) return takewhile(bool, (list(islice(iterator, n)) for _ in repeat(None))) @@ -83,47 +86,21 @@ def process_loop( try: result = task(args) except Exception as e: - conn.send(('exception', str(e))) + conn.send((MessageType.exception, str(e))) break else: - conn.send(('result', result)) + conn.send((MessageType.result, result)) - conn.send(('complete', None)) + conn.send((MessageType.complete, None)) conn.close() return -def exec_progress_pool( +def lambda_pool_impl( *, use_threads: bool, max_workers: int, tqdm_kwargs: dict, - worker_initializer: Optional[Callable], - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Callable, -): - - if not worker_initializer: - - def _noop(): - return - - worker_initializer = _noop - - with pool_lock: - _exec_progress_pool( - max_workers=max_workers, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - ) - - -def _exec_progress_pool( - *, - max_workers: int, worker_initializer: Callable, task: Callable, task_arguments: Optional[Iterable] = None, @@ -131,6 +108,32 @@ def _exec_progress_pool( ): pbar = Mock() + if use_threads and max_workers == 1: + for args in task_arguments: + result = task(args) + task_finished(result, pbar) + return + + with pool_lock: + _lambda_pool_impl( + max_workers=max_workers, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + pbar=pbar, + ) + + +def _lambda_pool_impl( + *, + max_workers: int, + worker_initializer: Callable, + task: Callable, + task_arguments: Optional[Iterable] = None, + task_finished: Callable, + pbar, +): task_arguments = list(task_arguments) grouped_args = list(zip_longest(*list(split_every(max_workers, task_arguments)))) if not grouped_args: @@ -165,18 +168,23 @@ def _exec_progress_pool( msg_type, msg = r.recv() except EOFError: connections.remove(r) - else: - if msg_type == 'result': - if task_finished: - task_finished(msg, pbar) - elif msg_type == 'log': - record = msg - logger = logging.getLogger(record.name) - logger.handle(record) - elif msg_type == 'exception': - print(msg) - elif msg_type == 'complete': - connections.remove(r) + continue + + if msg_type == MessageType.result: + if task_finished: + task_finished(msg, pbar) + elif msg_type == 'log': + record = msg + logger = logging.getLogger(record.name) + logger.handle(record) + elif msg_type == MessageType.complete: + connections.remove(r) + elif msg_type == MessageType.exception: + logger = logging.getLogger(__name__) + logger.error(msg) + for process in processes: + process.terminate() + raise RuntimeError("Failed") for process in processes: process.join() @@ -184,4 +192,4 @@ def _exec_progress_pool( @hookimpl def get_parallel_executor(): - return exec_progress_pool + return lambda_pool_impl From 1a3ce59476df8f77a9a7fb297963358833516b1d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 20:05:14 -0800 Subject: [PATCH 780/880] lambda: Don't be paranoid about exception marshalling It works --- src/ocrmypdf/_lambda_plugin.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/_lambda_plugin.py b/src/ocrmypdf/_lambda_plugin.py index d6a4e4a3..eec248cb 100644 --- a/src/ocrmypdf/_lambda_plugin.py +++ b/src/ocrmypdf/_lambda_plugin.py @@ -86,7 +86,7 @@ def process_loop( try: result = task(args) except Exception as e: - conn.send((MessageType.exception, str(e))) + conn.send((MessageType.exception, e)) break else: conn.send((MessageType.result, result)) @@ -180,11 +180,9 @@ def _lambda_pool_impl( elif msg_type == MessageType.complete: connections.remove(r) elif msg_type == MessageType.exception: - logger = logging.getLogger(__name__) - logger.error(msg) for process in processes: process.terminate() - raise RuntimeError("Failed") + raise msg for process in processes: process.join() From 6083b4f0a7e71c7547136aaa91b14c0a0309e87e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 20:52:41 -0800 Subject: [PATCH 781/880] lambda: don't overrun number of workers needed --- src/ocrmypdf/_lambda_plugin.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_lambda_plugin.py b/src/ocrmypdf/_lambda_plugin.py index eec248cb..1cab2f61 100644 --- a/src/ocrmypdf/_lambda_plugin.py +++ b/src/ocrmypdf/_lambda_plugin.py @@ -141,10 +141,10 @@ def _lambda_pool_impl( processes = [] connections = [] - for n in range(max_workers): + for chunk in grouped_args: parent_conn, child_conn = Pipe() - worker_args = [args for args in grouped_args[n] if args is not None] + worker_args = [args for args in chunk if args is not None] process = Process( target=process_loop, args=( From 6a8dd65aa28ac87ef8ec38167991de29948829fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 21:16:11 -0800 Subject: [PATCH 782/880] lambda: more issues related to new executor semantics Now all tests pass, except for: -tests that check the progress bar -tests where xdist may or may not load a _lambda_plugin by running some other test first before a test in optimize --- src/ocrmypdf/_exec/jbig2enc.py | 8 ++++++++ src/ocrmypdf/_exec/pngquant.py | 4 ++++ src/ocrmypdf/optimize.py | 33 +++++++++++++++------------------ 3 files changed, 27 insertions(+), 18 deletions(-) diff --git a/src/ocrmypdf/_exec/jbig2enc.py b/src/ocrmypdf/_exec/jbig2enc.py index 9311ec3f..2e8a058b 100644 --- a/src/ocrmypdf/_exec/jbig2enc.py +++ b/src/ocrmypdf/_exec/jbig2enc.py @@ -41,9 +41,17 @@ def convert_group(*, cwd, infiles, out_prefix): return proc +def convert_group_mp(args): + return convert_group(cwd=args[0], infiles=args[1], out_prefix=args[2]) + + def convert_single(*, cwd, infile, outfile): args = ['jbig2', '-p', infile] with open(outfile, 'wb') as fstdout: proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE) proc.check_returncode() return proc + + +def convert_single_mp(args): + return convert_single(cwd=args[0], infile=args[1], outfile=args[2]) diff --git a/src/ocrmypdf/_exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py index d7e345f2..12dc79bc 100644 --- a/src/ocrmypdf/_exec/pngquant.py +++ b/src/ocrmypdf/_exec/pngquant.py @@ -61,3 +61,7 @@ def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: if result.returncode == 0: # input_file could be the same as output_file, so we defer the write output_file.write_bytes(result.stdout) + + +def quantize_mp(args): + return quantize(*args) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index c097e32e..3724f731 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -212,7 +212,10 @@ def extract_image_generic( def extract_images( - pike: Pdf, root: Path, options, extract_fn: Callable[..., Optional[XrefExt]], + pike: Pdf, + root: Path, + options, + extract_fn: Callable[..., Optional[XrefExt]], ) -> Iterator[Tuple[int, XrefExt]]: """Extract image using extract_fn @@ -305,10 +308,10 @@ def _produce_jbig2_images( def jbig2_group_args(root: Path, groups: Dict[int, List[XrefExt]]): for group, xref_exts in groups.items(): prefix = f'group{group:08d}' - yield dict( - cwd=fspath(root), - infiles=(img_name(root, xref, ext) for xref, ext in xref_exts), - out_prefix=prefix, + yield ( + fspath(root), # =cwd + (img_name(root, xref, ext) for xref, ext in xref_exts), # =infiles + prefix, # =out_prefix ) def jbig2_single_args(root, groups: Dict[int, List[XrefExt]]): @@ -317,21 +320,18 @@ def _produce_jbig2_images( # Second loop is to ensure multiple images per page are unpacked for n, xref_ext in enumerate(xref_exts): xref, ext = xref_ext - yield dict( - cwd=fspath(root), - infile=img_name(root, xref, ext), - outfile=root / f'{prefix}.{n:04d}', + yield ( + fspath(root), + img_name(root, xref, ext), + root / f'{prefix}.{n:04d}', ) - def convert_generic(fn, kwargs_dict): - return fn(**kwargs_dict) - if options.jbig2_page_group_size > 1: jbig2_args = jbig2_group_args - jbig2_convert = partial(convert_generic, jbig2enc.convert_group) + jbig2_convert = jbig2enc.convert_group_mp else: jbig2_args = jbig2_single_args - jbig2_convert = partial(convert_generic, jbig2enc.convert_single) + jbig2_convert = jbig2enc.convert_single_mp exec_progress_pool( use_threads=True, @@ -476,9 +476,6 @@ def transcode_pngs( ) modified.add(xref) - def pngquant_fn(args): - pngquant.quantize(*args) - exec_progress_pool( use_threads=True, max_workers=options.jobs, @@ -488,7 +485,7 @@ def transcode_pngs( unit='image', disable=not options.progress_bar, ), - task=pngquant_fn, + task=pngquant.quantize_mp, task_arguments=pngquant_args(), ) From 3bd5054634446cec9fa2152e7dc6fa66ed2c4fa8 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 23:40:10 -0800 Subject: [PATCH 783/880] lambda: move to extra_plugins folder --- src/ocrmypdf/extra_plugins/__init__.py | 0 .../awslambda.py} | 23 ++++--------------- 2 files changed, 5 insertions(+), 18 deletions(-) create mode 100644 src/ocrmypdf/extra_plugins/__init__.py rename src/ocrmypdf/{_lambda_plugin.py => extra_plugins/awslambda.py} (80%) diff --git a/src/ocrmypdf/extra_plugins/__init__.py b/src/ocrmypdf/extra_plugins/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/src/ocrmypdf/_lambda_plugin.py b/src/ocrmypdf/extra_plugins/awslambda.py similarity index 80% rename from src/ocrmypdf/_lambda_plugin.py rename to src/ocrmypdf/extra_plugins/awslambda.py index 1cab2f61..3a5f802d 100644 --- a/src/ocrmypdf/_lambda_plugin.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -1,22 +1,9 @@ -# © 2021 James R Barlow: https://github.com/jbarlow83 +# © 2021 James R. Barlow: github.com/jbarlow83 # -# Permission is hereby granted, free of charge, to any person obtaining a copy -# of this software and associated documentation files (the "Software"), to deal -# in the Software without restriction, including without limitation the rights -# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -# copies of the Software, and to permit persons to whom the Software is -# furnished to do so, subject to the following conditions: -# -# The above copyright notice and this permission notice shall be included in all -# copies or substantial portions of the Software. -# -# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -# SOFTWARE. +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + """Alternate executor to support OCRmyPDF in AWS Lambda""" From 386cabff001dabfb8aa3e8644fc02ba6ce7ba083 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 24 Jan 2021 23:56:09 -0800 Subject: [PATCH 784/880] Make progress pool common rather than plugin-specific --- src/ocrmypdf/_concurrent.py | 22 ++++++++++-------- src/ocrmypdf/builtin_plugins/concurrency.py | 25 +++++++++------------ src/ocrmypdf/extra_plugins/awslambda.py | 20 +++++++---------- 3 files changed, 32 insertions(+), 35 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 87c9656c..e3fb5b91 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -4,8 +4,11 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. +import threading from typing import Callable, Iterable, Optional +pool_lock = threading.Lock() + def _task_noop(*_args, **_kwargs): return @@ -35,12 +38,13 @@ def exec_progress_pool( if not task_finished: task_finished = _task_noop - _model( - use_threads=use_threads, - max_workers=max_workers, - tqdm_kwargs=tqdm_kwargs, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - ) + with pool_lock: + _model( + use_threads=use_threads, + max_workers=max_workers, + tqdm_kwargs=tqdm_kwargs, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + ) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 7d7fbbfc..3b3e0d74 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -31,8 +31,6 @@ from ocrmypdf.exceptions import InputFileError Queue = Union[multiprocessing.Queue, queue.Queue] -pool_lock = threading.Lock() - def log_listener(q: Queue): """Listen to the worker processes and forward the messages to logging @@ -120,18 +118,17 @@ def exec_progress_pool( worker_initializer = _noop - with pool_lock: - _exec_progress_pool( - max_workers=max_workers, - tqdm_kwargs=tqdm_kwargs, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - log_queue=log_queue, - pool_class=pool_class, - initializer=initializer, - ) + _exec_progress_pool( + max_workers=max_workers, + tqdm_kwargs=tqdm_kwargs, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + log_queue=log_queue, + pool_class=pool_class, + initializer=initializer, + ) def _exec_progress_pool( diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 3a5f802d..4f7a7af7 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -11,7 +11,6 @@ import logging import logging.handlers import signal -import threading from contextlib import suppress from enum import Enum, auto from itertools import islice, repeat, takewhile, zip_longest @@ -23,8 +22,6 @@ from unittest.mock import Mock from ocrmypdf import hookimpl from ocrmypdf.exceptions import InputFileError -pool_lock = threading.Lock() - class MessageType(Enum): exception = auto() @@ -101,15 +98,14 @@ def lambda_pool_impl( task_finished(result, pbar) return - with pool_lock: - _lambda_pool_impl( - max_workers=max_workers, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - pbar=pbar, - ) + _lambda_pool_impl( + max_workers=max_workers, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + pbar=pbar, + ) def _lambda_pool_impl( From ecb0109d79fcbfc015a3a329e113aadc8605dac3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 Jan 2021 01:29:45 -0800 Subject: [PATCH 785/880] docs: fix rst formatting error --- docs/installation.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4fc3738a..f0524f07 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -520,7 +520,7 @@ DLLs or other Windows patches, and may require a reboot. You may then use ``pip`` to install ocrmypdf. (This can performed by a user or Administrator.): -* ``pip install ocrmypdf +* ``pip install ocrmypdf`` Chocolatey automatically selects appropriate versions of these applications. If you are installing them manually, please install 64-bit versions of all applications for From 108472493762cb529b158e238432e8a1f1e44cbf Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 Jan 2021 01:40:40 -0800 Subject: [PATCH 786/880] docs: improve API docs --- docs/apiref.rst | 12 ++++++++++++ docs/plugins.rst | 21 +++++++++++++++++++++ src/ocrmypdf/_jobcontext.py | 31 ++++++++++++++++++++++++++++--- src/ocrmypdf/helpers.py | 22 ++++++++++++++-------- src/ocrmypdf/hocrtransform.py | 11 +++++++++-- src/ocrmypdf/pdfa.py | 2 +- 6 files changed, 85 insertions(+), 14 deletions(-) diff --git a/docs/apiref.rst b/docs/apiref.rst index caeb277a..5a3bd488 100644 --- a/docs/apiref.rst +++ b/docs/apiref.rst @@ -5,6 +5,15 @@ API Reference This page summarizes the rest of the public API. Generally speaking this should mainly of interest to plugin developers. +ocrmypdf +======== + +.. autoclass:: ocrmypdf.PageContext + :members: + +.. autoclass:: ocrmypdf.PdfContext + :members: + ocrmypdf.exceptions =================== @@ -17,6 +26,9 @@ ocrmypdf.helpers .. automodule:: ocrmypdf.helpers :members: + :noindex: deprecated + + .. autodecorator:: deprecated ocrmypdf.hocrtransform ====================== diff --git a/docs/plugins.rst b/docs/plugins.rst index 12abf0ab..e73523af 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -65,6 +65,27 @@ similar to ``pytest`` packages such as ``pytest-cov`` (the package) and ``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the module), just like pytest plugins. +Setuptools plugins +================== + +You can also create a plugin that OCRmyPDF will always automatically load if both are +installed in the same virtual environment, using a setuptools entrypoint. + +Your package's ``setup.py`` would need to contain the following, for a plugin +named ``ocrmypdf-exampleplugin``: + +.. code-block:: python + + # sample ./setup.py file + from setuptools import setup + + setup( + name="ocrmypdf-exampleplugin", + packages=["exampleplugin"], + # the following makes a plugin available to pytest + entry_points={"ocrmypdf": ["exampleplugin = exampleplugin.pluginmodule"]}, + ) + Plugin requirements =================== diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index f0767814..61391404 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -14,11 +14,19 @@ from io import IOBase from pathlib import Path from typing import Iterator +from pluggy import PluginManager + from ocrmypdf.pdfinfo import PdfInfo +from ocrmypdf.pdfinfo.info import PageInfo class PdfContext: - """Holds our context for a particular run of the pipeline""" + """Holds the context for a particular run of the pipeline.""" + + options: Namespace #: The specified options for processing this PDF. + origin: Path #: The filename of the original input file. + pdfinfo: PdfInfo #: Detailed data for this PDF. + plugin_manager: PluginManager #: PluginManager for processing the current PDF. def __init__( self, @@ -35,21 +43,33 @@ class PdfContext: self.plugin_manager = plugin_manager def get_path(self, name: str) -> Path: + """Generate a ``Path`` for an intermediate file involved in processing. + + The path will be in a temporary folder that is common for all processing + of this particular PDF. + """ return self.work_folder / name def get_page_contexts(self) -> Iterator['PageContext']: + """Get all ``PageContext`` for this PDF.""" npages = len(self.pdfinfo) for n in range(npages): yield PageContext(self, n) class PageContext: - """Holds our context for a page + """Holds our context for a page. Must be pickable, so stores only intrinsic/simple data elements or those - capable of their serializing themselves via __getstate__. + capable of their serializing themselves via ``__getstate__``. """ + options: Namespace #: The specified options for processing this PDF. + origin: Path #: The filename of the original input file. + pageno: int #: This page number (zero-based). + pageinfo: PageInfo #: Information on this page. + plugin_manager: PluginManager #: PluginManager for processing the current PDF. + def __init__(self, pdf_context: PdfContext, pageno): self.work_folder = pdf_context.work_folder self.origin = pdf_context.origin @@ -59,6 +79,11 @@ class PageContext: self.plugin_manager = pdf_context.plugin_manager def get_path(self, name: str) -> Path: + """Generate a ``Path`` for a file that is part of processing this page. + + The path will be based in a common temporary folder and have a prefix based + on the page number. + """ return self.work_folder / ("%06d_%s" % (self.pageno + 1, name)) def __getstate__(self): diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index afd8eaa7..4823ee46 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -63,8 +63,13 @@ class NeverRaise(Exception): def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): - """ - Helper function: relinks soft symbolic link if necessary + """Create a symbolic link at ``soft_link_name``, which references ``input_file``. + + Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead. + + Use symlinks safely. Self-linking loops are prevented. On Windows, file copy is + used since symlinks may require administrator privileges. An existing link at the + destination is removed. """ input_file = os.fspath(input_file) soft_link_name = os.fspath(soft_link_name) @@ -72,8 +77,8 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): # Guard against soft linking to oneself if input_file == soft_link_name: log.warning( - "No symbolic link made. You are using " - "the original data directory as the working directory." + "No symbolic link created. You are using the original data directory " + "as the working directory." ) return @@ -114,7 +119,7 @@ def is_iterable_notstr(thing: Any) -> bool: def monotonic(L: Sequence) -> bool: - """Does list increase monotonically?""" + """Does this sequence increase monotonically?""" return all(b > a for a, b in zip(L, L[1:])) @@ -173,7 +178,8 @@ def is_file_writable(test_file: os.PathLike) -> bool: def check_pdf(input_file: Path) -> bool: """Check if a PDF complies with the PDF specification. - Checks for proper formatting and proper linearization. + Checks for proper formatting and proper linearization. Uses pikepdf (which in + turn, uses QPDF) to perform the checks. """ pdf = None try: @@ -217,7 +223,7 @@ def check_pdf(input_file: Path) -> bool: def clamp(n, smallest, largest): # mypy doesn't understand types for this - """Clamps the value of n to between smallest and largest.""" + """Clamps the value of ``n`` to between ``smallest`` and ``largest``.""" return max(smallest, min(n, largest)) @@ -235,7 +241,7 @@ def pikepdf_enable_mmap(): def deprecated(func): - """Warn that function is deprecated""" + """Warn that function is deprecated.""" @wraps(func) def new_func(*args, **kwargs): diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 13e266cb..4e2b2def 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -35,7 +35,7 @@ from collections import namedtuple from itertools import chain from math import atan, cos, sin from pathlib import Path -from typing import Optional, Tuple, Union +from typing import Any, NamedTuple, Optional, Tuple, Union from xml.etree import ElementTree from reportlab.lib.colors import black, cyan, magenta, red @@ -44,7 +44,14 @@ from reportlab.pdfgen.canvas import Canvas Element = ElementTree.Element -Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2']) + +class Rect(NamedTuple): + """A rectangle for managing PDF coordinates.""" + + x1: Any + y1: Any + x2: Any + y2: Any class HocrTransformError(Exception): diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 8313169a..4eaee8f1 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -112,7 +112,7 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'): def file_claims_pdfa(filename: Path): - """Determines if the file claims to be PDF/A compliant + """Determines if the file claims to be PDF/A compliant. This only checks if the XMP metadata contains a PDF/A marker. It does not do full PDF/A validation. From ef1e7a814ec7578c81589fa5e34042f3f7a7e9ae Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 Jan 2021 01:45:04 -0800 Subject: [PATCH 787/880] Delinting --- src/ocrmypdf/_exec/pngquant.py | 2 -- src/ocrmypdf/_jobcontext.py | 1 - src/ocrmypdf/_plugin_manager.py | 13 +++++++++---- src/ocrmypdf/helpers.py | 2 +- src/ocrmypdf/hocrtransform.py | 3 +-- src/ocrmypdf/leptonica.py | 1 - src/ocrmypdf/optimize.py | 13 +++++++------ src/ocrmypdf/pdfinfo/info.py | 2 +- 8 files changed, 19 insertions(+), 18 deletions(-) diff --git a/src/ocrmypdf/_exec/pngquant.py b/src/ocrmypdf/_exec/pngquant.py index d7e345f2..88a6370c 100644 --- a/src/ocrmypdf/_exec/pngquant.py +++ b/src/ocrmypdf/_exec/pngquant.py @@ -9,10 +9,8 @@ from contextlib import contextmanager from io import BytesIO -from os import fspath from pathlib import Path from subprocess import PIPE -from tempfile import NamedTemporaryFile from PIL import Image diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 61391404..2363ccbb 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -10,7 +10,6 @@ import shutil import sys from argparse import Namespace from copy import copy -from io import IOBase from pathlib import Path from typing import Iterator diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 510ae65b..c7e77c07 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -10,9 +10,8 @@ import importlib import importlib.util import pkgutil import sys -from functools import partial from pathlib import Path -from typing import Callable, List, Tuple, Union +from typing import List, Tuple, Union import pluggy @@ -32,7 +31,11 @@ class OcrmypdfPluginManager(pluggy.PluginManager): """ def __init__( - self, *args, plugins: List[Union[str, Path]], builtins: bool = True, **kwargs, + self, + *args, + plugins: List[Union[str, Path]], + builtins: bool = True, + **kwargs, ): self.__init_args = args self.__init_kwargs = kwargs @@ -88,7 +91,9 @@ class OcrmypdfPluginManager(pluggy.PluginManager): def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True): pm = OcrmypdfPluginManager( - project_name='ocrmypdf', plugins=plugins, builtins=builtins, + project_name='ocrmypdf', + plugins=plugins, + builtins=builtins, ) return pm diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 4823ee46..9fc27709 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -17,7 +17,7 @@ from functools import wraps from io import StringIO from math import isclose from pathlib import Path -from typing import Any, Sequence, TypeVar +from typing import Any, Sequence import pikepdf diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index 4e2b2def..ea75f51e 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -31,7 +31,6 @@ import argparse import os import re -from collections import namedtuple from itertools import chain from math import atan, cos, sin from pathlib import Path @@ -45,7 +44,7 @@ from reportlab.pdfgen.canvas import Canvas Element = ElementTree.Element -class Rect(NamedTuple): +class Rect(NamedTuple): # pylint: disable=inherit-non-class """A rectangle for managing PDF coordinates.""" x1: Any diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 69f6e444..91715939 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -13,7 +13,6 @@ import argparse import logging import os -import platform import sys import threading import warnings diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index c097e32e..93658076 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -10,11 +10,9 @@ import sys import tempfile from collections import defaultdict from functools import partial -from io import BytesIO from os import fspath from pathlib import Path from typing import ( - Any, Callable, Dict, Iterator, @@ -49,7 +47,7 @@ DEFAULT_PNG_QUALITY = 70 Xref = NewType('Xref', int) -class XrefExt(NamedTuple): +class XrefExt(NamedTuple): # pylint: disable=inherit-non-class xref: Xref ext: str @@ -212,7 +210,10 @@ def extract_image_generic( def extract_images( - pike: Pdf, root: Path, options, extract_fn: Callable[..., Optional[XrefExt]], + pike: Pdf, + root: Path, + options, + extract_fn: Callable[..., Optional[XrefExt]], ) -> Iterator[Tuple[int, XrefExt]]: """Extract image using extract_fn @@ -564,10 +565,10 @@ def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cove # ncolors == 0 means we are using a colorspace without a palette if compdata.spp == 1: cs = Name.DeviceGray - elif compdata.spp == 3: - cs = Name.DeviceRGB elif compdata.spp == 4: cs = Name.DeviceCMYK + else: # spp == 3 + cs = Name.DeviceRGB im_obj.ColorSpace = cs im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index ae02fb98..0ee921b0 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -15,7 +15,7 @@ from functools import partial from math import hypot, isclose from os import PathLike from pathlib import Path -from typing import Any, Container, Dict, Iterator, List, Optional, Tuple, Union +from typing import Container, Iterator, Optional, Tuple, Union from warnings import warn import pikepdf From 46d0632fe27f75e583e957edcc21beec63a2223e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 26 Jan 2021 01:47:49 -0800 Subject: [PATCH 788/880] v11.6.0 release notes --- docs/release_notes.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 173473d7..19c4f60a 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,16 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.6.0 +======= + +- OCRmyPDF will now automatically register plugins from the same virtual environment + with an appropriate setuptools entrypoint. +- Refactor the plugin manager to remove unnecessary complications and make plugin + registration more automatic. +- ``PageContext`` and ``PdfContext`` are now formally part of the API, as they + should have been, since they were part of ``ocrmypdf.pluginspec``. + v11.5.0 ======= From 327df5cbbc78ce160e877eb27bae603791fe397b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 1 Dec 2020 21:14:40 -0800 Subject: [PATCH 789/880] Use ColorConversionStrategy "LeaveColorUnchanged" Faster, still produces PDF/A --- src/ocrmypdf/_exec/ghostscript.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 3582a6eb..8f17719a 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -195,10 +195,11 @@ def generate_pdfa( "-dAutoFilterGrayImages=true", ] + strategy = 'LeaveColorUnchanged' # Older versions of Ghostscript expect a leading slash in # sColorConversionStrategy, newer ones should not have it. See Ghostscript # git commit fe1c025d. - strategy = 'RGB' if version() >= '9.19' else '/RGB' + strategy = ('/' + strategy) if version() < '9.19' else strategy if version() == '9.23': # 9.23: added JPEG passthrough as a new feature, but with a bug that From d274d88929d7d87b0e41313325ca56b1e4739b87 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 30 Jan 2021 17:36:30 -0800 Subject: [PATCH 790/880] Refactor to eliminate global state in _concurrent --- src/ocrmypdf/__init__.py | 1 + src/ocrmypdf/_concurrent.py | 123 ++++++++++---- src/ocrmypdf/_pipeline.py | 8 +- src/ocrmypdf/_sync.py | 19 +-- src/ocrmypdf/builtin_plugins/concurrency.py | 149 +++++++---------- src/ocrmypdf/extra_plugins/awslambda.py | 174 ++++++++++---------- src/ocrmypdf/optimize.py | 29 +++- src/ocrmypdf/pdfinfo/info.py | 23 ++- src/ocrmypdf/pluginspec.py | 37 +---- tests/test_validation.py | 2 +- 10 files changed, 301 insertions(+), 264 deletions(-) diff --git a/src/ocrmypdf/__init__.py b/src/ocrmypdf/__init__.py index 081324e9..80586658 100644 --- a/src/ocrmypdf/__init__.py +++ b/src/ocrmypdf/__init__.py @@ -8,6 +8,7 @@ from pluggy import HookimplMarker as _HookimplMarker from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo +from ocrmypdf._concurrent import Executor from ocrmypdf._jobcontext import PageContext, PdfContext from ocrmypdf._version import PROGRAM_NAME, __version__ from ocrmypdf.api import Verbosity, configure_logging, ocr diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index e3fb5b91..48c13e08 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -4,47 +4,110 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. +import sys import threading +from abc import ABC, abstractmethod +from functools import partial from typing import Callable, Iterable, Optional -pool_lock = threading.Lock() +from tqdm import tqdm def _task_noop(*_args, **_kwargs): return -def _model(**_kwargs) -> None: - raise RuntimeError("Parallel executor not set up") +class Executor(ABC): + pool_lock = threading.Lock() + + def __call__( + self, + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Optional[Callable] = None, + task: Optional[Callable] = None, + task_arguments: Optional[Iterable] = None, + task_finished: Optional[Callable] = None, + ) -> None: + """ + Args: + use_threads: If False, the workload is the sort that will benefit from + running in a multiprocessing context (for example, it uses Python + heavily, and parallelizing it with threads is not expected to be + performant). + max_workers: The maximum number of workers that should be run. + tdqm_kwargs: Arguments to set up the progress bar. + worker_initializer: Called when the worker is initialized, in the worker's + execution context. Must be possible to marshall to the worker. + task: Called when the worker starts a new task, in the worker's execution + context. Must be possible to marshallable to the worker. + task_finished: Called when a worker finishes a task, in the parent's + context. + task_arguments: An iterable that generates a group of parameters for each + task. This runs in the parent's context, but the parameters must be + marshallable to the worker. + """ + + if not task_arguments: + return # Nothing to do! + if not worker_initializer: + worker_initializer = _task_noop + if not task_finished: + task_finished = _task_noop + if not task: + task = _task_noop + + with self.pool_lock: + self._execute( + use_threads=use_threads, + max_workers=max_workers, + tqdm_kwargs=tqdm_kwargs, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + ) + + @abstractmethod + def _execute( + self, + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Iterable, + task_finished: Callable, + ): + """Custom executors should override this method.""" -def set_execution_model(model): - global _model - _model = model +def setup_executor(plugin_manager) -> Executor: + return plugin_manager.hook.get_executor() -def exec_progress_pool( - *, - use_threads: bool, - max_workers: int, - tqdm_kwargs: dict, - worker_initializer: Optional[Callable] = None, - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Optional[Callable] = None, -): - if not worker_initializer: - worker_initializer = _task_noop - if not task_finished: - task_finished = _task_noop +class SerialExecutor(Executor): + """Implements a purely sequential executor using the parallel protocol. - with pool_lock: - _model( - use_threads=use_threads, - max_workers=max_workers, - tqdm_kwargs=tqdm_kwargs, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - ) + The current process/thread will be the worker that executes all tasks + in order. As such, ``worker_initializer`` will never be called. + """ + + def _execute( + self, + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Iterable, + task_finished: Callable, + ): + with tqdm(**tqdm_kwargs) as pbar: + for args in task_arguments: + result = task(args) + task_finished(result, pbar) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6de1f2e9..d4c171d8 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -22,6 +22,7 @@ from PIL import Image, ImageColor, ImageDraw from tqdm import tqdm from ocrmypdf import leptonica +from ocrmypdf._concurrent import Executor from ocrmypdf._exec import unpaper from ocrmypdf._jobcontext import PageContext, PdfContext from ocrmypdf._version import PROGRAM_NAME @@ -146,6 +147,8 @@ def triage(original_filename, input_file, output_file, options): def get_pdfinfo( input_file, + *, + executor: Executor, detailed_analysis=False, progbar=False, max_workers=None, @@ -158,6 +161,7 @@ def get_pdfinfo( progbar=progbar, max_workers=max_workers, check_pages=check_pages, + executor=executor, ) except pikepdf.PasswordError: raise EncryptedPdfError() @@ -792,7 +796,7 @@ def metadata_fixup(working_file: Path, context: PdfContext): return output_file -def optimize_pdf(input_file: Path, context: PdfContext): +def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor): output_file = context.get_path('optimize.pdf') save_settings = dict( compress_streams=True, @@ -800,7 +804,7 @@ def optimize_pdf(input_file: Path, context: PdfContext): object_stream_mode=pikepdf.ObjectStreamMode.generate, linearize=should_linearize(input_file, context), ) - optimize(input_file, output_file, context, save_settings) + optimize(input_file, output_file, context, save_settings, executor) return output_file diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 1d78f1d9..68577a5d 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -15,10 +15,9 @@ from pathlib import Path from tempfile import mkdtemp from typing import List, NamedTuple, Optional, Tuple -import pikepdf import PIL -from ocrmypdf._concurrent import exec_progress_pool, set_execution_model +from ocrmypdf._concurrent import Executor, setup_executor from ocrmypdf._graft import OcrGrafter from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files from ocrmypdf._logging import PageNumberFilter @@ -224,14 +223,14 @@ def exec_page_sync(page_context: PageContext): ) -def post_process(pdf_file, context: PdfContext): +def post_process(pdf_file, context: PdfContext, executor: Executor): pdf_out = pdf_file if context.options.output_type.startswith('pdfa'): ps_stub_out = generate_postscript_stub(context) pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context) pdf_out = metadata_fixup(pdf_out, context) - return optimize_pdf(pdf_out, context) + return optimize_pdf(pdf_out, context, executor) def worker_init(max_pixels: int): @@ -242,7 +241,7 @@ def worker_init(max_pixels: int): pikepdf_enable_mmap() -def exec_concurrent(context: PdfContext): +def exec_concurrent(context: PdfContext, executor: Executor): """Execute the pipeline concurrently""" # Run exec_page_sync on every page context @@ -269,7 +268,7 @@ def exec_concurrent(context: PdfContext): finally: tls.pageno = None - exec_progress_pool( + executor( use_threads=options.use_threads, max_workers=max_workers, tqdm_kwargs=dict( @@ -296,7 +295,7 @@ def exec_concurrent(context: PdfContext): # PDF/A and metadata log.info("Postprocessing...") - pdf = post_process(pdf, context) + pdf = post_process(pdf, context, executor) # Copy PDF file to destination copy_final(pdf, options.output_file, context) @@ -346,8 +345,7 @@ def run_pipeline(options, *, plugin_manager, api=False): pikepdf_enable_mmap() - set_execution_model(plugin_manager.hook.get_parallel_executor()) - + executor = setup_executor(plugin_manager) try: check_requested_output_file(options) start_input_file, original_filename = create_input_file(options, work_folder) @@ -360,6 +358,7 @@ def run_pipeline(options, *, plugin_manager, api=False): # Gather pdfinfo and create context pdfinfo = get_pdfinfo( origin_pdf, + executor=executor, detailed_analysis=options.redo_ocr, progbar=options.progress_bar, max_workers=options.jobs if not options.use_threads else 1, # To help debug @@ -372,7 +371,7 @@ def run_pipeline(options, *, plugin_manager, api=False): validate_pdfinfo_options(context) # Execute the pipeline - exec_concurrent(context) + exec_concurrent(context, executor) if options.output_file == '-': log.info("Output sent to stdout") diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 3b3e0d74..b61654dc 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -26,7 +26,7 @@ from typing import Callable, Iterable, Optional, Union from tqdm import tqdm -from ocrmypdf import hookimpl +from ocrmypdf import Executor, hookimpl from ocrmypdf.exceptions import InputFileError Queue = Union[multiprocessing.Queue, queue.Queue] @@ -91,99 +91,68 @@ def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel): return -def exec_progress_pool( - *, - use_threads: bool, - max_workers: int, - tqdm_kwargs: dict, - worker_initializer: Optional[Callable], - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Callable, -): +class StandardExecutor(Executor): + def _execute( + self, + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Iterable, + task_finished: Callable, + ): + if use_threads: + log_queue = queue.Queue(-1) + pool_class = ThreadPool + initializer = thread_init + else: + log_queue = multiprocessing.Queue(-1) + pool_class = ProcessPool + initializer = process_init - if use_threads: - log_queue = queue.Queue(-1) - pool_class = ThreadPool - initializer = thread_init - else: - log_queue = multiprocessing.Queue(-1) - pool_class = ProcessPool - initializer = process_init + # Regardless of whether we use_threads for worker processes, the log_listener + # must be a thread + listener = threading.Thread(target=log_listener, args=(log_queue,)) + listener.start() - if not worker_initializer: - - def _noop(): - return - - worker_initializer = _noop - - _exec_progress_pool( - max_workers=max_workers, - tqdm_kwargs=tqdm_kwargs, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - log_queue=log_queue, - pool_class=pool_class, - initializer=initializer, - ) - - -def _exec_progress_pool( - *, - max_workers: int, - tqdm_kwargs: dict, - worker_initializer: Callable, - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Callable, - log_queue: Queue, - pool_class: Callable, - initializer: Callable, -): - # Regardless of whether we use_threads for worker processes, the log_listener - # must be a thread - listener = threading.Thread(target=log_listener, args=(log_queue,)) - listener.start() - - with tqdm(**tqdm_kwargs) as pbar: - pool = pool_class( - processes=max_workers, - initializer=initializer, - initargs=(log_queue, worker_initializer, logging.getLogger("").level), - ) - try: - results = pool.imap_unordered(task, task_arguments) - for result in results: - if task_finished: - task_finished(result, pbar) - else: - pbar.update() - except KeyboardInterrupt: - # Terminate pool so we exit instantly - pool.terminate() - # Don't try listener.join() here, will deadlock - raise - except Exception: - if not os.environ.get("PYTEST_CURRENT_TEST", ""): - # Unless inside pytest, exit immediately because no one wants - # to wait for child processes to finalize results that will be - # thrown away. Inside pytest, we want child processes to exit - # cleanly so that they output an error messages or coverage data - # we need from them. + with tqdm(**tqdm_kwargs) as pbar: + pool = pool_class( + processes=max_workers, + initializer=initializer, + initargs=(log_queue, worker_initializer, logging.getLogger("").level), + ) + try: + results = pool.imap_unordered(task, task_arguments) + for result in results: + if task_finished: + task_finished(result, pbar) + else: + pbar.update() + except KeyboardInterrupt: + # Terminate pool so we exit instantly pool.terminate() - raise - finally: - # Terminate log listener - log_queue.put_nowait(None) - pool.close() - pool.join() + # Don't try listener.join() here, will deadlock + raise + except Exception: + if not os.environ.get("PYTEST_CURRENT_TEST", ""): + # Unless inside pytest, exit immediately because no one wants + # to wait for child processes to finalize results that will be + # thrown away. Inside pytest, we want child processes to exit + # cleanly so that they output an error messages or coverage data + # we need from them. + pool.terminate() + raise + finally: + # Terminate log listener + log_queue.put_nowait(None) + pool.close() + pool.join() - listener.join() + listener.join() @hookimpl -def get_parallel_executor(): - return exec_progress_pool +def get_executor(): + return StandardExecutor() diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 4f7a7af7..8579683d 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -19,7 +19,7 @@ from multiprocessing.connection import Connection, wait from typing import Callable, Iterable, Optional from unittest.mock import Mock -from ocrmypdf import hookimpl +from ocrmypdf import Executor, hookimpl from ocrmypdf.exceptions import InputFileError @@ -80,97 +80,101 @@ def process_loop( return -def lambda_pool_impl( - *, - use_threads: bool, - max_workers: int, - tqdm_kwargs: dict, - worker_initializer: Callable, - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Callable, -): - pbar = Mock() +class LambdaExecutor(Executor): + def _execute( + self, + *, + use_threads: bool, + max_workers: int, + tqdm_kwargs: dict, + worker_initializer: Callable, + task: Callable, + task_arguments: Iterable, + task_finished: Callable, + ): + pbar = Mock() - if use_threads and max_workers == 1: - for args in task_arguments: - result = task(args) - task_finished(result, pbar) - return + if use_threads and max_workers == 1: + for args in task_arguments: + result = task(args) + task_finished(result, pbar) + return - _lambda_pool_impl( - max_workers=max_workers, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - pbar=pbar, - ) - - -def _lambda_pool_impl( - *, - max_workers: int, - worker_initializer: Callable, - task: Callable, - task_arguments: Optional[Iterable] = None, - task_finished: Callable, - pbar, -): - task_arguments = list(task_arguments) - grouped_args = list(zip_longest(*list(split_every(max_workers, task_arguments)))) - if not grouped_args: - return - - processes = [] - connections = [] - for chunk in grouped_args: - parent_conn, child_conn = Pipe() - - worker_args = [args for args in chunk if args is not None] - process = Process( - target=process_loop, - args=( - child_conn, - worker_initializer, - logging.getLogger("").level, - task, - worker_args, - ), + self._lambda_pool_impl( + max_workers=max_workers, + worker_initializer=worker_initializer, + task=task, + task_arguments=task_arguments, + task_finished=task_finished, + pbar=pbar, ) - process.daemon = True - processes.append(process) - connections.append(parent_conn) - for process in processes: - process.start() + def _lambda_pool_impl( + self, + *, + max_workers: int, + worker_initializer: Callable, + task: Callable, + task_arguments: Iterable, + task_finished: Callable, + pbar, + ): + task_arguments = list(task_arguments) + grouped_args = list( + zip_longest(*list(split_every(max_workers, task_arguments))) + ) + if not grouped_args: + return - while connections: - for r in wait(connections): - try: - msg_type, msg = r.recv() - except EOFError: - connections.remove(r) - continue + processes = [] + connections = [] + for chunk in grouped_args: + parent_conn, child_conn = Pipe() - if msg_type == MessageType.result: - if task_finished: - task_finished(msg, pbar) - elif msg_type == 'log': - record = msg - logger = logging.getLogger(record.name) - logger.handle(record) - elif msg_type == MessageType.complete: - connections.remove(r) - elif msg_type == MessageType.exception: - for process in processes: - process.terminate() - raise msg + worker_args = [args for args in chunk if args is not None] + process = Process( + target=process_loop, + args=( + child_conn, + worker_initializer, + logging.getLogger("").level, + task, + worker_args, + ), + ) + process.daemon = True + processes.append(process) + connections.append(parent_conn) - for process in processes: - process.join() + for process in processes: + process.start() + + while connections: + for r in wait(connections): + try: + msg_type, msg = r.recv() + except EOFError: + connections.remove(r) + continue + + if msg_type == MessageType.result: + if task_finished: + task_finished(msg, pbar) + elif msg_type == 'log': + record = msg + logger = logging.getLogger(record.name) + logger.handle(record) + elif msg_type == MessageType.complete: + connections.remove(r) + elif msg_type == MessageType.exception: + for process in processes: + process.terminate() + raise msg + + for process in processes: + process.join() @hookimpl -def get_parallel_executor(): - return lambda_pool_impl +def get_executor(): + return LambdaExecutor() diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 3724f731..5993014d 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -34,7 +34,7 @@ from PIL import Image from tqdm import tqdm from ocrmypdf import leptonica -from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._concurrent import Executor, SerialExecutor from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._jobcontext import PdfContext from ocrmypdf.exceptions import OutputFileAccessError @@ -301,7 +301,7 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefE def _produce_jbig2_images( - jbig2_groups: Dict[int, List[XrefExt]], root: Path, options + jbig2_groups: Dict[int, List[XrefExt]], root: Path, options, executor: Executor ) -> None: """Produce JBIG2 images from their groups""" @@ -333,7 +333,7 @@ def _produce_jbig2_images( jbig2_args = jbig2_single_args jbig2_convert = jbig2enc.convert_single_mp - exec_progress_pool( + executor( use_threads=True, max_workers=options.jobs, tqdm_kwargs=dict( @@ -348,7 +348,11 @@ def _produce_jbig2_images( def convert_to_jbig2( - pike: Pdf, jbig2_groups: Dict[int, List[XrefExt]], root: Path, options + pike: Pdf, + jbig2_groups: Dict[int, List[XrefExt]], + root: Path, + options, + executor: Executor, ) -> None: """Convert images to JBIG2 and insert into PDF. @@ -363,7 +367,7 @@ def convert_to_jbig2( and needs no dictionary. Currently this must be lossless JBIG2. """ - _produce_jbig2_images(jbig2_groups, root, options) + _produce_jbig2_images(jbig2_groups, root, options, executor) for group, xref_exts in jbig2_groups.items(): prefix = f'group{group:08d}' @@ -457,6 +461,7 @@ def transcode_pngs( image_name_fn: Callable[[Path, Xref], Path], root: Path, options, + executor, ) -> None: modified: MutableSet[Xref] = set() if options.optimize >= 2: @@ -476,7 +481,7 @@ def transcode_pngs( ) modified.add(xref) - exec_progress_pool( + executor( use_threads=True, max_workers=options.jobs, tqdm_kwargs=dict( @@ -569,7 +574,13 @@ def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cove im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) -def optimize(input_file: Path, output_file: Path, context, save_settings) -> None: +def optimize( + input_file: Path, + output_file: Path, + context, + save_settings, + executor: Executor = SerialExecutor(), +) -> None: options = context.options if options.optimize == 0: safe_symlink(input_file, output_file) @@ -591,10 +602,10 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non # if options.optimize >= 2: # Try pngifying the jpegs # transcode_pngs(pike, jpegs, jpg_name, root, options) - transcode_pngs(pike, pngs, png_name, root, options) + transcode_pngs(pike, pngs, png_name, root, options, executor) jbig2_groups = extract_images_jbig2(pike, root, options) - convert_to_jbig2(pike, jbig2_groups, root, options) + convert_to_jbig2(pike, jbig2_groups, root, options, executor) target_file = output_file.with_suffix('.opt.pdf') pike.remove_unreferenced_resources() diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 1ee82cf9..3733feaf 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -10,6 +10,7 @@ import atexit import logging import re from collections import defaultdict, namedtuple +from contextlib import ExitStack from decimal import Decimal from enum import Enum from functools import partial @@ -22,7 +23,7 @@ from warnings import warn import pikepdf from pikepdf import Object, Pdf, PdfMatrix -from ocrmypdf._concurrent import exec_progress_pool +from ocrmypdf._concurrent import Executor, SerialExecutor from ocrmypdf.exceptions import EncryptedPdfError, InputFileError from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes @@ -592,12 +593,21 @@ def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel): def _pdf_pageinfo_sync(args): pageno, thread_pdf, infile, check_pages, detailed_analysis = args pdf = thread_pdf if thread_pdf is not None else worker_pdf - page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis) - return page + with ExitStack() as stack: + if not pdf: # When called with SerialExecutor + pdf = stack.enter_context(pikepdf.open(infile)) + page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis) + return page def _pdf_pageinfo_concurrent( - pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False + pdf, + executor: Executor, + infile, + progbar, + max_workers, + check_pages, + detailed_analysis=False, ): pages = [None] * len(pdf.pages) @@ -629,7 +639,7 @@ def _pdf_pageinfo_concurrent( (n, initial_pdf, infile, check_pages, detailed_analysis) for n in range(total) ) assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable" - exec_progress_pool( + executor( use_threads=use_threads, max_workers=n_workers, tqdm_kwargs=dict( @@ -829,10 +839,12 @@ class PdfInfo: def __init__( self, infile, + *, detailed_analysis: bool = False, progbar: bool = False, max_workers: int = None, check_pages=None, + executor: Executor = SerialExecutor(), ): self._infile = infile if check_pages is None: @@ -843,6 +855,7 @@ class PdfInfo: raise EncryptedPdfError() # Triggered by encryption with empty passwd self._pages = _pdf_pageinfo_concurrent( pdf, + executor, infile, progbar, max_workers, diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 538be7a1..f3d5d6fa 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -13,6 +13,7 @@ from typing import TYPE_CHECKING, AbstractSet, Callable, Iterable, List, Optiona import pluggy +from ocrmypdf._concurrent import Executor from ocrmypdf.helpers import Resolution if TYPE_CHECKING: @@ -62,44 +63,16 @@ def check_options(options: Namespace) -> None: """ -class ParallelExecutor(ABC): - @abstractstaticmethod - def __call__( - use_threads: bool, - max_workers: int, - tqdm_kwargs: dict, - worker_initializer: Callable, - task: Callable, - task_finished: Callable, - task_arguments: Optional[Iterable] = None, - ): - """ - Args: - use_threads: If False, the workload is the sort that will benefit from - running in a multiprocessing context (for example, it uses Python - heavily, and parallelizing it with threads is not expected to be - performant). - max_workers: The maximum number of workers that should be run. - tdqm_kwargs: Arguments to set up the progress bar. - worker_initializer: Called when the worker is initialized, in the worker's - execution context. Must be possible to marshall to the worker. - task: Called when the worker starts a new task, in the worker's execution - context. Must be possible to marshallable to the worker. - task_finished: Called when a worker finishes a task, in the parent's - context. - task_arguments: An iterable that generates a group of parameters for each - task. This runs in the parent's context, but the parameters must be - marshallable to the worker. - """ - - @hookspec(firstresult=True) -def get_parallel_executor() -> Callable: +def get_executor() -> Executor: """Called to perform parallel execution This may be used to replace OCRmyPDF's default parallel execution system with a third party alternative. For example, you could make OCRmyPDF run in a distributed environment. + + OCRmyPDF's executors are analogous to the standard Python executors in + conconcurrent.futures, but they do not work the same way. """ diff --git a/tests/test_validation.py b/tests/test_validation.py index deed9769..f7ce5286 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -173,7 +173,7 @@ def test_false_action_store_true(): def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) plugin_manager = get_plugin_manager(opts.plugins) - with patch('ocrmypdf.builtin_plugins.concurrency.tqdm', autospec=True) as tqdmpatch: + with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: vd._check_options(opts, plugin_manager, set()) pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) assert pdfinfo is not None From 16bda74974df2802f44f421402ed969b3f119e3b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 30 Jan 2021 20:42:00 -0800 Subject: [PATCH 791/880] Refactor - decouple progressbar from executor --- docs/plugins.rst | 10 ++++++ src/ocrmypdf/_concurrent.py | 38 ++++++++++++++++----- src/ocrmypdf/builtin_plugins/concurrency.py | 7 +++- src/ocrmypdf/extra_plugins/awslambda.py | 5 +-- src/ocrmypdf/pluginspec.py | 37 ++++++++++++++++++-- tests/test_validation.py | 26 +++++++++----- 6 files changed, 100 insertions(+), 23 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index 12abf0ab..fb97f7f9 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -130,6 +130,16 @@ Custom command line arguments .. autofunction:: ocrmypdf.pluginspec.check_options +Execution and progress reporting +-------------------------------- + +.. autoclass: ocrmypdf.pluginspec.Executor + :members: + +.. autofunction:: ocrmypdf.pluginspec.get_executor + +.. autofunction:: ocrmypdf.pluginspec.get_progress_bar + Applying special behavior before processing ------------------------------------------- diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 48c13e08..ea975bae 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -10,15 +10,32 @@ from abc import ABC, abstractmethod from functools import partial from typing import Callable, Iterable, Optional -from tqdm import tqdm - def _task_noop(*_args, **_kwargs): return +class NullProgressBar: + def __init__(self, **kwargs): + pass + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_value, traceback): + return False + + def update(self, _arg=None): + return + + class Executor(ABC): pool_lock = threading.Lock() + pbar_class = NullProgressBar + + def __init__(self, *, pbar_class=None): + if pbar_class: + self.pbar_class = pbar_class def __call__( self, @@ -32,17 +49,21 @@ class Executor(ABC): task_finished: Optional[Callable] = None, ) -> None: """ + Set up parallel execution and progress reporting. + Args: - use_threads: If False, the workload is the sort that will benefit from + use_threads: If ``False``, the workload is the sort that will benefit from running in a multiprocessing context (for example, it uses Python heavily, and parallelizing it with threads is not expected to be performant). max_workers: The maximum number of workers that should be run. tdqm_kwargs: Arguments to set up the progress bar. - worker_initializer: Called when the worker is initialized, in the worker's - execution context. Must be possible to marshall to the worker. + worker_initializer: Called when a worker is initialized, in the worker's + execution context. If the child workers are processes, it must be + possible to marshall/pickle the worker initializer. + ``functools.partial`` can be used to bind parameters. task: Called when the worker starts a new task, in the worker's execution - context. Must be possible to marshallable to the worker. + context. Must be possible to marshall to the worker. task_finished: Called when a worker finishes a task, in the parent's context. task_arguments: An iterable that generates a group of parameters for each @@ -86,7 +107,8 @@ class Executor(ABC): def setup_executor(plugin_manager) -> Executor: - return plugin_manager.hook.get_executor() + pbar_class = plugin_manager.hook.get_progress_bar() + return plugin_manager.hook.get_executor(pbar_class=pbar_class) class SerialExecutor(Executor): @@ -107,7 +129,7 @@ class SerialExecutor(Executor): task_arguments: Iterable, task_finished: Callable, ): - with tqdm(**tqdm_kwargs) as pbar: + with self.pbar_class(**tqdm_kwargs) as pbar: for args in task_arguments: result = task(args) task_finished(result, pbar) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index b61654dc..df6d42c6 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -117,7 +117,7 @@ class StandardExecutor(Executor): listener = threading.Thread(target=log_listener, args=(log_queue,)) listener.start() - with tqdm(**tqdm_kwargs) as pbar: + with self.pbar_class(**tqdm_kwargs) as pbar: pool = pool_class( processes=max_workers, initializer=initializer, @@ -156,3 +156,8 @@ class StandardExecutor(Executor): @hookimpl def get_executor(): return StandardExecutor() + + +@hookimpl +def get_progress_bar(): + return tqdm diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 8579683d..64297f67 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -92,12 +92,10 @@ class LambdaExecutor(Executor): task_arguments: Iterable, task_finished: Callable, ): - pbar = Mock() - if use_threads and max_workers == 1: for args in task_arguments: result = task(args) - task_finished(result, pbar) + task_finished(result, self.pbar_class) return self._lambda_pool_impl( @@ -106,7 +104,6 @@ class LambdaExecutor(Executor): task=task, task_arguments=task_arguments, task_finished=task_finished, - pbar=pbar, ) def _lambda_pool_impl( diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index f3d5d6fa..cf68fd77 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -65,14 +65,47 @@ def check_options(options: Namespace) -> None: @hookspec(firstresult=True) def get_executor() -> Executor: - """Called to perform parallel execution + """Called to obtain an object that manages parallel execution. This may be used to replace OCRmyPDF's default parallel execution system with a third party alternative. For example, you could make OCRmyPDF run in a distributed environment. OCRmyPDF's executors are analogous to the standard Python executors in - conconcurrent.futures, but they do not work the same way. + ``conconcurrent.futures``, but they do not work the same way. + + Should be of type :class:`Executor` or otherwise conforming to the protocol + of that call. + + Note: + This hook will be called from the main process, and may modify global state + before child worker processes are forked. + Note: + This is a :ref:`firstresult hook`. + """ + + +@hookspec(firstresult=True) +def get_progress_bar(): + """Called to obtain a class that can be used to create progress bars. + + The class should follow a tqdm-like protocol. Calling the class should return + a new progress bar object, which is activated with ``__enter__`` and terminated + ``__exit__``. An update method is called whenever the progress bar is updated. + + The progress bar is held in the main process/thread and not updated by child + process/threads. When a child notifies the parent of completed work, the + parent updates the progress bar. + + The arguments are the same as `tqdm `_ accepts. + + Here is how OCRmyPDF will use the progress bar: + + Example: + pbar_class = pm.hook.get_progress_bar() + with pbar_class(**tqdm_kwargs) as pbar: + ... + pbar.update(1) """ diff --git a/tests/test_validation.py b/tests/test_validation.py index f7ce5286..c037fea9 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -6,12 +6,13 @@ import logging -from unittest.mock import patch +from unittest.mock import MagicMock, patch import pikepdf import pytest from ocrmypdf import _validation as vd +from ocrmypdf._concurrent import NullProgressBar, SerialExecutor from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.api import create_options from ocrmypdf.cli import get_parser @@ -173,13 +174,22 @@ def test_false_action_store_true(): def test_no_progress_bar(progress_bar, resources): opts = make_opts(progress_bar=progress_bar, input_file=(resources / 'trivial.pdf')) plugin_manager = get_plugin_manager(opts.plugins) - with patch('ocrmypdf._concurrent.tqdm', autospec=True) as tqdmpatch: - vd._check_options(opts, plugin_manager, set()) - pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar) - assert pdfinfo is not None - assert tqdmpatch.called - _args, kwargs = tqdmpatch.call_args - assert kwargs['disable'] != progress_bar + + vd._check_options(opts, plugin_manager, set()) + + pbar_disabled = None + + class CheckProgressBar(NullProgressBar): + def __init__(self, disable, **kwargs): + nonlocal pbar_disabled + pbar_disabled = disable + super().__init__(disable=disable, **kwargs) + + executor = SerialExecutor(pbar_class=CheckProgressBar) + pdfinfo = PdfInfo(opts.input_file, progbar=opts.progress_bar, executor=executor) + + assert pdfinfo is not None + assert pbar_disabled is not None and pbar_disabled != progress_bar def test_language_warning(caplog): From a9ad805347e48df3cd6ed53be46bc3519b46a429 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 00:08:20 -0800 Subject: [PATCH 792/880] optimize: Remove shim for unsupported pikepdf version --- src/ocrmypdf/optimize.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 93658076..3fa92355 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -150,11 +150,6 @@ def extract_image_generic( if pim.bits_per_component == 1: return None - try: - pim.indexed # pikepdf 1.6.3 can't handle [/Indexed [/Array...]] - except NotImplementedError: - return None - if filtdp[0] == Name.DCTDecode and options.optimize >= 2: # This is a simple heuristic derived from some training data, that has # about a 70% chance of guessing whether the JPEG is high quality, From 42c84531e42d909b1e1a295483f32bceca7b8d37 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 02:18:46 -0800 Subject: [PATCH 793/880] optimize: rewrite JPEG optimize to avoid use of tqdm and parallelize For some reason JPEG optimization was not done in parallel, and was perhaps never done in parallel. Strange oversight. --- src/ocrmypdf/optimize.py | 67 +++++++++++++++++++++++++++------------- 1 file changed, 45 insertions(+), 22 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 5993014d..112a26c3 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -9,8 +9,6 @@ import logging import sys import tempfile from collections import defaultdict -from functools import partial -from io import BytesIO from os import fspath from pathlib import Path from typing import ( @@ -31,7 +29,6 @@ import img2pdf import pikepdf from pikepdf import Dictionary, Name, Object, Pdf, PdfImage from PIL import Image -from tqdm import tqdm from ocrmypdf import leptonica from ocrmypdf._concurrent import Executor, SerialExecutor @@ -391,27 +388,53 @@ def convert_to_jbig2( ) -def transcode_jpegs(pike: Pdf, jpegs: Sequence[Xref], root: Path, options) -> None: - for xref in tqdm( - jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar - ): - in_jpg = jpg_name(root, xref) - opt_jpg = in_jpg.with_suffix('.opt.jpg') +def _optimize_jpeg(args): + xref, in_jpg, opt_jpg, jpeg_quality = args - # This produces a debug warning from PIL - # DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute - # 'close'. Seems to be mostly harmless - # https://github.com/python-pillow/Pillow/issues/1144 - with Image.open(in_jpg) as im: - im.save(opt_jpg, optimize=True, quality=options.jpeg_quality) + # This may produce a debug warning from PIL + # DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute + # 'close'. Seems to be mostly harmless + # https://github.com/python-pillow/Pillow/issues/1144 + with Image.open(in_jpg) as im: + im.save(opt_jpg, optimize=True, quality=jpeg_quality) - if opt_jpg.stat().st_size > in_jpg.stat().st_size: - log.debug("xref %s, jpeg, made larger - skip", xref) - continue + if opt_jpg.stat().st_size > in_jpg.stat().st_size: + log.debug("xref %s, jpeg, made larger - skip", xref) + opt_jpg.unlink() + opt_jpg = None + return xref, opt_jpg - compdata = leptonica.CompressedData.open(opt_jpg) - im_obj = pike.get_object(xref, 0) - im_obj.write(compdata.read(), filter=Name.DCTDecode) + +def transcode_jpegs( + pike: Pdf, jpegs: Sequence[Xref], root: Path, options, executor +) -> None: + def jpeg_args(): + for xref in jpegs: + in_jpg = jpg_name(root, xref) + opt_jpg = in_jpg.with_suffix('.opt.jpg') + yield xref, in_jpg, opt_jpg, options.jpeg_quality + + def finish_jpeg(result, pbar): + xref, opt_jpg = result + if opt_jpg: + compdata = leptonica.CompressedData.open(opt_jpg) + im_obj = pike.get_object(xref, 0) + im_obj.write(compdata.read(), filter=Name.DCTDecode) + pbar.update() + + executor( + use_threads=True, + max_workers=options.jobs, + tqdm_kwargs=dict( + desc="JPEGs", + total=len(jpegs), + unit='image', + disable=not options.progress_bar, + ), + task=_optimize_jpeg, + task_arguments=jpeg_args(), + task_finished=finish_jpeg, + ) def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool: @@ -598,7 +621,7 @@ def optimize( root.mkdir(exist_ok=True) jpegs, pngs = extract_images_generic(pike, root, options) - transcode_jpegs(pike, jpegs, root, options) + transcode_jpegs(pike, jpegs, root, options, executor) # if options.optimize >= 2: # Try pngifying the jpegs # transcode_pngs(pike, jpegs, jpg_name, root, options) From b1da09f141fe1c9af0aa6c7e3b4cc8b39b5c31ed Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 02:21:03 -0800 Subject: [PATCH 794/880] Add plugin for setting logging console So that we are not tied to tqdm. --- docs/plugins.rst | 2 ++ src/ocrmypdf/__main__.py | 5 +++- src/ocrmypdf/_pipeline.py | 7 ++++-- src/ocrmypdf/api.py | 26 ++++++++++++++------- src/ocrmypdf/builtin_plugins/concurrency.py | 6 +++++ src/ocrmypdf/optimize.py | 2 +- src/ocrmypdf/pluginspec.py | 6 +++++ 7 files changed, 42 insertions(+), 12 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index fb97f7f9..f1c3037c 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -136,6 +136,8 @@ Execution and progress reporting .. autoclass: ocrmypdf.pluginspec.Executor :members: +.. autofunction:: ocrmypdf.pluginspec.get_logging_console + .. autofunction:: ocrmypdf.pluginspec.get_executor .. autofunction:: ocrmypdf.pluginspec.get_progress_bar diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 3c897b72..1046a50c 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -47,7 +47,10 @@ def run(args=None): verbosity = Verbosity.quiet options.progress_bar = False configure_logging( - verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True + verbosity, + progress_bar_friendly=options.progress_bar, + manage_root_logger=True, + plugin_manager=plugin_manager, ) log.debug('ocrmypdf %s', __version__) try: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index d4c171d8..30ab9751 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -19,7 +19,6 @@ import img2pdf import pikepdf from pikepdf.models.metadata import encode_pdf_date from PIL import Image, ImageColor, ImageDraw -from tqdm import tqdm from ocrmypdf import leptonica from ocrmypdf._concurrent import Executor @@ -726,7 +725,11 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext): output_file=output_file, compression=options.pdfa_image_compression, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 - progressbar_class=tqdm if options.progress_bar else None, + progressbar_class=( + context.plugin_manager.hook.get_progress_bar() + if options.progress_bar + else None + ), ) return output_file diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 9a37cad6..60de98fe 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -15,7 +15,10 @@ from pathlib import Path from typing import AnyStr, BinaryIO, Iterable, Optional, Union from warnings import warn -from ocrmypdf._logging import PageNumberFilter, TqdmConsole +from ocrmypdf._logging import ( # pylint: disable=unused-import + PageNumberFilter, + TqdmConsole, +) from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf._sync import run_pipeline from ocrmypdf._validation import check_options @@ -47,6 +50,7 @@ def configure_logging( verbosity: Verbosity, progress_bar_friendly: bool = True, manage_root_logger: bool = False, + plugin_manager=None, ): """Set up logging. @@ -74,12 +78,13 @@ def configure_logging( their own debug logging. Args: - verbosity (Verbosity): Verbosity level. - progress_bar_friendly (bool): Install the TqdmConsole log handler, which is + verbosity: Verbosity level. + progress_bar_friendly: Install the TqdmConsole log handler, which is compatible with the tqdm progress bar; without this log messages will - overwrite the progress bar - manage_root_logger (bool): Configure the process's root logger, to ensure + overwrite the progress bar. + manage_root_logger: Configure the process's root logger, to ensure all log output is sent through + plugin_manager: The plugin manager. Returns: The toplevel logger for ocrmypdf (or the root logger, if we are managing it). @@ -90,8 +95,8 @@ def configure_logging( log = logging.getLogger(prefix) log.setLevel(logging.DEBUG) - if progress_bar_friendly: - console = logging.StreamHandler(stream=TqdmConsole(sys.stderr)) + if plugin_manager and progress_bar_friendly: + console = plugin_manager.hook.get_logging_console() else: console = logging.StreamHandler(stream=sys.stderr) @@ -245,6 +250,7 @@ def ocr( # pylint: disable=unused-argument user_patterns: os.PathLike = None, fast_web_view: float = None, plugins: Iterable[StrPath] = None, + plugin_manager=None, keep_temporary_files: bool = None, progress_bar: bool = None, **kwargs, @@ -296,6 +302,9 @@ def ocr( # pylint: disable=unused-argument Returns: :class:`ocrmypdf.ExitCode` """ + if plugins and plugin_manager: + raise ValueError("plugins= and plugin_manager are mutually exclusive") + if not plugins: plugins = [] elif isinstance(plugins, (str, Path)): @@ -315,7 +324,8 @@ def ocr( # pylint: disable=unused-argument # they might install different plugins, and generally speaking we have areas # of code that use global state. - plugin_manager = get_plugin_manager(plugins) + if not plugin_manager: + plugin_manager = get_plugin_manager(plugins) plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member if 'verbose' in kwargs: diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index df6d42c6..5bfc05b3 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -27,6 +27,7 @@ from typing import Callable, Iterable, Optional, Union from tqdm import tqdm from ocrmypdf import Executor, hookimpl +from ocrmypdf._logging import TqdmConsole from ocrmypdf.exceptions import InputFileError Queue = Union[multiprocessing.Queue, queue.Queue] @@ -161,3 +162,8 @@ def get_executor(): @hookimpl def get_progress_bar(): return tqdm + + +@hookimpl +def get_logging_console(): + return logging.StreamHandler(stream=TqdmConsole(sys.stderr)) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 112a26c3..7acb9241 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -423,7 +423,7 @@ def transcode_jpegs( pbar.update() executor( - use_threads=True, + use_threads=True, # Processes are significantly slower at this task max_workers=options.jobs, tqdm_kwargs=dict( desc="JPEGs", diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index cf68fd77..ee4e8dc4 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -8,6 +8,7 @@ from abc import ABC, abstractmethod, abstractstaticmethod from argparse import ArgumentParser, Namespace from collections import namedtuple +from logging import Handler from pathlib import Path from typing import TYPE_CHECKING, AbstractSet, Callable, Iterable, List, Optional @@ -27,6 +28,11 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument +@hookspec +def get_logging_console() -> Handler: + """Returns a logging handler. Should be configured to handle progress bars.""" + + @hookspec def add_options(parser: ArgumentParser) -> None: """Allows the plugin to add its own command line and API arguments. From dccdcfaa913db3c4f4258927a806f7eb36a8ed36 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 02:46:09 -0800 Subject: [PATCH 795/880] leptonica: tidy --- src/ocrmypdf/leptonica.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 69f6e444..c8759cfa 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -13,10 +13,8 @@ import argparse import logging import os -import platform import sys import threading -import warnings from collections import deque from collections.abc import Sequence from contextlib import suppress @@ -25,6 +23,7 @@ from functools import lru_cache from io import BytesIO, UnsupportedOperation from os import fspath from tempfile import TemporaryFile +from warnings import warn from ocrmypdf.exceptions import MissingDependencyError from ocrmypdf.lib._leptonica import ffi @@ -390,7 +389,7 @@ class Pix(LeptonicaObject): @classmethod def read(cls, path): - warnings.warn('Use Pix.open() instead', DeprecationWarning) + warn('Use Pix.open() instead', DeprecationWarning) return cls.open(path) @classmethod From 85c6a974ca4ef8335d6e413ad008d959535d951a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 02:46:44 -0800 Subject: [PATCH 796/880] Fix calls to hook.get_executor --- docs/plugins.rst | 2 +- src/ocrmypdf/_concurrent.py | 4 ++-- src/ocrmypdf/_pipeline.py | 2 +- src/ocrmypdf/builtin_plugins/concurrency.py | 6 +++--- src/ocrmypdf/pluginspec.py | 8 ++++---- 5 files changed, 11 insertions(+), 11 deletions(-) diff --git a/docs/plugins.rst b/docs/plugins.rst index f1c3037c..fad890e7 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -140,7 +140,7 @@ Execution and progress reporting .. autofunction:: ocrmypdf.pluginspec.get_executor -.. autofunction:: ocrmypdf.pluginspec.get_progress_bar +.. autofunction:: ocrmypdf.pluginspec.get_progressbar_class Applying special behavior before processing ------------------------------------------- diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index ea975bae..5882d962 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -107,8 +107,8 @@ class Executor(ABC): def setup_executor(plugin_manager) -> Executor: - pbar_class = plugin_manager.hook.get_progress_bar() - return plugin_manager.hook.get_executor(pbar_class=pbar_class) + pbar_class = plugin_manager.hook.get_progressbar_class() + return plugin_manager.hook.get_executor(progressbar_class=pbar_class) class SerialExecutor(Executor): diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 30ab9751..0871aeba 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -726,7 +726,7 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext): compression=options.pdfa_image_compression, pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3 progressbar_class=( - context.plugin_manager.hook.get_progress_bar() + context.plugin_manager.hook.get_progressbar_class() if options.progress_bar else None ), diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 5bfc05b3..b58211e4 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -155,12 +155,12 @@ class StandardExecutor(Executor): @hookimpl -def get_executor(): - return StandardExecutor() +def get_executor(progressbar_class): + return StandardExecutor(pbar_class=progressbar_class) @hookimpl -def get_progress_bar(): +def get_progressbar_class(): return tqdm diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index ee4e8dc4..3ed8c4af 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -28,7 +28,7 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument -@hookspec +@hookspec(firstresult=True) def get_logging_console() -> Handler: """Returns a logging handler. Should be configured to handle progress bars.""" @@ -70,7 +70,7 @@ def check_options(options: Namespace) -> None: @hookspec(firstresult=True) -def get_executor() -> Executor: +def get_executor(progressbar_class) -> Executor: """Called to obtain an object that manages parallel execution. This may be used to replace OCRmyPDF's default parallel execution system @@ -92,7 +92,7 @@ def get_executor() -> Executor: @hookspec(firstresult=True) -def get_progress_bar(): +def get_progressbar_class(): """Called to obtain a class that can be used to create progress bars. The class should follow a tqdm-like protocol. Calling the class should return @@ -108,7 +108,7 @@ def get_progress_bar(): Here is how OCRmyPDF will use the progress bar: Example: - pbar_class = pm.hook.get_progress_bar() + pbar_class = pm.hook.get_progressbar_class() with pbar_class(**tqdm_kwargs) as pbar: ... pbar.update(1) From 6c8f9223e9e3e18f95ccd9cd60c77e0b6a6d86d0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 03:00:05 -0800 Subject: [PATCH 797/880] Update awslambda to new pluginspec --- src/ocrmypdf/api.py | 4 +- src/ocrmypdf/extra_plugins/awslambda.py | 74 ++++++++++++------------- 2 files changed, 37 insertions(+), 41 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 60de98fe..859e4a82 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -95,9 +95,11 @@ def configure_logging( log = logging.getLogger(prefix) log.setLevel(logging.DEBUG) + console = None if plugin_manager and progress_bar_friendly: console = plugin_manager.hook.get_logging_console() - else: + + if not console: console = logging.StreamHandler(stream=sys.stderr) if verbosity < 0: diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 64297f67..23fe9799 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -20,6 +20,7 @@ from typing import Callable, Iterable, Optional from unittest.mock import Mock from ocrmypdf import Executor, hookimpl +from ocrmypdf._concurrent import NullProgressBar from ocrmypdf.exceptions import InputFileError @@ -98,24 +99,6 @@ class LambdaExecutor(Executor): task_finished(result, self.pbar_class) return - self._lambda_pool_impl( - max_workers=max_workers, - worker_initializer=worker_initializer, - task=task, - task_arguments=task_arguments, - task_finished=task_finished, - ) - - def _lambda_pool_impl( - self, - *, - max_workers: int, - worker_initializer: Callable, - task: Callable, - task_arguments: Iterable, - task_finished: Callable, - pbar, - ): task_arguments = list(task_arguments) grouped_args = list( zip_longest(*list(split_every(max_workers, task_arguments))) @@ -146,32 +129,43 @@ class LambdaExecutor(Executor): for process in processes: process.start() - while connections: - for r in wait(connections): - try: - msg_type, msg = r.recv() - except EOFError: - connections.remove(r) - continue + with self.pbar_class(**tqdm_kwargs) as pbar: + while connections: + for r in wait(connections): + try: + msg_type, msg = r.recv() + except EOFError: + connections.remove(r) + continue - if msg_type == MessageType.result: - if task_finished: - task_finished(msg, pbar) - elif msg_type == 'log': - record = msg - logger = logging.getLogger(record.name) - logger.handle(record) - elif msg_type == MessageType.complete: - connections.remove(r) - elif msg_type == MessageType.exception: - for process in processes: - process.terminate() - raise msg + if msg_type == MessageType.result: + if task_finished: + task_finished(msg, pbar) + elif msg_type == 'log': + record = msg + logger = logging.getLogger(record.name) + logger.handle(record) + elif msg_type == MessageType.complete: + connections.remove(r) + elif msg_type == MessageType.exception: + for process in processes: + process.terminate() + raise msg for process in processes: process.join() @hookimpl -def get_executor(): - return LambdaExecutor() +def get_executor(progressbar_class): + return LambdaExecutor(pbar_class=progressbar_class) + + +@hookimpl +def get_logging_console(): + return logging.StreamHandler() + + +@hookimpl +def get_progressbar_class(): + return NullProgressBar From 206c675df68bbca06a7358b6c7a137d5f27d3032 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 31 Jan 2021 19:26:35 -0800 Subject: [PATCH 798/880] docs: api --- src/ocrmypdf/api.py | 17 +++++++++-------- src/ocrmypdf/pluginspec.py | 28 +++++++++++++++++++++++++--- 2 files changed, 34 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 859e4a82..9bce5352 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -55,12 +55,15 @@ def configure_logging( """Set up logging. Before calling :func:`ocrmypdf.ocr()`, you can use this function to - configure logging, if you want ocrmypdf's output to look like the ocrmypdf + configure logging if you want ocrmypdf's output to look like the ocrmypdf command line interface. It will register log handlers, log filters, and formatters, configure color logging to standard error, and adjust the log levels of third party libraries. Details of this are fine-tuned and subject to change. The ``verbosity`` argument is equivalent to the argument - ``--verbose`` and applies those settings. + ``--verbose`` and applies those settings. If you have a wrapper + script for ocrmypdf and you want it to be very similar to ocrmypdf, use this + function; if you are using ocrmypdf as part of an application that manages + its own logging, you probably do not want this function. If this function is not called, ocrmypdf will not configure logging, and it is up to the caller of ``ocrmypdf.ocr()`` to set up logging as it wishes using @@ -79,12 +82,10 @@ def configure_logging( Args: verbosity: Verbosity level. - progress_bar_friendly: Install the TqdmConsole log handler, which is - compatible with the tqdm progress bar; without this log messages will - overwrite the progress bar. - manage_root_logger: Configure the process's root logger, to ensure - all log output is sent through - plugin_manager: The plugin manager. + progress_bar_friendly: If True (the default), install a custom log handler + that is compatible with progress bars and colored output. + manage_root_logger: Configure the process's root logger. + plugin_manager: The plugin manager, used for obtaining the custom log handler. Returns: The toplevel logger for ocrmypdf (or the root logger, if we are managing it). diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 3ed8c4af..d0c10239 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -30,7 +30,14 @@ hookspec = pluggy.HookspecMarker('ocrmypdf') @hookspec(firstresult=True) def get_logging_console() -> Handler: - """Returns a logging handler. Should be configured to handle progress bars.""" + """Returns a custom logging handler. + + Generally this is necessary when both logging output and a progress bar are both + outputting to ``sys.stderr``. + + Note: + This is a :ref:`firstresult hook`. + """ @hookspec @@ -78,11 +85,16 @@ def get_executor(progressbar_class) -> Executor: distributed environment. OCRmyPDF's executors are analogous to the standard Python executors in - ``conconcurrent.futures``, but they do not work the same way. + ``conconcurrent.futures``, but they do not work the same way. Executors may + be reused for different, unrelated batch operations, since all of the context + for a given job are passed to :meth:`Executor.__call__`. Should be of type :class:`Executor` or otherwise conforming to the protocol of that call. + Arguments: + progressbar_class: A progress bar class, which will be created when + Note: This hook will be called from the main process, and may modify global state before child worker processes are forked. @@ -93,11 +105,15 @@ def get_executor(progressbar_class) -> Executor: @hookspec(firstresult=True) def get_progressbar_class(): - """Called to obtain a class that can be used to create progress bars. + """Called to obtain a class that can be used to monitor progress. + + A progress bar is assumed, but this could be used for any type of monitoring. The class should follow a tqdm-like protocol. Calling the class should return a new progress bar object, which is activated with ``__enter__`` and terminated ``__exit__``. An update method is called whenever the progress bar is updated. + Progress bar objects will not be reused; a new one will be created for each + group of tasks. The progress bar is held in the main process/thread and not updated by child process/threads. When a child notifies the parent of completed work, the @@ -105,6 +121,12 @@ def get_progressbar_class(): The arguments are the same as `tqdm `_ accepts. + Progress bars should never write to ``sys.stdout``, or they will corrupt the + output if OCRmyPDF writes a PDF to standard output. + + The type of events that OCRmyPDF reports to a progress bar may change in + minor releases. + Here is how OCRmyPDF will use the progress bar: Example: From 390fdf8c05f5a07f25748add31d3775d7a7cf1fb Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 7 Dec 2020 21:57:10 -0800 Subject: [PATCH 799/880] Package OCR in Form XObject Should improve results in some situations where the initial content stream is messy or not well-formed. --- src/ocrmypdf/_graft.py | 51 +- .../pdf.bin | Bin 4036 -> 4036 bytes .../pdf.bin | Bin 3501 -> 3501 bytes .../pdf.bin | Bin 2962 -> 2962 bytes .../pdf.bin | Bin 4068 -> 4068 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 2989 -> 2989 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 10251 -> 10251 bytes .../hocr.bin | 1776 +++++++-------- .../txt.bin | 6 +- .../pdf.bin | Bin 11311 -> 11311 bytes .../hocr.bin | 1984 ++++++++--------- .../txt.bin | 169 +- .../pdf.bin | Bin 11615 -> 11615 bytes .../hocr.bin | 1974 ++++++++-------- .../txt.bin | 165 +- .../pdf.bin | Bin 12553 -> 12553 bytes .../hocr.bin | 68 +- .../pdf.bin | Bin 10251 -> 10251 bytes .../pdf.bin | Bin 3626 -> 3626 bytes .../pdf.bin | Bin 3310 -> 3310 bytes .../hocr.bin | 754 +++---- .../txt.bin | 18 +- .../pdf.bin | Bin 5972 -> 5972 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 2853 -> 2853 bytes tests/cache/manifest.jsonl | 108 +- .../hocr.bin | 2 +- .../pdf.bin | Bin 5225 -> 5225 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 4291 -> 4291 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 4042 -> 4042 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 8641 -> 8641 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 8230 -> 8213 bytes .../txt.bin | 4 +- .../hocr.bin | 2 +- .../pdf.bin | Bin 5766 -> 5766 bytes .../pdf.bin | Bin 11558 -> 10903 bytes .../stderr.bin | 2 - .../txt.bin | 2 - .../stderr.bin | 3 - .../hocr.bin | 2 +- .../pdf.bin | Bin 2798 -> 2798 bytes .../hocr.bin | 2 +- .../pdf.bin | Bin 12736 -> 12736 bytes 49 files changed, 3530 insertions(+), 3576 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 33d18e13..41db5e5e 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -6,22 +6,31 @@ import logging +import uuid from contextlib import suppress from pathlib import Path from typing import Optional import pikepdf +from pikepdf.objects import Dictionary, Name log = logging.getLogger(__name__) MAX_REPLACE_PAGES = 100 -def _update_page_resources(*, page, font, font_key, procset): - """Update this page's fonts with a reference to the Glyphless font""" +def _ensure_dictionary(obj, name): + if name not in obj: + obj[name] = pikepdf.Dictionary({}) + return obj[name] - if '/Resources' not in page: - page['/Resources'] = pikepdf.Dictionary({}) - resources = page['/Resources'] + +def _update_resources(*, obj, font, font_key, procset): + """Update this obj's fonts with a reference to the Glyphless font. + + obj can be a page or Form XObject. + """ + + resources = _ensure_dictionary(obj, '/Resources') try: fonts = resources['/Font'] except KeyError: @@ -32,7 +41,8 @@ def _update_page_resources(*, page, font, font_key, procset): # Reassign /ProcSet to one that just lists everything - ProcSet is # obsolete and doesn't matter but recommended for old viewer support - resources['/ProcSet'] = procset + if procset: + resources['/ProcSet'] = procset def strip_invisible_text(pdf, page): @@ -169,13 +179,13 @@ class OcrGrafter: """ page0 = self.pdf_base.pages[0] - _update_page_resources( - page=page0, font=self.font, font_key=self.font_key, procset=self.procset + _update_resources( + obj=page0, font=self.font, font_key=self.font_key, procset=self.procset ) # We cannot read and write the same file, that will corrupt it # but we don't to keep more copies than we need to. Delete intermediates. - # {interim_count} is the opened file we were updateing + # {interim_count} is the opened file we were updating # {interim_count - 1} can be deleted # {interim_count + 1} is the new file will produce and open old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf') @@ -210,6 +220,7 @@ class OcrGrafter: pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {}) except (AttributeError, IndexError, KeyError): return None, None + pdf_text_font = None for f in possible_font_names: pdf_text_font = pdf_text_fonts.get(f, None) if pdf_text_font is not None: @@ -279,17 +290,29 @@ class OcrGrafter: # finally move the lower left corner to match the mediabox ctm = translate @ rotate @ scale @ untranslate @ corner - pdf_text_contents = ( - b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n' + base_resources = _ensure_dictionary(base_page, '/Resources') + base_xobjs = _ensure_dictionary(base_resources, '/XObject') + text_xobj_name = Name('/' + str(uuid.uuid4())) + xobj = self.pdf_base.make_stream(pdf_text_contents) + base_xobjs[text_xobj_name] = xobj + xobj.Type = Name.XObject + xobj.Subtype = Name.Form + xobj.FormType = 1 + xobj.BBox = mediabox + _update_resources( + obj=xobj, font=font, font_key=font_key, procset=[Name.PDF] ) - new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents) + pdf_draw_xobj = ( + (b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n' + ) + new_text_layer = pikepdf.Stream(self.pdf_base, pdf_draw_xobj) if strip_old_text: strip_invisible_text(self.pdf_base, base_page) base_page.page_contents_add(new_text_layer, prepend=True) - _update_page_resources( - page=base_page, font=font, font_key=font_key, procset=procset + _update_resources( + obj=base_page, font=font, font_key=font_key, procset=procset ) diff --git a/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/2400dpi/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 86270f4985ce464c4149388899fd2a2a7b326a26..023d341fe7f1307cf8e660c74b5f5738d2911594 100644 GIT binary patch delta 26 hcmX>ie?)#m5Ff9hk%769p{cQvfv$nY=6JqTMgVA<2KE2| delta 26 hcmX>ie?)#m5Ff9BnURs9nW3eTv95vn=6JqTMgVC02L1p5 diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 1617a5b9bcc00c3bc95a3f7faa1d6eb204b8d01a..edfe4b01d78a195baea095045821ed779bcc78f4 100644 GIT binary patch delta 26 hcmZ20y;gdIH4m?$k%769p{bFvsjh*=W)GfJMgU -

+

diff --git a/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/aspect/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 4601a61056deed113e83ae5dd0077b8590e8ef60..a0e93b4114a87d718e276a7a65074e736c6f0b1f 100644 GIT binary patch delta 26 hcmZ20zE*sLH5ad;k%769p{bFnxvqi5W)H4ZMgU+_28#dy delta 26 hcmZ20zE*sLH5adenURs9nW3eLv95vnW)H4ZMgU-T28;jz diff --git a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index ac662824..51333eae 100644 --- a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index ec8b419ea1e3a4a16e88c88f5196802e2d7d3b59..394e7bbdd0ca074128d8b9007ca16786301ff787 100644 GIT binary patch delta 26 hcmeAU=nmL0Q;pZq$iUpl(A3z_Lf61z^D?zmMgVX92de-8 delta 26 hcmeAU=nmL0Q;pZa%*e>l%+S)*T-U&S^D?zmMgVYP2eSYG diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin index b8ce17c6..23a18626 100644 --- a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/hocr.bin @@ -9,1054 +9,1054 @@ -

-
-

- - The - LinnSequencer +

+
+

+ + The + LinnSequencer - - 32 - Track - MIDI - Sequence - Recorder + + 32 + Track + MIDI + Sequence + Recorder

-
-

- - The - LinnSequencer - is - a - state-of-the-art - composition - and - performance - tool - for - the - professional - musician. - It - is +

+

+ + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

-

- - extremely - powerful, - yet - amazingly - simple - to - learn - and - use. - It’s - many - remarkable - features - include: +

+ + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

-

- - ¢ - Operation - is - similar - to - multi-track - tape - recorder - with - PLAY, - STOP, - RECORD, - FAST +

+ + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST - - FORWARD, - REWIND, - and - LOCATE - controls. + + FORWARD, + REWIND, + and + LOCATE + controls.

-
-

- - e - Each - of - the - 100 - sequences - contains - 32 - simultaneous, - polyphonic - tracks. - Each - track - may +

+

+ + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may - - be - assigned - to - one - of - 16 - MIDI - channels. - Simultaneously - plays - up - to - 16 - polyphonic + + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic

-
-

- - synthesizers! +

+

+ + synthesizers!

-
-

- - ¢ - Ultra-fast - 3%” - disk - drive - stores - complex - songs - in - seconds - and - holds - over - 110,000 - notes +

+

+ + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes

-
-

- - per - disk! +

+

+ + per + disk!

-
-

- - ¢ - One - or - all - tracks - may - be - TRANSPOSED - at - the - touch - of - a - key. +

+

+ + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. - - e - Exclusive - real-time - ERASE - function - makes - editing - FAST. + + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. - - * - Exclusive - REPEAT - function - automatically - repeats - any - held - notes - at - a - pre-selected + + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected

-
-

- - rhythmic - value. +

+

+ + rhythmic + value.

-
-

- - ¢ - TIMING - CORRECTION - works - during - playback - and - operates - without - ‘chopping’ - notes. +

+

+ + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes.

-
-

- - ¢ - Optional - SMPTE - time - code - synchronization. +

+

+ + ¢ + Optional + SMPTE + time + code + synchronization.

-
-

- - © - Optional - remote - control. +

+

+ + © + Optional + remote + control.

-
-

- - Recording - a - Sequence +

+

+ + Recording + a + Sequence

-

- - To - record - a - sequence, - simply - press - RECORD - and - PLAY, +

+ + To + record + a + sequence, + simply + press + RECORD + and + PLAY, - - then - play - your - MIDI - keyboard - in - time - to - the - Sequencer’s + + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s - - click - track. - When - the - sequence - loops - back - around - to - bar - 1, + + click + track. + When + the + sequence + loops + back + around + to + bar + 1, - - you’ - ll - hear - what - you - played—only - all - timing - errors - will - be + + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be

-
-

- - corrected! - (Timing - correction - may - be - adjusted - or - defeated). +

+

+ + corrected! + (Timing + correction + may + be + adjusted + or + defeated).

-
-

- - Any - additional - notes - played - will - be - added - into - the - track +

+

+ + Any + additional + notes + played + will + be + added + into + the + track - - — - existing - notes - are - not - erased - while - recording! + + — + existing + notes + are + not + erased + while + recording!

-

- - FAST - FORWARD, - REWIND, - and - LOCATE - controls +

+ + FAST + FORWARD, + REWIND, + and + LOCATE + controls - - may - be - used - at - any - time - to - quickly - access - any - location - in + + may + be + used + at + any + time + to + quickly + access + any + location + in - - your - sequence - for - spot-recording. - To - overdub - a - new - part, + + your + sequence + for + spot-recording. + To + overdub + a + new + part, - - select - a - different - track - and - start - recording—while - you + + select + a + different + track + and + start + recording—while + you - - record, - the - first - track - will - play - in - perfect - sync - (unless - you + + record, + the + first + track + will + play + in + perfect + sync + (unless + you - - MUTE - it, - or - SOLO - another - track). - In - this - way, - up - to - 32 + + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 - - tracks - may - be - overdubbed! - All - MIDI - effects - are - recorded + + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded - - including - pitch - bend, - modulation, - velocity, - aftertouch, + + including + pitch + bend, + modulation, + velocity, + aftertouch, - - sustain - pedal, - and - program - changes! + + sustain + pedal, + and + program + changes!

-
-

- - Editing +

+

+ + Editing

-

- - To - erase - a - wrong - note, - simply - hold - ERASE - and - press +

+ + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - - the - note - to - be - erased - just - before - it - plays - in - the - sequence— + + the + note + to + be + erased + just + before + it + plays + in + the + sequence— - - when - played - back, - it - will - be - gone. - Notes - may - also - be + + when + played + back, + it + will + be + gone. + Notes + may + also + be

-
-

- - added, - erased, - or - changed - using - the - SINGLE - STEP - func- +

+

+ + added, + erased, + or + changed + using + the + SINGLE + STEP + func- - - tion. - To - overdub - notes - at - specific - points - within - a - sequence, + + tion. + To + overdub + notes + at + specific + points + within + a + sequence,

-
-

- - Additional - Features +

+

+ + Additional + Features

-
-

- - simply - use - LOCATE, - FAST - FORWARD, - or - REWIND - to +

+

+ + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to - - find - the - desired - bar - number, - then - start - recording. + + find + the + desired + bar + number, + then + start + recording.

-

- - The - INSERT/COPY - function - allows - you - to - move - bars +

+ + The + INSERT/COPY + function + allows + you + to + move + bars - - from - one - location - to - another—in - the - same - sequence - or - a + + from + one + location + to + another—in + the + same + sequence + or + a - - different - one. - For - example, - you - might - insert - a - copy - of - the + + different + one. + For + example, + you + might + insert + a + copy + of + the - - first - verse - between - the - second - chorus - and - the - bridge. + + first + verse + between + the + second + chorus + and + the + bridge. - - DELETE - BARS - operates - the - same - way - to - remove + + DELETE + BARS + operates + the + same + way + to + remove - - unwanted - sections, + + unwanted + sections,

-
-

- - Creating - a - Song +

+

+ + Creating + a + Song

-

- - One - way - to - create - a - song - is - to - record - each - track - all - the +

+ + One + way + to + create + a + song + is + to + record + each + track + all + the - - way - through - (up - to - 999 - bars). - Another - way - is - to - record + + way + through + (up + to + 999 + bars). + Another + way + is + to + record - - each - basic - section - (verse, - chorus, - etc.) - in - individual + + each + basic + section + (verse, + chorus, + etc.) + in + individual - - sequences, - then - use - the - CREATE - SONG - function - to - “chain” + + sequences, + then + use + the + CREATE + SONG + function + to + “chain” - - them - together. - CREATE - SONG - will - then - automatically + + them + together. + CREATE + SONG + will + then + automatically - - copy - all - the - parts - into - a - new - sequence. - If - desired, - you - can + + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can - - even - set - the - last - few - bars - to - repeat - infinitely, - for - a - fadeout. + + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

-
-

- - Composition - Without - Compromise +

+

+ + Composition + Without + Compromise

-

- - The - technology - you - use - should - never - be - so - complex - that +

+ + The + technology + you + use + should + never + be + so + complex + that - - it - interferes - with - the - creative - process. - That’s - precisely - why + + it + interferes + with + the + creative + process. + That’s + precisely + why - - the - LinnSequencer - is - designed - to - let - you - compose, - record + + the + LinnSequencer + is + designed + to + let + you + compose, + record - - and - edit - while - devoting - your - undivided - attention - to - your + + and + edit + while + devoting + your + undivided + attention + to + your - - music. - See - your - Linn - dealer - today - for - a - demonstration! + + music. + See + your + Linn + dealer + today + for + a + demonstration!

-
-

- - * - Simple, - easy - to - learn - operation—the - 32 - character - LCD - display - clearly - guides - you - through - all - operations. - If - needed, - the +

+

+ + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

-
-

- - HELP - button - displays - additional - explanations. +

+

+ + HELP + button + displays + additional + explanations.

-
-

- - * - Non-destructive - recording—existing - notes - are - not - erased - while - recording. +

+

+ + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - - ¢ - Two - FOOTSWITCH - INPUTS - may - be - assigned - to - remotely - control - many - of - the - commonly - used - functions, - including + + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

-
-

- - ERASE, - REPEAT, - PLAY/STOP, - or - LOCATE. +

+

+ + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

-
-

- - ¢ - Iwo - TRIGGER - OUTPUTS - may - be - programmed - to - output - pulses - at - any - selected - note - value. +

+

+ + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

-
-

- - © - Will - sync - to - standard - LinnDrum - or - Linn - 9000 - sync - tone. +

+

+ + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

-
-

- - ® - Utilizes - ultra - high-speed, - 8 - MHz - 80186 - 16 - bit - computer - internally - for - FAST - operation. +

+

+ + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. - - * - TEMPO - may - be - specified - in - BEATS-PER-MINUTE - or - FRAMES-PER-BEAT - at - 24, - 25, - or - 30 - frames - per - second, + + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

-
-

- - (even - drop - frame!) +

+

+ + (even + drop + frame!)

-
-

- - ¢ - TEMPO - may - be - entered - numerically, - adjustable - in - tenths - of - a - Beat-Per-Minute - increments, - or - by - tapping - quarter - notes +

+

+ + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

-
-

- - on - the - TAP - TEMPO - button. +

+

+ + on + the + TAP + TEMPO + button.

-
-

- - ¢ - TEMPO - CHANGES - may - be - programmed - into - a - sequence, - with - smooth - transitions - if - desired. +

+

+ + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - - ¢ - Any - TIME - SIGNATURE - may - be - used, - and - may - be - changed - within - a - song. + + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

-
-

- - nn +

+

+ + linn + + + Linn + Electronics, + Inc.

-
-

- - Linn - Electronics, - Inc. +

+

+ + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - - 18720 - Oxnard - Street, - Tarzana, - CA - 91356 - - - (818) - 708-8131 - TELEX - #298949 - LINN - UR + + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin index 686fd1ac..d3c2e860 100644 --- a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt/txt.bin @@ -103,7 +103,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE. © Will sync to standard LinnDrum or Linn 9000 sync tone. -® Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. * TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, (even drop frame!) @@ -115,9 +115,9 @@ on the TAP TEMPO button. ¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. ¢ Any TIME SIGNATURE may be used, and may be changed within a song. -nn - +linn Linn Electronics, Inc. + 18720 Oxnard Street, Tarzana, CA 91356 (818) 708-8131 TELEX #298949 LINN UR \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin index d8df92f633d9e488eed9062da9460747c099fe92..dd3620450b6bd9011fd5ddc795757f5e23ba5235 100644 GIT binary patch delta 26 hcmZ1 -
-
-

- - 2A - NNI‘I - 6F6867# - XATALL - IE18-80L - (818) +

+
+

+ + The + LinnSequencer + + + 32 + Track + MIDI + Sequence + Recorder

-
-

- - 9SEI6 - VO - “BUBZIRY, - “J0aNS - PIPUXO - OZLEI - - - “Uy - ‘soTUOMOI,q - UUrT - -

-
-
-

- - uut] - -

-
-
-

- - “‘SUOS - B - UIJIM - pasueyo - oq - ABU - pue - ‘posn - oq - AWW - AYN - IVNOIS - AWLL - AUV - - - “parlsop - Jr - SUOTIISUBI} - YIOOUIS - YIM - “BoueNbas - eB - OJUI - pourtueIZOId - 9q - ABU - SFONWHO - OdINAL - e - -

-
-
-

- - ‘uonng - OdNAL - dV - L - 9) - uO - -

-
-
-

- - sojou - Jayienb - Suiddy} - Aq - 10 - ‘syUSTIOIOUI - oINUTIAI-J8g-Jesg - & - JO - sys} - UL - ofquisn(pe - ‘ATTeouIAUINU - paiajus - oq - ABU - OdINALL - e - -

-
-
-

- - (jouer - doup - u3a9) - -

-
-
-

- - “puooes - Jed - souely - O€ - 10 - “SZ - “pz - 18 - [LVAG-MAd-SHN - VU - 10 - ALOANIWAAd-SLVAd - U! - patyoeds - aq - kewl - OAL - « - - - ‘uoTe1odo - [SVx - JO} - Aj[eusoyUT - JoyndUIOd - 11g - 9] - 98108 - ZHI - 8g - ‘poeds-ysry - Bann - soz] - e - -

-
-
-

- - "9U0} - DUAS - 0006 - UUL] - Jo - wNIqUUr] - prepue}s - 0} - OUAS - [ITAA - © - -

-
-
-

- - “ONYBA - 9}OU - poloapes - Aue - Je - sas—nd - jndyno - 07 - pewureigold - 3q - ACW - SL - Ad - LNO - YADONAL - OML - -

-
-
-

- - "ALVOOT - 10 - GOLS/AV - 1d - ‘LWddad - “ASV - -

-
-
-

- - SUIpNpoUr - ‘suOTIOUN] - posn - A[UOUILUOS - 94] - JO - AUBUT - [O1]UOD - AJ9]OWIAI - 0} - PousIsse - oq - ACUI - ST - AdNI - HOLIMSLOO - OME - « - - - “SUIPIONAI - I[IYM - P2sesd - JOU - Iv - $3}OU - BUTISIXO—ZUIPIOIA - SATON.ASOP-UON - -

-
-
-

- - ‘suoneurldxa - peuoyippe - sdeydsip - uowng - g1TqH - -

-
-
-

- - oy] - ‘pepsau - JI - ‘suoneiodo - [ye - yYsnosy] - NOA - sapins - ApIespo - Avfdsip - QO] - Joey - Z7¢ - 9y3—uoeIodo - Urea] - 0} - Ased - ‘aus - « - -

-
-
-

- - jUorel]suowtap - & - IO} - Aepol - Jayeap - uur’] - INOA - dag - ‘dISHUL - - - INOA - 0} - UONUS}]¥ - PaplAIPUN - INOA - SUTJOASp - ITY - ps - pue - - - p1osai - ‘asoduod - no - Jay - 0} - pausisap - st - 1s0uenbesuur’] - oy) - - - Aum - Aposiooid - $,Jeu], - ‘SS9d0Id - SATTBS1D - OY} - YIM - SOIOJIOIUT - - - yey) - xo]dwWI0d - Os - dq - JOA9U - P[NoUsS - osn - NOA - AZopOuYdE} - oy - -

-
-
-

- - ISTUMOIAUIO?) - NOAA - UOHISOdWIO) - -

-
-
-

- - "NOSpr] - B - Oy - ‘AONUTJUT - yada - 0} - seq - Maz - Se] - BY] - Jas - UdAd - - - uvd - NOA - ‘palisap - JJ - ‘souanbes - Mou - ¥B - OVUT - sjied - ou] - [Te - Adoo - - - ATesrewO - Ne - WI) - [IM - ONOS - ALVAAO - JeyIe80} - wey} - - - ,deyd,, - 0} - UOTOUNJ - ONOS - ALVA - ou] - asn - usy] - ‘saouanbes - - - JENPIAIpUt - UI - (“949 - ‘snJOYD - ‘aS1OA) - UOTIDIS - JIseq - Yes - - - Pl0da1 - OF - ST - ABM - JOuIOUY - “(812g - 666 - 01 - dn) - ysnory) - ABM +

+

+ + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

-

- - dU} - [fe - YORI] - YORs - p10991 - 0} - ST - SUOS - B - 9789I9 - 0} - ABM - SUG, - -

-
-
-

- - SUOS - & - SUTVAID - -

-
-
-

- - *suoT}oes - poJUBMUN +

+ + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

-

- - SAOUIOI - 0} - ABM - SWS - dU} - SoyeIodo - SUV - ALATAaG +

+ + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST + + + FORWARD, + REWIND, + and + LOCATE + controls. + +

+
+
+

+ + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may + + + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic + +

+
+
+

+ + synthesizers! + +

+
+
+

+ + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes + +

+
+
+

+ + per + disk! + +

+
+
+

+ + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. + + + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. + + + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected + +

+
+
+

+ + rhythmic + value. + +

+
+
+

+ + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes. + +

+
+
+

+ + ¢ + Optional + SMPTE + time + code + synchronization. + +

+
+
+

+ + © + Optional + remote + control. + +

+
+
+

+ + Recording + a + Sequence

-

- - “OBPLIq - dy} - PUB - SNIOY - PUOdAS - dT]] - Ud9MIAQ - SIDA - ISI +

+ + To + record + a + sequence, + simply + press + RECORD + and + PLAY, + + + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s + + + click + track. + When + the + sequence + loops + back + around + to + bar + 1, + + + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be + +

+
+
+

+ + corrected! + (Timing + correction + may + be + adjusted + or + defeated). + +

+
+
+

+ + Any + additional + notes + played + will + be + added + into + the + track + + + — + existing + notes + are + not + erased + while + recording!

-

- - ay) - Jo - Adoo - B - JJasuT - WYSE - NOAA - ‘afdwexs - 10.f - ‘UO - JUSIN]JIP +

+ + FAST + FORWARD, + REWIND, + and + LOCATE + controls + + + may + be + used + at + any + time + to + quickly + access + any + location + in + + + your + sequence + for + spot-recording. + To + overdub + a + new + part, + + + select + a + different + track + and + start + recording—while + you + + + record, + the + first + track + will + play + in + perfect + sync + (unless + you + + + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 + + + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded + + + including + pitch + bend, + modulation, + velocity, + aftertouch, + + + sustain + pedal, + and + program + changes! + +

+
+
+

+ + Editing

-

- - B - IO - aouaNbas - sues - OY} - UI—JOY - OUP - 0} - UOTIEIO] - 9UO - WOT] +

+ + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - - $1Bq - JAOUI - OF - NOA - sMOTIe - WOTIOUNS - AdOO/IMASNI - OULL + + the + note + to + be + erased + just + before + it + plays + in + the + sequence— + + + when + played + back, + it + will + be + gone. + Notes + may + also + be + +

+
+
+

+ + added, + erased, + or + changed + using + the + SINGLE + STEP + func- + + + tion. + To + overdub + notes + at + specific + points + within + a + sequence, + +

+
+
+

+ + Additional + Features + +

+
+
+

+ + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to + + + find + the + desired + bar + number, + then + start + recording.

-

- - ‘SUIPIONAI - JIVIS - Udy) - “OQuINU - eq - porisop - ay} - puy +

+ + The + INSERT/COPY + function + allows + you + to + move + bars + + + from + one + location + to + another—in + the + same + sequence + or + a + + + different + one. + For + example, + you + might + insert + a + copy + of + the + + + first + verse + between + the + second + chorus + and + the + bridge. + + + DELETE + BARS + operates + the + same + way + to + remove + + + unwanted + sections, + +

+
+
+

+ + Creating + a + Song

-

- - 0} - CNIMAY - 10 - ‘CYVM - Od - LSWA - “AEVOOT - esn - Apduns +

+ + One + way + to + create + a + song + is + to + record + each + track + all + the + + + way + through + (up + to + 999 + bars). + Another + way + is + to + record + + + each + basic + section + (verse, + chorus, + etc.) + in + individual + + + sequences, + then + use + the + CREATE + SONG + function + to + “chain” + + + them + together. + CREATE + SONG + will + then + automatically + + + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can + + + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

-
-

- - sainjeay - [PUOHIPPY - -

-
-
-

- - ‘gouanbas - & - UTYIIM - s]UTOd - a1y1dads - 3¥ - $9100 - QnPIOAO - OL - "UOT} - - - -ouns - dALLS - ATONIS - 24) - Suisn - pasueyo - Jo - ‘pasesa - ‘pappe - - - aq - osye - ABUT - S9]ON - ‘U0 - 9q - ]IIM - 1 - “yoeq - podeyd - uayM - - - —aouanbas - oy] - ul - skeyd - 71 - a10J9q - Isnf - posers - oq - 0} - d]0U - ayy - - - ssaid - pue - ASvwug - ploy - Aydunis - ‘jou - Suomm - & - aseso - OL - -

-
-
-

- - sunipa - -

-
-
-

- - jsesdueyo - ureisoid - pue - ‘fepod - ureysns - - - ‘yonoplalje - ‘AWOOTOA - ‘UOTyeTNpow - ‘pusg - youd - Surpnyour - - - pep10del - are - $199JJ2 - TCTIN - [WV - iPeqqnpseao - aq - Aeur - syoen - - - Ze - 07 - dn - ‘Kem - sie - Uy - *(foeI} - JOyOUR - OJOS - 10 - ALLAN - - - NOA - ssofum) - duAS - yOaysod - ul - Avy - [[IM - Yow] - ISI - 93 - “prooar - - - NOA - 3[IYM—SUIPIOIA - LIBIS - PU - YORI) - TUdIOTJIP - B - JOaTas - - - *y1ed - MOU - B - QNPIsA0 - OL, - “SuIps0daJ-jods - 10} - aouanbes - mno0k - - - UI - UOHBIO] - Aue - ssad0e - ATYOIND - 0} - owt} - Aue - ye - pasn - aq - AvUE - - - SJONUOD - FLIVOOT - pur - ‘ANIMA - ‘CYVMaYOd - LSVd - - - {SUIPIOSAI - {IY - posesa - JOU - se - So]OU - SuTsTXO— - - - yous} - 3U} - OUT - poppe - aq - JIM - poteyd - sajou - yeuonippe - Auy - -

-
-
-

- - *(povesjap - 10 - poysn{pe - oq - ABW - UOTIIII0D - BUTUTT]) - j{paqoeLI09 - -

-
-
-

- - 2q - ][IM - S1OLIe - Sur - [fe - ATUO—patey]d - nod - Jey - Jedy - ]],NOA +

+

+ + Composition + Without + Compromise

-

- - ‘] - req - 0] - punose - yoeq - sdoo] - sduanbas - ay] - Udy - AA - “YOu - Yor +

+ + The + technology + you + use + should + never + be + so + complex + that -

- -

- - §,sa0uaNbas - at} - O] - SUIT) - UI - preogday - [IW] - INO - Avy - usy3 + + it + interferes + with + the + creative + process. + That’s + precisely + why - - AV'1d - pue - (YOON - ssoid - Ayduus - ‘aousnbes - & - p1o09es - OF, + + the + LinnSequencer + is + designed + to + let + you + compose, + record + + + and + edit + while + devoting + your + undivided + attention + to + your + + + music. + See + your + Linn + dealer + today + for + a + demonstration!

-
-

- - g0uaNbas - & - SUIP10I0y] +

+

+ + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

-
-

- - ‘JONWOD - s}JouNaI - TeuONdGO - e +

+

+ + HELP + button + displays + additional + explanations.

-
-

- - "UOTJEZIUOIYUAS - OPOS - UIT} - FLAWS - [euondo - e +

+

+ + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. + + + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

-
-

- - ‘sou - .sulddoys, - noyyM - sayelodo - pue - yoegdvyd - ZuLINp - S¥IOM - NOLLOANNYOO - ONIWILL - e +

+

+ + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

-
-

- - ‘onqea - ory - AY +

+

+ + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

-
-

- - pojoojes-oid - & - ye - sajou - pyoy - Aue - syeadas - ATTeONewWO - Ne - UOTOUNS - [WAdAY - OAISNOX - e - - - ‘LSVJ - SUnIpS - soyeu - UOTOUN - ASV - UA - OUlN-[eal - SAISNIOXY - e - - - ‘Koy - B - JO - YONO} - 941 - 12 - CASOdSNVALL - 0g - ABU - Syde] - [Te - 10 - 9UC - e +

+

+ + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

-
-

- - i - ASIP - Jed +

+

+ + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. + + + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

-
-

- - S9}0U - OOO‘OTT - JOA - SpfOy - puv - SpUOdeS - UT - SBUOS - Xa[AUIOD - So10}S - DALIP - YSIP - , - 74 - € - ISCJ-CNIN +

+

+ + (even + drop + frame!)

-
-

- - jSIOZISOUJUAS +

+

+ + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

-
-

- - stuoydAjod - of - 0} - dn - skeyd - A[snoourynuls - ‘spouueYd - [IW - 9T - JO - duo - 0} - pousisse - oq - - - ABUL - YORI] - YOR - ‘syous) - oruoydAjod - ‘snoouelnurs - 7¢ - SuTeJUOS - ssouUaNbas - QO] - OY} - JO - YORA - e +

+

+ + on + the + TAP + TEMPO + button.

-
-

- - ‘SJONUOS - ATWOOT - pur - ‘GNIMAY - ‘GaVM - OA +

+

+ + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. - - LSVd - ‘GYOOde - AOLS - ‘AV - Td - YIM - Jopsocas - ade} - Yowsj-N[NU - O} - eps - st - UOTLISdO - @ - - - LOPNOUT - SaINjeoy - s[quyIeUlss - AUB - S.JJ - ‘OSN - pue - UIes] - 0} - o[duns - A[suIzeUe - JOA - ‘PnJsomod - APOUIOITXO - - - St - 1] - “UeIOIsNUL - feUOIssajoid - oY} - 10 - JOO} - soUBULIOJIJAd - pue - UOTIsOduIOS - 11e-dY1-JO-9}e)s - B - SI - IONUANbDaguUT] - ay + + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

-
-

- - JOps1odady - soUINbIS - [GTI - YVAL - ZE +

+

+ + linn - - Jgouanbaguury - oy + + Linn + Electronics, + Inc. + +

+
+
+

+ + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 + + + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin index d80b111f..d3c2e860 100644 --- a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/txt.bin @@ -1,128 +1,123 @@ -2A NNI‘I 6F6867# XATALL IE18-80L (818) +The LinnSequencer +32 Track MIDI Sequence Recorder -9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI -“Uy ‘soTUOMOI,q UUrT +The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is -uut] +extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: -“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV -“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e +¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST +FORWARD, REWIND, and LOCATE controls. -‘uonng OdNAL dV L 9) uO +e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may +be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic -sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e +synthesizers! -(jouer doup u3a9) +¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes -“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « -‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e +per disk! -"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © +¢ One or all tracks may be TRANSPOSED at the touch of a key. +e Exclusive real-time ERASE function makes editing FAST. +* Exclusive REPEAT function automatically repeats any held notes at a pre-selected -“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML +rhythmic value. -"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV +¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. -SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « -“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON +¢ Optional SMPTE time code synchronization. -‘suoneurldxa peuoyippe sdeydsip uowng g1TqH +© Optional remote control. -oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « +Recording a Sequence -jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL -INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue -p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) -Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT -yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy +To record a sequence, simply press RECORD and PLAY, +then play your MIDI keyboard in time to the Sequencer’s +click track. When the sequence loops back around to bar 1, +you’ ll hear what you played—only all timing errors will be -ISTUMOIAUIO?) NOAA UOHISOdWIO) +corrected! (Timing correction may be adjusted or defeated). -"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd -uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo -ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} -,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes -JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes -Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM +Any additional notes played will be added into the track +— existing notes are not erased while recording! -dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, +FAST FORWARD, REWIND, and LOCATE controls +may be used at any time to quickly access any location in +your sequence for spot-recording. To overdub a new part, +select a different track and start recording—while you +record, the first track will play in perfect sync (unless you +MUTE it, or SOLO another track). In this way, up to 32 +tracks may be overdubbed! All MIDI effects are recorded +including pitch bend, modulation, velocity, aftertouch, +sustain pedal, and program changes! -SUOS & SUTVAID +Editing -*suoT}oes poJUBMUN +To erase a wrong note, simply hold ERASE and press +the note to be erased just before it plays in the sequence— +when played back, it will be gone. Notes may also be -SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG +added, erased, or changed using the SINGLE STEP func- +tion. To overdub notes at specific points within a sequence, -“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI +Additional Features -ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP +simply use LOCATE, FAST FORWARD, or REWIND to +find the desired bar number, then start recording. -B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] -$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL +The INSERT/COPY function allows you to move bars +from one location to another—in the same sequence or a +different one. For example, you might insert a copy of the +first verse between the second chorus and the bridge. +DELETE BARS operates the same way to remove +unwanted sections, -‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy +Creating a Song -0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns +One way to create a song is to record each track all the +way through (up to 999 bars). Another way is to record +each basic section (verse, chorus, etc.) in individual +sequences, then use the CREATE SONG function to “chain” +them together. CREATE SONG will then automatically +copy all the parts into a new sequence. If desired, you can +even set the last few bars to repeat infinitely, for a fadeout. -sainjeay [PUOHIPPY +Composition Without Compromise -‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} --ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe -aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM -—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy -ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL +The technology you use should never be so complex that +it interferes with the creative process. That’s precisely why +the LinnSequencer is designed to let you compose, record +and edit while devoting your undivided attention to your +music. See your Linn dealer today for a demonstration! -sunipa +* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the -jsesdueyo ureisoid pue ‘fepod ureysns -‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour -pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen -Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN -NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar -NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas -*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k -UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE -SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd -{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— -yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy +HELP button displays additional explanations. -*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 +* Non-destructive recording—existing notes are not erased while recording. +¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including -2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA +ERASE, REPEAT, PLAY/STOP, or LOCATE. -‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor +¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. -§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 -AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, +© Will sync to standard LinnDrum or Linn 9000 sync tone. -g0uaNbas & SUIP10I0y] +© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, -‘JONWOD s}JouNaI TeuONdGO e +(even drop frame!) -"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e +¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes -‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e +on the TAP TEMPO button. -‘onqea ory AY +¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. +¢ Any TIME SIGNATURE may be used, and may be changed within a song. -pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e -‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e -‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e +linn +Linn Electronics, Inc. -i ASIP Jed - -S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN - -jSIOZISOUJUAS - -stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq -ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e - -‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA -LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ -LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO -St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay - -JOps1odady soUINbIS [GTI YVAL ZE -Jgouanbaguury oy +18720 Oxnard Street, Tarzana, CA 91356 +(818) 708-8131 TELEX #298949 LINN UR \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin index 9e0083188715b4a4128af76c618f96f90e2b1c01..a9c86e15b00193034bcf2486275e3c6bfe323f43 100644 GIT binary patch delta 26 hcmcZ~bw6su8*N@gBLj0ILsMfT16>1)%|Er%7y*ix2@wDQ delta 26 hcmcZ~bw6su8*N?#Gb1BIGeb*L3ta>A%|Er%7y*kz2_XOg diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin index 2567ab7e..518ac636 100644 --- a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin @@ -9,1070 +9,1054 @@ -
-
-

- - 2A - NNI‘I - 6F6867# - XATALL - IE18-80L - (818) +

+
+

+ + The + LinnSequencer + + + 32 + Track + MIDI + Sequence + Recorder

-
-

- - 9SEI6 - VO - “BUBZIRY, - “J0aNS - PIPUXO - OZLEI - - - “Uy - ‘soTUOMOI,q - UUrT - -

-
-
-

- - uu - -

-
-
-

- - “‘SUOS - B - UIJIM - pasueyo - oq - ABU - pue - ‘posn - oq - AWW - AYN - IVNOIS - AWLL - AUV - - - “parlsop - Jr - SUOTIISUBI} - YIOOUIS - YIM - “BoueNbas - eB - OJUI - pourtueIZOId - 9q - ABU - SFONWHO - OdINAL - e - -

-
-
-

- - ‘uonng - OdNAL - dV - L - 9) - uO - -

-
-
-

- - sojou - Jayienb - Suiddy} - Aq - 10 - ‘syUSTIOIOUI - oINUTIAI-J8g-Jesg - & - JO - sys} - UL - ofquisn(pe - ‘ATTeouIAUINU - paiajus - oq - ABU - OdINALL - e - -

-
-
-

- - (jouer - doup - u3a9) - -

-
-
-

- - “puooes - Jed - souely - O€ - 10 - “SZ - “pz - 18 - [LVAG-MAd-SHN - VU - 10 - ALOANIWAAd-SLVAd - U! - patyoeds - aq - kewl - OAL - « - - - ‘uoTe1odo - [SVx - JO} - Aj[eusoyUT - JoyndUIOd - 11g - 9] - 98108 - ZHI - 8g - ‘poeds-ysry - Bann - soz] - e - -

-
-
-

- - "9U0} - DUAS - 0006 - UUL] - Jo - wNIqUUr] - prepue}s - 0} - OUAS - [ITAA - © - -

-
-
-

- - “ONYBA - 9}OU - poloapes - Aue - Je - sas—nd - jndyno - 07 - pewureigold - 3q - ACW - SL - Ad - LNO - YADONAL - OML - -

-
-
-

- - "ALVOOT - 10 - GOLS/AV - 1d - ‘LWddad - “ASV - -

-
-
-

- - SUIpNpoUr - ‘suOTIOUN] - posn - A[UOUILUOS - 94] - JO - AUBUT - [O1]UOD - AJ9]OWIAI - 0} - PousIsse - oq - ACUI - ST - AdNI - HOLIMSLOO - OME - « - - - “SUIPIONAI - I[IYM - P2sesd - JOU - Iv - $3}OU - BUTISIXO—ZUIPIOIA - SATON.ASOP-UON - -

-
-
-

- - ‘suoneurldxa - peuoyippe - sdeydsip - uowng - g1TqH - -

-
-
-

- - oy] - ‘pepsau - JI - ‘suoneiodo - [ye - yYsnosy] - NOA - sapins - ApIespo - Avfdsip - QO] - Joey - Z7¢ - 9y3—uoeIodo - Urea] - 0} - Ased - ‘aus - « - -

-
-
-

- - jUorel]suowtap - & - IO} - Aepol - Jayeap - uur’] - INOA - dag - ‘dISHUL - - - INOA - 0} - UONUS}]¥ - PaplAIPUN - INOA - SUTJOASp - ITY - ps - pue - - - p1osai - ‘asoduod - no - Jay - 0} - pausisap - st - 1s0uenbesuur’] - oy) - - - Aum - Aposiooid - $,Jeu], - ‘SS9d0Id - SATTBS1D - OY} - YIM - SOIOJIOIUT - - - yey) - xo]dwWI0d - Os - dq - JOA9U - P[NoUsS - osn - NOA - AZopOuYdE} - oy - -

-
-
-

- - ISTUMOIAUIO?) - NOAA - UOHISOdWIO) - -

-
-
-

- - "NOSpr] - B - Oy - ‘AONUTJUT - yada - 0} - seq - Maz - Se] - BY] - Jas - UdAd - - - uvd - NOA - ‘palisap - JJ - ‘souanbes - Mou - ¥B - OVUT - sjied - ou] - [Te - Adoo - - - ATesrewO - Ne - WI) - [IM - ONOS - ALVAAO - JeyIe80} - wey} - - - ,deyd,, - 0} - UOTOUNJ - ONOS - ALVA - ou] - asn - usy] - ‘saouanbes - - - JENPIAIpUt - UI - (“949 - ‘snJOYD - ‘aS1OA) - UOTIDIS - JIseq - Yes - - - Pl0da1 - OF - ST - ABM - JOuIOUY - “(812g - 666 - 01 - dn) - ysnory) - ABM +

+

+ + The + LinnSequencer + is + a + state-of-the-art + composition + and + performance + tool + for + the + professional + musician. + It + is

-

- - dU} - [fe - YORI] - YORs - p10991 - 0} - ST - SUOS - B - 9789I9 - 0} - ABM - SUG, - -

-
-
-

- - SUOS - & - SUTVAID - -

-
-
-

- - *suoT}oes - poJUBMUN +

+ + extremely + powerful, + yet + amazingly + simple + to + learn + and + use. + It’s + many + remarkable + features + include:

-

- - SAOUIOI - 0} - ABM - SWS - dU} - SoyeIodo - SUV - ALATAaG +

+ + ¢ + Operation + is + similar + to + multi-track + tape + recorder + with + PLAY, + STOP, + RECORD, + FAST + + + FORWARD, + REWIND, + and + LOCATE + controls. + +

+
+
+

+ + e + Each + of + the + 100 + sequences + contains + 32 + simultaneous, + polyphonic + tracks. + Each + track + may + + + be + assigned + to + one + of + 16 + MIDI + channels. + Simultaneously + plays + up + to + 16 + polyphonic + +

+
+
+

+ + synthesizers! + +

+
+
+

+ + ¢ + Ultra-fast + 3%” + disk + drive + stores + complex + songs + in + seconds + and + holds + over + 110,000 + notes + +

+
+
+

+ + per + disk! + +

+
+
+

+ + ¢ + One + or + all + tracks + may + be + TRANSPOSED + at + the + touch + of + a + key. + + + e + Exclusive + real-time + ERASE + function + makes + editing + FAST. + + + * + Exclusive + REPEAT + function + automatically + repeats + any + held + notes + at + a + pre-selected + +

+
+
+

+ + rhythmic + value. + +

+
+
+

+ + ¢ + TIMING + CORRECTION + works + during + playback + and + operates + without + ‘chopping’ + notes. + +

+
+
+

+ + ¢ + Optional + SMPTE + time + code + synchronization. + +

+
+
+

+ + © + Optional + remote + control. + +

+
+
+

+ + Recording + a + Sequence

-

- - “OBPLIq - dy} - PUB - SNIOY - PUOdAS - dT]] - Ud9MIAQ - SIDA - ISI +

+ + To + record + a + sequence, + simply + press + RECORD + and + PLAY, + + + then + play + your + MIDI + keyboard + in + time + to + the + Sequencer’s + + + click + track. + When + the + sequence + loops + back + around + to + bar + 1, + + + you’ + ll + hear + what + you + played—only + all + timing + errors + will + be + +

+
+
+

+ + corrected! + (Timing + correction + may + be + adjusted + or + defeated). + +

+
+
+

+ + Any + additional + notes + played + will + be + added + into + the + track + + + — + existing + notes + are + not + erased + while + recording!

-

- - ay) - Jo - Adoo - B - JJasuT - WYSE - NOAA - ‘afdwexs - 10.f - ‘UO - JUSIN]JIP +

+ + FAST + FORWARD, + REWIND, + and + LOCATE + controls + + + may + be + used + at + any + time + to + quickly + access + any + location + in + + + your + sequence + for + spot-recording. + To + overdub + a + new + part, + + + select + a + different + track + and + start + recording—while + you + + + record, + the + first + track + will + play + in + perfect + sync + (unless + you + + + MUTE + it, + or + SOLO + another + track). + In + this + way, + up + to + 32 + + + tracks + may + be + overdubbed! + All + MIDI + effects + are + recorded + + + including + pitch + bend, + modulation, + velocity, + aftertouch, + + + sustain + pedal, + and + program + changes! + +

+
+
+

+ + Editing

-

- - B - IO - aouaNbas - sues - OY} - UI—JOY - OUP - 0} - UOTIEIO] - 9UO - WOT] +

+ + To + erase + a + wrong + note, + simply + hold + ERASE + and + press - - $1Bq - JAOUI - OF - NOA - sMOTIe - WOTIOUNS - AdOO/IMASNI - OULL + + the + note + to + be + erased + just + before + it + plays + in + the + sequence— + + + when + played + back, + it + will + be + gone. + Notes + may + also + be + +

+
+
+

+ + added, + erased, + or + changed + using + the + SINGLE + STEP + func- + + + tion. + To + overdub + notes + at + specific + points + within + a + sequence, + +

+
+
+

+ + Additional + Features + +

+
+
+

+ + simply + use + LOCATE, + FAST + FORWARD, + or + REWIND + to + + + find + the + desired + bar + number, + then + start + recording.

-

- - ‘SUIPIONAI - JIVIS - Udy) - “OQuINU - eq - porisop - ay} - puy +

+ + The + INSERT/COPY + function + allows + you + to + move + bars + + + from + one + location + to + another—in + the + same + sequence + or + a + + + different + one. + For + example, + you + might + insert + a + copy + of + the + + + first + verse + between + the + second + chorus + and + the + bridge. + + + DELETE + BARS + operates + the + same + way + to + remove + + + unwanted + sections, + +

+
+
+

+ + Creating + a + Song

-

- - 0} - CNIMAY - 10 - ‘CYVM - Od - LSWA - “AEVOOT - esn - Apduns +

+ + One + way + to + create + a + song + is + to + record + each + track + all + the + + + way + through + (up + to + 999 + bars). + Another + way + is + to + record + + + each + basic + section + (verse, + chorus, + etc.) + in + individual + + + sequences, + then + use + the + CREATE + SONG + function + to + “chain” + + + them + together. + CREATE + SONG + will + then + automatically + + + copy + all + the + parts + into + a + new + sequence. + If + desired, + you + can + + + even + set + the + last + few + bars + to + repeat + infinitely, + for + a + fadeout.

-
-

- - sainjeay - [PUOHIPPY +

+

+ + Composition + Without + Compromise + +

+ +

+ + The + technology + you + use + should + never + be + so + complex + that + + + it + interferes + with + the + creative + process. + That’s + precisely + why + + + the + LinnSequencer + is + designed + to + let + you + compose, + record + + + and + edit + while + devoting + your + undivided + attention + to + your + + + music. + See + your + Linn + dealer + today + for + a + demonstration!

-
-

- - ‘gouanbas - & - UTYIIM - s]UTOd - a1y1dads - 3¥ - $9100 - QnPIOAO - OL - "UOT} - - - -ouns - dALLS - ATONIS - 24) - Suisn - pasueyo - Jo - ‘pasesa - ‘pappe - - - aq - osye - ABUT - S9]ON - ‘U0 - 9q - ]IIM - 1 - “yoeq - podeyd - uayM - - - —aouanbas - oy] - ul - skeyd - 71 - a10J9q - Isnf - posers - oq - 0} - d]0U - ayy - - - ssaid - pue - ASvwug - ploy - Aydunis - ‘jou - Suomm - & - aseso - OL +

+

+ + * + Simple, + easy + to + learn + operation—the + 32 + character + LCD + display + clearly + guides + you + through + all + operations. + If + needed, + the

-
-

- - sunipa +

+

+ + HELP + button + displays + additional + explanations.

-
-

- - jsesdueyo - ureisoid - pue - ‘fepod - ureysns +

+

+ + * + Non-destructive + recording—existing + notes + are + not + erased + while + recording. - - ‘yonoplalje - ‘AWOOTOA - ‘UOTyeTNpow - ‘pusg - youd - Surpnyour - - - pep10del - are - $199JJ2 - TCTIN - [WV - iPeqqnpseao - aq - Aeur - syoen - - - Ze - 07 - dn - ‘Kem - sie - Uy - *(foeI} - JOyOUR - OJOS - 10 - ALLAN - - - NOA - ssofum) - duAS - yOaysod - ul - Avy - [[IM - Yow] - ISI - 93 - “prooar - - - NOA - 3[IYM—SUIPIOIA - LIBIS - PU - YORI) - TUdIOTJIP - B - JOaTas - - - *y1ed - MOU - B - QNPIsA0 - OL, - “SuIps0daJ-jods - 10} - aouanbes - mno0k - - - UI - UOHBIO] - Aue - ssad0e - ATYOIND - 0} - owt} - Aue - ye - pasn - aq - AvUE - - - SJONUOD - FLIVOOT - pur - ‘ANIMA - ‘CYVMaYOd - LSVd - - - {SUIPIOSAI - {IY - posesa - JOU - se - So]OU - SuTsTXO— - - - yous} - 3U} - OUT - poppe - aq - JIM - poteyd - sajou - yeuonippe - Auy - - - *(povesjap - 10 - poysn{pe - oq - ABW - UOTIIII0D - BUTUTT]) - j{paqoeLI09 - - - 2q - ][IM - S1OLIe - Sur - [fe - ATUO—patey]d - nod - Jey - Jedy - ]],NOA - - - ‘] - req - 0] - punose - yoeq - sdoo] - sduanbas - ay] - Udy - AA - “YOu - Yor - - - §,sa0uaNbas - at} - O] - SUIT) - UI - preogday - [IW] - INO - Avy - usy3 - - - AV'1d - pue - (YOON - ssoid - Ayduus - ‘aousnbes - & - p1o09es - OF, + + ¢ + Two + FOOTSWITCH + INPUTS + may + be + assigned + to + remotely + control + many + of + the + commonly + used + functions, + including

-
-

- - g0uaNbas - & - SUIP10I0y] +

+

+ + ERASE, + REPEAT, + PLAY/STOP, + or + LOCATE.

-
-

- - ‘JONWOD - s}JouNaI - TeuONdGO - e +

+

+ + ¢ + Iwo + TRIGGER + OUTPUTS + may + be + programmed + to + output + pulses + at + any + selected + note + value.

-
-

- - "UOTJEZIUOIYUAS - OPOS - UIT} - FLAWS - [euondo - e +

+

+ + © + Will + sync + to + standard + LinnDrum + or + Linn + 9000 + sync + tone.

-
-

- - ‘sou - .sulddoys, - noyyM - sayelodo - pue - yoegdvyd - ZuLINp - S¥IOM - NOLLOANNYOO - ONIWILL - e +

+

+ + © + Utilizes + ultra + high-speed, + 8 + MHz + 80186 + 16 + bit + computer + internally + for + FAST + operation. + + + * + TEMPO + may + be + specified + in + BEATS-PER-MINUTE + or + FRAMES-PER-BEAT + at + 24, + 25, + or + 30 + frames + per + second,

-
-

- - ‘onqea - ory - AY +

+

+ + (even + drop + frame!)

-
-

- - pojoojes-oid - & - ye - sajou - pyoy - Aue - syeadas - ATTeONewWO - Ne - UOTOUNS - [WAdAY - OAISNOX - e - - - ‘LSVJ - SUnIpS - soyeu - UOTOUN - ASV - UA - OUlN-[eal - SAISNIOXY - e - - - ‘Koy - B - JO - YONO} - 941 - 12 - CASOdSNVALL - 0g - ABU - Syde] - [Te - 10 - 9UC - e +

+

+ + ¢ + TEMPO + may + be + entered + numerically, + adjustable + in + tenths + of + a + Beat-Per-Minute + increments, + or + by + tapping + quarter + notes

-
-

- - i - ASIP - Jed +

+

+ + on + the + TAP + TEMPO + button.

-
-

- - S9}0U - OOO‘OTT - JOA - SpfOy - puv - SpUOdeS - UT - SBUOS - Xa[AUIOD - So10}S - DALIP - YSIP - , - 74 - € - ISCJ-CNIN +

+

+ + ¢ + TEMPO + CHANGES + may + be + programmed + into + a + sequence, + with + smooth + transitions + if + desired. + + + ¢ + Any + TIME + SIGNATURE + may + be + used, + and + may + be + changed + within + a + song.

-
-

- - jSIOZISOUJUAS +

+

+ + linn + + + Linn + Electronics, + Inc.

-
-

- - stuoydAjod - of - 0} - dn - skeyd - A[snoourynuls - ‘spouueYd - [IW - 9T - JO - duo - 0} - pousisse - oq +

+

+ + 18720 + Oxnard + Street, + Tarzana, + CA + 91356 - - ABUL - YORI] - YOR - ‘syous) - oruoydAjod - ‘snoouelnurs - 7¢ - SuTeJUOS - ssouUaNbas - QO] - OY} - JO - YORA - e - -

-
-
-

- - ‘SJONUOS - ATWOOT - pur - ‘GNIMAY - ‘GaVM - OA - - - LSVd - ‘GYOOde - AOLS - ‘AV - Td - YIM - Jopsocas - ade} - Yowsj-N[NU - O} - eps - st - UOTLISdO - @ - - - LOPNOUT - SaINjeoy - s[quyIeUlss - AUB - S.JJ - ‘OSN - pue - UIes] - 0} - o[duns - A[suIzeUe - JOA - ‘PnJsomod - APOUIOITXO - - - St - 1] - “UeIOIsNUL - feUOIssajoid - oY} - 10 - JOO} - soUBULIOJIJAd - pue - UOTIsOduIOS - 11e-dY1-JO-9}e)s - B - SI - IONUANbDaguUT] - ay - -

-
-
-

- - JOps1odady - soUINbIS - [GTI - YVAL - ZE - - - Jgouanbaguury - oy + + (818) + 708-8131 + TELEX + #298949 + LINN + UR

diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin index 137fef56..d3c2e860 100644 --- a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/txt.bin @@ -1,124 +1,123 @@ -2A NNI‘I 6F6867# XATALL IE18-80L (818) +The LinnSequencer +32 Track MIDI Sequence Recorder -9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI -“Uy ‘soTUOMOI,q UUrT +The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is -uu +extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include: -“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV -“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e +¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST +FORWARD, REWIND, and LOCATE controls. -‘uonng OdNAL dV L 9) uO +e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may +be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic -sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e +synthesizers! -(jouer doup u3a9) +¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes -“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL « -‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e +per disk! -"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA © +¢ One or all tracks may be TRANSPOSED at the touch of a key. +e Exclusive real-time ERASE function makes editing FAST. +* Exclusive REPEAT function automatically repeats any held notes at a pre-selected -“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML +rhythmic value. -"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV +¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes. -SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME « -“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON +¢ Optional SMPTE time code synchronization. -‘suoneurldxa peuoyippe sdeydsip uowng g1TqH +© Optional remote control. -oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus « +Recording a Sequence -jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL -INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue -p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy) -Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT -yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy +To record a sequence, simply press RECORD and PLAY, +then play your MIDI keyboard in time to the Sequencer’s +click track. When the sequence loops back around to bar 1, +you’ ll hear what you played—only all timing errors will be -ISTUMOIAUIO?) NOAA UOHISOdWIO) +corrected! (Timing correction may be adjusted or defeated). -"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd -uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo -ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey} -,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes -JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes -Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM +Any additional notes played will be added into the track +— existing notes are not erased while recording! -dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG, +FAST FORWARD, REWIND, and LOCATE controls +may be used at any time to quickly access any location in +your sequence for spot-recording. To overdub a new part, +select a different track and start recording—while you +record, the first track will play in perfect sync (unless you +MUTE it, or SOLO another track). In this way, up to 32 +tracks may be overdubbed! All MIDI effects are recorded +including pitch bend, modulation, velocity, aftertouch, +sustain pedal, and program changes! -SUOS & SUTVAID +Editing -*suoT}oes poJUBMUN +To erase a wrong note, simply hold ERASE and press +the note to be erased just before it plays in the sequence— +when played back, it will be gone. Notes may also be -SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG +added, erased, or changed using the SINGLE STEP func- +tion. To overdub notes at specific points within a sequence, -“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI +Additional Features -ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP +simply use LOCATE, FAST FORWARD, or REWIND to +find the desired bar number, then start recording. -B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT] -$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL +The INSERT/COPY function allows you to move bars +from one location to another—in the same sequence or a +different one. For example, you might insert a copy of the +first verse between the second chorus and the bridge. +DELETE BARS operates the same way to remove +unwanted sections, -‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy +Creating a Song -0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns +One way to create a song is to record each track all the +way through (up to 999 bars). Another way is to record +each basic section (verse, chorus, etc.) in individual +sequences, then use the CREATE SONG function to “chain” +them together. CREATE SONG will then automatically +copy all the parts into a new sequence. If desired, you can +even set the last few bars to repeat infinitely, for a fadeout. -sainjeay [PUOHIPPY +Composition Without Compromise -‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT} --ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe -aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM -—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy -ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL +The technology you use should never be so complex that +it interferes with the creative process. That’s precisely why +the LinnSequencer is designed to let you compose, record +and edit while devoting your undivided attention to your +music. See your Linn dealer today for a demonstration! -sunipa +* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the -jsesdueyo ureisoid pue ‘fepod ureysns -‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour -pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen -Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN -NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar -NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas -*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k -UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE -SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd -{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO— -yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy -*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09 -2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA -‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor -§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3 -AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF, +HELP button displays additional explanations. -g0uaNbas & SUIP10I0y] +* Non-destructive recording—existing notes are not erased while recording. +¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including -‘JONWOD s}JouNaI TeuONdGO e +ERASE, REPEAT, PLAY/STOP, or LOCATE. -"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e +¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value. -‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e +© Will sync to standard LinnDrum or Linn 9000 sync tone. -‘onqea ory AY +© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation. +* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second, -pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e -‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e -‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e +(even drop frame!) -i ASIP Jed +¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes -S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN +on the TAP TEMPO button. -jSIOZISOUJUAS +¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired. +¢ Any TIME SIGNATURE may be used, and may be changed within a song. -stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq -ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e +linn +Linn Electronics, Inc. -‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA -LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @ -LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO -St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay - -JOps1odady soUINbIS [GTI YVAL ZE -Jgouanbaguury oy +18720 Oxnard Street, Tarzana, CA 91356 +(818) 708-8131 TELEX #298949 LINN UR \ No newline at end of file diff --git a/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin b/tests/cache/cardinal/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin index 5c1ac320d98c4e47d50d19b57d4ce6dd21b32076..aa26441b2e506f838b4802fb43f79ce688f05324 100644 GIT binary patch delta 26 hcmeB7>P*@&O`q4$$iUpl(A3z-K-a)x^J4upMgVg`2hjik delta 26 hcmeB7>P*@&O`q4m%*e>l%+S)*QrEzI^J4upMgVj62jTz# diff --git a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 123d2a1e..d06920de 100644 --- a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -
+

@@ -29,7 +29,7 @@

The - LinnSequencer + LinnSequencer is a state-of-the-art @@ -39,7 +39,7 @@ tool for the - professional + professional musician. It is @@ -209,7 +209,7 @@

- rhythmic + rhythmic value.

@@ -246,7 +246,7 @@

- © + © Optional remote control. @@ -300,8 +300,8 @@ 1, - you’ - ll + you’ + ll hear what you @@ -361,7 +361,7 @@ REWIND, and LOCATE - controls + controls may @@ -400,9 +400,9 @@ record, - the - first - track + the + first + track will play in @@ -412,11 +412,11 @@ you - MUTE - it, + MUTE + it, or SOLO - another + another track). In this @@ -518,13 +518,13 @@ tion. To - overdub + overdub notes at specific points - within - a + within + a sequence,

@@ -681,7 +681,7 @@ SONG function to - “chain” + “chain” them @@ -874,10 +874,10 @@

- ¢ - Iwo + ¢ + Iwo TRIGGER - OUTPUTS + OUTPUTS may be programmed @@ -887,7 +887,7 @@ at any selected - note + note value.

@@ -911,15 +911,15 @@

- - © - Utilizes - ultra + + © + Utilizes + ultra high-speed, 8 MHz - 80186 - 16 + 80186 + 16 bit computer internally @@ -960,18 +960,18 @@

- ¢ + ¢ TEMPO may - be + be entered numerically, - adjustable - in + adjustable + in tenths of - a - Beat-Per-Minute + a + Beat-Per-Minute increments, or by diff --git a/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/ccitt/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 13d33a38a99252af3e8e00cd3186c174fc1d3e6a..88bb9fc93541fbb3ab6292c2f7a1d81e0fc4f120 100644 GIT binary patch delta 26 hcmeAU=nmL0Q;pZq$iUpl(A3D-QrEy@^D?zmMgVXR2dw}A delta 26 hcmeAU=nmL0Q;pZa%*e>l%+S) -

-
-

- - 41st - CONGRESS; - | - SENATE. +

+
+

+ + 4ist + ConGREss, + } + SENATE. + { + Ex. + Doc, - - 3d - Session. - } + + 3d + Session. + No. + 25.

-
-

- - MESSAGE +

+

+ + MESSAGE

-
-

- - OF - THE +

+

+ + OF + THE

-
-

- - PRESIDENT - OF - THE - UNITED - STATES, +

+

+ + PRESIDENT + OF + THE + UNITED + STATES,

-
-

- - COMMUNICATING +

+

+ + COMMUNICATING

-
-

- - A - copy - of - regulations - for - the - consular - courts - of - the - United - States - in - Japan, +

+

+ + A + copy + of + regulations + for + the + consular + courts + of + the + United + States + in + Japan, - - decreed - and - issued - by - the - minister - of - the - United - States - in - that - country. + + decreed + and + issued + by + the + minister + of + the + United + States + in + that + country.

-
-

- - January - 27, - 1871,—Read, - referred - to - the - Committee - on - Commerce, - and - ordered - to - be +

+

+ + JANUARY + 27, + 1871,—Read, + referred + to + the + Committee + on + Commerce, + and + ordered + to + be - - printed. + + printed.

-
-

- - To - the - Senate - and - House - of - Representatives - : +

+

+ + To + the + Senate + and + House + of + Representatives + :

-

- - I - transmit - herewith, - for - the - consideration - of - Congress, - a - report - from +

+ + I + transmit + herewith, + for + the + consideration + of + Congress, + a + report + from - - the - Secretary - of - State, - and - the - papers - which - accompanied - it, - concern- + + the + Secretary + of + State, + and + the + papers + which + accompanied + it, + concern- - - ing - regulations - for - the - consular - courts - of - the - United - States - in - Japan. + + ing + regulations + for + the + consular + courts + of + the + United + States + in + Japan.

-

- - U. - 8. - GRANT. +

+ + U. + 8. + GRANT.

-

- - ‘WASHINGTON, - January - 27, - 1871. +

+ + ‘WASHINGTON, + January + 27, + 1871.

-
-

- - DEPARTMENT - OF - STATE, +

+

+ + DEPARTMENT + OF + STATE, - - . - Washington, - January - 26, - 1870, + + Washington, + January + 26, + 1870,

-

- - The - Secretary - of - State - has - the - honor - to - submit - herewith, - for - revision +

+ + The + Secretary + of + State + has + the + honor + to + submit + herewith, + for + revision - - by - Congress, - in - conformity - with - the - provisions - of - section - 6 - of - the - act + + by + Congress, + in + conformity + with + the + provisions + of + section + 6 + of + the + act - - approved - 22d - of - June, - 1860, - a - copy - of - “regulations - for - the - consular + + approved + 22d + of + June, + 1860, + a + copy + of + “regulations + for + the + consular - - courts - of - the - United - States - in - Japan,” - decreed - and - issued - by - C. - E. + + courts + of + the + United + States + in + Japan,” + decreed + and + issued + by + C. + BE. - - De - Long, - the - miniater - of - the - United - States - in - that - country, - in - Septem- + + De + Long, + the + minister + of + the + United + States + in + that + country, + in + Septem- - - ber, - 1870; - and - also - the - papers - mentioned - in - the - subjoined - list, - which, + + ber, + 1870; + and + also + the + papers + mentioned + in + the + subjoined + list, + which, - - contain - suggestions - on - the - subject - thereof. + + contain + suggestions + on + the + subject + thereof.

-

- - A - copy - of - Article - XXVI - of - the - consular - regulations - is - also - submitted, +

+ + A + copy + of + Article + XXVI + of + the + consular + regulations + is + also + submitted, - - and - the - Secretary - of - State - respectfully - suggests, - for - the - consideration + + and + the + Secretary + of + State + respectfully + suggests, + for + the + consideration - - of - Congress, - the - propriety - of - limiting - the - power - of - ministers - to - make + + of + Congress, + the + propriety + of + limiting + the + power + of + ministers + to + make - - decrees - and - regulation, - in - the - sense - in - which - it - is - limited - by - paragraph + + decrees + and + regulation, + in + the + sense + in + which + it + is + limited + by + paragraph - - 431 - of - the - article - before - named—that - is, - “to - acts - necessary - to - organize + + 431 + of + the + article + before + named—that + is, + “to + acts + necessary + to + organize - - and - give - efficiency - to - the - courts - created - by - the - act.” + + and + give + efficiency + to + the + courts + created + by + the + act.”

-

- - Respectfully - submitted. - . +

+ + Respectfully + submitted.

-

- - HAMILTON - FISH. +

+ + HAMILTON + FISH.

-
-

- - The - PRESIDENT, +

+

+ + The + PRESIDENT,

-
-

- - List - of - accompanying - papers. +

+

+ + List + of + accompanying + papers.

-
-

- - 1, - Regulations - for - the - consular - courts - of - the - United - States - in - Japan. +

+

+ + 1, + Regulations + for + the + consular + courts + of + the + United + States + in + Japan. - - 2. - Mr. - Fish - to - Mr. - De - Long, - September - 10, - 1870. - . + + 2, + Mr. + Fish + to + Mr. + De + Long, + September + 10, + 1870,

- - + +

-
-

- - +

+

+ + + +

+
+
+

+ +

diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin index a5ca4032..a450f78e 100644 --- a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin +++ b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/txt.bin @@ -1,5 +1,5 @@ -41st CONGRESS; | SENATE. -3d Session. } +4ist ConGREss, } SENATE. { Ex. Doc, +3d Session. No. 25. MESSAGE @@ -12,7 +12,7 @@ COMMUNICATING A copy of regulations for the consular courts of the United States in Japan, decreed and issued by the minister of the United States in that country. -January 27, 1871,—Read, referred to the Committee on Commerce, and ordered to be +JANUARY 27, 1871,—Read, referred to the Committee on Commerce, and ordered to be printed. To the Senate and House of Representatives : @@ -26,13 +26,13 @@ U. 8. GRANT. ‘WASHINGTON, January 27, 1871. DEPARTMENT OF STATE, -. Washington, January 26, 1870, +Washington, January 26, 1870, The Secretary of State has the honor to submit herewith, for revision by Congress, in conformity with the provisions of section 6 of the act approved 22d of June, 1860, a copy of “regulations for the consular -courts of the United States in Japan,” decreed and issued by C. E. -De Long, the miniater of the United States in that country, in Septem- +courts of the United States in Japan,” decreed and issued by C. BE. +De Long, the minister of the United States in that country, in Septem- ber, 1870; and also the papers mentioned in the subjoined list, which, contain suggestions on the subject thereof. @@ -43,7 +43,7 @@ decrees and regulation, in the sense in which it is limited by paragraph 431 of the article before named—that is, “to acts necessary to organize and give efficiency to the courts created by the act.” -Respectfully submitted. . +Respectfully submitted. HAMILTON FISH. @@ -52,7 +52,9 @@ The PRESIDENT, List of accompanying papers. 1, Regulations for the consular courts of the United States in Japan. -2. Mr. Fish to Mr. De Long, September 10, 1870. . +2, Mr. Fish to Mr. De Long, September 10, 1870, + + diff --git a/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/jbig2/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 556f0121c8e71ec43bb5f6ccc98907ba6183f104..96c1415ab53c8f6d48704535f6f399229cdc9ddc 100644 GIT binary patch delta 26 hcmcbjcSUc*BT-&MBLj0ILsKJDLtO)l&F@8183BD)2xR~O delta 26 hcmcbjcSUc*BT-%hGb1BIGeb)gV_gID&F@8183BE;2y6fV diff --git a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 8bd5f1a7..fc7f0c29 100644 --- a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -
+

diff --git a/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/lichtenstein/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 06ef4e3c8670c8abbbdc66cc115e7f63dc363f57..2987783d198f8c4f937889ce253f0e5471459ad5 100644 GIT binary patch delta 26 hcmZ1~wp47xW=>v1BLj0ILsKIYb6o?A%?CJB83Af42W9{O delta 26 hcmZ1~wp47xW=>uMGb1BIGeb*bQ(Xh|%?CJB83Afv2WbER diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 5e7a374f..f482aaaa 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -1,64 +1,44 @@ -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/2400dpi.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/poster.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} -{"tesseract_version": "4.1.1", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.7", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "--psm", "7", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/2400dpi.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/aspect.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/ccitt.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/jbig2.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/palette.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/poster.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/skew.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000004_ocr.png", "$TMPDIR/000004_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000005_ocr.png", "$TMPDIR/000005_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_hocr", "hocr", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt", "sourcefile": "resources/multipage.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000006_ocr.png", "$TMPDIR/000006_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__osd__--psm__0__000001_rasterize_preview.jpg__stdout", "sourcefile": "resources/poster.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000001_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__osd__--psm__0__000002_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000002_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__osd__--psm__0__000003_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000003_rasterize_preview.jpg", "stdout"]} +{"tesseract_version": "4.1.1", "platform": "macOS-10.14.6-x86_64-i386-64bit", "python": "3.9.0", "argv_slug": "__-l__osd__--psm__0__000004_rasterize_preview.jpg__stdout", "sourcefile": "resources/cardinal.pdf", "args": ["-l", "osd", "--psm", "0", "$TMPDIR/000004_rasterize_preview.jpg", "stdout"]} \ No newline at end of file diff --git a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 99cfda8e..38166827 100644 --- a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 8f5f36cacbe771655aa1dfdf89c3124e359537ad..66a303254a50005811f54500d345c73b7a249bb7 100644 GIT binary patch delta 26 hcmaE<@ls>McOhOwBLj0ILsKJTb6o?A%`C#Hi~xQ<2dn@9 delta 26 hcmaE<@ls>McOhN_Gb1BIGeb)wb6o@T%`C#Hi~xRx2eAME diff --git a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin index 8d0a20d2..2c989a38 100644 --- a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin index fcabb2c4b9cd85684b32a055aaa416e1fe19143e..82d5f1418bb33cb05dd4f240e0d0970383075efa 100644 GIT binary patch delta 26 hcmX@Ccvx{mAV066k%769p{bFPrLKX+<~aUTMgVCH2LJ#7 delta 26 hcmX@Ccvx{mAV05xnURs9nW3eTp{{}X<~aUTMgVCQ2L1p5 diff --git a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin index abcbaecb..182a705a 100644 --- a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000004_ocr.png__000004_ocr_tess__pdf__txt/pdf.bin index 2f7ca9b94c53c3c7f4c15e5a55159d2c4317f58c..a5591ccec9a6ce25ccb58d2c39346ab7cd01d5bb 100644 GIT binary patch delta 26 hcmX>le@cEs1Rt-Vk%769p{bFvp{{|&=2X5^MgVGL2M+)M delta 26 hcmX>le@cEs1Rt+~nURs9nW3eTfv$o1=2X5^MgVG}2NM7Q diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin index 384e7c3c..19e890e9 100644 --- a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000005_ocr.png__000005_ocr_tess__pdf__txt/pdf.bin index e43434bc93231f0ad62682c58e899536f7599996..99d3c4eb1a7fd73a4761a69b7e63a66572ee0ac6 100644 GIT binary patch delta 26 hcmX@;e9(D=zXGqJk%769p{bFvg|30c<`{)kMgVYL2VMXG delta 26 hcmX@;e9(D=zXGp;nURs9nW3eTg|315<`{)kMgVZ72V(#L diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin index ddb2a438..3e3c4012 100644 --- a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/pdf.bin index 2e797b57553a3ed0daed6a9796578fc932e941b9..4a339f17ead4e515779e94c2b738ab6dd7fd3955 100644 GIT binary patch delta 5556 zcmZ4HFx6oL69>1Eo`HdanIVXp%*(Nu$=G7@Lq?_gxv{6$$?W94`&;`LdraWVNTw!+ zZ+pKf?Pj>g8mW3BPwb>oynN_Z4Y{D8>9-_!-u2v#RSm3>k(;#Zf8~-K`DK5Ezdt`; z|L^0E+tc6g-?!uV`u}#{UiW`Ho_8qhpWTy^e?Pt+f9(JNU)H{9`!(nP`mWr1HuC?E zultYRm#;TdQ%z2)*R!wvSt*fxHeT+3#h+iV55JAJIrX{j@9T&C-#+d*>|VC}>5o4i zx6I2A1pEGcas2b+=a(Pz*A(4os;$#)J8M-}vF3ik)B6GYJ~rMjx}Ddq-)(=Dd(F+h zU0+^o-{Bq?e)s9uFW(CHi#FQKmoe|Zy+Yo9+I^*cuX2CZ)CT{|+ENr)FW6eTrC%=h`Bsg>(PM-+Q<3u<|~eeYb4$_3ZY&JH9n$^V@~-?{*%^KIu3wecdPPQ}t>q z8O`s{$ZF6$b;TvU;Q#%fPtNyl-10fC|6%yDRT2A~xE?RQu`%Y%zL`~hA8*WQG!0)F z7wuCuyNy3h{MC=Ob>2GmCJU!LU7Yu@e&fQtrPY#sUzO&*=R1G(*n*NjS6Iu}E`7x$ zxzo0HPmYS-iLV*U#m|%!E}I#Yzv8E6aNVM8_ka@PV>>c#v*{G%U9ZksCH%S9ce?Xh zr`p4F?xx0^FWqDG^nzX4hFw1*=KEARSRd8>@u-G6=T6S-NNt-HD(F1<3f+`V#D zv;M34#akb~{M4Pzw6*N3Wa!d!>9McRUFEl^$V>QZ+u9trJxj!;d0WE6DM4G6YS#3Y z*c_eCUVh?eZAk6?o0^=upQ}H!pUicC5t?{i@Zk-ooGMM5j%qubr_Y7fyT3W_`hk0P zXPd&~IrXQCmPmOoOux@p7Ag7YlWNQSP=>h9_f~Hot*$?J*-kfI^$$zk1BJNSsXZ^bWPW#hIv2uwd^@G_97jGA7wfmF5^V24c z!my3Anwx*T-qL@%f}!%kG>=7}W(fcIvFL4{{V#o|Ia95Bco&E|i)G!fob&0s%JD6d zC5Jl?`yb|<+q$FoY`wkXi@A%r4?nWm#!}I;VjX%nPJZ%1M$|}R)|%w?FSa*u zOv%ubZa$_^X3cX|wdsi1!&atxlZO`$-8r&A z-y-(g)iv2`+6o)?Mg8diF_k--rB3+eo{DUPj0~l3ynFo?C|jK^s^7WibkeEF+x{Os z3mcreb}MtlH7uIE@2-G%@ypfrrWv}03lC12!TLaL#=F(ctJw9| zEV6KEnz6d_%te(tnM=Q9Gp0X2{EDw$_tvu9at;QkT~)VN+vJP<+9T4}$H)4}Xo`+T zg1|*}M%F@S!}H7=^I94>r`=t>QCNcKe%=F_7b1JR+F02q+f4pZ_MDaRi+Hbaon!v` zBL=5_A6EP$>2~V%OD35#zo-V=?_GAv-l8e4ZlR}rvC!+PkHHUEKP|JaKhNkWX(%8dF*_*7O?v9S z)R%fKu{GDes%%yD3M*0CGW&*;|0J&hRew>Rn1K7d-%3y2h^tOM^Hh*`V)T}ReJt&p zFMUd|te)l*E*@{GZ&UtYj#Gf?yrbtg##g&`8wAw{t-7V~^L*A8_7f)dS39-~N!;As z_d5Gv)17udo;|Wv=fe$3I6rSO*{`wCJ^pu~yxdiDJ?UK>9v>Q0^XItyI>MMgqjqkk znQ?^Dyq+~;FIE}_s$7R0e_f9v9nY^@``j=;a ziB@L!jN6vyV)R;-a^|Zo54w;xON8f8@J6?rXJYf1O6v74>UgWyJA61ddtLyK*vr?| zY!4WgU2{^BdtuWl{WFDei@kK(4A)JTj0$=}_YQvloym0Q>I7G-;~VmrKYB3y`15mb zTYSs1L-WeNR5!)93%LG%y0l8{6Vr*t7I!0$i5Gm;xz*-!`5mv}(GGYet2^Ul$j+7B zZ9Zn^S8GLc{3mycbLuHY*~`Ahdn1c& z^VC_NIjoE=;}y5m1iTBJ(jKaq`ozFybM>wz%o5d0o<-y?)CpTE`q*TW%&)JRbN5sg zpPIQk->L41af9)MPu3QlZ}y(eDfNE#c85WdX{Vvj3sL@C7O$nJaGf@amY6D6zqzWy zaI1ypcgDABKD?_PSG!Eh%#?R}?xvA3pJihl<8kk(#p^>JoLIMOhk6O~w}ckepa=3j zrcq|RjKMj7Pi$VZi}CO*1JftWH<(wOo!*oGXhP2RyWZz77FG1T*ivz6X)*JT&WbI! zX5a0*d)~KhN{wZ(aEw;Yl65!k-`=b7Bi-EBn|>a-xo1(1-C33uH@_Vd+9Z4^ z<%7KQbOy;#neeG@k%BYtMg_TK2d|t}@czryKA8leNOmEwb}Rk&Gvs%7+l#O8F?AB^ zY)q(j$=dZXQ}R2PLe5w038A7>Ul zR6LzpSSdp0s*NwYdIU6qtGG zQ6o!1lC}C{*8W``>x9xQc2xUws+1=2oabD@s{c#f<&&Sl1>Q5vuei(eo*KFOy*ssB zR^u4ozMex4U(?@7?=vvko3||`V*90{mruI@!SO_5s&WY zv7LVP%ekRLzS?Rgk5T6S%eCEBrYRQ`;_4C(rPVhOg?^?#sB~g`mOFu7+c$BBF+jgxz{kWn|QT;Uz^%#5Q zNfn>Cn5Oh98D!+2oD{_Lep}4uEpNQywo7i4$jjULt!?$1p6d}?e+I-eb4Zs9mR74V zNS!+49bUs=e)Hq8ll^}tGH+0migdeayY$tofD?)ub!n$;dZIZ09F*LzrMCaW4V7uF zpOmjWJoR}aTiE3neM|rLo_Y57p8fI0dR@!Tue1Iu<}8YE`p;(2{wllhV3=-?-}2!8 z4cCn#O@jifm(S_E(42Mkd2eJ0*C)#moeN@Ixv}|YG*k2 z*T$|bjjX6WUZrB=b>MnNWQ=;brQ@c_Te)(zPo7@uy2|!?*U>H8xIEtQRhsOJH-6#l zfAX%kR`vI*nx^S~a>o37jCJ&HK4N_o78$>)~m$_UP;_c3Y4+fiKBb ztdg^`tK{OV&%5g19@skLUr|tJaa5?t>XmY3{k=!_8MIwEW09YBVUbaiPL%z&S1J-$ zOWh=!wukH3vGXTimyX?1c|*&2@21K7N?mna(%HV)Tdx(^p4@QJ$~gO6apLL&%f){m z*d{WmdNF^q(5mfkmW9scthvu~V_k^dg62d)gNF?BZuf-ln|Vblz20O?`_7&lr`=rc z&a|90|3s~t@iYtPNdwIRyVsDb~PGyO|I&i z;x+fNw^hgVE$Y5r!d83U{<(N`-ew!W{uPa0!F`GClX{fO^*(q^T72ND=fv#8t2b};7d_sNn6*ZnO7d5<>8@Ha1d zmg2}gd6(grsi`kA-b#Jm(>(Fh9Z9H_lr8zbB_{6z87%h#iM=VYL!nPn))HWzs^J1w@=I33~rn^H_55Gw`>3bw^ zOWwnV7lAQ5rCiUpWqw%aXZLn~Mnh$o(1*3oVfBg|*ZGE@?cSadajmOEHsWpDjlEy^ zcwO#qKX&7B?~KDOu7%-#=N*`?+7*X)uNA(~uPt+RO1Zq{s_zWZX$PLYv{?IaE!SXK3QB-kz({N4cu&ZWk~>+ymqDl;CGRAk!CcVDDZx98E{?n94sOPFM|=f)>bDWBG&(y~Cn zMqHoWzJ&`ZlJkrKg zLtM%0sm9XHdmCq~Pi*>BW#+Ef%r+x_>daZY&#`Y)@kusN_o&}}(AqOD_+j|N?TqK5 z=k3UiI9t`pqP4I*>_eiws`?#YLF09Ee*M~-^PrAF=FWnKC69N_nwS5+D)z}@=Sh#; zC)K|RXr7{!V6mk7X=Bj0^%4FiC&JiXEcn^!?7n%w%pSi7TK!_rK1ZxtE4{SF$zg}s zs}-J_g$Z31xj`k5Plrrt&FdH^`#bp^zWYzxSF=B z%+d^(=iM)=eV<+A%$~bfyeqCS8oo#rnN;h#^>@tL>LdRSEwxztGG~dH!RNx;Hri<% zht@@i`Gjy>OKIKRczeg*M~%5xv|kl9?5K==&-Cv8@!K`?FDw1(m}9a}@l|VE{f2Za z?lXr7ONh>N}q~r_ziMna~{Zd2HyA9h9Z8iI4 z-Z5#Zoy{{(_pc7C_)1=!jOBg(WUr{}ig(5rO6-uEeGTlZ1g!Z$}dBSJPx;@yc2 z%)+bA&R545A2(jY;Q9^wJKD+lznLKYw6PoiPNW?R@_tK zU9eVgGPlvyrm(}#ck&MLf?WLbCZFzdR^BlGSa z3t=eRxIaYI<7)PwpFyU~>!K`omak-6JDp`&y57s38~^FXi2v|=u=&Pv#k83xVz1A8 zobAhEz44uDR($2HnAEckVca*a?CZK3y=~1p^%)c0nS3L5G$r>c)fLw^s9ce_J?Z12 z3)^n@SrkZKKEm$StLCg&?^x}&#@F^wnl^k}O76?s{J1dvooQp#RvC>M zXY)AM@Ubo2ICIJFX3f81f5l5z`AzKBQ0vTCEV(Ms>WH}cs+D`Mujt5s6roms=*0ij zXX|dS5lYu*w%>jFM4z-@dvX$6x{Gg3WA7|+J%NZhSESCnZ{rqBkXmBu zU$*_D!^M-iiALMiA9MD1oMm%Y6U#VOs!+ej_x({--(IJan{r%QSyUdk8sFW?mH1@# zE+H9_t13rt%kMh;`cH|W*P_iG!gm8TWq-NcedN34+wmVq&fU&B6I-xh@^tg*8=94s zgjty0m${wYoz0fsA8tN7&(U`J>K)O(@n-)k(p=(97O0nk&s-LrQfHxu-_yGer@twCz0Ldh>6de0y=01S!i=|$v)=JbzTZ`O zva3LB^{+-l`-_XS`-+bqYz>NCIZyh^wEXJ_PRIK%SrOf)?O*Dfn2_&R&lWVLw|16o{6b~nW3?sx#479j>Sx-rjs8sD%Ed|J-zPfjZ&D>YsC@rp@7|A~=ai86$ZJN44Q*_0JSbATDfxa#TIeY1}<+n6cJAMw^zZz=y><2fUVq)c-dq+^)74?5ATz@pi&);wK<9qVMy^nrquca;R-X&sarDQwO~sp^-hO>NQ#i43+wY7!x1?{>C+9^w zP3zq?asBtSS!=$1yJA@6JEH?|qEqmGo^#^Ndd=Ov_z<`C*sXK670Qvzt%miOs&gz}7g#ujW~i zt#(Fa%(ty?r?$j=lX_aW*@F3`#_EZUwKdlB>yxWLcz=3%$2rDq_vWt`9GlkZW~Tj2 z>-;G4{N1#xjTwEaKfZ5j7N1rBqJy`)U;7ou%*M9{%$4B>^Mohe6Hm`r`FrE5+s*uC zoEN`dx7^!$ci%0&wAJR8yJ{~n9k}VTdxym~hU_y&o1d?`r&XNGN& zNguxJ5fNrm`t8rEVA1~dI=4NRyKemO^`7o-d;bf^bj*It3ugbw=KlWb?1+*w(ffN_ zFNew<-=J3~p8dyS{pnk4WP-iQ?{G(N%`V@|b>ZRz*_$t|(oMVT(~{rY?6Y6?-RgAt zn)jt!%4e-%dT#yqq4-bj{Xa6pbZ39eyq$E%YFmB$a*;U`k^&pH-Y^Mhi2nYqqVV3U z?#~LVJ}oauxHgf|<22v4XALr6BXX_pNtl@S@Os_J)7_#xsqd{wO62Nqa%qzDEZYS> zePc1QGF|zrT>DM(wPmJTf^XD+Jl4RrIXdi|b@oRoi4`ZFEvgr*vACa><8UGW4nvvS zN!{6*71`b!j^B`pI;wQHcjL`pIaeFbB}!ag<99<`Ap5FFdEVrnmdDE@-ey?rxqGbt zskg+Wo{O)9mX=jsnBsfm+x@yfZwzbyZTb_N$A5Ibr^kH7j1<0x=ztOiV=)n%fAwc{ zw{^#tewnsoZK(bSzg7PGb~U<0UE5F#X`&XD77fmT{`=FxQuf+0CBAc&}l)$g?-cRGz+` zb*OIVnWFwr5A)|IY-(ys{!|`%Tysw8M}}n!>)n{xE^0qJu%2~w#{J;EjC=nF&)TcQ z_3r76+!cvU)@N_56nt}PN5sT67j|wmy)~Ql&S&Y{y2=`@FZ3;*3OKvxu(4?u)GI zxas+y^TDk8?ybgCCF1X#WIbCjExl)Ngx~pny&IP9So!WVi}@!>v4`n7I|~y+TixF; z-Cvw^vf|Ahi%^L>bCb0C7laD?HTw0K{C5Am^JvCGRKi68ycCG6P6-gkZR@Oc?h^1n3hsqB*3 z_pGjdd#vd7VvR$oW=crPgo%b%6+_N)#WdcOJkY{slwZ~?{kO2LR^a@U)la)4ota;5 z4rQ+qo?RWXcvsJ?TBT#I^(;!aTF=Gpe>BHben#+m{>tR5?x|~wqJJ%|Fx_ut$#?1W zyBAeUw%2_O?taTwo%W+Ez^9lo)MHZm%6nIt1t-`}eEhxZ=BmdPe}4Nt_?U8G9ph^G zwm{XVXQd2h`G$2X=x|SJn)r1oEE}yKb^lY=|f`4nWbAH@`S@hzP`?0*6dJp+12{%#!J8PR+-Bx?%Ube||TQ+wnKXM4z z)O%)kuttoe`oCQ#eD}88tUvUr;d^rVgr!^4I!!8$y+0~dx~FwZRD%G^FP)&va+;?E zmiRsoJiX1l-2P$u)uoZ17f1T?+0C+_|WG0POXqt z42leE=ZCz$!Te77%>?(5H;a!f3c7Lgjr^;V!TKk6NpC(dbNWHiRg8(pr3&1%91py; zJInAqS2aDsPkw&b-7PEHnD|!z+VR_C^)2Dbx9^>F9p5aHU*K-d#gydGxxB&s-5*9T z*NxB5>=KFj6?V`tb5Xs}nzDkj;D8p!O9C!UIS934_pwp-)Ic=fCJE1xI z<$SoW-!_~*!6MuJjky0LkDD3IOAEv<_=wEv(7Zi!&b^bpvXc%cKl88J#C{}yj*h@` zb>&GRVM`WfRUUfo@Mro{HR%*lmjnGvN^+((@6?xbNJcwMxTIyyo_S3Ab?~}Z*0M)EB`)j~<&m%KyZqy3zQ(c(T&v0^2O8?k zP2#I6n=Zw0^8B7$^+|49zc{h{v^scp@8j6Lj!b16jdmVNDq!i@nJ@T7pkDE$o)14) zSGQq9ih%X|Oe;0DYJcUJ%hLnzNb1-3Mz6ilcK&l*UE|{j(?5a-?%nnLvGL$)tNBG1 zsXH8$H(G9TYn9ZQyj_z|X!Tl!iIHe+OCm;4sOnUsAO@yW2CYpDKjY@%9)FnMn ztt%Tpul=Zg=fG*1xP6|bo0)P2|4p-yTJ&=@kGs?htyx~4WnEdDH_g^KGvBBVoIEoBb zcb&Jk>7+&p=E4ZmNzboWo@mTCDx9}I{xi2*+%@Y}m7!~#Y?Sx|pEhlsIDO+4;aB-; zHY~MW?rV-4GhPi7JRuo4GdQNYdq+*%wU|>SZgZD?k?gLST+2`yJi|>O<%YK6H+QzK zO){UPl%37)8r3t!GTq#-$Mcn#*5!FKF4G z&ZM}eS5RO!&y)oFV!gw648s&`jiQWQwGwgHHWor%XLx~JNyL^qi(UcFIg<_ecS*?sr- z=)6r@;i_rQaj5f}#*0mN6d23vCmS~{`lDm1FEAlw)$1hg+kftscu%ufYrN&iEvwL9 z{8cu~EQNLQ#IFiRHFlcv{&3F+NdWMhA@SFR#hI%+}4tXZ*-ToO1p86Le6S^l7H^Y!n$8_cE%Ea*N` zVEjexv=OUH`8lyO8h7od&E!{Y(llR;cadA#Nysc`kSJw&i1gE;D&v{1~AH6G^fAm(@hBZQaYY!gu zP-!oFdQa8ly~)ki)ht~B+;^_(e_P@bHraQcveTh)X8JA?t4y{ zhbuX~zH05gW_NW1C;!iw@OrClLQ}8Y+FGvi!cXvPNcx;(UEAwjZ-=-3z3yATH01mV zPKGx#{jM9$$?IM0wYk8kchmb-Oi9`c?|e@co+=jcN$0Y}1shYX^-8v)$qZVLrfwIw z@ty16-3Ma#cB(yDk#6DIU1!A6CKbeVr$_O~oRt!l@|sfFc}LnLnRw-GTdJ$0WVVG~ zdYHG&bV}J>)fu7{2CG>*>es%!RU9|@$;Gat%VZAE%Dc8n;YQZ=d&`e~dVfb{9yh!9 zRrN~FX$En#-poo~o7i+vQRqR>?0^87<|S?2TimjiT2`36TJWG*Z(U{6ZUOG%(-E2P zww`o0+Z%uFam|5)b7$;wcDQN8ZGZa5(vlAwz1q^5H@_&^Emj^UrJijcxvf;KUXbnL zqpVL84P5daz?6zA@$Un?$b5rU8O?y!mzR0Zt9EB$Wg?E+qEz7^!$|q+hwd~kQ zA*GIYSF^6?_p=|;E3}!hLV#`OLQlQcM3wrYigbN0>yCZ_m93=C+u#y2man@o0U~A)$tMcS}})-X;atgRrh26N^MPqwZgWUlDON7N zYvJ&1ZOzH$>s~M$-OhT-DR#w1^1-JBuEi4N;f|kY{&{hrKHC1BpV`~4hhm3IT+MPe zIjvym&ritQ_-N5F-U(l7nM$4cUM0?*_d?y{(Wyzd!e)L{nDQxS!mSt$)&w7(n!_yl z8#(&w-R<8mn;mHxyN|DSn(=mtC(T$I-ch~hCljF-w zSmJIUp&S%>mSafab=sLk*$J@8HAqHMWg@3!8aN8)W}=T^zy zczP^I*HGv~PWHN4_Fl|SIr4g&9{#>~{KjF13I0D6WLBjz-OT#g5MBD_!0p0*?IT%Q z23pU%BVV5E`*_rR9!pBvbdiS7N(o=2g3|OQ_#`Uwrlv*H<3O zYfe82Z_s*hAminkBffUM%+72HyAH^k1g`m~m0Enj{N5|o?9S|?ykD$DBX2b(*+~Zd zEcmyKL(q2S*Uh(f>vV6>e)j8zi8=Gs`ph#HJI=1Sl5lm^%a3Xw&N$XQoV30|pUq_( zpQhi(`}~ZIO%vjG-V^lJ-RgPd@dB>(-xFHaPFE8-uJ%bH=}75=($mN1MoCTZ-KOZQ z!nHn2MDt-=Wa`YB8E$g=9PX#OlNW6d&TO7zbcGdNOHq$$Np3|h3^>272I%m&Q7rx-&gAX=ZS2{Rn zuv*s~IQKT-?B4Y>Kk>J84Gh)!>RzR+S>_~PHSI`(S^=@#MHG3&fzSfzL%~Yj#0i-A6@?Z>ov}w zpCxD9IvN{Lc8#2(`OWbMc^<32 z?~#5w|NY{*-RU2W??^F}c=gj(|9eXP8#O;qf$7tD^uB5eKbSQAc-~SSK}o-ZNx|it z-&|p0+atQBNNt)0-_-fXXRT4@5!f!j^-#bvWX>b!V;pI%LsXSr0!(66@V_=at!r_L|x zvaf%0x6wMQ7O`rH;PSg-8jd-CN~Zl~oN>#%e#SZH zqA6{Qdn?1A8ib_Szq|H7qe(tO|JZNM#}=Hw>gS$TRbFEH|3N|58`VvoSGZm)Jo;x| zrhGj=H7}*Oq$o9UvV>eWuYsA7k)fHPrLnoLf%)VOa#2ji29tT@H!)e7PCg*-&1Gp| M%B8C6>hHz{0Cji6ZvX%Q diff --git a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/txt.bin b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/txt.bin index 48cfff56..bad62d40 100644 --- a/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/txt.bin +++ b/tests/cache/multipage/__-l__eng__000006_ocr.png__000006_ocr_tess__pdf__txt/txt.bin @@ -45,8 +45,8 @@ when that sufferer was put to death, already marked by the Woodman, Fate, to come down and be sawn into boards, to make a certain mov- able framework with a sack and a knife in it, ter- -tible in history. It is likely enough that in the -tough outhouses of some tillers of the heavy +rible in history. It is likely enough that in the +rough outhouses of some tillers of the heavy lands adjacent to Paris, there were sheltered from the weather that very day, rude carts, bespattered with rustic mire, snuffed about by diff --git a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index 4c20ae20..fd60c85f 100644 --- a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -

+

diff --git a/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/palette/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 80c3c42595d0f3ffd5441a067e6349709385e549..55a7fbdf369c91ee73e672e253fc779464481d13 100644 GIT binary patch delta 26 hcmZqEZPVQ#BFbxMWMFP&Xli6?s%v1eSy42V5dc~71}^{r delta 26 hcmZqEZPVQ#BFbxEW@Kb&W@u?*sB2)pSy42V5dc~g1~32s diff --git a/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/poster/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 450cbc3c9656e4e9555fb99440195118c17b17a2..33a5425605a9f1bba7d1068a6b412b4fc121aa29 100644 GIT binary patch delta 8278 zcmZ1$H9d3z2fv|(fq|ZZfr7Can4Zkfv548w$aL~UMy2{av8Vg)?kbu)fBg;ngx35Z zhXTg=dDRml8O}HL2+g@Et8%#JuWnGy8m*vPLB~0n4)0Uf?d(swx+-+_jsIUSd9Doq zXD|KA{@>&G>+AnMJ|ACyfdAc2AJo-?v}iuaE!#>5^K`wdksc>;Koh&D^~3 z|DN9q=HGpJzW;Im_T!raFU!~e+5PALt$Onh_w3{DukoFI?d4b96}qb?*b7|UcKpiz zgqNP({;GSIrt}?Lvf=364Huv5v&8QIp?*%O?W}99&5F=FljJY|SYmI&`@C;|{c-1) z5!;eqk~!UU%)`?yE_Ef0({qSD1HO_OI-s`)A%I zl>2Jzo^AH{(ARfW?6ak}XIy@NGj{HQtT}gO_E_vUiHcF$Zy9TTHQVUs#xs@1$Irj) z{kAc0-=g~O@~78#zx(kqLi|FFT({JrkMBD^*Z*H*pWXML@A0?jf{pK|$=}bIE%Ir1 z$X$cz<&R%Q{k6H4vgiEvkF#`F9(Wci#&19G`MtBcIh;C6Yc}W=%?sE*&A(#0iS66V z(|WAqJk+&+H~n)8cc1;g=7GW=v1a>uoOO%KdmIn9|2}VRE2i^i_QpTLmmZwR_-*Oa z$7e7*X=c#f`prhM*4K8d&7Jv$}1?pqpre~q=-qCU|arG8`C`JHEtYp(6jJT&pA+Tv-#@1{xaJb2`n(BZH? zCYQ1UtIwW`*VaGadD&%ZzF3y|Ztm{9*LH~Q@9#dg@$HUZ57=*bxm%x^@<-CK(a>Vc#%%)i*S^m)HO9Z)d9j)|kJN>^_ z9;0~3&9X{Pdr4;K@x&tg*5|noQ{TQi$UAX%@Uc17EUVorRBp4FG(C&|_fKLsGxtk( zjU~kbK|3ti3uiFQHRauQ?M`y!2dj-@+UGZhFv$BgfA^AcvJK0XS;bO!KS%FuTKx|` zmelWe_%Y{B`9!so8FpJlZA$z4!t98~&1I(Eh5way8GZnF*K#|+?Q?H#ljaQ{K4p+&u5MN69;}CFOo~EpJCxz^?3bWjlUJgwNjTC-jF}O;q;-G zyMEmM~j7IN5-opm_%Jf#vxoM}#|GfL=n{xs) z8q=-jy-l}vWVmns{HN!f2m4qzxT$l*G%j6u!Q<5Y)QrU?OwNfl)A@nJ?RHUC)v!GIy%^{g`FX>irk8#Z`IvPI%fIGQ)v8&+YE}ymPs&VgLN13(Q#mN{Bt5 zH)k^EQ)%XxPAW$pxj6kcJkvI}#OodVfe$HHKg_Y8C;P3*qvMmm&5O%n_P2x<CG-9(^}EoaOpG$&UBThEum*dCzuX(X7i+7u=6C zyog~)a=YK<*vn_3x0BsEbKa5(x{Q7L)~i~k9JjhWhq?UK6iLm$ofj|mdnj4-K480) zwocn+gY(|V$2W{q7yR6*&{xGa^K5TG$z8+En-A}swv(^*bRY9cW9<^{NuI0fJHvcl zsoa_vUo=bFdHXgEdl|L~pZaXCAA0qqJV?#_`HjAXJ9R|3lJn4vDIujbYjlV#~;ti zf8Cw8@%8J)bt=`zu4QQWNAY!Lp8cxZdCR8W(noLU&HHx0u2s#{wYT^+MJeLJ^t(n8 zhfZ=^+LzV5|8Q5N>(|L;F;-aYKpJB_H@jSk1dFY=z#R~h68O_++b|B%8 zYQsgd{)yK$yuK~zU$gQ_;i_jT+tS*5QcD7wCM=KSj*oJ@mNn&d4#P!T`Hb@1v)oIB zuZuln)~pW@`21^{tsb@1Iv%*rniv2dnhm zv{oIDlvLj0yupd#h_1vox5({AE|1G^bn7g+eXda9e$O&V&K6DsflpIAZFG+igGgepSgnc*$Uc^X#Xroxf@}7qL#4U-{zr%eVX1>^)r+e(K^bBjtI)HQ(x< zy@<+q<)fcoew|;xGL!#-qmbEEe^Y}3u`3T3#2$OX^7+#fG3Olx3w0+4wA?s%YQlrn zF7K{pM{GTIqcLQ)`3*z2db6#mGnV~)=8@#%8fD~j`I)q?nda3wOMd?L-?~6nW8Zgy zsfH#hwT3*xnOn~Bw!JyAy|?f9Z1rDr&uOScFQ3qKz>bTj#<#j!YEGxyGY6Rh$y>W} z+t0er5?y)y-?jP2r4v=GwOJQfYw;8x{v|x?!^46SCj5avsrKKsZ{ZH%5;RCMA z7cE{|UUcA>P#yotIV`SmuBE@YgI1oAxZPaz(=GUC@yoaH zC?rqp{$`?DE){UxnCsluZo*t-a~AX zn_{l`Jlpp?avIM`g?gD4Jf~OvVL8&uE$bjNy>;V+vUvxCd_HMBS*2m5Z|ryNiM7kM zV;_!A<#juJZEtUHw%9^F*PRm>XP-T@;ft$dx_wEFGusj!5l88l=bRn{t7xCxs(9a8 za5h^6)59;jI8SOWyu5fR--GMP#!~?b_1BaSE;Y^Y zInwVcaGCF0Z%tpH(!3L`8#TV1d7;0i_9oX!mlG2fJ$K1kJ=?U} zX6A>Biocf@AG2uR&6RfW%nJq$W9K7w7ewbCNa$jld_q9rN8x0xvRzRd=dL#s>(FE0 ztg?1XdX?5+H=Bl_7aNSH9^Fzi?R5Rc`Y+w0PYrA3_uQLQ@2F%P5Mj1w!}-2hHy^Ih z6jO?trXo7=Z#6gH?c2?5_T6#2ox9HKo-|SUbX6>NR=7ja#^suQZ5%WBPX%PeZ_4Qj z3Rtsidl{el&Xfx_eDmh%ANMkPxUDtiQcj|x_UacKEk7!;UwdfCXUXto`=iRFslMm8 zeqLVhH7CiW{XN@a&w!R zZ{V(ez}0eJpp{KKVM60kHHLc@4~{Dv^-hjB+t3h{bSTv0Kw4wyqd(S{?Go8+SsVXk z9jfDG-Q8(gR5G94D`?w7kMMg@+;1!wF*n4mO0@}F`|<>zfR6B;*|lA4XYO74=vi|9 zN#BCaAG<;&bj?^!)h~2>RoBJ-(|6irXVZkMhmK0ISgZ}cwr{IQYW1HzFWDr2a35ga zoBBa#$|B3Gsaf;GqB$*Q%1Rt?_1l>8vFPT}u7D@~78`rsZkq1bTOhvl@R@554yztI zy@ydN#J{UeS;XPw{lZy|{)UE)Jx5kno3h6Q+W&bOaq&X=J3XhBynD~qOD@{KZTai$iy_DT!7Yp-F-lQpWS~>EkDnGw#w!6o4&Y@-hYnrFJNhej^ zaFu`FqAhy+$AM5bR=!yqH+f&5eD&Tt-tVkuE?i))bo+d#>3z$&6!yoH7Pn^nnef#_ zNiy@9+?8J~s_R}RUJMr5er-$f4W*iTS2I^WRZ%v-aDG{s zjQ6o5;VH}+nseS?37g!jCl`<*c$V+Kn?$_GT>Ygp64Pr{|EDjA+2AK0 zv1o7FvMT|3PHCbBh0$Ii3vWzZZMGwQxvG*{u~E#c+dEVr?%1JRc=-E7>w8B_xmGW! znX^@<v>p&XU z+KA0J`IjGF^Im<18@HE{`vWe<15-5l9Ovvk1LjW)NkD_^zQ3fnU=n9Q$wd@aUbHlk?`oiC4bHx_C^Vs6Kk&7PVvx7 zx}lXVW#_zE>wT9-^t6VIgQnMeb&WZ_Hf(v$(cf9ERPMU?>Ek)?XD<<4W8@lqyV9*H zUTWfh-6=~iq#5mdF0gosF2nzN&HZW1Lqyk?&gP9ynYesc@0<9!FBAIfv+wrrnBDz; zV(Z=0zYN*$Ue42T{BEpy;O`<|J#m}Fvc~@>b0cRw>UQE^^4@m+$2vaViluzPJ7-Qj zk(M}f$E8NYk`7*tjFq3Kv^|zTF;#x4skxA2pZy!Lw?4*eUM^9#OuLwD{@ipGYfEgW zEX&S+Pn86pe^&p?$o)YfT43wH)4KI%sxC#B-qXLf&?PvMh5wbSQBw_nrSqP5$793N z7o<2?AF+8Douca%=PQ(Vbf&oK`qMAt#jU%A9~{czB zIOY6Txr<-AqGlgEX};oqN6gLV6PXt1bA4Cp{bi|>cTQ1P|NNV@Pc0{MKd>;CZVCVL z<(yWUPJ_+Gi|Yb^73OT2`!2^)&;5*f!~xL>{h>m9_21OFQyWYQ%hw-{;;p&v{w;5c z?U8K;ABt4C=IV(YpI#tQ>d$24kXZDCKPUd`9?{QR;!d-wyjkTNW@52EkNNq;Xo(Ub z)sM%ozIETCq8KipA9Cop%)y&`g4M+;r|%45oUp%WTIE?eZi$Y&j!R3F#SHf7b1i+a zAj&bJE3${}`IP!^(XpzjZjQ$LBltc&KfwAl(5=$KqwihU-oW+JGm4&;Jze49dw+V0 z-|N|t&nEJfG_Xh{UQr18rWwIpb!z%G6^}KuA8wtqc}D#_C+minit~>pK+#D*v)QP7YYWvEGGECAMl=;shVwKSf8~UCZjV zCR&{3HjIDlvM&1GmF$_16tyPS>nz>K_~zc#{>@kSlpbKtXAF3B;a$5uujd-}XX~aU zys@~!7WFnF_@9o+vU_4~2cnw~o`HAeq^qr35nq3i_aU-ipEJ#{K?YB>As+5gPR=3KVN@}K(} z`__r5$)4EMuw{oO*Lt}8BfRk{I}KX9wb zeVcig_l}}xeNwbn=JYht$lH5j88WV^N*?V>=%3cm>sxXx|1?Y4^ON_wO76USe14PA zq}OS$)y^bk1?9Ddzw3_LZ|&<NB+L?O- z&q-!x7wz`xXX38KX6|FlWKnqaYigqClcU>$gF2EAtIE!Z-udl!&W|lBQSJf(cT+Dd z_^lj&ZMQyGNV?0%$A_l0Y!O>Gzv4E>l=pjX9}j$^6jW&~qBXUSUHscQ=7e~SUv-PW zv_A?xlw|TK?Y5Asn$*O}|LbMqHtzkC;=b#71CQJl#qxubzbwyrHevBoZO+Nw@ptvl z-aYiQ?P2ZiM^2}#Yp)#&ythkHMLF<%<~;rQh12`Xbb02=sKw` zP|#}eu&c4iR=zmhb)q%a^7n-|Vb>F0h^|bX znbj7$@X#rf57#Q!6nZ2#Ey zi|l4T&=OrPq*Qc@Nung8$R*or0Tj1dv^gPFgp6{ukb*xjza@8RYJHiHZ8}I=jg4(Xmq}t)}caaDCoO zVgLGrVrje&W$M;y7MF>Y2XUenZ*( zZo~9v@nQv$_ZQtY*2rJK+SXD=Oq6A7!S6WHdwrM9_Bc-4?8R(t5Wmy!V1en%@a1RD zE;4%@@i6+%hL}%FzTM$3t!E6lA})Djx3J!KUKzQrc(t~kC$FwQ`t&RBc+@H}^I{Vwh^S$!+rT+HSJNC=){tv+s!o&z|$zznD!Y>^W)2Hz8F2<^TN)RU^Gw zwni8kILzP09~wE+YbFm{^TDWhp<7J8l-;a9)P9niZTl(q?Kj&06eQ`ygx>CG6yxC9 zeRgZngIG_a?$fnbHxe1fUUxaKNQ#y7YefZOPPUNXgmwh@j zuT65#VX4oxXSeDUNc1sI{1zg;crm-Xt3dZ{7oXY2&QCk?)%Kmx zp6Sl_DN!Y&j#+#9;{UbVoA#XDqQU;NZSublhb1=$&9TiF%e%FJd&SzAiKTL1UV8=| z)&BocD=Nn1oQwPx_n(XW(ZH?K^LT-e``&wcBwS@FA1_1FIXvT|bOTQcnk-w~UK zhtIUWlM8K~ak_lU`kf-2{_!$wA76df^_kRS$a}vv>ir6)@~VS=Q*7HM7V3KxL=~JHqKG7^K$s=&f>h~ z>%IyWGd-!_(|$-l+4HSlN&3gXUz2V(sM)ej4_qRXeK6)!Ga7^EKsU@D{Vz2_>dIXFjZc`{M7KQ-2qoU1@k!>DaT9I|b*i9eh-uxXR%0 zgHt>AipDs4JQTa}QU6E&1Fp{p)#~T6czy2q@FDQi4`%zPuz$|C z-KXDwoYUxO)|FVCRQ!^F{t!bV5_U`fQo$nb~4>c~``u<0v zhv3QB`iml-4IXQpCrwT8S*4@o{Hak%Y6j~=QNC++vnTr=$}2ziB4huNV?Xbyz29S> z+18?Mn{eYrf&QEIEqwpoD?0p^I(>dw|1aZ+S2b_=&V$ZoZUJ16&8$nK96S@kK2@lH z71N*Mc=guK1LjjUn0%Z3#ObN}rC)NNUaxuNdz>@)caT`U^?`EH+jFKb-TlY6<)s== zQTJJqB?mc18=G?&h7tf*Y#~}*x#h-%qi@~$1K;)algA@cT}e0FZDg$*LDlNmz&`~XHH`A zH#sJuNt(yctT$?#m{8xlFZciBP1_dA*GrkT-BygN64mUT!DRTQLgUy*Ry&(aEr)en zI!OX2EBfkRUuH0w^)us`P-LR$l~AGjn>L(jI^H2F*IBwOPAM}SWSqmyyEox-z`8Xn z=FH1q66-%-Q{|_&$u>EKRQs@paYt50l%5muGw!Q73sy$fj_&FkKRbNlpo-H>k_UlKk+m!n%xX`&{?W3@d$CcOr5B?cj zUvqEn>diBjcrBOO>!Ecv;o9}A4L2vKhi&=XSUG2=si5gq(I@lbFRuOd{lj6UBHvzx zC4O^#H?jw?{kyr#Xw7-cqQl?j9eUe#a>1pR`&(UKPi}^j$NE0#ooKQAnEeazIGv_=862$ZbA!{ z?l*q_rlQG_ZFRT)#Nn`Zn-7~GmM--z>dCc{UF-aIAGh8`OP02cjq5Fbh3GLJ-ZxFV z>F><;&Cf4>ojdKfri!7@0()l9*>TCdyVqWR&c^amCx$Py(e+~Di=01?jzIKNYQ6nwpnVTryczyN}n<$iUpl v(A3!2NY}t(@&WA#76TIlqscNl8(0iX%*-cW(DCLrFf}*lQdM>JcjE#8MU`1-y4 z|1aC~U(NpI@$>QZ`*uN9Po~t{)%|^ce1H9)%|Z7xMTP!WEbaLzXHURr%lqo!HEeT5CTa?t8ezVfyBW)74q-&i|nvlU`Gkk$$^$)-u(s&G#O6*S)B> zos(U^Znaa5#pK_ocej6h{WN~>n?0Su)5H?*9C}^KetErC_@6h;<;z$8`j{OZ_4=^# zWtSUM7DvR)e)Y&L>dy0Q>CZ9}dzb(27Q5#1_e1ikxwR2n+XC~XbpLGXQMJ=^ zl3#j@oH`p4uXp?Y@vkR;-Be4xd%X)|?ojUdxA4WnT!u!q!kHPh8S`I$ z3cDP)??K$9Pd!&p*Svh26T`e>@xPn(b`KXVIeF%{5Bv8MiM&kmHud%^cD3JrpC09V zc)nt9-@BQ=rd-IYXUGqV+qf?%e#fs}WoJLnxc_U{{G54TZ{2<(EV3sp)_~t!vi_T; zvHgtomrqNzKHa5rhE@0b)Txu#U+%2&)H-)=b-Q)%mL8+6?MEI;?)x?W%Pv9L$$fcm z*$WqE&Yw8j=Z(41{5Nh-)Q;Whe5&Wa#{9r_zGs5R-={}AGjD#mareIFdiKuc<{p(d zBNzQhk-2^8>!Xdg?NS34Y25fO`tvKtyzCiYKkcitZTl8}CVR)}GpNB6n}jjJ5p-Kkf_@(LNTsB}0;l z*?6m=aQ!#wnTPL4PnFNK6790RS;%f!zwFA_aPw^K^@5h0B0oGg`+V28u8b{5BMSaT@g?rL(J|$^!l`YZ zd5ih$`;W-4y*_7c?Ud{XaEYo8bmtft(dLe$LL{xUWy)WA^Xh+Be@V`xM$lkDJyvZJ6k+|CZl@Rd>b~DsrpI}_-n!?Eh#EMGLo?dR2&hR*RWC`}*~p zn*3S^mo#a|Ojh2p=x+VOOwLoTEBo|Rw((7Kb7Nj@Z&Y@*yC!yzPoHPYKbeVJ|hk&FHg6I+ifDZi8MYbdab%rrQ( z#_;$8hiN9oIy#;m8>aL{3V)EuXZ1V!{RO|_HWm@bioNsw&#&J->F>Igc6;5MFW4N= z-sBX0KI!eW{c0P&7`m!2&(J72m!Q&6Ke4`;J9o$ZNsH%K3Nse*Ocj1zzWL!h2X}{0 zjDcab5>;Csh`XpJoS3(9?N{s9P77<3xQepP{{H)BKgsH5=z_f3DLr$$S$=#XW zcevciX2z8x-&vQxN_iE|%+V#c&iAnV>xT<%<{VnYbnT7xuDBzYOgx&^nA42zc&a8c z<;2&!9m!Cr=5VNrne&7?!1Ks&khv5xs$oGt&BTOA$iHuT?*f< zEhkE`v#p%8HG-eLpzL%t&%-@4+anhUo42ie9y@X88#$rf$x7cfn{FFy+hJ?2)9)s> z-f;d~wvsP)D;G>TBF{9VHo@wj!K`y8cLTPpPf)m!S3iZzEoRTgMmzuhgXLXQ!X1~a z=QglVndTjEZcduoL>8l0FIF7++c-%g=u(T|U6Z2q4;$@&Ip0mZdOmvZTZz|AEB##G zZ3}p`I-&H*Z5<|au`|N!*8Zz{EWTyQ3%#%J?=p*=+V$qZ{jUtEo{1*!M9g3Rlisti zlI6*HmE#_DJO8WQtY22%)crGL4nL1nnb}g&&g%clOCC9v>|txw`ptAPp-I|Hs)GNh zK$_B?ZSP`*=FCccIW6{TjQ`g=XTR}Gy4Wu#;jnY@)(

DiZ8Y6pJyYggxYaBxm=y zfa&F%yG(oU3G1-TOggxJ=I##V<65~-&hL}|H1ATNk!PKI>}Dr(VVkG*{43YI_LF?M zQZhevgF)AZysoc#YNh4eXF8Q`1j7O=XbspcbS9cG2T-p36p?=0i1L+T^cLvpe zExJE5`L$%VxZ?KRDtp++$5VXyq)VUcqdW8G8LugJFXs8je~7wwdi* zwdY~&2UfdFXDa2oh%#KyJ+ivQ_vR_{wy1m%9diF?l!3jllIgSk z>r|wFReR4o)$8jz9CM)0dov}1N`Kup)X}Y;Ws|#C6;G2(gBfo6^;a4%`TdFBv`1Q$uX8Zcz zsV`)YIQ&mjFzw{6ldPBC-*Qee?@rsh$&&GwM&1q6v~RoE1p?e%>h&KS+45-9#=S?B z8yHwC+K+T)3QIS3&NVT6Y?Cpmwxr?8L$$=4mln+YvF>hrO!&-yaY4(j1gZP3oc#F4 z(+JKThxAeng+GMudw$dSz22Sll=mSa5_LyPy)qf=!<1fk`+V5*`h3}OQ<3Sj@-2RB z@ha=TlH{((>D9!-m*QmXCF!2u#bc(MC$>jS<>|vof~PnmCR_hmpY)V-HIFmT@n5I6 zeSLAr^24cUDZAK=iOtg!Zp(2m=)9K@<8S%w!}Cnu6?|Fe60|SOl`JqiJR@F(SeL-}ZX{^N7XxIHpY>|z1A=^_zZfL#U^};DE@W2L{)sMQ@i!Cv$ zn|D1&>8*z2cG-`-*K)cRH#^<(u$a4tt@wWsr}>^mW-lggoUmEug5^xFiE~y}Z@=F5 ztNS+V~FJ9A3)t>dCSmuGKp4g}CNaTsRUFa=nU0+2H+&Z5+y%zAdYHHapvB-c%Qn zCEvC?NwJDX%z1E%IplxFHvjcr^EU`zs_|q$cJSUMmj~6HpEJ^a_{UEYw^3gicw^p) z_ev=eG2si|YSnbVbVxAb_|cjDS!wAQg4 zjha=^(HQb~SHJ?E53AcgSV~ANxtjQmNx&wC}w{E=jYrU?}A?N**KImHMxr%JrS7N;>!?GfR zn>#H1sk(yRoEJIrt9f2DX-LiscNwq8*JGdX zE*mFhvCD@OD~$x>3Kgb)`udW$;7j7pwX@dl_FjJLQf$@lgVBF1TJ()yJa7I`F`G|Z zEKW-`BXD_o{h_tGyDv%oiC-ywXixvAI|eSwzH*#t*Sp>-9lR-7u%iCbn#wO{Cr#D+ zCK~mP{o7QLVjGSSfjtIUj2k=k{^c#3ESQ%Xbll57{Y;eD)30|o%2=+OFfCliC$;L} z%f)=ZrT9pqmUy;jUb((}^YGAE;VS$h70CZqeG5VepT9n<`foHHepcK-P5xw$9Q zC+nKd`po~EY!~c!{>HaY?4(-MiU*ZJqEBBXbgnXKyYV==Z>w!ri`LS!hxykYPBe6! z=euFLsE+#!8MfnID-;%p%zhUX(>;0BPba}IZ-1T>JtM$*Y^t89^EACkEr$EQo9aWR zx-H!ovQs2vXWq#`%Z=`xS^v||DGWa}ILG*jq`>V&d zvMlYG^!?JU_v|_u9U6c2Zdg9j*O_uJW7W0Pb?Sz(=j;1!d=#ae?3!FX+AXEugmChsJoc+TIY1HwunSP)s&!TN4c5Sdvi__T47e6eSq`p>BWay z(r)B^5A=F(AIAQ={-h?C!NFUBsok2-POz*^wlVlNecG)Fmx3J8wKmU(7K6y}DXKSgqE3A})%`d$MvFnCepdDCxpyD^_wsxD zZqe^Ry@OUaOc!zo{`74NvCsZ%`__0}pIBnZl!NDVN*}66PO{V7xX|TpleAW1YR29T z+fu3?6+MjaS|Qx=J}Ul;`T^Go7mls^Z@BVV$D%uCM!sp&&-IA$-PJS*x%*~inc?Ya zmrOMyPV&l@UHz_g^gz47cg~vnXVYAMUD(OEJ~l<=klex0$iMDZUtb4A z?OyrYd&!nuQD51G=00yXI6Ih1&)H=EX8OdCI}0UT`>+0TTQ;ftLp!^LntsymY7MJ( zCy)5fWPECPI{8EOsh$(&+cQ>08m3R1>z%E^^J(X*Id8-E-O;Z!YfPV)>83S#j;i&s zZH1bDl{pJTcZao^@7;7xai@ZglKC#XTjm|=4zlH5o^Ls2-_}2EH}~%i|8Sx;=hND+ zyX*zFnr`CT8n(nId-mk8j|)<_WEd*=p?!yI%7;sHGb^RGT~)i3Gj9qvWB*c*bKevEb(Zd1KUZwOjqVfaQ!5kv>#G=L zt$qDXF31(|nAfRI67RXVqHc}>=cgI(wa#z_@i=H+6X2T5GSzm6%2Dr9fyZZTI*)yy zoN;bz+X)fJP4Xvd3ZGB$x_mhH?XBRICR#$86U$GF%)9jb$)(aC?k_&ypTMz#xqQRa z%@T&`-%Sz~{w7+?U6*1Ly;998LQ;)Y?m1h%!^t?6T@kX&I{rQRve?#hrqv_$JJl)2 z-Ii3CA3Le@`b9?6^i}7+h};XjF2=lk%dq91ZTB0KyhCliz1c=T zub;45aBJR!x18n-mR<)eUbU`z-Bou(+vhr4#MIMSCgK+x{C4eNE8F)x^r@W3%%JZ_ z-aISk`o6uM)%{OBAg=cp6foeO6? z-+8Zz;ZqlU_t8&v%?V9IzWvdgbuUd7*zHu-ov`k5k=vIqZ)5A<2bk!}MO&Vm*uU-1 zMBeMY#@iC@#m*VmYR;3>+1j{Wxlnk7b(e2PYt+L)IY|=VSBc~;P&lkiA+`f{TpSsAdQN!3y?nO%}&vkXaeut{) z&cIXVysP&+OK#csYf*%K{RhqHt1C{w*!Cm7`bgHVr!`g|eolX|RXiffxa8EEE4?@F z*Ia%YqP1&o#x<_A{kq(Rl_G1`oZfzkX~os6p~br;rn@)CwS1Aa)iPQB-GzNZFvrOg zsrx#=${w`a{aEw#gInI+`^5GHL}?%LJM;IB+SPmC*&ME@u6s25(A#rC1!vh$)c=1~ z`EHMvHQz7iUv;tEOl{^>_l>G137Stikg)W)M2PgIIX5^}_XgEi-`5sl1fOqHL ze<7zE9xl&N`1a`Jj-GtW7x{|cvo7rJwXjc7-thQ9)nS+H@=v!~GlG=Uj$E6j?mYLR zPvW+>iskZ~>N+p(yz;>By8MlIvD@;m6rFll&ue3Fc3i}=uEo$ z=CUhxoBlU1oAPPqF#*#pt4t-W4CKz~h`js0;r>ywv6S9{HFJf*us ztY3ZCwI|{l=KY8I`%4~fUfVb;Jt_bB#q1AJ54KNy9{V+sdwsc0PS4kCp90EqLsozE zf5GAZXwt^CD3eS1#wR`)xSYEE%m4S68~e9TKl3O=b5`W}#RmWDx!Bj7iLT0;*MBD? z`}xm~KmQJXoD(JTvvBIIV&C+*Ik$K7)|D&oOPsUvin2T}H}}u1YNl-W-Xx`n;5iJl zoHrcrcVrcMIX6}Gm*ec~{MzTw2w96srpl>*&G?bZs+U?Y{U$H-My@>^{;%Wno^Jnk z@Ne$_cQ5(Qm%7x+eCewDl2u=PfOCS+JDcf=0+ZW+y>zgSP^lJ6(|cw5R9K)*`Vr@b zqMzQg_*g8DdCQ#8JJ!~dP`N&_eRraZOIRWExs{(TRUf`H@tgkX>Wp=>RlX{4x%WCv zy25H0aeueYxjV^8fuFB6zTR`))45@X_Z>%J--CNNp2Y`VD&OJ1ror<~_QNIp_0#`` zwDvf^D_)iN)HB9sQfQwISCeAZ?1Imor*D|)8+1>6D$(hgB(>CDbiGH|V!`KP;WJmh z3R=EdUV(qQ{sXh;I~%feg1Qc?7cSf2y({eBw*s5;zdSAf?}jd};pb1P+mtoowrFXC z(1SOj?Ylor-OiV)$+_ad%Kf$H5^vs*JIz;Lb=s@5uXE>bQNzN|DVmDVI)fxV&h67SEg4vD;rXH^zNo{C-l{m+{*Ir{+(_r4FJIWiqM- z{BiB_cSWaN*(&pB;n62KZzn4>mH(XhsMk01^+&!p+ZOx0pZ@lm_}+t#CfgG$LRTqy!*f2Db>9;zPznL`d4+jQ`VnTQ4=q2^%xdm?>Uk#lgd+P--|mS8DyEZHM6T>oI`d*iL(aa7>K*Ry_pFRl`0cn{Qg{B3 zzcO(PRg*6r5ft=u|NS_V zAD-{ZpMTBvMDQY+Px8OnHZ*JenN+%QjU}suV&{f+``@)({?2z*b?W0}?#E~AZ#vG) zU$p(w(^`w8VQrgx6(pa}`_Fee>c!GKp@$CmMfI7-g`KqtxBa4d{|+BF@1l=CzCAQu zq?)PoyTM@kRQ7Gzbq#cbDg)U9nhTAbVhE`;lhXfrM=-C&WDztytnXv z{=UMGqNlz;`eOB{ck{I6`l{V-NmC5E78|zJ$0wg--IrNeUC3kY$NT!%CB?PbGGD4Y znWK*V>3>qWc=AV`5YGqvMhP7^F8o*Ol|2%uyPi+4QY+`d-&8SyUBB%a7#RNlKO5^k z?TMPeo!@%jgqc*fe3?CKXYc#{rV4g~oE#zZTFozSez7Vna6*ckn@ir6nP=wOziBe5 z&e40Y`*{5c?dMXe2d*5Ins2JHzvc9%2Zy&g_GMiBc5O4q0qbh<67Q|$)8?FTt)KI^ zr{H+1rBScc^VGAGz1`n$y35;gY9>qU4HM0y(>t2e{c9P0-aSyd*_gE3GVI@fk2kKy zKfW~0Yzbc4wod$t%!HepV!ox%GMRigSv1*Fp*v$;8k zdXuWzY_q$|)>?4;t&oTkd95MH=kMsZ^v$twnaNF|ypOdzYtwz)WqKd||N8OR-OZ{~ zKc`N8`{5OH+xmxUrt2=H=$wpQbmVGT_Zj8yeLlThi<(LY0t*$P` zukXE*$m?s?Emv(kx^l^p`u3CgpG1Y19I1&fW{q0&gKf+1)Tx$_{>!-jncTH)&4ljP ze;z8&TEAk=1$nWO-K`($a$g+Wo;4xl#f0#`x9>g)T;1ytpH_N@pON=}y7o!Gr~4-L z&KLLX_}$d^UjF^H$)Y>FmkA!3%)VPN@YgM$*d^DRRtnpPFo?W1l}Nl@j?6!qLU)1(@-G!@@hr}|&s>v<>l z=OU(`S0DNo-D%r6C9fn`WyXTOv*`yK+dhBzF?p6$`eIvwn!RD?e*Iwex}|b2QS>(7 zs}T7~g$BPi_vKo5Mj3^?){Vb>{G$I!wfW zyCr4obh!SqW3m3a0=@-@6RL7cZ|iHn*w3+SA@6bq&-zUlPKu;j&0kP3=h&6GC3(}Y zSUguLy=|4NVN|%WKJv}G<>DgU4cjD|;vV}v@sSC8IO{_9PKSS+P3+Pt|0D!d-*Q}J zv@OB$*d(W%Cr{rvR34aeqFK2{?!;Z|=W%x*t9IPldH9A%L%qt+$kL3OVOEnfeRPZ3y~z^ZBT@1G&Zt(vt8%yh639=!J6OPD)wnLh%Y*q-h4(}K_Z!iWYYKAuC%q(z>XclWLTIZzV9mf}L?r`)R-@Ejz)v`~@BhRGI+Ie@2 zr@n+Hr}HN6^PCYMbIa7`>^)KZc4hmAj%6=j@0mQczP`nIqW)Swz2!eGsg=F~OP28;*xx{eAtQnUp%ip&8NKhWS4^ zH6)W`3J|lA9herNhg`OSn4fVD%MmkEIT-RZ@ujKuYvXZSb+BQp8KH?Db z=9>Lux$(6Bt3TIA*}Y9PVwk&r)3TcnmR$b%v|G`~0%LlHz7{^Xqc%sT{eq?^x z{kQ+#hWU5xNv%4du=QEZyh=w)+5T&xldE>f>$A+4y*Bfx_Gh2vTH%^Ir#tv}&-j}^ ze-p#W2_0X2mD^>CCqDmiY3GyDw?F08To*(b#!fo_)@5Z?-1kK>Jv{Y`STg4DEz6p` zfGfmNIoTsqX!n-#@~&Np%11u(%m1jI@zdB@ZEab3{R7|cji=}KwY_wB_C#fV(POi< zTp|a9ihoxNd=O)x?X?xsQzu?wTA0jyzP58Oj5Xg@&JR1#S8b=xjgN{5(@?S zryPG_ULViD;r@SC+ik-3lP79S<~1-gGBPwXGB7mLH87uSs2RaxXkcnSxmR-oi=m;B V(PT$0Z*D^)V -

+
diff --git a/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__--psm__7__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 979d31171a64014eab3ac1aaf63f864f4f72288a..31627deb2cd877f56eb8490bd58feb4de2163324 100644 GIT binary patch delta 26 hcmaDS`c8C14JWUmk%769p{bFvv95u|=1$I3MgVfJ2ZI0r delta 26 hcmaDS`c8C14JWUGnURs9nW3erxvqiv=1$I3MgVg-2af;% diff --git a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin index f5eed6db..17df1412 100644 --- a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin +++ b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_hocr__hocr__txt/hocr.bin @@ -9,7 +9,7 @@ -
+

diff --git a/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/skew/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin index 11a1c3114fabfddd0850dc781980fc90d00b3409..b143cc9bc98c28128f9f9292443391ccce24ce87 100644 GIT binary patch delta 26 hcmX?*d?0y)uK};2k%769p{bFvnXZAw<|u Date: Wed, 9 Dec 2020 10:15:15 -0800 Subject: [PATCH 800/880] Stricter parameter checking for many public functions --- src/ocrmypdf/_exec/ghostscript.py | 1 + src/ocrmypdf/_exec/tesseract.py | 1 + src/ocrmypdf/_exec/unpaper.py | 5 +++-- src/ocrmypdf/_pipeline.py | 11 ++++++++--- src/ocrmypdf/api.py | 1 + src/ocrmypdf/hocrtransform.py | 11 ++++++----- tests/test_hocrtransform.py | 6 ++++-- tests/test_optimize.py | 2 +- 8 files changed, 25 insertions(+), 13 deletions(-) diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 8f17719a..5c357f1b 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -166,6 +166,7 @@ class GhostscriptFollower: def generate_pdfa( pdf_pages, output_file: os.PathLike, + *, compression: str, pdf_version: str = '1.5', pdfa_part: str = '2', diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 46e810c7..7f7c06a4 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -221,6 +221,7 @@ def _generate_null_hocr(output_hocr, output_text, image): def generate_hocr( + *, input_file: Path, output_hocr: Path, output_text: Path, diff --git a/src/ocrmypdf/_exec/unpaper.py b/src/ocrmypdf/_exec/unpaper.py index 5798ed55..5226326d 100644 --- a/src/ocrmypdf/_exec/unpaper.py +++ b/src/ocrmypdf/_exec/unpaper.py @@ -69,7 +69,7 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]: def run( - input_file: Path, output_file: Path, dpi: DecFloat, mode_args: List[str] + input_file: Path, output_file: Path, *, dpi: DecFloat, mode_args: List[str] ) -> None: args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args @@ -114,6 +114,7 @@ def validate_custom_args(args: str) -> List[str]: def clean( input_file: Path, output_file: Path, + *, dpi: DecFloat, unpaper_args: Optional[List[str]] = None, ): @@ -130,4 +131,4 @@ def clean( ] if not unpaper_args: unpaper_args = default_args - run(input_file, output_file, dpi, unpaper_args) + run(input_file, output_file, dpi=dpi, mode_args=unpaper_args) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6de1f2e9..3a7c7123 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -480,7 +480,12 @@ def preprocess_deskew(input_file: Path, page_context: PageContext): def preprocess_clean(input_file: Path, page_context: PageContext): output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args) + unpaper.clean( + input_file, + output_file, + dpi=dpi.x, + unpaper_args=page_context.options.unpaper_args, + ) return output_file @@ -616,9 +621,9 @@ def render_hocr_page(hocr: Path, page_context: PageContext): dpi = get_page_square_dpi(page_context.pageinfo, options) debug_mode = options.pdf_renderer == 'hocrdebug' - hocrtransform = HocrTransform(hocr, dpi.x) # square + hocrtransform = HocrTransform(hocr_filename=hocr, dpi=dpi.x) # square hocrtransform.to_pdf( - output_file, + out_filename=output_file, image_filename=None, show_bounding_boxes=False if not debug_mode else True, invisible_text=True if not debug_mode else False, diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 9a37cad6..06f630bb 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -45,6 +45,7 @@ class Verbosity(IntEnum): def configure_logging( verbosity: Verbosity, + *, progress_bar_friendly: bool = True, manage_root_logger: bool = False, ): diff --git a/src/ocrmypdf/hocrtransform.py b/src/ocrmypdf/hocrtransform.py index ea75f51e..6c64bcff 100755 --- a/src/ocrmypdf/hocrtransform.py +++ b/src/ocrmypdf/hocrtransform.py @@ -77,7 +77,7 @@ class HocrTransform: {'ff': 'ff', 'ffi': 'f‌f‌i', 'ffl': 'f‌f‌l', 'fi': 'fi', 'fl': 'fl'} ) - def __init__(self, hocr_filename: Union[str, Path], dpi: float): + def __init__(self, *, hocr_filename: Union[str, Path], dpi: float): self.dpi = dpi self.hocr = ElementTree.parse(os.fspath(hocr_filename)) @@ -182,6 +182,7 @@ class HocrTransform: def to_pdf( self, + *, out_filename: Path, image_filename: Optional[Path] = None, show_bounding_boxes: bool = False, @@ -433,10 +434,10 @@ if __name__ == "__main__": parser.add_argument('outputfile', help='Path to the PDF file to be generated') args = parser.parse_args() - hocr = HocrTransform(args.hocrfile, args.resolution) + hocr = HocrTransform(hocr_filename=args.hocrfile, dpi=args.resolution) hocr.to_pdf( - args.outputfile, - args.image, - args.boundingboxes, + out_filename=args.outputfile, + image_filename=args.image, + show_bounding_boxes=args.boundingboxes, interword_spaces=args.interword_spaces, ) diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index f2f0c660..136c9a49 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -53,8 +53,10 @@ def test_mono_image(blank_hocr, outdir): im.putpixel((n, n), 1) im.save(outdir / 'mono.tif', format='TIFF') - hocr = hocrtransform.HocrTransform(str(blank_hocr), 300) - hocr.to_pdf(str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif')) + hocr = hocrtransform.HocrTransform(hocr_filename=str(blank_hocr), dpi=300) + hocr.to_pdf( + out_filename=str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif') + ) check_pdf(str(outdir / 'mono.pdf')) diff --git a/tests/test_optimize.py b/tests/test_optimize.py index e49e4878..e90b724a 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -143,7 +143,7 @@ def test_multiple_pngs(resources, outdir): outputstream=inpdf, ) - def mockquant(input_file, output_file, _quality_min, _quality_max): + def mockquant(input_file, output_file, *args): with Image.open(input_file) as im: draw = ImageDraw.Draw(im) draw.rectangle((0, 0, im.width, im.height), fill=128) From 9cba738b485fec161c50768c0eaa8aacc1783995 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 9 Dec 2020 15:07:34 -0800 Subject: [PATCH 801/880] Remove deprecated code --- src/ocrmypdf/optimize.py | 81 +--------------------------------------- 1 file changed, 1 insertion(+), 80 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 3fa92355..413c561c 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -36,7 +36,7 @@ from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._jobcontext import PdfContext from ocrmypdf.exceptions import OutputFileAccessError -from ocrmypdf.helpers import deprecated, safe_symlink +from ocrmypdf.helpers import safe_symlink log = logging.getLogger(__name__) @@ -64,10 +64,6 @@ def jpg_name(root: Path, xref: Xref) -> Path: return img_name(root, xref, '.jpg') -def tif_name(root: Path, xref: Xref) -> Path: - return img_name(root, xref, '.tif') - - def extract_image_filter( pike: Pdf, root: Path, image: Object, xref: Xref ) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]: @@ -493,81 +489,6 @@ def transcode_pngs( _transcode_png(pike, filename, xref) -@deprecated -def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cover - im_obj.BitsPerComponent = 1 - im_obj.Width = compdata.w - im_obj.Height = compdata.h - - im_obj.write(compdata.read()) - - log.debug(f"PNG to G4 {im_obj.objgen}") - if Name.Predictor in im_obj: - del im_obj.Predictor - if Name.DecodeParms in im_obj: - del im_obj.DecodeParms - im_obj.DecodeParms = Dictionary( - K=-1, BlackIs1=bool(compdata.minisblack), Columns=compdata.w - ) - - im_obj.Filter = Name.CCITTFaxDecode - return - - -@deprecated -def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None: # pragma: no cover - # When a PNG is inserted into a PDF, we more or less copy the IDAT section from - # the PDF and transfer the rest of the PNG headers to PDF image metadata. - # One thing we have to do is tell the PDF reader whether a predictor was used - # on the image before Flate encoding. (Typically one is.) - # According to Leptonica source, PDF readers don't actually need us - # to specify the correct predictor, they just need a value of either: - # 1 - no predictor - # 10-14 - there is a predictor - # Leptonica's compdata->predictor only tells TRUE or FALSE - # 10-14 means the actual predictor is specified in the data, so for any - # number >= 10 the PDF reader will use whatever the PNG data specifies. - # In practice Leptonica should use Paeth, 14, but 15 seems to be the - # designated value for "optimal". So we will use 15. - # See: - # - PDF RM 7.4.4.4 Table 10 - # - https://github.com/DanBloomberg/leptonica/blob/master/src/pdfio2.c#L757 - predictor = 15 if compdata.predictor > 0 else 1 - dparms = Dictionary(Predictor=predictor) - if predictor > 1: - dparms.BitsPerComponent = compdata.bps # Yes, this is redundant - dparms.Colors = compdata.spp - dparms.Columns = compdata.w - - im_obj.BitsPerComponent = compdata.bps - im_obj.Width = compdata.w - im_obj.Height = compdata.h - - log.debug( - f"PNG {im_obj.objgen}: palette={compdata.ncolors} spp={compdata.spp} bps={compdata.bps}" - ) - if compdata.ncolors > 0: - # .ncolors is the number of colors in the palette, not the number of - # colors used in a true color image. The palette string is always - # given as RGB tuples even when the image is grayscale; see - # https://github.com/DanBloomberg/leptonica/blob/master/src/colormap.c#L2067 - palette_pdf_string = compdata.get_palette_pdf_string() - palette_data = pikepdf.Object.parse(palette_pdf_string) - palette_stream = pikepdf.Stream(pike, bytes(palette_data)) - palette = [Name.Indexed, Name.DeviceRGB, compdata.ncolors - 1, palette_stream] - cs = palette - else: - # ncolors == 0 means we are using a colorspace without a palette - if compdata.spp == 1: - cs = Name.DeviceGray - elif compdata.spp == 4: - cs = Name.DeviceCMYK - else: # spp == 3 - cs = Name.DeviceRGB - im_obj.ColorSpace = cs - im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms) - - def optimize(input_file: Path, output_file: Path, context, save_settings) -> None: options = context.options if options.optimize == 0: From a48ca556c7f568d3245096219c0a1c34be184803 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 14 Feb 2021 01:22:33 -0800 Subject: [PATCH 802/880] Add filter_pdf_page hook --- src/ocrmypdf/_pipeline.py | 5 ++++ src/ocrmypdf/pluginspec.py | 54 ++++++++++++++++++++++++++++++++++---- 2 files changed, 54 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 0871aeba..a5b8bc5a 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -610,6 +610,11 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext): imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf ) log.debug('convert done') + + output_file = page_context.plugin_manager.hook.filter_pdf_page( + page=page_context, image_filename=image, output_pdf=output_file + ) + return output_file diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index d0c10239..5fa0b3ca 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -238,11 +238,11 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: will be resized and the OCR layer misaligned. OCRmyPDF does not nothing to enforce these constraints; it is up to the plugin to do sensible things. - OCRmyPDF will create the PDF page based on the image format used. If you - convert the image to a JPEG, the output page will be created as a JPEG, etc. - If you change the colorspace, that change will be kept. Note that the - OCRmyPDF image optimization stage, if enabled, may ultimately chose a - different format. + OCRmyPDF will create the PDF page based on the image format used (unless the + hook is overriden). If you convert the image to a JPEG, the output page will + be created as a JPEG, etc. If you change the colorspace, that change will be + kept. Note that the OCRmyPDF image optimization stage, if enabled, may + ultimately chose a different format. If the return value is a file that does not exist, ``FileNotFoundError`` will occur. The return value should be a path to a file in the same folder @@ -260,6 +260,50 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path: """ +@hookspec(firstresult=True) +def filter_pdf_page( + page: 'PageContext', image_filename: Path, output_pdf: Path +) -> Path: + """Called to convert a filtered whole page image into a PDF. + + A whole page image is only produced when preprocessing command line arguments + are issued or when ``--force-ocr`` is issued. If no whole page is image is + produced for a given page, this function will not be called. This is not + the image that will be shown to OCR. The whole page image is filtered in + the hook above, ``filter_page_image``, then this function is called for + PDF conversion. + + This function will only be called when OCRmyPDF runs in a mode such as + "force OCR" mode where rasterizing of all content is performed. + + Clever things could be done at this stage such as segmenting the page image into + color regions or vector equivalents. + + The provider of the hook implementation is responsible for ensuring that the + OCR text layer is aligned with the PDF produced here, or text misalignment + will result. + + Currently this function must produce a single page PDF or the pipeline will + fail. If the intent is to remove the PDF, then create a single page empty + PDF. + + Args: + page: Context for this page. + image_filename: Filename of the input image used to create output_pdf, + for "reference" if recreating the output_pdf entirely. + output_pdf: The previous created output_pdf. + + Returns: + output_pdf + + Note: + This hook will be called from child processes. Modifying global state + will not affect the main process or other child processes. + Note: + This is a :ref:`firstresult hook`. + """ + + OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence')) """Expresses an OCR engine's confidence in page rotation. From 18e613657cac71e7c12b0a3aa950747a181299b4 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 14 Feb 2021 01:23:01 -0800 Subject: [PATCH 803/880] docker-compose: fix typo --- misc/docker-compose.example.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/docker-compose.example.yml b/misc/docker-compose.example.yml index 0db102b9..9668b2b9 100644 --- a/misc/docker-compose.example.yml +++ b/misc/docker-compose.example.yml @@ -9,7 +9,7 @@ services: - "/media/scan:/input" - "/mnt/scan:/output" environment: - - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 + - OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0 user: ":" entrypoint: python3 command: watcher.py From 2898879be7bdf4b526c3924626ea673aa6768a56 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 14 Feb 2021 01:23:01 -0800 Subject: [PATCH 804/880] docker-compose: fix typo --- misc/docker-compose.example.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/misc/docker-compose.example.yml b/misc/docker-compose.example.yml index 0db102b9..9668b2b9 100644 --- a/misc/docker-compose.example.yml +++ b/misc/docker-compose.example.yml @@ -9,7 +9,7 @@ services: - "/media/scan:/input" - "/mnt/scan:/output" environment: - - OCR_OUTPUT_DIRECTORY_YEAR_MONT=0 + - OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0 user: ":" entrypoint: python3 command: watcher.py From 2a52c6dec2ea58a7ba0a108b5dd60b14691c8d7b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 14 Feb 2021 01:35:33 -0800 Subject: [PATCH 805/880] optimize: skip images with unusually small dimensions They're unlikely to be handled well by our recompressors. It seems that JBIG2 cannot handle very small widths. Fixes #732 --- src/ocrmypdf/optimize.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 3fa92355..05057d5c 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -76,6 +76,9 @@ def extract_image_filter( if image.Length < 100: log.debug(f"Skipping small image, xref {xref}") return None + if image.Width < 8 or image.Height < 8: # Issue 732 + log.debug(f"Skipping oddly sized image, xref {xref}") + return None pim = PdfImage(image) From 82de78b6b0ec35b038a9e9db560bd3e99f70d650 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 14 Feb 2021 01:51:26 -0800 Subject: [PATCH 806/880] v11.6.1 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 19c4f60a..35467958 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.6.1 +======= + +- Fixed an issue with attempting optimize unusually narrow-width images by excluding + these images from optimization (#732). +- Remove an obsolete compatibility shim for a version of pikepdf that is no longer + supported. + v11.6.0 ======= From 8770fff96885dd16b2bc95e5815592bb18d7ef94 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Feb 2021 01:05:08 -0800 Subject: [PATCH 807/880] tests: remove unreliable/incomplete test --- tests/test_rotation.py | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/tests/test_rotation.py b/tests/test_rotation.py index bb699124..8ee08344 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -249,16 +249,6 @@ def test_rotate_page_level(image_angle, page_angle, resources, outdir): assert check_monochrome_correlation(outdir, reference, 1, out, 1) > 0.2 -def test_tesseract_orientation(resources, tmp_path): - pix = leptonica.Pix.open(resources / 'crom.png') - pix_rotated = pix.rotate_orth(2) # 180 degrees clockwise - pix_rotated.write_implied_format(tmp_path / '000001.png') - - tesseract.get_orientation( # Test results of this are unreliable - tmp_path / '000001.png', engine_mode='3', timeout=10 - ) - - def test_rasterize_rotates(resources, tmp_path): pm = get_plugin_manager([]) From 064f935699e9763045f89f8128d18e6ba9b26635 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Feb 2021 01:47:09 -0800 Subject: [PATCH 808/880] Fix page rotation regression Page size fixes in commit b26749 did accounted for a "kept" rotation, but not a corrected rotation. Fixes #730. --- src/ocrmypdf/_pipeline.py | 8 +- src/ocrmypdf/_sync.py | 2 +- tests/plugins/tesseract_debug_rotate.py | 112 ++++++++++++++++++++++++ tests/test_rotation.py | 51 ++++++++++- 4 files changed, 169 insertions(+), 4 deletions(-) create mode 100644 tests/plugins/tesseract_debug_rotate.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 6de1f2e9..28ddf6e2 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -584,7 +584,9 @@ def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path: return output_file -def create_pdf_page_from_image(image: Path, page_context: PageContext): +def create_pdf_page_from_image( + image: Path, page_context: PageContext, orientation_correction +): # We rasterize a square DPI version of each page because most image # processing tools don't support rectangular DPI. Use the square DPI as it # accurately describes the image. It would be possible to resample the image @@ -595,7 +597,8 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext): pageinfo = page_context.pageinfo pagesize = 72.0 * float(pageinfo.width_inches), 72.0 * float(pageinfo.height_inches) - if pageinfo.rotation % 180 == 90: + effective_rotation = (pageinfo.rotation - orientation_correction) % 360 + if effective_rotation % 180 == 90: pagesize = pagesize[1], pagesize[0] # This create a single page PDF @@ -607,6 +610,7 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext): imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf ) log.debug('convert done') + return output_file diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index af32ad8c..43e5008a 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -204,7 +204,7 @@ def exec_page_sync(page_context: PageContext): if filtered_image: visible_image_out = filtered_image pdf_page_from_image_out = create_pdf_page_from_image( - visible_image_out, page_context + visible_image_out, page_context, orientation_correction ) if options.pdf_renderer.startswith('hocr'): diff --git a/tests/plugins/tesseract_debug_rotate.py b/tests/plugins/tesseract_debug_rotate.py new file mode 100644 index 00000000..30c613bd --- /dev/null +++ b/tests/plugins/tesseract_debug_rotate.py @@ -0,0 +1,112 @@ +# © 2020 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +"""Tesseract no-op/fixed rotate plugin + +To quickly run tests where getting OCR output is not necessary and we want to test +the rotation pipeline. + +In 'hocr' mode, create a .hocr file that specifies no text found. + +In 'pdf' mode, convert the image to PDF using another program. + +In orientation check mode, report 0, 90, 180, 270... based on page number. +""" + +import pikepdf +from PIL import Image + +from ocrmypdf import OcrEngine, OrientationConfidence, hookimpl +from ocrmypdf.helpers import page_number + +HOCR_TEMPLATE = ''' + + + + + + + + + +

+
+

+ + +

+
+
+ +''' + + +class FixedRotateNoopOcrEngine(OcrEngine): + @staticmethod + def version(): + return '4.0.0' + + @staticmethod + def creator_tag(options): + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + return f"NO-OP {tag} {FixedRotateNoopOcrEngine.version()}" + + def __str__(self): + return f"NO-OP {FixedRotateNoopOcrEngine.version()}" + + @staticmethod + def languages(options): + return {'eng'} + + @staticmethod + def get_orientation(input_file, options): + page = page_number(input_file) + + angle = ((page - 1) * 90) % 360 + + return OrientationConfidence(angle=angle, confidence=99.9) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with Image.open(input_file) as im, open( + output_hocr, 'w', encoding='utf-8' + ) as f: + w, h = im.size + f.write(HOCR_TEMPLATE.format(str(w), str(h))) + with open(output_text, 'w') as f: + f.write('') + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with Image.open(input_file) as im: + dpi = im.info['dpi'] + pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1] + ptsize = pagesize[0] * 72, pagesize[1] * 72 + pdf = pikepdf.new() + pdf.add_blank_page(page_size=ptsize) + pdf.save(output_pdf, static_id=True) + output_text.write_text('') + + +@hookimpl +def get_ocr_engine(): + return FixedRotateNoopOcrEngine() diff --git a/tests/test_rotation.py b/tests/test_rotation.py index 8ee08344..b826a09e 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -6,15 +6,17 @@ from io import BytesIO +from math import cos, pi, sin from os import fspath import img2pdf import pikepdf import pytest from PIL import Image +from reportlab.pdfgen.canvas import Canvas from ocrmypdf import leptonica -from ocrmypdf._exec import ghostscript, tesseract +from ocrmypdf._exec import ghostscript from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import PdfInfo @@ -277,3 +279,50 @@ def test_rasterize_rotates(resources, tmp_path): filter_vector=False, ) assert Image.open(img).size == (151, 123), "Image not rotated" + + +def test_simulated_scan(outdir): + canvas = Canvas( + fspath(outdir / 'fakescan.pdf'), + pagesize=(209.8, 297.6), + ) + + page_vars = [(2, 36, 250), (91, 170, 240), (179, 190, 36), (271, 36, 36)] + + for n, page_var in enumerate(page_vars): + text = canvas.beginText() + text.setFont('Helvetica', 20) + + angle, x, y = page_var + cos_a, sin_a = cos(angle / 180.0 * pi), sin(angle / 180.0 * pi) + + text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, x, y) + text.textOut(f'Page {n + 1}') + canvas.drawText(text) + canvas.showPage() + canvas.save() + + check_ocrmypdf( + outdir / 'fakescan.pdf', + outdir / 'out.pdf', + '--force-ocr', + '--deskew', + '--rotate-pages', + '--plugin', + 'tests/plugins/tesseract_debug_rotate.py', + ) + + with pikepdf.open(outdir / 'out.pdf') as pdf: + assert ( + pdf.pages[1].MediaBox[2] > pdf.pages[1].MediaBox[3] + ), "Wrong orientation: not landscape" + assert ( + pdf.pages[3].MediaBox[2] > pdf.pages[3].MediaBox[3] + ), "Wrong orientation: Not landscape" + + assert ( + pdf.pages[0].MediaBox[2] < pdf.pages[0].MediaBox[3] + ), "Wrong orientation: Not portrait" + assert ( + pdf.pages[2].MediaBox[2] < pdf.pages[2].MediaBox[3] + ), "Wrong orientation: Not portrait" From 3692868004c2c4255603b3f75b4b1d872ffd1bec Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 15 Feb 2021 01:48:14 -0800 Subject: [PATCH 809/880] v11.6.2 release notes --- docs/release_notes.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 35467958..15357e00 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.6.2 +======= + +- Fixed a regression where the wrong page orientation would be produced when using + arguments such as ``--deskew --rotate-pages`` (#730). + v11.6.1 ======= From 079ee86d4354470616051abaf07acb48d54bd1a7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 18 Feb 2021 01:48:56 -0800 Subject: [PATCH 810/880] pyproject: also target py39 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index a28f55c0..920174e5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,7 @@ build-backend = "setuptools.build_meta" [tool.black] line-length = 88 -target-version = ["py36", "py37", "py38"] +target-version = ["py36", "py37", "py38", "py39"] skip-string-normalization = true include = '\.pyi?$' exclude = ''' From 5e2206bae76fe4b98f353676fdb37d8092a305e4 Mon Sep 17 00:00:00 2001 From: Dima Kuznetsov Date: Sat, 20 Feb 2021 02:55:35 +0200 Subject: [PATCH 811/880] Allow --sidecar along --pages (#735) --- src/ocrmypdf/_pipeline.py | 26 +++++++++-- src/ocrmypdf/_validation.py | 2 - tests/test_pipeline.py | 87 +++++++++++++++++++++++++++++++++++++ tests/test_validation.py | 2 - 4 files changed, 110 insertions(+), 7 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 28ddf6e2..3ca5e97f 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -808,11 +808,27 @@ def optimize_pdf(input_file: Path, context: PdfContext): return output_file +def enumerate_compress_ranges(iterable): + skipped_from = None + for index, txt_file in enumerate(iterable): + index += 1 + if txt_file: + if skipped_from is not None: + yield (skipped_from, index - 1), None + skipped_from = None + yield (index, index), txt_file + else: + if skipped_from is None: + skipped_from = index + if skipped_from is not None: + yield (skipped_from, index), None + + def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext): output_file = context.get_path('sidecar.txt') with open(output_file, 'w', encoding="utf-8") as stream: - for page_num, txt_file in enumerate(txt_files): - if page_num != 0: + for (frm, to), txt_file in enumerate_compress_ranges(txt_files): + if frm != 1: stream.write('\f') # Form feed between pages if txt_file: with open(txt_file, 'r', encoding="utf-8") as in_: @@ -825,7 +841,11 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext): else: stream.write(txt) else: - stream.write(f'[OCR skipped on page {(page_num + 1)}]') + if frm != to: + pages = f'{frm}-{to}' + else: + pages = f'{frm}' + stream.write(f'[OCR skipped on page(s) {pages}]') return output_file diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index a6ee9a08..2e2cd529 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -184,8 +184,6 @@ def check_options_ocr_behavior(options): ) if exclusive_options >= 2: raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.") - if options.pages and options.sidecar: - raise BadArgsError("--pages and --sidecar are mutually exclusive") if options.pages: options.pages = _pages_from_ranges(options.pages) diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index 64254e84..becef2c4 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -61,3 +61,90 @@ def test_dpi_needed(image, text, vector, result, rgb_image, outdir): assert _pipeline.get_canvas_square_dpi(pi[0], mock) == result assert _pipeline.get_page_square_dpi(pi[0], mock) == result + + +@pytest.mark.parametrize( + # Name for nicer -v output + 'name,input,output', + ( + ( + 'empty_input', + # Input: + (), + # Output: + (), + ), + ( + 'no_values', + # Input: + ('', '', '', '', ''), + # Output: + ( + ((1, 5), None), + ), + ), + ( + 'no_empty_values', + # Input: + ('v', 'w', 'x', 'y', 'z'), + # Output: + ( + ((1, 1), 'v'), + ((2, 2), 'w'), + ((3, 3), 'x'), + ((4, 4), 'y'), + ((5, 5), 'z'), + ), + ), + ( + 'skip_head', + # Input: + ('', '', 'x', 'y', 'z'), + # Output: + ( + ((1, 2), None), + ((3, 3), 'x'), + ((4, 4), 'y'), + ((5, 5), 'z'), + ), + ), + ( + 'skip_tail', + # Input: + ('x', 'y', 'z', '', ''), + # Output: + ( + ((1, 1), 'x'), + ((2, 2), 'y'), + ((3, 3), 'z'), + ((4, 5), None), + ), + ), + ( + 'range_in_middle', + # Input: + ('x', '', '', '', 'y'), + # Output: + ( + ((1, 1), 'x'), + ((2, 4), None), + ((5, 5), 'y'), + ), + ), + ( + 'range_in_middle_2', + # Input: + ('x', '', '', 'y', '', '', '', 'z'), + # Output: + ( + ((1, 1), 'x'), + ((2, 3), None), + ((4, 4), 'y'), + ((5, 7), None), + ((8, 8), 'z'), + ), + ), + ), +) +def test_enumerate_compress_ranges(name, input, output): + assert output == tuple(_pipeline.enumerate_compress_ranges(input)) \ No newline at end of file diff --git a/tests/test_validation.py b/tests/test_validation.py index f7ce5286..6eb53e48 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -90,8 +90,6 @@ def test_mutex_options(): vd.check_options_ocr_behavior(make_opts(redo_ocr=True, skip_text=True)) with pytest.raises(BadArgsError): vd.check_options_ocr_behavior(make_opts(redo_ocr=True, force_ocr=True)) - with pytest.raises(BadArgsError): - vd.check_options_ocr_behavior(make_opts(pages='1-3', sidecar='file.txt')) def test_optimizing(caplog): From dd1f5f7215ed6cd12fa1c9065a4226befc41e1ca Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 25 Feb 2021 16:10:20 -0800 Subject: [PATCH 812/880] pyproject: black doesn't like py39 yet --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 920174e5..a28f55c0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,7 @@ build-backend = "setuptools.build_meta" [tool.black] line-length = 88 -target-version = ["py36", "py37", "py38", "py39"] +target-version = ["py36", "py37", "py38"] skip-string-normalization = true include = '\.pyi?$' exclude = ''' From a23c22b0e8b373cd421f11aa6df2ce9b81ed8621 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 25 Feb 2021 22:51:53 -0800 Subject: [PATCH 813/880] helpers: tidy check_pdf --- src/ocrmypdf/helpers.py | 59 +++++++++++++++++++---------------------- 1 file changed, 28 insertions(+), 31 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 9fc27709..42039cab 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -181,45 +181,42 @@ def check_pdf(input_file: Path) -> bool: Checks for proper formatting and proper linearization. Uses pikepdf (which in turn, uses QPDF) to perform the checks. """ - pdf = None try: pdf = pikepdf.open(input_file) except pikepdf.PdfError as e: log.error(e) return False else: - messages = pdf.check() - for msg in messages: - if 'error' in msg.lower(): - log.error(msg) + with pdf: + messages = pdf.check() + for msg in messages: + if 'error' in msg.lower(): + log.error(msg) + else: + log.warning(msg) + + sio = StringIO() + linearize_msgs = '' + try: + # If linearization is missing entirely, we do not complain. We do + # complain if linearization is present but incorrect. + pdf.check_linearization(sio) + except RuntimeError: + pass + except ( # Workaround for a problematic pikepdf version + getattr(pikepdf, 'ForeignObjectError') + if pikepdf.__version__ == '2.1.0' + else NeverRaise + ): + pass else: - log.warning(msg) + linearize_msgs = sio.getvalue() + if linearize_msgs: + log.warning(linearize_msgs) - sio = StringIO() - linearize_msgs = '' - try: - # If linearization is missing entirely, we do not complain. We do - # complain if linearization is present but incorrect. - pdf.check_linearization(sio) - except RuntimeError: - pass - except ( - getattr(pikepdf, 'ForeignObjectError') - if pikepdf.__version__ == '2.1.0' # This version may throw wrong exception - else NeverRaise - ): - pass - else: - linearize_msgs = sio.getvalue() - if linearize_msgs: - log.warning(linearize_msgs) - - if not messages and not linearize_msgs: - return True - return False - finally: - if pdf: - pdf.close() + if not messages and not linearize_msgs: + return True + return False def clamp(n, smallest, largest): # mypy doesn't understand types for this From 4124889f360446bdfdab3ba0aaef1b154579c63c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Feb 2021 00:23:57 -0800 Subject: [PATCH 814/880] Don't generate PDF/A-1b with object streams Acrobat insists that PDF/A-1b should not have object streams. Other programs like veraPDF disagree with this restriction, but we can accommodate Acrobat so we will. Also add more tests around this. --- src/ocrmypdf/_pipeline.py | 26 ++++++++++++++++++++------ tests/test_pdfa.py | 34 ++++++++++++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 6 deletions(-) create mode 100644 tests/test_pdfa.py diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 3ca5e97f..747b2936 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -739,6 +739,24 @@ def should_linearize(working_file: Path, context: PdfContext): return False +def get_pdf_save_settings(output_type: str): + if output_type == 'pdfa-1': + # Trigger recompression to ensure object streams are removed, because + # Acrobat complains about them in PDF/A-1b validation. + return dict( + preserve_pdfa=True, + compress_streams=True, + stream_decode_level=pikepdf.StreamDecodeLevel.generalized, + object_stream_mode=pikepdf.ObjectStreamMode.disable, + ) + else: + return dict( + preserve_pdfa=True, + compress_streams=True, + object_stream_mode=(pikepdf.ObjectStreamMode.generate), + ) + + def metadata_fixup(working_file: Path, context: PdfContext): output_file = context.get_path('metafix.pdf') options = context.options @@ -783,9 +801,7 @@ def metadata_fixup(working_file: Path, context: PdfContext): pdf.save( output_file, - compress_streams=True, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, + **get_pdf_save_settings(options.output_type), linearize=( # Don't linearize if optimize() will be linearizing too should_linearize(working_file, context) if options.optimize == 0 @@ -799,10 +815,8 @@ def metadata_fixup(working_file: Path, context: PdfContext): def optimize_pdf(input_file: Path, context: PdfContext): output_file = context.get_path('optimize.pdf') save_settings = dict( - compress_streams=True, - preserve_pdfa=True, - object_stream_mode=pikepdf.ObjectStreamMode.generate, linearize=should_linearize(input_file, context), + **get_pdf_save_settings(context.options.output_type), ) optimize(input_file, output_file, context, save_settings) return output_file diff --git a/tests/test_pdfa.py b/tests/test_pdfa.py new file mode 100644 index 00000000..d0c269ff --- /dev/null +++ b/tests/test_pdfa.py @@ -0,0 +1,34 @@ +# © 2021 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +import pikepdf +import pytest + +check_ocrmypdf = pytest.helpers.check_ocrmypdf + + +@pytest.mark.parametrize('optimize', (0, 3)) +@pytest.mark.parametrize('pdfa_level', (1, 2, 3)) +def test_pdfa(resources, outpdf, optimize, pdfa_level): + check_ocrmypdf( + resources / 'francais.pdf', + outpdf, + '--plugin', + 'tests/plugins/tesseract_noop.py', + f'--output-type=pdfa-{pdfa_level}', + f'--optimize={optimize}', + ) + if pdfa_level in (2, 3): + # PDF/A-2 allows ObjStm + assert b'/ObjStm' in outpdf.read_bytes() + elif pdfa_level == 1: + # PDF/A-1 might allow ObjStm, but Acrobat does not approve it, so + # we don't use it + assert b'/ObjStm' not in outpdf.read_bytes() + + with pikepdf.open(outpdf) as pdf: + with pdf.open_metadata() as m: + assert m.pdfa_status == f'{pdfa_level}B' From 5c470778a339238daf296df3b5bbe181762d21c0 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Feb 2021 00:29:52 -0800 Subject: [PATCH 815/880] v11.7.0 release notes --- docs/release_notes.rst | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 15357e00..4d8836b8 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,16 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.7.0 +======= + +- We now support using ``--sidecar`` in conjunction with ``--pages``; these arguments + used to be mutually exclusive. (#735) +- Fixed a possible issue with PDF/A-1b generation. Acrobat complained that our PDFs use + object streams. More robust PDF/A validators like veraPDF don't consider this a + problem, but we'll honor Acrobat's objection from here on. This may increase file + size of PDF/A-1b files. PDF/A-2b files will not be affected. + v11.6.2 ======= From 2261c51effb0a318d4aa9df697a34a6850432e73 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 26 Feb 2021 01:19:03 -0800 Subject: [PATCH 816/880] Reactivate pngquant on windows --- azure-pipelines.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index cece2435..db05746c 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -32,7 +32,7 @@ stages: choco install --yes --no-progress --pre tesseract choco install --yes --no-progress python3 choco install --yes --no-progress ghostscript - # choco install --yes --no-progress pngquant + choco install --yes --no-progress pngquant displayName: "Install system packages" - pwsh: | refreshenv From 8ffc99f64868b0979d33c7f1b8b0017c17644839 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 28 Feb 2021 16:39:00 -0800 Subject: [PATCH 817/880] optimize: log errors more loudly --- src/ocrmypdf/optimize.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 05057d5c..f62376b8 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -258,8 +258,8 @@ def extract_images( result = extract_fn( pike=pike, root=root, image=image, xref=xref, options=options ) - except Exception as e: # pylint: disable=broad-except - log.debug("Image xref %s, error %s", xref, repr(e)) + except Exception: # pylint: disable=broad-except + log.exception(f"While extracting image xref {xref}, an error occurred") errors += 1 else: if result: From 0885799010096a743510a8d3ea85f09a4eda4dee Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sun, 28 Feb 2021 19:31:21 -0800 Subject: [PATCH 818/880] Update docs for conda Closes #743 --- README.md | 26 ++++++++--------------- docs/installation.rst | 48 ++++++++++++++++++++++++------------------- 2 files changed, 36 insertions(+), 38 deletions(-) diff --git a/README.md b/README.md index c4c16856..0c3b20cc 100644 --- a/README.md +++ b/README.md @@ -59,23 +59,15 @@ I searched the web for a free command line tool to OCR PDF files: I found many, Linux, Windows, macOS and FreeBSD are supported. Docker images are also available. -Users of Debian 9 or later or Ubuntu 16.10 or later may simply - -```bash -apt-get install ocrmypdf -``` - -and users of Fedora 29 or later may simply - -```bash -dnf install ocrmypdf -``` - -and Homebrew users (macOS, Linux, Windows Subsystem for Linux) may simply - -```bash -brew install ocrmypdf -``` +| Operating system | Install command | +| ----------------------------- | ------------------------------| +| Debian, Ubuntu | ``apt install ocrmypdf`` | +| Windows Subsystem for Linux | ``apt install ocrmypdf`` | +| Fedora | ``dnf install ocrmypdf`` | +| macOS | ``brew install ocrmypdf`` | +| LinuxBrew | ``brew install ocrmypdf`` | +| FreeBSD | ``pkg install py37-ocrmypdf`` | +| Conda | ``conda install ocrmypdf`` | For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps. diff --git a/docs/installation.rst b/docs/installation.rst index f0524f07..9dc46286 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -12,19 +12,21 @@ system/platform. This version may be out of date, however. These platforms have one-liner installs: -+-----------------------------+-------------------------------+ -| Debian, Ubuntu | ``apt install ocrmypdf`` | -+-----------------------------+-------------------------------+ -| Windows Subsystem for Linux | ``apt install ocrmypdf`` | -+-----------------------------+-------------------------------+ -| Fedora | ``dnf install ocrmypdf`` | -+-----------------------------+-------------------------------+ -| macOS | ``brew install ocrmypdf`` | -+-----------------------------+-------------------------------+ -| LinuxBrew | ``brew install ocrmypdf`` | -+-----------------------------+-------------------------------+ -| FreeBSD | ``pkg install py37-ocrmypdf`` | -+-----------------------------+-------------------------------+ ++-------------------------------+-------------------------------+ +| Debian, Ubuntu | ``apt install ocrmypdf`` | ++-------------------------------+-------------------------------+ +| Windows Subsystem for Linux | ``apt install ocrmypdf`` | ++-------------------------------+-------------------------------+ +| Fedora | ``dnf install ocrmypdf`` | ++-------------------------------+-------------------------------+ +| macOS | ``brew install ocrmypdf`` | ++-------------------------------+-------------------------------+ +| LinuxBrew | ``brew install ocrmypdf`` | ++-------------------------------+-------------------------------+ +| FreeBSD | ``pkg install py37-ocrmypdf`` | ++-------------------------------+-------------------------------+ +| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` | ++-------------------------------+-------------------------------+ More detailed procedures are outlined below. If you want to do a manual install, or install a more recent version than your platform provides, read on. @@ -54,6 +56,9 @@ Debian and Ubuntu 18.04 or newer .. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg :alt: Ubuntu 20.04 LTS +.. |ubu-2010| image:: https://repology.org/badge/version-for-repo/ubuntu_20_10/ocrmypdf.svg + :alt: Ubuntu 20.10 + +-----------------------------------------------+ | **OCRmyPDF versions in Debian & Ubuntu** | +-----------------------------------------------+ @@ -61,7 +66,7 @@ Debian and Ubuntu 18.04 or newer +-----------------------------------------------+ | |deb-stable| |deb-testing| |deb-unstable| | +-----------------------------------------------+ -| |ubu-1804| |ubu-2004| | +| |ubu-1804| |ubu-2004| |ubu-2010| | +-----------------------------------------------+ Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users @@ -90,15 +95,15 @@ For full details on version availability for your platform, check the automatically detect it (specifically the ``jbig2`` binary) on the ``PATH``. To add JBIG2 encoding, see :ref:`jbig2`. -Fedora 29 or newer ------------------- - -.. |fedora-31| image:: https://repology.org/badge/version-for-repo/fedora_31/ocrmypdf.svg - :alt: Fedora 31 +Fedora +------ .. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg :alt: Fedora 32 +.. |fedora-33| image:: https://repology.org/badge/version-for-repo/fedora_33/ocrmypdf.svg + :alt: Fedora 33 + .. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg :alt: Fedore Rawhide @@ -107,7 +112,7 @@ Fedora 29 or newer +-----------------------------------------------+ | |latest| | +-----------------------------------------------+ -| |fedora-31| |fedora-32| |fedora-rawhide| | +| |fedora-32| |fedora-33| |fedora-rawhide| | +-----------------------------------------------+ Users of Fedora 29 or later may simply @@ -355,7 +360,8 @@ To install OCRmyPDF for Alpine Linux: Mageia 7 -------- -Install the following dependencies: +There is no OS-level packaging available for Mageia, so you must install the +dependencies: .. code-block:: bash From 6e71fe118676400605924ae620d4152e0e22545e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Mar 2021 00:44:21 -0800 Subject: [PATCH 819/880] Clarify --unpaper-args errors --- src/ocrmypdf/_validation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 2e2cd529..58ea18bf 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -133,7 +133,7 @@ def check_options_preprocessing(options): options.unpaper_args ) except Exception as e: - raise BadArgsError(str(e)) + raise BadArgsError("--unpaper-args") from e def _pages_from_ranges(ranges: str) -> Set[int]: From ffcae9a1a03ed269542838a91367644660ddfe95 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Mar 2021 00:46:35 -0800 Subject: [PATCH 820/880] v11.7.1 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 4d8836b8..ffad6255 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.7.1 +======= + +- Some exceptions while attempting image optimization were only logged at the debug + level, causing them to be suppressed. These errors are now logged appropriately. +- Improved the error message related to ``--unpaper-args``. +- Updated documentation to mention the new conda distribution. + v11.7.0 ======= From 25c8c4656f5a85f7d5643afa0feb77b8a6384bf9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Mar 2021 01:02:59 -0800 Subject: [PATCH 821/880] Fix error message change --- src/ocrmypdf/_validation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 58ea18bf..82a9cc35 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -133,7 +133,7 @@ def check_options_preprocessing(options): options.unpaper_args ) except Exception as e: - raise BadArgsError("--unpaper-args") from e + raise BadArgsError("--unpaper-args: " + str(e)) from e def _pages_from_ranges(ranges: str) -> Set[int]: From 079c162a96f043dfd4117f8335708f1378afab81 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 5 Mar 2021 00:27:50 -0800 Subject: [PATCH 822/880] Ensure sidecar is not input or output file --- src/ocrmypdf/_validation.py | 4 ++++ src/ocrmypdf/cli.py | 3 ++- tests/test_validation.py | 8 ++++++++ 3 files changed, 14 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 82a9cc35..a3dd7448 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -112,6 +112,10 @@ def check_options_sidecar(options): "--sidecar filename must be specified when output file is stdout." ) options.sidecar = options.output_file + '.txt' + if options.sidecar == options.input_file or options.sidecar == options.output_file: + raise BadArgsError( + "--sidecar file must be different from the input and output files" + ) def check_options_preprocessing(options): diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 24d5e036..204b7b31 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -167,7 +167,8 @@ Online documentation is located at: metavar='FILE', help="Generate sidecar text files that contain the same text recognized " "by Tesseract. This may be useful for building a OCR text database. " - "If FILE is omitted, the sidecar file be named {output_file}.txt " + "If FILE is omitted, the sidecar file be named {output_file}.txt; the next " + "argument must NOT be the name of the input PDF. " "If FILE is set to '-', the sidecar is written to stdout (a " "convenient way to preview OCR quality). The output file and sidecar " "may not both use stdout at the same time.", diff --git a/tests/test_validation.py b/tests/test_validation.py index 6eb53e48..3393fd9f 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -18,6 +18,8 @@ from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.pdfinfo import PdfInfo +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api + def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): if language is not None: @@ -270,3 +272,9 @@ def test_two_languages(): *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} ) mock.assert_called() + + +def test_sidecar_equals_output(resources, no_outpdf): + op = no_outpdf + with pytest.raises(BadArgsError, match=r'--sidecar'): + run_ocrmypdf_api(resources / 'trivial.pdf', op, '--sidecar', op) From 873f915212b1253936ae3c79d7dc737f5c0055ba Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 9 Mar 2021 23:24:10 -0800 Subject: [PATCH 823/880] docs: mention problems with Debian/Ubuntu and other tidying --- docs/installation.rst | 28 ++++++++++++++++++++++++---- 1 file changed, 24 insertions(+), 4 deletions(-) diff --git a/docs/installation.rst b/docs/installation.rst index 4fc3738a..047d4bb5 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -158,7 +158,7 @@ To install ocrmypdf for the system: .. code-block:: bash - sudo pip3 install ocrmypdf + pip3 install ocrmypdf To install for the current user only: @@ -380,12 +380,16 @@ Install the following dependencies: To install ocrmypdf for the system: +.. code-block:: bash + # As root user pip3 install ocrmypdf ldconfig Or, to install for the current user only: +.. code-block:: bash + export PATH=$HOME/.local/bin:$PATH pip3 install --user ocrmypdf @@ -520,7 +524,7 @@ DLLs or other Windows patches, and may require a reboot. You may then use ``pip`` to install ocrmypdf. (This can performed by a user or Administrator.): -* ``pip install ocrmypdf +* ``pip install ocrmypdf`` Chocolatey automatically selects appropriate versions of these applications. If you are installing them manually, please install 64-bit versions of all applications for @@ -536,8 +540,9 @@ to change the PATH. .. warning:: As of early 2021, users have reported problems with the Microsoft Store version of - Python affected most third party Python packages including OCRmyPDF. Please use - Python downloaded from Python.org or Chocolatey as recommended here. + Python and OCRmyPDF. These issues affect many other third party Python packages. + Please download Python from Python.org or Chocolatey instead, and do not use the + Microsoft Store version. Windows Subsystem for Linux --------------------------- @@ -645,6 +650,21 @@ the latest version. However, PyPI and ``pip`` cannot address the fact that ``ocrmypdf`` depends on certain non-Python system libraries and programs being installed. +.. warning:: + + Debian and Ubuntu users: unfortunately, Debian and Ubuntu customize + Python in non-standard ways, and the nature of these customizations + varies from release to release. This can make for a frustrating + user experience. The instructions below work on almost all platforms that + have Python installed, except for Debian and Ubuntu, where you may need + to take additional steps. For best results on Debian and Ubuntu, use the + ``apt`` packages; or if these are too old, run + ``apt install python3-pip python3-venv``, create a virtual environment, + and install OCRmyPDF in that environment. + + `See here for more inforation on Debian-Python issues + `__. + For best results, first install `your platform's version `__ of ``ocrmypdf``, using the instructions elsewhere in this document. Then From c9594a4a5fc2b052f3cef07afec28f982d7dbdb6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 19 Mar 2021 00:31:27 -0700 Subject: [PATCH 824/880] Update pinned versions to avoid Pillow vulnerabilties See https://github.com/python-pillow/Pillow/blob/master/CHANGES.rst --- requirements/main.txt | 12 ++++++------ requirements/test.txt | 6 +++--- requirements/watcher.txt | 2 +- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index fde1197e..63ca302a 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -1,12 +1,12 @@ # requirements.txt can be used to replicate the developer's build environment # setup.py lists a separate set of requirements that are looser to simplify # installation -cffi == 1.14.3 -coloredlogs == 14.0 # technically optional +cffi == 1.14.5 +coloredlogs == 15.0 # technically optional img2pdf == 0.4.0 pdfminer.six == 20201018 -pikepdf == 2.0.0 +pikepdf == 2.9.0 pluggy == 0.13.1 -Pillow == 8.0.1 -reportlab == 3.5.55 -tqdm == 4.51.0 +Pillow == 8.1.2 +reportlab == 3.5.65 +tqdm == 4.59.0 diff --git a/requirements/test.txt b/requirements/test.txt index 8bdf1fe6..fea870d7 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,7 +1,7 @@ -pytest >= 5.0.0 +pytest >= 6.0.0 pytest-helpers-namespace >= 2019.1.8 -pytest-xdist >= 1.31.0 -pytest-cov >= 2.10.0 +pytest-xdist >= 2.2.0 +pytest-cov >= 2.11.1 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 # or brew install exempi #PyMuPDF == 1.13.4 # optional diff --git a/requirements/watcher.txt b/requirements/watcher.txt index cdbc5325..660d7af4 100644 --- a/requirements/watcher.txt +++ b/requirements/watcher.txt @@ -1 +1 @@ -watchdog == 0.10.2 +watchdog == 1.0.2 From d8f47768f99bdb0b938ad2b3934041c447c25fbc Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 19 Mar 2021 00:31:38 -0700 Subject: [PATCH 825/880] v11.7.2 release notes --- docs/release_notes.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index ffad6255..54adfb15 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,14 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.7.2 +======= + +- Updated pinned versions in main.txt, primarily to upgrade Pillow to 8.1.2, due + to recently disclosed security vulnerabilities in that software. +- The ``--sidecar`` parameter now causes an exception if set to the same file as + the input or output PDF. + v11.7.1 ======= From 0a42934c083da902c08255592407f6c302bf0941 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 20 Mar 2021 23:28:21 -0700 Subject: [PATCH 826/880] Exclude Group 3 images from optimization --- src/ocrmypdf/optimize.py | 4 ++++ tests/test_optimize.py | 13 +++++++++++++ 2 files changed, 17 insertions(+) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index f62376b8..b420948d 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -95,6 +95,10 @@ def extract_image_filter( log.debug(f"Skipping JPEG2000 iamge, xref {xref}") return None # Don't do JPEG2000 + if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0: + log.debug(f"Skipping CCITT Group 3 image, xref {xref}") + return None # pikepdf doesn't support Group 3 yet + if Name.Decode in image: log.debug(f"Skipping image with Decode table, xref {xref}") return None # Don't mess with custom Decode tables diff --git a/tests/test_optimize.py b/tests/test_optimize.py index e49e4878..7c979a87 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -185,3 +185,16 @@ def test_optimize_off(resources, outpdf): '--plugin', 'tests/plugins/tesseract_noop.py', ) + + +def test_group3(resources, outdir): + with pikepdf.open(resources / 'ccitt.pdf') as pdf: + im = pdf.pages[0].Resources.XObject['/Im1'] + assert ( + opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is not None + ), "Group 4 should be allowed" + + im.DecodeParms['/K'] = 0 + assert ( + opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is None + ), "Group 3 should be disallowed" From f4f0f3c022f6a0b0bb11b1ddd2d0a8a256296769 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 20 Mar 2021 23:30:42 -0700 Subject: [PATCH 827/880] v11.7.3 release notes --- docs/release_notes.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 54adfb15..6d239aff 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v11.7.3 +======= + +- Exclude CCITT Group 3 images from being optimized. Some libraries + OCRmyPDF uses do not seem to handle this obscure compression format properly. + You may get errors or possible corrupted output images without this fix. + v11.7.2 ======= From e96770c5e4658a2e695d1ddab69ad16903932fed Mon Sep 17 00:00:00 2001 From: Timo Klerx Date: Sat, 27 Mar 2021 00:13:28 +0100 Subject: [PATCH 828/880] Updated documentation for windows and additional languages (#753) Changed the documentation for how to add new languages on Windows. Co-authored-by: jbarlow83 --- docs/languages.rst | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/languages.rst b/docs/languages.rst index cc26fa68..ed414a50 100644 --- a/docs/languages.rst +++ b/docs/languages.rst @@ -70,4 +70,7 @@ derived Docker image as Windows users ============= -The Tesseract installer provided by Chocolatey already includes 100 languages. +The Tesseract installer provided by Chocolatey currently includes only English language. +To install other languages, download the respective language pack (``.traineddata`` file) +from https://github.com/tesseract-ocr/tessdata/ and place it in +``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed). From c5bf1dd90dc71ad324dceff839e3ce0ee5116d5f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 1 Apr 2021 16:01:23 -0700 Subject: [PATCH 829/880] reqs: bump reqs to avoid reportlab issue --- requirements/main.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/main.txt b/requirements/main.txt index 63ca302a..2362e6a4 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -8,5 +8,5 @@ pdfminer.six == 20201018 pikepdf == 2.9.0 pluggy == 0.13.1 Pillow == 8.1.2 -reportlab == 3.5.65 +reportlab == 3.5.66 tqdm == 4.59.0 From e09ae9c68a3c244fd198cf8b525ad523d81634e3 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 1 Apr 2021 16:25:40 -0700 Subject: [PATCH 830/880] Fix test suite failure if filter_pdf_page is missing --- src/ocrmypdf/builtin_plugins/default_filters.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) create mode 100644 src/ocrmypdf/builtin_plugins/default_filters.py diff --git a/src/ocrmypdf/builtin_plugins/default_filters.py b/src/ocrmypdf/builtin_plugins/default_filters.py new file mode 100644 index 00000000..83a8f3fc --- /dev/null +++ b/src/ocrmypdf/builtin_plugins/default_filters.py @@ -0,0 +1,12 @@ +# © 2021 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +from ocrmypdf import hookimpl + + +@hookimpl +def filter_pdf_page(page, image_filename, output_pdf): + return output_pdf From 2e155c31bf42fe5c0a1efefdad1b49ece04d4fde Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 1 Apr 2021 16:30:42 -0700 Subject: [PATCH 831/880] Ensure builtin module registration is deterministic --- src/ocrmypdf/_plugin_manager.py | 4 +++- src/ocrmypdf/builtin_plugins/__init__.py | 4 ++++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 76e12bd7..0cd69673 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -67,7 +67,9 @@ class OcrmypdfPluginManager(pluggy.PluginManager): # 1. Register builtins if self.__builtins: - for module in pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__): + for module in sorted( + pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__) + ): name = f'ocrmypdf.builtin_plugins.{module.name}' module = importlib.import_module(name) self.register(module) diff --git a/src/ocrmypdf/builtin_plugins/__init__.py b/src/ocrmypdf/builtin_plugins/__init__.py index 61f9f6d6..05d8c70e 100644 --- a/src/ocrmypdf/builtin_plugins/__init__.py +++ b/src/ocrmypdf/builtin_plugins/__init__.py @@ -3,3 +3,7 @@ # This Source Code Form is subject to the terms of the Mozilla Public # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. + +# This file exists only mark builtin_plugins as a package. +# The plugin manager will not load it, so anything defined here may not be +# processed as a module. From a2033698fa12077bb24f77ee81f951b681e7772a Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 1 Apr 2021 16:55:34 -0700 Subject: [PATCH 832/880] graft: use newer pikepdf style --- src/ocrmypdf/_graft.py | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 41db5e5e..262ec685 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -30,14 +30,10 @@ def _update_resources(*, obj, font, font_key, procset): obj can be a page or Form XObject. """ - resources = _ensure_dictionary(obj, '/Resources') - try: - fonts = resources['/Font'] - except KeyError: - fonts = pikepdf.Dictionary({}) + resources = _ensure_dictionary(obj, Name.Resources) + fonts = _ensure_dictionary(resources, Name.Font) if font_key is not None and font_key not in fonts: fonts[font_key] = font - resources['/Font'] = fonts # Reassign /ProcSet to one that just lists everything - ProcSet is # obsolete and doesn't matter but recommended for old viewer support @@ -290,8 +286,8 @@ class OcrGrafter: # finally move the lower left corner to match the mediabox ctm = translate @ rotate @ scale @ untranslate @ corner - base_resources = _ensure_dictionary(base_page, '/Resources') - base_xobjs = _ensure_dictionary(base_resources, '/XObject') + base_resources = _ensure_dictionary(base_page, Name.Resources) + base_xobjs = _ensure_dictionary(base_resources, Name.XObject) text_xobj_name = Name('/' + str(uuid.uuid4())) xobj = self.pdf_base.make_stream(pdf_text_contents) base_xobjs[text_xobj_name] = xobj From 653e2e23df58ae684defa6d44b67de1401f48137 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 2 Apr 2021 01:10:39 -0700 Subject: [PATCH 833/880] Update deps for next release --- requirements/main.txt | 2 +- setup.py | 11 +++-------- 2 files changed, 4 insertions(+), 9 deletions(-) diff --git a/requirements/main.txt b/requirements/main.txt index 2362e6a4..a49cf9fa 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ cffi == 1.14.5 coloredlogs == 15.0 # technically optional img2pdf == 0.4.0 pdfminer.six == 20201018 -pikepdf == 2.9.0 +pikepdf == 2.10.0 pluggy == 0.13.1 Pillow == 8.1.2 reportlab == 3.5.66 diff --git a/setup.py b/setup.py index 4e1c2bb3..98d395ec 100644 --- a/setup.py +++ b/setup.py @@ -17,10 +17,6 @@ if sys.version_info < (3, 6): print("Python 3.6 or newer is required", file=sys.stderr) sys.exit(1) -if 'upload' in sys.argv[1:]: - print('Use twine to upload the package - setup.py upload is insecure') - sys.exit(1) - tests_require = open('requirements/test.txt', encoding='utf-8').read().splitlines() @@ -73,11 +69,10 @@ setup( 'coloredlogs >= 14.0', # strictly optional 'img2pdf >= 0.3.0, < 0.5', # pure Python, so track HEAD closely 'pdfminer.six >= 20191110, != 20200720, <= 20201018', - "pikepdf >= 1.14.0, < 3 ; python_version < '3.9'", - "pikepdf >= 2.0.0 ; python_version >= '3.9'", - 'Pillow >= 7.0.0', + "pikepdf >= 2.10.0", + 'Pillow >= 8.1.2', 'pluggy >= 0.13.0, < 1.0', - 'reportlab >= 3.3.0', # oldest released version with sane image handling + 'reportlab >= 3.5.66', 'tqdm >= 4', ], tests_require=tests_require, From dd6cb7ce2003e3e69c8596d9989f723403e59364 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 2 Apr 2021 01:10:53 -0700 Subject: [PATCH 834/880] Preparing release notes --- docs/release_notes.rst | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 6d239aff..55f5d4b2 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,39 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v12.0.0 +======= + +**Breaking changes** + +- Due to recent security issues in pikepdf, Pillow and reportlab, we now require + newer versions of these libraries and some of their dependencies. (If necessary, + package maintainers may override these versions, and lower versions will often + work.) +- We now use the "LeaveColorUnchanged" color conversion strategy when directing + Ghostscript to create a PDF/A. Generally this is faster than performing a + color conversion, which is not always necessary. +- OCR text is now packaged in a Form XObject. This makes it easier to isolate + OCR from other document content. However, some poor implemented PDF text + extraction algorithms may fail to find the text. +- Many API functions have stricter parameter checking or expect keyword arguments + were they previously did not. +- Some deprecated functions in ``ocrmypdf.optimize`` were removed. +- The ``ocrmypdf.leptonica`` module is now deprecated. + +**New features** + +- New plugin hook: ``get_progressbar_class``, for progress reporting, + allowing developers to replace the standard console progress bar with some + other mechanism, such as updating a GUI progress bar. +- New plugin hook: ``get_executor``, for replacing the concurrency model. + This is primarily to support execution on AWS Lambda, which does not support + standard Python ``multiprocessing`` due to its lack of shared memory. +- New plugin hook: ``get_logging_console``, for replacing the standard + way OCRmyPDF outputs its messages. +- New plugin hook: ``filter_pdf_page``, for modifying individual PDF + pages produced by OCRmyPDF. + v11.7.3 ======= From 16438c131234841df22ca8b7acd6df1933ffbb5d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 2 Apr 2021 01:34:14 -0700 Subject: [PATCH 835/880] Begin GitHub Actions migration --- .github/workflows/build.yml | 227 ++++++++++++++++++++++++++++++++++++ .gitignore | 1 + azure-pipelines.yml | 1 + setup.cfg | 4 +- 4 files changed, 230 insertions(+), 3 deletions(-) create mode 100644 .github/workflows/build.yml diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml new file mode 100644 index 00000000..d94fbd7b --- /dev/null +++ b/.github/workflows/build.yml @@ -0,0 +1,227 @@ +name: Build and upload to PyPI + +on: + push: + branches: + - master + - ci + - release/* + paths-ignore: + - README* + pull_request: + +jobs: + test_linux: + name: Test ${{ matrix.os }} with Python ${{ matrix.python }} + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-18.04] #, ubuntu-20.04] + python: ["3.6"] #, "3.7", "3.8", "3.9"] + + env: + OS: ${{ matrix.os }} + PYTHON: ${{ matrix.python }} + + steps: + - uses: actions/checkout@v2 + with: + fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags + + - uses: actions/setup-python@v2 + name: Install Python + with: + python-version: ${{ matrix.python }} + + - name: Install common packages + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + curl \ + ghostscript \ + img2pdf \ + libffi-dev \ + liblept5 \ + libsm6 libxext6 libxrender-dev \ + pngquant \ + poppler-utils \ + tesseract-ocr \ + tesseract-ocr-deu \ + tesseract-ocr-eng \ + unpaper \ + zlib1g + + - name: Install Ubuntu 18.04 packages + if: matrix.os == 'ubuntu-18.04' + run: | + sudo apt-get install -y --no-install-recommends \ + libexempi3 + + - name: Install Ubuntu 20.04 packages + if: matrix.os == 'ubuntu-20.04' + run: | + sudo apt-get install -y --no-install-recommends \ + libexempi8 + + - name: Install Python packages + run: | + python -m pip install -r requirements/main.txt -r requirements/test.txt . + + - name: Report versions + run: | + tesseract --version + gs --version + pngquant --version + unpaper --version + img2pdf --version + + - name: Test + run: | + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + + # - name: Upload coverage to Codecov + # uses: codecov/codecov-action@v1 + # with: + # files: ./coverage.xml + # env_vars: OS,PYTHON + + test_macos: + name: Test macOS + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [macos-latest] + python: ["3.9"] + + env: + OS: ${{ matrix.os }} + PYTHON: ${{ matrix.python }} + + steps: + - uses: actions/checkout@v2 + with: + fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags + + - uses: actions/setup-python@v2 + name: Install Python + with: + python-version: ${{ matrix.python }} + + - name: Install Homebrew deps + run: | + brew update + brew install \ + exempi \ + ghostscript \ + jbig2enc \ + leptonica \ + openjpeg \ + pngquant \ + tesseract + + - name: Install Python packages + run: | + python -m pip install --upgrade pip + python -m pip install -r requirements/main.txt -r requirements/test.txt . + + - name: Report versions + run: | + tesseract --version + gs --version + pngquant --version + img2pdf --version + + - name: Test + run: | + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + + # - name: Upload coverage to Codecov + # uses: codecov/codecov-action@v1 + # with: + # files: ./coverage.xml + # env_vars: OS,PYTHON + + test_windows: + name: Test Windows + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [windows-latest] + python: ["3.9"] + + env: + OS: ${{ matrix.os }} + PYTHON: ${{ matrix.python }} + + steps: + - uses: actions/checkout@v2 + with: + fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags + + - uses: actions/setup-python@v2 + name: Install Python + with: + python-version: ${{ matrix.python }} + + - name: Install system packages + run: | + choco install --yes --no-progress --pre tesseract + choco install --yes --no-progress ghostscript + choco install --yes --no-progress pngquant + + - name: Install Python packages + run: | + python -m pip install --upgrade pip + python -m pip install -r requirements/main.txt -r requirements/test.txt . + + - name: Test + run: | + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + + # - name: Upload coverage to Codecov + # uses: codecov/codecov-action@v1 + # with: + # files: ./coverage.xml + # env_vars: OS,PYTHON + + wheel_sdist_linux: + name: Build sdist and wheels + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v2 + with: + fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags + + - uses: actions/setup-python@v2 + name: Install Python + with: + python-version: "3.6" + + - name: Make wheels and sdist + run: | + python -m pip install --upgrade pip wheel + python setup.py sdist + python setup.py bdist_wheel + + - uses: actions/upload-artifact@v2 + with: + path: | + ./dist/*.whl + ./dist/*.tar.gz + + upload_pypi: + name: Deploy artifacts to PyPI + needs: [wheel_sdist_linux, test_linux, test_macos, test_windows] + runs-on: ubuntu-latest + if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v') + steps: + - uses: actions/download-artifact@v2 + with: + name: artifact + path: dist + + - uses: pypa/gh-action-pypi-publish@master + with: + user: __token__ + password: ${{ secrets.TOKEN_PYPI }} + # repository_url: https://test.pypi.org/legacy/ diff --git a/.gitignore b/.gitignore index 7fc65e63..60de406b 100644 --- a/.gitignore +++ b/.gitignore @@ -24,6 +24,7 @@ venv*/ /debug_tests.py *.traineddata /private +/coverage.xml # Package building *.egg-info/ diff --git a/azure-pipelines.yml b/azure-pipelines.yml index db05746c..2f50df24 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -7,6 +7,7 @@ trigger: - "*" exclude: - "travis" + - "ci" stages: - stage: "Test" diff --git a/setup.cfg b/setup.cfg index 44ed90c5..36545d66 100644 --- a/setup.cfg +++ b/setup.cfg @@ -32,14 +32,12 @@ license_file = LICENSE [coverage:paths] source = - src/ + src/ocrmypdf [coverage:run] branch = true parallel = true concurrency = multiprocessing -source = - src/ocrmypdf [coverage:report] # Regexes for lines to exclude from consideration From af526f078d1d291f395447c1fb77b15159788267 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Apr 2021 00:08:47 -0700 Subject: [PATCH 836/880] Activate codecov --- .github/workflows/build.yml | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index d94fbd7b..cd4cf868 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -79,11 +79,11 @@ jobs: run: | python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick - # - name: Upload coverage to Codecov - # uses: codecov/codecov-action@v1 - # with: - # files: ./coverage.xml - # env_vars: OS,PYTHON + - name: Upload coverage to Codecov + uses: codecov/codecov-action@v1 + with: + files: ./coverage.xml + env_vars: OS,PYTHON test_macos: name: Test macOS @@ -135,11 +135,11 @@ jobs: run: | python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick - # - name: Upload coverage to Codecov - # uses: codecov/codecov-action@v1 - # with: - # files: ./coverage.xml - # env_vars: OS,PYTHON + - name: Upload coverage to Codecov + uses: codecov/codecov-action@v1 + with: + files: ./coverage.xml + env_vars: OS,PYTHON test_windows: name: Test Windows @@ -178,11 +178,11 @@ jobs: run: | python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick - # - name: Upload coverage to Codecov - # uses: codecov/codecov-action@v1 - # with: - # files: ./coverage.xml - # env_vars: OS,PYTHON + - name: Upload coverage to Codecov + uses: codecov/codecov-action@v1 + with: + files: ./coverage.xml + env_vars: OS,PYTHON wheel_sdist_linux: name: Build sdist and wheels From e1f4813d94a8b6b2f9583d80f195686c77b8eb41 Mon Sep 17 00:00:00 2001 From: andkrause Date: Mon, 5 Apr 2021 01:38:48 -0700 Subject: [PATCH 837/880] Dockerfile: support arm64 Thanks to @andkrause for submitting the initial PR that made this change possible, @0x326 for review comments on the same PR, and both for their patience in waiting for the rest of OCRmyPDF to catch up. Co-authored-by: James R. Barlow Closes #564 --- .docker/Dockerfile | 7 ++++-- .github/workflows/build.yml | 46 +++++++++++++++++++++++++++++++++++++ 2 files changed, 51 insertions(+), 2 deletions(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 2378adb3..89cc336f 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -10,8 +10,10 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential autoconf automake libtool \ libleptonica-dev \ zlib1g-dev \ - python3 \ + python3-dev \ python3-distutils \ + libffi-dev \ + libqpdf-dev \ ca-certificates \ curl \ git @@ -62,7 +64,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ tesseract-ocr-fra \ tesseract-ocr-por \ tesseract-ocr-spa \ - unpaper + unpaper \ + && rm -rf /var/lib/apt/lists/* WORKDIR /app diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index cd4cf868..e45fc839 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -225,3 +225,49 @@ jobs: user: __token__ password: ${{ secrets.TOKEN_PYPI }} # repository_url: https://test.pypi.org/legacy/ + + docker: + name: Build Docker images + needs: [wheel_sdist_linux, test_linux, test_macos, test_windows] + runs-on: ubuntu-latest + steps: + - name: Set image tag to release or branch + run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV + + - name: If master, set to latest + run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV + if: env.DOCKER_IMAGE_TAG == 'master' + + - name: Set Docker Hub repository to username + run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV + + - name: Set image name + run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV + + - uses: actions/checkout@v2 + with: + fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags + + - name: Login to Docker Hub + uses: docker/login-action@v1 + with: + username: jbarlow83 + password: ${{ secrets.DOCKERHUB_TOKEN }} + + - name: Set up QEMU + uses: docker/setup-qemu-action@v1 + + - name: Set up Docker Buildx + id: buildx + uses: docker/setup-buildx-action@v1 + + - name: Print image tag + run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" + + - name: Build + run: | + docker buildx build \ + --push \ + --platform linux/arm64/v8,linux/amd64 \ + --tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \ + --file .docker/Dockerfile . From a25f8ecc6294c3df333fd85aa2ae4ced393ecc46 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Apr 2021 00:17:29 -0700 Subject: [PATCH 838/880] Remove Azure Pipelines --- .github/workflows/build.yml | 2 +- README.md | 2 +- azure-pipelines.yml | 253 ------------------------------------ docs/installation.rst | 2 +- 4 files changed, 3 insertions(+), 256 deletions(-) delete mode 100644 azure-pipelines.yml diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index e45fc839..9329c503 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -1,4 +1,4 @@ -name: Build and upload to PyPI +name: Test and deploy on: push: diff --git a/README.md b/README.md index 0c3b20cc..a0bf0a35 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ OCRmyPDF -[![Build Status][azure]](https://dev.azure.com/jim0585/ocrmypdf/_build/latest?definitionId=2&branchName=master) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] +[![Build Status](https://github.com/jbarlow83/OCRmyPDF/actions/workflows/build.yml/badge.svg)](https://github.com/jbarlow83/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions] [azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master [travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status" diff --git a/azure-pipelines.yml b/azure-pipelines.yml deleted file mode 100644 index 2f50df24..00000000 --- a/azure-pipelines.yml +++ /dev/null @@ -1,253 +0,0 @@ -trigger: - tags: - include: - - v* - branches: - include: - - "*" - exclude: - - "travis" - - "ci" - -stages: - - stage: "Test" - jobs: - - job: Windows - pool: - vmImage: "vs2017-win2016" - strategy: - matrix: - Python36: - python.version: "3.6" - Python37: - python.version: "3.7" - Python38: - python.version: "3.8" - Python39: - python.version: "3.9" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" - - pwsh: | - choco install --yes --no-progress --pre tesseract - choco install --yes --no-progress python3 - choco install --yes --no-progress ghostscript - choco install --yes --no-progress pngquant - displayName: "Install system packages" - - pwsh: | - refreshenv - python -m pip install --upgrade pip wheel - python -m pip install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - pwsh: | - refreshenv - $env:pathext += ';.py' - # -n auto helps Windows - python -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() - - job: "Ubuntu_1804" - pool: - vmImage: "ubuntu-18.04" - strategy: - matrix: - Python36: - python.version: "3.6" - Python37: - python.version: "3.7" - Python38: - python.version: "3.8" - Python39: - python.version: "3.9" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" - - bash: | - sudo apt-get update - sudo apt-get install -y --no-install-recommends \ - python3-software-properties \ - curl \ - ghostscript \ - img2pdf \ - libexempi3 \ - libffi-dev \ - liblept5 \ - libsm6 libxext6 libxrender-dev \ - pngquant \ - poppler-utils \ - tesseract-ocr \ - tesseract-ocr-deu \ - tesseract-ocr-eng \ - unpaper \ - zlib1g - displayName: "Install system packages" - - bash: | - curl https://bootstrap.pypa.io/get-pip.py | python3 - pip3 install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - bash: | - tesseract --version - displayName: "Record versions" - - bash: | - # -n auto is slower on Linux and breaks on Python 3.8 - pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() - - job: "Ubuntu_1604" - pool: - vmImage: "ubuntu-16.04" - strategy: - matrix: - Python36: - python.version: "3.6" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "$(python.version)" - - bash: | - sudo apt-get update - sudo apt-get install -y --no-install-recommends \ - software-properties-common - sudo add-apt-repository -y ppa:alex-p/tesseract-ocr - sudo apt-get update - sudo apt-get install -y --no-install-recommends \ - ghostscript \ - img2pdf \ - libexempi3 \ - libffi-dev \ - liblept5 \ - libsm6 libxext6 libxrender-dev \ - pngquant \ - poppler-utils \ - tesseract-ocr \ - tesseract-ocr-deu \ - tesseract-ocr-eng \ - unpaper \ - zlib1g - displayName: "Install system packages" - - bash: | - curl https://bootstrap.pypa.io/get-pip.py | python3 - pip3 install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - bash: | - tesseract --version - displayName: "Record versions" - - bash: | - # -n auto is slower on Linux and breaks on Python 3.8 - pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() - - job: "macOS_Mojave" - pool: - vmImage: "macos-10.14" - steps: - # https://github.com/actions/virtual-environments/issues/664 - # - task: UsePythonVersion@0 - # inputs: - # versionSpec: "$(python.version)" - - bash: | - brew update - brew upgrade python - echo "Using `python3 --version`" - displayName: "Update brew and Python" - - bash: | - brew install \ - exempi \ - ghostscript \ - jbig2enc \ - leptonica \ - openjpeg \ - pngquant \ - tesseract - displayName: "Install system packages" - - bash: | - pip3 install --upgrade pip - pip3 install -r requirements/main.txt -r requirements/test.txt . - displayName: "Install Python packages" - - bash: | - tesseract --version - displayName: "Record versions" - - bash: pytest -nauto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml - displayName: "Test" - - task: PublishTestResults@2 - inputs: - testResultsFiles: "test.xml" - testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)" - condition: succeededOrFailed() - - task: PublishCodeCoverageResults@1 - inputs: - codeCoverageTool: Cobertura - summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml" - - - stage: "Artifacts" - jobs: - - job: "sdist_wheel" - pool: - vmImage: "ubuntu-18.04" - steps: - - task: UsePythonVersion@0 - inputs: - versionSpec: "3.7" - - bash: | - python -m pip install --upgrade pip wheel - python setup.py sdist bdist_wheel - - publish: dist - artifact: sdist_wheel - - - stage: "Deploy" - jobs: - - deployment: "PyPI" - pool: - vmImage: "ubuntu-18.04" - environment: "deploy" - strategy: - runOnce: - deploy: - steps: - - download: current - artifact: sdist_wheel - - script: | - mkdir -p dist - mv $(Pipeline.Workspace)/sdist_wheel/* dist - displayName: "Move dist files" - - task: UsePythonVersion@0 - inputs: - versionSpec: "3.8" - architecture: x64 - - script: | - pip install --upgrade twine - displayName: "Generate artifacts" - - script: | - cat <.pypirc - [distutils] - index-servers = - pypi - - [pypi] - username: __token__ - password: $(TOKEN_PYPI) - - FILE - displayName: "Generate PyPI auth file" - - script: | - python -m twine upload --config-file .pypirc dist/* - displayName: "Upload to PyPI" - condition: and(succeeded(), startsWith(variables['Build.SourceBranch'], 'refs/tags/')) - - script: | - curl -X POST -d "token=$(TOKEN_RTD)" https://readthedocs.org/api/v2/webhook/pikepdf/39557/ - displayName: "Trigger ReadTheDocs" - condition: and(succeeded(), or(startsWith(variables['Build.SourceBranch'], 'refs/tags/'), startsWith(variables['Build.SourceBranch'], 'refs/heads/master'))) diff --git a/docs/installation.rst b/docs/installation.rst index f3c952fc..6d096996 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -464,7 +464,7 @@ Update Homebrew: Install or upgrade the required Homebrew packages, if any are missing. To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew -dependencies. You could also check the ``azure-pipelines.yml``. +dependencies. You could also check the ``.workflows/build.yml``. This will include the English, French, German and Spanish language packs. If you need other languages you can optionally install them all: From 6c942ecefd9213c7c17a73a571ed8a237643aff2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Apr 2021 00:17:36 -0700 Subject: [PATCH 839/880] Update release notes --- docs/release_notes.rst | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 55f5d4b2..42163ae7 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -19,8 +19,8 @@ v12.0.0 - Due to recent security issues in pikepdf, Pillow and reportlab, we now require newer versions of these libraries and some of their dependencies. (If necessary, - package maintainers may override these versions, and lower versions will often - work.) + package maintainers may override these versions at their discretion; lower + versions will often work.) - We now use the "LeaveColorUnchanged" color conversion strategy when directing Ghostscript to create a PDF/A. Generally this is faster than performing a color conversion, which is not always necessary. @@ -31,6 +31,7 @@ v12.0.0 were they previously did not. - Some deprecated functions in ``ocrmypdf.optimize`` were removed. - The ``ocrmypdf.leptonica`` module is now deprecated. +- Continuous integration moved to GitHub Actions. **New features** @@ -44,6 +45,8 @@ v12.0.0 way OCRmyPDF outputs its messages. - New plugin hook: ``filter_pdf_page``, for modifying individual PDF pages produced by OCRmyPDF. +- We now generate an ARM64-compatible Docker image alongside the x64 image. + Thanks to @andkrause for contributing the change and @0x326 for review comments. v11.7.3 ======= From 3a6eb383dc356a027f537aff68ed07ed55385b15 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Apr 2021 00:28:06 -0700 Subject: [PATCH 840/880] Run all tests --- .github/workflows/build.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 9329c503..1a096821 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -77,7 +77,7 @@ jobs: - name: Test run: | - python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/ - name: Upload coverage to Codecov uses: codecov/codecov-action@v1 @@ -133,7 +133,7 @@ jobs: - name: Test run: | - python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/ - name: Upload coverage to Codecov uses: codecov/codecov-action@v1 @@ -176,7 +176,7 @@ jobs: - name: Test run: | - python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/test_main.py::test_quick + python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/ - name: Upload coverage to Codecov uses: codecov/codecov-action@v1 From b1306bd7a8b3fd9903e026a911e44cd201e69943 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 6 Apr 2021 01:15:00 -0700 Subject: [PATCH 841/880] tests: skip test_bash on Windows --- tests/test_completion.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tests/test_completion.py b/tests/test_completion.py index fa7cde15..4ac753b8 100644 --- a/tests/test_completion.py +++ b/tests/test_completion.py @@ -5,6 +5,7 @@ # file, You can obtain one at http://mozilla.org/MPL/2.0/. +import os from subprocess import PIPE, run import pytest @@ -29,6 +30,9 @@ def test_fish(): pytest.xfail('fish is not installed') +@pytest.mark.skipif( + os.name == 'nt', reason="Windows CI workers have bash but are best left alone" +) def test_bash(): try: proc = run( From e0441c4aa1a430dbeaa9199d929967a4119494bd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tom=C3=A1=C5=A1=20Hrn=C4=8Diar?= Date: Tue, 6 Apr 2021 11:23:44 +0200 Subject: [PATCH 842/880] Explicitly require setuptools, since pdfa.py imports pkg_resources (#755) Co-authored-by: jbarlow83 --- setup.py | 1 + 1 file changed, 1 insertion(+) diff --git a/setup.py b/setup.py index 98d395ec..e2342651 100644 --- a/setup.py +++ b/setup.py @@ -73,6 +73,7 @@ setup( 'Pillow >= 8.1.2', 'pluggy >= 0.13.0, < 1.0', 'reportlab >= 3.5.66', + 'setuptools', 'tqdm >= 4', ], tests_require=tests_require, From aa115a8be37f7d62773a7ca23cd22ca5329189f7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 01:56:51 -0700 Subject: [PATCH 843/880] Remove pytest_helpers_namespace --- docs/release_notes.rst | 1 + requirements/test.txt | 1 - src/ocrmypdf/RELEASE.md | 35 +++++++++++++++++++++++++++++++++++ tests/__init__.py | 7 +++++++ tests/conftest.py | 12 ------------ tests/test_acroform.py | 2 +- tests/test_completion.py | 4 +++- tests/test_ghostscript.py | 4 +--- tests/test_helpers.py | 6 +++--- tests/test_hocrtransform.py | 4 ++-- tests/test_image_input.py | 3 +-- tests/test_main.py | 31 ++++++++++++++++--------------- tests/test_metadata.py | 7 ++----- tests/test_optimize.py | 2 +- tests/test_pdfa.py | 2 +- tests/test_preprocessing.py | 10 ++-------- tests/test_rotation.py | 10 ++++------ tests/test_stdio.py | 6 +----- tests/test_tesseract.py | 5 ++--- tests/test_unpaper.py | 8 ++------ tests/test_userunit.py | 4 +--- tests/test_validation.py | 2 +- 22 files changed, 87 insertions(+), 79 deletions(-) create mode 100644 src/ocrmypdf/RELEASE.md create mode 100644 tests/__init__.py diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 42163ae7..4c1c8646 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -32,6 +32,7 @@ v12.0.0 - Some deprecated functions in ``ocrmypdf.optimize`` were removed. - The ``ocrmypdf.leptonica`` module is now deprecated. - Continuous integration moved to GitHub Actions. +- We no longer depend on ``pytest_helpers_namespace`` for testing. **New features** diff --git a/requirements/test.txt b/requirements/test.txt index fea870d7..72225e2d 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,5 +1,4 @@ pytest >= 6.0.0 -pytest-helpers-namespace >= 2019.1.8 pytest-xdist >= 2.2.0 pytest-cov >= 2.11.1 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 diff --git a/src/ocrmypdf/RELEASE.md b/src/ocrmypdf/RELEASE.md new file mode 100644 index 00000000..41a40e97 --- /dev/null +++ b/src/ocrmypdf/RELEASE.md @@ -0,0 +1,35 @@ +# Release checklist + +## Patch release + +- Check `pytest` + +- Update release notes + +## Minor release + +## Major release + +- Run `pre-commit autoupdate` + +- Check README.md + +- Check setup.py + + - Are classifiers up to date? + - Is `python_requires` correct? + - Python 3.6 is EOL on December 2021-12. Could drop support then. + - Can we tighten any `install_requires` dependencies? + +- Search for old version shims we can remove + + - "shim" + - ` pikepdf.__version__` + +- Search for deprecation: search all files for deprec*, etc. + +- Check requirements/* + +- Delete `tests/cache`, do `pytest --runslow`, and update cache. + +- Do `pytest --cov-report html` diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 00000000..f714d74d --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1,7 @@ +# © 2021 James R. Barlow: github.com/jbarlow83 +# +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at http://mozilla.org/MPL/2.0/. + +# Empty __init__.py file diff --git a/tests/conftest.py b/tests/conftest.py index 70741548..d3b3ebb7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -20,37 +20,29 @@ from ocrmypdf._plugin_manager import get_parser_options_plugins pytest_plugins = ['helpers_namespace'] -# pylint: disable=E1101 -# pytest.helpers is dynamic so it confuses pylint - if sys.version_info < (3, 5): print("Requires Python 3.5+") sys.exit(1) -@pytest.helpers.register def is_linux(): return platform.system() == 'Linux' -@pytest.helpers.register def is_macos(): return platform.system() == 'Darwin' -@pytest.helpers.register def running_in_docker(): # Docker creates a file named /.dockerenv (newer versions) or # /.dockerinit (older) -- this is undocumented, not an offical test return Path('/.dockerenv').exists() or Path('/.dockerinit').exists() -@pytest.helpers.register def running_in_travis(): return os.environ.get('TRAVIS') == 'true' -@pytest.helpers.register def have_unpaper(): try: unpaper.version() @@ -93,7 +85,6 @@ def no_outpdf(tmp_path): return tmp_path / 'no_output.pdf' -@pytest.helpers.register def check_ocrmypdf(input_file, output_file, *args): """Run ocrmypdf and confirmed that a valid file was created""" args = [str(input_file), str(output_file)] + [ @@ -111,7 +102,6 @@ def check_ocrmypdf(input_file, output_file, *args): return output_file -@pytest.helpers.register def run_ocrmypdf_api(input_file, output_file, *args): """Run ocrmypdf via API and let caller deal with results @@ -127,7 +117,6 @@ def run_ocrmypdf_api(input_file, output_file, *args): return api.run_pipeline(options, plugin_manager=None, api=False) -@pytest.helpers.register def run_ocrmypdf(input_file, output_file, *args, text=True): "Run ocrmypdf and let caller deal with results" @@ -150,7 +139,6 @@ def run_ocrmypdf(input_file, output_file, *args, text=True): return p, p.stdout, p.stderr -@pytest.helpers.register def first_page_dimensions(pdf): info = pdfinfo.PdfInfo(pdf) page0 = info[0] diff --git a/tests/test_acroform.py b/tests/test_acroform.py index f6932ebe..910bacc8 100644 --- a/tests/test_acroform.py +++ b/tests/test_acroform.py @@ -11,7 +11,7 @@ import pytest import ocrmypdf -check_ocrmypdf = pytest.helpers.check_ocrmypdf +from .conftest import check_ocrmypdf @pytest.fixture diff --git a/tests/test_completion.py b/tests/test_completion.py index 4ac753b8..20d716ff 100644 --- a/tests/test_completion.py +++ b/tests/test_completion.py @@ -10,8 +10,10 @@ from subprocess import PIPE, run import pytest +from .conftest import running_in_docker + pytestmark = pytest.mark.skipif( - pytest.helpers.running_in_docker(), # pylint: disable=no-member + running_in_docker(), reason="docker can't complete", ) diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 37a1a1e9..4995fb94 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -16,9 +16,7 @@ from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.exceptions import ExitCode from ocrmypdf.helpers import Resolution -check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member -run_ocrmypdf = pytest.helpers.run_ocrmypdf # pylint: disable=no-member -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api # pylint: disable=no-member +from .conftest import check_ocrmypdf, run_ocrmypdf, run_ocrmypdf_api # pylint: disable=redefined-outer-name diff --git a/tests/test_helpers.py b/tests/test_helpers.py index f0a5e4d0..7b885221 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -14,6 +14,8 @@ import pytest from ocrmypdf import helpers as helpers +from .conftest import running_in_docker + class TestSafeSymlink: def test_safe_symlink_link_self(self, tmp_path, caplog): @@ -58,9 +60,7 @@ def test_deprecated(): assert old_function() == 42 -skipif_docker = pytest.mark.skipif( - pytest.helpers.running_in_docker(), reason="fails on Docker" -) +skipif_docker = pytest.mark.skipif(running_in_docker(), reason="fails on Docker") class TestFileIsWritable: diff --git a/tests/test_hocrtransform.py b/tests/test_hocrtransform.py index 136c9a49..b0046e94 100644 --- a/tests/test_hocrtransform.py +++ b/tests/test_hocrtransform.py @@ -21,6 +21,8 @@ from ocrmypdf import hocrtransform from ocrmypdf._exec.tesseract import HOCR_TEMPLATE from ocrmypdf.helpers import check_pdf +from .conftest import check_ocrmypdf + def text_from_pdf(filename): output_string = StringIO() @@ -37,8 +39,6 @@ def text_from_pdf(filename): # pylint: disable=redefined-outer-name -check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member - @pytest.fixture def blank_hocr(tmp_path): diff --git a/tests/test_image_input.py b/tests/test_image_input.py index 5efad8c0..ad738f84 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -14,8 +14,7 @@ from PIL import Image import ocrmypdf -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +from .conftest import check_ocrmypdf, run_ocrmypdf_api @pytest.fixture diff --git a/tests/test_main.py b/tests/test_main.py index 33d45fba..4f5aedcf 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -24,12 +24,18 @@ from ocrmypdf.pdfa import file_claims_pdfa from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo from ocrmypdf.subprocess import get_version -# pytest.helpers is dynamic -# pylint: disable=no-member,redefined-outer-name +from .conftest import ( + check_ocrmypdf, + first_page_dimensions, + have_unpaper, + is_macos, + run_ocrmypdf, + run_ocrmypdf_api, + running_in_docker, + running_in_travis, +) -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +# pylint: disable=redefined-outer-name RENDERERS = ['hocr', 'sandwich'] @@ -155,7 +161,7 @@ def test_maximum_options(renderer, output_type, resources, outpdf): resources / 'multipage.pdf', outpdf, '-d', - '-ci' if pytest.helpers.have_unpaper() else None, + '-ci' if have_unpaper() else None, '-f', '-k', '--oversample', @@ -208,7 +214,7 @@ def test_force_ocr_on_pdf_with_no_images(resources, no_outpdf): @pytest.mark.skipif( - pytest.helpers.is_macos() and pytest.helpers.running_in_travis(), + is_macos() and running_in_travis(), reason="takes too long to install language packs in Travis macOS homebrew", ) def test_german(resources, outdir): @@ -269,9 +275,7 @@ def test_input_file_not_found(caplog, no_outpdf): assert input_file in caplog.text -@pytest.mark.skipif( - os.name == 'nt' or pytest.helpers.running_in_docker(), reason="chmod" -) +@pytest.mark.skipif(os.name == 'nt' or running_in_docker(), reason="chmod") def test_input_file_not_readable(caplog, resources, outdir, no_outpdf): input_file = outdir / 'trivial.pdf' shutil.copy(resources / 'trivial.pdf', input_file) @@ -519,9 +523,6 @@ def test_form_xobject(resources, outpdf): @pytest.mark.parametrize('renderer', RENDERERS) def test_pagesize_consistency(renderer, resources, outpdf): - - first_page_dimensions = pytest.helpers.first_page_dimensions - infile = resources / '3small.pdf' before_dims = first_page_dimensions(infile) @@ -531,10 +532,10 @@ def test_pagesize_consistency(renderer, resources, outpdf): outpdf, '--pdf-renderer', renderer, - '--clean' if pytest.helpers.have_unpaper() else None, + '--clean' if have_unpaper() else None, '--deskew', '--remove-background', - '--clean-final' if pytest.helpers.have_unpaper() else None, + '--clean-final' if have_unpaper() else None, '--pages', '1', ) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index e6496d36..87df840a 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -24,19 +24,16 @@ from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfa import SRGB_ICC_PROFILE, file_claims_pdfa, generate_pdfa_ps from ocrmypdf.pdfinfo import PdfInfo +from .conftest import check_ocrmypdf, run_ocrmypdf + try: import fitz except ImportError: fitz = None -# pytest.helpers is dynamic -# pylint: disable=no-member pytestmark = pytest.mark.filterwarnings('ignore:.*XMLParser.*:DeprecationWarning') -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf - @pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) def test_preserve_docinfo(output_type, resources, outpdf): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index a319b812..96cb803b 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -19,7 +19,7 @@ from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.helpers import Resolution -check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101 +from .conftest import check_ocrmypdf needs_pngquant = pytest.mark.skipif( not pngquant.available(), reason="pngquant not installed" diff --git a/tests/test_pdfa.py b/tests/test_pdfa.py index d0c269ff..75b22a57 100644 --- a/tests/test_pdfa.py +++ b/tests/test_pdfa.py @@ -7,7 +7,7 @@ import pikepdf import pytest -check_ocrmypdf = pytest.helpers.check_ocrmypdf +from .conftest import check_ocrmypdf @pytest.mark.parametrize('optimize', (0, 3)) diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 2fc81c10..593505e8 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -15,13 +15,7 @@ from ocrmypdf.helpers import Resolution from ocrmypdf.leptonica import Pix from ocrmypdf.pdfinfo import PdfInfo -# pytest.helpers is dynamic -# pylint: disable=no-member,redefined-outer-name - -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api - +from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf, run_ocrmypdf_api RENDERERS = ['hocr', 'sandwich'] @@ -96,7 +90,7 @@ def test_exotic_image(pdf, renderer, output_type, resources, outdir): check_ocrmypdf( resources / pdf, outfile, - '-dc' if pytest.helpers.have_unpaper() else '-d', + '-dc' if have_unpaper() else '-d', '-v', '1', '--output-type', diff --git a/tests/test_rotation.py b/tests/test_rotation.py index b826a09e..cb7fcb7b 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -21,18 +21,16 @@ from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import PdfInfo -# pytest.helpers is dynamic -# pylint: disable=no-member -# pylint: disable=w0612 +from .conftest import check_ocrmypdf, run_ocrmypdf + +# pylintx: disable=unused-variable + pytestmark = pytest.mark.skipif( leptonica.get_leptonica_version() < 'leptonica-1.72', reason="Leptonica is too old, correlation doesn't work", ) -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf - RENDERERS = ['hocr', 'sandwich'] diff --git a/tests/test_stdio.py b/tests/test_stdio.py index cfd7f7b3..2e43cff5 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -15,11 +15,7 @@ import pytest from ocrmypdf.exceptions import ExitCode from ocrmypdf.helpers import check_pdf -# pytest.helpers is dynamic -# pylint: disable=no-member,redefined-outer-name - -run_ocrmypdf = pytest.helpers.run_ocrmypdf -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf +from .conftest import run_ocrmypdf def test_stdin(ocrmypdf_exec, resources, outpdf): diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index f7331e48..d4d1a521 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -17,10 +17,9 @@ from ocrmypdf import pdfinfo from ocrmypdf._exec import tesseract from ocrmypdf.exceptions import MissingDependencyError -# pylint: disable=no-member,redefined-outer-name +from .conftest import check_ocrmypdf, run_ocrmypdf -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf +# pylint: disable=redefined-outer-name @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index eefe9852..72a32c5a 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -14,13 +14,9 @@ from ocrmypdf._plugin_manager import get_parser_options_plugins from ocrmypdf._validation import check_options from ocrmypdf.exceptions import ExitCode, MissingDependencyError -# pytest.helpers is dynamic -# pylint: disable=no-member,redefined-outer-name -# pylint: disable=w0612 +from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf -check_ocrmypdf = pytest.helpers.check_ocrmypdf -run_ocrmypdf = pytest.helpers.run_ocrmypdf -have_unpaper = pytest.helpers.have_unpaper +# pylint: disable=redefined-outer-name def test_no_unpaper(resources, no_outpdf): diff --git a/tests/test_userunit.py b/tests/test_userunit.py index 6926d160..e3b9fe31 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -12,9 +12,7 @@ import pytest from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfinfo import PdfInfo -check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=no-member -run_ocrmypdf = pytest.helpers.run_ocrmypdf # pylint: disable=no-member -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api # pylint: disable=no-member +from .conftest import check_ocrmypdf, run_ocrmypdf, run_ocrmypdf_api # pylint: disable=redefined-outer-name diff --git a/tests/test_validation.py b/tests/test_validation.py index 7b2f1ead..81a4e5d5 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -19,7 +19,7 @@ from ocrmypdf.cli import get_parser from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.pdfinfo import PdfInfo -run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api +from .conftest import run_ocrmypdf_api def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): From 173a80864d4825576fbc11ce2375f6505a1a9c8f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 02:09:45 -0700 Subject: [PATCH 844/880] Delinting --- src/ocrmypdf/_concurrent.py | 4 +--- src/ocrmypdf/_graft.py | 2 +- src/ocrmypdf/_pipeline.py | 12 ++++++------ src/ocrmypdf/_sync.py | 2 +- src/ocrmypdf/builtin_plugins/concurrency.py | 2 +- src/ocrmypdf/builtin_plugins/default_filters.py | 4 +++- src/ocrmypdf/pluginspec.py | 5 ++++- tests/test_acroform.py | 2 ++ tests/test_ghostscript.py | 2 +- tests/test_helpers.py | 3 ++- tests/test_image_input.py | 2 ++ tests/test_optimize.py | 2 +- tests/test_preprocessing.py | 2 +- tests/test_rotation.py | 8 ++++---- tests/test_tesseract.py | 2 +- tests/test_unpaper.py | 4 ++-- tests/test_userunit.py | 2 +- tests/test_validation.py | 4 ++-- 18 files changed, 36 insertions(+), 28 deletions(-) diff --git a/src/ocrmypdf/_concurrent.py b/src/ocrmypdf/_concurrent.py index 5882d962..af505eb3 100644 --- a/src/ocrmypdf/_concurrent.py +++ b/src/ocrmypdf/_concurrent.py @@ -4,10 +4,8 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at http://mozilla.org/MPL/2.0/. -import sys import threading from abc import ABC, abstractmethod -from functools import partial from typing import Callable, Iterable, Optional @@ -128,7 +126,7 @@ class SerialExecutor(Executor): task: Callable, task_arguments: Iterable, task_finished: Callable, - ): + ): # pylint: disable=unused-argument with self.pbar_class(**tqdm_kwargs) as pbar: for args in task_arguments: result = task(args) diff --git a/src/ocrmypdf/_graft.py b/src/ocrmypdf/_graft.py index 262ec685..ba4b050f 100644 --- a/src/ocrmypdf/_graft.py +++ b/src/ocrmypdf/_graft.py @@ -20,7 +20,7 @@ MAX_REPLACE_PAGES = 100 def _ensure_dictionary(obj, name): if name not in obj: - obj[name] = pikepdf.Dictionary({}) + obj[name] = Dictionary({}) return obj[name] diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index fd8d3ac1..1ee0c104 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -13,7 +13,7 @@ from contextlib import suppress from datetime import datetime, timezone from pathlib import Path from shutil import copyfileobj -from typing import BinaryIO, Dict, Iterable, Optional, Union, cast +from typing import Dict, Iterable, Optional import img2pdf import pikepdf @@ -162,10 +162,10 @@ def get_pdfinfo( check_pages=check_pages, executor=executor, ) - except pikepdf.PasswordError: - raise EncryptedPdfError() - except pikepdf.PdfError: - raise InputFileError() + except pikepdf.PasswordError as e: + raise EncryptedPdfError() from e + except pikepdf.PdfError as e: + raise InputFileError() from e def validate_pdfinfo_options(context: PdfContext): @@ -839,7 +839,7 @@ def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor): def enumerate_compress_ranges(iterable): - skipped_from = None + skipped_from, index = None, None for index, txt_file in enumerate(iterable): index += 1 if txt_file: diff --git a/src/ocrmypdf/_sync.py b/src/ocrmypdf/_sync.py index 60e9192f..d9432cf4 100644 --- a/src/ocrmypdf/_sync.py +++ b/src/ocrmypdf/_sync.py @@ -65,7 +65,7 @@ from ocrmypdf.pdfa import file_claims_pdfa log = logging.getLogger(__name__) -class PageResult(NamedTuple): +class PageResult(NamedTuple): # pylint: disable=inherit-non-class pageno: int pdf_page_from_image: Optional[Path] ocr: Optional[Path] diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index b58211e4..16005984 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -22,7 +22,7 @@ import threading from contextlib import suppress from multiprocessing import Pool as ProcessPool from multiprocessing.pool import ThreadPool -from typing import Callable, Iterable, Optional, Union +from typing import Callable, Iterable, Union from tqdm import tqdm diff --git a/src/ocrmypdf/builtin_plugins/default_filters.py b/src/ocrmypdf/builtin_plugins/default_filters.py index 83a8f3fc..da3f3aae 100644 --- a/src/ocrmypdf/builtin_plugins/default_filters.py +++ b/src/ocrmypdf/builtin_plugins/default_filters.py @@ -8,5 +8,7 @@ from ocrmypdf import hookimpl @hookimpl -def filter_pdf_page(page, image_filename, output_pdf): +def filter_pdf_page( + page, image_filename, output_pdf +): # pylint: disable=unused-argument return output_pdf diff --git a/src/ocrmypdf/pluginspec.py b/src/ocrmypdf/pluginspec.py index 5fa0b3ca..47de64b6 100644 --- a/src/ocrmypdf/pluginspec.py +++ b/src/ocrmypdf/pluginspec.py @@ -10,7 +10,7 @@ from argparse import ArgumentParser, Namespace from collections import namedtuple from logging import Handler from pathlib import Path -from typing import TYPE_CHECKING, AbstractSet, Callable, Iterable, List, Optional +from typing import TYPE_CHECKING, AbstractSet, List, Optional import pluggy @@ -20,9 +20,12 @@ from ocrmypdf.helpers import Resolution if TYPE_CHECKING: from PIL import Image + # pylint: disable=ungrouped-imports from ocrmypdf._jobcontext import PageContext from ocrmypdf.pdfinfo import PdfInfo + # pylint: enable=ungrouped-imports + hookspec = pluggy.HookspecMarker('ocrmypdf') # pylint: disable=unused-argument diff --git a/tests/test_acroform.py b/tests/test_acroform.py index 910bacc8..82fe8d1f 100644 --- a/tests/test_acroform.py +++ b/tests/test_acroform.py @@ -13,6 +13,8 @@ import ocrmypdf from .conftest import check_ocrmypdf +# pylint: disable=redefined-outer-name + @pytest.fixture def acroform(resources): diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py index 4995fb94..0907b819 100644 --- a/tests/test_ghostscript.py +++ b/tests/test_ghostscript.py @@ -16,7 +16,7 @@ from ocrmypdf._exec.ghostscript import rasterize_pdf from ocrmypdf.exceptions import ExitCode from ocrmypdf.helpers import Resolution -from .conftest import check_ocrmypdf, run_ocrmypdf, run_ocrmypdf_api +from .conftest import check_ocrmypdf, run_ocrmypdf # pylint: disable=redefined-outer-name diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 7b885221..3a6c0b5a 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -12,7 +12,7 @@ from unittest.mock import MagicMock import pytest -from ocrmypdf import helpers as helpers +from ocrmypdf import helpers from .conftest import running_in_docker @@ -100,6 +100,7 @@ class TestFileIsWritable: @pytest.mark.skipif(os.name != 'nt', reason="Windows test") def test_shim_paths(tmp_path): + # pylint: disable=import-outside-toplevel from ocrmypdf.subprocess._windows import shim_env_path progfiles = tmp_path / 'Program Files' diff --git a/tests/test_image_input.py b/tests/test_image_input.py index ad738f84..78ad5fd1 100644 --- a/tests/test_image_input.py +++ b/tests/test_image_input.py @@ -16,6 +16,8 @@ import ocrmypdf from .conftest import check_ocrmypdf, run_ocrmypdf_api +# pylint: disable=redefined-outer-name + @pytest.fixture def baiona(resources): diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 96cb803b..3c5b3b38 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -143,7 +143,7 @@ def test_multiple_pngs(resources, outdir): outputstream=inpdf, ) - def mockquant(input_file, output_file, *args): + def mockquant(input_file, output_file, *_args): with Image.open(input_file) as im: draw = ImageDraw.Draw(im) draw.rectangle((0, 0, im.width, im.height), fill=128) diff --git a/tests/test_preprocessing.py b/tests/test_preprocessing.py index 593505e8..fa6bc687 100644 --- a/tests/test_preprocessing.py +++ b/tests/test_preprocessing.py @@ -15,7 +15,7 @@ from ocrmypdf.helpers import Resolution from ocrmypdf.leptonica import Pix from ocrmypdf.pdfinfo import PdfInfo -from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf, run_ocrmypdf_api +from .conftest import check_ocrmypdf, have_unpaper RENDERERS = ['hocr', 'sandwich'] diff --git a/tests/test_rotation.py b/tests/test_rotation.py index cb7fcb7b..3639eec7 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -89,7 +89,7 @@ def test_monochrome_correlation(resources, outdir): def test_autorotate(renderer, resources, outdir): # cardinal.pdf contains four copies of an image rotated in each cardinal # direction - these ones are "burned in" not tagged with /Rotate - out = check_ocrmypdf( + check_ocrmypdf( resources / 'cardinal.pdf', outdir / 'out.pdf', '-r', @@ -119,7 +119,7 @@ def test_autorotate(renderer, resources, outdir): ], ) def test_autorotate_threshold(threshold, correlation_test, resources, outdir): - out = check_ocrmypdf( + check_ocrmypdf( resources / 'cardinal.pdf', outdir / 'out.pdf', '--rotate-pages-threshold', @@ -131,14 +131,14 @@ def test_autorotate_threshold(threshold, correlation_test, resources, outdir): 'tests/plugins/tesseract_cache.py', ) - correlation = check_monochrome_correlation( + correlation = check_monochrome_correlation( # pylint: disable=unused-variable outdir, reference_pdf=resources / 'cardinal.pdf', reference_pageno=1, test_pdf=outdir / 'out.pdf', test_pageno=3, ) - assert eval(correlation_test) # pylint: disable=w0123 + assert eval(correlation_test) # pylint: disable=eval-used def test_rotated_skew_timeout(resources, outpdf): diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index d4d1a521..f9223c48 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -17,7 +17,7 @@ from ocrmypdf import pdfinfo from ocrmypdf._exec import tesseract from ocrmypdf.exceptions import MissingDependencyError -from .conftest import check_ocrmypdf, run_ocrmypdf +from .conftest import check_ocrmypdf # pylint: disable=redefined-outer-name diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 72a32c5a..de5a7f1f 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -71,7 +71,7 @@ def test_unpaper_args_valid(resources, outpdf): @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") def test_unpaper_args_invalid_filename(resources, outpdf): - p, out, err = run_ocrmypdf( + p, _out, err = run_ocrmypdf( resources / "skew.pdf", outpdf, "-c", @@ -86,7 +86,7 @@ def test_unpaper_args_invalid_filename(resources, outpdf): @pytest.mark.skipif(not have_unpaper(), reason="requires unpaper") def test_unpaper_args_invalid(resources, outpdf): - p, out, err = run_ocrmypdf( + p, _out, _err = run_ocrmypdf( resources / "skew.pdf", outpdf, "-c", diff --git a/tests/test_userunit.py b/tests/test_userunit.py index e3b9fe31..14ac653c 100644 --- a/tests/test_userunit.py +++ b/tests/test_userunit.py @@ -12,7 +12,7 @@ import pytest from ocrmypdf.exceptions import ExitCode from ocrmypdf.pdfinfo import PdfInfo -from .conftest import check_ocrmypdf, run_ocrmypdf, run_ocrmypdf_api +from .conftest import check_ocrmypdf, run_ocrmypdf_api # pylint: disable=redefined-outer-name diff --git a/tests/test_validation.py b/tests/test_validation.py index 81a4e5d5..745ea15e 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -6,7 +6,7 @@ import logging -from unittest.mock import MagicMock, patch +from unittest.mock import patch import pikepdf import pytest @@ -194,7 +194,7 @@ def test_no_progress_bar(progress_bar, resources): def test_language_warning(caplog): opts = make_opts(language=None) - plugin_manager = get_plugin_manager(opts.plugins) + _plugin_manager = get_plugin_manager(opts.plugins) caplog.set_level(logging.DEBUG) with patch( 'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8') From e788dde607c761a16518c72103fb711b74043391 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 02:11:31 -0700 Subject: [PATCH 845/880] tests: eliminate unnecessary mmap --- tests/test_metadata.py | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 87df840a..ebcbf031 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -6,7 +6,6 @@ import datetime -import mmap from datetime import timezone from os import fspath from shutil import copyfile @@ -356,17 +355,16 @@ def test_prevent_gs_invalid_xml(resources, outdir): str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context ) - with open(outdir / 'pdfa.pdf', 'r+b') as f: - with mmap.mmap(f.fileno(), 0) as mm: - # Since the XML may be invalid, we scan instead of actually feeding it - # to a parser. - XMP_MAGIC = b'W5M0MpCehiHzreSzNTczkc9d' - xmp_start = mm.find(XMP_MAGIC) - xmp_end = mm.rfind(b' Date: Wed, 7 Apr 2021 02:18:08 -0700 Subject: [PATCH 846/880] Delinting: unused args --- src/ocrmypdf/optimize.py | 5 +++++ tests/test_stdio.py | 6 ++---- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/optimize.py b/src/ocrmypdf/optimize.py index 038e454d..fce66464 100644 --- a/src/ocrmypdf/optimize.py +++ b/src/ocrmypdf/optimize.py @@ -65,6 +65,9 @@ def jpg_name(root: Path, xref: Xref) -> Path: def extract_image_filter( pike: Pdf, root: Path, image: Object, xref: Xref ) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]: + del pike # unused args + del root + if image.Subtype != Name.Image: return None if image.Length < 100: @@ -103,6 +106,8 @@ def extract_image_filter( def extract_image_jbig2( *, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options ) -> Optional[XrefExt]: + del options # unused arg + result = extract_image_filter(pike, root, image, xref) if result is None: return None diff --git a/tests/test_stdio.py b/tests/test_stdio.py index 2e43cff5..f5993704 100644 --- a/tests/test_stdio.py +++ b/tests/test_stdio.py @@ -30,8 +30,7 @@ def test_stdin(ocrmypdf_exec, resources, outpdf): '--plugin', 'tests/plugins/tesseract_noop.py', ] - p = run(p_args, stdout=PIPE, stderr=PIPE, stdin=input_stream) - assert p.returncode == ExitCode.ok + run(p_args, stdout=PIPE, stderr=PIPE, stdin=input_stream, check=True) def test_stdout(ocrmypdf_exec, resources, outpdf): @@ -49,8 +48,7 @@ def test_stdout(ocrmypdf_exec, resources, outpdf): '--plugin', 'tests/plugins/tesseract_noop.py', ] - p = run(p_args, stdout=output_stream, stderr=PIPE, stdin=DEVNULL) - assert p.returncode == ExitCode.ok + run(p_args, stdout=output_stream, stderr=PIPE, stdin=DEVNULL, check=True) assert check_pdf(output_file) From 9416e850ffa84ed9eb8299d6eebfad9b933ab729 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 02:23:04 -0700 Subject: [PATCH 847/880] Remove another instance of helpers_namespace --- tests/conftest.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index d3b3ebb7..bae8f5be 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -17,9 +17,6 @@ from ocrmypdf import api, pdfinfo from ocrmypdf._exec import unpaper from ocrmypdf._plugin_manager import get_parser_options_plugins -pytest_plugins = ['helpers_namespace'] - - if sys.version_info < (3, 5): print("Requires Python 3.5+") sys.exit(1) From 906d77b389f76c9600c2f03c6a224bc913a7ec08 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 02:25:10 -0700 Subject: [PATCH 848/880] tests: remove obsolete running_in_travis() --- tests/conftest.py | 4 ---- tests/test_main.py | 5 ++--- 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index bae8f5be..9a059c83 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -36,10 +36,6 @@ def running_in_docker(): return Path('/.dockerenv').exists() or Path('/.dockerinit').exists() -def running_in_travis(): - return os.environ.get('TRAVIS') == 'true' - - def have_unpaper(): try: unpaper.version() diff --git a/tests/test_main.py b/tests/test_main.py index 4f5aedcf..0d827fa3 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -32,7 +32,6 @@ from .conftest import ( run_ocrmypdf, run_ocrmypdf_api, running_in_docker, - running_in_travis, ) # pylint: disable=redefined-outer-name @@ -214,8 +213,8 @@ def test_force_ocr_on_pdf_with_no_images(resources, no_outpdf): @pytest.mark.skipif( - is_macos() and running_in_travis(), - reason="takes too long to install language packs in Travis macOS homebrew", + is_macos(), + reason="takes too long to install language packs in macOS homebrew", ) def test_german(resources, outdir): # Produce a sidecar too - implicit test that system locale is set up From 336d274a54c6898a508e36bd43c1f9f501490542 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 22:46:39 -0700 Subject: [PATCH 849/880] Drop remnants of support for Tesseract without has_textonly_pdf Also improve Tesseract version checking so it can compare all of their weird conventions. --- src/ocrmypdf/_exec/tesseract.py | 42 +++++++++---------- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 8 +--- src/ocrmypdf/subprocess/__init__.py | 38 +++++++++++++---- tests/test_validation.py | 38 ++++++++--------- 4 files changed, 68 insertions(+), 58 deletions(-) diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 7f7c06a4..5dc5024b 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -9,8 +9,10 @@ import logging import os +import re import shutil from collections import namedtuple +from distutils.version import StrictVersion from os import fspath from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired @@ -53,31 +55,29 @@ class TesseractLoggerAdapter(logging.LoggerAdapter): return '[tesseract] %s' % (msg), kwargs +class TesseractVersion(StrictVersion): + version_re = re.compile( + r''' + ^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch + [-]? # optional hyphen separator + (?:(alpha|beta|rc|dev)[.\-\ ]?(\d+)?)? # 5/prerelease, 6/prerelease_num + $ + ''', + re.VERBOSE | re.ASCII, + ) + + def parse(self, vstring): + try: + super().parse(vstring) + except TypeError as e: + if 'int() argument must be a string' in str(e): + super().parse(vstring + '0') + + def version(): return get_version('tesseract', regex=r'tesseract\s(.+)') -def has_textonly_pdf(langs=None): - """Does Tesseract have textonly_pdf capability? - - Available in v4.00.00alpha since January 2017. Best to - parse the parameter list. - """ - args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf'] - params = '' - try: - proc = run(args_tess, check=True, stdout=PIPE, stderr=STDOUT) - params = proc.stdout - except CalledProcessError as e: - raise MissingDependencyError( - "Could not --print-parameters from tesseract. This can happen if the " - "TESSDATA_PREFIX environment is not set to a valid tessdata folder. " - ) from e - if b'textonly_pdf' in params: - return True - return False - - def has_user_words(): """Does Tesseract have --user-words capability? diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index bbcb6720..bcb5de7f 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -81,19 +81,13 @@ def check_options(options): package={'linux': 'tesseract-ocr'}, version_checker=tesseract.version, need_version='4.0.0', # using backport for Travis CI + version_parser=tesseract.TesseractVersion, ) # Decide on what renderer to use if options.pdf_renderer == 'auto': options.pdf_renderer = 'sandwich' - if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf( - set(options.languages) - ): - raise MissingDependencyError( - "You are using an alpha version of Tesseract 4.0 that does not support " - "the textonly_pdf parameter. We don't support versions this old." - ) if not tesseract.has_user_words() and (options.user_words or options.user_patterns): log.warning( "Tesseract 4.0 ignores --user-words and --user-patterns, so these " diff --git a/src/ocrmypdf/subprocess/__init__.py b/src/ocrmypdf/subprocess/__init__.py index 2755337b..99f052c6 100644 --- a/src/ocrmypdf/subprocess/__init__.py +++ b/src/ocrmypdf/subprocess/__init__.py @@ -13,14 +13,17 @@ import re import sys from collections.abc import Mapping from contextlib import suppress -from distutils.version import LooseVersion +from distutils.version import LooseVersion, Version from functools import lru_cache from pathlib import Path from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen from subprocess import run as subprocess_run +from typing import Callable, Optional, Type, Union from ocrmypdf.exceptions import MissingDependencyError +# pylint: disable=logging-format-interpolation + log = logging.getLogger(__name__) @@ -264,13 +267,30 @@ def _error_old_version(program, package, need_version, found_version, required_f def check_external_program( *, - program, - package, - version_checker, - need_version, - required_for=None, + program: str, + package: str, + version_checker: Union[str, Callable], + need_version: str, + required_for: Optional[str] = None, recommended=False, + version_parser: Type[Version] = LooseVersion, ): + """Check for required version of external program and raise exception if not. + + Args: + program: The name of the program to test. + package: The name of a software package that typically supplies this program. + Usually the same as program. + version_check: A callable without arguments that retrieves the installed + version of program. + need_version: The minimum required version. + required_for: The name of an argument of feature that requires this program. + recommended: If this external program is recommended, instead of raising + an exception, log a warning and allow execution to continue. + version_parser: A class that should be used to parse and compare version + numbers. Used when version numbers do not follow standard conventions. + """ + try: if callable(version_checker): found_version = version_checker() @@ -279,7 +299,7 @@ def check_external_program( except (CalledProcessError, FileNotFoundError, MissingDependencyError): _error_missing_program(program, package, required_for, recommended) if not recommended: - raise MissingDependencyError() + raise MissingDependencyError(program) return def remove_leading_v(s): @@ -290,9 +310,9 @@ def check_external_program( found_version = remove_leading_v(found_version) need_version = remove_leading_v(need_version) - if found_version and LooseVersion(found_version) < LooseVersion(need_version): + if found_version and version_parser(found_version) < version_parser(need_version): _error_old_version(program, package, need_version, found_version, required_for) if not recommended: - raise MissingDependencyError() + raise MissingDependencyError(program) log.debug('Found %s %s', program, found_version) diff --git a/tests/test_validation.py b/tests/test_validation.py index 745ea15e..6bec04e3 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -13,6 +13,7 @@ import pytest from ocrmypdf import _validation as vd from ocrmypdf._concurrent import NullProgressBar, SerialExecutor +from ocrmypdf._exec.tesseract import TesseractVersion from ocrmypdf._plugin_manager import get_plugin_manager from ocrmypdf.api import create_options from ocrmypdf.cli import get_parser @@ -52,29 +53,23 @@ def test_hocr_notlatin_warning(caplog): def test_old_ghostscript(caplog): - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'), patch( - 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True - ): + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'): vd._check_options( *make_opts_pm(language='chi_sim', output_type='pdfa'), {'chi_sim'} ) assert 'does not work correctly' in caplog.text - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'), patch( - 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True - ): + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'): with pytest.raises(MissingDependencyError): vd._check_options(*make_opts_pm(output_type='pdfa-3'), set()) - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.24'), patch( - 'ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True - ): + with patch('ocrmypdf._exec.ghostscript.version', return_value='9.24'): with pytest.raises(MissingDependencyError): vd._check_options(*make_opts_pm(), set()) def test_old_tesseract_error(): - with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=False): + with patch('ocrmypdf._exec.tesseract.version', return_value='4.00.00alpha'): with pytest.raises(MissingDependencyError): opts = make_opts(pdf_renderer='sandwich', language='eng') plugin_manager = get_plugin_manager(opts.plugins) @@ -227,17 +222,20 @@ def test_version_comparison(): version_checker=lambda: '10.0', need_version='8.0.2', ) - vd.check_external_program( - program="tesseract", - package="tesseract", - version_checker=lambda: '4.0.0-beta.1', - need_version='4.0.0', - ) + with pytest.raises(MissingDependencyError): + vd.check_external_program( + program="tesseract", + package="tesseract", + version_checker=lambda: '4.0.0-beta.1', + need_version='4.0.0', + version_parser=TesseractVersion, + ) vd.check_external_program( program="tesseract", package="tesseract", version_checker=lambda: 'v5.0.0-alpha.20200201', need_version='4.0.0', + version_parser=TesseractVersion, ) with pytest.raises(MissingDependencyError): vd.check_external_program( @@ -277,11 +275,9 @@ def test_pagesegmode_warning(caplog): def test_two_languages(): - with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock: - vd._check_options( - *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} - ) - mock.assert_called() + vd._check_options( + *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} + ) def test_sidecar_equals_output(resources, no_outpdf): From 8423bd549bfe1ade5f505b0f34501120a00f9abd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 23:16:10 -0700 Subject: [PATCH 850/880] helpers: don't trap exception on failure to unlink If we can't unlink a file we expect to unlink, logging and moving on is probably the wrong action. Coverage never hits this line. --- src/ocrmypdf/helpers.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 42039cab..538ecec9 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -87,10 +87,7 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike): # do not delete or overwrite real (non-soft link) file if not os.path.islink(soft_link_name): raise FileExistsError(f"{soft_link_name} exists and is not a link") - try: - os.unlink(soft_link_name) - except OSError: - log.debug("Can't unlink %s", soft_link_name) + os.unlink(soft_link_name) if not os.path.exists(input_file): raise FileNotFoundError(f"trying to create a broken symlink to {input_file}") From 9db9a3d6ec7674ab7d6146bfb91f1dff3bb2f230 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 7 Apr 2021 23:26:37 -0700 Subject: [PATCH 851/880] helpers: improve test coverage of Resolution --- src/ocrmypdf/helpers.py | 6 ++++-- tests/test_helpers.py | 11 +++++++++++ 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 538ecec9..1dae1f31 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -54,7 +54,7 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))): def __str__(self): return f"{self.x:f}x{self.y:f}" - def __repr__(self): + def __repr__(self): # pragma: no cover return f"Resolution({self.x}x{self.y} dpi)" @@ -200,7 +200,9 @@ def check_pdf(input_file: Path) -> bool: pdf.check_linearization(sio) except RuntimeError: pass - except ( # Workaround for a problematic pikepdf version + except ( + # Workaround for a problematic pikepdf version + # pragma: no cover getattr(pikepdf, 'ForeignObjectError') if pikepdf.__version__ == '2.1.0' else NeverRaise diff --git a/tests/test_helpers.py b/tests/test_helpers.py index 3a6c0b5a..5af7f610 100644 --- a/tests/test_helpers.py +++ b/tests/test_helpers.py @@ -117,3 +117,14 @@ def test_shim_paths(tmp_path): assert results[-3].endswith('tesseract-ocr'), results assert results[-2].endswith(os.path.join('gs', '9.52', 'bin')), results assert results[-1].endswith(os.path.join('gs', '9.51', 'bin')), results + + +def test_resolution(): + Resolution = helpers.Resolution + dpi_100 = Resolution(100, 100) + dpi_200 = Resolution(200, 200) + assert dpi_100.is_square + assert not Resolution(100, 200).is_square + assert dpi_100 == Resolution(100, 100) + assert str(dpi_100) != str(dpi_200) + assert dpi_100.take_max([200, 300], [400]) == Resolution(300, 400) From 913c939dc9fb08eddd925f31ba41a2b606be13e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 01:51:44 -0700 Subject: [PATCH 852/880] docs: update wrt AWS --- docs/batch.rst | 7 ------- docs/plugins.rst | 2 ++ 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/docs/batch.rst b/docs/batch.rst index 63f6e41b..e8e97374 100644 --- a/docs/batch.rst +++ b/docs/batch.rst @@ -202,13 +202,6 @@ Alternatives - `Watchman `__ is a more powerful alternative to ``watchmedo``. -AWS Lambda is not viable ------------------------- - -AWS Lambda and its equivalents have low limits on execution time and payload -size, relative to OCRmyPDF's needs. As of this writing, the request/response -payload for AWS Lambda was 6 MB, which means many PDFs will not fit. - macOS Automator =============== diff --git a/docs/plugins.rst b/docs/plugins.rst index f0ae2d85..952af3fd 100644 --- a/docs/plugins.rst +++ b/docs/plugins.rst @@ -180,6 +180,8 @@ Modifying intermediate images .. autofunction:: ocrmypdf.pluginspec.filter_page_image +.. autofunction:: ocrmypdf.pluginspec.filter_pdf_page + OCR engine ---------- From 9de38afb135afdc8c0a4179a6c49abec1360f5e5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 13:07:13 -0700 Subject: [PATCH 853/880] Fix Tesseract version for ubuntu 18.04 --- src/ocrmypdf/builtin_plugins/tesseract_ocr.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index bcb5de7f..a6973b01 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -80,7 +80,7 @@ def check_options(options): program='tesseract', package={'linux': 'tesseract-ocr'}, version_checker=tesseract.version, - need_version='4.0.0', # using backport for Travis CI + need_version='4.0.0-beta.1', # using backport for Travis CI version_parser=tesseract.TesseractVersion, ) From 532d65a35564d045a968180d24f01f7b40dce551 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 14:19:59 -0700 Subject: [PATCH 854/880] ci: push on tag --- .github/workflows/build.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 1a096821..cad70f3f 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -6,6 +6,8 @@ on: - master - ci - release/* + tags: + - v* paths-ignore: - README* pull_request: From 139d9f9841e761f4c170da977bb8e0e75844dcc6 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 20:58:46 -0700 Subject: [PATCH 855/880] Shut up pikepdf mmap disabled message --- src/ocrmypdf/helpers.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 1dae1f31..0f26e025 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -232,7 +232,7 @@ def pikepdf_enable_mmap(): # We found a race condition probably related to pybind issue #2252 that can # cause a crash. For now, disable pikepdf mmap to be on the safe side. # Fix is not in pybind11 2.6.0 - log.debug("pikepdf mmap disabled") + # log.debug("pikepdf mmap disabled") return From 051b9da99182043483f8f9724afd5836fd5d50e2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 21:08:05 -0700 Subject: [PATCH 856/880] Remove undocumented/unused debug environment variables --- src/ocrmypdf/_pipeline.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 1ee0c104..ce3b8d88 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -686,11 +686,6 @@ def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]: pdfmark['/Creator'] = f'{PROGRAM_NAME} {VERSION} / {creator_tag}' pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}' - if 'OCRMYPDF_CREATOR' in os.environ: - pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR'] - if 'OCRMYPDF_PRODUCER' in os.environ: - pdfmark['/Producer'] = os.environ['OCRMYPDF_PRODUCER'] - pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc)) return pdfmark From a5852ba199a1dfc1c9759d3c99d5a5c98161f408 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 8 Apr 2021 21:39:18 -0700 Subject: [PATCH 857/880] Remove parent process's log handlers properly --- src/ocrmypdf/builtin_plugins/concurrency.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 16005984..fd26e373 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -72,12 +72,15 @@ def process_init(q: Queue, user_init: Callable[[], None], loglevel): # Windows and Cygwin do not have pthread_sigmask or SIGBUS signal.signal(signal.SIGBUS, process_sigbus) - # Reconfigure the root logger for this process to send all messages to a queue - h = logging.handlers.QueueHandler(q) + # Remove any log handlers that belong to the parent process root = logging.getLogger() + for handler in root.handlers[:]: + root.removeHandler(handler) + handler.close() # To ensure handlers with opened resources are released + + # Set up our single log handler to forward messages to the parent root.setLevel(loglevel) - root.handlers = [] - root.addHandler(h) + root.addHandler(logging.handlers.QueueHandler(q)) user_init() return From e4f69cc1d6fba6e9fde8d6900c97c0b2ba5ba509 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Apr 2021 01:51:50 -0700 Subject: [PATCH 858/880] Maybe fix a deadlock on attempting sys.stderr.flush() Closes #758 Closes #733 --- src/ocrmypdf/leptonica.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index c8759cfa..8ec02e0a 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -88,9 +88,9 @@ except ffi.error as e: class _LeptonicaErrorTrap_Redirect: """ - Context manager to trap errors reported by Leptonica. + Context manager to trap errors reported by Leptonica < 1.79 or on Apple Silicon. - Leptonica's error return codes don't provide much informatino about what + Leptonica's error return codes don't provide much information about what went wrong. Leptonica does, however, write more detailed errors to stderr (provided this is not disabled at compile time). The Leptonica source code is very consistent in its use of macros to generate errors. @@ -114,8 +114,10 @@ class _LeptonicaErrorTrap_Redirect: # Save the old stderr, and redirect stderr to temporary file self.leptonica_lock.acquire() try: - with suppress(AttributeError): - sys.stderr.flush() + # It would make sense to do sys.stderr.flush() here, but that can deadlock + # due to https://bugs.python.org/issue6721. So don't flush. Pretend + # there's nothing important sys.stderr. If the user cared they would + # be use Leptonica 1.79 or later anyway and avoid this mess. self.copy_of_stderr = os.dup(sys.stderr.fileno()) os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False) except AttributeError: From a90b9e669fb2b33591d260c407203375c697d59b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 9 Apr 2021 13:22:48 -0700 Subject: [PATCH 859/880] Add reminder to not mess with pool/listener order --- src/ocrmypdf/builtin_plugins/concurrency.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index fd26e373..74b3500f 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -117,7 +117,8 @@ class StandardExecutor(Executor): initializer = process_init # Regardless of whether we use_threads for worker processes, the log_listener - # must be a thread + # must be a thread. Make sure we create the listener after the worker pool, + # so that it does not get forked into the workers. listener = threading.Thread(target=log_listener, args=(log_queue,)) listener.start() From 8f8aaa93ed65b0b7067562b5ad52b356177f923c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 13 Apr 2021 13:16:05 -0700 Subject: [PATCH 860/880] Refactor removing log handlers --- src/ocrmypdf/builtin_plugins/concurrency.py | 5 ++--- src/ocrmypdf/extra_plugins/awslambda.py | 3 ++- src/ocrmypdf/helpers.py | 7 +++++++ 3 files changed, 11 insertions(+), 4 deletions(-) diff --git a/src/ocrmypdf/builtin_plugins/concurrency.py b/src/ocrmypdf/builtin_plugins/concurrency.py index 74b3500f..797087ae 100644 --- a/src/ocrmypdf/builtin_plugins/concurrency.py +++ b/src/ocrmypdf/builtin_plugins/concurrency.py @@ -29,6 +29,7 @@ from tqdm import tqdm from ocrmypdf import Executor, hookimpl from ocrmypdf._logging import TqdmConsole from ocrmypdf.exceptions import InputFileError +from ocrmypdf.helpers import remove_all_log_handlers Queue = Union[multiprocessing.Queue, queue.Queue] @@ -74,9 +75,7 @@ def process_init(q: Queue, user_init: Callable[[], None], loglevel): # Remove any log handlers that belong to the parent process root = logging.getLogger() - for handler in root.handlers[:]: - root.removeHandler(handler) - handler.close() # To ensure handlers with opened resources are released + remove_all_log_handlers(root) # Set up our single log handler to forward messages to the parent root.setLevel(loglevel) diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 23fe9799..0f98f470 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -22,6 +22,7 @@ from unittest.mock import Mock from ocrmypdf import Executor, hookimpl from ocrmypdf._concurrent import NullProgressBar from ocrmypdf.exceptions import InputFileError +from ocrmypdf.helpers import remove_all_log_handlers class MessageType(Enum): @@ -61,8 +62,8 @@ def process_loop( # Reconfigure the root logger for this process to send all messages to a queue h = ConnectionLogHandler(conn) root = logging.getLogger() + remove_all_log_handlers(root) root.setLevel(loglevel) - root.handlers = [] root.addHandler(h) user_init() diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 0f26e025..07ff66ae 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -223,6 +223,13 @@ def clamp(n, smallest, largest): # mypy doesn't understand types for this return max(smallest, min(n, largest)) +def remove_all_log_handlers(logger): + "Remove all log handlers, usually used in a child process." + for handler in logger.handlers[:]: + logger.removeHandler(handler) + handler.close() # To ensure handlers with opened resources are released + + def pikepdf_enable_mmap(): # try: # if pikepdf._qpdf.set_access_default_mmap(True): From f453e94f146b149f0d11f2889824407b9959e98e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 13 Apr 2021 13:16:35 -0700 Subject: [PATCH 861/880] awslambda: better documentation --- src/ocrmypdf/extra_plugins/awslambda.py | 26 ++++++++++++++++++++++--- 1 file changed, 23 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index 0f98f470..bf7569f3 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -5,7 +5,20 @@ # file, You can obtain one at http://mozilla.org/MPL/2.0/. -"""Alternate executor to support OCRmyPDF in AWS Lambda""" +"""Alternate executor to support OCRmyPDF in AWS Lambda. + +AWS Lambda does not support the standard multiprocessing module because it +environment. + +This alternate executor avoids that. However, it has drawbacks. Most notably, +it divvies up work among worker processes at the beginning, to avoid coordinating +effort in ways that require shared queues and semaphores. In this implementation, +there is no shared queue, so no possible lock contention between workers. The +main process has a shared pipe with each worker. + +If some tasks are larger than others, some workers will may fall far behind +while others have deep queues. The last worker may end up with fewer tasks. +""" import logging @@ -16,7 +29,7 @@ from enum import Enum, auto from itertools import islice, repeat, takewhile, zip_longest from multiprocessing import Pipe, Process from multiprocessing.connection import Connection, wait -from typing import Callable, Iterable, Optional +from typing import Callable, Iterable, Iterator from unittest.mock import Mock from ocrmypdf import Executor, hookimpl @@ -31,7 +44,14 @@ class MessageType(Enum): complete = auto() -def split_every(n: int, iterable: Iterable): +def split_every(n: int, iterable: Iterable) -> Iterator: + """Split iterable into groups of n. + + >>> list(split_every(4, range(10))) + [[0, 1, 2], [3, 4, 5], [6, 7, 8], [9]] + + https://stackoverflow.com/a/22919323 + """ iterator = iter(iterable) return takewhile(bool, (list(islice(iterator, n)) for _ in repeat(None))) From fc75254c603620557b30fe3e5a80feba1ba280d2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 13 Apr 2021 14:37:30 -0700 Subject: [PATCH 862/880] Fix awslambda progressbar.update() error Fixes #759 --- src/ocrmypdf/extra_plugins/awslambda.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/awslambda.py index bf7569f3..0effcc1d 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/awslambda.py @@ -115,9 +115,10 @@ class LambdaExecutor(Executor): task_finished: Callable, ): if use_threads and max_workers == 1: - for args in task_arguments: - result = task(args) - task_finished(result, self.pbar_class) + with self.pbar_class(**tqdm_kwargs) as pbar: + for args in task_arguments: + result = task(args) + task_finished(result, pbar) return task_arguments = list(task_arguments) From 710d797299685ffcc41155e0441df16b47877df2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 14 Apr 2021 00:36:58 -0700 Subject: [PATCH 863/880] Fix comment typos --- src/ocrmypdf/leptonica.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 8ec02e0a..a6757391 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -116,8 +116,8 @@ class _LeptonicaErrorTrap_Redirect: try: # It would make sense to do sys.stderr.flush() here, but that can deadlock # due to https://bugs.python.org/issue6721. So don't flush. Pretend - # there's nothing important sys.stderr. If the user cared they would - # be use Leptonica 1.79 or later anyway and avoid this mess. + # there's nothing important in sys.stderr. If the user cared they would + # be using Leptonica 1.79 or later anyway to avoid this mess. self.copy_of_stderr = os.dup(sys.stderr.fileno()) os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False) except AttributeError: From d89a633ba73af4a6bdacda6b9a4c0638b39167bd Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 15 Apr 2021 23:25:18 -0700 Subject: [PATCH 864/880] Remove apparently unused portion of a test --- tests/test_pdfinfo.py | 8 +------- 1 file changed, 1 insertion(+), 7 deletions(-) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 90e75ac3..e1e58244 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -87,14 +87,8 @@ def test_single_page_image(eight_by_eight, outpdf): assert isclose(pdfimage.dpi.y, 8) -def test_single_page_inline_image(eight_by_eight, outdir): +def test_single_page_inline_image(outdir): filename = outdir / 'image-mono-inline.pdf' - pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) - - # Draw image in a 72x72 pt or 1"x1" area - pdf.drawInlineImage(eight_by_eight, 0, 0, width=72, height=72) - pdf.showPage() - pdf.save() info = pdfinfo.PdfInfo(filename) print(info) From d6731269940239f8bfaa82e510a4531128b733b1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 15 Apr 2021 23:26:14 -0700 Subject: [PATCH 865/880] Fix ZeroDivisionError on files containing images drawn at scale 0 Fixes #761 --- src/ocrmypdf/helpers.py | 6 +++++- src/ocrmypdf/pdfinfo/info.py | 15 +++++++++++---- tests/test_pdfinfo.py | 21 +++++++++++++++++++++ 3 files changed, 37 insertions(+), 5 deletions(-) diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 07ff66ae..d7080f34 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -15,7 +15,7 @@ from collections.abc import Iterable from contextlib import suppress from functools import wraps from io import StringIO -from math import isclose +from math import isclose, isfinite from pathlib import Path from typing import Any, Sequence @@ -39,6 +39,10 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))): def is_square(self) -> bool: return isclose(self.x, self.y, rel_tol=1e-3) + @property + def is_finite(self) -> bool: + return isfinite(self.x) and isfinite(self.y) + def take_max(self, vals, yvals=None): if yvals is not None: return Resolution(max(self.x, *vals), max(self.y, *yvals)) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index 50c77b0b..aac1f867 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -14,7 +14,7 @@ from contextlib import ExitStack from decimal import Decimal from enum import Enum from functools import partial -from math import hypot, isclose +from math import hypot, isclose, ulp from os import PathLike from pathlib import Path from typing import Container, Iterator, Optional, Tuple, Union @@ -260,8 +260,9 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution: image_drawn_height = hypot(c, d) # The scale of the image is pixels per unit of default user space (1/72") - scale_w = image_size[0] / image_drawn_width - scale_h = image_size[1] / image_drawn_height + # use ulp to turn divide by zero into infinity + scale_w = image_size[0] / (image_drawn_width + ulp(0)) + scale_h = image_size[1] / (image_drawn_height + ulp(0)) # DPI = scale * 72 dpi_w = scale_w * 72.0 @@ -365,6 +366,10 @@ class ImageInfo: def enc(self): return self._enc + @property + def renderable(self): + return self.dpi.is_finite and self.width >= 0 and self.height >= 0 + @property def dpi(self): return _get_dpi(self._shorthand, (self._width, self._height)) @@ -734,7 +739,9 @@ class PageInfo: self._dpi = None if self._images: - dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in self._images) + dpi = Resolution(0.0, 0.0).take_max( + image.dpi for image in self._images if image.renderable + ) self._dpi = dpi self._width_pixels = int(round(dpi.x * float(self._width_inches))) self._height_pixels = int(round(dpi.y * float(self._height_inches))) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index e1e58244..24f6234a 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -17,6 +17,7 @@ from reportlab.pdfgen.canvas import Canvas from ocrmypdf import pdfinfo from ocrmypdf.exceptions import InputFileError +from ocrmypdf.helpers import Resolution from ocrmypdf.pdfinfo import Colorspace, Encoding from ocrmypdf.pdfinfo.layout import PDFPage @@ -194,3 +195,23 @@ def test_pages_issue700(monkeypatch, resources): progbar=False, max_workers=1, ) + + +def test_image_scale0(resources, outpdf): + with pikepdf.open(resources / 'cmyk.pdf') as cmyk: + xobj = pikepdf.Page(cmyk.pages[0]).as_form_xobject() + + p = pikepdf.Pdf.new() + p.add_blank_page(page_size=(72, 72)) + objname = pikepdf.Page(p.pages[0]).add_resource( + p.copy_foreign(xobj), pikepdf.Name.XObject, pikepdf.Name.Im0 + ) + print(objname) + p.pages[0].Contents = pikepdf.Stream( + p, b"q 0 0 0 0 0 0 cm %s Do Q" % bytes(objname) + ) + p.save(outpdf) + + pi = pdfinfo.PdfInfo(outpdf, detailed_analysis=True, progbar=False, max_workers=1) + assert not pi.pages[0]._images[0].dpi.is_finite + assert pi.pages[0].dpi == Resolution(0, 0) From 75c5b92cb93cb33c6132cb99e8fa8ecff82359d1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 15 Apr 2021 23:49:58 -0700 Subject: [PATCH 866/880] v12.0.0 update release notes --- docs/release_notes.rst | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 4c1c8646..75ab6510 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -46,9 +46,26 @@ v12.0.0 way OCRmyPDF outputs its messages. - New plugin hook: ``filter_pdf_page``, for modifying individual PDF pages produced by OCRmyPDF. +- Using the new plugin hooks, it is now possible to run OCRmyPDF on alternative + execution environments that do not have interprocess semaphores, such as + AWS Lambda and Android Termux. +- Continuous integration moved to GitHub Actions. - We now generate an ARM64-compatible Docker image alongside the x64 image. Thanks to @andkrause for contributing the change and @0x326 for review comments. +**Fixes** + +- Fixed a possible deadlock on attempting to flush ``sys.stderr`` when older + versions of Leptonica are in use. +- Some worker processes inherited resources from their parents such as log + handlers that may have also lead to deadlocks. These resources are now released. +- Improvements to test coverage. +- Removed vestiges of support for Tesseract versions older than 4.0.0-beta1 ( + which ships with Ubuntu 18.04). +- OCRmyPDF can now parse all of Tesseract version numbers, since several + schemes have been in use. +- Fixed an issue with parsing PDFs that contain images drawn at a scale of 0. (#761) + v11.7.3 ======= From 757b72b0af7b1a5adae9a193a8a5b4c09dadde4b Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 16 Apr 2021 00:21:11 -0700 Subject: [PATCH 867/880] Revert "Remove apparently unused portion of a test" This reverts commit d89a633ba73af4a6bdacda6b9a4c0638b39167bd. --- tests/test_pdfinfo.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index 24f6234a..cc93293d 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -88,8 +88,14 @@ def test_single_page_image(eight_by_eight, outpdf): assert isclose(pdfimage.dpi.y, 8) -def test_single_page_inline_image(outdir): +def test_single_page_inline_image(eight_by_eight, outdir): filename = outdir / 'image-mono-inline.pdf' + pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) + + # Draw image in a 72x72 pt or 1"x1" area + pdf.drawInlineImage(eight_by_eight, 0, 0, width=72, height=72) + pdf.showPage() + pdf.save() info = pdfinfo.PdfInfo(filename) print(info) From 5112e9e8578243bd994d330a7ef798b5f42e7947 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 16 Apr 2021 00:54:41 -0700 Subject: [PATCH 868/880] Redo dpi calc to avoid 'math.ulp' --- src/ocrmypdf/pdfinfo/info.py | 19 ++++++++----------- 1 file changed, 8 insertions(+), 11 deletions(-) diff --git a/src/ocrmypdf/pdfinfo/info.py b/src/ocrmypdf/pdfinfo/info.py index aac1f867..64b59b9c 100644 --- a/src/ocrmypdf/pdfinfo/info.py +++ b/src/ocrmypdf/pdfinfo/info.py @@ -14,7 +14,7 @@ from contextlib import ExitStack from decimal import Decimal from enum import Enum from functools import partial -from math import hypot, isclose, ulp +from math import hypot, inf, isclose from os import PathLike from pathlib import Path from typing import Container, Iterator, Optional, Tuple, Union @@ -256,18 +256,15 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution: a, b, c, d, _, _ = ctm_shorthand # Calculate the width and height of the image in PDF units - image_drawn_width = hypot(a, b) - image_drawn_height = hypot(c, d) + image_drawn = hypot(a, b), hypot(c, d) - # The scale of the image is pixels per unit of default user space (1/72") - # use ulp to turn divide by zero into infinity - scale_w = image_size[0] / (image_drawn_width + ulp(0)) - scale_h = image_size[1] / (image_drawn_height + ulp(0)) - - # DPI = scale * 72 - dpi_w = scale_w * 72.0 - dpi_h = scale_h * 72.0 + def calc(drawn, pixels, inches_per_pt=72.0): + # The scale of the image is pixels per unit of default user space (1/72") + scale = pixels / drawn if drawn != 0 else inf + dpi = scale * inches_per_pt + return dpi + dpi_w, dpi_h = (calc(image_drawn[n], image_size[n]) for n in range(2)) return Resolution(dpi_w, dpi_h) From d25c49ba81c5eca4ef4b5b4ebdbdb4c4e829de21 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 19 Apr 2021 00:06:22 -0700 Subject: [PATCH 869/880] docs: remove incorrect value of rotate-pages-threshold from docs Closes #762 --- docs/cookbook.rst | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/docs/cookbook.rst b/docs/cookbook.rst index 59e7b267..8ee4d8fd 100644 --- a/docs/cookbook.rst +++ b/docs/cookbook.rst @@ -58,9 +58,11 @@ portrait pages. You can increase (decrease) the parameter ``--rotate-pages-threshold`` to make page rotation more (less) aggressive. The threshold number is the ratio of how confidence the OCR engine is that the document image should be changed, -compared to kept the same. A value of ``15.0`` is the default, and is fairly -conservative. A value of ``2.0`` will produce more rotations, and more false -positives. +compared to kept the same. The default value is quite conservative; on some files +it may not attempt rotations at all unless it is very confident that the current +rotation is wrong. A lower value of ``2.0`` will produce more rotations, and +more false positives. Run with ``-v1`` to see the confidence level for each +page to see if there may be a better value for your files. If the page is "just a little off horizontal", like a crooked picture, then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal From 252221fd8bbc6214f0123cab0d92a42e8a6af99c Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 21 Apr 2021 23:17:42 -0700 Subject: [PATCH 870/880] dockerfile: remove unnecessary copy of /app/src --- .docker/Dockerfile | 1 - 1 file changed, 1 deletion(-) diff --git a/.docker/Dockerfile b/.docker/Dockerfile index 89cc336f..4700004e 100644 --- a/.docker/Dockerfile +++ b/.docker/Dockerfile @@ -79,6 +79,5 @@ COPY --from=builder /app/misc/watcher.py /app/ COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/ COPY --from=builder /app/requirements /app/requirements COPY --from=builder /app/tests /app/tests -COPY --from=builder /app/src /app/src ENTRYPOINT ["/usr/local/bin/ocrmypdf"] From be45871d10946ac79b8c67f50ede9ee7baeaea1d Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 21 Apr 2021 23:18:29 -0700 Subject: [PATCH 871/880] docker: add special hint for using docker --- src/ocrmypdf/_validation.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index a3dd7448..ab48195e 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -341,7 +341,17 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]: safe_symlink(options.input_file, target) return target, os.fspath(options.input_file) except FileNotFoundError: - raise InputFileError(f"File not found - {options.input_file}") + msg = f"File not found - {options.input_file}" + if Path('/.dockerenv').exists(): # pragma: no cover + msg += ( + "\nDocker cannot your working directory unless you " + "explicitly share it with the Docker container and set up" + "permissions correctly.\n" + "You may find it easier to use stdin/stdout:" + "\n" + "\tdocker run -i --rm jbarlow83/ocrmypdf - - output.pdf\n" + ) + raise InputFileError(msg) def check_requested_output_file(options): From ad0126185f5317110b7a3ba8329f671284fb30d1 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 21 Apr 2021 23:29:55 -0700 Subject: [PATCH 872/880] Rename awslambda -> semfree.py --- .../{awslambda.py => semfree.py} | 22 +++++++++---------- 1 file changed, 10 insertions(+), 12 deletions(-) rename src/ocrmypdf/extra_plugins/{awslambda.py => semfree.py} (87%) diff --git a/src/ocrmypdf/extra_plugins/awslambda.py b/src/ocrmypdf/extra_plugins/semfree.py similarity index 87% rename from src/ocrmypdf/extra_plugins/awslambda.py rename to src/ocrmypdf/extra_plugins/semfree.py index 0effcc1d..c84b1b2d 100644 --- a/src/ocrmypdf/extra_plugins/awslambda.py +++ b/src/ocrmypdf/extra_plugins/semfree.py @@ -5,22 +5,21 @@ # file, You can obtain one at http://mozilla.org/MPL/2.0/. -"""Alternate executor to support OCRmyPDF in AWS Lambda. +"""Semaphore-free alternate executor. -AWS Lambda does not support the standard multiprocessing module because it -environment. +There are two popular environments that do not fully support the standard Python +multiprocessing module: AWS Lambda, and Termux (a terminal emulator for Android). -This alternate executor avoids that. However, it has drawbacks. Most notably, -it divvies up work among worker processes at the beginning, to avoid coordinating -effort in ways that require shared queues and semaphores. In this implementation, -there is no shared queue, so no possible lock contention between workers. The -main process has a shared pipe with each worker. +This alternate executor divvies up work among worker processes before processing, +rather than having each worker consume work from a shared queue when they finish +their task. This means workers have no need to coordinate with each other. Each +worker communicates only with the main process. -If some tasks are larger than others, some workers will may fall far behind -while others have deep queues. The last worker may end up with fewer tasks. +This is not without drawbacks. If the tasks are not "even" in size, which cannot +be guaranteed, some workers may end up with too much work while others are idle. +It is less efficient than the standard implementation, so not th edefault. """ - import logging import logging.handlers import signal @@ -30,7 +29,6 @@ from itertools import islice, repeat, takewhile, zip_longest from multiprocessing import Pipe, Process from multiprocessing.connection import Connection, wait from typing import Callable, Iterable, Iterator -from unittest.mock import Mock from ocrmypdf import Executor, hookimpl from ocrmypdf._concurrent import NullProgressBar From a613722e967b0f0dfe4cffc3a7f7e86a7226f2a2 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 21 Apr 2021 23:39:38 -0700 Subject: [PATCH 873/880] Auto-register semfree.py if needed --- src/ocrmypdf/_plugin_manager.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/src/ocrmypdf/_plugin_manager.py b/src/ocrmypdf/_plugin_manager.py index 3d245676..0ac14578 100644 --- a/src/ocrmypdf/_plugin_manager.py +++ b/src/ocrmypdf/_plugin_manager.py @@ -73,10 +73,19 @@ class OcrmypdfPluginManager(pluggy.PluginManager): module = importlib.import_module(name) self.register(module) - # 2. Register setuptools plugins + # 2. Install semfree if needed + try: + # pylint: disable=import-outside-toplevel + from multiprocessing.synchronize import SemLock + + del SemLock + except ImportError: + self.register(importlib.import_module('ocrmypdf.extra_plugins.semfree')) + + # 3. Register setuptools plugins self.load_setuptools_entrypoints('ocrmypdf') - # 3. Register plugins specified on command line + # 4. Register plugins specified on command line for name in self.__plugins: if isinstance(name, Path) or name.endswith('.py'): # Import by filename From 33e0b16174b8deb17e41eafdbe2e77a1b7ee0070 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Fri, 23 Apr 2021 00:04:21 -0700 Subject: [PATCH 874/880] v12.0.0 release notes --- docs/release_notes.rst | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 75ab6510..d4b1341b 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -25,12 +25,15 @@ v12.0.0 Ghostscript to create a PDF/A. Generally this is faster than performing a color conversion, which is not always necessary. - OCR text is now packaged in a Form XObject. This makes it easier to isolate - OCR from other document content. However, some poor implemented PDF text - extraction algorithms may fail to find the text. + OCR from other document content. However, some poorly implemented PDF text + extraction algorithms may fail to detect the text. - Many API functions have stricter parameter checking or expect keyword arguments were they previously did not. - Some deprecated functions in ``ocrmypdf.optimize`` were removed. -- The ``ocrmypdf.leptonica`` module is now deprecated. +- The ``ocrmypdf.leptonica`` module is now deprecated, due to difficulties with + the current strategy of ABI binding on newer platforms like Apple Silicon. + It will be removed and replaced, either by repackaging Leptonica as an + independent library using or using a different image processing library. - Continuous integration moved to GitHub Actions. - We no longer depend on ``pytest_helpers_namespace`` for testing. @@ -46,12 +49,15 @@ v12.0.0 way OCRmyPDF outputs its messages. - New plugin hook: ``filter_pdf_page``, for modifying individual PDF pages produced by OCRmyPDF. -- Using the new plugin hooks, it is now possible to run OCRmyPDF on alternative - execution environments that do not have interprocess semaphores, such as - AWS Lambda and Android Termux. +- OCRmyPDF now runs on nonstandard execution environments that do not have + interprocess semaphores, such as AWS Lambda and Android Termux. If the environment + does not have semaphores, OCRmyPDF will automatically select an alternate + process executor that does not use semaphores. - Continuous integration moved to GitHub Actions. - We now generate an ARM64-compatible Docker image alongside the x64 image. - Thanks to @andkrause for contributing the change and @0x326 for review comments. + Thanks to @andkrause for doing most of the work in a pull request several months + ago, which we were finally able to integrate now. Also thanks to @0x326 for + review comments. **Fixes** @@ -65,6 +71,7 @@ v12.0.0 - OCRmyPDF can now parse all of Tesseract version numbers, since several schemes have been in use. - Fixed an issue with parsing PDFs that contain images drawn at a scale of 0. (#761) +- Removed a frequently repeated message about disabling mmap. v11.7.3 ======= From 7b1e5b4f41a9ca487678779d696e5b5f032c816f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 26 Apr 2021 01:18:07 -0700 Subject: [PATCH 875/880] Fix "invalid version number" for untagged tesseract versions Fixes #770 --- src/ocrmypdf/_exec/tesseract.py | 1 + tests/test_validation.py | 7 +++++++ 2 files changed, 8 insertions(+) diff --git a/src/ocrmypdf/_exec/tesseract.py b/src/ocrmypdf/_exec/tesseract.py index 5dc5024b..119771ec 100644 --- a/src/ocrmypdf/_exec/tesseract.py +++ b/src/ocrmypdf/_exec/tesseract.py @@ -61,6 +61,7 @@ class TesseractVersion(StrictVersion): ^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch [-]? # optional hyphen separator (?:(alpha|beta|rc|dev)[.\-\ ]?(\d+)?)? # 5/prerelease, 6/prerelease_num + (?:-(\d+)-g[0-9a-f]+)? # untagged git version $ ''', re.VERBOSE | re.ASCII, diff --git a/tests/test_validation.py b/tests/test_validation.py index 6bec04e3..fd4d6fc2 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -237,6 +237,13 @@ def test_version_comparison(): need_version='4.0.0', version_parser=TesseractVersion, ) + vd.check_external_program( + program="tesseract", + package="tesseract", + version_checker=lambda: '4.1.1-rc2-25-g9707', + need_version='4.0.0', + version_parser=TesseractVersion, + ) with pytest.raises(MissingDependencyError): vd.check_external_program( program="dummy_fails", From 399b5548cae5227fddc822526e6307f7a85d31e7 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Mon, 26 Apr 2021 01:18:57 -0700 Subject: [PATCH 876/880] v12.0.1 release notes --- docs/release_notes.rst | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index d4b1341b..c527732a 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,11 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. +v12.0.1 +======= + +- Fix "invalid version number" for untagged tesseract versions (#770). + v12.0.0 ======= From 352f009c77a88e52449d893b643e40aa8466654e Mon Sep 17 00:00:00 2001 From: Matthias Braun Date: Sun, 9 May 2021 10:03:13 +0200 Subject: [PATCH 877/880] Fix code of French language pack in languages.rst (#776) * Fix code of French language pack in languages.rst * Update languages.rst Reviewed-by: jbarlow83 --- docs/languages.rst | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/docs/languages.rst b/docs/languages.rst index ed414a50..45dfac6f 100644 --- a/docs/languages.rst +++ b/docs/languages.rst @@ -12,9 +12,9 @@ languages ``, +After you have installed a language pack, you can use it with ``ocrmypdf -l ``, for example ``ocrmypdf -l spa``. For multilingual documents, you can specify all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French. English is assumed by default unless other language(s) are specified. @@ -35,8 +35,8 @@ Debian and Ubuntu users You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be -requested using either ``-l eng+fre`` (English and French) or -``-l eng -l fre``. +requested using either ``-l eng+fra`` (English and French) or +``-l eng -l fra``. Fedora users ============ @@ -51,8 +51,8 @@ Fedora users You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be -requested using either ``-l eng+fre`` (English and French) or -``-l eng -l fre``. +requested using either ``-l eng+fra`` (English and French) or +``-l eng -l fra``. macOS users =========== From 43e7765efd8582123432c516c18174a1995dd7c5 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Thu, 13 May 2021 23:24:54 -0700 Subject: [PATCH 878/880] Expand documentation for adding language packs Closes #777 --- docs/docker.rst | 33 +++++++++++++++++++++++++++++---- 1 file changed, 29 insertions(+), 4 deletions(-) diff --git a/docs/docker.rst b/docs/docker.rst index 788efbcd..16eccb7c 100644 --- a/docs/docker.rst +++ b/docs/docker.rst @@ -103,16 +103,41 @@ Adding languages to the Docker image By default the Docker image includes English, German, Simplified Chinese, French, Portuguese and Spanish, the most popular languages for OCRmyPDF users based on feedback. You may add other languages by creating a new -Dockerfile based on the public one: +Dockerfile based on the public one. .. code-block:: dockerfile FROM jbarlow83/ocrmypdf - # Add French - RUN apt install tesseract-ocr-fra + # Example: add Italian + RUN apt install tesseract-ocr-ita -You can also copy training data to ``/usr/share/tesseract-ocr//tessdata``. +To install language packs (training data) such as the +`tessdata_best `_ suite or +custom data, you first need to determine the version of Tesseract data files, which +may differ from the Tesseract program version. Use this command to determine the data +file version: + +.. code-block:: bash + + docker run -i --rm --entrypoint /bin/ls jbarlow83/ocrmypdf /usr/share/tesseract-ocr + +As of 2021, the data file version is probably ``4.00``. + +You can then add new data with either a Dockerfile: + +.. code-block:: dockerfile + + FROM jbarlow83/ocrmypdf + + # Example: add a tessdata_best file + COPY chi_tra_vert.traineddata /usr/share/tesseract-ocr//tessdata/ + +Alternately, you can copy training data into a Docker container as follows: + +.. code-block:: bash + + docker cp mycustomtraining.traineddata name_of_container:/usr/share/tesseract-ocr//tessdata/ Executing the test suite ======================== From 09c485bd888903c1013974e61dc6f130edaedb1f Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 18 May 2021 23:19:51 -0700 Subject: [PATCH 879/880] Convert harmless Leptonica exception to warning Closes Error when trying --remove-background on pdf #769 --- src/ocrmypdf/leptonica.py | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index a6757391..807560ff 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -192,11 +192,17 @@ class _LeptonicaErrorTrap_Queue: if 'Error' in output: if 'image file not found' in output: raise FileNotFoundError() - if 'pixWrite: stream not opened' in output: + elif 'pixWrite: stream not opened' in output: raise LeptonicaIOError() - if 'index not valid' in output: + elif 'index not valid' in output: raise IndexError() - raise LeptonicaError(output) + elif 'pixGetInvBackgroundMap: w and h must be >= 5' in output: + logger.warning( + "Leptonica attempted to remove background from a low resolution - " + "you may want to review in a PDF viewer" + ) + else: + raise LeptonicaError(output) return False @@ -656,6 +662,9 @@ class Pix(LeptonicaObject): bg_val=200, smooth_kernel=(2, 1), ): + if self.width < tile_size[0] or self.height < tile_size[1]: + logger.info("Skipped pixMaskedThreshOnBackgroundNorm on small image") + return self # Background norm doesn't work on color mapped Pix, so remove colormap target_pix = self.remove_colormap(lept.REMOVE_CMAP_BASED_ON_SRC) with _LeptonicaErrorTrap(): From c409fa58251a7a7bca0f4379ed323e9888400b11 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 18 May 2021 23:22:11 -0700 Subject: [PATCH 880/880] v12.0.2 release notes --- docs/release_notes.rst | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/release_notes.rst b/docs/release_notes.rst index c527732a..2d171340 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -12,6 +12,15 @@ may be unreliable. Use the API to depend on precise behavior. The public API may be useful in scripts that launch OCRmyPDF processes or that wish to use some of its features for working with PDFs. + +v12.0.2 +======= + +- Fix exception thrown when using ``--remove-background`` on files containing small + images (#769). +- Improve documentation for description of adding language packs to the Docker image + and corrected name of French language pack. + v12.0.1 =======