Fix handling of DPI for rare case of JPEG recompression after deskew/clean

This test is exercised by page 4 of multipage.pdf. If all images are
JPEGs, and one of deskew/clean removes DPI information, make sure that
we can get the right information back and that the DPI stays square.
This commit is contained in:
James R. Barlow
2016-07-29 01:34:52 -07:00
parent 8f77576dc4
commit fef35e4eb2
4 changed files with 19 additions and 11 deletions
+1 -1
View File
@@ -9,6 +9,7 @@ Download software here: https://github.com/jbarlow83/OCRmyPDF/tags
v4.2 (planned):
===============
- Improved support for PDFs containing images with "non-square" pixel aspect ratios, such as 200x100 DPI.
- Support for some older programs dropped - should not be a problem for anyone except Ubuntu 12.04 LTS users
+ Ghostscript 9.15 is now required
@@ -16,7 +17,6 @@ v4.2 (planned):
- Ghostscript now runs in "safer" mode where possible
v4.1.4:
=======
+13 -5
View File
@@ -738,12 +738,20 @@ def select_image_for_pdf(
image = next(ii for ii in infiles if ii.endswith(image_suffix))
pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock)
if all(image['enc'] == 'jpeg' for image in pageinfo['images']):
if all(orig_image['enc'] == 'jpeg' for orig_image in pageinfo['images']):
# If all images were JPEGs originally, produce a JPEG as output
im = Image.open(image)
fallback_dpi = get_page_dpi(pageinfo)
dpi = im.info.get('dpi', fallback_dpi)
dpi = round(dpi[0]), round(dpi[1]) # Pillow requires integer DPI
# At this point the image should be a .png, but deskew, unpaper might
# have removed the DPI information. In this case, fall back to square
# DPI used to rasterize. When the preview image was rasterized, it
# was also converted to square resolution, which is what we want to
# give tesseract, so keep it square.
fallback_dpi = get_page_square_dpi(pageinfo)
dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi))
# Pillow requires integer DPI
dpi = round(dpi[0]), round(dpi[1])
im.save(output_file, format='JPEG', dpi=dpi)
else:
re_symlink(image, output_file)
@@ -871,7 +879,7 @@ def add_text_layer(
y2 = page_image.mediaBox.getUpperRight_y()
# Rotation occurs about the page's (0, 0). Most pages will have the media
# box at (0, 0) will all content in the first quadrant but some cropped
# box at (0, 0) with all content in the first quadrant but some cropped
# files may have an offset mediabox. We translate the page so that its
# bottom left corner after rotation is pinned to (0, 0) with the image
# in the first quadrant.
+2 -2
View File
@@ -14,9 +14,9 @@ In some cases they were converted from one image format to another without other
+=====================+================================================================================+
| c02-22.pdf | `Project Gutenberg`_, Adventures of Huckleberry Finn, page 22 |
+---------------------+--------------------------------------------------------------------------------+
| congress.jpg | `US Congressional Records`_ |
| congress.jpg | `US Congressional Records`_ (Public Domain) |
+---------------------+--------------------------------------------------------------------------------+
| graph.pdf | `Wikimedia: Pandas text analysis.png`_ |
| graph.pdf | `Wikimedia: Pandas text analysis.png`_ (Public Domain) |
+---------------------+--------------------------------------------------------------------------------+
| lichtenstein.pdf | `Wikimedia: JPEG2000 Lichtenstein`_ (Creative Commons BY-SA 3.0) |
+---------------------+--------------------------------------------------------------------------------+
+3 -3
View File
@@ -565,8 +565,8 @@ def test_algo4():
@pytest.mark.parametrize('renderer', [
'hocr'])
def test_non_square_resolution(renderer):
'hocr']) # tesseract cannot pass this test yet
def test_non_square_resolution(renderer, spoof_tesseract_cache):
# Confirm input image is non-square resolution
in_pageinfo = pdf_get_all_pageinfo(_infile('aspect.pdf'))
assert in_pageinfo[0]['xres'] != in_pageinfo[0]['yres']
@@ -574,7 +574,7 @@ def test_non_square_resolution(renderer):
out = 'aspect_%s.pdf' % renderer
check_ocrmypdf(
'aspect.pdf', out,
'--pdf-renderer', renderer)
'--pdf-renderer', renderer, env=spoof_tesseract_cache)
out_pageinfo = pdf_get_all_pageinfo(_outfile(out))