From 8de7b05fb9d16e21e76efaa1508dec1368fff42e Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 1 Jul 2026 00:10:57 -0700 Subject: [PATCH] Fix flaky tagged-PDF skip-text test under Ghostscript 10.x The test asserted that --mode skip preserves the structure tree, but with the default --output-type auto the output runs through Ghostscript PDF/A conversion, which discards /StructTreeRoot on Ghostscript 10.x (9.x kept it). This failed on macOS CI and locally while passing on the Ubuntu runners' Ghostscript 9.55. Pin the test to --output-type pdf so it exercises OCRmyPDF's own structure-tree handling without the version-dependent GS step, and document the caveat in advanced.md. --- docs/advanced.md | 10 ++++++++++ tests/test_tagged.py | 6 +++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/docs/advanced.md b/docs/advanced.md index fd60a21f..5fc5c66a 100644 --- a/docs/advanced.md +++ b/docs/advanced.md @@ -135,6 +135,16 @@ OCRmyPDF cannot rebuild a structure tree to match newly recognized text. When the structure tree no longer corresponds to the page content, so it is discarded. `--mode skip` leaves text pages untouched, so their structural markup is preserved. +:::{note} +Preservation under `--mode skip` only holds when the output is not converted to +PDF/A. PDF/A conversion is performed by Ghostscript, and Ghostscript 10.x discards +the structure tree during conversion (Ghostscript 9.x preserved it). Because the +default `--output-type auto` may fall back to Ghostscript, use +`--output-type pdf` if you need to guarantee that a Tagged PDF's structural markup +survives. For best results, install veraPDF so that speculative PDF/A +conversion can sidestep this issue entirely in most real cases. +::: + ### Time and image size limits By default, OCRmyPDF permits tesseract to run for three minutes (180 diff --git a/tests/test_tagged.py b/tests/test_tagged.py index a661891b..16ab3a1c 100644 --- a/tests/test_tagged.py +++ b/tests/test_tagged.py @@ -54,10 +54,14 @@ def test_tagged_pdf_mode_ignore_with_skip_text(resources, outpdf, caplog): outpdf, tagged_pdf_mode='ignore', skip_text=True, # Tagged PDF has text, so skip pages with text + # output_type=pdf avoids the Ghostscript PDF/A step, whose treatment of + # the structure tree is version-dependent (Ghostscript >= 10 discards it, + # 9.x preserves it). We only want to assert OCRmyPDF's own behavior here. + output_type='pdf', plugins=['tests/plugins/tesseract_noop.py'], ) assert 'structural markup' in caplog.text - # skip-text leaves the text pages untouched, so the structure tree remains valid + # skip-text leaves the text pages untouched, so OCRmyPDF keeps the structure tree with pikepdf.open(outpdf) as pdf: assert Name.StructTreeRoot in pdf.Root