diff --git a/OCRmyPDF.sh b/OCRmyPDF.sh index c338a1ff..c03f7043 100644 --- a/OCRmyPDF.sh +++ b/OCRmyPDF.sh @@ -1,29 +1,36 @@ #!/bin/sh +FILE_INPUT_PDF="$1" + VERSION="alpha0" -# Initialization of configuration parameters -FILE_INPUT_PDF="$1" -LOG_ERR="0" # 0=only error messages -LOG_INFO="1" # 1=error messages and some infos -LOG_DEBUG="2" # 2=debug level logging -VERBOSITY="$LOG_ERR" -LAN="eng" # language of the PDF file (required to get good OCR results) -KEEP_TMP="0" # do not delete the temporary files -IMG_PREPROCESS="0" # 0=none - # 1=only deskew - # 2=full preprocessing for OCR, but only deskew in final PDF - # 3=full preprocessing +# Initialization of constants +EXIT_BAD_INPUT_FILE="1" # possible exit codes +EXIT_MISSING_DEPENDENCY="2" +EXIT_INVALID_OUPUT_PDFA="3" +LOG_ERR="0" # 0=only error messages +LOG_INFO="1" # 1=error messages and some infos +LOG_DEBUG="2" # 2=debug level logging + +# Initialization the configuration parameters with default values +VERBOSITY="$LOG_ERR" # default verbosity level +LAN="eng" # default language of the PDF file (required to get good OCR results) +KEEP_TMP="0" # do not delete the temporary files (default) +PREPROCESS_DESKEW="1" # 0=no, 1=yes +PREPROCESS_CLEAN="2" # 0=no, + # 1=clean image to improve OCR, but do not put cleaned image in final PDF + # 2=clean image to improve OCR, AND put it in final PDF # check if the required utilities are installed -! which gs > /dev/null && echo "Please install ghostcript. Exiting..." && exit 1 -! which identify > /dev/null && echo "Please install ImageMagick. Exiting..." && exit 1 -! which pdfimages > /dev/null && echo "Please install xpdf. Exiting..." && exit 1 -! which pdftoppm > /dev/null && echo "Please install xpdf. Exiting..." && exit 1 -! which pdftk > /dev/null && echo "Please install pdftk. Exiting..." && exit 1 -! which unpaper > /dev/null && echo "Please install unpaper. Exiting..." && exit 1 -! which tesseract > /dev/null && echo "Please install tesseract and tesseract-data. Exiting..." && exit 1 - +echo "Checking if all dependencies are installed" +! command -v gs > /dev/null && echo "Please install ghostcript. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v identify > /dev/null && echo "Please install ImageMagick. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v pdfimages > /dev/null && echo "Please install xpdf. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v pdftoppm > /dev/null && echo "Please install xpdf. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v pdftk > /dev/null && echo "Please install pdftk. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v unpaper > /dev/null && echo "Please install unpaper. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v tesseract > /dev/null && echo "Please install tesseract and tesseract-data. Exiting..." && exit $EXIT_MISSING_DEPENDENCY +! command -v java > /dev/null && echo "Please install java. Exiting..." && exit $EXIT_MISSING_DEPENDENCY @@ -36,6 +43,7 @@ FILE_OUTPUT_PDFA="${tmp}/ocred-pdfa.pdf" # name of the final PDF/A file FILE_VALIDATION_LOG="${tmp}/pdf_validation.log" # log file containing the results of the validation of the PDF/A file # delete tmp files +echo "Removing old temporary files" rm -r -f "${tmp}" mkdir -p "${tmp}" @@ -55,7 +63,8 @@ while read pageSize ; do # add leading zeros to the page number page=`printf "%04d" $cpt` - + echo "Processing page $page" + # create the name of the required file curOrigImg="$tmp/${page}_Image" # original image available in the current PDF page # (the image file may have a different orientation than in the pdf file) @@ -72,19 +81,21 @@ while read pageSize ; do pdfimages -f $page -l $page -j "$FILE_INPUT_PDF" "$curOrigImg" 1>&2 # count number of extracted images nbImg=`ls -1 "$curOrigImg"* | wc -l` - [ $nbImg -ne "1" ] && echo "Not exactly 1 image on page $page. Exiting..." && exit 1 + [ $nbImg -ne "1" ] && echo "Not exactly 1 image on page $page. Exiting..." && exit $EXIT_BAD_INPUT_FILE - # Get the characteristic of the extracted image + # Get characteristics of the extracted image curOrigImg01=`ls -1 "$curOrigImg"*` propCurOrigImg01=`identify -format "%w %h %[colorspace]" "$curOrigImg01"` heightCurOrigImg01=`echo "$propCurOrigImg01" | cut -f1 -d" "` widthICurOrigImg01=`echo "$propCurOrigImg01" | cut -f2 -d" "` colorspaceCurOrigImg01=`echo "$propCurOrigImg01" | cut -f3 -d" "` # compute the resolution of the whole page (taking into account all images) - dpi=$(($heightCurOrigImg01*72/$heightPDF)) + dpi_x=$(($widthICurOrigImg01*72/$widthPDF)) + dpi_y=$(($heightCurOrigImg01*72/$heightPDF)) + [ "$dpi_x" -ne "$dpi_y" ] && echo "X/Y Resolutions not equal (Not supported currently). Exiting..." && exit $EXIT_BAD_INPUT_FILE + dpi="$dpi_x" # Identify if page image should be saved as ppm (color) or pgm (gray) - echo "Page $page: Extracting image as ppm/pgm (${dpi} dpi)" ext="ppm" opt="" if [ $colorspaceCurOrigImg01 == "Gray" ]; then @@ -92,17 +103,31 @@ while read pageSize ; do opt="-gray" fi curImgPixmap="$tmp/$page.$ext" - curImgPixmapClean="$tmp/$page.for-ocr.$ext" + curImgPixmapDeskewed="$tmp/$page.deskewed.$ext" + curImgPixmapClean="$tmp/$page.cleaned.$ext" # extract current page as image with right orientation and resoltution + echo "Page $page: Extracting image as $ext file (${dpi} dpi)" pdftoppm -f $page -l $page -r $dpi $opt "$FILE_INPUT_PDF" > "$curImgPixmap" - # improve quality of the image with unpaper to get better OCR results - echo "Page $page: Preprocessing image with unpaper" - unpaper --dpi $dpi --mask-scan-size 100 \ - --no-grayfilter --no-blackfilter --no-mask-center --no-border-align \ - "$curImgPixmap" "$curImgPixmapClean" 1> /dev/null - + # if requested deskew image (without changing its size in pixel) + if [ "$PREPROCESS_DESKEW" -eq "1" ]; then + echo "Page $page: Deskewing image" + convert "$curImgPixmap" -deskew 40% -gravity center -extent ${heightCurOrigImg01}x${widthICurOrigImg01} "$curImgPixmapDeskewed" + else + cp "$curImgPixmap" "$curImgPixmapDeskewed" + fi + + # if requested clean image with unpaper to get better OCR results + if [ "$PREPROCESS_CLEAN" -ge "1" ]; then + echo "Page $page: Cleaning image with unpaper" + unpaper --dpi $dpi --mask-scan-size 100 \ + --no-deskew --no-grayfilter --no-blackfilter --no-mask-center --no-border-align \ + "$curImgPixmapDeskewed" "$curImgPixmapClean" 1> /dev/null + else + cp "$curImgPixmapDeskewed" "$curImgPixmapClean" + fi + # perform OCR echo "Page $page: Performing OCR" tesseract -l "$LAN" "$curImgPixmapClean" "$curHocr" hocr 1> /dev/null 2> /dev/null @@ -110,7 +135,12 @@ while read pageSize ; do # embed text and image to new pdf file echo "Page $page: Embedding text in PDF" - python hocrTransform.py -r $dpi -i "$curImgPixmapClean" "$curHocr" "$curOCRedPDF" + if [ "$PREPROCESS_CLEAN" -eq "2" ]; then + image4finalPDF="$curImgPixmapClean" + else + image4finalPDF="$curImgPixmapDeskewed" + fi + python hocrTransform.py -r $dpi -i "$image4finalPDF" "$curHocr" "$curOCRedPDF" # delete temporary files created for the current page # to avoid using to much disk space in case of PDF files having many pages @@ -118,6 +148,7 @@ while read pageSize ; do rm "$curOrigImg"*.* rm "$curHocr" rm "$curImgPixmap" + rm "$curImgPixmapDeskewed" rm "$curImgPixmapClean" fi @@ -135,6 +166,7 @@ pdftk $FILES_OCRed_PDFS cat output "$FILE_OUTPUT_PDF" # insert metadata (copy metadata from input file) #echo "Output file: Inserting metadata" # TODO (may work with pdftk update_info) +# the name of the file may be used as title # convert the pdf file to match PDF/A format echo "Output file: Conversion to PDF/A" @@ -156,10 +188,12 @@ grep -i "Status.*Not well-formed" "$FILE_VALIDATION_LOG" && pdf_valid=0 - + # delete temporary files if [ $KEEP_TMP -eq 0 ]; then rm $FILES_OCRed_PDFS rm "$FILE_SIZE_PAGES" rm "$FILE_OUTPUT_PDF" -fi \ No newline at end of file +fi + +exit 0 \ No newline at end of file