+15
-15
@@ -91,7 +91,7 @@ FILE_INPUT_PDF="$1"
|
||||
|
||||
|
||||
# check if the required utilities are installed
|
||||
echo "Checking if all dependencies are installed"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Checking if all dependencies are installed"
|
||||
! command -v gs > /dev/null && echo "Please install ghostcript. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||
! command -v identify > /dev/null && echo "Please install ImageMagick. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||
! command -v pdfimages > /dev/null && echo "Please install xpdf. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||
@@ -112,7 +112,7 @@ FILE_OUTPUT_PDFA="${tmp}/ocred-pdfa.pdf" # name of the final PDF/A file
|
||||
FILE_VALIDATION_LOG="${tmp}/pdf_validation.log" # log file containing the results of the validation of the PDF/A file
|
||||
|
||||
# delete tmp files
|
||||
echo "Removing old temporary files"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Removing old temporary files"
|
||||
rm -r -f "${tmp}"
|
||||
mkdir -p "${tmp}"
|
||||
|
||||
@@ -120,12 +120,12 @@ mkdir -p "${tmp}"
|
||||
|
||||
|
||||
# get the size of each pdf page (width / height) in pt (inch*72)
|
||||
echo "Input file: Extracting size of each page (in pt)"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Input file: Extracting size of each page (in pt)"
|
||||
! identify -format "%w %h\n" "$FILE_INPUT_PDF" > "$FILE_SIZE_PAGES" \
|
||||
&& echo "Could not get size of PDF pages. Exiting..." && exit $EXIT_BAD_INPUT_FILE
|
||||
sed -I "" '/^$/d' "$FILE_SIZE_PAGES" # removing empty lines (last one should be)
|
||||
numpages=`cat "$FILE_SIZE_PAGES" | wc -l | sed 's/^ *//g'`
|
||||
echo "Input file: The file has $numpages pages"
|
||||
[ $VERBOSITY -ge $LOG_INFO ] && echo "Input file: The file has $numpages pages"
|
||||
|
||||
# Itterate the pages of the input pdf file
|
||||
cpt="1"
|
||||
@@ -133,7 +133,7 @@ while read pageSize ; do
|
||||
|
||||
# add leading zeros to the page number
|
||||
page=`printf "%04d" $cpt`
|
||||
echo "Processing page $page"
|
||||
[ $VERBOSITY -ge $LOG_INFO ] && echo "Processing page $page"
|
||||
|
||||
# create the name of the required file
|
||||
curOrigImg="$tmp/${page}_Image" # original image available in the current PDF page
|
||||
@@ -141,7 +141,7 @@ while read pageSize ; do
|
||||
curHocr="$tmp/$page.hocr" # hocr file to be generated by the OCR SW for the current page
|
||||
curOCRedPDF="$tmp/${page}-ocred.pdf" # PDF file containing the image + the OCRed text for the current page
|
||||
|
||||
echo "Page $page: Computing embedded image resolution"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Computing embedded image resolution"
|
||||
# get width / height of PDF page
|
||||
heightPDF=`echo $pageSize | cut -f1 -d" "`
|
||||
widthPDF=`echo $pageSize | cut -f2 -d" "`
|
||||
@@ -177,13 +177,13 @@ while read pageSize ; do
|
||||
curImgPixmapClean="$tmp/$page.cleaned.$ext"
|
||||
|
||||
# extract current page as image with right orientation and resoltution
|
||||
echo "Page $page: Extracting image as $ext file (${dpi} dpi)"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Extracting image as $ext file (${dpi} dpi)"
|
||||
! pdftoppm -f $page -l $page -r $dpi $opt "$FILE_INPUT_PDF" > "$curImgPixmap" \
|
||||
&& echo "Could not extract page $page as $ext from $FILE_INPUT_PDF. Exiting..." && exit $EXIT_OTHER_ERROR
|
||||
|
||||
# if requested deskew image (without changing its size in pixel)
|
||||
if [ "$PREPROCESS_DESKEW" -eq "1" ]; then
|
||||
echo "Page $page: Deskewing image"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Deskewing image"
|
||||
! convert "$curImgPixmap" -deskew 40% -gravity center -extent ${heightCurOrigImg01}x${widthICurOrigImg01} "$curImgPixmapDeskewed" \
|
||||
&& echo "Could not deskew \"$curImgPixmap\". Exiting..." && exit $EXIT_OTHER_ERROR
|
||||
else
|
||||
@@ -192,7 +192,7 @@ while read pageSize ; do
|
||||
|
||||
# if requested clean image with unpaper to get better OCR results
|
||||
if [ "$PREPROCESS_CLEAN" -eq "1" ]; then
|
||||
echo "Page $page: Cleaning image with unpaper"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Cleaning image with unpaper"
|
||||
! unpaper --dpi $dpi --mask-scan-size 100 \
|
||||
--no-deskew --no-grayfilter --no-blackfilter --no-mask-center --no-border-align \
|
||||
"$curImgPixmapDeskewed" "$curImgPixmapClean" 1> /dev/null \
|
||||
@@ -202,13 +202,13 @@ while read pageSize ; do
|
||||
fi
|
||||
|
||||
# perform OCR
|
||||
echo "Page $page: Performing OCR"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Performing OCR"
|
||||
! tesseract -l "$LAN" "$curImgPixmapClean" "$curHocr" hocr 1> /dev/null 2> /dev/null \
|
||||
&& echo "Could not OCR file \"$curImgPixmapClean\". Exiting..." && exit $EXIT_OTHER_ERROR
|
||||
mv "$curHocr.html" "$curHocr"
|
||||
|
||||
# embed text and image to new pdf file
|
||||
echo "Page $page: Embedding text in PDF"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF"
|
||||
if [ "$PREPROCESS_CLEANINPDF" -eq "1" ]; then
|
||||
image4finalPDF="$curImgPixmapClean"
|
||||
else
|
||||
@@ -235,7 +235,7 @@ done < "$FILE_SIZE_PAGES"
|
||||
|
||||
|
||||
# concatenate all pages
|
||||
echo "Output file: Concatenating all pages"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Concatenating all pages"
|
||||
! pdftk $FILES_OCRed_PDFS cat output "$FILE_OUTPUT_PDF" \
|
||||
&& echo "Could not concatenate individual PDF pages (\"$FILES_OCRed_PDFS\") to one file. Exiting..." && exit $EXIT_OTHER_ERROR
|
||||
|
||||
@@ -245,14 +245,14 @@ echo "Output file: Concatenating all pages"
|
||||
# the name of the file may be used as title
|
||||
|
||||
# convert the pdf file to match PDF/A format
|
||||
echo "Output file: Conversion to PDF/A"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Converting to PDF/A"
|
||||
! gs -dQUIET -dPDFA -dBATCH -dNOPAUSE -dUseCIEColor \
|
||||
-sProcessColorModel=DeviceCMYK -sDEVICE=pdfwrite -sPDFACompatibilityPolicy=2 \
|
||||
-sOutputFile=$FILE_OUTPUT_PDFA "$FILE_OUTPUT_PDF" \
|
||||
-sOutputFile=$FILE_OUTPUT_PDFA "$FILE_OUTPUT_PDF" 1> /dev/null 2> /dev/null \
|
||||
&& echo "Could not convert PDF file \"$FILE_OUTPUT_PDF\" to PDF/A. Exiting..." && exit $EXIT_OTHER_ERROR
|
||||
|
||||
# validate generated pdf file (compliance to PDF/A)
|
||||
echo "Output file: Checking compliance to PDF/A standard"
|
||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Checking compliance to PDF/A standard"
|
||||
java -jar /root/jhove-1_9/jhove/bin/JhoveApp.jar -m PDF-hul "$FILE_OUTPUT_PDFA" > "$FILE_VALIDATION_LOG"
|
||||
grep -i "Status|Message" "$FILE_VALIDATION_LOG" # summary of the validation
|
||||
# check if the validation was successful
|
||||
|
||||
Reference in New Issue
Block a user