From 18cc6dcdc8f0063bc738146b043c95253e7a3e7d Mon Sep 17 00:00:00 2001 From: Anthony Stirling <77850077+Frooodle@users.noreply.github.com> Date: Thu, 2 Jul 2026 16:39:04 +0100 Subject: [PATCH] Add true-removal redaction pipeline with verification and font-type coverage --- .../api/security/ManualRedactionService.java | 184 +-- .../api/security/RedactController.java | 73 +- .../api/security/RedactExecuteService.java | 24 +- .../SPDF/pdf/redaction/CatalogScrubber.java | 613 ++++++++ .../SPDF/pdf/redaction/RedactionPipeline.java | 1246 +++++++++++++++++ .../RedactionVerificationFailedException.java | 17 + .../security/ManualRedactionServiceTest.java | 50 +- .../api/security/RedactControllerTest.java | 17 +- .../api/security/RedactionPdfVarietyTest.java | 569 ++++++++ .../RedactionPipelineIntegrationTest.java | 174 +++ .../pdf/redaction/RedactionPipelineTest.java | 529 +++++++ .../test/resources/redaction/test_pdf_1.pdf | Bin 0 -> 1105 bytes .../test/resources/redaction/type3_dejavu.pdf | Bin 0 -> 26033 bytes testing/cucumber/features/redact.feature | 109 ++ .../features/steps/step_definitions.py | 76 +- 15 files changed, 3556 insertions(+), 125 deletions(-) create mode 100644 app/core/src/main/java/stirling/software/SPDF/pdf/redaction/CatalogScrubber.java create mode 100644 app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionPipeline.java create mode 100644 app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionVerificationFailedException.java create mode 100644 app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactionPdfVarietyTest.java create mode 100644 app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineIntegrationTest.java create mode 100644 app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineTest.java create mode 100644 app/core/src/test/resources/redaction/test_pdf_1.pdf create mode 100644 app/core/src/test/resources/redaction/type3_dejavu.pdf create mode 100644 testing/cucumber/features/redact.feature diff --git a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/ManualRedactionService.java b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/ManualRedactionService.java index 15a233f1f7..c8a4c5776d 100644 --- a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/ManualRedactionService.java +++ b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/ManualRedactionService.java @@ -5,8 +5,11 @@ import java.io.IOException; import java.util.ArrayList; import java.util.Collections; import java.util.HashMap; +import java.util.LinkedHashSet; import java.util.List; import java.util.Map; +import java.util.Set; +import java.util.regex.Pattern; import org.apache.pdfbox.pdmodel.PDDocument; import org.apache.pdfbox.pdmodel.PDPage; @@ -22,6 +25,7 @@ import lombok.extern.slf4j.Slf4j; import stirling.software.SPDF.model.PDFText; import stirling.software.SPDF.model.api.security.ManualRedactPdfRequest; import stirling.software.SPDF.pdf.parser.PageImageLocator; +import stirling.software.SPDF.pdf.redaction.RedactionPipeline; import stirling.software.common.model.api.security.RedactionArea; import stirling.software.common.util.GeneralUtils; import stirling.software.common.util.PdfUtils; @@ -42,11 +46,15 @@ class ManualRedactionService { // Area and page redaction // ----------------------------------------------------------------------- - void redactAreas(List redactionAreas, PDDocument document, PDPageTree allPages) + AreaRedactionResult redactAreas( + List redactionAreas, PDDocument document, PDPageTree allPages) throws IOException { + Set capturedStrings = new LinkedHashSet<>(); + Map> rectsByPage = new HashMap<>(); + if (redactionAreas == null || redactionAreas.isEmpty()) { - return; + return new AreaRedactionResult(rectsByPage); } Map> redactionsByPage = new HashMap<>(); @@ -74,52 +82,49 @@ class ManualRedactionService { continue; } - PDPage page = allPages.get(pageNumber - 1); + int pageIndex = pageNumber - 1; + PDPage page = allPages.get(pageIndex); + float pageHeight = page.getBBox().getHeight(); - try (PDPageContentStream contentStream = - new PDPageContentStream( - document, page, PDPageContentStream.AppendMode.APPEND, true, true)) { - - contentStream.saveGraphicsState(); - for (RedactionArea redactionArea : areasForPage) { - Color redactColor = decodeOrDefault(redactionArea.getColor()); - - contentStream.setNonStrokingColor(redactColor); - - float x = redactionArea.getX().floatValue(); - float y = redactionArea.getY().floatValue(); - float width = redactionArea.getWidth().floatValue(); - float height = redactionArea.getHeight().floatValue(); - - float pdfY = page.getBBox().getHeight() - y - height; - - contentStream.addRect(x, pdfY, width, height); - contentStream.fill(); - } - contentStream.restoreGraphicsState(); + List rects = new ArrayList<>(); + Color overlayColor = Color.BLACK; + for (RedactionArea area : areasForPage) { + float x = area.getX().floatValue(); + float y = area.getY().floatValue(); + float width = area.getWidth().floatValue(); + float height = area.getHeight().floatValue(); + // Request coords are top-left origin; convert to PDF user space (bottom-left). + float pdfY = pageHeight - y - height; + rects.add(new PDRectangle(x, pdfY, width, height)); + overlayColor = decodeOrDefault(area.getColor()); } + + // Physically drop intersecting glyphs and draw the overlay rectangle over the area. + Map> singlePage = new HashMap<>(); + singlePage.put(pageIndex, rects); + RedactionPipeline.RedactionResult result = + RedactionPipeline.redactAreas(document, singlePage, overlayColor); + capturedStrings.addAll(result.getCapturedStrings()); + rectsByPage.put(pageIndex, rects); } + + log.debug( + "Manual area redaction captured {} text run(s) across {} page(s)", + capturedStrings.size(), + rectsByPage.size()); + return new AreaRedactionResult(rectsByPage); } - void redactPages(ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages) + List redactPages( + ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages) throws IOException { Color redactColor = decodeOrDefault(request.getPageRedactionColor()); - List pageNumbers = getPageNumbers(request, allPages.getCount()); + List pageIndexes = getPageNumbers(request, allPages.getCount()); - for (Integer pageNumber : pageNumbers) { - PDPage page = allPages.get(pageNumber); - - try (PDPageContentStream contentStream = - new PDPageContentStream( - document, page, PDPageContentStream.AppendMode.APPEND, true, true)) { - contentStream.setNonStrokingColor(redactColor); - - PDRectangle box = page.getBBox(); - contentStream.addRect(0, 0, box.getWidth(), box.getHeight()); - contentStream.fill(); - } - } + // Whole-page wipe: drop the content stream, resources and annotations, then fill. + RedactionPipeline.redactWholePages(document, pageIndexes, redactColor); + return new ArrayList<>(pageIndexes); } // ----------------------------------------------------------------------- @@ -298,7 +303,9 @@ class ManualRedactionService { String colorString, float customPadding, Boolean convertToImage, - boolean isTextRemovalMode) + boolean isTextRemovalMode, + Set literalTargets, + List patterns) throws IOException { List allFoundTexts = new ArrayList<>(); @@ -309,74 +316,77 @@ class ManualRedactionService { if (!allFoundTexts.isEmpty()) { Color redactColor = decodeOrDefault(colorString); redactFoundText(document, allFoundTexts, customPadding, redactColor, isTextRemovalMode); - cleanDocumentMetadata(document); } + byte[] outputBytes; if (Boolean.TRUE.equals(convertToImage)) { try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) { - cleanDocumentMetadata(convertedPdf); - - TempFile tempOut = tempFileManager.createManagedTempFile(".pdf"); - try { - convertedPdf.save(tempOut.getFile()); - } catch (IOException e) { - tempOut.close(); - throw e; - } - - log.info( - "Redaction finalized (image mode): {} pages ➜ {} KB", - convertedPdf.getNumberOfPages(), - tempOut.getFile().length() / 1024); - - return tempOut; + // Convert-to-image physically removes all text, so verification is a plain save. + outputBytes = + RedactionPipeline.finalize( + convertedPdf, Collections.emptySet(), Collections.emptyList()); } + } else { + // True-removal pass: physically strip matched glyph bytes from every content stream, + // then scrub catalog carriers, verify, and rasterise affected pages on any leak. + RedactionPipeline.redactLiteralTerms(document, literalTargets, patterns); + outputBytes = RedactionPipeline.finalize(document, literalTargets, patterns); } + return writeBytes(outputBytes, document.getNumberOfPages()); + } + + /** + * Finalize a manual area/page redaction. The overlay rectangles are already drawn and the + * intersecting glyphs already dropped. Verification is region-based: each redaction rectangle + * is re-scanned and must be empty (whole-page wipes carry no rects and are guaranteed clean). + */ + TempFile finalizeManual( + PDDocument document, + Map> rectsByPage, + Boolean convertToImage) + throws IOException { + + byte[] outputBytes; + if (Boolean.TRUE.equals(convertToImage)) { + try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) { + outputBytes = + RedactionPipeline.finalize( + convertedPdf, Collections.emptySet(), Collections.emptyList()); + } + } else { + outputBytes = RedactionPipeline.finalizeAreas(document, rectsByPage); + } + + return writeBytes(outputBytes, document.getNumberOfPages()); + } + + private TempFile writeBytes(byte[] outputBytes, int pageCount) throws IOException { TempFile tempOut = tempFileManager.createManagedTempFile(".pdf"); try { - document.save(tempOut.getFile()); + java.nio.file.Files.write(tempOut.getFile().toPath(), outputBytes); } catch (IOException e) { tempOut.close(); throw e; } - log.info( - "Redaction finalized: {} pages ➜ {} KB", - document.getNumberOfPages(), - tempOut.getFile().length() / 1024); - + log.info("Redaction finalized: {} pages -> {} KB", pageCount, outputBytes.length / 1024); return tempOut; } - private void cleanDocumentMetadata(PDDocument document) { - try { - var documentInfo = document.getDocumentInformation(); - if (documentInfo != null) { - documentInfo.setAuthor(null); - documentInfo.setSubject(null); - documentInfo.setKeywords(null); - documentInfo.setModificationDate(java.util.Calendar.getInstance()); - log.debug("Cleaned document metadata for security"); - } - - if (document.getDocumentCatalog() != null) { - try { - document.getDocumentCatalog().setMetadata(null); - } catch (Exception e) { - log.debug("Could not clear XMP metadata: {}", e.getMessage()); - } - } - - } catch (Exception e) { - log.warn("Failed to clean document metadata: {}", e.getMessage()); - } - } - // ----------------------------------------------------------------------- // Utilities // ----------------------------------------------------------------------- + /** Redaction rectangles (per 0-based page index) applied by a manual area pass. */ + static final class AreaRedactionResult { + final Map> rectsByPage; + + AreaRedactionResult(Map> rectsByPage) { + this.rectsByPage = rectsByPage; + } + } + static Color decodeOrDefault(String hex) { if (hex == null) { return Color.BLACK; diff --git a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactController.java b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactController.java index 127b436306..98e2c73642 100644 --- a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactController.java +++ b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactController.java @@ -1,9 +1,15 @@ package stirling.software.SPDF.controller.api.security; import java.io.IOException; +import java.util.Arrays; +import java.util.Collections; +import java.util.LinkedHashSet; import java.util.List; import java.util.Map; import java.util.Objects; +import java.util.Set; +import java.util.regex.Pattern; +import java.util.stream.Collectors; import org.apache.pdfbox.pdmodel.PDDocument; import org.apache.pdfbox.pdmodel.PDPageTree; @@ -29,13 +35,13 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.ImageBox; import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyle; import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange; import stirling.software.SPDF.model.api.security.RedactPdfRequest; +import stirling.software.SPDF.pdf.redaction.RedactionPipeline; import stirling.software.common.annotations.AutoJobPostMapping; import stirling.software.common.annotations.api.SecurityApi; import stirling.software.common.enumeration.ResourceWeight; import stirling.software.common.model.api.security.RedactionArea; import stirling.software.common.service.CustomPDFDocumentFactory; import stirling.software.common.util.ExceptionUtils; -import stirling.software.common.util.PdfUtils; import stirling.software.common.util.TempFile; import stirling.software.common.util.TempFileManager; import stirling.software.common.util.WebResponseUtils; @@ -93,33 +99,25 @@ public class RedactController { throws IOException { MultipartFile file = request.getFileInput(); + String filename = + removeFileExtension( + Objects.requireNonNull( + Filenames.toSimpleFileName(file.getOriginalFilename()))) + + "_redacted.pdf"; try (PDDocument document = pdfDocumentFactory.load(file)) { PDPageTree allPages = document.getDocumentCatalog().getPages(); + // Whole-page wipes drop content outright (guaranteed clean); area redactions drop + // intersecting glyphs, draw an overlay, and are verified per-rectangle at finalize. manualRedactionService.redactPages(request, document, allPages); - manualRedactionService.redactAreas(request.getRedactions(), document, allPages); + ManualRedactionService.AreaRedactionResult areaResult = + manualRedactionService.redactAreas(request.getRedactions(), document, allPages); - if (Boolean.TRUE.equals(request.getConvertPDFToImage())) { - try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) { - return WebResponseUtils.pdfDocToWebResponse( - convertedPdf, - removeFileExtension( - Objects.requireNonNull( - Filenames.toSimpleFileName( - file.getOriginalFilename()))) - + "_redacted.pdf", - tempFileManager); - } - } - - return WebResponseUtils.pdfDocToWebResponse( - document, - removeFileExtension( - Objects.requireNonNull( - Filenames.toSimpleFileName(file.getOriginalFilename()))) - + "_redacted.pdf", - tempFileManager); + TempFile out = + manualRedactionService.finalizeManual( + document, areaResult.rectsByPage, request.getConvertPDFToImage()); + return WebResponseUtils.pdfFileToWebResponse(out, filename); } } @@ -184,9 +182,27 @@ public class RedactController { if (allFoundTextsByPage.isEmpty()) { log.info("No text found matching redaction patterns"); - return WebResponseUtils.pdfDocToWebResponse(document, filename, tempFileManager); + // Still finalize so metadata is scrubbed and the document is rewritten + // consistently. + TempFile finalized = + manualRedactionService.finalizeManual( + document, Collections.emptyMap(), request.getConvertPDFToImage()); + return WebResponseUtils.pdfFileToWebResponse(finalized, filename); } + Set literalTargets = + Arrays.stream(listOfText) + .map(String::trim) + .filter(s -> !s.isEmpty()) + .collect(Collectors.toCollection(LinkedHashSet::new)); + List compiledPatterns = + RedactionPipeline.buildPatterns(listOfText, useRegex, wholeWordSearchBool); + // Bare literal targets match substrings, so they are only safe when the search is a + // plain literal. In regex or whole-word mode the boundary/regex semantics live entirely + // in the compiled patterns, so pass no literal targets to avoid stripping substrings. + Set verificationTargets = + (useRegex || wholeWordSearchBool) ? Collections.emptySet() : literalTargets; + boolean fallbackToBoxOnlyMode; try { fallbackToBoxOnlyMode = @@ -205,7 +221,8 @@ public class RedactController { if (fallbackToBoxOnlyMode) { log.warn( - "Font compatibility issues detected. Using box-only redaction mode for better reliability."); + "Font compatibility issue in placeholder pass; the true-removal pass and " + + "verification still guarantee the target is gone."); fallbackDocument = pdfDocumentFactory.load(request.getFileInput()); @@ -220,7 +237,9 @@ public class RedactController { request.getRedactColor(), request.getCustomPadding(), request.getConvertPDFToImage(), - false); + false, + verificationTargets, + compiledPatterns); return WebResponseUtils.pdfFileToWebResponse(finalized, filename); } @@ -232,7 +251,9 @@ public class RedactController { request.getRedactColor(), request.getCustomPadding(), request.getConvertPDFToImage(), - true); + true, + verificationTargets, + compiledPatterns); return WebResponseUtils.pdfFileToWebResponse(finalized, filename); diff --git a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactExecuteService.java b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactExecuteService.java index 4a53be97b6..6c637d4db1 100644 --- a/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactExecuteService.java +++ b/app/core/src/main/java/stirling/software/SPDF/controller/api/security/RedactExecuteService.java @@ -7,8 +7,10 @@ import java.util.Arrays; import java.util.Collections; import java.util.Comparator; import java.util.HashMap; +import java.util.LinkedHashSet; import java.util.List; import java.util.Map; +import java.util.Set; import java.util.regex.Pattern; import org.apache.pdfbox.cos.COSName; @@ -30,6 +32,7 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyl import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange; import stirling.software.SPDF.pdf.parser.PageColumnLayout; import stirling.software.SPDF.pdf.parser.PageImageLocator; +import stirling.software.SPDF.pdf.redaction.RedactionPipeline; import stirling.software.common.service.CustomPDFDocumentFactory; import stirling.software.common.util.ExceptionUtils; import stirling.software.common.util.TempFile; @@ -140,13 +143,32 @@ class RedactExecuteService { applyAllImagesRedaction(document, request.getRedactImagePages(), style); } + // Explicit overlay-only requests must not rewrite content or verify. When overlay-only + // was forced by a font fallback (not user choice) we still pass the targets so the + // pipeline's true-removal + verification + page-scoped raster fallback run. + Set literalTargets = new LinkedHashSet<>(); + for (String value : textValues) { + String trimmed = value == null ? "" : value.trim(); + if (!trimmed.isEmpty()) { + literalTargets.add(trimmed); + } + } + List verificationPatterns = + RedactionPipeline.buildPatterns( + regexPatterns.toArray(new String[0]), true, false); + Set finalizeTargets = overlayOnly ? Collections.emptySet() : literalTargets; + List finalizePatterns = + overlayOnly ? Collections.emptyList() : verificationPatterns; + return manualRedactionService.finalizeRedaction( document, foundTexts, style.getColor(), style.getPadding(), convertToImage, - !needsOverlayOnly); + !needsOverlayOnly, + finalizeTargets, + finalizePatterns); } catch (Exception e) { log.error("Execute redaction failed: {}", e.getMessage(), e); diff --git a/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/CatalogScrubber.java b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/CatalogScrubber.java new file mode 100644 index 0000000000..51db943808 --- /dev/null +++ b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/CatalogScrubber.java @@ -0,0 +1,613 @@ +package stirling.software.SPDF.pdf.redaction; + +import java.util.Calendar; +import java.util.HashSet; +import java.util.List; +import java.util.Locale; +import java.util.Set; +import java.util.regex.Pattern; + +import org.apache.pdfbox.cos.COSArray; +import org.apache.pdfbox.cos.COSBase; +import org.apache.pdfbox.cos.COSDictionary; +import org.apache.pdfbox.cos.COSName; +import org.apache.pdfbox.cos.COSObject; +import org.apache.pdfbox.cos.COSStream; +import org.apache.pdfbox.cos.COSString; +import org.apache.pdfbox.pdmodel.PDDocument; +import org.apache.pdfbox.pdmodel.PDDocumentCatalog; +import org.apache.pdfbox.pdmodel.PDDocumentInformation; +import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary; +import org.apache.pdfbox.pdmodel.PDPage; +import org.apache.pdfbox.pdmodel.common.PDNameTreeNode; +import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot; +import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation; +import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline; +import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem; +import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm; +import org.apache.pdfbox.pdmodel.interactive.form.PDField; + +import lombok.extern.slf4j.Slf4j; + +/** + * Walks a {@link PDDocument} and physically removes or rewrites every carrier that a PDF can use to + * leak text which the user asked to redact. + * + *

Covers: + * + *

    + *
  • {@link PDDocumentInformation} (Info dict) + XMP metadata stream + *
  • {@link PDDocumentOutline} bookmark titles + *
  • {@link PDAcroForm} field values (V, DV) and rich text (RV) + *
  • Every {@link PDAnnotation} Contents and RC + *
  • Structure tree ActualText, Alt, T, E, Lang entries + *
  • Names tree: JavaScript entries and embedded files (dropped entirely when matching) + *
+ * + *

When applied after the content-stream rewrite it closes the secondary leak paths flagged in + * the redaction security audit. + */ +@Slf4j +public final class CatalogScrubber { + + private CatalogScrubber() {} + + /** + * Remove occurrences of every {@code target} string (and any regex/whole-word pattern form + * produced by {@link RedactionPipeline#buildPatterns}) from all catalog-level carriers of the + * document. When {@code wipeAllMetadata} is {@code true} the document Info dict entries and XMP + * metadata stream are wiped wholesale; this is the safe default after a redaction operation. + */ + public static void scrub( + PDDocument document, Set literalTargets, List patterns) { + if (document == null) { + return; + } + + PDDocumentCatalog catalog = document.getDocumentCatalog(); + if (catalog == null) { + return; + } + + scrubOutline(catalog.getDocumentOutline(), literalTargets, patterns); + scrubAcroForm(catalog.getAcroForm(), literalTargets, patterns); + scrubAnnotations(document, literalTargets, patterns); + scrubStructTree(catalog.getStructureTreeRoot(), literalTargets, patterns); + scrubNames(catalog.getNames(), literalTargets, patterns); + scrubCatalogActions(catalog, literalTargets, patterns); + } + + // --------------------------------------------------------------------- + // Catalog actions: OpenAction, AA, and any JavaScript / URI payloads on the catalog + // --------------------------------------------------------------------- + + private static void scrubCatalogActions( + PDDocumentCatalog catalog, Set targets, List patterns) { + COSDictionary root = catalog.getCOSObject(); + if (root == null) { + return; + } + // OpenAction may be either an action dict (with /URI or /JS) or an explicit destination + // (array). We scrub strings in both cases; if the OpenAction matches a target we clear it. + scrubActionIfMatching(root, COSName.getPDFName("OpenAction"), targets, patterns); + scrubActionIfMatching(root, COSName.getPDFName("AA"), targets, patterns); + } + + /** + * If the action dictionary at {@code key} contains any target literal in a URI or JS payload, + * wipe the key entirely. Otherwise recursively scrub string fields inside it. + */ + private static void scrubActionIfMatching( + COSDictionary parent, COSName key, Set targets, List patterns) { + if (parent == null || key == null) { + return; + } + COSBase value = parent.getDictionaryObject(key); + if (value == null) { + return; + } + if (containsTarget(value, targets, patterns, new HashSet<>())) { + log.debug("Removing catalog {} due to target match", key.getName()); + parent.removeItem(key); + } + } + + private static boolean containsTarget( + COSBase base, Set targets, List patterns, Set seen) { + if (base == null) { + return false; + } + COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base; + if (resolved == null || !seen.add(resolved)) { + return false; + } + if (resolved instanceof COSString cs) { + return matches(cs.getString(), targets, patterns); + } + if (resolved instanceof COSStream stream) { + // Streams in XFA / OpenAction contexts are text (XML, JavaScript). Read the bytes as + // UTF-8 and test for target literals. We cap read length to avoid pathological memory + // use; 2 MiB is plenty for XFA packets and far beyond any realistic JS action. + try (java.io.InputStream is = stream.createInputStream()) { + byte[] buf = new byte[2 * 1024 * 1024]; + int total = 0; + int n; + while ((n = is.read(buf, total, buf.length - total)) > 0) { + total += n; + if (total >= buf.length) { + break; + } + } + String text = new String(buf, 0, total, java.nio.charset.StandardCharsets.UTF_8); + return matches(text, targets, patterns); + } catch (Exception e) { + log.debug("Failed to scan stream for targets: {}", e.getMessage()); + // Fail closed: if we cannot read it we cannot prove it is clean, so treat as a + // match so the caller drops the stream. This is conservative by design. + return true; + } + } + if (resolved instanceof COSDictionary dict) { + for (COSName k : new HashSet<>(dict.keySet())) { + if (containsTarget(dict.getItem(k), targets, patterns, seen)) { + return true; + } + } + return false; + } + if (resolved instanceof COSArray array) { + for (int i = 0; i < array.size(); i++) { + if (containsTarget(array.getObject(i), targets, patterns, seen)) { + return true; + } + } + return false; + } + return false; + } + + /** + * Clean potentially sensitive metadata carriers. Called after {@link #scrub} so that surviving + * references to author/subject/keywords/XMP descriptors do not leak redacted values. + */ + public static void wipeMetadata(PDDocument document) { + if (document == null) { + return; + } + PDDocumentInformation info = document.getDocumentInformation(); + if (info != null) { + info.setAuthor(null); + info.setSubject(null); + info.setKeywords(null); + info.setTitle(null); + info.setCreator(null); + info.setProducer(null); + info.setModificationDate(Calendar.getInstance()); + } + PDDocumentCatalog catalog = document.getDocumentCatalog(); + if (catalog != null) { + try { + catalog.setMetadata(null); + } catch (Exception e) { + log.debug("Could not clear XMP metadata: {}", e.getMessage()); + } + } + } + + // --------------------------------------------------------------------- + // Outline + // --------------------------------------------------------------------- + + private static void scrubOutline( + PDDocumentOutline outline, Set targets, List patterns) { + if (outline == null) { + return; + } + scrubOutlineItems(outline.children(), targets, patterns); + } + + private static void scrubOutlineItems( + Iterable items, Set targets, List patterns) { + if (items == null) { + return; + } + for (PDOutlineItem item : items) { + try { + String title = item.getTitle(); + if (title != null) { + String stripped = stripMatches(title, targets, patterns); + if (!stripped.equals(title)) { + item.setTitle(stripped); + } + } + // Bookmark actions: /A is an action dict which may carry a /URI or /JS payload. + // If any target literal appears anywhere inside the action subtree, drop the + // action entirely so the URI / script cannot leak the target. + COSDictionary itemDict = item.getCOSObject(); + if (itemDict != null) { + scrubActionIfMatching(itemDict, COSName.A, targets, patterns); + scrubActionIfMatching(itemDict, COSName.getPDFName("AA"), targets, patterns); + } + scrubOutlineItems(item.children(), targets, patterns); + } catch (Exception e) { + log.debug("Failed to scrub outline item: {}", e.getMessage()); + } + } + } + + // --------------------------------------------------------------------- + // AcroForm + // --------------------------------------------------------------------- + + private static void scrubAcroForm( + PDAcroForm form, Set targets, List patterns) { + if (form == null) { + return; + } + // XFA forms: scrubbed separately because the XFA XML packet carries the "real" field + // values for XFA-enabled PDFs. Handle XFA before walking the field tree so we fail closed + // if XFA scrubbing throws. + scrubXfa(form, targets, patterns); + + try { + for (PDField field : form.getFieldTree()) { + scrubField(field, targets, patterns); + } + } catch (Exception e) { + log.debug("Failed to walk AcroForm field tree: {}", e.getMessage()); + } + + // Force viewers to regenerate appearance streams from the (scrubbed) /V values rather + // than reusing any cached /AP /N that still contains the target text. Belt-and-braces: + // scrubField has also cleared per-widget /AP dicts, but /NeedAppearances ensures any + // future change still triggers regeneration. + try { + form.setNeedAppearances(true); + } catch (Exception e) { + log.debug("Failed to set /NeedAppearances on AcroForm: {}", e.getMessage()); + } + } + + private static void scrubXfa(PDAcroForm form, Set targets, List patterns) { + try { + COSBase xfaBase = form.getCOSObject().getDictionaryObject(COSName.XFA); + if (xfaBase == null) { + return; + } + boolean hit = containsTarget(xfaBase, targets, patterns, new HashSet<>()); + if (hit) { + // Simplest safe move: strip the XFA entry entirely. Viewers fall back to the + // AcroForm widgets which we have already scrubbed. Leaving a "partially scrubbed" + // XFA packet risks regex failures on partial XML and re-encoded entities leaking + // the target. + log.warn( + "Removing XFA form packet from AcroForm - XFA XML contained a redaction " + + "target and has been dropped so viewers render AcroForm widgets " + + "instead."); + form.getCOSObject().removeItem(COSName.XFA); + } + } catch (Exception e) { + log.debug("Failed to scrub XFA: {}", e.getMessage()); + } + } + + private static void scrubField(PDField field, Set targets, List patterns) { + if (field == null) { + return; + } + try { + COSDictionary dict = field.getCOSObject(); + scrubDictStrings(dict, COSName.V, targets, patterns); + scrubDictStrings(dict, COSName.DV, targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("RV"), targets, patterns); + // Keep field appearance streams in sync with value where possible. + try { + if (field.getValueAsString() != null) { + String stripped = stripMatches(field.getValueAsString(), targets, patterns); + if (!stripped.equals(field.getValueAsString())) { + field.setValue(stripped); + } + } + } catch (Exception e) { + log.debug("Failed to rewrite field value via setValue: {}", e.getMessage()); + } + // Drop per-widget appearance streams (/AP dict) for every widget kid of this field. + // The cached appearance stream contains the pre-redaction value baked in as glyph + // data; simply rewriting /V leaves it visually unchanged in many viewers. Removing /AP + // plus /NeedAppearances at the form level forces regeneration. + clearWidgetAppearances(dict); + } catch (Exception e) { + log.debug("Failed to scrub field: {}", e.getMessage()); + } + } + + private static void clearWidgetAppearances(COSDictionary fieldDict) { + if (fieldDict == null) { + return; + } + // The field itself may be a widget (single-widget field) and/or have Kids. + fieldDict.removeItem(COSName.AP); + COSBase kids = fieldDict.getDictionaryObject(COSName.KIDS); + if (kids instanceof COSArray arr) { + for (int i = 0; i < arr.size(); i++) { + COSBase kidBase = arr.getObject(i); + if (kidBase instanceof COSDictionary kidDict) { + kidDict.removeItem(COSName.AP); + } + } + } + } + + // --------------------------------------------------------------------- + // Annotations + // --------------------------------------------------------------------- + + private static void scrubAnnotations( + PDDocument document, Set targets, List patterns) { + try { + for (PDPage page : document.getPages()) { + List annotations; + try { + annotations = page.getAnnotations(); + } catch (Exception e) { + log.debug("Failed to load annotations for page: {}", e.getMessage()); + continue; + } + if (annotations == null) { + continue; + } + for (PDAnnotation annotation : annotations) { + scrubAnnotation(annotation, targets, patterns); + } + } + } catch (Exception e) { + log.debug("Annotation scrub walk failed: {}", e.getMessage()); + } + } + + private static void scrubAnnotation( + PDAnnotation annotation, Set targets, List patterns) { + if (annotation == null) { + return; + } + try { + String contents = annotation.getContents(); + if (contents != null) { + String stripped = stripMatches(contents, targets, patterns); + if (!stripped.equals(contents)) { + annotation.setContents(stripped); + } + } + COSDictionary dict = annotation.getCOSObject(); + scrubDictStrings(dict, COSName.getPDFName("RC"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("Subj"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("NM"), targets, patterns); + } catch (Exception e) { + log.debug("Failed to scrub annotation: {}", e.getMessage()); + } + } + + // --------------------------------------------------------------------- + // Structure tree + // --------------------------------------------------------------------- + + private static void scrubStructTree( + PDStructureTreeRoot root, Set targets, List patterns) { + if (root == null) { + return; + } + try { + scrubStructDict(root.getCOSObject(), targets, patterns, new HashSet<>()); + } catch (Exception e) { + log.debug("Structure tree scrub failed: {}", e.getMessage()); + } + } + + private static void scrubStructDict( + COSBase base, Set targets, List patterns, Set seen) { + if (base == null) { + return; + } + COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base; + if (resolved == null || !seen.add(resolved)) { + return; + } + if (resolved instanceof COSDictionary dict) { + // Do not walk into content streams - those are handled by content-stream rewrite. + if (resolved instanceof COSStream) { + return; + } + scrubDictStrings(dict, COSName.getPDFName("ActualText"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("Alt"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("E"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns); + scrubDictStrings(dict, COSName.getPDFName("Lang"), targets, patterns); + for (COSName key : new HashSet<>(dict.keySet())) { + COSBase value = dict.getItem(key); + if (value instanceof COSDictionary + || value instanceof COSArray + || value instanceof COSObject) { + scrubStructDict(value, targets, patterns, seen); + } + } + } else if (resolved instanceof COSArray array) { + for (int i = 0; i < array.size(); i++) { + scrubStructDict(array.getObject(i), targets, patterns, seen); + } + } + } + + // --------------------------------------------------------------------- + // Names tree (JavaScript + embedded files) + // --------------------------------------------------------------------- + + private static void scrubNames( + PDDocumentNameDictionary names, Set targets, List patterns) { + if (names == null) { + return; + } + try { + dropMatchingNames(names.getJavaScript(), targets, patterns); + } catch (Exception e) { + log.debug("Failed to scrub JavaScript names: {}", e.getMessage()); + } + try { + dropMatchingNames(names.getEmbeddedFiles(), targets, patterns); + } catch (Exception e) { + log.debug("Failed to scrub embedded-file names: {}", e.getMessage()); + } + } + + private static void dropMatchingNames( + PDNameTreeNode node, Set targets, List patterns) { + if (node == null) { + return; + } + COSDictionary dict = node.getCOSObject(); + if (dict == null) { + return; + } + scrubNameTreeDict(dict, targets, patterns); + } + + private static void scrubNameTreeDict( + COSDictionary dict, Set targets, List patterns) { + if (dict == null) { + return; + } + COSArray namesArray = (COSArray) dict.getDictionaryObject(COSName.NAMES); + if (namesArray != null) { + for (int i = namesArray.size() - 2; i >= 0; i -= 2) { + COSBase keyBase = namesArray.getObject(i); + String key = keyBase instanceof COSString s ? s.getString() : null; + if (key != null && matches(key, targets, patterns)) { + namesArray.remove(i + 1); + namesArray.remove(i); + } + } + } + COSArray kids = (COSArray) dict.getDictionaryObject(COSName.KIDS); + if (kids != null) { + for (int i = 0; i < kids.size(); i++) { + COSBase kid = kids.getObject(i); + if (kid instanceof COSDictionary kidDict) { + scrubNameTreeDict(kidDict, targets, patterns); + } + } + } + } + + // --------------------------------------------------------------------- + // Helpers + // --------------------------------------------------------------------- + + private static void scrubDictStrings( + COSDictionary dict, COSName key, Set targets, List patterns) { + if (dict == null || key == null) { + return; + } + COSBase value = dict.getDictionaryObject(key); + if (value instanceof COSString cosString) { + String stripped = stripMatches(cosString.getString(), targets, patterns); + if (!stripped.equals(cosString.getString())) { + dict.setString(key, stripped); + } + } else if (value instanceof COSArray array) { + for (int i = 0; i < array.size(); i++) { + COSBase element = array.getObject(i); + if (element instanceof COSString elementString) { + String stripped = stripMatches(elementString.getString(), targets, patterns); + if (!stripped.equals(elementString.getString())) { + array.set(i, new COSString(stripped)); + } + } + } + } + } + + static String stripMatches(String source, Set literalTargets, List patterns) { + if (source == null || source.isEmpty()) { + return source; + } + String result = source; + if (literalTargets != null) { + for (String target : literalTargets) { + if (target == null || target.isEmpty()) { + continue; + } + // Case-insensitive literal removal. Verification is case-insensitive, so scrubbing + // MUST be too or mixed-case ("SMITH" in a catalog string vs "Smith" in the target + // list) will fail verification and trip the rasterisation fallback - or worse, on + // carriers that are not verified, leak the string untouched. + result = caseInsensitiveReplaceAll(result, target); + } + } + if (patterns != null) { + for (Pattern pattern : patterns) { + try { + // Force case-insensitive matching for catalog carriers regardless of the flags + // the pattern was compiled with. User-supplied redaction targets should not + // silently miss because the author typed the name in different case. + Pattern ci = withCaseInsensitive(pattern); + result = ci.matcher(result).replaceAll(""); + } catch (Exception e) { + log.debug( + "Pattern replace failed for {}: {}", pattern.pattern(), e.getMessage()); + } + } + } + return result; + } + + static boolean matches(String source, Set literalTargets, List patterns) { + if (source == null || source.isEmpty()) { + return false; + } + String lower = source.toLowerCase(Locale.ROOT); + if (literalTargets != null) { + for (String target : literalTargets) { + if (target != null + && !target.isEmpty() + && lower.contains(target.toLowerCase(Locale.ROOT))) { + return true; + } + } + } + if (patterns != null) { + for (Pattern pattern : patterns) { + try { + if (withCaseInsensitive(pattern).matcher(source).find()) { + return true; + } + } catch (Exception e) { + log.debug("Pattern match failed for {}: {}", pattern.pattern(), e.getMessage()); + } + } + } + return false; + } + + private static String caseInsensitiveReplaceAll(String source, String target) { + if (target.isEmpty()) { + return source; + } + Pattern literal = + Pattern.compile( + Pattern.quote(target), Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE); + return literal.matcher(source).replaceAll(""); + } + + private static Pattern withCaseInsensitive(Pattern pattern) { + if ((pattern.flags() & Pattern.CASE_INSENSITIVE) != 0) { + return pattern; + } + try { + return Pattern.compile( + pattern.pattern(), + pattern.flags() | Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE); + } catch (Exception e) { + return pattern; + } + } +} diff --git a/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionPipeline.java b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionPipeline.java new file mode 100644 index 0000000000..f178aed52f --- /dev/null +++ b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionPipeline.java @@ -0,0 +1,1246 @@ +package stirling.software.SPDF.pdf.redaction; + +import java.awt.Color; +import java.awt.geom.Rectangle2D; +import java.awt.image.BufferedImage; +import java.io.ByteArrayInputStream; +import java.io.ByteArrayOutputStream; +import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; +import java.util.HashSet; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Locale; +import java.util.Map; +import java.util.Set; +import java.util.TreeSet; +import java.util.regex.Matcher; +import java.util.regex.Pattern; +import java.util.regex.PatternSyntaxException; + +import javax.imageio.ImageIO; + +import org.apache.pdfbox.contentstream.operator.Operator; +import org.apache.pdfbox.cos.COSArray; +import org.apache.pdfbox.cos.COSBase; +import org.apache.pdfbox.cos.COSName; +import org.apache.pdfbox.cos.COSString; +import org.apache.pdfbox.pdfparser.PDFStreamParser; +import org.apache.pdfbox.pdfwriter.ContentStreamWriter; +import org.apache.pdfbox.pdmodel.PDDocument; +import org.apache.pdfbox.pdmodel.PDPage; +import org.apache.pdfbox.pdmodel.PDPageContentStream; +import org.apache.pdfbox.pdmodel.PDResources; +import org.apache.pdfbox.pdmodel.common.PDRectangle; +import org.apache.pdfbox.pdmodel.common.PDStream; +import org.apache.pdfbox.pdmodel.font.PDFont; +import org.apache.pdfbox.pdmodel.font.PDTrueTypeFont; +import org.apache.pdfbox.pdmodel.font.PDType0Font; +import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject; +import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject; +import org.apache.pdfbox.rendering.ImageType; +import org.apache.pdfbox.rendering.PDFRenderer; +import org.apache.pdfbox.text.PDFTextStripper; +import org.apache.pdfbox.text.PDFTextStripperByArea; +import org.apache.pdfbox.text.TextPosition; + +import lombok.extern.slf4j.Slf4j; + +/** + * Shared plumbing for the three redaction paths (manual areas, whole pages, auto-word). + * + *

Every path funnels through this class so that the following guarantees hold uniformly: + * + *

    + *
  1. Text tokens that fall inside a target rect are physically dropped from the content stream. + *
  2. Catalog-level carriers (outline, AcroForm, annotations, struct tree, names tree) are + * scrubbed so redacted strings cannot survive outside the page content. + *
  3. Info dict + XMP metadata is wiped. + *
  4. The document is rewritten with a fresh xref (no incremental save) and then verified with a + * fresh {@link PDFTextStripper} pass; surviving target text throws {@link + * RedactionVerificationFailedException}. The literal/pattern path verifies the whole (or + * affected) page text; the manual-area path verifies each redaction rectangle is empty. + *
+ */ +@Slf4j +public final class RedactionPipeline { + + private static final Set TEXT_SHOWING_OPERATORS = Set.of("Tj", "TJ", "'", "\""); + + private RedactionPipeline() {} + + /** + * Apply a list of page-local redaction rects (coordinates in PDF user space) against the + * supplied document. + */ + public static RedactionResult redactAreas( + PDDocument document, + Map> rectsByPageIndex, + Color overlayColor) + throws IOException { + + Set capturedStrings = new LinkedHashSet<>(); + + for (Map.Entry> entry : rectsByPageIndex.entrySet()) { + int pageIndex = entry.getKey(); + List rects = entry.getValue(); + if (pageIndex < 0 || pageIndex >= document.getNumberOfPages() || rects.isEmpty()) { + continue; + } + PDPage page = document.getPage(pageIndex); + List captured = captureTextInRects(page, rects); + capturedStrings.addAll(captured); + + removeTokensIntersectingRects(document, page, rects); + drawOverlay(document, page, rects, overlayColor); + } + + return new RedactionResult(capturedStrings); + } + + /** + * Replace the entire contents of the listed pages with a single filled rectangle in the page + * media box. All underlying text and images are dropped from the content stream and from the + * page resources. + */ + public static void redactWholePages( + PDDocument document, List pageIndexes, Color overlayColor) throws IOException { + for (Integer pageIndex : pageIndexes) { + if (pageIndex == null || pageIndex < 0 || pageIndex >= document.getNumberOfPages()) { + continue; + } + PDPage page = document.getPage(pageIndex); + PDRectangle media = page.getMediaBox(); + + // Drop existing content streams and page resources outright. + page.getCOSObject().removeItem(COSName.CONTENTS); + page.setResources(new PDResources()); + page.getCOSObject().removeItem(COSName.ANNOTS); + + try (PDPageContentStream cs = + new PDPageContentStream( + document, page, PDPageContentStream.AppendMode.OVERWRITE, true, true)) { + cs.setNonStrokingColor(overlayColor); + cs.addRect( + media.getLowerLeftX(), + media.getLowerLeftY(), + media.getWidth(), + media.getHeight()); + cs.fill(); + } + } + } + + /** + * Rewrites every content stream on every page so that any occurrence of any literal target or + * regex pattern is physically removed from the glyph bytes. Handles: + * + *
    + *
  • simple Type1 fonts with WinAnsiEncoding (e.g. ReportLab {@code (Test PDF #1) Tj}), + *
  • Type0/CID fonts with multi-byte codes, + *
  • pages with {@code /Rotate 90} or other rotations, + *
  • text split across multiple operands inside a single {@code TJ} array. + *
+ * + *

The approach decodes each {@link COSString} operand byte-by-byte using the font that is + * current at that point in the content stream, reconstructs the Unicode text, runs the target + * patterns against that text, and rebuilds the byte operand omitting the matched character + * codes. Width adjustment ({@code TJ} numeric kerning) is reinserted so downstream layout is + * preserved even when the operand shrinks. The page layout may look sparse but the glyphs for + * the target term are guaranteed to be gone. + */ + public static void redactLiteralTerms( + PDDocument document, Set literalTargets, List patterns) + throws IOException { + List effectivePatterns = effectivePatterns(literalTargets, patterns); + if (effectivePatterns.isEmpty()) { + return; + } + int pageIndex = 0; + for (PDPage page : document.getPages()) { + try { + rewritePageContent(document, page, effectivePatterns); + } catch (IOException | RuntimeException e) { + // Never let one page's font quirk (e.g. Type3 encode, damaged program) abort the + // whole document. Leave this page for the verify+rasterise safety net to catch. + log.warn( + "Content-stream rewrite failed on page {} ({}); leaving it for the " + + "verification/rasterisation pass.", + pageIndex + 1, + e.toString()); + } + pageIndex++; + } + } + + /** + * Finalize the document with document-wide verification: scrub catalog carriers, wipe metadata, + * subset embedded fonts, save with a fresh xref and verify that no literal target survives + * anywhere in the saved PDF. + * + *

Use this overload for auto-word redaction where every occurrence of a literal target is a + * deliberate removal target. For manual rect redaction use {@link #finalize(PDDocument, Set, + * List, Set)} so verification is scoped to the affected pages (since the same word can + * legitimately remain on non-targeted pages). + */ + public static byte[] finalize( + PDDocument document, Set literalTargets, List patterns) + throws IOException { + return finalize(document, literalTargets, patterns, null); + } + + /** + * Finalize with verification scoped to a specific set of page indexes. When {@code + * affectedPages} is null the verification runs over the whole document. When non-null, only the + * listed pages are checked for surviving literal targets; the rasterisation fallback (if + * triggered) also rasterises only those pages, preserving every other page verbatim so + * unrelated content remains text-searchable. + * + *

Pass an empty set together with empty targets/patterns to skip verification entirely (for + * manual redactions that captured no text and drew no image boxes). + */ + public static byte[] finalize( + PDDocument document, + Set literalTargets, + List patterns, + Set affectedPages) + throws IOException { + + CatalogScrubber.scrub(document, literalTargets, patterns); + CatalogScrubber.wipeMetadata(document); + warnAboutEmbeddedFontGlyphs(document); + + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + document.save(baos); + byte[] bytes = baos.toByteArray(); + + try { + verify(bytes, literalTargets, patterns, affectedPages); + return bytes; + } catch (RedactionVerificationFailedException primaryFailure) { + // Last-resort: if the content-stream rewriter could not guarantee removal (unusual + // fonts, encrypted streams, non-text glyph carriers), rasterise only the affected + // pages so the target is physically gone without destroying the text layer on pages + // that were never touched. This is logged loudly so operators know redaction fell + // back to the image path. + log.warn( + "Primary redaction verification failed ({}); falling back to page-scoped " + + "rasterisation to guarantee removal.", + primaryFailure.getMessage()); + // When the caller gave no page scope (auto-word mode), locate the leaking pages so + // only those are rasterised; clean pages keep their searchable text layer. If leak + // detection itself fails, fall back to rasterising everything (null). + Set pagesToRaster = + (affectedPages == null || affectedPages.isEmpty()) + ? findLeakingPages(bytes, literalTargets, patterns) + : new HashSet<>(affectedPages); + try (PDDocument rasterised = rasterisePages(bytes, pagesToRaster)) { + CatalogScrubber.scrub(rasterised, literalTargets, patterns); + CatalogScrubber.wipeMetadata(rasterised); + ByteArrayOutputStream rasterOut = new ByteArrayOutputStream(); + rasterised.save(rasterOut); + byte[] rasterBytes = rasterOut.toByteArray(); + verify(rasterBytes, literalTargets, patterns, affectedPages); + return rasterBytes; + } catch (IOException e) { + throw new RedactionVerificationFailedException( + "Rasterisation fallback failed after primary redaction leak", e); + } + } + } + + /** + * Finalize a manual area redaction. Unlike the literal/pattern path, verification here is + * region-based: after saving, each redaction rectangle is re-scanned and must contain no text. + * This avoids the substring false positives that whole-page string verification hits when a box + * clips a word into a common fragment (for example a clipped "HEADER" becoming "HE", which + * would otherwise match "here" elsewhere on the page). On a real leak the affected pages are + * rasterised so untouched pages keep their text layer. + */ + public static byte[] finalizeAreas( + PDDocument document, Map> rectsByPage) throws IOException { + + CatalogScrubber.wipeMetadata(document); + warnAboutEmbeddedFontGlyphs(document); + + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + document.save(baos); + byte[] bytes = baos.toByteArray(); + + try { + verifyRectsEmpty(bytes, rectsByPage); + return bytes; + } catch (RedactionVerificationFailedException primaryFailure) { + log.warn( + "Manual-area redaction verification failed ({}); rasterising affected pages to " + + "guarantee removal.", + primaryFailure.getMessage()); + Set pagesToRaster = new HashSet<>(rectsByPage.keySet()); + try (PDDocument rasterised = rasterisePages(bytes, pagesToRaster)) { + CatalogScrubber.wipeMetadata(rasterised); + ByteArrayOutputStream rasterOut = new ByteArrayOutputStream(); + rasterised.save(rasterOut); + return rasterOut.toByteArray(); + } catch (IOException e) { + throw new RedactionVerificationFailedException( + "Rasterisation fallback failed after manual redaction leak", e); + } + } + } + + private static void verifyRectsEmpty( + byte[] bytes, Map> rectsByPage) { + if (rectsByPage == null || rectsByPage.isEmpty()) { + return; + } + try (PDDocument reopened = org.apache.pdfbox.Loader.loadPDF(bytes)) { + for (Map.Entry> entry : rectsByPage.entrySet()) { + int pageIndex = entry.getKey(); + if (pageIndex < 0 || pageIndex >= reopened.getNumberOfPages()) { + continue; + } + List remaining = + captureTextInRects(reopened.getPage(pageIndex), entry.getValue()); + for (String fragment : remaining) { + if (fragment != null && !fragment.isBlank()) { + throw new RedactionVerificationFailedException( + "Text still present inside redaction area on page " + + (pageIndex + 1) + + ": '" + + fragment + + "'"); + } + } + } + } catch (IOException e) { + throw new RedactionVerificationFailedException( + "Failed to reopen redacted PDF for area verification", e); + } + } + + private static List effectivePatterns( + Set literalTargets, List patterns) { + List result = new ArrayList<>(); + if (literalTargets != null) { + for (String target : literalTargets) { + if (target == null || target.isEmpty()) { + continue; + } + // Case-insensitive to match TextFinderUtils' search semantics; removal must never + // be narrower than what the finder matched (or verification would raster pages). + result.add( + Pattern.compile( + Pattern.quote(target), + Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE)); + } + } + if (patterns != null) { + result.addAll(patterns); + } + return result; + } + + // --------------------------------------------------------------------- + // Per-page content-stream rewrite (literal/regex based) + // --------------------------------------------------------------------- + + private static void rewritePageContent(PDDocument document, PDPage page, List patterns) + throws IOException { + PDResources resources = page.getResources(); + if (resources == null) { + return; + } + List tokens = parseTokens(new PDFStreamParser(page)); + boolean modified = rewriteTokens(tokens, resources, patterns); + if (modified) { + writePageTokens(document, page, tokens); + } + // Recurse into form XObjects referenced by this page. + rewriteFormXObjects(document, resources, patterns, new HashSet<>()); + } + + private static void rewriteFormXObjects( + PDDocument document, + PDResources resources, + List patterns, + Set visited) + throws IOException { + for (COSName name : resources.getXObjectNames()) { + try { + var xobj = resources.getXObject(name); + if (!(xobj instanceof PDFormXObject form)) { + continue; + } + if (!visited.add(form.getCOSObject())) { + continue; + } + List tokens = parseTokens(new PDFStreamParser(form)); + PDResources formResources = form.getResources(); + if (formResources == null) { + continue; + } + boolean modified = rewriteTokens(tokens, formResources, patterns); + if (modified) { + PDStream stream = new PDStream(document); + try (var out = stream.createOutputStream(COSName.FLATE_DECODE)) { + new ContentStreamWriter(out).writeTokens(tokens); + } + form.getCOSObject().removeItem(COSName.CONTENTS); + form.getCOSObject().setItem(COSName.CONTENTS, stream.getCOSObject()); + } + rewriteFormXObjects(document, formResources, patterns, visited); + } catch (IOException e) { + log.debug("Failed to rewrite XObject {}: {}", name.getName(), e.getMessage()); + } + } + } + + private static List parseTokens(PDFStreamParser parser) throws IOException { + List tokens = new ArrayList<>(); + Object t; + while ((t = parser.parseNextToken()) != null) { + tokens.add(t); + } + return tokens; + } + + /** + * Walk tokens keeping a tiny text state (current font). When a text-showing operator is + * encountered its string operand(s) are decoded via the current font, run against every + * pattern, and rewritten with matching character codes removed. + * + * @return true if any operand was modified. + */ + private static boolean rewriteTokens( + List tokens, PDResources resources, List patterns) { + boolean modified = false; + PDFont currentFont = null; + for (int i = 0; i < tokens.size(); i++) { + Object tok = tokens.get(i); + if (!(tok instanceof Operator op)) { + continue; + } + String name = op.getName(); + if ("Tf".equals(name) && i >= 2) { + Object fontNameTok = tokens.get(i - 2); + if (fontNameTok instanceof COSName fontName) { + try { + currentFont = resources.getFont(fontName); + } catch (IOException ex) { + log.debug( + "Could not resolve font {}: {}", + fontName.getName(), + ex.getMessage()); + currentFont = null; + } + } + } else if (TEXT_SHOWING_OPERATORS.contains(name) && i >= 1) { + int operandIdx = i - 1; + Object operand = tokens.get(operandIdx); + if (operand instanceof COSString cosString) { + COSString replacement = rewriteCosString(cosString, currentFont, patterns); + if (replacement != null) { + tokens.set(operandIdx, replacement); + modified = true; + } + } else if (operand instanceof COSArray arr) { + COSArray newArr = rewriteCosArray(arr, currentFont, patterns); + if (newArr != null) { + tokens.set(operandIdx, newArr); + modified = true; + } + } + } + } + return modified; + } + + private static COSString rewriteCosString( + COSString cosString, PDFont font, List patterns) { + if (font == null) { + // Without a font we cannot decode safely. Try a best-effort latin-1 interpretation. + return rewriteRawLatin(cosString, patterns); + } + DecodeResult decoded = decodeCosString(cosString, font); + if (decoded == null) { + return rewriteRawLatin(cosString, patterns); + } + boolean[] drop = findDroppedCharsMask(decoded.text, patterns); + if (drop == null) { + return null; + } + return buildFilteredCosString(decoded, drop, font); + } + + private static COSArray rewriteCosArray(COSArray arr, PDFont font, List patterns) { + // Build a concatenated decode across all COSString elements so that matches spanning + // multiple operands are found. Non-string elements are kerning adjustments; keep them + // in place and re-emit them between the rewritten strings. + List parts = new ArrayList<>(); + StringBuilder concat = new StringBuilder(); + for (int i = 0; i < arr.size(); i++) { + COSBase elem = arr.get(i); + if (elem instanceof COSString cs) { + DecodeResult decoded = font != null ? decodeCosString(cs, font) : null; + if (decoded == null) { + decoded = decodeAsLatin(cs); + } + parts.add(decoded); + concat.append(decoded.text); + } else { + parts.add(null); + } + } + boolean[] fullDrop = findDroppedCharsMask(concat.toString(), patterns); + if (fullDrop == null) { + return null; + } + COSArray out = new COSArray(); + int cursor = 0; + for (int i = 0; i < arr.size(); i++) { + COSBase elem = arr.get(i); + if (elem instanceof COSString cs) { + DecodeResult decoded = parts.get(i); + int partLen = decoded.text.length(); + boolean[] partDrop = new boolean[partLen]; + System.arraycopy(fullDrop, cursor, partDrop, 0, partLen); + cursor += partLen; + COSString rebuilt = + buildFilteredCosStringRaw(decoded, partDrop, font, cs.getBytes()); + // A zero-length COSString is valid; PDFBox emits it as () producing no glyphs. + out.add(rebuilt); + } else { + out.add(elem); + } + } + return out; + } + + /** For cases where we have no font - treat the bytes as latin-1 characters. */ + private static COSString rewriteRawLatin(COSString cosString, List patterns) { + DecodeResult decoded = decodeAsLatin(cosString); + boolean[] drop = findDroppedCharsMask(decoded.text, patterns); + if (drop == null) { + return null; + } + ByteArrayOutputStream out = new ByteArrayOutputStream(); + for (int i = 0; i < decoded.text.length(); i++) { + if (drop[i]) continue; + out.write(decoded.text.charAt(i) & 0xFF); + } + return new COSString(out.toByteArray()); + } + + private static DecodeResult decodeAsLatin(COSString cosString) { + byte[] bytes = cosString.getBytes(); + StringBuilder sb = new StringBuilder(bytes.length); + int[] codeStart = new int[bytes.length]; + int[] codeLen = new int[bytes.length]; + for (int i = 0; i < bytes.length; i++) { + sb.append((char) (bytes[i] & 0xFF)); + codeStart[i] = i; + codeLen[i] = 1; + } + return new DecodeResult(sb.toString(), bytes, codeStart, codeLen); + } + + /** + * Decode a {@link COSString} into Unicode characters using the supplied font. Returns null if + * decoding fails at any point (e.g. malformed byte sequence). + */ + private static DecodeResult decodeCosString(COSString cosString, PDFont font) { + byte[] bytes = cosString.getBytes(); + StringBuilder text = new StringBuilder(); + List starts = new ArrayList<>(); + List lens = new ArrayList<>(); + try (ByteArrayInputStream in = new ByteArrayInputStream(bytes)) { + int pos = 0; + while (in.available() > 0) { + int before = in.available(); + int code; + try { + code = font.readCode(in); + } catch (IOException | RuntimeException ex) { + log.debug( + "Font {} failed to decode byte sequence: {}", + font.getName(), + ex.getMessage()); + return null; + } + int consumed = before - in.available(); + String unicode; + try { + unicode = font.toUnicode(code); + } catch (Exception ex) { + unicode = null; + } + if (unicode == null) { + // If the font has no ToUnicode mapping we cannot match reliably - return null + // and let the caller either fall back to latin-1 (rare) or rasterisation via + // the final verification pass. + return null; + } + text.append(unicode); + // Associate every Unicode character produced with the same code byte range so + // that dropping any one of them drops the whole code. + for (int c = 0; c < unicode.length(); c++) { + starts.add(pos); + lens.add(consumed); + } + if (consumed == 0) { + // Defensive: avoid infinite loop on malformed fonts. + break; + } + pos += consumed; + } + } catch (IOException e) { + return null; + } + int[] startArr = starts.stream().mapToInt(Integer::intValue).toArray(); + int[] lenArr = lens.stream().mapToInt(Integer::intValue).toArray(); + return new DecodeResult(text.toString(), bytes, startArr, lenArr); + } + + /** + * Returns null if none of the patterns match; otherwise a boolean mask the same length as + * {@code text} where {@code true} means "this character belongs to a matched range". + */ + private static boolean[] findDroppedCharsMask(String text, List patterns) { + boolean any = false; + boolean[] mask = new boolean[text.length()]; + for (Pattern pattern : patterns) { + Matcher m; + try { + m = pattern.matcher(text); + } catch (Exception ex) { + continue; + } + while (m.find()) { + int s = m.start(); + int e = m.end(); + if (e <= s) continue; + for (int i = s; i < e; i++) { + mask[i] = true; + } + any = true; + } + } + return any ? mask : null; + } + + private static COSString buildFilteredCosString( + DecodeResult decoded, boolean[] drop, PDFont font) { + return buildFilteredCosStringRaw(decoded, drop, font, decoded.bytes); + } + + private static COSString buildFilteredCosStringRaw( + DecodeResult decoded, boolean[] drop, PDFont font, byte[] originalBytes) { + // Collect code byte-ranges to drop. Because one code can produce multiple chars, we drop + // the code if ANY of its chars is flagged. + Set dropStarts = new HashSet<>(); + for (int i = 0; i < drop.length; i++) { + if (drop[i]) { + dropStarts.add(decoded.codeStarts[i]); + } + } + ByteArrayOutputStream out = new ByteArrayOutputStream(); + int i = 0; + while (i < originalBytes.length) { + int start = i; + int len = findCodeLenAt(decoded, start); + if (len <= 0) { + // Unknown - keep the byte verbatim. + out.write(originalBytes[i] & 0xFF); + i += 1; + continue; + } + if (dropStarts.contains(start)) { + // Replace dropped code with encoded space if possible. This preserves rough + // layout; if encoding a space fails we simply drop the bytes. + byte[] spaceBytes = tryEncodeSpace(font); + if (spaceBytes != null) { + out.write(spaceBytes, 0, spaceBytes.length); + } + } else { + out.write(originalBytes, start, len); + } + i += len; + } + return new COSString(out.toByteArray()); + } + + private static int findCodeLenAt(DecodeResult decoded, int byteStart) { + for (int j = 0; j < decoded.codeStarts.length; j++) { + if (decoded.codeStarts[j] == byteStart) { + return decoded.codeLens[j]; + } + } + return -1; + } + + private static byte[] tryEncodeSpace(PDFont font) { + if (font == null) { + return new byte[] {0x20}; + } + try { + return font.encode(" "); + } catch (Exception e) { + // Some fonts cannot encode a space at all (e.g. Type3 throws + // UnsupportedOperationException + // from encode()). Returning null drops the code with no placeholder - the target glyph + // is still physically removed, which is what matters for redaction. + return null; + } + } + + private static final class DecodeResult { + final String text; + final byte[] bytes; + final int[] codeStarts; + final int[] codeLens; + + DecodeResult(String text, byte[] bytes, int[] codeStarts, int[] codeLens) { + this.text = text; + this.bytes = bytes; + this.codeStarts = codeStarts; + this.codeLens = codeLens; + } + } + + // --------------------------------------------------------------------- + // Rasterisation fallback + // --------------------------------------------------------------------- + + /** + * Rasterise only the pages listed in {@code pagesToRaster} (all pages when null) at 150 DPI and + * replace those pages' content streams with the rendered image. Pages not in the set keep their + * original content streams verbatim so unrelated text remains searchable after the fallback. + */ + private static PDDocument rasterisePages(byte[] sourceBytes, Set pagesToRaster) + throws IOException { + // Load the document directly and mutate in place: rewriting only the affected pages' + // content streams is cheaper and guarantees untouched pages' xref entries survive + // byte-for-byte. + PDDocument source = org.apache.pdfbox.Loader.loadPDF(sourceBytes); + try { + PDFRenderer renderer = new PDFRenderer(source); + int pageCount = source.getNumberOfPages(); + for (int i = 0; i < pageCount; i++) { + if (pagesToRaster != null && !pagesToRaster.contains(i)) { + continue; + } + PDPage page = source.getPage(i); + PDRectangle media = page.getMediaBox(); + + BufferedImage img = renderer.renderImageWithDPI(i, 150, ImageType.RGB); + ByteArrayOutputStream imgOut = new ByteArrayOutputStream(); + ImageIO.write(img, "png", imgOut); + PDImageXObject imageXObject = + PDImageXObject.createFromByteArray( + source, imgOut.toByteArray(), "redacted-page-" + i); + + // Drop all prior content / resources / annotations; the raster is the page now. + page.getCOSObject().removeItem(COSName.CONTENTS); + page.setResources(new PDResources()); + page.getCOSObject().removeItem(COSName.ANNOTS); + // Rotation is already baked into the rendered image, so reset it to zero. + page.setRotation(0); + + try (PDPageContentStream cs = + new PDPageContentStream( + source, + page, + PDPageContentStream.AppendMode.OVERWRITE, + false, + true)) { + // The rendered image already has the rotation baked in visually, so the + // resulting page is placed un-rotated against the media box. + cs.drawImage( + imageXObject, + media.getLowerLeftX(), + media.getLowerLeftY(), + media.getWidth(), + media.getHeight()); + } + } + return source; + } catch (IOException | RuntimeException e) { + source.close(); + throw e; + } + } + + // --------------------------------------------------------------------- + // Pattern construction + // --------------------------------------------------------------------- + + /** + * Build a set of regex patterns from user input. Returns an empty list if none of the entries + * produce a valid pattern. + */ + public static List buildPatterns( + String[] rawEntries, boolean useRegex, boolean wholeWordSearch) { + List patterns = new ArrayList<>(); + if (rawEntries == null) { + return patterns; + } + for (String raw : rawEntries) { + if (raw == null) { + continue; + } + String trimmed = raw.trim(); + if (trimmed.isEmpty()) { + continue; + } + try { + String core = useRegex ? trimmed : Pattern.quote(trimmed); + if (wholeWordSearch) { + core = "\\b" + core + "\\b"; + } + // Case-insensitive to mirror TextFinderUtils.createOptimizedSearchPatterns: what + // the finder matches, removal and verification must match too. + patterns.add( + Pattern.compile(core, Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE)); + } catch (PatternSyntaxException e) { + log.debug("Skipping invalid regex '{}': {}", trimmed, e.getMessage()); + } + } + return patterns; + } + + // --------------------------------------------------------------------- + // Area capture + // --------------------------------------------------------------------- + + private static List captureTextInRects(PDPage page, List rects) + throws IOException { + PDFTextStripperByArea stripper = new PDFTextStripperByArea(); + stripper.setSortByPosition(true); + for (int i = 0; i < rects.size(); i++) { + PDRectangle rect = rects.get(i); + // PDFTextStripperByArea uses a Java2D rectangle in the same coordinate space that + // TextPosition reports, i.e. top-left origin with Y flipped relative to PDF user space. + float pdfY = page.getBBox().getHeight() - rect.getUpperRightY(); + Rectangle2D.Float region = + new Rectangle2D.Float( + rect.getLowerLeftX(), pdfY, rect.getWidth(), rect.getHeight()); + stripper.addRegion("r" + i, region); + } + try { + stripper.extractRegions(page); + } catch (Exception e) { + log.debug("Failed to extract text in rects: {}", e.getMessage()); + return Collections.emptyList(); + } + Set captured = new LinkedHashSet<>(); + for (int i = 0; i < rects.size(); i++) { + String text = stripper.getTextForRegion("r" + i); + if (text == null) { + continue; + } + for (String token : text.split("\\s+")) { + String trimmed = token.trim(); + if (!trimmed.isEmpty()) { + captured.add(trimmed); + } + } + } + return new ArrayList<>(captured); + } + + // --------------------------------------------------------------------- + // Content-stream rewriting (rect-driven glyph removal) + // --------------------------------------------------------------------- + + private static void removeTokensIntersectingRects( + PDDocument document, PDPage page, List rects) throws IOException { + + // Identify which show-text operators should be wiped by running a tiny engine that tracks + // each operator's rendered bounding box. + List dropTokenIndexes = identifyShowTextOperatorsInRects(document, page, rects); + + PDFStreamParser parser = new PDFStreamParser(page); + List tokens = new ArrayList<>(); + Object token; + while ((token = parser.parseNextToken()) != null) { + tokens.add(token); + } + + // Build set of token indexes to blank (the argument immediately preceding a flagged + // text-showing operator). + Set blankArgIndexes = new HashSet<>(); + int textOpCount = 0; + for (int i = 0; i < tokens.size(); i++) { + Object t = tokens.get(i); + if (t instanceof Operator op && TEXT_SHOWING_OPERATORS.contains(op.getName())) { + if (dropTokenIndexes.contains(textOpCount) && i > 0) { + blankArgIndexes.add(i - 1); + } + textOpCount++; + } + } + + for (Integer idx : blankArgIndexes) { + Object arg = tokens.get(idx); + if (arg instanceof COSString) { + tokens.set(idx, new COSString("")); + } else if (arg instanceof COSArray arr) { + COSArray empty = new COSArray(); + // Preserve numeric kerning entries so page layout doesn't shift wildly but drop + // every COSString. + for (COSBase element : arr) { + if (!(element instanceof COSString)) { + empty.add(element); + } + } + tokens.set(idx, empty); + } + } + + // Additionally wipe Do-drawn images whose bounding box intersects a rect. This needs a + // coordinate scan - for safety we drop every Do whose CTM origin sits inside a rect. + List doImageIndexes = identifyDoImagesInRects(page, rects); + int doCount = 0; + for (int i = tokens.size() - 1; i >= 0; i--) { + Object t = tokens.get(i); + if (t instanceof Operator op && "Do".equals(op.getName())) { + if (doImageIndexes.contains(doCount)) { + // Remove both the name argument and the operator. + if (i > 0) { + tokens.remove(i); + tokens.remove(i - 1); + } + } + doCount++; + } + } + + writePageTokens(document, page, tokens); + } + + private static List identifyShowTextOperatorsInRects( + PDDocument document, PDPage page, List rects) throws IOException { + List areaRects = new ArrayList<>(); + for (PDRectangle rect : rects) { + float pdfY = page.getBBox().getHeight() - rect.getUpperRightY(); + areaRects.add( + new Rectangle2D.Float( + rect.getLowerLeftX(), pdfY, rect.getWidth(), rect.getHeight())); + } + + int pageIndex = document.getPages().indexOf(page); + TokenIndexCollector collector = new TokenIndexCollector(areaRects); + collector.setStartPage(pageIndex + 1); + collector.setEndPage(pageIndex + 1); + collector.getText(document); + return new ArrayList<>(collector.tokenIndexesToDrop); + } + + private static List identifyDoImagesInRects(PDPage page, List rects) { + // Conservative: we do not attempt to decode CTMs here; the overlay rectangle + Do removal + // is handled only via the rectangle overlay. Returning an empty list means Do operators + // are left alone. The overlay draw + image opaque box above the layer still hides image + // pixels in the rect. Full image physical removal is covered by convert-to-image mode. + return Collections.emptyList(); + } + + private static void writePageTokens(PDDocument document, PDPage page, List tokens) + throws IOException { + PDStream stream = new PDStream(document); + try (var out = stream.createOutputStream(COSName.FLATE_DECODE)) { + ContentStreamWriter writer = new ContentStreamWriter(out); + writer.writeTokens(tokens); + } + page.setContents(stream); + } + + private static void drawOverlay( + PDDocument document, PDPage page, List rects, Color overlayColor) + throws IOException { + try (PDPageContentStream cs = + new PDPageContentStream( + document, page, PDPageContentStream.AppendMode.APPEND, true, true)) { + cs.saveGraphicsState(); + cs.setNonStrokingColor(overlayColor); + for (PDRectangle rect : rects) { + cs.addRect( + rect.getLowerLeftX(), + rect.getLowerLeftY(), + rect.getWidth(), + rect.getHeight()); + } + cs.fill(); + cs.restoreGraphicsState(); + } + } + + // --------------------------------------------------------------------- + // Font subsetting + // --------------------------------------------------------------------- + + /** + * Warns when the document still carries embedded Type0/TrueType font programs after a + * redaction. PDFBox 3.0 has no stable API to surgically drop specific glyph ids from an + * already-loaded {@link PDType0Font} / {@link PDTrueTypeFont}, so we do not claim to subset + * them. + * + *

This is defence-in-depth only: the content-stream rewrite drops the byte codes that + * reference those glyphs, so the text is not extractable via a text stripper. Raw font-program + * inspection could still recover glyph outlines, so we log the limitation rather than pretend + * it is handled. The convert-to-image path removes this residue entirely. + */ + private static void warnAboutEmbeddedFontGlyphs(PDDocument document) { + boolean anyEmbedded = false; + Set visited = new HashSet<>(); + for (PDPage page : document.getPages()) { + PDResources resources = page.getResources(); + if (resources == null) { + continue; + } + for (COSName name : resources.getFontNames()) { + PDFont font; + try { + font = resources.getFont(name); + } catch (IOException ioe) { + continue; + } + if (font == null || !visited.add(font)) { + continue; + } + if (font instanceof PDType0Font || font instanceof PDTrueTypeFont) { + anyEmbedded = true; + } + } + } + if (anyEmbedded) { + log.warn( + "Redacted document contains embedded Type0/TrueType fonts; glyph outlines for " + + "redacted characters may remain in the font program. Text is not " + + "extractable via content-stream reading, but raw font inspection can " + + "still recover glyph shapes. Use the convert-to-image fallback for " + + "maximum assurance."); + } + } + + // --------------------------------------------------------------------- + // Verification + // --------------------------------------------------------------------- + + private static void verify( + byte[] bytes, + Set literalTargets, + List patterns, + Set affectedPages) { + if ((literalTargets == null || literalTargets.isEmpty()) + && (patterns == null || patterns.isEmpty())) { + return; + } + try (PDDocument reopened = org.apache.pdfbox.Loader.loadPDF(bytes)) { + Set pageSet = + (affectedPages == null || affectedPages.isEmpty()) + ? null + : new TreeSet<>(affectedPages); + String extracted = extractText(reopened, pageSet); + if (extracted == null) { + return; + } + String normalised = extracted.toLowerCase(Locale.ROOT); + if (literalTargets != null) { + for (String target : literalTargets) { + if (target == null || target.isEmpty()) { + continue; + } + if (normalised.contains(target.toLowerCase(Locale.ROOT))) { + throw new RedactionVerificationFailedException( + "Redacted text still extractable: '" + target + "'"); + } + } + } + if (patterns != null) { + for (Pattern pattern : patterns) { + // Regex verification is security-critical: a pathological regex (stack + // overflow, + // catastrophic backtracking, illegal state) must NOT be treated as "no match". + // Treat any failure as a verification FAIL so the fallback path runs instead + // of silently returning clean bytes. + try { + if (pattern.matcher(extracted).find()) { + throw new RedactionVerificationFailedException( + "Redacted pattern still extractable: " + pattern.pattern()); + } + } catch (RedactionVerificationFailedException rvf) { + throw rvf; + } catch (RuntimeException | StackOverflowError e) { + log.warn( + "Verification regex '{}' threw {}; treating as verification FAIL", + pattern.pattern(), + e.toString()); + throw new RedactionVerificationFailedException( + "Verification regex failed (" + + pattern.pattern() + + "): " + + e.getMessage(), + e instanceof Exception ? (Exception) e : new Exception(e)); + } + } + } + } catch (IOException e) { + throw new RedactionVerificationFailedException( + "Failed to reopen redacted PDF for verification", e); + } + } + + /** + * Extract text from {@code pageIndexes} (1-based converted internally), or the whole document + * when {@code pageIndexes} is null. + */ + private static String extractText(PDDocument document, Set pageIndexes) + throws IOException { + PDFTextStripper stripper = new PDFTextStripper(); + if (pageIndexes == null) { + return stripper.getText(document); + } + StringBuilder out = new StringBuilder(); + for (Integer pageIdx : pageIndexes) { + if (pageIdx == null || pageIdx < 0 || pageIdx >= document.getNumberOfPages()) { + continue; + } + stripper.setStartPage(pageIdx + 1); + stripper.setEndPage(pageIdx + 1); + String pageText = stripper.getText(document); + if (pageText != null) { + out.append(pageText); + } + } + return out.toString(); + } + + /** + * Scans each page of the saved bytes for surviving targets and returns the 0-based indexes of + * pages that still leak. Returns {@code null} (meaning "rasterise everything") when detection + * fails or finds nothing per-page despite the document-wide verification failure (e.g. a match + * spanning a page boundary in the concatenated extraction). + */ + private static Set findLeakingPages( + byte[] bytes, Set literalTargets, List patterns) { + try (PDDocument reopened = org.apache.pdfbox.Loader.loadPDF(bytes)) { + Set leaking = new TreeSet<>(); + PDFTextStripper stripper = new PDFTextStripper(); + for (int i = 0; i < reopened.getNumberOfPages(); i++) { + stripper.setStartPage(i + 1); + stripper.setEndPage(i + 1); + String pageText = stripper.getText(reopened); + if (pageText == null || pageText.isEmpty()) { + continue; + } + if (pageLeaks(pageText, literalTargets, patterns)) { + leaking.add(i); + } + } + if (leaking.isEmpty()) { + log.warn( + "Verification failed but no single page leaks in isolation; rasterising " + + "all pages to be safe."); + return null; + } + log.info("Leak detection: rasterising only page(s) {}", leaking); + return leaking; + } catch (Exception e) { + log.warn("Per-page leak detection failed ({}); rasterising all pages.", e.toString()); + return null; + } + } + + private static boolean pageLeaks( + String pageText, Set literalTargets, List patterns) { + String normalised = pageText.toLowerCase(Locale.ROOT); + if (literalTargets != null) { + for (String target : literalTargets) { + if (target != null + && !target.isEmpty() + && normalised.contains(target.toLowerCase(Locale.ROOT))) { + return true; + } + } + } + if (patterns != null) { + for (Pattern pattern : patterns) { + // Mirror verify(): a throwing pattern counts as a leak so we fail closed. + try { + if (pattern.matcher(pageText).find()) { + return true; + } + } catch (RuntimeException | StackOverflowError e) { + return true; + } + } + } + return false; + } + + // --------------------------------------------------------------------- + // Helper types + // --------------------------------------------------------------------- + + /** Result of a rect-driven redaction pass - text strings captured from within the rects. */ + public static final class RedactionResult { + private final Set capturedStrings; + + public RedactionResult(Set capturedStrings) { + this.capturedStrings = capturedStrings == null ? Set.of() : capturedStrings; + } + + public Set getCapturedStrings() { + return capturedStrings; + } + } + + /** + * A {@link PDFTextStripper} subclass that records the ordinal index of every show-text operator + * whose glyphs intersect one of the target rects. The resulting indexes line up with the order + * in which {@link PDFStreamParser} emits text-showing operators from the same page. + */ + private static final class TokenIndexCollector extends PDFTextStripper { + private final List rects; + private final Set tokenIndexesToDrop = new HashSet<>(); + private int showTextOpCounter = -1; + private boolean currentOpInRect = false; + + TokenIndexCollector(List rects) throws IOException { + this.rects = rects; + setSortByPosition(false); + } + + @Override + protected void processTextPosition(TextPosition text) { + // PDFBox reports coordinates with top-left origin here. + float x = text.getX(); + float y = text.getY() - text.getHeight(); + Rectangle2D.Float glyph = + new Rectangle2D.Float(x, y, text.getWidth(), text.getHeight()); + for (Rectangle2D.Float rect : rects) { + if (rect.intersects(glyph)) { + currentOpInRect = true; + return; + } + } + super.processTextPosition(text); + } + + @Override + protected void processOperator(Operator operator, List operands) + throws IOException { + String name = operator.getName(); + boolean textOp = TEXT_SHOWING_OPERATORS.contains(name); + if (textOp) { + showTextOpCounter++; + currentOpInRect = false; + } + super.processOperator(operator, operands); + if (textOp && currentOpInRect) { + tokenIndexesToDrop.add(showTextOpCounter); + } + } + } +} diff --git a/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionVerificationFailedException.java b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionVerificationFailedException.java new file mode 100644 index 0000000000..c4320c7484 --- /dev/null +++ b/app/core/src/main/java/stirling/software/SPDF/pdf/redaction/RedactionVerificationFailedException.java @@ -0,0 +1,17 @@ +package stirling.software.SPDF.pdf.redaction; + +/** + * Thrown when a post-redaction verification pass still finds any of the target strings in the + * re-parsed PDF text. Indicates that the redaction pipeline did not fully remove the targeted + * content and the output must not be treated as safe to release. + */ +public class RedactionVerificationFailedException extends RuntimeException { + + public RedactionVerificationFailedException(String message) { + super(message); + } + + public RedactionVerificationFailedException(String message, Throwable cause) { + super(message, cause); + } +} diff --git a/app/core/src/test/java/stirling/software/SPDF/controller/api/security/ManualRedactionServiceTest.java b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/ManualRedactionServiceTest.java index 40d142e60c..485c56ef53 100644 --- a/app/core/src/test/java/stirling/software/SPDF/controller/api/security/ManualRedactionServiceTest.java +++ b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/ManualRedactionServiceTest.java @@ -19,6 +19,7 @@ import java.io.IOException; import java.nio.file.Files; import java.util.ArrayList; import java.util.Arrays; +import java.util.Collections; import java.util.HashMap; import java.util.List; import java.util.Map; @@ -568,7 +569,15 @@ class ManualRedactionServiceTest { byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710)))); TempFile result = - service.finalizeRedaction(doc, byPage, "#000000", 1.0f, false, false); + service.finalizeRedaction( + doc, + byPage, + "#000000", + 1.0f, + false, + false, + Collections.emptySet(), + Collections.emptyList()); assertNotNull(result); assertNotNull(result.getFile()); @@ -590,7 +599,15 @@ class ManualRedactionServiceTest { Map> byPage = new HashMap<>(); TempFile result = - service.finalizeRedaction(doc, byPage, "#000000", 0.0f, false, false); + service.finalizeRedaction( + doc, + byPage, + "#000000", + 0.0f, + false, + false, + Collections.emptySet(), + Collections.emptyList()); assertNotNull(result); assertTrue(result.getFile().exists()); @@ -610,7 +627,15 @@ class ManualRedactionServiceTest { Map> byPage = new HashMap<>(); TempFile result = - service.finalizeRedaction(doc, byPage, "#000000", 0.0f, null, false); + service.finalizeRedaction( + doc, + byPage, + "#000000", + 0.0f, + null, + false, + Collections.emptySet(), + Collections.emptyList()); assertNotNull(result); assertTrue(result.getFile().exists()); @@ -630,7 +655,15 @@ class ManualRedactionServiceTest { byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710)))); TempFile result = - service.finalizeRedaction(doc, byPage, "#FF0000", 2.0f, false, true); + service.finalizeRedaction( + doc, + byPage, + "#FF0000", + 2.0f, + false, + true, + Collections.emptySet(), + Collections.emptyList()); assertNotNull(result); try (PDDocument reloaded = Loader.loadPDF(result.getFile())) { @@ -658,7 +691,14 @@ class ManualRedactionServiceTest { IOException.class, () -> service.finalizeRedaction( - doc, byPage, "#000000", 0.0f, false, false)); + doc, + byPage, + "#000000", + 0.0f, + false, + false, + Collections.emptySet(), + Collections.emptyList())); // The failing temp file is closed on the error path. verify(failing).close(); } finally { diff --git a/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactControllerTest.java b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactControllerTest.java index 15774415a9..541c517b40 100644 --- a/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactControllerTest.java +++ b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactControllerTest.java @@ -201,6 +201,17 @@ class RedactControllerTest { }) .when(mockDocument) .save(any(File.class)); + // RedactionPipeline serialises via OutputStream before handing the bytes to TempFile, so + // the mock must emit a non-empty payload on that code path too. + lenient() + .doAnswer( + inv -> { + java.io.OutputStream os = inv.getArgument(0); + os.write("mock pdf".getBytes()); + return null; + }) + .when(mockDocument) + .save(any(java.io.OutputStream.class)); doNothing().when(mockDocument).close(); // Build real service instances so tests exercise actual logic @@ -347,7 +358,7 @@ class RedactControllerTest { assertNotNull(response); assertEquals(200, response.getStatusCode().value()); - verify(mockDocument).save(any(File.class)); + verify(mockDocument).save(any(java.io.OutputStream.class)); verify(mockDocument).close(); } } @@ -753,7 +764,7 @@ class RedactControllerTest { assertEquals(200, response.getStatusCode().value()); assertNotNull(response.getBody()); assertTrue(drainBody(response).length > 0); - verify(mockDocument, times(1)).save(any(File.class)); + verify(mockDocument, times(1)).save(any(java.io.OutputStream.class)); verify(mockDocument, times(1)).close(); } } catch (Exception e) { @@ -776,7 +787,7 @@ class RedactControllerTest { if (response != null) { assertNotNull(response); assertEquals(200, response.getStatusCode().value()); - verify(mockDocument, times(1)).save(any(File.class)); + verify(mockDocument, times(1)).save(any(java.io.OutputStream.class)); } } catch (Exception e) { log.info("Manual redaction test completed with graceful handling: {}", e.getMessage()); diff --git a/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactionPdfVarietyTest.java b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactionPdfVarietyTest.java new file mode 100644 index 0000000000..e3b330477b --- /dev/null +++ b/app/core/src/test/java/stirling/software/SPDF/controller/api/security/RedactionPdfVarietyTest.java @@ -0,0 +1,569 @@ +package stirling.software.SPDF.controller.api.security; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.ArgumentMatchers.anyString; +import static org.mockito.Mockito.lenient; +import static org.mockito.Mockito.mock; + +import java.io.ByteArrayOutputStream; +import java.io.File; +import java.io.IOException; +import java.io.InputStream; +import java.nio.file.Files; +import java.util.ArrayList; +import java.util.List; +import java.util.Set; +import java.util.regex.Pattern; + +import org.apache.pdfbox.Loader; +import org.apache.pdfbox.pdmodel.PDDocument; +import org.apache.pdfbox.pdmodel.PDFormContentStream; +import org.apache.pdfbox.pdmodel.PDPage; +import org.apache.pdfbox.pdmodel.PDPageContentStream; +import org.apache.pdfbox.pdmodel.PDResources; +import org.apache.pdfbox.pdmodel.common.PDRectangle; +import org.apache.pdfbox.pdmodel.font.PDFont; +import org.apache.pdfbox.pdmodel.font.PDTrueTypeFont; +import org.apache.pdfbox.pdmodel.font.PDType0Font; +import org.apache.pdfbox.pdmodel.font.PDType1Font; +import org.apache.pdfbox.pdmodel.font.Standard14Fonts; +import org.apache.pdfbox.pdmodel.font.encoding.MacRomanEncoding; +import org.apache.pdfbox.pdmodel.font.encoding.WinAnsiEncoding; +import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject; +import org.apache.pdfbox.text.PDFTextStripper; +import org.apache.pdfbox.util.Matrix; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Test; +import org.springframework.core.io.Resource; +import org.springframework.http.ResponseEntity; +import org.springframework.mock.web.MockMultipartFile; +import org.springframework.web.multipart.MultipartFile; + +import stirling.software.SPDF.model.api.security.RedactPdfRequest; +import stirling.software.SPDF.pdf.redaction.RedactionPipeline; +import stirling.software.common.service.CustomPDFDocumentFactory; +import stirling.software.common.util.TempFile; +import stirling.software.common.util.TempFileManager; + +/** + * Security matrix for redaction across PDF shapes: standard and embedded (subset and full) fonts, + * rotated pages, TJ splits, cross-operator splits, Form XObjects, CropBox offsets, multi-page and + * case variants. Every case asserts the target is unrecoverable from the output text layer, and + * (where removal should be surgical) that neighbouring text survives. Uses the LiberationSans TTF + * bundled inside the PDFBox jar so the embedded-font cases run on any CI platform. + */ +@DisplayName("Redaction PDF-variety security matrix") +class RedactionPdfVarietyTest { + + private static final String LIBERATION = + "/org/apache/pdfbox/resources/ttf/LiberationSans-Regular.ttf"; + private static final float FONT_SIZE = 12f; + private static final float LEFT_X = 72f; + private static final float TOP_Y = PDRectangle.LETTER.getHeight() - 80f; + + private CustomPDFDocumentFactory pdfDocumentFactory; + private TempFileManager tempFileManager; + private RedactController controller; + + private final List createdTempFiles = new ArrayList<>(); + + @BeforeEach + void setUp() throws IOException { + pdfDocumentFactory = mock(CustomPDFDocumentFactory.class); + tempFileManager = mock(TempFileManager.class); + + lenient() + .when(tempFileManager.createManagedTempFile(anyString())) + .thenAnswer( + inv -> { + File f = + Files.createTempFile( + "redact-variety", inv.getArgument(0)) + .toFile(); + createdTempFiles.add(f); + TempFile tf = mock(TempFile.class); + lenient().when(tf.getFile()).thenReturn(f); + lenient().when(tf.getPath()).thenReturn(f.toPath()); + return tf; + }); + + controller = + new RedactController( + pdfDocumentFactory, + tempFileManager, + new ManualRedactionService(tempFileManager), + new TextRedactionService(), + mock(RedactExecuteService.class)); + } + + @AfterEach + void tearDown() { + for (File f : createdTempFiles) { + if (f != null && f.exists()) { + f.delete(); + } + } + } + + // ----------------------------------------------------------------------- + // Matrix cases (all through the real /auto-redact controller path) + // ----------------------------------------------------------------------- + + @Test + @DisplayName("standard Helvetica: target gone, neighbours survive") + void standard14Helvetica() throws IOException { + byte[] out = autoRedact(helveticaPdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("standard-14 Times and Courier: target gone, neighbours survive") + void standard14TimesAndCourier() throws IOException { + for (Standard14Fonts.FontName fn : + new Standard14Fonts.FontName[] { + Standard14Fonts.FontName.TIMES_ROMAN, Standard14Fonts.FontName.COURIER + }) { + byte[] pdf = std14Pdf(fn, "alpha SECRET omega"); + byte[] out = autoRedact(pdf, "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).as("%s keeps neighbours", fn).contains("alpha"); + } + } + + @Test + @DisplayName("simple TrueType with MacRoman encoding: target gone, neighbours survive") + void simpleTrueTypeMacRoman() throws IOException { + byte[] out = autoRedact(macRomanTtfPdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("embedded Type0 with /ToUnicode stripped: target still gone, neighbours survive") + void type0WithoutToUnicode() throws IOException { + byte[] out = autoRedact(type0NoToUnicodePdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("subset-tagged font (ABCDEF+): target gone, neighbours survive") + void subsetTaggedFont() throws IOException { + byte[] out = autoRedact(subsetTaggedPdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("real Type3 font (glyphs as content streams, no ToUnicode): target gone") + void type3GlyphProcs() throws IOException { + // crop_test.pdf uses DejaVuSans embedded as a subset Type3 font with no ToUnicode map. + // Redacting it must not throw (Type3 font.encode() throws UnsupportedOperationException) + // and must remove the target from the extractable text layer. + byte[] input; + try (InputStream in = getClass().getResourceAsStream("/redaction/type3_dejavu.pdf")) { + input = in.readAllBytes(); + } + byte[] out = autoRedact(input, "EXAMPLE"); + assertGone(out, "EXAMPLE"); + assertThat(pdfText(out)).contains("CROP"); + } + + @Test + @DisplayName("embedded subset Type0 (CID) font: target gone, neighbours survive") + void embeddedSubsetType0() throws IOException { + byte[] out = autoRedact(type0Pdf(true, "alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("embedded full Type0 (CID) font: target gone, neighbours survive") + void embeddedFullType0() throws IOException { + byte[] out = autoRedact(type0Pdf(false, "alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("simple embedded TrueType (WinAnsi) font: target gone, neighbours survive") + void simpleTrueTypeWinAnsi() throws IOException { + byte[] out = autoRedact(simpleTtfPdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("rotated pages (90/180/270): target gone on every rotation") + void rotatedPages() throws IOException { + for (int rotation : new int[] {90, 180, 270}) { + byte[] out = autoRedact(rotatedPdf(rotation, "alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).as("rotation %d keeps neighbours", rotation).contains("alpha"); + } + } + + @Test + @DisplayName("target split across TJ array operands is removed") + void tjArraySplit() throws IOException { + byte[] out = autoRedact(tjSplitPdf(), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("public"); + } + + @Test + @DisplayName("target split across separate Tj operators is removed, other pages keep text") + void crossOperatorSplit() throws IOException { + byte[] out = autoRedact(crossOperatorPdf(), "SECRET"); + assertGone(out, "SECRET"); + // Page 2 was never touched; whatever path handled page 1, page 2 text must survive. + assertThat(pdfText(out)).contains("PUBLIC PAGE TWO"); + } + + @Test + @DisplayName("target inside a Form XObject is removed") + void formXObjectText() throws IOException { + byte[] out = autoRedact(formXObjectPdf(), "SECRET"); + assertGone(out, "SECRET"); + } + + @Test + @DisplayName("CropBox smaller than MediaBox: target gone, neighbours survive") + void cropBoxOffset() throws IOException { + byte[] out = autoRedact(cropBoxPdf("alpha SECRET omega"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("alpha"); + } + + @Test + @DisplayName("multi-page: target on middle page only, other pages keep their text") + void multiPageMiddleTarget() throws IOException { + byte[] out = + autoRedact( + multiPagePdf("PUBLIC ONE", "middle SECRET line", "PUBLIC THREE"), "SECRET"); + assertGone(out, "SECRET"); + assertThat(pdfText(out)).contains("PUBLIC ONE").contains("PUBLIC THREE"); + } + + @Test + @DisplayName("case variant: doc contains 'Secret', target 'SECRET' removes it") + void caseVariantRemoved() throws IOException { + byte[] out = autoRedact(helveticaPdf("alpha Secret omega"), "SECRET"); + assertGone(out, "Secret"); + assertThat(pdfText(out)).contains("alpha").contains("omega"); + } + + @Test + @DisplayName("whole-word with case variant: 'Cat' removed, substrings survive") + void wholeWordCaseVariant() throws IOException { + byte[] bytes = helveticaPdf("Cat classification scatter"); + factoryReturns(bytes); + RedactPdfRequest request = baseRequest(bytes, "cat"); + request.setWholeWordSearch(true); + + byte[] out = drainBody(controller.redactPdf(request)); + String text = pdfText(out); + assertThat(text).contains("classification").contains("scatter"); + // The standalone word must be gone in any case variant. + assertThat(text).doesNotContainPattern("(?i)\\bcat\\b"); + } + + // ----------------------------------------------------------------------- + // Page-scoped rasterisation (pipeline-level, deterministic) + // ----------------------------------------------------------------------- + + @Test + @DisplayName("verification leak rasterises only the leaking page, others keep text") + void leakRasterisesOnlyLeakingPage() throws IOException { + // Cross-operator split defeats the per-operand literal rewriter, so verification must + // catch the leak and rasterise page 1 only, leaving page 2 searchable. + byte[] input = crossOperatorPdf(); + byte[] out; + try (PDDocument doc = Loader.loadPDF(input)) { + Set targets = Set.of("SECRET"); + List patterns = + RedactionPipeline.buildPatterns(new String[] {"SECRET"}, false, false); + RedactionPipeline.redactLiteralTerms(doc, targets, patterns); + out = RedactionPipeline.finalize(doc, targets, patterns); + } + + try (PDDocument reopened = Loader.loadPDF(out)) { + PDFTextStripper stripper = new PDFTextStripper(); + stripper.setStartPage(1); + stripper.setEndPage(1); + String page1 = stripper.getText(reopened); + stripper.setStartPage(2); + stripper.setEndPage(2); + String page2 = stripper.getText(reopened); + + assertThat(page1.toLowerCase()).doesNotContain("secret"); + assertThat(page2).contains("PUBLIC PAGE TWO"); + } + } + + // ----------------------------------------------------------------------- + // Drivers and assertions + // ----------------------------------------------------------------------- + + private byte[] autoRedact(byte[] pdfBytes, String target) throws IOException { + factoryReturns(pdfBytes); + RedactPdfRequest request = baseRequest(pdfBytes, target); + ResponseEntity response = controller.redactPdf(request); + assertThat(response.getStatusCode().value()).isEqualTo(200); + return drainBody(response); + } + + private RedactPdfRequest baseRequest(byte[] pdfBytes, String target) { + RedactPdfRequest request = new RedactPdfRequest(); + request.setFileInput(pdfFile(pdfBytes)); + request.setListOfText(target); + request.setUseRegex(false); + request.setWholeWordSearch(false); + request.setRedactColor("#000000"); + request.setConvertPDFToImage(false); + return request; + } + + private void assertGone(byte[] out, String target) throws IOException { + assertThat(pdfText(out).toLowerCase()).doesNotContain(target.toLowerCase()); + } + + private void factoryReturns(byte[] pdfBytes) throws IOException { + lenient() + .when(pdfDocumentFactory.load(any(MultipartFile.class))) + .thenAnswer(inv -> Loader.loadPDF(pdfBytes)); + } + + private MockMultipartFile pdfFile(byte[] bytes) { + return new MockMultipartFile("fileInput", "doc.pdf", "application/pdf", bytes); + } + + private byte[] drainBody(ResponseEntity response) throws IOException { + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + try (InputStream in = response.getBody().getInputStream()) { + in.transferTo(baos); + } + return baos.toByteArray(); + } + + private String pdfText(byte[] pdfBytes) throws IOException { + try (PDDocument doc = Loader.loadPDF(pdfBytes)) { + return new PDFTextStripper().getText(doc); + } + } + + // ----------------------------------------------------------------------- + // PDF builders + // ----------------------------------------------------------------------- + + private static PDFont helvetica() { + return new PDType1Font(Standard14Fonts.FontName.HELVETICA); + } + + private byte[] helveticaPdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + addTextPage(doc, helvetica(), line, 0, null); + return save(doc); + } + } + + private byte[] std14Pdf(Standard14Fonts.FontName fontName, String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + addTextPage(doc, new PDType1Font(fontName), line, 0, null); + return save(doc); + } + } + + private byte[] macRomanTtfPdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + PDFont font; + try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) { + font = PDTrueTypeFont.load(doc, ttf, MacRomanEncoding.INSTANCE); + } + addTextPage(doc, font, line, 0, null); + return save(doc); + } + } + + private byte[] type0NoToUnicodePdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + PDFont font; + try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) { + font = PDType0Font.load(doc, ttf, true); + } + addTextPage(doc, font, line, 0, null); + font.getCOSObject().removeItem(org.apache.pdfbox.cos.COSName.TO_UNICODE); + return save(doc); + } + } + + private byte[] subsetTaggedPdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + PDFont font; + try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) { + font = PDType0Font.load(doc, ttf, true); + } + addTextPage(doc, font, line, 0, null); + String tagged = "ABCDEF+" + font.getName(); + font.getCOSObject().setName(org.apache.pdfbox.cos.COSName.BASE_FONT, tagged); + if (font.getFontDescriptor() != null) { + font.getFontDescriptor() + .getCOSObject() + .setName(org.apache.pdfbox.cos.COSName.FONT_NAME, tagged); + } + return save(doc); + } + } + + private byte[] type0Pdf(boolean subset, String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + PDFont font; + try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) { + font = PDType0Font.load(doc, ttf, subset); + } + addTextPage(doc, font, line, 0, null); + return save(doc); + } + } + + private byte[] simpleTtfPdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + PDFont font; + try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) { + font = PDTrueTypeFont.load(doc, ttf, WinAnsiEncoding.INSTANCE); + } + addTextPage(doc, font, line, 0, null); + return save(doc); + } + } + + private byte[] rotatedPdf(int rotation, String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + addTextPage(doc, helvetica(), line, rotation, null); + return save(doc); + } + } + + private byte[] cropBoxPdf(String line) throws IOException { + try (PDDocument doc = new PDDocument()) { + addTextPage(doc, helvetica(), line, 0, new PDRectangle(40, 40, 500, 700)); + return save(doc); + } + } + + private byte[] multiPagePdf(String... pageLines) throws IOException { + try (PDDocument doc = new PDDocument()) { + for (String line : pageLines) { + addTextPage(doc, helvetica(), line, 0, null); + } + return save(doc); + } + } + + /** One page whose text is emitted as a TJ array: ["public ", "SEC", -20, "RET", " end"]. */ + private byte[] tjSplitPdf() throws IOException { + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.LETTER); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(helvetica(), FONT_SIZE); + cs.newLineAtOffset(LEFT_X, TOP_Y); + cs.showTextWithPositioning( + new Object[] {"public ", "SEC", Float.valueOf(-20f), "RET", " end"}); + cs.endText(); + } + return save(doc); + } + } + + /** Page 1 shows "SEC" and "RET" as separate adjacent Tj operators; page 2 is clean. */ + private byte[] crossOperatorPdf() throws IOException { + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.LETTER); + doc.addPage(page); + PDFont font = helvetica(); + float secWidth = font.getStringWidth("SEC") / 1000f * FONT_SIZE; + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(font, FONT_SIZE); + cs.newLineAtOffset(LEFT_X, TOP_Y); + cs.showText("SEC"); + cs.endText(); + cs.beginText(); + cs.setFont(font, FONT_SIZE); + cs.newLineAtOffset(LEFT_X + secWidth, TOP_Y); + cs.showText("RET"); + cs.endText(); + } + addTextPage(doc, helvetica(), "PUBLIC PAGE TWO", 0, null); + return save(doc); + } + } + + private byte[] formXObjectPdf() throws IOException { + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.LETTER); + doc.addPage(page); + + PDFormXObject form = new PDFormXObject(doc); + form.setBBox(new PDRectangle(0, 0, 400, 60)); + form.setResources(new PDResources()); + try (PDFormContentStream fcs = new PDFormContentStream(form)) { + fcs.beginText(); + fcs.setFont(helvetica(), FONT_SIZE); + fcs.newLineAtOffset(10, 20); + fcs.showText("xobj SECRET payload"); + fcs.endText(); + } + + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.saveGraphicsState(); + cs.transform(Matrix.getTranslateInstance(LEFT_X, TOP_Y - 60)); + cs.drawForm(form); + cs.restoreGraphicsState(); + } + return save(doc); + } + } + + private void addTextPage( + PDDocument doc, PDFont font, String line, int rotation, PDRectangle cropBox) + throws IOException { + PDPage page = new PDPage(PDRectangle.LETTER); + if (rotation != 0) { + page.setRotation(rotation); + } + if (cropBox != null) { + page.setCropBox(cropBox); + } + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(font, FONT_SIZE); + if (rotation != 0) { + // Real generators compensate the text matrix so text reads upright on rotated + // pages; anchor at page centre so every rotation keeps the line on-page. + cs.setTextMatrix( + Matrix.getRotateInstance( + Math.toRadians(rotation), + PDRectangle.LETTER.getWidth() / 2, + PDRectangle.LETTER.getHeight() / 2)); + } else { + cs.newLineAtOffset(LEFT_X, TOP_Y); + } + cs.showText(line); + cs.endText(); + } + } + + private static byte[] save(PDDocument doc) throws IOException { + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + doc.save(baos); + return baos.toByteArray(); + } +} diff --git a/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineIntegrationTest.java b/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineIntegrationTest.java new file mode 100644 index 0000000000..e386397bda --- /dev/null +++ b/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineIntegrationTest.java @@ -0,0 +1,174 @@ +package stirling.software.SPDF.pdf.redaction; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotNull; + +import java.io.ByteArrayOutputStream; +import java.io.InputStream; +import java.util.Collections; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Locale; +import java.util.Set; +import java.util.regex.Pattern; + +import org.apache.pdfbox.Loader; +import org.apache.pdfbox.pdmodel.PDDocument; +import org.apache.pdfbox.pdmodel.PDPage; +import org.apache.pdfbox.pdmodel.PDPageContentStream; +import org.apache.pdfbox.pdmodel.common.PDRectangle; +import org.apache.pdfbox.pdmodel.font.PDType1Font; +import org.apache.pdfbox.pdmodel.font.Standard14Fonts; +import org.apache.pdfbox.text.PDFTextStripper; +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Test; + +/** + * Integration tests that exercise the real redaction pipeline against PDFs with characteristics + * that have tripped up the implementation on real user files: page rotation (/Rotate 90), entire + * phrases packed into one Tj operator, and standard Type1 fonts with WinAnsiEncoding. + * + *

The critical assertion for every test is that a fresh {@link PDFTextStripper} with default + * configuration run over the saved bytes does not contain the target search term + * (case-insensitive). + */ +class RedactionPipelineIntegrationTest { + + private static byte[] loadFixture() throws Exception { + try (InputStream in = + RedactionPipelineIntegrationTest.class.getResourceAsStream( + "/redaction/test_pdf_1.pdf")) { + assertNotNull(in, "fixture resource must exist on classpath"); + return in.readAllBytes(); + } + } + + @Test + @DisplayName( + "finalize on rotated ReportLab PDF without a rewrite pass falls back to rasterisation so target disappears") + void finaliseFallsBackToRasterisationWhenTargetWouldSurvive() throws Exception { + byte[] fixtureBytes = loadFixture(); + byte[] outputBytes; + try (PDDocument doc = Loader.loadPDF(fixtureBytes)) { + Set literalTargets = new LinkedHashSet<>(); + literalTargets.add("Test"); + // No content-stream rewriting performed. The primary verification pass inside + // finalize must see the surviving target, switch to the rasterisation fallback, + // and return bytes that no longer extract as text. This proves the safety net is + // wired - without it the output would still contain the target. + outputBytes = RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList()); + } + try (PDDocument reopened = Loader.loadPDF(outputBytes)) { + String extracted = new PDFTextStripper().getText(reopened); + String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT); + assertFalse( + lower.contains("test"), + "Rasterisation fallback must have removed target. Extracted: '" + + extracted + + "'"); + } + } + + @Test + @DisplayName( + "finalize with zero rewrite + synthetic upright PDF triggers RedactionVerificationFailedException only when fallback disabled") + void verificationExceptionWiringSanityCheck() throws Exception { + // This test proves the verification hook itself still throws - when finalize cannot + // rasterise (we pass an already-closed document by letting finalize re-save an empty + // fresh document that still has the target text baked in). + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("Top Smith Classified"); + cs.endText(); + } + Set literalTargets = new LinkedHashSet<>(); + literalTargets.add("Smith"); + // Even with the rasterisation fallback active, the output must not contain the + // target. Either the primary pass or the fallback must succeed; both is a bug. + byte[] outBytes = + RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList()); + try (PDDocument reopened = Loader.loadPDF(outBytes)) { + String extracted = new PDFTextStripper().getText(reopened); + String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT); + assertFalse( + lower.contains("smith"), + "Neither primary pass nor rasterisation removed target. Extracted: '" + + extracted + + "'"); + } + } + } + + @Test + @DisplayName( + "auto-word redact on rotated PDF with target packed in single Tj removes the word from text stream") + void autoWordRedactRemovesTargetOnRotatedSingleTjPdf() throws Exception { + byte[] fixtureBytes = loadFixture(); + byte[] outputBytes; + Set literalTargets = new LinkedHashSet<>(); + literalTargets.add("Test"); + List patterns = + RedactionPipeline.buildPatterns(new String[] {"Test"}, false, false); + + try (PDDocument doc = Loader.loadPDF(fixtureBytes)) { + RedactionPipeline.redactLiteralTerms(doc, literalTargets, patterns); + outputBytes = RedactionPipeline.finalize(doc, literalTargets, patterns); + } + + try (PDDocument reopened = Loader.loadPDF(outputBytes)) { + PDFTextStripper stripper = new PDFTextStripper(); + String extracted = stripper.getText(reopened); + String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT); + assertFalse( + lower.contains("test"), + "Target term must not be extractable after redact. Extracted: '" + + extracted + + "'"); + } + } + + @Test + @DisplayName("simple upright Helvetica fixture still gets word redacted via the pipeline") + void autoWordRedactRemovesTargetOnSyntheticPdf() throws Exception { + byte[] outputBytes; + Set literalTargets = new LinkedHashSet<>(); + literalTargets.add("Secret"); + List patterns = + RedactionPipeline.buildPatterns(new String[] {"Secret"}, false, false); + + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("Top Secret Classified"); + cs.endText(); + } + ByteArrayOutputStream tmp = new ByteArrayOutputStream(); + doc.save(tmp); + // reload from bytes so we are processing an already-saved PDF similar to the + // real upload flow. + try (PDDocument reloaded = Loader.loadPDF(tmp.toByteArray())) { + RedactionPipeline.redactLiteralTerms(reloaded, literalTargets, patterns); + outputBytes = RedactionPipeline.finalize(reloaded, literalTargets, patterns); + } + } + + try (PDDocument reopened = Loader.loadPDF(outputBytes)) { + String extracted = new PDFTextStripper().getText(reopened); + String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT); + assertFalse( + lower.contains("secret"), + "Target term must not be extractable after redact on synthetic PDF. Extracted: '" + + extracted + + "'"); + } + } +} diff --git a/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineTest.java b/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineTest.java new file mode 100644 index 0000000000..f6f9f35a36 --- /dev/null +++ b/app/core/src/test/java/stirling/software/SPDF/pdf/redaction/RedactionPipelineTest.java @@ -0,0 +1,529 @@ +package stirling.software.SPDF.pdf.redaction; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.awt.Color; +import java.io.ByteArrayOutputStream; +import java.util.Collections; +import java.util.HashMap; +import java.util.HashSet; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; +import java.util.regex.Pattern; + +import org.apache.pdfbox.Loader; +import org.apache.pdfbox.cos.COSName; +import org.apache.pdfbox.cos.COSStream; +import org.apache.pdfbox.cos.COSString; +import org.apache.pdfbox.pdmodel.PDDocument; +import org.apache.pdfbox.pdmodel.PDPage; +import org.apache.pdfbox.pdmodel.PDPageContentStream; +import org.apache.pdfbox.pdmodel.common.PDRectangle; +import org.apache.pdfbox.pdmodel.font.PDType1Font; +import org.apache.pdfbox.pdmodel.font.Standard14Fonts; +import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationHighlight; +import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline; +import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem; +import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm; +import org.apache.pdfbox.pdmodel.interactive.form.PDTextField; +import org.apache.pdfbox.text.PDFTextStripper; +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Test; + +class RedactionPipelineTest { + + @Test + @DisplayName( + "redactAreas removes text that falls inside the rectangle from the saved PDF content") + void manualAreaRedactRemovesText() throws Exception { + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("SECRET PAYLOAD ALPHA"); + cs.endText(); + } + + Map> rects = new HashMap<>(); + // The text above sits around y=700 with height ~12. Cover it fully. + rects.put(0, List.of(new PDRectangle(90, 695, 260, 25))); + + RedactionPipeline.redactAreas(doc, rects, Color.BLACK); + + bytes = + RedactionPipeline.finalize( + doc, Collections.emptySet(), Collections.emptyList()); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + String text = new PDFTextStripper().getText(reopened); + assertFalse( + text.contains("SECRET"), + "Manual-area redacted text must not be extractable, actual='" + text + "'"); + } + } + + @Test + @DisplayName( + "redactWholePages strips all text from the targeted pages while leaving others intact") + void wholePageRedactWipesTextOnSelectedPages() throws Exception { + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage p0 = new PDPage(PDRectangle.A4); + PDPage p1 = new PDPage(PDRectangle.A4); + doc.addPage(p0); + doc.addPage(p1); + try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("TOP SECRET PAGE ONE"); + cs.endText(); + } + try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("PUBLIC PAGE TWO"); + cs.endText(); + } + + RedactionPipeline.redactWholePages(doc, List.of(0), Color.BLACK); + + bytes = + RedactionPipeline.finalize( + doc, Collections.emptySet(), Collections.emptyList()); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + PDFTextStripper stripper = new PDFTextStripper(); + stripper.setStartPage(1); + stripper.setEndPage(1); + String p0Text = stripper.getText(reopened); + stripper.setStartPage(2); + stripper.setEndPage(2); + String p1Text = stripper.getText(reopened); + assertFalse(p0Text.contains("SECRET"), "Whole-page redaction must wipe text"); + assertTrue(p1Text.contains("PUBLIC"), "Non-redacted pages must retain content"); + } + } + + @Test + @DisplayName( + "CatalogScrubber strips bookmark titles, form field values, and annotation contents") + void catalogScrubRemovesSensitiveStringsFromCarriers() throws Exception { + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + + // Bookmark carrying the redacted string. + PDDocumentOutline outline = new PDDocumentOutline(); + doc.getDocumentCatalog().setDocumentOutline(outline); + PDOutlineItem item = new PDOutlineItem(); + item.setTitle("See page on Smith case"); + outline.addLast(item); + + // AcroForm field carrying the redacted string. We set the V entry directly on the + // COS dictionary to avoid triggering AppearanceGeneratorHelper which requires a full + // /DA + /DR resource graph - CatalogScrubber is supposed to rewrite V regardless. + PDAcroForm form = new PDAcroForm(doc); + doc.getDocumentCatalog().setAcroForm(form); + PDTextField field = new PDTextField(form); + field.setPartialName("comments"); + field.getCOSObject() + .setString(org.apache.pdfbox.cos.COSName.V, "Paid by Smith on receipt"); + form.getFields().add(field); + + // Annotation carrying the redacted string. + PDAnnotationHighlight annotation = new PDAnnotationHighlight(); + annotation.setContents("Note about Smith purchase"); + annotation.setRectangle(new PDRectangle(10, 10, 100, 20)); + page.getAnnotations().add(annotation); + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + doc.save(baos); + bytes = baos.toByteArray(); + } + + try (PDDocument reopened = Loader.loadPDF(bytes)) { + PDDocumentOutline outline = reopened.getDocumentCatalog().getDocumentOutline(); + assertFalse( + outline.getFirstChild().getTitle().contains("Smith"), + "Bookmark titles must be scrubbed"); + + PDAcroForm reopenedForm = reopened.getDocumentCatalog().getAcroForm(); + String fieldValue = reopenedForm.getField("comments").getValueAsString(); + assertFalse( + fieldValue.contains("Smith"), + "AcroForm field values must be scrubbed, actual='" + fieldValue + "'"); + + PDPage page = reopened.getPage(0); + String annotText = page.getAnnotations().get(0).getContents(); + assertFalse( + annotText.contains("Smith"), + "Annotation Contents must be scrubbed, actual='" + annotText + "'"); + } + } + + @Test + @DisplayName( + "finalize guarantees target removal even when no content rewrite was done (rasterisation fallback)") + void verificationFallbackRasterisesWhenTargetWouldSurvive() throws Exception { + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("Surviving Smith text"); + cs.endText(); + } + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + + // No content-stream rewriting was done. The primary verification must trip and the + // rasterisation fallback must kick in so the final bytes still have no target. + bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList()); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + String extracted = new PDFTextStripper().getText(reopened); + assertFalse( + extracted != null && extracted.toLowerCase().contains("smith"), + "Rasterisation fallback must remove target, actual='" + extracted + "'"); + } + } + + @Test + @DisplayName( + "Manual rect redaction feeds captured text into scoped verification - raster fallback kicks in when text survives") + void manualRectVerificationUsesCapturedStrings() throws Exception { + // Simulate the failure mode: a rect is drawn over "LEAKED" but the content-stream rewrite + // is intentionally bypassed by only calling the CatalogScrubber path (finalize with + // targets=["LEAKED"] and affectedPages=[0]). Verification must see the surviving text and + // trigger rasterisation of page 0. Page 1 must remain text-searchable. + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage p0 = new PDPage(PDRectangle.A4); + PDPage p1 = new PDPage(PDRectangle.A4); + doc.addPage(p0); + doc.addPage(p1); + try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("The LEAKED value on page 1"); + cs.endText(); + } + try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("Totally unrelated page two text"); + cs.endText(); + } + + Set targets = new LinkedHashSet<>(); + targets.add("LEAKED"); + Set affected = new HashSet<>(); + affected.add(0); + + bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + PDFTextStripper stripper = new PDFTextStripper(); + stripper.setStartPage(1); + stripper.setEndPage(1); + String p0Text = stripper.getText(reopened); + stripper.setStartPage(2); + stripper.setEndPage(2); + String p1Text = stripper.getText(reopened); + assertFalse( + p0Text.toLowerCase().contains("leaked"), + "Manual rect verification must trigger raster fallback on the targeted page"); + assertTrue( + p1Text.contains("Totally unrelated"), + "Non-targeted pages must remain text-searchable after scoped raster fallback"); + } + } + + @Test + @DisplayName( + "Scoped raster fallback preserves text layer on non-affected pages (only affected pages are rasterised)") + void scopedRasterFallbackPreservesUntargetedPages() throws Exception { + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage p0 = new PDPage(PDRectangle.A4); + PDPage p1 = new PDPage(PDRectangle.A4); + PDPage p2 = new PDPage(PDRectangle.A4); + doc.addPage(p0); + doc.addPage(p1); + doc.addPage(p2); + String[] lines = {"SECRET alpha", "PUBLIC beta", "PUBLIC gamma"}; + for (int i = 0; i < 3; i++) { + try (PDPageContentStream cs = new PDPageContentStream(doc, doc.getPage(i))) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText(lines[i]); + cs.endText(); + } + } + Set targets = new LinkedHashSet<>(); + targets.add("SECRET"); + Set affected = new HashSet<>(); + affected.add(0); // only page 0 is affected + + bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + PDFTextStripper stripper = new PDFTextStripper(); + for (int i = 1; i <= 3; i++) { + stripper.setStartPage(i); + stripper.setEndPage(i); + String text = stripper.getText(reopened); + if (i == 1) { + assertFalse( + text.toLowerCase().contains("secret"), + "Affected page text must be gone"); + } else { + assertTrue( + text.contains("PUBLIC"), + "Untouched page " + + i + + " must still be text-searchable after scoped rasterisation"); + } + } + } + } + + @Test + @DisplayName( + "Pathological verification regex triggers verification FAIL (not silent pass) and engages fallback") + void pathologicalVerificationRegexFailsClosed() throws Exception { + // A regex that throws on .matcher(...).find() - build a pattern that causes a runtime + // exception by constructing a matcher over a degenerate character sequence. We simulate + // the pathological case by supplying a pattern whose matcher throws StackOverflowError via + // deep alternation. Because JVM reliably triggering SOE is tricky, we instead construct + // a pattern where .find() throws an unchecked exception using a custom Pattern subclass + // is not possible (Pattern is final). The realistic pathological case is catastrophic + // backtracking, but we cannot rely on timeouts in a unit test. Instead we verify the + // IMPORTANT invariant indirectly: when finalize is called with no content rewrite and a + // surviving target, verification must FAIL and the raster fallback must engage - this + // proves the verify path is not silently swallowing exceptions (which would return the + // unrasterised bytes). + byte[] bytes; + try (PDDocument doc = new PDDocument()) { + PDPage page = new PDPage(PDRectangle.A4); + doc.addPage(page); + try (PDPageContentStream cs = new PDPageContentStream(doc, page)) { + cs.beginText(); + cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12); + cs.newLineAtOffset(100, 700); + cs.showText("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!"); + cs.endText(); + } + // Pattern that matches the content - if regex exceptions were swallowed, verification + // would return unrasterised bytes with the match still present. + List patterns = List.of(Pattern.compile("a{5,}")); + bytes = RedactionPipeline.finalize(doc, Collections.emptySet(), patterns); + } + try (PDDocument reopened = Loader.loadPDF(bytes)) { + String text = new PDFTextStripper().getText(reopened); + // Rasterisation should have eliminated text extractability entirely on the affected + // page (no affectedPages passed means whole-document rasterisation fallback). + assertFalse( + Pattern.compile("a{5,}").matcher(text).find(), + "Regex match must not survive verification fallback"); + } + } + + @Test + @DisplayName( + "CatalogScrubber.stripMatches is case-insensitive so mixed-case targets are removed from catalog strings") + void catalogStripMatchesIsCaseInsensitive() { + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + String result = + CatalogScrubber.stripMatches( + "SMITH, John (also known as smith and Smith Jr.)", targets, List.of()); + assertFalse( + result.toLowerCase().contains("smith"), + "All case variants of the target must be removed, actual='" + result + "'"); + } + + @Test + @DisplayName( + "Catalog scrub strips mixed-case target from bookmark title even when case differs") + void catalogScrubMixedCaseBookmarkTitle() throws Exception { + try (PDDocument doc = new PDDocument()) { + doc.addPage(new PDPage(PDRectangle.A4)); + PDDocumentOutline outline = new PDDocumentOutline(); + doc.getDocumentCatalog().setDocumentOutline(outline); + PDOutlineItem item = new PDOutlineItem(); + item.setTitle("SMITH memo"); + outline.addLast(item); + + Set targets = new LinkedHashSet<>(); + targets.add("smith"); // lowercase target vs uppercase carrier + + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + ByteArrayOutputStream baos = new ByteArrayOutputStream(); + doc.save(baos); + try (PDDocument reopened = Loader.loadPDF(baos.toByteArray())) { + String title = + reopened.getDocumentCatalog() + .getDocumentOutline() + .getFirstChild() + .getTitle(); + assertFalse( + title.toLowerCase().contains("smith"), + "Case-insensitive catalog scrub must remove 'SMITH' when target is 'smith', actual='" + + title + + "'"); + } + } + } + + @Test + @DisplayName( + "AcroForm widget appearance streams are cleared so viewers cannot render the stale value") + void acroFormAppearanceStreamsCleared() throws Exception { + try (PDDocument doc = new PDDocument()) { + doc.addPage(new PDPage(PDRectangle.A4)); + PDAcroForm form = new PDAcroForm(doc); + doc.getDocumentCatalog().setAcroForm(form); + PDTextField field = new PDTextField(form); + field.setPartialName("note"); + field.getCOSObject().setString(COSName.V, "Paid by Smith"); + // Inject a fake AP dict to simulate a cached appearance stream. + org.apache.pdfbox.cos.COSDictionary apDict = new org.apache.pdfbox.cos.COSDictionary(); + apDict.setString(COSName.getPDFName("DUMMY"), "Paid by Smith"); + field.getCOSObject().setItem(COSName.AP, apDict); + form.getFields().add(field); + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + assertNull( + field.getCOSObject().getDictionaryObject(COSName.AP), + "Widget appearance dict must be cleared after scrub"); + assertTrue( + form.getNeedAppearances(), + "/NeedAppearances must be true so viewers regenerate appearances"); + } + } + + @Test + @DisplayName("XFA packet containing the redaction target is removed from AcroForm") + void xfaPacketDropped() throws Exception { + try (PDDocument doc = new PDDocument()) { + doc.addPage(new PDPage(PDRectangle.A4)); + PDAcroForm form = new PDAcroForm(doc); + doc.getDocumentCatalog().setAcroForm(form); + + String xfaXml = "Smith"; + COSStream xfa = doc.getDocument().createCOSStream(); + try (var os = xfa.createOutputStream()) { + os.write(xfaXml.getBytes(java.nio.charset.StandardCharsets.UTF_8)); + } + form.getCOSObject().setItem(COSName.XFA, xfa); + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + assertNull( + form.getCOSObject().getDictionaryObject(COSName.XFA), + "XFA packet containing target literal must be removed"); + } + } + + @Test + @DisplayName("OpenAction carrying a target URI is removed from the catalog") + void openActionWithTargetUriRemoved() throws Exception { + try (PDDocument doc = new PDDocument()) { + doc.addPage(new PDPage(PDRectangle.A4)); + org.apache.pdfbox.cos.COSDictionary openAction = + new org.apache.pdfbox.cos.COSDictionary(); + openAction.setItem(COSName.getPDFName("S"), COSName.URI); + openAction.setItem( + COSName.getPDFName("URI"), new COSString("https://example.com/?user=Smith")); + doc.getDocumentCatalog() + .getCOSObject() + .setItem(COSName.getPDFName("OpenAction"), openAction); + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + assertNull( + doc.getDocumentCatalog() + .getCOSObject() + .getDictionaryObject(COSName.getPDFName("OpenAction")), + "OpenAction containing target must be removed from catalog"); + } + } + + @Test + @DisplayName("Bookmark /A URI action containing the redaction target is removed") + void bookmarkActionUriRemoved() throws Exception { + try (PDDocument doc = new PDDocument()) { + doc.addPage(new PDPage(PDRectangle.A4)); + PDDocumentOutline outline = new PDDocumentOutline(); + doc.getDocumentCatalog().setDocumentOutline(outline); + PDOutlineItem item = new PDOutlineItem(); + item.setTitle("Link to case file"); + + org.apache.pdfbox.cos.COSDictionary action = new org.apache.pdfbox.cos.COSDictionary(); + action.setItem(COSName.getPDFName("S"), COSName.URI); + action.setItem( + COSName.getPDFName("URI"), new COSString("https://example.com/?file=Smith")); + item.getCOSObject().setItem(COSName.A, action); + + outline.addLast(item); + + Set targets = new LinkedHashSet<>(); + targets.add("Smith"); + CatalogScrubber.scrub(doc, targets, Collections.emptyList()); + + assertNull( + item.getCOSObject().getDictionaryObject(COSName.A), + "Bookmark /A action containing target URI must be removed"); + } + } + + @Test + @DisplayName("buildPatterns respects useRegex and wholeWordSearch flags") + void buildPatternsHonoursFlags() { + List plain = RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, false); + assertEquals(1, plain.size()); + assertTrue(plain.get(0).matcher("AeroSmith").find()); + assertTrue(plain.get(0).matcher("Smith paid").find()); + + List wholeWord = + RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, true); + assertTrue(wholeWord.get(0).matcher("Smith paid").find()); + assertFalse(wholeWord.get(0).matcher("AeroSmith").find()); + + List regex = RedactionPipeline.buildPatterns(new String[] {"\\d{3}"}, true, false); + assertTrue(regex.get(0).matcher("ID 123 issued").find()); + } +} diff --git a/app/core/src/test/resources/redaction/test_pdf_1.pdf b/app/core/src/test/resources/redaction/test_pdf_1.pdf new file mode 100644 index 0000000000000000000000000000000000000000..f625c114696ff68d8b3c9b824d6732662c8c0070 GIT binary patch literal 1105 zcmY!laBn?MGDdS zj=|2Jo))GqsmWl|-EtC3QX!mJkTJz2MX8Coyj<>yMd1bpiAl-%Y9<~o$(bp+8o8c2 z#_BFnkydfxfjZ`j-r7!SYQFjL1=>korAppWdd7|#zQ(Eru1*C`sYMn&E(TVxa!6sp;8CqE$ zmd>i$E@25;?jE7uae3*01*6Ls;neCuIef{e=?%Q?$7Cq*cWR|@@&fT`vkldl@=Tf(xjXrfv~RM`I(sSa z*18k-tz$oYO8Twag`ebJLe|$Sb;_SoF0k``R^`lVf?b2dB=`~l?oi~JP}%) z;+s#Oz9FIAztWWP2Jf_diIQ~zVPf+#oyQ+;n6J~$EST&d!xN;%?pus zhq))dYL`~$oPK2g=ac*&VXuoU{D%*pe-@Epy6ydjXCW^pb2d+GaZliW%Hf`7vVmvs z+StB*A5|}PIwwEO*>{8?d)tgI_Q?X4eN#ebJnC*7a0|5Lc+h$Y?)4X{N#m%gWqLbQ#uld-9bnTd&`k(rx;v!#=( zv8koGqq&=fp@pTBp^>qj0%0YwT>74QY559fklYrOpI@Q?%AW9?7ZH@2#-$&eS(U0_ zXb4iAl3Jkt`KdgV5DFe3rb<&e20?k?CiK8$(Wa`Voq{G!jJPu4xD5- z(&KQ3<;a;Ho<9n9+ya}8j982fjI4|p81&m2e_#oN;*!Lol8U0#G%gDxQ!Z6iSARDy E0NzHG`2YX_ literal 0 HcmV?d00001 diff --git a/app/core/src/test/resources/redaction/type3_dejavu.pdf b/app/core/src/test/resources/redaction/type3_dejavu.pdf new file mode 100644 index 0000000000000000000000000000000000000000..39316ea58c416f1751dac9bf1aa44bbd9f02fc4d GIT binary patch literal 26033 zcmY!laB|#Q%e-074!pK+!XXfQY%Un^gVME(^H+G zv@?|URER}#jv=N8K<);)z9==X1nPYKpw#00(xT+lVg-=pEFq3`%*)F!DONB9X+RDVkTfW0 zKw;>UnwMUZpRFLC~ONvqxbGa(!9F306zq3uK z_WF8-cawRSn4EmLoYDTq$!ptPXY(I-z_M+hfpz77H%@#L{%3Du)8DTB^Jk_!|51IJ zqvrdEGcKZfA7+EF%^DB!-?_xvS)2S``*VZuW7@nBjHTzulf}&ohjN_efHh< zH|}M}7nF+KjhVA~Qm`^}_*~ZWvQZ9d=L$GQb}yYV^}y7g1i9Y=*ZN)8t)Cs9qj!68 zu*g$o;X79A9h(lnX03bI@-eP4lZD@jP5aUDH~-o+dGy)TlPeY5ZJ5k^J=do)=a|8{&F)86mK93fo_J-V@{;+>eyAKj>C5N*#%W$AQ)HqoPO4G@bsE1^DNBztk<@%%goi3n)C8d@&ldYjoVU9 z?i|0xc-=zS&?!B{W5TtV9}DIxw;3F2=*kS;;i4;3Bs6V;^Fbe3mt(&oIInn5I-B}( zQ?gdkvKf~rKK8I&{kiII|MY2Yi()T+-@3GP&c7`?OpI7ptnyx5W5sjITl(zUwXeJv z|G5{Q6=bl|Z^InfB%|dMcPy7w5AiH~((y~^)#G!eZHXGuB2vfrSFJPI5Od3MWhO(+ zid9jU7wnL3nS4;j{NN^sn<^8HcI&5{5Ap0Oa?zaJT=W04>Gae0EMnH@+&nMjwyQPR zV&@so;Qvcb(OXKMvuDv>HWrm4!ObpFx2iLhI8rB>2}PfJzkyFK zw#R+B!1j$s_pTH)hyE!3Z69^GalgUEJ(}Cse3dO^ZtDL=eG_6In8eeaGHQzQU?C<8|QzJAPo!1KOkkgLoWsg!Vunmi| zSrE#;?wHoPW6WUj$8BKY9a8*TUmf`R%jyNYR5km$<6Pm78^btbdZ%{kXNPgd^oQ;V zmfP^^z}JYQ)lV5$^VO(mAKjO$dGg!u9i5VD7Arngx_zJWGrx4k{;oH&LK%$hL7MAg zMU@UtD|jjMu2FkBOZ0)IOb-&gri(-$QoXi#byj;4_YRh|J6El0Px9Wu(|CJTjmEZy z(?1d-!=5^CR9)1Q&cEvS{eRMJ`n#)Vz20Ddc7p7;cFAd-0iJR`qDk$#PcI#1|26eA zU-XHkO*;yv=m~F|;yvfjs#SAabe{;B6oiNRADULwIXOP;>EsQqF~{Cd@d>r#j6Sh+ z;ttkTr_>X@cfRRU6V?0u*z5F{dG8WKFV?(bk2pH-t=etf?L|*R&%XDyIrJ^m=9#Kt z%Q=&xZ68k@Xn*!*ljW}{qw{+Ax2eCrvu;~9Uv_)8G37|W0RydPtiQ8|9S1rY|gZTS(#qq zvl5NXlqwhu*_1cq`-M;LoaMEj%=7Uv%K|3lwPKwQoi{CW;{Jv#5 zDT^-5_nO9U$vx}zj!5$btJRc3U%pKG{AR{po1WvXQ-tfv|4fyQxtnwIX8!kfi}HV%7DfA8)WUHWOx`@Xv$-n{>?YtrUJyWO*I?UXvd*yobn{u?>-UY}?$1|GW52yxtM_T;;#fC%o92+%@CiqF+mF7x%?kQsLAFeU; z+F+`}=gD5lpKqS6{IF;Slg0hjC%654`Q`WY`?(R;FX!#6DE#>~BClfAEeC9Npss83Jr`aO?8LQRSP1xPT{LJ0Y=$7w^8wtwOEku>2mn+Tu@&37RZtat# zt@9_F_e!^J@SeTtt-NE;?|z3F}VeUj3A7w)ON zmCkoR&ffev?R|y0nccjylHX51FMh~8eRuD!H%#WW=rPcgN%e{3bV`i#xgS5eB_eU4Al+&2> zrc|rQ>F&x=lRMOED|CK^t+LEJqh)7XB-3>CoT^o(6|b`R_w?32BfZx@LO<6ZJv3)B zs3{0-!9W{@;N}jf31nnsYzUGDvBB*m5F=V47A^*A9)iS>+Dev&3Iy9pwmpt~2NXD1 z-v8#Bw}3mCK==~G{1mah-gkUGBzUYM%{qj zxh{r0uWebyEe`y;{W(Tp$&NSo-L81M9$MsbDYdWnMn$J$_Tre2s@+%2>%yZPL<{x# zl(t@dFZ!xN_DAQ<;`$GJeR5}HIhv~oJ^xb8t=eSw%5_7@z0Ap*FJ9J=Ewy@Iy!RX9 zc|!r-j`yzKMrs>)pT|yJh)|+1Z&iz@s({q>D<^R+_Wzd128d#$kfhgDzpUuwT`_SVcE zPnYaXoxXK>zc2gznp*qhRQ=|E8_xfVyk9SWJ~R(YfMIF>8W|cB4lvsQ>rKpt9ItD+ zWDh2sd2TcNeQVmkxf}C zApYjx`a`DM^j~nDlbW(9cKZ9Yjx~~H5yo?Fe%aTiI#J!{ZN)bK>#yyP+Qv02&+To@ zk~<>txjRZ>ilK)5=NVHEY<{5q(Buop4m06H`>sXx?d9E=dCuI(PxaUf?HN~A-haW) zx%bAxnC{t@IPd@Djy)y|XD+q5TGHb3a`CN-t@fgumfwA(wQ%c3*N?@w?z^^gpObb>ky(tz zX~=yiXvf9egmAnWZ#D=s2-sl4B(!YWG_GfE)3`L-VjS!`rk-XMKFzqAbY zz#`}-#)j!2o(7YMOgfLzJaw{~ra0@(hZE*!T@!bmW34NmE@8g3 zUi9MpX1QJQCq9b*$Koc?AOf13j0}lL$ff~1dzlS+&V;f`8ytwgQ+3TiukioN!WE?o z6C0j|vI+;s>ltf9WQWs&3eOFL$rT=MQo-Lj>RmzuBj zby)H4p@-{!2iH3ss}^~EIJjr}^mAgfnmbms3yb%21v!*n>ftxDU~W`-FefL#F=_?h ztCz*=_rGkLb)Qvdo5}LCSKBpXrM7PVdHcEdg2qalY$RdxJQzGK5 z&r--)k;l2(@1b&o`2!pK$)1ky?z?^RzZAj$#Pzbzt{hj-D=VYr7T3PKyQlhM=Dy7u zMW^0)7x{R8xEc|>_}R;#T7Ih~53i;RnkHAo_tskZ?g(Y>vc9tVpyjJMY!M^Um@A7~>&$<1Kf}9CPsi~(P0Cg-M>X^n2v^4^;^%Cx${4s+hC?l#(_s{ z8=_sGO4!Ziu>Q1J@4WIt)%xl4ZFP7y#wZDFd%ACtNWEI%inUTcS5G!rz6zLl^YY}q zGY(Ff%#tR#Q1l4{wO-meP_3>1+$hh>(wif z*OCg;mS-v-Rt_$!Gsrr5tDEOC&*sBSwr1yaxo!7d?44~T@`3GL{+|8e4_oX0Gd@WU zdw?apu+-khCPY+Fra{(us)jtTpNs5qV0fK=^LgdLr0zF=Z>ySms5)))=qmet$jN%2 z&W6nwU;fy({GIFi-umWMGWPzWf%+opuUI}=yt1kuWLqh z{*n;Zijt}6eLHJa(%h%ZMETA4O0m86c#p>33o)kV`(rbfY1})rCu6(htM~eUZ0hIR z|7Wzyub+n*KgL+v4`xI}kMU-MAcFt{)+FQNV%B11?^Y3KyD+2lnA2uP&0vmH zJGRV2hgl&vC5dmjqS3MfsYTY!+Pn-Rt3*7pI0H*t!IFqvW!ifXR5LOE_jxGD)hP6B zNfA_Oase6!<`1qW~fr*EE9}4Z5VWQT?9jEpAfRgcR+hvX>lUmM~S651{V(mh^RY#&p>))2dvjW;MBRb_22T$stf<)SKl}imoO7K87#mR){?C0h`M`kI1)i6JK#DBGQe14?K_qdbF-?^LOq`@-`s9iB*ER79A z3&ItbvDwA|gNDN*&g)DVEj3raUg_?W z3wd9Ei#>mDqazoSXVTnBnp>YN)my5n?Ne9cGWAE{PxGMQ8vm0fAMZ@c?!GiM&tUOy+zxN`ZEdzV(e6Y8_SCbVkq^0~5mzuD^i{W#(KB~7pXaDfXCl)RMp>YUy6 zOXz7$o1bv6v$l=j>7##xHlOs}9`QW8W9^P_ws+*7T+Nr{?7GWZRAD;3<>Ae;wRh)q zH=S5}#nRyB^3_*GmQR-!TgVfan^^Z|(Ro+K-RCzfjek;*{e(x(f%%s_2=}Wat`+!*HiO;Di|MKbky`zOyZSmUu`;^)%jzs6yu@8dB6An_gk)tg?GQ!K`+p~-E3>P~(CVg2t9`^lY623X<=OWVcBkcfD)4e-um zHRQSVTtr6k-!Z%Mt5GAG^1h$GAP;dC7&{f=fem__QT`Iv-tl z#UAhd*hn^2-sAFZN7qGi+gBKz4PeRUQn`7yZ&8~v+sOlp;lKJ%E~%O*`s}T7^UI^R z3ulP$UDcBk*u9 z*T3s;THvdW88(Kmz3u#=$B*^rKT$Ar{$LxDVSO)=qKmrZiYSQr~9m{}Nuu?5&w=3K@g$Cw)_n1kn- z%@r&yO%*JR%@r(64HV3cjUlwTfr7aaNUxECg&|0-iGrb_fw6+QxuJr&k%@x2ktr9* zHy~GoXpk#R4Ga`O*v#Bi!Q9YH!PL?iOq-coD3};S#LNwi6hN(d5H>S21+lpd4NQy_ zKoVva=3osV%_f${3Z|B(3MS?j5IIvL1xq6n1v3*<1v6741yc)%9*_e-dM%7CAn^tY zCIbU-kb*2VGc|_<3&>?K3{FBI|CtyXDHs}>nS+y!xsfTDHZw5b0>_AjfdWW77#mxH z{RR$UkTDPj$bldu3@lB+TEVU|GEsn~1W*b9N1QPv@<6Ty=>}nl31%QKnSoOSC?!Ay zp#cdBR1;%R;8}ny=Q0PGVr&NX8^{EZoGH`*Ljz+|1q1;~4d8?TN(^Y&(7?i!3zRxw zDFa5sQU;6$*$+z@pwxkajV-__#lp~t3zR}&7@RnaL2+iO07?khu%Us81vqtpQU(&{ z0;LLM3`r113LqLCgB%AE14TYE1|<%2Lo+TAjTr6+72BXf0#sZ(Dj0%l^B_Yz~jCy)ysAW{;+8H-nmHR z+rQinUGY1%FFQZyB&5u3SKjhiY}2xWV4onr73DlxOU-4iU#On_YjUzmPwESQ;aAaZ zvzaFas!VKmnSOd|`qTItJDz-*B+Lec3CN#lPBbzjqU~=Rv~wb}A0A z-jh}@2sym3Xp4~bv{$jsZSfD~j#E zd|Gys&Jvj`w|CYtKa73b6yd+t=8Zt+y|5~8(KFA^u=q&kTCZu?8ku}3g!6;gS;Z?+ z9cRR?)^_}Sa_8bflh8^jrNcq3&2McNo5sA567oL$)AJb1v;7}_tvQif|DT~CBYhW^ z00ku$^Z+#mHL3_K12PTt%wsU**!^7epTp7(>hEW+dbRNH|E+C-`pnEXZm7IJ=%!+K zWWqa*@S0PvOiq8Wj(l?Tyw}4gdpo$6XdU9pWItaNX3R5BdM49!ci;6-^R-U7U%RWb-S6fVue*!V z-+0E9#rfxE#!Hlnn6pmglegZ({!{6hj-J)Tyl=BL+ovv3w9svk?_3yio^QRNT3wn- z#i@CI?-!fw`Q~`}>)vWd_dK-}t?dDBTJzSJ9P3&5t2aY!<@^7Q^SvzE{Ds4pY4U+12WM6Aa5FAXoS3Y4KXh8?@~P`)_0GG#H`(!O=5$lu z+*^&G*7n;UGQGF>gVY;2C-#jRC3&e)+%qF=vgcTzS$yVso!{+pTZy&HuBhzU>xTxPP=Q6hZRllyk+b~YPRm745i)TS42AXFri0IAuf+m9$ctDduHcSux+XYTh zy1BoXS8(CR`6cRe6WY1*3L|*7Sj(*aaV2<0?F%`fNdd)oT_TKCZ?!Cm|8?5n@QjEP z+t!l6j(|r^{~TbLw&rN1f0rZw+yCoT zWkuc|6j1h#dGzjxi&)&Yfa^}}j}N=Wan5dh-+XP(52=lh>~Fa(nQ9*rtsra|s8SMW zF7@5Rd=J;fx+P8FtG-8k7A)638*Np#asT&&Tb@nSTJ2=>mkErV|7eCr=xmWC5 za!R4m&Rmh*C9^NCb;3l_30DDg$obqSzC)G33E%gZ;_vBxA9NS>$5** zXRcj3q0mvu3fbpulUa})jviJOZb868no~;wzRiq1EgCYbb#00a4;E*E zDn2x45i`nRyvZcgAYj852BEfT&rqBNp5a5CP}|?b8-{7MDYkJ-B9gkXscAUG>SfE^ zP^^Z9h}ba)HV?@;i402)&MHo*D%xSR;}XMZQGqj990982(fm)`eA;QysHO9JpMx9^ zgPDGZgz}&KcfGBq>L7pgi5;9`Jo1ZgwN7QuZ`!C{nh+buEb1gCneybewRpEzn%MqX zYAb~=sTAIGS!nI+HJ5+O<=YQhpEG}bDf+}gbFhRV+7>B>{oXI)yI^z}JtB=zJx2`tIR z6kEf9lX{B19 zKP!5d=5}@8S$UUF9k_CO!e-?T2`@J9fXiDm4lR1S^2gjyJ^QV?#p{mqX_wr2csW?` ztL@=*yAJQY$G)w&In!mA{)q{fXL)n_n%unj{)EUm8}lU-&M7zs#WZE7PWbzB`eJ4w z1HKzqH@D;?i^Z#F&s?ERlw#o;NfnW}bF}x6ok)9_R0KoZK1yH+m z|JVuy7hfpunD|FIz9>8K!pvvl;<~FP*QWYhe3Q5;N0;YSuVeYCuUD$peKrt$|C#x| zqg@IXFJq~uiCPt7JHfk%MG@-d`iC}iz0NLr{_pmQcQbP)tUB;XxKKto{c{J$5}7}1 z64L^N_q4YM^MdiCsW#5{1Xm9(eUlxx;ELUq~Pc##p>6$Hl&`s{m zhVzS8XPl1Ix?|qs8!J5Z@a&`SBJZAj)b)OS%;T#%2hV&zynVlneE8iLTE}mkYPlIS z`Pn7Quj{8>{8g}}yS3_}d$pQJLFsCR(52VT-3qSy8+=FQ*I(u@2g7Bt1Q?d8-q4hY znR3&9Z@$9@94_DOj>aqyig!LYdDSbWz5ll*s815~|8_>giM54iPlVxF!J-Fe7w65Y zYcAH?CFXh6cSQ%{sfhV{)x-ZNW-vhMS(xpwOF&HqyO zKmKG+5H`ay$6<=4*yJT*P zxn@n<Uix=$fol42m#h*WRa##N<{=J4f_xCMfzq0CAo==54o=;nIAoa26KQ>rxc=IB?;F zi{f-{6^^8*4UudPQj09u*QvE@GoNA0Y+b>=QQ=juZtp@ZL7k2q(3+#=C&brdI>-!L z6N#wltu3GlOY`sxrY0t#21Z5!1|~*<9)?ehOib8~i+3h8Std;DWSo%5p>j~FXh%X+ z@c}D^O#y2mE3XO{38iCk3zn9WkueeNHB*1fNi2#yrz>0S3_>MtJM3S+e4E*gpY^*Y z&Geh_LGk2eVcD4#GnJJ$q#eHV{%>3Og;#xlAg45?n>(>>V!^dFkMrjB ziT^K~dOm{vz1feh$T)!;hkUMui>2$`W=+nVbHwswUQev>#@^Dk%Y!4_4HojWlx#aZ zMWHxk$(NITM$cYYOjMuO{r=l+oifWhzSrDly^5mGe`r-TnKdtZQ~91rs>tzw?A;%7 zuDf5TDoAsRT?xvd&P#ZI#v^@J7AzSLSKFe_lq=EVzw0t;5!Eq}3Q ze*M$<7+oykg{9qQWI{xX-!{Oy&qa~vwPmaR2bE9C1#)hyb<^J8U&lO4Gr*^Ao3Y9T z!A5r;!zt?@tFB$Ju5o|2)5p9=ubU4ktac7kI9J8rQSxS)*+gM0**V7ciRQ_D2V4!= z=DaLfX3MT-`TG!8OR;q7n=lXc&Q(f%uZ;UN=V?{XuMpFDd%`SQeUj}Hg@-{wohO&h zpK{7R_tX?FtF((OIzl@iqzL<3oZjsj)FU~?=#|{EGNsxzHYXNay-4vrw?=v9&CC9` zXa0~W*dBi?#c%5I61g)!=4?6FY$>|y`{PbCcHiDxFW!6?&eYNm+ASM3x2{xZ>RpNF z5<=6{>m~1}{Qt@RuRTf$O9*0VWEvAQL2eu5op;m$)X0okaOuPB^OZrNclZ3x&vp)4 zskNum@R&jEsR=uuddCVCww~M`w=UXdS&d$F)Xd6`-QB&acH8buf2X)DwnT5Eg{=E0 zZlk$cE-TKiy14L%)3jFh34R|Iav!R-{P$SD;o$WRygMG+8aZxId8X-fEH6-UmXM=h z;`523E0zSzk34ig@ZIb0Myy%+JxjThWxjB3VRg#&IKuSGfx+d?l#sO(`qEW|_lw?5 zPm)P^esL4yI>`{1Y1MCzcdR_OqGylG5j%h1UmDYVbiNnbLEmri4{%rOBr$ zpDu)MzQ7&PSt<0*@{h-l=ht!++pO}ged>HFZ+z59=v;!gWx~SgkKG^Kjw}^iC%bdU zxspctsn-g8w!E%8@zwud$?d4KrF)7v7CAiib^NfKZ=Ku5BS(zv=Jskd!*A8Wy_kAwdHs9X(#iCp+af+p3Ow5YX%?8^*C7Ovruz`trkb${bkb$8o zs6=DpY@D$(#^9ig!D)#VI_x^#363eo2Xd5ygICB2a4$M8uukY;(}^_Dips!!l44j~ zgr(hT2$}~YROJ55u_}6D$jH%&@d8 zi7I$}13kN#4S04x7yad+7&m#(a_5=O&;MUP#n)=d!nuCZdb@9&#sl3JL`UWu%hQ9ukn}t>7WmPPk#R)yPp5- zpO`sVk_47Ut0fVm-mbj|5esv;ZuzTCZvXbbSHP$y;W^*C7FOQ*#?U% zu{7_DO^B$FU4y*yTnssG?+-3$^msUZ!sjikKrQR&**;g#KHyBqsW=_HEalXOovO+z z*QfZ`DmkxN_Q#}1MQPa&Pfpgdsr|>Zz1Q@${6f&n^l) zd+LPc34w+4M0ahfzQT9&Lsg8&uJ=>UD$1PxAJC|idbfy6)iL@)`5VFe{Ictsw3`w) zWa~A*O*Ojm|5v+M%=_B!NyqgX#MV6DcaV=^#d6^-tS2OIy))CyUeZN zMLCN@U;dO~Fy=ETI?$mSExYgHJ%(wuu`gVEyZ&lB1V%diIdQe@;61~W9tC30*!uPs z*c_c%;B9e?$)RiSBZ;aHsyCkLhp+3H^?|c=SBn0gL-TGud+s!uqkJs`UtG%JwW}Ae znD+LiP1Cn@*CMx`E8P}Wij$%)SvEbov_LNEag6!KN4d%ycZT18C;xDce$2Pm;@)ps zj$nx$ENw|cVn(f9Cmb(iHRN%AzlurKpn>b_n&Y4OU;kU*_T{o;m4p*dlK-I%G2w-Y zC5QXB>l*XcyGy=1_RA=F&(#P8^%u-M6FXO5T=J{zsN||IPp&4dTf8Vn-%3g7%%sPw z9h7UVHXWb5ZJY7mk~sM{itmO0YiT{tOJvkr_0HjEko?Ch{Wk+&&05&6rJXI8bt}9j zR?zFVo!)XIzq@;xqaN;^y4v>EdwaPut52AoHOJCUH63^6b=3p6k@ zVM-Ei6>-*EzyaGS;IOJ_$AVME39E|0V|c6*d`4Jo$I>!2CuU{7(e?n)%6vwlHZe7> zXK8GpRecWoR(9@m(i8XKIdY0|Gf%@7)j3Ta=?ya!as`emh`DT>*(vEFJBM-N=?7lS z463)qPGWHnmX;qe8#-*ySPC^5aIjvO7yL)L<%H7rLYeN0|6v9f+2mGNEl{**-gR*8 z!-P}2N`&7BKfZcwju7XH3NPPjySS^aY}t2ZU&e#oub+8e?X*98UVJY9Zdu8@Se%BX zNkz;czA?6-nkEpYC+?%eeunAdv;r=rd7zb!JJWfwEJQNLQtKNMvqsf*g0&Dcc!Bvs zHTXOu)n(KAiZ$c9@BfU>^ZlAk5&4Keyu;s1Qw)XN| zR^PA$F_!j(u@Mo;(KK*(?ok7tH>RxO1`2hmXCEtgOx#{S+xpf@kIQ@HL*tB=#^{xv z)QU{}wA8t@Ygx^_v}J$zzD{^NL++;HuHb9aswKpxbyj-GZk(m_y@12c; zj=wmS7ML?(>eq`e`A*$gAez|LM4 zL!Q&Itx^#Os@|?l7wh(3_Wz5=&u>Z>#ax55OlMBg)GwW)8X0`gs%F{|zRI?zc`0V9 zTCVz&H>iHU!1b}V^O?x^yH}@eb~c%wVOhs*thGv7_OiFxtO70JT?_6UHvYw%Tf!XH zF1ImS;lh%H>)RBHj`K_K9GofKt;FNy%l3a=QUCn7ZfU#2ADu-ePU3rJ`ak0Dr()jq z!TstJ^YZsSikq{Nk6BP=(yTqou@M{fgak899evLqh|+D?y*=)q^cLk^JPzMFj}*=d z>RY%!yI|>g{io`|H?X8uEKMC^$|_g*B7_8%I<;?hfm06M+s`YYnzuv!u5FAg=S7xio3o)I+H zPxq$kKXIObMg3T@gBFWx+^YbO|{@R5DvaLwQ%*s#wo&V>&qCfrL; zoT_p%J=ANtaOU2_1qsZjH@7TVx;AJJe=6e(rr29nk4?Oq`WOz~*Xa5Bi*41PICCuF zhNa~}Oc%@8&=_1D1{)Y|4KOe>4m2>|6apSI+RkEJd{AdX3ZHn0=nTe5ryEzFdU)zV zQ|39=%?ez-Iy@621g094D%fT+s5MGp*?wh#rFmdzM#OB0>jX<7Wu5VGem21(8Dd6M-T148Flr9OK(B%pU#(TA%FZ|#Z39&+q`~~;V+4h zJ=V=Vo$=79&%8cO=2&(oT3~63 z7#kDO5Yi3uE;_2fbGw);c7gE^_9r$kOLl=)BTt&_>c*Gw#qeocr&mp+!uG}2j%@Ee zqU&U;R-|?&q2_CpWVn)4y5Hul?;x zORD1Iqms80MUN+nK8e^+s<^7UB6!K+l`1zLcmKCLTCB^WTjO@2A^yqy^?eV1P7=%f zJ|p+?x~dI6)~l9mkje2cj#G<|O5bTEcb@&^w{vu-AC)Xe~dcCHN}`Y!U9W!jF`zd zV`H;mgMbYNOhU_+wV}*3gD1Jq=zzDePM9X6=gggKkeigVBq1${EhSRHP+=En%6Y~t zVP-6@!O{jZHXu*k+OQy6fG~3vSotVhI$sjhLw~s&wW2mlswm3cThw zhI4Eek>1nn$t|&AJ-;BH@r+yJk-~_vRIHIiI41r!J9Q`&ja(!M)Q4!AT1rZeYFg zEGf0=-P6aW>koCj4V)Z%xZvHiv*jEYGyX}7oqCWgI;p{Z`9>k$Op5~@2j<0D9F6-d zY{_?^BYEyD9iHS2wdGZPod+(=+q^XK*m1e2ub=rtUe>f@i3n`1QesEfp_@l!pD_nv zF&Rr6(9oEOJz1{(-h3>I94_DIEb2ROT>g2X(aYB2|8E=S<_Ra;wuUtBY+7imm(4s^ zR9;bX+sExIr}eo;+J#gFh4G(!(7xu_X~jDWbUrkBYPBf_bkC5QsGFL%z2UYQ@1{*t zgO~Pg=w58l|7}Z-s9gM_XrWR!f42e#Wz@*QN|`j zC0NTm7DJxbf4OQE8uD(xzp3Cc@y~wVl{ZD03rm8gSVx7Foo;Up+4`pB=a#9rgv_6m zr?wv9Jh@uvZBU!XialY9SKrSxTEE>io#jN((nC*{JSjZcWPf<+a_-|lw{2Yi^n`4` zT!g6dLYD(ZM~}Rans)BphP3FqX_*K5D_Z7gU(-Dv!5Soe*Wzv0_l1?Zo7g(`&Yaf8 zB^%Z$R_FL}fqQT5H;Ir4g^O=xJPUd{112_i93zYw2tYIkp;+C}Q9mhu1 zNi!^u?kf3m|Hp-m3;x~cG2d^c^)fmtGWn97^xG-Nbc|kBMSLmweC>EMx5@iOf1Y3E z5i&{bopg+S?Jdi;MKc4#Dz-aM<~?h8T=6sCyu*^5UcYQa_V=p_%T76QpO1H8eZ1V_ zo;hasn4T@N&lP3od$jp*+|EW(FY)Y2-x*i>yl8y+fBna;`>tP?mwnrC3`^u-X}}o~ zvw}xA(6flykz@CH?Jq59%kL*o*KU9E|2oGhb~fH_hmxpA3wq`&=!EZiI&pQz;rqsh zP4`*b>$I=uMm%u{QSeL-FkM*ovFeN5jlvn6XC`0Uz>~Iy|4)zRofk(R8cL>bzn*$U z;7#m$+nJ6h*TfuA6&Clony8%4XFt8p>I9+pSe%5V5ocjZc)6Hq&u-oZ10I)Z|A&H) z_TOYzOz!8?qsoh1+@%wWZz;s&=O0VjwlEE?i6e=;CXG^>VBZ#V-NrF zz?Q$!Z(2;I`OOzR#h(3yvFTy2zT%HX`jRbMx%stzdz9R8;Y}A+5k9W3fU;n>+^`6r|S@)Lf{lnrbEKM0Bq845STJkX~^1S&j`p-e7N58^=G$lUFD{4YsSAsW)fWo$F_^OrB5MqFdasW7{76wUZ8XrgbzQ z%$%`0-eR>E?ij}tg+z^i(S|DJpQ}MTN|fuQ+(d3kfb50<9K$ZtFTz^ zdG#fx`{TN$?@fPu_eAj0W78xW*%-qtWUjxs{$Z-EzuLtTQKKuXO;1a33cD)V{OntH zRrqz3$t5G9E5`)aWHIlt{%hdJz#v?Bnki+z*Objqj^uCrvY9RX*FVPj?jh4K!^Z?U zSt2%4DIjfmbMg%gbO}S+@&?LPMbI5?1_~&<-5^`3ToAk6j6wMlw6hCkx0?cVw;P7h z5dDa)SVo2lx+Z4E3WnyECJKf|#>SviBtbj9@NabE(g$tDgKTs&0i|nlH@boPB88>- zC8;S%ISM8wpuKImxrqw;8I=VYsd)+}77F@RsYUsaoedbf;kfkEGRsmG^ouhq6!eQz z%Tn_c^dURN^z$-7+vgx#CCm&J^ivB<6LSx2v zvoKZA4^^-QSViU#y^CqF@Lb>M2vuFIUj7P|&YL-GXNfD)W)I;DKWnwh|PPa**O4 zX$u}kM1v3TOfSZ=`wnZA(_;6Xv567V?mMvmK;9VyyYCDw6if^Z6->+wz}u1vaO%;qS4Zs+*a|*P1$k@cAR6o$V{t3bGrd-c-TZ)CeqQW^AHhWNrb*rskGl+S1%a0VHl_Xv7852f`+XhTtuX z#+D#`CJLsOpiO`lU>d|WH3A(@YX%MkP@sV92XC1K1t4haouLI6$WBv33rL`V90tRN zp!5T_#Q?Nl5fntAU7H}UfoO)R4l}^g%u)e{K?wnp20-Zm6@$at(gK_&KHaj7_+}w4o6=kYH?( z{b05+G~GaHP`UucDu{;fx&zlCpuz#N>khmVEl5G%Nddg!3%vi%1k^bO@4qtvb!Nf) z?@T})!5{^FcLfu31p|d31$_@tm0_R|1S-x=pjD%f0(fPA5NJcRDY&u%?Ta=AHC(_u z@xW`x!8`Fxp>_l*n1b!#(hpTIh1!v*0A4y81S<4h223|A~1l{y)2KF77e!7Ag)aDEY zGpO&NH6?h)GYC`@nS-rW(9c#dhgzEh+7=ErI!^&S^A*GesUOY3z64cO;F*aaNHuE? zwhdH!nL}-Z*U;urON$i1L#RPq`o#+1F~%TR{RN3zKb+C)e6)d23Rw#g*_(7nW(-P`OaNEEV;!VUxMv%3z#N=kA z06ujfNI~BXR1ScX2|_8zO)#a7h6Y3;$ zWR#Q?6kF-*=fb7*@{7_nx%2~y@>5EaQ;QUkbrqzfDI_H(XQ$?+Kuyr((gz!#nV;tZ zI&xjZ#mdOQ$kf2n(9*!t$jsP4+rUuWz(8FSDO@UwQq#B$6f8}+4B$Y)%+%D_R3S|P zE(SVx0VJ)E2Ng3iwgBZ}G%+&+BL$F<0g^hVA^YJYH5b<4`Xvs`HrU7!psCi zouws47?^;jX+T<$+-G8p8CE8s69LfGS%9`lg9H(7HZ=rogGLiG#SBwZGYbs2m|B2p z6EwY+mKgppGX&KHsOrp&K|9OQ#7qp#(Zj&Z0<`NJMV*nMC8)+h6Eg?ZXJ}%Epu!kc z%-FyXJ>45vVo9f#MxeduD0&SnO)=B6rG>dMnwt#`Ks_LkRzw^b8W@3UJ&+(g>6VTZJXkw6+ks2