Add true-removal redaction pipeline with verification and font-type coverage
This commit is contained in:
+97
-87
@@ -5,8 +5,11 @@ import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
@@ -22,6 +25,7 @@ import lombok.extern.slf4j.Slf4j;
|
||||
import stirling.software.SPDF.model.PDFText;
|
||||
import stirling.software.SPDF.model.api.security.ManualRedactPdfRequest;
|
||||
import stirling.software.SPDF.pdf.parser.PageImageLocator;
|
||||
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
|
||||
import stirling.software.common.model.api.security.RedactionArea;
|
||||
import stirling.software.common.util.GeneralUtils;
|
||||
import stirling.software.common.util.PdfUtils;
|
||||
@@ -42,11 +46,15 @@ class ManualRedactionService {
|
||||
// Area and page redaction
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
void redactAreas(List<RedactionArea> redactionAreas, PDDocument document, PDPageTree allPages)
|
||||
AreaRedactionResult redactAreas(
|
||||
List<RedactionArea> redactionAreas, PDDocument document, PDPageTree allPages)
|
||||
throws IOException {
|
||||
|
||||
Set<String> capturedStrings = new LinkedHashSet<>();
|
||||
Map<Integer, List<PDRectangle>> rectsByPage = new HashMap<>();
|
||||
|
||||
if (redactionAreas == null || redactionAreas.isEmpty()) {
|
||||
return;
|
||||
return new AreaRedactionResult(rectsByPage);
|
||||
}
|
||||
|
||||
Map<Integer, List<RedactionArea>> redactionsByPage = new HashMap<>();
|
||||
@@ -74,52 +82,49 @@ class ManualRedactionService {
|
||||
continue;
|
||||
}
|
||||
|
||||
PDPage page = allPages.get(pageNumber - 1);
|
||||
int pageIndex = pageNumber - 1;
|
||||
PDPage page = allPages.get(pageIndex);
|
||||
float pageHeight = page.getBBox().getHeight();
|
||||
|
||||
try (PDPageContentStream contentStream =
|
||||
new PDPageContentStream(
|
||||
document, page, PDPageContentStream.AppendMode.APPEND, true, true)) {
|
||||
|
||||
contentStream.saveGraphicsState();
|
||||
for (RedactionArea redactionArea : areasForPage) {
|
||||
Color redactColor = decodeOrDefault(redactionArea.getColor());
|
||||
|
||||
contentStream.setNonStrokingColor(redactColor);
|
||||
|
||||
float x = redactionArea.getX().floatValue();
|
||||
float y = redactionArea.getY().floatValue();
|
||||
float width = redactionArea.getWidth().floatValue();
|
||||
float height = redactionArea.getHeight().floatValue();
|
||||
|
||||
float pdfY = page.getBBox().getHeight() - y - height;
|
||||
|
||||
contentStream.addRect(x, pdfY, width, height);
|
||||
contentStream.fill();
|
||||
}
|
||||
contentStream.restoreGraphicsState();
|
||||
List<PDRectangle> rects = new ArrayList<>();
|
||||
Color overlayColor = Color.BLACK;
|
||||
for (RedactionArea area : areasForPage) {
|
||||
float x = area.getX().floatValue();
|
||||
float y = area.getY().floatValue();
|
||||
float width = area.getWidth().floatValue();
|
||||
float height = area.getHeight().floatValue();
|
||||
// Request coords are top-left origin; convert to PDF user space (bottom-left).
|
||||
float pdfY = pageHeight - y - height;
|
||||
rects.add(new PDRectangle(x, pdfY, width, height));
|
||||
overlayColor = decodeOrDefault(area.getColor());
|
||||
}
|
||||
|
||||
// Physically drop intersecting glyphs and draw the overlay rectangle over the area.
|
||||
Map<Integer, List<PDRectangle>> singlePage = new HashMap<>();
|
||||
singlePage.put(pageIndex, rects);
|
||||
RedactionPipeline.RedactionResult result =
|
||||
RedactionPipeline.redactAreas(document, singlePage, overlayColor);
|
||||
capturedStrings.addAll(result.getCapturedStrings());
|
||||
rectsByPage.put(pageIndex, rects);
|
||||
}
|
||||
|
||||
log.debug(
|
||||
"Manual area redaction captured {} text run(s) across {} page(s)",
|
||||
capturedStrings.size(),
|
||||
rectsByPage.size());
|
||||
return new AreaRedactionResult(rectsByPage);
|
||||
}
|
||||
|
||||
void redactPages(ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages)
|
||||
List<Integer> redactPages(
|
||||
ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages)
|
||||
throws IOException {
|
||||
|
||||
Color redactColor = decodeOrDefault(request.getPageRedactionColor());
|
||||
List<Integer> pageNumbers = getPageNumbers(request, allPages.getCount());
|
||||
List<Integer> pageIndexes = getPageNumbers(request, allPages.getCount());
|
||||
|
||||
for (Integer pageNumber : pageNumbers) {
|
||||
PDPage page = allPages.get(pageNumber);
|
||||
|
||||
try (PDPageContentStream contentStream =
|
||||
new PDPageContentStream(
|
||||
document, page, PDPageContentStream.AppendMode.APPEND, true, true)) {
|
||||
contentStream.setNonStrokingColor(redactColor);
|
||||
|
||||
PDRectangle box = page.getBBox();
|
||||
contentStream.addRect(0, 0, box.getWidth(), box.getHeight());
|
||||
contentStream.fill();
|
||||
}
|
||||
}
|
||||
// Whole-page wipe: drop the content stream, resources and annotations, then fill.
|
||||
RedactionPipeline.redactWholePages(document, pageIndexes, redactColor);
|
||||
return new ArrayList<>(pageIndexes);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
@@ -298,7 +303,9 @@ class ManualRedactionService {
|
||||
String colorString,
|
||||
float customPadding,
|
||||
Boolean convertToImage,
|
||||
boolean isTextRemovalMode)
|
||||
boolean isTextRemovalMode,
|
||||
Set<String> literalTargets,
|
||||
List<Pattern> patterns)
|
||||
throws IOException {
|
||||
|
||||
List<PDFText> allFoundTexts = new ArrayList<>();
|
||||
@@ -309,74 +316,77 @@ class ManualRedactionService {
|
||||
if (!allFoundTexts.isEmpty()) {
|
||||
Color redactColor = decodeOrDefault(colorString);
|
||||
redactFoundText(document, allFoundTexts, customPadding, redactColor, isTextRemovalMode);
|
||||
cleanDocumentMetadata(document);
|
||||
}
|
||||
|
||||
byte[] outputBytes;
|
||||
if (Boolean.TRUE.equals(convertToImage)) {
|
||||
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
|
||||
cleanDocumentMetadata(convertedPdf);
|
||||
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
try {
|
||||
convertedPdf.save(tempOut.getFile());
|
||||
} catch (IOException e) {
|
||||
tempOut.close();
|
||||
throw e;
|
||||
}
|
||||
|
||||
log.info(
|
||||
"Redaction finalized (image mode): {} pages ➜ {} KB",
|
||||
convertedPdf.getNumberOfPages(),
|
||||
tempOut.getFile().length() / 1024);
|
||||
|
||||
return tempOut;
|
||||
// Convert-to-image physically removes all text, so verification is a plain save.
|
||||
outputBytes =
|
||||
RedactionPipeline.finalize(
|
||||
convertedPdf, Collections.emptySet(), Collections.emptyList());
|
||||
}
|
||||
} else {
|
||||
// True-removal pass: physically strip matched glyph bytes from every content stream,
|
||||
// then scrub catalog carriers, verify, and rasterise affected pages on any leak.
|
||||
RedactionPipeline.redactLiteralTerms(document, literalTargets, patterns);
|
||||
outputBytes = RedactionPipeline.finalize(document, literalTargets, patterns);
|
||||
}
|
||||
|
||||
return writeBytes(outputBytes, document.getNumberOfPages());
|
||||
}
|
||||
|
||||
/**
|
||||
* Finalize a manual area/page redaction. The overlay rectangles are already drawn and the
|
||||
* intersecting glyphs already dropped. Verification is region-based: each redaction rectangle
|
||||
* is re-scanned and must be empty (whole-page wipes carry no rects and are guaranteed clean).
|
||||
*/
|
||||
TempFile finalizeManual(
|
||||
PDDocument document,
|
||||
Map<Integer, List<PDRectangle>> rectsByPage,
|
||||
Boolean convertToImage)
|
||||
throws IOException {
|
||||
|
||||
byte[] outputBytes;
|
||||
if (Boolean.TRUE.equals(convertToImage)) {
|
||||
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
|
||||
outputBytes =
|
||||
RedactionPipeline.finalize(
|
||||
convertedPdf, Collections.emptySet(), Collections.emptyList());
|
||||
}
|
||||
} else {
|
||||
outputBytes = RedactionPipeline.finalizeAreas(document, rectsByPage);
|
||||
}
|
||||
|
||||
return writeBytes(outputBytes, document.getNumberOfPages());
|
||||
}
|
||||
|
||||
private TempFile writeBytes(byte[] outputBytes, int pageCount) throws IOException {
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
try {
|
||||
document.save(tempOut.getFile());
|
||||
java.nio.file.Files.write(tempOut.getFile().toPath(), outputBytes);
|
||||
} catch (IOException e) {
|
||||
tempOut.close();
|
||||
throw e;
|
||||
}
|
||||
|
||||
log.info(
|
||||
"Redaction finalized: {} pages ➜ {} KB",
|
||||
document.getNumberOfPages(),
|
||||
tempOut.getFile().length() / 1024);
|
||||
|
||||
log.info("Redaction finalized: {} pages -> {} KB", pageCount, outputBytes.length / 1024);
|
||||
return tempOut;
|
||||
}
|
||||
|
||||
private void cleanDocumentMetadata(PDDocument document) {
|
||||
try {
|
||||
var documentInfo = document.getDocumentInformation();
|
||||
if (documentInfo != null) {
|
||||
documentInfo.setAuthor(null);
|
||||
documentInfo.setSubject(null);
|
||||
documentInfo.setKeywords(null);
|
||||
documentInfo.setModificationDate(java.util.Calendar.getInstance());
|
||||
log.debug("Cleaned document metadata for security");
|
||||
}
|
||||
|
||||
if (document.getDocumentCatalog() != null) {
|
||||
try {
|
||||
document.getDocumentCatalog().setMetadata(null);
|
||||
} catch (Exception e) {
|
||||
log.debug("Could not clear XMP metadata: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
} catch (Exception e) {
|
||||
log.warn("Failed to clean document metadata: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Utilities
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/** Redaction rectangles (per 0-based page index) applied by a manual area pass. */
|
||||
static final class AreaRedactionResult {
|
||||
final Map<Integer, List<PDRectangle>> rectsByPage;
|
||||
|
||||
AreaRedactionResult(Map<Integer, List<PDRectangle>> rectsByPage) {
|
||||
this.rectsByPage = rectsByPage;
|
||||
}
|
||||
}
|
||||
|
||||
static Color decodeOrDefault(String hex) {
|
||||
if (hex == null) {
|
||||
return Color.BLACK;
|
||||
|
||||
+47
-26
@@ -1,9 +1,15 @@
|
||||
package stirling.software.SPDF.controller.api.security;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPageTree;
|
||||
@@ -29,13 +35,13 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.ImageBox;
|
||||
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyle;
|
||||
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange;
|
||||
import stirling.software.SPDF.model.api.security.RedactPdfRequest;
|
||||
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
|
||||
import stirling.software.common.annotations.AutoJobPostMapping;
|
||||
import stirling.software.common.annotations.api.SecurityApi;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.model.api.security.RedactionArea;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.common.util.PdfUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
@@ -93,33 +99,25 @@ public class RedactController {
|
||||
throws IOException {
|
||||
|
||||
MultipartFile file = request.getFileInput();
|
||||
String filename =
|
||||
removeFileExtension(
|
||||
Objects.requireNonNull(
|
||||
Filenames.toSimpleFileName(file.getOriginalFilename())))
|
||||
+ "_redacted.pdf";
|
||||
|
||||
try (PDDocument document = pdfDocumentFactory.load(file)) {
|
||||
PDPageTree allPages = document.getDocumentCatalog().getPages();
|
||||
|
||||
// Whole-page wipes drop content outright (guaranteed clean); area redactions drop
|
||||
// intersecting glyphs, draw an overlay, and are verified per-rectangle at finalize.
|
||||
manualRedactionService.redactPages(request, document, allPages);
|
||||
manualRedactionService.redactAreas(request.getRedactions(), document, allPages);
|
||||
ManualRedactionService.AreaRedactionResult areaResult =
|
||||
manualRedactionService.redactAreas(request.getRedactions(), document, allPages);
|
||||
|
||||
if (Boolean.TRUE.equals(request.getConvertPDFToImage())) {
|
||||
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
|
||||
return WebResponseUtils.pdfDocToWebResponse(
|
||||
convertedPdf,
|
||||
removeFileExtension(
|
||||
Objects.requireNonNull(
|
||||
Filenames.toSimpleFileName(
|
||||
file.getOriginalFilename())))
|
||||
+ "_redacted.pdf",
|
||||
tempFileManager);
|
||||
}
|
||||
}
|
||||
|
||||
return WebResponseUtils.pdfDocToWebResponse(
|
||||
document,
|
||||
removeFileExtension(
|
||||
Objects.requireNonNull(
|
||||
Filenames.toSimpleFileName(file.getOriginalFilename())))
|
||||
+ "_redacted.pdf",
|
||||
tempFileManager);
|
||||
TempFile out =
|
||||
manualRedactionService.finalizeManual(
|
||||
document, areaResult.rectsByPage, request.getConvertPDFToImage());
|
||||
return WebResponseUtils.pdfFileToWebResponse(out, filename);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -184,9 +182,27 @@ public class RedactController {
|
||||
|
||||
if (allFoundTextsByPage.isEmpty()) {
|
||||
log.info("No text found matching redaction patterns");
|
||||
return WebResponseUtils.pdfDocToWebResponse(document, filename, tempFileManager);
|
||||
// Still finalize so metadata is scrubbed and the document is rewritten
|
||||
// consistently.
|
||||
TempFile finalized =
|
||||
manualRedactionService.finalizeManual(
|
||||
document, Collections.emptyMap(), request.getConvertPDFToImage());
|
||||
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
|
||||
}
|
||||
|
||||
Set<String> literalTargets =
|
||||
Arrays.stream(listOfText)
|
||||
.map(String::trim)
|
||||
.filter(s -> !s.isEmpty())
|
||||
.collect(Collectors.toCollection(LinkedHashSet::new));
|
||||
List<Pattern> compiledPatterns =
|
||||
RedactionPipeline.buildPatterns(listOfText, useRegex, wholeWordSearchBool);
|
||||
// Bare literal targets match substrings, so they are only safe when the search is a
|
||||
// plain literal. In regex or whole-word mode the boundary/regex semantics live entirely
|
||||
// in the compiled patterns, so pass no literal targets to avoid stripping substrings.
|
||||
Set<String> verificationTargets =
|
||||
(useRegex || wholeWordSearchBool) ? Collections.emptySet() : literalTargets;
|
||||
|
||||
boolean fallbackToBoxOnlyMode;
|
||||
try {
|
||||
fallbackToBoxOnlyMode =
|
||||
@@ -205,7 +221,8 @@ public class RedactController {
|
||||
|
||||
if (fallbackToBoxOnlyMode) {
|
||||
log.warn(
|
||||
"Font compatibility issues detected. Using box-only redaction mode for better reliability.");
|
||||
"Font compatibility issue in placeholder pass; the true-removal pass and "
|
||||
+ "verification still guarantee the target is gone.");
|
||||
|
||||
fallbackDocument = pdfDocumentFactory.load(request.getFileInput());
|
||||
|
||||
@@ -220,7 +237,9 @@ public class RedactController {
|
||||
request.getRedactColor(),
|
||||
request.getCustomPadding(),
|
||||
request.getConvertPDFToImage(),
|
||||
false);
|
||||
false,
|
||||
verificationTargets,
|
||||
compiledPatterns);
|
||||
|
||||
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
|
||||
}
|
||||
@@ -232,7 +251,9 @@ public class RedactController {
|
||||
request.getRedactColor(),
|
||||
request.getCustomPadding(),
|
||||
request.getConvertPDFToImage(),
|
||||
true);
|
||||
true,
|
||||
verificationTargets,
|
||||
compiledPatterns);
|
||||
|
||||
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
|
||||
|
||||
|
||||
+23
-1
@@ -7,8 +7,10 @@ import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
@@ -30,6 +32,7 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyl
|
||||
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange;
|
||||
import stirling.software.SPDF.pdf.parser.PageColumnLayout;
|
||||
import stirling.software.SPDF.pdf.parser.PageImageLocator;
|
||||
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
@@ -140,13 +143,32 @@ class RedactExecuteService {
|
||||
applyAllImagesRedaction(document, request.getRedactImagePages(), style);
|
||||
}
|
||||
|
||||
// Explicit overlay-only requests must not rewrite content or verify. When overlay-only
|
||||
// was forced by a font fallback (not user choice) we still pass the targets so the
|
||||
// pipeline's true-removal + verification + page-scoped raster fallback run.
|
||||
Set<String> literalTargets = new LinkedHashSet<>();
|
||||
for (String value : textValues) {
|
||||
String trimmed = value == null ? "" : value.trim();
|
||||
if (!trimmed.isEmpty()) {
|
||||
literalTargets.add(trimmed);
|
||||
}
|
||||
}
|
||||
List<Pattern> verificationPatterns =
|
||||
RedactionPipeline.buildPatterns(
|
||||
regexPatterns.toArray(new String[0]), true, false);
|
||||
Set<String> finalizeTargets = overlayOnly ? Collections.emptySet() : literalTargets;
|
||||
List<Pattern> finalizePatterns =
|
||||
overlayOnly ? Collections.emptyList() : verificationPatterns;
|
||||
|
||||
return manualRedactionService.finalizeRedaction(
|
||||
document,
|
||||
foundTexts,
|
||||
style.getColor(),
|
||||
style.getPadding(),
|
||||
convertToImage,
|
||||
!needsOverlayOnly);
|
||||
!needsOverlayOnly,
|
||||
finalizeTargets,
|
||||
finalizePatterns);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.error("Execute redaction failed: {}", e.getMessage(), e);
|
||||
|
||||
@@ -0,0 +1,613 @@
|
||||
package stirling.software.SPDF.pdf.redaction;
|
||||
|
||||
import java.util.Calendar;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.cos.COSArray;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.cos.COSObject;
|
||||
import org.apache.pdfbox.cos.COSStream;
|
||||
import org.apache.pdfbox.cos.COSString;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentInformation;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.common.PDNameTreeNode;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
|
||||
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline;
|
||||
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDField;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Walks a {@link PDDocument} and physically removes or rewrites every carrier that a PDF can use to
|
||||
* leak text which the user asked to redact.
|
||||
*
|
||||
* <p>Covers:
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@link PDDocumentInformation} (Info dict) + XMP metadata stream
|
||||
* <li>{@link PDDocumentOutline} bookmark titles
|
||||
* <li>{@link PDAcroForm} field values (V, DV) and rich text (RV)
|
||||
* <li>Every {@link PDAnnotation} Contents and RC
|
||||
* <li>Structure tree ActualText, Alt, T, E, Lang entries
|
||||
* <li>Names tree: JavaScript entries and embedded files (dropped entirely when matching)
|
||||
* </ul>
|
||||
*
|
||||
* <p>When applied after the content-stream rewrite it closes the secondary leak paths flagged in
|
||||
* the redaction security audit.
|
||||
*/
|
||||
@Slf4j
|
||||
public final class CatalogScrubber {
|
||||
|
||||
private CatalogScrubber() {}
|
||||
|
||||
/**
|
||||
* Remove occurrences of every {@code target} string (and any regex/whole-word pattern form
|
||||
* produced by {@link RedactionPipeline#buildPatterns}) from all catalog-level carriers of the
|
||||
* document. When {@code wipeAllMetadata} is {@code true} the document Info dict entries and XMP
|
||||
* metadata stream are wiped wholesale; this is the safe default after a redaction operation.
|
||||
*/
|
||||
public static void scrub(
|
||||
PDDocument document, Set<String> literalTargets, List<Pattern> patterns) {
|
||||
if (document == null) {
|
||||
return;
|
||||
}
|
||||
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
if (catalog == null) {
|
||||
return;
|
||||
}
|
||||
|
||||
scrubOutline(catalog.getDocumentOutline(), literalTargets, patterns);
|
||||
scrubAcroForm(catalog.getAcroForm(), literalTargets, patterns);
|
||||
scrubAnnotations(document, literalTargets, patterns);
|
||||
scrubStructTree(catalog.getStructureTreeRoot(), literalTargets, patterns);
|
||||
scrubNames(catalog.getNames(), literalTargets, patterns);
|
||||
scrubCatalogActions(catalog, literalTargets, patterns);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Catalog actions: OpenAction, AA, and any JavaScript / URI payloads on the catalog
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubCatalogActions(
|
||||
PDDocumentCatalog catalog, Set<String> targets, List<Pattern> patterns) {
|
||||
COSDictionary root = catalog.getCOSObject();
|
||||
if (root == null) {
|
||||
return;
|
||||
}
|
||||
// OpenAction may be either an action dict (with /URI or /JS) or an explicit destination
|
||||
// (array). We scrub strings in both cases; if the OpenAction matches a target we clear it.
|
||||
scrubActionIfMatching(root, COSName.getPDFName("OpenAction"), targets, patterns);
|
||||
scrubActionIfMatching(root, COSName.getPDFName("AA"), targets, patterns);
|
||||
}
|
||||
|
||||
/**
|
||||
* If the action dictionary at {@code key} contains any target literal in a URI or JS payload,
|
||||
* wipe the key entirely. Otherwise recursively scrub string fields inside it.
|
||||
*/
|
||||
private static void scrubActionIfMatching(
|
||||
COSDictionary parent, COSName key, Set<String> targets, List<Pattern> patterns) {
|
||||
if (parent == null || key == null) {
|
||||
return;
|
||||
}
|
||||
COSBase value = parent.getDictionaryObject(key);
|
||||
if (value == null) {
|
||||
return;
|
||||
}
|
||||
if (containsTarget(value, targets, patterns, new HashSet<>())) {
|
||||
log.debug("Removing catalog {} due to target match", key.getName());
|
||||
parent.removeItem(key);
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean containsTarget(
|
||||
COSBase base, Set<String> targets, List<Pattern> patterns, Set<COSBase> seen) {
|
||||
if (base == null) {
|
||||
return false;
|
||||
}
|
||||
COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base;
|
||||
if (resolved == null || !seen.add(resolved)) {
|
||||
return false;
|
||||
}
|
||||
if (resolved instanceof COSString cs) {
|
||||
return matches(cs.getString(), targets, patterns);
|
||||
}
|
||||
if (resolved instanceof COSStream stream) {
|
||||
// Streams in XFA / OpenAction contexts are text (XML, JavaScript). Read the bytes as
|
||||
// UTF-8 and test for target literals. We cap read length to avoid pathological memory
|
||||
// use; 2 MiB is plenty for XFA packets and far beyond any realistic JS action.
|
||||
try (java.io.InputStream is = stream.createInputStream()) {
|
||||
byte[] buf = new byte[2 * 1024 * 1024];
|
||||
int total = 0;
|
||||
int n;
|
||||
while ((n = is.read(buf, total, buf.length - total)) > 0) {
|
||||
total += n;
|
||||
if (total >= buf.length) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
String text = new String(buf, 0, total, java.nio.charset.StandardCharsets.UTF_8);
|
||||
return matches(text, targets, patterns);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scan stream for targets: {}", e.getMessage());
|
||||
// Fail closed: if we cannot read it we cannot prove it is clean, so treat as a
|
||||
// match so the caller drops the stream. This is conservative by design.
|
||||
return true;
|
||||
}
|
||||
}
|
||||
if (resolved instanceof COSDictionary dict) {
|
||||
for (COSName k : new HashSet<>(dict.keySet())) {
|
||||
if (containsTarget(dict.getItem(k), targets, patterns, seen)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (resolved instanceof COSArray array) {
|
||||
for (int i = 0; i < array.size(); i++) {
|
||||
if (containsTarget(array.getObject(i), targets, patterns, seen)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clean potentially sensitive metadata carriers. Called after {@link #scrub} so that surviving
|
||||
* references to author/subject/keywords/XMP descriptors do not leak redacted values.
|
||||
*/
|
||||
public static void wipeMetadata(PDDocument document) {
|
||||
if (document == null) {
|
||||
return;
|
||||
}
|
||||
PDDocumentInformation info = document.getDocumentInformation();
|
||||
if (info != null) {
|
||||
info.setAuthor(null);
|
||||
info.setSubject(null);
|
||||
info.setKeywords(null);
|
||||
info.setTitle(null);
|
||||
info.setCreator(null);
|
||||
info.setProducer(null);
|
||||
info.setModificationDate(Calendar.getInstance());
|
||||
}
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
if (catalog != null) {
|
||||
try {
|
||||
catalog.setMetadata(null);
|
||||
} catch (Exception e) {
|
||||
log.debug("Could not clear XMP metadata: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Outline
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubOutline(
|
||||
PDDocumentOutline outline, Set<String> targets, List<Pattern> patterns) {
|
||||
if (outline == null) {
|
||||
return;
|
||||
}
|
||||
scrubOutlineItems(outline.children(), targets, patterns);
|
||||
}
|
||||
|
||||
private static void scrubOutlineItems(
|
||||
Iterable<PDOutlineItem> items, Set<String> targets, List<Pattern> patterns) {
|
||||
if (items == null) {
|
||||
return;
|
||||
}
|
||||
for (PDOutlineItem item : items) {
|
||||
try {
|
||||
String title = item.getTitle();
|
||||
if (title != null) {
|
||||
String stripped = stripMatches(title, targets, patterns);
|
||||
if (!stripped.equals(title)) {
|
||||
item.setTitle(stripped);
|
||||
}
|
||||
}
|
||||
// Bookmark actions: /A is an action dict which may carry a /URI or /JS payload.
|
||||
// If any target literal appears anywhere inside the action subtree, drop the
|
||||
// action entirely so the URI / script cannot leak the target.
|
||||
COSDictionary itemDict = item.getCOSObject();
|
||||
if (itemDict != null) {
|
||||
scrubActionIfMatching(itemDict, COSName.A, targets, patterns);
|
||||
scrubActionIfMatching(itemDict, COSName.getPDFName("AA"), targets, patterns);
|
||||
}
|
||||
scrubOutlineItems(item.children(), targets, patterns);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub outline item: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// AcroForm
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubAcroForm(
|
||||
PDAcroForm form, Set<String> targets, List<Pattern> patterns) {
|
||||
if (form == null) {
|
||||
return;
|
||||
}
|
||||
// XFA forms: scrubbed separately because the XFA XML packet carries the "real" field
|
||||
// values for XFA-enabled PDFs. Handle XFA before walking the field tree so we fail closed
|
||||
// if XFA scrubbing throws.
|
||||
scrubXfa(form, targets, patterns);
|
||||
|
||||
try {
|
||||
for (PDField field : form.getFieldTree()) {
|
||||
scrubField(field, targets, patterns);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to walk AcroForm field tree: {}", e.getMessage());
|
||||
}
|
||||
|
||||
// Force viewers to regenerate appearance streams from the (scrubbed) /V values rather
|
||||
// than reusing any cached /AP /N that still contains the target text. Belt-and-braces:
|
||||
// scrubField has also cleared per-widget /AP dicts, but /NeedAppearances ensures any
|
||||
// future change still triggers regeneration.
|
||||
try {
|
||||
form.setNeedAppearances(true);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to set /NeedAppearances on AcroForm: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void scrubXfa(PDAcroForm form, Set<String> targets, List<Pattern> patterns) {
|
||||
try {
|
||||
COSBase xfaBase = form.getCOSObject().getDictionaryObject(COSName.XFA);
|
||||
if (xfaBase == null) {
|
||||
return;
|
||||
}
|
||||
boolean hit = containsTarget(xfaBase, targets, patterns, new HashSet<>());
|
||||
if (hit) {
|
||||
// Simplest safe move: strip the XFA entry entirely. Viewers fall back to the
|
||||
// AcroForm widgets which we have already scrubbed. Leaving a "partially scrubbed"
|
||||
// XFA packet risks regex failures on partial XML and re-encoded entities leaking
|
||||
// the target.
|
||||
log.warn(
|
||||
"Removing XFA form packet from AcroForm - XFA XML contained a redaction "
|
||||
+ "target and has been dropped so viewers render AcroForm widgets "
|
||||
+ "instead.");
|
||||
form.getCOSObject().removeItem(COSName.XFA);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub XFA: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void scrubField(PDField field, Set<String> targets, List<Pattern> patterns) {
|
||||
if (field == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
COSDictionary dict = field.getCOSObject();
|
||||
scrubDictStrings(dict, COSName.V, targets, patterns);
|
||||
scrubDictStrings(dict, COSName.DV, targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("RV"), targets, patterns);
|
||||
// Keep field appearance streams in sync with value where possible.
|
||||
try {
|
||||
if (field.getValueAsString() != null) {
|
||||
String stripped = stripMatches(field.getValueAsString(), targets, patterns);
|
||||
if (!stripped.equals(field.getValueAsString())) {
|
||||
field.setValue(stripped);
|
||||
}
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to rewrite field value via setValue: {}", e.getMessage());
|
||||
}
|
||||
// Drop per-widget appearance streams (/AP dict) for every widget kid of this field.
|
||||
// The cached appearance stream contains the pre-redaction value baked in as glyph
|
||||
// data; simply rewriting /V leaves it visually unchanged in many viewers. Removing /AP
|
||||
// plus /NeedAppearances at the form level forces regeneration.
|
||||
clearWidgetAppearances(dict);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub field: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void clearWidgetAppearances(COSDictionary fieldDict) {
|
||||
if (fieldDict == null) {
|
||||
return;
|
||||
}
|
||||
// The field itself may be a widget (single-widget field) and/or have Kids.
|
||||
fieldDict.removeItem(COSName.AP);
|
||||
COSBase kids = fieldDict.getDictionaryObject(COSName.KIDS);
|
||||
if (kids instanceof COSArray arr) {
|
||||
for (int i = 0; i < arr.size(); i++) {
|
||||
COSBase kidBase = arr.getObject(i);
|
||||
if (kidBase instanceof COSDictionary kidDict) {
|
||||
kidDict.removeItem(COSName.AP);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Annotations
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubAnnotations(
|
||||
PDDocument document, Set<String> targets, List<Pattern> patterns) {
|
||||
try {
|
||||
for (PDPage page : document.getPages()) {
|
||||
List<PDAnnotation> annotations;
|
||||
try {
|
||||
annotations = page.getAnnotations();
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to load annotations for page: {}", e.getMessage());
|
||||
continue;
|
||||
}
|
||||
if (annotations == null) {
|
||||
continue;
|
||||
}
|
||||
for (PDAnnotation annotation : annotations) {
|
||||
scrubAnnotation(annotation, targets, patterns);
|
||||
}
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("Annotation scrub walk failed: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void scrubAnnotation(
|
||||
PDAnnotation annotation, Set<String> targets, List<Pattern> patterns) {
|
||||
if (annotation == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
String contents = annotation.getContents();
|
||||
if (contents != null) {
|
||||
String stripped = stripMatches(contents, targets, patterns);
|
||||
if (!stripped.equals(contents)) {
|
||||
annotation.setContents(stripped);
|
||||
}
|
||||
}
|
||||
COSDictionary dict = annotation.getCOSObject();
|
||||
scrubDictStrings(dict, COSName.getPDFName("RC"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("Subj"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("NM"), targets, patterns);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub annotation: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Structure tree
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubStructTree(
|
||||
PDStructureTreeRoot root, Set<String> targets, List<Pattern> patterns) {
|
||||
if (root == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
scrubStructDict(root.getCOSObject(), targets, patterns, new HashSet<>());
|
||||
} catch (Exception e) {
|
||||
log.debug("Structure tree scrub failed: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void scrubStructDict(
|
||||
COSBase base, Set<String> targets, List<Pattern> patterns, Set<COSBase> seen) {
|
||||
if (base == null) {
|
||||
return;
|
||||
}
|
||||
COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base;
|
||||
if (resolved == null || !seen.add(resolved)) {
|
||||
return;
|
||||
}
|
||||
if (resolved instanceof COSDictionary dict) {
|
||||
// Do not walk into content streams - those are handled by content-stream rewrite.
|
||||
if (resolved instanceof COSStream) {
|
||||
return;
|
||||
}
|
||||
scrubDictStrings(dict, COSName.getPDFName("ActualText"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("Alt"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("E"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns);
|
||||
scrubDictStrings(dict, COSName.getPDFName("Lang"), targets, patterns);
|
||||
for (COSName key : new HashSet<>(dict.keySet())) {
|
||||
COSBase value = dict.getItem(key);
|
||||
if (value instanceof COSDictionary
|
||||
|| value instanceof COSArray
|
||||
|| value instanceof COSObject) {
|
||||
scrubStructDict(value, targets, patterns, seen);
|
||||
}
|
||||
}
|
||||
} else if (resolved instanceof COSArray array) {
|
||||
for (int i = 0; i < array.size(); i++) {
|
||||
scrubStructDict(array.getObject(i), targets, patterns, seen);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Names tree (JavaScript + embedded files)
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubNames(
|
||||
PDDocumentNameDictionary names, Set<String> targets, List<Pattern> patterns) {
|
||||
if (names == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
dropMatchingNames(names.getJavaScript(), targets, patterns);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub JavaScript names: {}", e.getMessage());
|
||||
}
|
||||
try {
|
||||
dropMatchingNames(names.getEmbeddedFiles(), targets, patterns);
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to scrub embedded-file names: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static void dropMatchingNames(
|
||||
PDNameTreeNode<?> node, Set<String> targets, List<Pattern> patterns) {
|
||||
if (node == null) {
|
||||
return;
|
||||
}
|
||||
COSDictionary dict = node.getCOSObject();
|
||||
if (dict == null) {
|
||||
return;
|
||||
}
|
||||
scrubNameTreeDict(dict, targets, patterns);
|
||||
}
|
||||
|
||||
private static void scrubNameTreeDict(
|
||||
COSDictionary dict, Set<String> targets, List<Pattern> patterns) {
|
||||
if (dict == null) {
|
||||
return;
|
||||
}
|
||||
COSArray namesArray = (COSArray) dict.getDictionaryObject(COSName.NAMES);
|
||||
if (namesArray != null) {
|
||||
for (int i = namesArray.size() - 2; i >= 0; i -= 2) {
|
||||
COSBase keyBase = namesArray.getObject(i);
|
||||
String key = keyBase instanceof COSString s ? s.getString() : null;
|
||||
if (key != null && matches(key, targets, patterns)) {
|
||||
namesArray.remove(i + 1);
|
||||
namesArray.remove(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
COSArray kids = (COSArray) dict.getDictionaryObject(COSName.KIDS);
|
||||
if (kids != null) {
|
||||
for (int i = 0; i < kids.size(); i++) {
|
||||
COSBase kid = kids.getObject(i);
|
||||
if (kid instanceof COSDictionary kidDict) {
|
||||
scrubNameTreeDict(kidDict, targets, patterns);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------
|
||||
|
||||
private static void scrubDictStrings(
|
||||
COSDictionary dict, COSName key, Set<String> targets, List<Pattern> patterns) {
|
||||
if (dict == null || key == null) {
|
||||
return;
|
||||
}
|
||||
COSBase value = dict.getDictionaryObject(key);
|
||||
if (value instanceof COSString cosString) {
|
||||
String stripped = stripMatches(cosString.getString(), targets, patterns);
|
||||
if (!stripped.equals(cosString.getString())) {
|
||||
dict.setString(key, stripped);
|
||||
}
|
||||
} else if (value instanceof COSArray array) {
|
||||
for (int i = 0; i < array.size(); i++) {
|
||||
COSBase element = array.getObject(i);
|
||||
if (element instanceof COSString elementString) {
|
||||
String stripped = stripMatches(elementString.getString(), targets, patterns);
|
||||
if (!stripped.equals(elementString.getString())) {
|
||||
array.set(i, new COSString(stripped));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static String stripMatches(String source, Set<String> literalTargets, List<Pattern> patterns) {
|
||||
if (source == null || source.isEmpty()) {
|
||||
return source;
|
||||
}
|
||||
String result = source;
|
||||
if (literalTargets != null) {
|
||||
for (String target : literalTargets) {
|
||||
if (target == null || target.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
// Case-insensitive literal removal. Verification is case-insensitive, so scrubbing
|
||||
// MUST be too or mixed-case ("SMITH" in a catalog string vs "Smith" in the target
|
||||
// list) will fail verification and trip the rasterisation fallback - or worse, on
|
||||
// carriers that are not verified, leak the string untouched.
|
||||
result = caseInsensitiveReplaceAll(result, target);
|
||||
}
|
||||
}
|
||||
if (patterns != null) {
|
||||
for (Pattern pattern : patterns) {
|
||||
try {
|
||||
// Force case-insensitive matching for catalog carriers regardless of the flags
|
||||
// the pattern was compiled with. User-supplied redaction targets should not
|
||||
// silently miss because the author typed the name in different case.
|
||||
Pattern ci = withCaseInsensitive(pattern);
|
||||
result = ci.matcher(result).replaceAll("");
|
||||
} catch (Exception e) {
|
||||
log.debug(
|
||||
"Pattern replace failed for {}: {}", pattern.pattern(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
static boolean matches(String source, Set<String> literalTargets, List<Pattern> patterns) {
|
||||
if (source == null || source.isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
String lower = source.toLowerCase(Locale.ROOT);
|
||||
if (literalTargets != null) {
|
||||
for (String target : literalTargets) {
|
||||
if (target != null
|
||||
&& !target.isEmpty()
|
||||
&& lower.contains(target.toLowerCase(Locale.ROOT))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (patterns != null) {
|
||||
for (Pattern pattern : patterns) {
|
||||
try {
|
||||
if (withCaseInsensitive(pattern).matcher(source).find()) {
|
||||
return true;
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("Pattern match failed for {}: {}", pattern.pattern(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
private static String caseInsensitiveReplaceAll(String source, String target) {
|
||||
if (target.isEmpty()) {
|
||||
return source;
|
||||
}
|
||||
Pattern literal =
|
||||
Pattern.compile(
|
||||
Pattern.quote(target), Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE);
|
||||
return literal.matcher(source).replaceAll("");
|
||||
}
|
||||
|
||||
private static Pattern withCaseInsensitive(Pattern pattern) {
|
||||
if ((pattern.flags() & Pattern.CASE_INSENSITIVE) != 0) {
|
||||
return pattern;
|
||||
}
|
||||
try {
|
||||
return Pattern.compile(
|
||||
pattern.pattern(),
|
||||
pattern.flags() | Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE);
|
||||
} catch (Exception e) {
|
||||
return pattern;
|
||||
}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
+17
@@ -0,0 +1,17 @@
|
||||
package stirling.software.SPDF.pdf.redaction;
|
||||
|
||||
/**
|
||||
* Thrown when a post-redaction verification pass still finds any of the target strings in the
|
||||
* re-parsed PDF text. Indicates that the redaction pipeline did not fully remove the targeted
|
||||
* content and the output must not be treated as safe to release.
|
||||
*/
|
||||
public class RedactionVerificationFailedException extends RuntimeException {
|
||||
|
||||
public RedactionVerificationFailedException(String message) {
|
||||
super(message);
|
||||
}
|
||||
|
||||
public RedactionVerificationFailedException(String message, Throwable cause) {
|
||||
super(message, cause);
|
||||
}
|
||||
}
|
||||
+45
-5
@@ -19,6 +19,7 @@ import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -568,7 +569,15 @@ class ManualRedactionServiceTest {
|
||||
byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710))));
|
||||
|
||||
TempFile result =
|
||||
service.finalizeRedaction(doc, byPage, "#000000", 1.0f, false, false);
|
||||
service.finalizeRedaction(
|
||||
doc,
|
||||
byPage,
|
||||
"#000000",
|
||||
1.0f,
|
||||
false,
|
||||
false,
|
||||
Collections.emptySet(),
|
||||
Collections.emptyList());
|
||||
|
||||
assertNotNull(result);
|
||||
assertNotNull(result.getFile());
|
||||
@@ -590,7 +599,15 @@ class ManualRedactionServiceTest {
|
||||
Map<Integer, List<PDFText>> byPage = new HashMap<>();
|
||||
|
||||
TempFile result =
|
||||
service.finalizeRedaction(doc, byPage, "#000000", 0.0f, false, false);
|
||||
service.finalizeRedaction(
|
||||
doc,
|
||||
byPage,
|
||||
"#000000",
|
||||
0.0f,
|
||||
false,
|
||||
false,
|
||||
Collections.emptySet(),
|
||||
Collections.emptyList());
|
||||
|
||||
assertNotNull(result);
|
||||
assertTrue(result.getFile().exists());
|
||||
@@ -610,7 +627,15 @@ class ManualRedactionServiceTest {
|
||||
Map<Integer, List<PDFText>> byPage = new HashMap<>();
|
||||
|
||||
TempFile result =
|
||||
service.finalizeRedaction(doc, byPage, "#000000", 0.0f, null, false);
|
||||
service.finalizeRedaction(
|
||||
doc,
|
||||
byPage,
|
||||
"#000000",
|
||||
0.0f,
|
||||
null,
|
||||
false,
|
||||
Collections.emptySet(),
|
||||
Collections.emptyList());
|
||||
|
||||
assertNotNull(result);
|
||||
assertTrue(result.getFile().exists());
|
||||
@@ -630,7 +655,15 @@ class ManualRedactionServiceTest {
|
||||
byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710))));
|
||||
|
||||
TempFile result =
|
||||
service.finalizeRedaction(doc, byPage, "#FF0000", 2.0f, false, true);
|
||||
service.finalizeRedaction(
|
||||
doc,
|
||||
byPage,
|
||||
"#FF0000",
|
||||
2.0f,
|
||||
false,
|
||||
true,
|
||||
Collections.emptySet(),
|
||||
Collections.emptyList());
|
||||
|
||||
assertNotNull(result);
|
||||
try (PDDocument reloaded = Loader.loadPDF(result.getFile())) {
|
||||
@@ -658,7 +691,14 @@ class ManualRedactionServiceTest {
|
||||
IOException.class,
|
||||
() ->
|
||||
service.finalizeRedaction(
|
||||
doc, byPage, "#000000", 0.0f, false, false));
|
||||
doc,
|
||||
byPage,
|
||||
"#000000",
|
||||
0.0f,
|
||||
false,
|
||||
false,
|
||||
Collections.emptySet(),
|
||||
Collections.emptyList()));
|
||||
// The failing temp file is closed on the error path.
|
||||
verify(failing).close();
|
||||
} finally {
|
||||
|
||||
+14
-3
@@ -201,6 +201,17 @@ class RedactControllerTest {
|
||||
})
|
||||
.when(mockDocument)
|
||||
.save(any(File.class));
|
||||
// RedactionPipeline serialises via OutputStream before handing the bytes to TempFile, so
|
||||
// the mock must emit a non-empty payload on that code path too.
|
||||
lenient()
|
||||
.doAnswer(
|
||||
inv -> {
|
||||
java.io.OutputStream os = inv.getArgument(0);
|
||||
os.write("mock pdf".getBytes());
|
||||
return null;
|
||||
})
|
||||
.when(mockDocument)
|
||||
.save(any(java.io.OutputStream.class));
|
||||
doNothing().when(mockDocument).close();
|
||||
|
||||
// Build real service instances so tests exercise actual logic
|
||||
@@ -347,7 +358,7 @@ class RedactControllerTest {
|
||||
assertNotNull(response);
|
||||
assertEquals(200, response.getStatusCode().value());
|
||||
|
||||
verify(mockDocument).save(any(File.class));
|
||||
verify(mockDocument).save(any(java.io.OutputStream.class));
|
||||
verify(mockDocument).close();
|
||||
}
|
||||
}
|
||||
@@ -753,7 +764,7 @@ class RedactControllerTest {
|
||||
assertEquals(200, response.getStatusCode().value());
|
||||
assertNotNull(response.getBody());
|
||||
assertTrue(drainBody(response).length > 0);
|
||||
verify(mockDocument, times(1)).save(any(File.class));
|
||||
verify(mockDocument, times(1)).save(any(java.io.OutputStream.class));
|
||||
verify(mockDocument, times(1)).close();
|
||||
}
|
||||
} catch (Exception e) {
|
||||
@@ -776,7 +787,7 @@ class RedactControllerTest {
|
||||
if (response != null) {
|
||||
assertNotNull(response);
|
||||
assertEquals(200, response.getStatusCode().value());
|
||||
verify(mockDocument, times(1)).save(any(File.class));
|
||||
verify(mockDocument, times(1)).save(any(java.io.OutputStream.class));
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.info("Manual redaction test completed with graceful handling: {}", e.getMessage());
|
||||
|
||||
+569
@@ -0,0 +1,569 @@
|
||||
package stirling.software.SPDF.controller.api.security;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
import static org.mockito.ArgumentMatchers.any;
|
||||
import static org.mockito.ArgumentMatchers.anyString;
|
||||
import static org.mockito.Mockito.lenient;
|
||||
import static org.mockito.Mockito.mock;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDFormContentStream;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.PDResources;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDFont;
|
||||
import org.apache.pdfbox.pdmodel.font.PDTrueTypeFont;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType0Font;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.apache.pdfbox.pdmodel.font.encoding.MacRomanEncoding;
|
||||
import org.apache.pdfbox.pdmodel.font.encoding.WinAnsiEncoding;
|
||||
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
|
||||
import org.apache.pdfbox.text.PDFTextStripper;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.springframework.core.io.Resource;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.mock.web.MockMultipartFile;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import stirling.software.SPDF.model.api.security.RedactPdfRequest;
|
||||
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
|
||||
/**
|
||||
* Security matrix for redaction across PDF shapes: standard and embedded (subset and full) fonts,
|
||||
* rotated pages, TJ splits, cross-operator splits, Form XObjects, CropBox offsets, multi-page and
|
||||
* case variants. Every case asserts the target is unrecoverable from the output text layer, and
|
||||
* (where removal should be surgical) that neighbouring text survives. Uses the LiberationSans TTF
|
||||
* bundled inside the PDFBox jar so the embedded-font cases run on any CI platform.
|
||||
*/
|
||||
@DisplayName("Redaction PDF-variety security matrix")
|
||||
class RedactionPdfVarietyTest {
|
||||
|
||||
private static final String LIBERATION =
|
||||
"/org/apache/pdfbox/resources/ttf/LiberationSans-Regular.ttf";
|
||||
private static final float FONT_SIZE = 12f;
|
||||
private static final float LEFT_X = 72f;
|
||||
private static final float TOP_Y = PDRectangle.LETTER.getHeight() - 80f;
|
||||
|
||||
private CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
private TempFileManager tempFileManager;
|
||||
private RedactController controller;
|
||||
|
||||
private final List<File> createdTempFiles = new ArrayList<>();
|
||||
|
||||
@BeforeEach
|
||||
void setUp() throws IOException {
|
||||
pdfDocumentFactory = mock(CustomPDFDocumentFactory.class);
|
||||
tempFileManager = mock(TempFileManager.class);
|
||||
|
||||
lenient()
|
||||
.when(tempFileManager.createManagedTempFile(anyString()))
|
||||
.thenAnswer(
|
||||
inv -> {
|
||||
File f =
|
||||
Files.createTempFile(
|
||||
"redact-variety", inv.<String>getArgument(0))
|
||||
.toFile();
|
||||
createdTempFiles.add(f);
|
||||
TempFile tf = mock(TempFile.class);
|
||||
lenient().when(tf.getFile()).thenReturn(f);
|
||||
lenient().when(tf.getPath()).thenReturn(f.toPath());
|
||||
return tf;
|
||||
});
|
||||
|
||||
controller =
|
||||
new RedactController(
|
||||
pdfDocumentFactory,
|
||||
tempFileManager,
|
||||
new ManualRedactionService(tempFileManager),
|
||||
new TextRedactionService(),
|
||||
mock(RedactExecuteService.class));
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
for (File f : createdTempFiles) {
|
||||
if (f != null && f.exists()) {
|
||||
f.delete();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Matrix cases (all through the real /auto-redact controller path)
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
@DisplayName("standard Helvetica: target gone, neighbours survive")
|
||||
void standard14Helvetica() throws IOException {
|
||||
byte[] out = autoRedact(helveticaPdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("standard-14 Times and Courier: target gone, neighbours survive")
|
||||
void standard14TimesAndCourier() throws IOException {
|
||||
for (Standard14Fonts.FontName fn :
|
||||
new Standard14Fonts.FontName[] {
|
||||
Standard14Fonts.FontName.TIMES_ROMAN, Standard14Fonts.FontName.COURIER
|
||||
}) {
|
||||
byte[] pdf = std14Pdf(fn, "alpha SECRET omega");
|
||||
byte[] out = autoRedact(pdf, "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).as("%s keeps neighbours", fn).contains("alpha");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("simple TrueType with MacRoman encoding: target gone, neighbours survive")
|
||||
void simpleTrueTypeMacRoman() throws IOException {
|
||||
byte[] out = autoRedact(macRomanTtfPdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("embedded Type0 with /ToUnicode stripped: target still gone, neighbours survive")
|
||||
void type0WithoutToUnicode() throws IOException {
|
||||
byte[] out = autoRedact(type0NoToUnicodePdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("subset-tagged font (ABCDEF+): target gone, neighbours survive")
|
||||
void subsetTaggedFont() throws IOException {
|
||||
byte[] out = autoRedact(subsetTaggedPdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("real Type3 font (glyphs as content streams, no ToUnicode): target gone")
|
||||
void type3GlyphProcs() throws IOException {
|
||||
// crop_test.pdf uses DejaVuSans embedded as a subset Type3 font with no ToUnicode map.
|
||||
// Redacting it must not throw (Type3 font.encode() throws UnsupportedOperationException)
|
||||
// and must remove the target from the extractable text layer.
|
||||
byte[] input;
|
||||
try (InputStream in = getClass().getResourceAsStream("/redaction/type3_dejavu.pdf")) {
|
||||
input = in.readAllBytes();
|
||||
}
|
||||
byte[] out = autoRedact(input, "EXAMPLE");
|
||||
assertGone(out, "EXAMPLE");
|
||||
assertThat(pdfText(out)).contains("CROP");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("embedded subset Type0 (CID) font: target gone, neighbours survive")
|
||||
void embeddedSubsetType0() throws IOException {
|
||||
byte[] out = autoRedact(type0Pdf(true, "alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("embedded full Type0 (CID) font: target gone, neighbours survive")
|
||||
void embeddedFullType0() throws IOException {
|
||||
byte[] out = autoRedact(type0Pdf(false, "alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("simple embedded TrueType (WinAnsi) font: target gone, neighbours survive")
|
||||
void simpleTrueTypeWinAnsi() throws IOException {
|
||||
byte[] out = autoRedact(simpleTtfPdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("rotated pages (90/180/270): target gone on every rotation")
|
||||
void rotatedPages() throws IOException {
|
||||
for (int rotation : new int[] {90, 180, 270}) {
|
||||
byte[] out = autoRedact(rotatedPdf(rotation, "alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).as("rotation %d keeps neighbours", rotation).contains("alpha");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("target split across TJ array operands is removed")
|
||||
void tjArraySplit() throws IOException {
|
||||
byte[] out = autoRedact(tjSplitPdf(), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("public");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("target split across separate Tj operators is removed, other pages keep text")
|
||||
void crossOperatorSplit() throws IOException {
|
||||
byte[] out = autoRedact(crossOperatorPdf(), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
// Page 2 was never touched; whatever path handled page 1, page 2 text must survive.
|
||||
assertThat(pdfText(out)).contains("PUBLIC PAGE TWO");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("target inside a Form XObject is removed")
|
||||
void formXObjectText() throws IOException {
|
||||
byte[] out = autoRedact(formXObjectPdf(), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("CropBox smaller than MediaBox: target gone, neighbours survive")
|
||||
void cropBoxOffset() throws IOException {
|
||||
byte[] out = autoRedact(cropBoxPdf("alpha SECRET omega"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("alpha");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("multi-page: target on middle page only, other pages keep their text")
|
||||
void multiPageMiddleTarget() throws IOException {
|
||||
byte[] out =
|
||||
autoRedact(
|
||||
multiPagePdf("PUBLIC ONE", "middle SECRET line", "PUBLIC THREE"), "SECRET");
|
||||
assertGone(out, "SECRET");
|
||||
assertThat(pdfText(out)).contains("PUBLIC ONE").contains("PUBLIC THREE");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("case variant: doc contains 'Secret', target 'SECRET' removes it")
|
||||
void caseVariantRemoved() throws IOException {
|
||||
byte[] out = autoRedact(helveticaPdf("alpha Secret omega"), "SECRET");
|
||||
assertGone(out, "Secret");
|
||||
assertThat(pdfText(out)).contains("alpha").contains("omega");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("whole-word with case variant: 'Cat' removed, substrings survive")
|
||||
void wholeWordCaseVariant() throws IOException {
|
||||
byte[] bytes = helveticaPdf("Cat classification scatter");
|
||||
factoryReturns(bytes);
|
||||
RedactPdfRequest request = baseRequest(bytes, "cat");
|
||||
request.setWholeWordSearch(true);
|
||||
|
||||
byte[] out = drainBody(controller.redactPdf(request));
|
||||
String text = pdfText(out);
|
||||
assertThat(text).contains("classification").contains("scatter");
|
||||
// The standalone word must be gone in any case variant.
|
||||
assertThat(text).doesNotContainPattern("(?i)\\bcat\\b");
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Page-scoped rasterisation (pipeline-level, deterministic)
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
@DisplayName("verification leak rasterises only the leaking page, others keep text")
|
||||
void leakRasterisesOnlyLeakingPage() throws IOException {
|
||||
// Cross-operator split defeats the per-operand literal rewriter, so verification must
|
||||
// catch the leak and rasterise page 1 only, leaving page 2 searchable.
|
||||
byte[] input = crossOperatorPdf();
|
||||
byte[] out;
|
||||
try (PDDocument doc = Loader.loadPDF(input)) {
|
||||
Set<String> targets = Set.of("SECRET");
|
||||
List<Pattern> patterns =
|
||||
RedactionPipeline.buildPatterns(new String[] {"SECRET"}, false, false);
|
||||
RedactionPipeline.redactLiteralTerms(doc, targets, patterns);
|
||||
out = RedactionPipeline.finalize(doc, targets, patterns);
|
||||
}
|
||||
|
||||
try (PDDocument reopened = Loader.loadPDF(out)) {
|
||||
PDFTextStripper stripper = new PDFTextStripper();
|
||||
stripper.setStartPage(1);
|
||||
stripper.setEndPage(1);
|
||||
String page1 = stripper.getText(reopened);
|
||||
stripper.setStartPage(2);
|
||||
stripper.setEndPage(2);
|
||||
String page2 = stripper.getText(reopened);
|
||||
|
||||
assertThat(page1.toLowerCase()).doesNotContain("secret");
|
||||
assertThat(page2).contains("PUBLIC PAGE TWO");
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Drivers and assertions
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
private byte[] autoRedact(byte[] pdfBytes, String target) throws IOException {
|
||||
factoryReturns(pdfBytes);
|
||||
RedactPdfRequest request = baseRequest(pdfBytes, target);
|
||||
ResponseEntity<Resource> response = controller.redactPdf(request);
|
||||
assertThat(response.getStatusCode().value()).isEqualTo(200);
|
||||
return drainBody(response);
|
||||
}
|
||||
|
||||
private RedactPdfRequest baseRequest(byte[] pdfBytes, String target) {
|
||||
RedactPdfRequest request = new RedactPdfRequest();
|
||||
request.setFileInput(pdfFile(pdfBytes));
|
||||
request.setListOfText(target);
|
||||
request.setUseRegex(false);
|
||||
request.setWholeWordSearch(false);
|
||||
request.setRedactColor("#000000");
|
||||
request.setConvertPDFToImage(false);
|
||||
return request;
|
||||
}
|
||||
|
||||
private void assertGone(byte[] out, String target) throws IOException {
|
||||
assertThat(pdfText(out).toLowerCase()).doesNotContain(target.toLowerCase());
|
||||
}
|
||||
|
||||
private void factoryReturns(byte[] pdfBytes) throws IOException {
|
||||
lenient()
|
||||
.when(pdfDocumentFactory.load(any(MultipartFile.class)))
|
||||
.thenAnswer(inv -> Loader.loadPDF(pdfBytes));
|
||||
}
|
||||
|
||||
private MockMultipartFile pdfFile(byte[] bytes) {
|
||||
return new MockMultipartFile("fileInput", "doc.pdf", "application/pdf", bytes);
|
||||
}
|
||||
|
||||
private byte[] drainBody(ResponseEntity<Resource> response) throws IOException {
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream();
|
||||
try (InputStream in = response.getBody().getInputStream()) {
|
||||
in.transferTo(baos);
|
||||
}
|
||||
return baos.toByteArray();
|
||||
}
|
||||
|
||||
private String pdfText(byte[] pdfBytes) throws IOException {
|
||||
try (PDDocument doc = Loader.loadPDF(pdfBytes)) {
|
||||
return new PDFTextStripper().getText(doc);
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// PDF builders
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
private static PDFont helvetica() {
|
||||
return new PDType1Font(Standard14Fonts.FontName.HELVETICA);
|
||||
}
|
||||
|
||||
private byte[] helveticaPdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
addTextPage(doc, helvetica(), line, 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] std14Pdf(Standard14Fonts.FontName fontName, String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
addTextPage(doc, new PDType1Font(fontName), line, 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] macRomanTtfPdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDFont font;
|
||||
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
|
||||
font = PDTrueTypeFont.load(doc, ttf, MacRomanEncoding.INSTANCE);
|
||||
}
|
||||
addTextPage(doc, font, line, 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] type0NoToUnicodePdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDFont font;
|
||||
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
|
||||
font = PDType0Font.load(doc, ttf, true);
|
||||
}
|
||||
addTextPage(doc, font, line, 0, null);
|
||||
font.getCOSObject().removeItem(org.apache.pdfbox.cos.COSName.TO_UNICODE);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] subsetTaggedPdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDFont font;
|
||||
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
|
||||
font = PDType0Font.load(doc, ttf, true);
|
||||
}
|
||||
addTextPage(doc, font, line, 0, null);
|
||||
String tagged = "ABCDEF+" + font.getName();
|
||||
font.getCOSObject().setName(org.apache.pdfbox.cos.COSName.BASE_FONT, tagged);
|
||||
if (font.getFontDescriptor() != null) {
|
||||
font.getFontDescriptor()
|
||||
.getCOSObject()
|
||||
.setName(org.apache.pdfbox.cos.COSName.FONT_NAME, tagged);
|
||||
}
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] type0Pdf(boolean subset, String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDFont font;
|
||||
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
|
||||
font = PDType0Font.load(doc, ttf, subset);
|
||||
}
|
||||
addTextPage(doc, font, line, 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] simpleTtfPdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDFont font;
|
||||
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
|
||||
font = PDTrueTypeFont.load(doc, ttf, WinAnsiEncoding.INSTANCE);
|
||||
}
|
||||
addTextPage(doc, font, line, 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] rotatedPdf(int rotation, String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
addTextPage(doc, helvetica(), line, rotation, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] cropBoxPdf(String line) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
addTextPage(doc, helvetica(), line, 0, new PDRectangle(40, 40, 500, 700));
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] multiPagePdf(String... pageLines) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
for (String line : pageLines) {
|
||||
addTextPage(doc, helvetica(), line, 0, null);
|
||||
}
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
/** One page whose text is emitted as a TJ array: ["public ", "SEC", -20, "RET", " end"]. */
|
||||
private byte[] tjSplitPdf() throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.LETTER);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(helvetica(), FONT_SIZE);
|
||||
cs.newLineAtOffset(LEFT_X, TOP_Y);
|
||||
cs.showTextWithPositioning(
|
||||
new Object[] {"public ", "SEC", Float.valueOf(-20f), "RET", " end"});
|
||||
cs.endText();
|
||||
}
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
/** Page 1 shows "SEC" and "RET" as separate adjacent Tj operators; page 2 is clean. */
|
||||
private byte[] crossOperatorPdf() throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.LETTER);
|
||||
doc.addPage(page);
|
||||
PDFont font = helvetica();
|
||||
float secWidth = font.getStringWidth("SEC") / 1000f * FONT_SIZE;
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(font, FONT_SIZE);
|
||||
cs.newLineAtOffset(LEFT_X, TOP_Y);
|
||||
cs.showText("SEC");
|
||||
cs.endText();
|
||||
cs.beginText();
|
||||
cs.setFont(font, FONT_SIZE);
|
||||
cs.newLineAtOffset(LEFT_X + secWidth, TOP_Y);
|
||||
cs.showText("RET");
|
||||
cs.endText();
|
||||
}
|
||||
addTextPage(doc, helvetica(), "PUBLIC PAGE TWO", 0, null);
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private byte[] formXObjectPdf() throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.LETTER);
|
||||
doc.addPage(page);
|
||||
|
||||
PDFormXObject form = new PDFormXObject(doc);
|
||||
form.setBBox(new PDRectangle(0, 0, 400, 60));
|
||||
form.setResources(new PDResources());
|
||||
try (PDFormContentStream fcs = new PDFormContentStream(form)) {
|
||||
fcs.beginText();
|
||||
fcs.setFont(helvetica(), FONT_SIZE);
|
||||
fcs.newLineAtOffset(10, 20);
|
||||
fcs.showText("xobj SECRET payload");
|
||||
fcs.endText();
|
||||
}
|
||||
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.saveGraphicsState();
|
||||
cs.transform(Matrix.getTranslateInstance(LEFT_X, TOP_Y - 60));
|
||||
cs.drawForm(form);
|
||||
cs.restoreGraphicsState();
|
||||
}
|
||||
return save(doc);
|
||||
}
|
||||
}
|
||||
|
||||
private void addTextPage(
|
||||
PDDocument doc, PDFont font, String line, int rotation, PDRectangle cropBox)
|
||||
throws IOException {
|
||||
PDPage page = new PDPage(PDRectangle.LETTER);
|
||||
if (rotation != 0) {
|
||||
page.setRotation(rotation);
|
||||
}
|
||||
if (cropBox != null) {
|
||||
page.setCropBox(cropBox);
|
||||
}
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(font, FONT_SIZE);
|
||||
if (rotation != 0) {
|
||||
// Real generators compensate the text matrix so text reads upright on rotated
|
||||
// pages; anchor at page centre so every rotation keeps the line on-page.
|
||||
cs.setTextMatrix(
|
||||
Matrix.getRotateInstance(
|
||||
Math.toRadians(rotation),
|
||||
PDRectangle.LETTER.getWidth() / 2,
|
||||
PDRectangle.LETTER.getHeight() / 2));
|
||||
} else {
|
||||
cs.newLineAtOffset(LEFT_X, TOP_Y);
|
||||
}
|
||||
cs.showText(line);
|
||||
cs.endText();
|
||||
}
|
||||
}
|
||||
|
||||
private static byte[] save(PDDocument doc) throws IOException {
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream();
|
||||
doc.save(baos);
|
||||
return baos.toByteArray();
|
||||
}
|
||||
}
|
||||
+174
@@ -0,0 +1,174 @@
|
||||
package stirling.software.SPDF.pdf.redaction;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.InputStream;
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.apache.pdfbox.text.PDFTextStripper;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/**
|
||||
* Integration tests that exercise the real redaction pipeline against PDFs with characteristics
|
||||
* that have tripped up the implementation on real user files: page rotation (/Rotate 90), entire
|
||||
* phrases packed into one Tj operator, and standard Type1 fonts with WinAnsiEncoding.
|
||||
*
|
||||
* <p>The critical assertion for every test is that a fresh {@link PDFTextStripper} with default
|
||||
* configuration run over the saved bytes does not contain the target search term
|
||||
* (case-insensitive).
|
||||
*/
|
||||
class RedactionPipelineIntegrationTest {
|
||||
|
||||
private static byte[] loadFixture() throws Exception {
|
||||
try (InputStream in =
|
||||
RedactionPipelineIntegrationTest.class.getResourceAsStream(
|
||||
"/redaction/test_pdf_1.pdf")) {
|
||||
assertNotNull(in, "fixture resource must exist on classpath");
|
||||
return in.readAllBytes();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"finalize on rotated ReportLab PDF without a rewrite pass falls back to rasterisation so target disappears")
|
||||
void finaliseFallsBackToRasterisationWhenTargetWouldSurvive() throws Exception {
|
||||
byte[] fixtureBytes = loadFixture();
|
||||
byte[] outputBytes;
|
||||
try (PDDocument doc = Loader.loadPDF(fixtureBytes)) {
|
||||
Set<String> literalTargets = new LinkedHashSet<>();
|
||||
literalTargets.add("Test");
|
||||
// No content-stream rewriting performed. The primary verification pass inside
|
||||
// finalize must see the surviving target, switch to the rasterisation fallback,
|
||||
// and return bytes that no longer extract as text. This proves the safety net is
|
||||
// wired - without it the output would still contain the target.
|
||||
outputBytes = RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList());
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
|
||||
String extracted = new PDFTextStripper().getText(reopened);
|
||||
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
|
||||
assertFalse(
|
||||
lower.contains("test"),
|
||||
"Rasterisation fallback must have removed target. Extracted: '"
|
||||
+ extracted
|
||||
+ "'");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"finalize with zero rewrite + synthetic upright PDF triggers RedactionVerificationFailedException only when fallback disabled")
|
||||
void verificationExceptionWiringSanityCheck() throws Exception {
|
||||
// This test proves the verification hook itself still throws - when finalize cannot
|
||||
// rasterise (we pass an already-closed document by letting finalize re-save an empty
|
||||
// fresh document that still has the target text baked in).
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("Top Smith Classified");
|
||||
cs.endText();
|
||||
}
|
||||
Set<String> literalTargets = new LinkedHashSet<>();
|
||||
literalTargets.add("Smith");
|
||||
// Even with the rasterisation fallback active, the output must not contain the
|
||||
// target. Either the primary pass or the fallback must succeed; both is a bug.
|
||||
byte[] outBytes =
|
||||
RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList());
|
||||
try (PDDocument reopened = Loader.loadPDF(outBytes)) {
|
||||
String extracted = new PDFTextStripper().getText(reopened);
|
||||
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
|
||||
assertFalse(
|
||||
lower.contains("smith"),
|
||||
"Neither primary pass nor rasterisation removed target. Extracted: '"
|
||||
+ extracted
|
||||
+ "'");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"auto-word redact on rotated PDF with target packed in single Tj removes the word from text stream")
|
||||
void autoWordRedactRemovesTargetOnRotatedSingleTjPdf() throws Exception {
|
||||
byte[] fixtureBytes = loadFixture();
|
||||
byte[] outputBytes;
|
||||
Set<String> literalTargets = new LinkedHashSet<>();
|
||||
literalTargets.add("Test");
|
||||
List<Pattern> patterns =
|
||||
RedactionPipeline.buildPatterns(new String[] {"Test"}, false, false);
|
||||
|
||||
try (PDDocument doc = Loader.loadPDF(fixtureBytes)) {
|
||||
RedactionPipeline.redactLiteralTerms(doc, literalTargets, patterns);
|
||||
outputBytes = RedactionPipeline.finalize(doc, literalTargets, patterns);
|
||||
}
|
||||
|
||||
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
|
||||
PDFTextStripper stripper = new PDFTextStripper();
|
||||
String extracted = stripper.getText(reopened);
|
||||
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
|
||||
assertFalse(
|
||||
lower.contains("test"),
|
||||
"Target term must not be extractable after redact. Extracted: '"
|
||||
+ extracted
|
||||
+ "'");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("simple upright Helvetica fixture still gets word redacted via the pipeline")
|
||||
void autoWordRedactRemovesTargetOnSyntheticPdf() throws Exception {
|
||||
byte[] outputBytes;
|
||||
Set<String> literalTargets = new LinkedHashSet<>();
|
||||
literalTargets.add("Secret");
|
||||
List<Pattern> patterns =
|
||||
RedactionPipeline.buildPatterns(new String[] {"Secret"}, false, false);
|
||||
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("Top Secret Classified");
|
||||
cs.endText();
|
||||
}
|
||||
ByteArrayOutputStream tmp = new ByteArrayOutputStream();
|
||||
doc.save(tmp);
|
||||
// reload from bytes so we are processing an already-saved PDF similar to the
|
||||
// real upload flow.
|
||||
try (PDDocument reloaded = Loader.loadPDF(tmp.toByteArray())) {
|
||||
RedactionPipeline.redactLiteralTerms(reloaded, literalTargets, patterns);
|
||||
outputBytes = RedactionPipeline.finalize(reloaded, literalTargets, patterns);
|
||||
}
|
||||
}
|
||||
|
||||
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
|
||||
String extracted = new PDFTextStripper().getText(reopened);
|
||||
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
|
||||
assertFalse(
|
||||
lower.contains("secret"),
|
||||
"Target term must not be extractable after redact on synthetic PDF. Extracted: '"
|
||||
+ extracted
|
||||
+ "'");
|
||||
}
|
||||
}
|
||||
}
|
||||
+529
@@ -0,0 +1,529 @@
|
||||
package stirling.software.SPDF.pdf.redaction;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.awt.Color;
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.cos.COSStream;
|
||||
import org.apache.pdfbox.cos.COSString;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationHighlight;
|
||||
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline;
|
||||
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDTextField;
|
||||
import org.apache.pdfbox.text.PDFTextStripper;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
class RedactionPipelineTest {
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"redactAreas removes text that falls inside the rectangle from the saved PDF content")
|
||||
void manualAreaRedactRemovesText() throws Exception {
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("SECRET PAYLOAD ALPHA");
|
||||
cs.endText();
|
||||
}
|
||||
|
||||
Map<Integer, List<PDRectangle>> rects = new HashMap<>();
|
||||
// The text above sits around y=700 with height ~12. Cover it fully.
|
||||
rects.put(0, List.of(new PDRectangle(90, 695, 260, 25)));
|
||||
|
||||
RedactionPipeline.redactAreas(doc, rects, Color.BLACK);
|
||||
|
||||
bytes =
|
||||
RedactionPipeline.finalize(
|
||||
doc, Collections.emptySet(), Collections.emptyList());
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
String text = new PDFTextStripper().getText(reopened);
|
||||
assertFalse(
|
||||
text.contains("SECRET"),
|
||||
"Manual-area redacted text must not be extractable, actual='" + text + "'");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"redactWholePages strips all text from the targeted pages while leaving others intact")
|
||||
void wholePageRedactWipesTextOnSelectedPages() throws Exception {
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage p0 = new PDPage(PDRectangle.A4);
|
||||
PDPage p1 = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(p0);
|
||||
doc.addPage(p1);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("TOP SECRET PAGE ONE");
|
||||
cs.endText();
|
||||
}
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("PUBLIC PAGE TWO");
|
||||
cs.endText();
|
||||
}
|
||||
|
||||
RedactionPipeline.redactWholePages(doc, List.of(0), Color.BLACK);
|
||||
|
||||
bytes =
|
||||
RedactionPipeline.finalize(
|
||||
doc, Collections.emptySet(), Collections.emptyList());
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
PDFTextStripper stripper = new PDFTextStripper();
|
||||
stripper.setStartPage(1);
|
||||
stripper.setEndPage(1);
|
||||
String p0Text = stripper.getText(reopened);
|
||||
stripper.setStartPage(2);
|
||||
stripper.setEndPage(2);
|
||||
String p1Text = stripper.getText(reopened);
|
||||
assertFalse(p0Text.contains("SECRET"), "Whole-page redaction must wipe text");
|
||||
assertTrue(p1Text.contains("PUBLIC"), "Non-redacted pages must retain content");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"CatalogScrubber strips bookmark titles, form field values, and annotation contents")
|
||||
void catalogScrubRemovesSensitiveStringsFromCarriers() throws Exception {
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
|
||||
// Bookmark carrying the redacted string.
|
||||
PDDocumentOutline outline = new PDDocumentOutline();
|
||||
doc.getDocumentCatalog().setDocumentOutline(outline);
|
||||
PDOutlineItem item = new PDOutlineItem();
|
||||
item.setTitle("See page on Smith case");
|
||||
outline.addLast(item);
|
||||
|
||||
// AcroForm field carrying the redacted string. We set the V entry directly on the
|
||||
// COS dictionary to avoid triggering AppearanceGeneratorHelper which requires a full
|
||||
// /DA + /DR resource graph - CatalogScrubber is supposed to rewrite V regardless.
|
||||
PDAcroForm form = new PDAcroForm(doc);
|
||||
doc.getDocumentCatalog().setAcroForm(form);
|
||||
PDTextField field = new PDTextField(form);
|
||||
field.setPartialName("comments");
|
||||
field.getCOSObject()
|
||||
.setString(org.apache.pdfbox.cos.COSName.V, "Paid by Smith on receipt");
|
||||
form.getFields().add(field);
|
||||
|
||||
// Annotation carrying the redacted string.
|
||||
PDAnnotationHighlight annotation = new PDAnnotationHighlight();
|
||||
annotation.setContents("Note about Smith purchase");
|
||||
annotation.setRectangle(new PDRectangle(10, 10, 100, 20));
|
||||
page.getAnnotations().add(annotation);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream();
|
||||
doc.save(baos);
|
||||
bytes = baos.toByteArray();
|
||||
}
|
||||
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
PDDocumentOutline outline = reopened.getDocumentCatalog().getDocumentOutline();
|
||||
assertFalse(
|
||||
outline.getFirstChild().getTitle().contains("Smith"),
|
||||
"Bookmark titles must be scrubbed");
|
||||
|
||||
PDAcroForm reopenedForm = reopened.getDocumentCatalog().getAcroForm();
|
||||
String fieldValue = reopenedForm.getField("comments").getValueAsString();
|
||||
assertFalse(
|
||||
fieldValue.contains("Smith"),
|
||||
"AcroForm field values must be scrubbed, actual='" + fieldValue + "'");
|
||||
|
||||
PDPage page = reopened.getPage(0);
|
||||
String annotText = page.getAnnotations().get(0).getContents();
|
||||
assertFalse(
|
||||
annotText.contains("Smith"),
|
||||
"Annotation Contents must be scrubbed, actual='" + annotText + "'");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"finalize guarantees target removal even when no content rewrite was done (rasterisation fallback)")
|
||||
void verificationFallbackRasterisesWhenTargetWouldSurvive() throws Exception {
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("Surviving Smith text");
|
||||
cs.endText();
|
||||
}
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
|
||||
// No content-stream rewriting was done. The primary verification must trip and the
|
||||
// rasterisation fallback must kick in so the final bytes still have no target.
|
||||
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList());
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
String extracted = new PDFTextStripper().getText(reopened);
|
||||
assertFalse(
|
||||
extracted != null && extracted.toLowerCase().contains("smith"),
|
||||
"Rasterisation fallback must remove target, actual='" + extracted + "'");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"Manual rect redaction feeds captured text into scoped verification - raster fallback kicks in when text survives")
|
||||
void manualRectVerificationUsesCapturedStrings() throws Exception {
|
||||
// Simulate the failure mode: a rect is drawn over "LEAKED" but the content-stream rewrite
|
||||
// is intentionally bypassed by only calling the CatalogScrubber path (finalize with
|
||||
// targets=["LEAKED"] and affectedPages=[0]). Verification must see the surviving text and
|
||||
// trigger rasterisation of page 0. Page 1 must remain text-searchable.
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage p0 = new PDPage(PDRectangle.A4);
|
||||
PDPage p1 = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(p0);
|
||||
doc.addPage(p1);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("The LEAKED value on page 1");
|
||||
cs.endText();
|
||||
}
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("Totally unrelated page two text");
|
||||
cs.endText();
|
||||
}
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("LEAKED");
|
||||
Set<Integer> affected = new HashSet<>();
|
||||
affected.add(0);
|
||||
|
||||
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected);
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
PDFTextStripper stripper = new PDFTextStripper();
|
||||
stripper.setStartPage(1);
|
||||
stripper.setEndPage(1);
|
||||
String p0Text = stripper.getText(reopened);
|
||||
stripper.setStartPage(2);
|
||||
stripper.setEndPage(2);
|
||||
String p1Text = stripper.getText(reopened);
|
||||
assertFalse(
|
||||
p0Text.toLowerCase().contains("leaked"),
|
||||
"Manual rect verification must trigger raster fallback on the targeted page");
|
||||
assertTrue(
|
||||
p1Text.contains("Totally unrelated"),
|
||||
"Non-targeted pages must remain text-searchable after scoped raster fallback");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"Scoped raster fallback preserves text layer on non-affected pages (only affected pages are rasterised)")
|
||||
void scopedRasterFallbackPreservesUntargetedPages() throws Exception {
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage p0 = new PDPage(PDRectangle.A4);
|
||||
PDPage p1 = new PDPage(PDRectangle.A4);
|
||||
PDPage p2 = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(p0);
|
||||
doc.addPage(p1);
|
||||
doc.addPage(p2);
|
||||
String[] lines = {"SECRET alpha", "PUBLIC beta", "PUBLIC gamma"};
|
||||
for (int i = 0; i < 3; i++) {
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, doc.getPage(i))) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText(lines[i]);
|
||||
cs.endText();
|
||||
}
|
||||
}
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("SECRET");
|
||||
Set<Integer> affected = new HashSet<>();
|
||||
affected.add(0); // only page 0 is affected
|
||||
|
||||
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected);
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
PDFTextStripper stripper = new PDFTextStripper();
|
||||
for (int i = 1; i <= 3; i++) {
|
||||
stripper.setStartPage(i);
|
||||
stripper.setEndPage(i);
|
||||
String text = stripper.getText(reopened);
|
||||
if (i == 1) {
|
||||
assertFalse(
|
||||
text.toLowerCase().contains("secret"),
|
||||
"Affected page text must be gone");
|
||||
} else {
|
||||
assertTrue(
|
||||
text.contains("PUBLIC"),
|
||||
"Untouched page "
|
||||
+ i
|
||||
+ " must still be text-searchable after scoped rasterisation");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"Pathological verification regex triggers verification FAIL (not silent pass) and engages fallback")
|
||||
void pathologicalVerificationRegexFailsClosed() throws Exception {
|
||||
// A regex that throws on .matcher(...).find() - build a pattern that causes a runtime
|
||||
// exception by constructing a matcher over a degenerate character sequence. We simulate
|
||||
// the pathological case by supplying a pattern whose matcher throws StackOverflowError via
|
||||
// deep alternation. Because JVM reliably triggering SOE is tricky, we instead construct
|
||||
// a pattern where .find() throws an unchecked exception using a custom Pattern subclass
|
||||
// is not possible (Pattern is final). The realistic pathological case is catastrophic
|
||||
// backtracking, but we cannot rely on timeouts in a unit test. Instead we verify the
|
||||
// IMPORTANT invariant indirectly: when finalize is called with no content rewrite and a
|
||||
// surviving target, verification must FAIL and the raster fallback must engage - this
|
||||
// proves the verify path is not silently swallowing exceptions (which would return the
|
||||
// unrasterised bytes).
|
||||
byte[] bytes;
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(100, 700);
|
||||
cs.showText("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
|
||||
cs.endText();
|
||||
}
|
||||
// Pattern that matches the content - if regex exceptions were swallowed, verification
|
||||
// would return unrasterised bytes with the match still present.
|
||||
List<Pattern> patterns = List.of(Pattern.compile("a{5,}"));
|
||||
bytes = RedactionPipeline.finalize(doc, Collections.emptySet(), patterns);
|
||||
}
|
||||
try (PDDocument reopened = Loader.loadPDF(bytes)) {
|
||||
String text = new PDFTextStripper().getText(reopened);
|
||||
// Rasterisation should have eliminated text extractability entirely on the affected
|
||||
// page (no affectedPages passed means whole-document rasterisation fallback).
|
||||
assertFalse(
|
||||
Pattern.compile("a{5,}").matcher(text).find(),
|
||||
"Regex match must not survive verification fallback");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"CatalogScrubber.stripMatches is case-insensitive so mixed-case targets are removed from catalog strings")
|
||||
void catalogStripMatchesIsCaseInsensitive() {
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
String result =
|
||||
CatalogScrubber.stripMatches(
|
||||
"SMITH, John (also known as smith and Smith Jr.)", targets, List.of());
|
||||
assertFalse(
|
||||
result.toLowerCase().contains("smith"),
|
||||
"All case variants of the target must be removed, actual='" + result + "'");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"Catalog scrub strips mixed-case target from bookmark title even when case differs")
|
||||
void catalogScrubMixedCaseBookmarkTitle() throws Exception {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
doc.addPage(new PDPage(PDRectangle.A4));
|
||||
PDDocumentOutline outline = new PDDocumentOutline();
|
||||
doc.getDocumentCatalog().setDocumentOutline(outline);
|
||||
PDOutlineItem item = new PDOutlineItem();
|
||||
item.setTitle("SMITH memo");
|
||||
outline.addLast(item);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("smith"); // lowercase target vs uppercase carrier
|
||||
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream();
|
||||
doc.save(baos);
|
||||
try (PDDocument reopened = Loader.loadPDF(baos.toByteArray())) {
|
||||
String title =
|
||||
reopened.getDocumentCatalog()
|
||||
.getDocumentOutline()
|
||||
.getFirstChild()
|
||||
.getTitle();
|
||||
assertFalse(
|
||||
title.toLowerCase().contains("smith"),
|
||||
"Case-insensitive catalog scrub must remove 'SMITH' when target is 'smith', actual='"
|
||||
+ title
|
||||
+ "'");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName(
|
||||
"AcroForm widget appearance streams are cleared so viewers cannot render the stale value")
|
||||
void acroFormAppearanceStreamsCleared() throws Exception {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
doc.addPage(new PDPage(PDRectangle.A4));
|
||||
PDAcroForm form = new PDAcroForm(doc);
|
||||
doc.getDocumentCatalog().setAcroForm(form);
|
||||
PDTextField field = new PDTextField(form);
|
||||
field.setPartialName("note");
|
||||
field.getCOSObject().setString(COSName.V, "Paid by Smith");
|
||||
// Inject a fake AP dict to simulate a cached appearance stream.
|
||||
org.apache.pdfbox.cos.COSDictionary apDict = new org.apache.pdfbox.cos.COSDictionary();
|
||||
apDict.setString(COSName.getPDFName("DUMMY"), "Paid by Smith");
|
||||
field.getCOSObject().setItem(COSName.AP, apDict);
|
||||
form.getFields().add(field);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
assertNull(
|
||||
field.getCOSObject().getDictionaryObject(COSName.AP),
|
||||
"Widget appearance dict must be cleared after scrub");
|
||||
assertTrue(
|
||||
form.getNeedAppearances(),
|
||||
"/NeedAppearances must be true so viewers regenerate appearances");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("XFA packet containing the redaction target is removed from AcroForm")
|
||||
void xfaPacketDropped() throws Exception {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
doc.addPage(new PDPage(PDRectangle.A4));
|
||||
PDAcroForm form = new PDAcroForm(doc);
|
||||
doc.getDocumentCatalog().setAcroForm(form);
|
||||
|
||||
String xfaXml = "<xdp><data><field>Smith</field></data></xdp>";
|
||||
COSStream xfa = doc.getDocument().createCOSStream();
|
||||
try (var os = xfa.createOutputStream()) {
|
||||
os.write(xfaXml.getBytes(java.nio.charset.StandardCharsets.UTF_8));
|
||||
}
|
||||
form.getCOSObject().setItem(COSName.XFA, xfa);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
assertNull(
|
||||
form.getCOSObject().getDictionaryObject(COSName.XFA),
|
||||
"XFA packet containing target literal must be removed");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("OpenAction carrying a target URI is removed from the catalog")
|
||||
void openActionWithTargetUriRemoved() throws Exception {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
doc.addPage(new PDPage(PDRectangle.A4));
|
||||
org.apache.pdfbox.cos.COSDictionary openAction =
|
||||
new org.apache.pdfbox.cos.COSDictionary();
|
||||
openAction.setItem(COSName.getPDFName("S"), COSName.URI);
|
||||
openAction.setItem(
|
||||
COSName.getPDFName("URI"), new COSString("https://example.com/?user=Smith"));
|
||||
doc.getDocumentCatalog()
|
||||
.getCOSObject()
|
||||
.setItem(COSName.getPDFName("OpenAction"), openAction);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
assertNull(
|
||||
doc.getDocumentCatalog()
|
||||
.getCOSObject()
|
||||
.getDictionaryObject(COSName.getPDFName("OpenAction")),
|
||||
"OpenAction containing target must be removed from catalog");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("Bookmark /A URI action containing the redaction target is removed")
|
||||
void bookmarkActionUriRemoved() throws Exception {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
doc.addPage(new PDPage(PDRectangle.A4));
|
||||
PDDocumentOutline outline = new PDDocumentOutline();
|
||||
doc.getDocumentCatalog().setDocumentOutline(outline);
|
||||
PDOutlineItem item = new PDOutlineItem();
|
||||
item.setTitle("Link to case file");
|
||||
|
||||
org.apache.pdfbox.cos.COSDictionary action = new org.apache.pdfbox.cos.COSDictionary();
|
||||
action.setItem(COSName.getPDFName("S"), COSName.URI);
|
||||
action.setItem(
|
||||
COSName.getPDFName("URI"), new COSString("https://example.com/?file=Smith"));
|
||||
item.getCOSObject().setItem(COSName.A, action);
|
||||
|
||||
outline.addLast(item);
|
||||
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add("Smith");
|
||||
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
|
||||
|
||||
assertNull(
|
||||
item.getCOSObject().getDictionaryObject(COSName.A),
|
||||
"Bookmark /A action containing target URI must be removed");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("buildPatterns respects useRegex and wholeWordSearch flags")
|
||||
void buildPatternsHonoursFlags() {
|
||||
List<Pattern> plain = RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, false);
|
||||
assertEquals(1, plain.size());
|
||||
assertTrue(plain.get(0).matcher("AeroSmith").find());
|
||||
assertTrue(plain.get(0).matcher("Smith paid").find());
|
||||
|
||||
List<Pattern> wholeWord =
|
||||
RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, true);
|
||||
assertTrue(wholeWord.get(0).matcher("Smith paid").find());
|
||||
assertFalse(wholeWord.get(0).matcher("AeroSmith").find());
|
||||
|
||||
List<Pattern> regex = RedactionPipeline.buildPatterns(new String[] {"\\d{3}"}, true, false);
|
||||
assertTrue(regex.get(0).matcher("ID 123 issued").find());
|
||||
}
|
||||
}
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,109 @@
|
||||
@security @redact
|
||||
Feature: PDF redaction physically removes text
|
||||
Redaction must destroy the underlying text, not merely cover it with a box.
|
||||
These scenarios push known text through the redaction endpoints and assert the
|
||||
target is gone from the extracted text layer (and catalog carriers) while other
|
||||
text survives.
|
||||
|
||||
Scenario: Auto-redact physically removes the target word
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "PUBLIC alpha SECRET99 omega PUBLIC"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | SECRET99 |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response content type should be "application/pdf"
|
||||
And the response PDF should not contain the text "SECRET99"
|
||||
And the response PDF should contain the text "omega"
|
||||
|
||||
Scenario: Auto-redact leaves substrings when whole-word search is on
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "cat classification scatter"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | cat |
|
||||
| wholeWordSearch | true |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should contain the text "classification"
|
||||
And the response PDF should contain the text "scatter"
|
||||
|
||||
Scenario: Auto-redact with a regex pattern removes matches
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "call 123-45-6789 today"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | \d{3}-\d{2}-\d{4} |
|
||||
| useRegex | true |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "123-45-6789"
|
||||
And the response PDF should contain the text "today"
|
||||
|
||||
Scenario: Auto-redact convert-to-image drops the entire text layer
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "keep SECRET77 hidden"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | SECRET77 |
|
||||
| convertPDFToImage | true |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "SECRET77"
|
||||
And the response PDF should not contain the text "keep"
|
||||
|
||||
Scenario: Auto-redact also scrubs the target from bookmark titles
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "body SECRET55 text"
|
||||
And the pdf has a bookmark titled "Chapter SECRET55 overview"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | SECRET55 |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "SECRET55"
|
||||
And the response PDF bookmarks should not contain "SECRET55"
|
||||
|
||||
Scenario: Manual whole-page redaction wipes every word on the page
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "TOP SECRET material"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| pageNumbers | 1 |
|
||||
When I send the API request to the endpoint "/api/v1/security/redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "SECRET"
|
||||
|
||||
Scenario: Auto-redact removes case variants of the target
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf pages all contain the text "alpha Secret99x omega"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | SECRET99X |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "Secret99x"
|
||||
And the response PDF should contain the text "omega"
|
||||
|
||||
Scenario: Auto-redact only touches pages containing the target
|
||||
Given I generate a PDF file as "fileInput"
|
||||
And the pdf contains 3 pages
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | Page 2 |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "Page 2"
|
||||
And the response PDF should contain the text "Page 1"
|
||||
And the response PDF should contain the text "Page 3"
|
||||
|
||||
Scenario: Auto-redact handles a Type3 font PDF (glyphs as content streams)
|
||||
Given I use an example file at "../crop_test.pdf" as parameter "fileInput"
|
||||
And the request data includes
|
||||
| parameter | value |
|
||||
| listOfText | EXAMPLE |
|
||||
When I send the API request to the endpoint "/api/v1/security/auto-redact"
|
||||
Then the response status code should be 200
|
||||
And the response PDF should not contain the text "EXAMPLE"
|
||||
And the response PDF should contain the text "CROP"
|
||||
@@ -17,6 +17,10 @@ from PIL import Image, ImageDraw
|
||||
|
||||
API_HEADERS = {"X-API-KEY": "123456789"}
|
||||
|
||||
# Base URL of the backend under test. Defaults to the CI value; override with
|
||||
# STIRLING_BASE_URL to point at a backend on another port (e.g. a local sidecar run).
|
||||
BASE_URL = os.environ.get("STIRLING_BASE_URL", "http://localhost:8080")
|
||||
|
||||
#########
|
||||
# GIVEN #
|
||||
#########
|
||||
@@ -581,7 +585,7 @@ def step_request_json_part(context, part_name, json_content):
|
||||
|
||||
@when('I send a GET request to "{endpoint}"')
|
||||
def step_send_get_request(context, endpoint):
|
||||
base_url = "http://localhost:8080"
|
||||
base_url = BASE_URL
|
||||
full_url = f"{base_url}{endpoint}"
|
||||
response = requests.get(full_url, headers=API_HEADERS, timeout=60)
|
||||
context.response = response
|
||||
@@ -589,7 +593,7 @@ def step_send_get_request(context, endpoint):
|
||||
|
||||
@when('I send a GET request to "{endpoint}" with parameters')
|
||||
def step_send_get_request_with_params(context, endpoint):
|
||||
base_url = "http://localhost:8080"
|
||||
base_url = BASE_URL
|
||||
params = {row["parameter"]: row["value"] for row in context.table}
|
||||
full_url = f"{base_url}{endpoint}"
|
||||
response = requests.get(full_url, params=params, headers=API_HEADERS, timeout=60)
|
||||
@@ -598,7 +602,7 @@ def step_send_get_request_with_params(context, endpoint):
|
||||
|
||||
@when('I send the API request to the endpoint "{endpoint}"')
|
||||
def step_send_api_request(context, endpoint):
|
||||
url = f"http://localhost:8080{endpoint}"
|
||||
url = f"{BASE_URL}{endpoint}"
|
||||
files = context.files if hasattr(context, "files") else {}
|
||||
|
||||
if not hasattr(context, "request_data") or context.request_data is None:
|
||||
@@ -803,3 +807,69 @@ def step_response_matches_regex(context, pattern):
|
||||
assert re.match(
|
||||
pattern, response_text
|
||||
), f"Response '{response_text}' does not match the expected pattern '{pattern}'"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Redaction: text-layer and catalog assertions
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _extract_response_pdf_text(context):
|
||||
reader = PdfReader(io.BytesIO(context.response.content))
|
||||
return "\n".join((page.extract_text() or "") for page in reader.pages)
|
||||
|
||||
|
||||
@then('the response PDF should contain the text "{text}"')
|
||||
def step_response_pdf_contains_text(context, text):
|
||||
extracted = _extract_response_pdf_text(context)
|
||||
assert text in extracted, (
|
||||
f"Expected redacted PDF to still contain '{text}', but it was missing. "
|
||||
f"Extracted text: {extracted!r}"
|
||||
)
|
||||
|
||||
|
||||
@then('the response PDF should not contain the text "{text}"')
|
||||
def step_response_pdf_not_contains_text(context, text):
|
||||
extracted = _extract_response_pdf_text(context)
|
||||
assert text not in extracted, (
|
||||
f"Redacted PDF still contains '{text}' - redaction did not remove it. "
|
||||
f"Extracted text: {extracted!r}"
|
||||
)
|
||||
|
||||
|
||||
def _collect_outline_titles(outline, titles):
|
||||
for item in outline:
|
||||
if isinstance(item, list):
|
||||
_collect_outline_titles(item, titles)
|
||||
else:
|
||||
title = getattr(item, "title", None)
|
||||
if title:
|
||||
titles.append(title)
|
||||
|
||||
|
||||
@then('the response PDF bookmarks should not contain "{text}"')
|
||||
def step_response_pdf_bookmarks_not_contain(context, text):
|
||||
reader = PdfReader(io.BytesIO(context.response.content))
|
||||
titles = []
|
||||
try:
|
||||
_collect_outline_titles(reader.outline, titles)
|
||||
except Exception:
|
||||
titles = []
|
||||
joined = " ".join(titles)
|
||||
assert text not in joined, (
|
||||
f"Redacted PDF bookmark titles still contain '{text}': {titles!r}"
|
||||
)
|
||||
|
||||
|
||||
@given('the pdf has a bookmark titled "{title}"')
|
||||
def step_pdf_has_bookmark_titled(context, title):
|
||||
"""Add a single top-level outline entry with an explicit title."""
|
||||
reader = PdfReader(context.file_name)
|
||||
writer = PdfWriter()
|
||||
for page in reader.pages:
|
||||
writer.add_page(page)
|
||||
writer.add_outline_item(title, 0)
|
||||
with open(context.file_name, "wb") as f:
|
||||
writer.write(f)
|
||||
context.files[context.param_name].close()
|
||||
context.files[context.param_name] = open(context.file_name, "rb")
|
||||
|
||||
Reference in New Issue
Block a user