Add true-removal redaction pipeline with verification and font-type coverage

This commit is contained in:
Anthony Stirling
2026-07-02 16:39:04 +01:00
parent 6c85200eb9
commit 18cc6dcdc8
15 changed files with 3556 additions and 125 deletions
@@ -5,8 +5,11 @@ import java.io.IOException;
import java.util.ArrayList;
import java.util.Collections;
import java.util.HashMap;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
@@ -22,6 +25,7 @@ import lombok.extern.slf4j.Slf4j;
import stirling.software.SPDF.model.PDFText;
import stirling.software.SPDF.model.api.security.ManualRedactPdfRequest;
import stirling.software.SPDF.pdf.parser.PageImageLocator;
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
import stirling.software.common.model.api.security.RedactionArea;
import stirling.software.common.util.GeneralUtils;
import stirling.software.common.util.PdfUtils;
@@ -42,11 +46,15 @@ class ManualRedactionService {
// Area and page redaction
// -----------------------------------------------------------------------
void redactAreas(List<RedactionArea> redactionAreas, PDDocument document, PDPageTree allPages)
AreaRedactionResult redactAreas(
List<RedactionArea> redactionAreas, PDDocument document, PDPageTree allPages)
throws IOException {
Set<String> capturedStrings = new LinkedHashSet<>();
Map<Integer, List<PDRectangle>> rectsByPage = new HashMap<>();
if (redactionAreas == null || redactionAreas.isEmpty()) {
return;
return new AreaRedactionResult(rectsByPage);
}
Map<Integer, List<RedactionArea>> redactionsByPage = new HashMap<>();
@@ -74,52 +82,49 @@ class ManualRedactionService {
continue;
}
PDPage page = allPages.get(pageNumber - 1);
int pageIndex = pageNumber - 1;
PDPage page = allPages.get(pageIndex);
float pageHeight = page.getBBox().getHeight();
try (PDPageContentStream contentStream =
new PDPageContentStream(
document, page, PDPageContentStream.AppendMode.APPEND, true, true)) {
contentStream.saveGraphicsState();
for (RedactionArea redactionArea : areasForPage) {
Color redactColor = decodeOrDefault(redactionArea.getColor());
contentStream.setNonStrokingColor(redactColor);
float x = redactionArea.getX().floatValue();
float y = redactionArea.getY().floatValue();
float width = redactionArea.getWidth().floatValue();
float height = redactionArea.getHeight().floatValue();
float pdfY = page.getBBox().getHeight() - y - height;
contentStream.addRect(x, pdfY, width, height);
contentStream.fill();
}
contentStream.restoreGraphicsState();
List<PDRectangle> rects = new ArrayList<>();
Color overlayColor = Color.BLACK;
for (RedactionArea area : areasForPage) {
float x = area.getX().floatValue();
float y = area.getY().floatValue();
float width = area.getWidth().floatValue();
float height = area.getHeight().floatValue();
// Request coords are top-left origin; convert to PDF user space (bottom-left).
float pdfY = pageHeight - y - height;
rects.add(new PDRectangle(x, pdfY, width, height));
overlayColor = decodeOrDefault(area.getColor());
}
// Physically drop intersecting glyphs and draw the overlay rectangle over the area.
Map<Integer, List<PDRectangle>> singlePage = new HashMap<>();
singlePage.put(pageIndex, rects);
RedactionPipeline.RedactionResult result =
RedactionPipeline.redactAreas(document, singlePage, overlayColor);
capturedStrings.addAll(result.getCapturedStrings());
rectsByPage.put(pageIndex, rects);
}
log.debug(
"Manual area redaction captured {} text run(s) across {} page(s)",
capturedStrings.size(),
rectsByPage.size());
return new AreaRedactionResult(rectsByPage);
}
void redactPages(ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages)
List<Integer> redactPages(
ManualRedactPdfRequest request, PDDocument document, PDPageTree allPages)
throws IOException {
Color redactColor = decodeOrDefault(request.getPageRedactionColor());
List<Integer> pageNumbers = getPageNumbers(request, allPages.getCount());
List<Integer> pageIndexes = getPageNumbers(request, allPages.getCount());
for (Integer pageNumber : pageNumbers) {
PDPage page = allPages.get(pageNumber);
try (PDPageContentStream contentStream =
new PDPageContentStream(
document, page, PDPageContentStream.AppendMode.APPEND, true, true)) {
contentStream.setNonStrokingColor(redactColor);
PDRectangle box = page.getBBox();
contentStream.addRect(0, 0, box.getWidth(), box.getHeight());
contentStream.fill();
}
}
// Whole-page wipe: drop the content stream, resources and annotations, then fill.
RedactionPipeline.redactWholePages(document, pageIndexes, redactColor);
return new ArrayList<>(pageIndexes);
}
// -----------------------------------------------------------------------
@@ -298,7 +303,9 @@ class ManualRedactionService {
String colorString,
float customPadding,
Boolean convertToImage,
boolean isTextRemovalMode)
boolean isTextRemovalMode,
Set<String> literalTargets,
List<Pattern> patterns)
throws IOException {
List<PDFText> allFoundTexts = new ArrayList<>();
@@ -309,74 +316,77 @@ class ManualRedactionService {
if (!allFoundTexts.isEmpty()) {
Color redactColor = decodeOrDefault(colorString);
redactFoundText(document, allFoundTexts, customPadding, redactColor, isTextRemovalMode);
cleanDocumentMetadata(document);
}
byte[] outputBytes;
if (Boolean.TRUE.equals(convertToImage)) {
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
cleanDocumentMetadata(convertedPdf);
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
try {
convertedPdf.save(tempOut.getFile());
} catch (IOException e) {
tempOut.close();
throw e;
}
log.info(
"Redaction finalized (image mode): {} pages ➜ {} KB",
convertedPdf.getNumberOfPages(),
tempOut.getFile().length() / 1024);
return tempOut;
// Convert-to-image physically removes all text, so verification is a plain save.
outputBytes =
RedactionPipeline.finalize(
convertedPdf, Collections.emptySet(), Collections.emptyList());
}
} else {
// True-removal pass: physically strip matched glyph bytes from every content stream,
// then scrub catalog carriers, verify, and rasterise affected pages on any leak.
RedactionPipeline.redactLiteralTerms(document, literalTargets, patterns);
outputBytes = RedactionPipeline.finalize(document, literalTargets, patterns);
}
return writeBytes(outputBytes, document.getNumberOfPages());
}
/**
* Finalize a manual area/page redaction. The overlay rectangles are already drawn and the
* intersecting glyphs already dropped. Verification is region-based: each redaction rectangle
* is re-scanned and must be empty (whole-page wipes carry no rects and are guaranteed clean).
*/
TempFile finalizeManual(
PDDocument document,
Map<Integer, List<PDRectangle>> rectsByPage,
Boolean convertToImage)
throws IOException {
byte[] outputBytes;
if (Boolean.TRUE.equals(convertToImage)) {
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
outputBytes =
RedactionPipeline.finalize(
convertedPdf, Collections.emptySet(), Collections.emptyList());
}
} else {
outputBytes = RedactionPipeline.finalizeAreas(document, rectsByPage);
}
return writeBytes(outputBytes, document.getNumberOfPages());
}
private TempFile writeBytes(byte[] outputBytes, int pageCount) throws IOException {
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
try {
document.save(tempOut.getFile());
java.nio.file.Files.write(tempOut.getFile().toPath(), outputBytes);
} catch (IOException e) {
tempOut.close();
throw e;
}
log.info(
"Redaction finalized: {} pages ➜ {} KB",
document.getNumberOfPages(),
tempOut.getFile().length() / 1024);
log.info("Redaction finalized: {} pages -> {} KB", pageCount, outputBytes.length / 1024);
return tempOut;
}
private void cleanDocumentMetadata(PDDocument document) {
try {
var documentInfo = document.getDocumentInformation();
if (documentInfo != null) {
documentInfo.setAuthor(null);
documentInfo.setSubject(null);
documentInfo.setKeywords(null);
documentInfo.setModificationDate(java.util.Calendar.getInstance());
log.debug("Cleaned document metadata for security");
}
if (document.getDocumentCatalog() != null) {
try {
document.getDocumentCatalog().setMetadata(null);
} catch (Exception e) {
log.debug("Could not clear XMP metadata: {}", e.getMessage());
}
}
} catch (Exception e) {
log.warn("Failed to clean document metadata: {}", e.getMessage());
}
}
// -----------------------------------------------------------------------
// Utilities
// -----------------------------------------------------------------------
/** Redaction rectangles (per 0-based page index) applied by a manual area pass. */
static final class AreaRedactionResult {
final Map<Integer, List<PDRectangle>> rectsByPage;
AreaRedactionResult(Map<Integer, List<PDRectangle>> rectsByPage) {
this.rectsByPage = rectsByPage;
}
}
static Color decodeOrDefault(String hex) {
if (hex == null) {
return Color.BLACK;
@@ -1,9 +1,15 @@
package stirling.software.SPDF.controller.api.security;
import java.io.IOException;
import java.util.Arrays;
import java.util.Collections;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Set;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPageTree;
@@ -29,13 +35,13 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.ImageBox;
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyle;
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange;
import stirling.software.SPDF.model.api.security.RedactPdfRequest;
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
import stirling.software.common.annotations.AutoJobPostMapping;
import stirling.software.common.annotations.api.SecurityApi;
import stirling.software.common.enumeration.ResourceWeight;
import stirling.software.common.model.api.security.RedactionArea;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.common.util.PdfUtils;
import stirling.software.common.util.TempFile;
import stirling.software.common.util.TempFileManager;
import stirling.software.common.util.WebResponseUtils;
@@ -93,33 +99,25 @@ public class RedactController {
throws IOException {
MultipartFile file = request.getFileInput();
String filename =
removeFileExtension(
Objects.requireNonNull(
Filenames.toSimpleFileName(file.getOriginalFilename())))
+ "_redacted.pdf";
try (PDDocument document = pdfDocumentFactory.load(file)) {
PDPageTree allPages = document.getDocumentCatalog().getPages();
// Whole-page wipes drop content outright (guaranteed clean); area redactions drop
// intersecting glyphs, draw an overlay, and are verified per-rectangle at finalize.
manualRedactionService.redactPages(request, document, allPages);
manualRedactionService.redactAreas(request.getRedactions(), document, allPages);
ManualRedactionService.AreaRedactionResult areaResult =
manualRedactionService.redactAreas(request.getRedactions(), document, allPages);
if (Boolean.TRUE.equals(request.getConvertPDFToImage())) {
try (PDDocument convertedPdf = PdfUtils.convertPdfToPdfImage(document)) {
return WebResponseUtils.pdfDocToWebResponse(
convertedPdf,
removeFileExtension(
Objects.requireNonNull(
Filenames.toSimpleFileName(
file.getOriginalFilename())))
+ "_redacted.pdf",
tempFileManager);
}
}
return WebResponseUtils.pdfDocToWebResponse(
document,
removeFileExtension(
Objects.requireNonNull(
Filenames.toSimpleFileName(file.getOriginalFilename())))
+ "_redacted.pdf",
tempFileManager);
TempFile out =
manualRedactionService.finalizeManual(
document, areaResult.rectsByPage, request.getConvertPDFToImage());
return WebResponseUtils.pdfFileToWebResponse(out, filename);
}
}
@@ -184,9 +182,27 @@ public class RedactController {
if (allFoundTextsByPage.isEmpty()) {
log.info("No text found matching redaction patterns");
return WebResponseUtils.pdfDocToWebResponse(document, filename, tempFileManager);
// Still finalize so metadata is scrubbed and the document is rewritten
// consistently.
TempFile finalized =
manualRedactionService.finalizeManual(
document, Collections.emptyMap(), request.getConvertPDFToImage());
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
}
Set<String> literalTargets =
Arrays.stream(listOfText)
.map(String::trim)
.filter(s -> !s.isEmpty())
.collect(Collectors.toCollection(LinkedHashSet::new));
List<Pattern> compiledPatterns =
RedactionPipeline.buildPatterns(listOfText, useRegex, wholeWordSearchBool);
// Bare literal targets match substrings, so they are only safe when the search is a
// plain literal. In regex or whole-word mode the boundary/regex semantics live entirely
// in the compiled patterns, so pass no literal targets to avoid stripping substrings.
Set<String> verificationTargets =
(useRegex || wholeWordSearchBool) ? Collections.emptySet() : literalTargets;
boolean fallbackToBoxOnlyMode;
try {
fallbackToBoxOnlyMode =
@@ -205,7 +221,8 @@ public class RedactController {
if (fallbackToBoxOnlyMode) {
log.warn(
"Font compatibility issues detected. Using box-only redaction mode for better reliability.");
"Font compatibility issue in placeholder pass; the true-removal pass and "
+ "verification still guarantee the target is gone.");
fallbackDocument = pdfDocumentFactory.load(request.getFileInput());
@@ -220,7 +237,9 @@ public class RedactController {
request.getRedactColor(),
request.getCustomPadding(),
request.getConvertPDFToImage(),
false);
false,
verificationTargets,
compiledPatterns);
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
}
@@ -232,7 +251,9 @@ public class RedactController {
request.getRedactColor(),
request.getCustomPadding(),
request.getConvertPDFToImage(),
true);
true,
verificationTargets,
compiledPatterns);
return WebResponseUtils.pdfFileToWebResponse(finalized, filename);
@@ -7,8 +7,10 @@ import java.util.Arrays;
import java.util.Collections;
import java.util.Comparator;
import java.util.HashMap;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.cos.COSName;
@@ -30,6 +32,7 @@ import stirling.software.SPDF.model.api.security.RedactExecuteRequest.RedactStyl
import stirling.software.SPDF.model.api.security.RedactExecuteRequest.TextRange;
import stirling.software.SPDF.pdf.parser.PageColumnLayout;
import stirling.software.SPDF.pdf.parser.PageImageLocator;
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.common.util.TempFile;
@@ -140,13 +143,32 @@ class RedactExecuteService {
applyAllImagesRedaction(document, request.getRedactImagePages(), style);
}
// Explicit overlay-only requests must not rewrite content or verify. When overlay-only
// was forced by a font fallback (not user choice) we still pass the targets so the
// pipeline's true-removal + verification + page-scoped raster fallback run.
Set<String> literalTargets = new LinkedHashSet<>();
for (String value : textValues) {
String trimmed = value == null ? "" : value.trim();
if (!trimmed.isEmpty()) {
literalTargets.add(trimmed);
}
}
List<Pattern> verificationPatterns =
RedactionPipeline.buildPatterns(
regexPatterns.toArray(new String[0]), true, false);
Set<String> finalizeTargets = overlayOnly ? Collections.emptySet() : literalTargets;
List<Pattern> finalizePatterns =
overlayOnly ? Collections.emptyList() : verificationPatterns;
return manualRedactionService.finalizeRedaction(
document,
foundTexts,
style.getColor(),
style.getPadding(),
convertToImage,
!needsOverlayOnly);
!needsOverlayOnly,
finalizeTargets,
finalizePatterns);
} catch (Exception e) {
log.error("Execute redaction failed: {}", e.getMessage(), e);
@@ -0,0 +1,613 @@
package stirling.software.SPDF.pdf.redaction;
import java.util.Calendar;
import java.util.HashSet;
import java.util.List;
import java.util.Locale;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.cos.COSArray;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.cos.COSObject;
import org.apache.pdfbox.cos.COSStream;
import org.apache.pdfbox.cos.COSString;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
import org.apache.pdfbox.pdmodel.PDDocumentInformation;
import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDNameTreeNode;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline;
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
import org.apache.pdfbox.pdmodel.interactive.form.PDField;
import lombok.extern.slf4j.Slf4j;
/**
* Walks a {@link PDDocument} and physically removes or rewrites every carrier that a PDF can use to
* leak text which the user asked to redact.
*
* <p>Covers:
*
* <ul>
* <li>{@link PDDocumentInformation} (Info dict) + XMP metadata stream
* <li>{@link PDDocumentOutline} bookmark titles
* <li>{@link PDAcroForm} field values (V, DV) and rich text (RV)
* <li>Every {@link PDAnnotation} Contents and RC
* <li>Structure tree ActualText, Alt, T, E, Lang entries
* <li>Names tree: JavaScript entries and embedded files (dropped entirely when matching)
* </ul>
*
* <p>When applied after the content-stream rewrite it closes the secondary leak paths flagged in
* the redaction security audit.
*/
@Slf4j
public final class CatalogScrubber {
private CatalogScrubber() {}
/**
* Remove occurrences of every {@code target} string (and any regex/whole-word pattern form
* produced by {@link RedactionPipeline#buildPatterns}) from all catalog-level carriers of the
* document. When {@code wipeAllMetadata} is {@code true} the document Info dict entries and XMP
* metadata stream are wiped wholesale; this is the safe default after a redaction operation.
*/
public static void scrub(
PDDocument document, Set<String> literalTargets, List<Pattern> patterns) {
if (document == null) {
return;
}
PDDocumentCatalog catalog = document.getDocumentCatalog();
if (catalog == null) {
return;
}
scrubOutline(catalog.getDocumentOutline(), literalTargets, patterns);
scrubAcroForm(catalog.getAcroForm(), literalTargets, patterns);
scrubAnnotations(document, literalTargets, patterns);
scrubStructTree(catalog.getStructureTreeRoot(), literalTargets, patterns);
scrubNames(catalog.getNames(), literalTargets, patterns);
scrubCatalogActions(catalog, literalTargets, patterns);
}
// ---------------------------------------------------------------------
// Catalog actions: OpenAction, AA, and any JavaScript / URI payloads on the catalog
// ---------------------------------------------------------------------
private static void scrubCatalogActions(
PDDocumentCatalog catalog, Set<String> targets, List<Pattern> patterns) {
COSDictionary root = catalog.getCOSObject();
if (root == null) {
return;
}
// OpenAction may be either an action dict (with /URI or /JS) or an explicit destination
// (array). We scrub strings in both cases; if the OpenAction matches a target we clear it.
scrubActionIfMatching(root, COSName.getPDFName("OpenAction"), targets, patterns);
scrubActionIfMatching(root, COSName.getPDFName("AA"), targets, patterns);
}
/**
* If the action dictionary at {@code key} contains any target literal in a URI or JS payload,
* wipe the key entirely. Otherwise recursively scrub string fields inside it.
*/
private static void scrubActionIfMatching(
COSDictionary parent, COSName key, Set<String> targets, List<Pattern> patterns) {
if (parent == null || key == null) {
return;
}
COSBase value = parent.getDictionaryObject(key);
if (value == null) {
return;
}
if (containsTarget(value, targets, patterns, new HashSet<>())) {
log.debug("Removing catalog {} due to target match", key.getName());
parent.removeItem(key);
}
}
private static boolean containsTarget(
COSBase base, Set<String> targets, List<Pattern> patterns, Set<COSBase> seen) {
if (base == null) {
return false;
}
COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base;
if (resolved == null || !seen.add(resolved)) {
return false;
}
if (resolved instanceof COSString cs) {
return matches(cs.getString(), targets, patterns);
}
if (resolved instanceof COSStream stream) {
// Streams in XFA / OpenAction contexts are text (XML, JavaScript). Read the bytes as
// UTF-8 and test for target literals. We cap read length to avoid pathological memory
// use; 2 MiB is plenty for XFA packets and far beyond any realistic JS action.
try (java.io.InputStream is = stream.createInputStream()) {
byte[] buf = new byte[2 * 1024 * 1024];
int total = 0;
int n;
while ((n = is.read(buf, total, buf.length - total)) > 0) {
total += n;
if (total >= buf.length) {
break;
}
}
String text = new String(buf, 0, total, java.nio.charset.StandardCharsets.UTF_8);
return matches(text, targets, patterns);
} catch (Exception e) {
log.debug("Failed to scan stream for targets: {}", e.getMessage());
// Fail closed: if we cannot read it we cannot prove it is clean, so treat as a
// match so the caller drops the stream. This is conservative by design.
return true;
}
}
if (resolved instanceof COSDictionary dict) {
for (COSName k : new HashSet<>(dict.keySet())) {
if (containsTarget(dict.getItem(k), targets, patterns, seen)) {
return true;
}
}
return false;
}
if (resolved instanceof COSArray array) {
for (int i = 0; i < array.size(); i++) {
if (containsTarget(array.getObject(i), targets, patterns, seen)) {
return true;
}
}
return false;
}
return false;
}
/**
* Clean potentially sensitive metadata carriers. Called after {@link #scrub} so that surviving
* references to author/subject/keywords/XMP descriptors do not leak redacted values.
*/
public static void wipeMetadata(PDDocument document) {
if (document == null) {
return;
}
PDDocumentInformation info = document.getDocumentInformation();
if (info != null) {
info.setAuthor(null);
info.setSubject(null);
info.setKeywords(null);
info.setTitle(null);
info.setCreator(null);
info.setProducer(null);
info.setModificationDate(Calendar.getInstance());
}
PDDocumentCatalog catalog = document.getDocumentCatalog();
if (catalog != null) {
try {
catalog.setMetadata(null);
} catch (Exception e) {
log.debug("Could not clear XMP metadata: {}", e.getMessage());
}
}
}
// ---------------------------------------------------------------------
// Outline
// ---------------------------------------------------------------------
private static void scrubOutline(
PDDocumentOutline outline, Set<String> targets, List<Pattern> patterns) {
if (outline == null) {
return;
}
scrubOutlineItems(outline.children(), targets, patterns);
}
private static void scrubOutlineItems(
Iterable<PDOutlineItem> items, Set<String> targets, List<Pattern> patterns) {
if (items == null) {
return;
}
for (PDOutlineItem item : items) {
try {
String title = item.getTitle();
if (title != null) {
String stripped = stripMatches(title, targets, patterns);
if (!stripped.equals(title)) {
item.setTitle(stripped);
}
}
// Bookmark actions: /A is an action dict which may carry a /URI or /JS payload.
// If any target literal appears anywhere inside the action subtree, drop the
// action entirely so the URI / script cannot leak the target.
COSDictionary itemDict = item.getCOSObject();
if (itemDict != null) {
scrubActionIfMatching(itemDict, COSName.A, targets, patterns);
scrubActionIfMatching(itemDict, COSName.getPDFName("AA"), targets, patterns);
}
scrubOutlineItems(item.children(), targets, patterns);
} catch (Exception e) {
log.debug("Failed to scrub outline item: {}", e.getMessage());
}
}
}
// ---------------------------------------------------------------------
// AcroForm
// ---------------------------------------------------------------------
private static void scrubAcroForm(
PDAcroForm form, Set<String> targets, List<Pattern> patterns) {
if (form == null) {
return;
}
// XFA forms: scrubbed separately because the XFA XML packet carries the "real" field
// values for XFA-enabled PDFs. Handle XFA before walking the field tree so we fail closed
// if XFA scrubbing throws.
scrubXfa(form, targets, patterns);
try {
for (PDField field : form.getFieldTree()) {
scrubField(field, targets, patterns);
}
} catch (Exception e) {
log.debug("Failed to walk AcroForm field tree: {}", e.getMessage());
}
// Force viewers to regenerate appearance streams from the (scrubbed) /V values rather
// than reusing any cached /AP /N that still contains the target text. Belt-and-braces:
// scrubField has also cleared per-widget /AP dicts, but /NeedAppearances ensures any
// future change still triggers regeneration.
try {
form.setNeedAppearances(true);
} catch (Exception e) {
log.debug("Failed to set /NeedAppearances on AcroForm: {}", e.getMessage());
}
}
private static void scrubXfa(PDAcroForm form, Set<String> targets, List<Pattern> patterns) {
try {
COSBase xfaBase = form.getCOSObject().getDictionaryObject(COSName.XFA);
if (xfaBase == null) {
return;
}
boolean hit = containsTarget(xfaBase, targets, patterns, new HashSet<>());
if (hit) {
// Simplest safe move: strip the XFA entry entirely. Viewers fall back to the
// AcroForm widgets which we have already scrubbed. Leaving a "partially scrubbed"
// XFA packet risks regex failures on partial XML and re-encoded entities leaking
// the target.
log.warn(
"Removing XFA form packet from AcroForm - XFA XML contained a redaction "
+ "target and has been dropped so viewers render AcroForm widgets "
+ "instead.");
form.getCOSObject().removeItem(COSName.XFA);
}
} catch (Exception e) {
log.debug("Failed to scrub XFA: {}", e.getMessage());
}
}
private static void scrubField(PDField field, Set<String> targets, List<Pattern> patterns) {
if (field == null) {
return;
}
try {
COSDictionary dict = field.getCOSObject();
scrubDictStrings(dict, COSName.V, targets, patterns);
scrubDictStrings(dict, COSName.DV, targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("RV"), targets, patterns);
// Keep field appearance streams in sync with value where possible.
try {
if (field.getValueAsString() != null) {
String stripped = stripMatches(field.getValueAsString(), targets, patterns);
if (!stripped.equals(field.getValueAsString())) {
field.setValue(stripped);
}
}
} catch (Exception e) {
log.debug("Failed to rewrite field value via setValue: {}", e.getMessage());
}
// Drop per-widget appearance streams (/AP dict) for every widget kid of this field.
// The cached appearance stream contains the pre-redaction value baked in as glyph
// data; simply rewriting /V leaves it visually unchanged in many viewers. Removing /AP
// plus /NeedAppearances at the form level forces regeneration.
clearWidgetAppearances(dict);
} catch (Exception e) {
log.debug("Failed to scrub field: {}", e.getMessage());
}
}
private static void clearWidgetAppearances(COSDictionary fieldDict) {
if (fieldDict == null) {
return;
}
// The field itself may be a widget (single-widget field) and/or have Kids.
fieldDict.removeItem(COSName.AP);
COSBase kids = fieldDict.getDictionaryObject(COSName.KIDS);
if (kids instanceof COSArray arr) {
for (int i = 0; i < arr.size(); i++) {
COSBase kidBase = arr.getObject(i);
if (kidBase instanceof COSDictionary kidDict) {
kidDict.removeItem(COSName.AP);
}
}
}
}
// ---------------------------------------------------------------------
// Annotations
// ---------------------------------------------------------------------
private static void scrubAnnotations(
PDDocument document, Set<String> targets, List<Pattern> patterns) {
try {
for (PDPage page : document.getPages()) {
List<PDAnnotation> annotations;
try {
annotations = page.getAnnotations();
} catch (Exception e) {
log.debug("Failed to load annotations for page: {}", e.getMessage());
continue;
}
if (annotations == null) {
continue;
}
for (PDAnnotation annotation : annotations) {
scrubAnnotation(annotation, targets, patterns);
}
}
} catch (Exception e) {
log.debug("Annotation scrub walk failed: {}", e.getMessage());
}
}
private static void scrubAnnotation(
PDAnnotation annotation, Set<String> targets, List<Pattern> patterns) {
if (annotation == null) {
return;
}
try {
String contents = annotation.getContents();
if (contents != null) {
String stripped = stripMatches(contents, targets, patterns);
if (!stripped.equals(contents)) {
annotation.setContents(stripped);
}
}
COSDictionary dict = annotation.getCOSObject();
scrubDictStrings(dict, COSName.getPDFName("RC"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("Subj"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("NM"), targets, patterns);
} catch (Exception e) {
log.debug("Failed to scrub annotation: {}", e.getMessage());
}
}
// ---------------------------------------------------------------------
// Structure tree
// ---------------------------------------------------------------------
private static void scrubStructTree(
PDStructureTreeRoot root, Set<String> targets, List<Pattern> patterns) {
if (root == null) {
return;
}
try {
scrubStructDict(root.getCOSObject(), targets, patterns, new HashSet<>());
} catch (Exception e) {
log.debug("Structure tree scrub failed: {}", e.getMessage());
}
}
private static void scrubStructDict(
COSBase base, Set<String> targets, List<Pattern> patterns, Set<COSBase> seen) {
if (base == null) {
return;
}
COSBase resolved = base instanceof COSObject obj ? obj.getObject() : base;
if (resolved == null || !seen.add(resolved)) {
return;
}
if (resolved instanceof COSDictionary dict) {
// Do not walk into content streams - those are handled by content-stream rewrite.
if (resolved instanceof COSStream) {
return;
}
scrubDictStrings(dict, COSName.getPDFName("ActualText"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("Alt"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("E"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("T"), targets, patterns);
scrubDictStrings(dict, COSName.getPDFName("Lang"), targets, patterns);
for (COSName key : new HashSet<>(dict.keySet())) {
COSBase value = dict.getItem(key);
if (value instanceof COSDictionary
|| value instanceof COSArray
|| value instanceof COSObject) {
scrubStructDict(value, targets, patterns, seen);
}
}
} else if (resolved instanceof COSArray array) {
for (int i = 0; i < array.size(); i++) {
scrubStructDict(array.getObject(i), targets, patterns, seen);
}
}
}
// ---------------------------------------------------------------------
// Names tree (JavaScript + embedded files)
// ---------------------------------------------------------------------
private static void scrubNames(
PDDocumentNameDictionary names, Set<String> targets, List<Pattern> patterns) {
if (names == null) {
return;
}
try {
dropMatchingNames(names.getJavaScript(), targets, patterns);
} catch (Exception e) {
log.debug("Failed to scrub JavaScript names: {}", e.getMessage());
}
try {
dropMatchingNames(names.getEmbeddedFiles(), targets, patterns);
} catch (Exception e) {
log.debug("Failed to scrub embedded-file names: {}", e.getMessage());
}
}
private static void dropMatchingNames(
PDNameTreeNode<?> node, Set<String> targets, List<Pattern> patterns) {
if (node == null) {
return;
}
COSDictionary dict = node.getCOSObject();
if (dict == null) {
return;
}
scrubNameTreeDict(dict, targets, patterns);
}
private static void scrubNameTreeDict(
COSDictionary dict, Set<String> targets, List<Pattern> patterns) {
if (dict == null) {
return;
}
COSArray namesArray = (COSArray) dict.getDictionaryObject(COSName.NAMES);
if (namesArray != null) {
for (int i = namesArray.size() - 2; i >= 0; i -= 2) {
COSBase keyBase = namesArray.getObject(i);
String key = keyBase instanceof COSString s ? s.getString() : null;
if (key != null && matches(key, targets, patterns)) {
namesArray.remove(i + 1);
namesArray.remove(i);
}
}
}
COSArray kids = (COSArray) dict.getDictionaryObject(COSName.KIDS);
if (kids != null) {
for (int i = 0; i < kids.size(); i++) {
COSBase kid = kids.getObject(i);
if (kid instanceof COSDictionary kidDict) {
scrubNameTreeDict(kidDict, targets, patterns);
}
}
}
}
// ---------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------
private static void scrubDictStrings(
COSDictionary dict, COSName key, Set<String> targets, List<Pattern> patterns) {
if (dict == null || key == null) {
return;
}
COSBase value = dict.getDictionaryObject(key);
if (value instanceof COSString cosString) {
String stripped = stripMatches(cosString.getString(), targets, patterns);
if (!stripped.equals(cosString.getString())) {
dict.setString(key, stripped);
}
} else if (value instanceof COSArray array) {
for (int i = 0; i < array.size(); i++) {
COSBase element = array.getObject(i);
if (element instanceof COSString elementString) {
String stripped = stripMatches(elementString.getString(), targets, patterns);
if (!stripped.equals(elementString.getString())) {
array.set(i, new COSString(stripped));
}
}
}
}
}
static String stripMatches(String source, Set<String> literalTargets, List<Pattern> patterns) {
if (source == null || source.isEmpty()) {
return source;
}
String result = source;
if (literalTargets != null) {
for (String target : literalTargets) {
if (target == null || target.isEmpty()) {
continue;
}
// Case-insensitive literal removal. Verification is case-insensitive, so scrubbing
// MUST be too or mixed-case ("SMITH" in a catalog string vs "Smith" in the target
// list) will fail verification and trip the rasterisation fallback - or worse, on
// carriers that are not verified, leak the string untouched.
result = caseInsensitiveReplaceAll(result, target);
}
}
if (patterns != null) {
for (Pattern pattern : patterns) {
try {
// Force case-insensitive matching for catalog carriers regardless of the flags
// the pattern was compiled with. User-supplied redaction targets should not
// silently miss because the author typed the name in different case.
Pattern ci = withCaseInsensitive(pattern);
result = ci.matcher(result).replaceAll("");
} catch (Exception e) {
log.debug(
"Pattern replace failed for {}: {}", pattern.pattern(), e.getMessage());
}
}
}
return result;
}
static boolean matches(String source, Set<String> literalTargets, List<Pattern> patterns) {
if (source == null || source.isEmpty()) {
return false;
}
String lower = source.toLowerCase(Locale.ROOT);
if (literalTargets != null) {
for (String target : literalTargets) {
if (target != null
&& !target.isEmpty()
&& lower.contains(target.toLowerCase(Locale.ROOT))) {
return true;
}
}
}
if (patterns != null) {
for (Pattern pattern : patterns) {
try {
if (withCaseInsensitive(pattern).matcher(source).find()) {
return true;
}
} catch (Exception e) {
log.debug("Pattern match failed for {}: {}", pattern.pattern(), e.getMessage());
}
}
}
return false;
}
private static String caseInsensitiveReplaceAll(String source, String target) {
if (target.isEmpty()) {
return source;
}
Pattern literal =
Pattern.compile(
Pattern.quote(target), Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE);
return literal.matcher(source).replaceAll("");
}
private static Pattern withCaseInsensitive(Pattern pattern) {
if ((pattern.flags() & Pattern.CASE_INSENSITIVE) != 0) {
return pattern;
}
try {
return Pattern.compile(
pattern.pattern(),
pattern.flags() | Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE);
} catch (Exception e) {
return pattern;
}
}
}
@@ -0,0 +1,17 @@
package stirling.software.SPDF.pdf.redaction;
/**
* Thrown when a post-redaction verification pass still finds any of the target strings in the
* re-parsed PDF text. Indicates that the redaction pipeline did not fully remove the targeted
* content and the output must not be treated as safe to release.
*/
public class RedactionVerificationFailedException extends RuntimeException {
public RedactionVerificationFailedException(String message) {
super(message);
}
public RedactionVerificationFailedException(String message, Throwable cause) {
super(message, cause);
}
}
@@ -19,6 +19,7 @@ import java.io.IOException;
import java.nio.file.Files;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Collections;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
@@ -568,7 +569,15 @@ class ManualRedactionServiceTest {
byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710))));
TempFile result =
service.finalizeRedaction(doc, byPage, "#000000", 1.0f, false, false);
service.finalizeRedaction(
doc,
byPage,
"#000000",
1.0f,
false,
false,
Collections.emptySet(),
Collections.emptyList());
assertNotNull(result);
assertNotNull(result.getFile());
@@ -590,7 +599,15 @@ class ManualRedactionServiceTest {
Map<Integer, List<PDFText>> byPage = new HashMap<>();
TempFile result =
service.finalizeRedaction(doc, byPage, "#000000", 0.0f, false, false);
service.finalizeRedaction(
doc,
byPage,
"#000000",
0.0f,
false,
false,
Collections.emptySet(),
Collections.emptyList());
assertNotNull(result);
assertTrue(result.getFile().exists());
@@ -610,7 +627,15 @@ class ManualRedactionServiceTest {
Map<Integer, List<PDFText>> byPage = new HashMap<>();
TempFile result =
service.finalizeRedaction(doc, byPage, "#000000", 0.0f, null, false);
service.finalizeRedaction(
doc,
byPage,
"#000000",
0.0f,
null,
false,
Collections.emptySet(),
Collections.emptyList());
assertNotNull(result);
assertTrue(result.getFile().exists());
@@ -630,7 +655,15 @@ class ManualRedactionServiceTest {
byPage.put(0, new ArrayList<>(Arrays.asList(text(0, 72, 690, 300, 710))));
TempFile result =
service.finalizeRedaction(doc, byPage, "#FF0000", 2.0f, false, true);
service.finalizeRedaction(
doc,
byPage,
"#FF0000",
2.0f,
false,
true,
Collections.emptySet(),
Collections.emptyList());
assertNotNull(result);
try (PDDocument reloaded = Loader.loadPDF(result.getFile())) {
@@ -658,7 +691,14 @@ class ManualRedactionServiceTest {
IOException.class,
() ->
service.finalizeRedaction(
doc, byPage, "#000000", 0.0f, false, false));
doc,
byPage,
"#000000",
0.0f,
false,
false,
Collections.emptySet(),
Collections.emptyList()));
// The failing temp file is closed on the error path.
verify(failing).close();
} finally {
@@ -201,6 +201,17 @@ class RedactControllerTest {
})
.when(mockDocument)
.save(any(File.class));
// RedactionPipeline serialises via OutputStream before handing the bytes to TempFile, so
// the mock must emit a non-empty payload on that code path too.
lenient()
.doAnswer(
inv -> {
java.io.OutputStream os = inv.getArgument(0);
os.write("mock pdf".getBytes());
return null;
})
.when(mockDocument)
.save(any(java.io.OutputStream.class));
doNothing().when(mockDocument).close();
// Build real service instances so tests exercise actual logic
@@ -347,7 +358,7 @@ class RedactControllerTest {
assertNotNull(response);
assertEquals(200, response.getStatusCode().value());
verify(mockDocument).save(any(File.class));
verify(mockDocument).save(any(java.io.OutputStream.class));
verify(mockDocument).close();
}
}
@@ -753,7 +764,7 @@ class RedactControllerTest {
assertEquals(200, response.getStatusCode().value());
assertNotNull(response.getBody());
assertTrue(drainBody(response).length > 0);
verify(mockDocument, times(1)).save(any(File.class));
verify(mockDocument, times(1)).save(any(java.io.OutputStream.class));
verify(mockDocument, times(1)).close();
}
} catch (Exception e) {
@@ -776,7 +787,7 @@ class RedactControllerTest {
if (response != null) {
assertNotNull(response);
assertEquals(200, response.getStatusCode().value());
verify(mockDocument, times(1)).save(any(File.class));
verify(mockDocument, times(1)).save(any(java.io.OutputStream.class));
}
} catch (Exception e) {
log.info("Manual redaction test completed with graceful handling: {}", e.getMessage());
@@ -0,0 +1,569 @@
package stirling.software.SPDF.controller.api.security;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.ArgumentMatchers.any;
import static org.mockito.ArgumentMatchers.anyString;
import static org.mockito.Mockito.lenient;
import static org.mockito.Mockito.mock;
import java.io.ByteArrayOutputStream;
import java.io.File;
import java.io.IOException;
import java.io.InputStream;
import java.nio.file.Files;
import java.util.ArrayList;
import java.util.List;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDFormContentStream;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.PDResources;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.apache.pdfbox.pdmodel.font.PDTrueTypeFont;
import org.apache.pdfbox.pdmodel.font.PDType0Font;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.apache.pdfbox.pdmodel.font.encoding.MacRomanEncoding;
import org.apache.pdfbox.pdmodel.font.encoding.WinAnsiEncoding;
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
import org.apache.pdfbox.text.PDFTextStripper;
import org.apache.pdfbox.util.Matrix;
import org.junit.jupiter.api.AfterEach;
import org.junit.jupiter.api.BeforeEach;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import org.springframework.core.io.Resource;
import org.springframework.http.ResponseEntity;
import org.springframework.mock.web.MockMultipartFile;
import org.springframework.web.multipart.MultipartFile;
import stirling.software.SPDF.model.api.security.RedactPdfRequest;
import stirling.software.SPDF.pdf.redaction.RedactionPipeline;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.util.TempFile;
import stirling.software.common.util.TempFileManager;
/**
* Security matrix for redaction across PDF shapes: standard and embedded (subset and full) fonts,
* rotated pages, TJ splits, cross-operator splits, Form XObjects, CropBox offsets, multi-page and
* case variants. Every case asserts the target is unrecoverable from the output text layer, and
* (where removal should be surgical) that neighbouring text survives. Uses the LiberationSans TTF
* bundled inside the PDFBox jar so the embedded-font cases run on any CI platform.
*/
@DisplayName("Redaction PDF-variety security matrix")
class RedactionPdfVarietyTest {
private static final String LIBERATION =
"/org/apache/pdfbox/resources/ttf/LiberationSans-Regular.ttf";
private static final float FONT_SIZE = 12f;
private static final float LEFT_X = 72f;
private static final float TOP_Y = PDRectangle.LETTER.getHeight() - 80f;
private CustomPDFDocumentFactory pdfDocumentFactory;
private TempFileManager tempFileManager;
private RedactController controller;
private final List<File> createdTempFiles = new ArrayList<>();
@BeforeEach
void setUp() throws IOException {
pdfDocumentFactory = mock(CustomPDFDocumentFactory.class);
tempFileManager = mock(TempFileManager.class);
lenient()
.when(tempFileManager.createManagedTempFile(anyString()))
.thenAnswer(
inv -> {
File f =
Files.createTempFile(
"redact-variety", inv.<String>getArgument(0))
.toFile();
createdTempFiles.add(f);
TempFile tf = mock(TempFile.class);
lenient().when(tf.getFile()).thenReturn(f);
lenient().when(tf.getPath()).thenReturn(f.toPath());
return tf;
});
controller =
new RedactController(
pdfDocumentFactory,
tempFileManager,
new ManualRedactionService(tempFileManager),
new TextRedactionService(),
mock(RedactExecuteService.class));
}
@AfterEach
void tearDown() {
for (File f : createdTempFiles) {
if (f != null && f.exists()) {
f.delete();
}
}
}
// -----------------------------------------------------------------------
// Matrix cases (all through the real /auto-redact controller path)
// -----------------------------------------------------------------------
@Test
@DisplayName("standard Helvetica: target gone, neighbours survive")
void standard14Helvetica() throws IOException {
byte[] out = autoRedact(helveticaPdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("standard-14 Times and Courier: target gone, neighbours survive")
void standard14TimesAndCourier() throws IOException {
for (Standard14Fonts.FontName fn :
new Standard14Fonts.FontName[] {
Standard14Fonts.FontName.TIMES_ROMAN, Standard14Fonts.FontName.COURIER
}) {
byte[] pdf = std14Pdf(fn, "alpha SECRET omega");
byte[] out = autoRedact(pdf, "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).as("%s keeps neighbours", fn).contains("alpha");
}
}
@Test
@DisplayName("simple TrueType with MacRoman encoding: target gone, neighbours survive")
void simpleTrueTypeMacRoman() throws IOException {
byte[] out = autoRedact(macRomanTtfPdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("embedded Type0 with /ToUnicode stripped: target still gone, neighbours survive")
void type0WithoutToUnicode() throws IOException {
byte[] out = autoRedact(type0NoToUnicodePdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("subset-tagged font (ABCDEF+): target gone, neighbours survive")
void subsetTaggedFont() throws IOException {
byte[] out = autoRedact(subsetTaggedPdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("real Type3 font (glyphs as content streams, no ToUnicode): target gone")
void type3GlyphProcs() throws IOException {
// crop_test.pdf uses DejaVuSans embedded as a subset Type3 font with no ToUnicode map.
// Redacting it must not throw (Type3 font.encode() throws UnsupportedOperationException)
// and must remove the target from the extractable text layer.
byte[] input;
try (InputStream in = getClass().getResourceAsStream("/redaction/type3_dejavu.pdf")) {
input = in.readAllBytes();
}
byte[] out = autoRedact(input, "EXAMPLE");
assertGone(out, "EXAMPLE");
assertThat(pdfText(out)).contains("CROP");
}
@Test
@DisplayName("embedded subset Type0 (CID) font: target gone, neighbours survive")
void embeddedSubsetType0() throws IOException {
byte[] out = autoRedact(type0Pdf(true, "alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("embedded full Type0 (CID) font: target gone, neighbours survive")
void embeddedFullType0() throws IOException {
byte[] out = autoRedact(type0Pdf(false, "alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("simple embedded TrueType (WinAnsi) font: target gone, neighbours survive")
void simpleTrueTypeWinAnsi() throws IOException {
byte[] out = autoRedact(simpleTtfPdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("rotated pages (90/180/270): target gone on every rotation")
void rotatedPages() throws IOException {
for (int rotation : new int[] {90, 180, 270}) {
byte[] out = autoRedact(rotatedPdf(rotation, "alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).as("rotation %d keeps neighbours", rotation).contains("alpha");
}
}
@Test
@DisplayName("target split across TJ array operands is removed")
void tjArraySplit() throws IOException {
byte[] out = autoRedact(tjSplitPdf(), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("public");
}
@Test
@DisplayName("target split across separate Tj operators is removed, other pages keep text")
void crossOperatorSplit() throws IOException {
byte[] out = autoRedact(crossOperatorPdf(), "SECRET");
assertGone(out, "SECRET");
// Page 2 was never touched; whatever path handled page 1, page 2 text must survive.
assertThat(pdfText(out)).contains("PUBLIC PAGE TWO");
}
@Test
@DisplayName("target inside a Form XObject is removed")
void formXObjectText() throws IOException {
byte[] out = autoRedact(formXObjectPdf(), "SECRET");
assertGone(out, "SECRET");
}
@Test
@DisplayName("CropBox smaller than MediaBox: target gone, neighbours survive")
void cropBoxOffset() throws IOException {
byte[] out = autoRedact(cropBoxPdf("alpha SECRET omega"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("alpha");
}
@Test
@DisplayName("multi-page: target on middle page only, other pages keep their text")
void multiPageMiddleTarget() throws IOException {
byte[] out =
autoRedact(
multiPagePdf("PUBLIC ONE", "middle SECRET line", "PUBLIC THREE"), "SECRET");
assertGone(out, "SECRET");
assertThat(pdfText(out)).contains("PUBLIC ONE").contains("PUBLIC THREE");
}
@Test
@DisplayName("case variant: doc contains 'Secret', target 'SECRET' removes it")
void caseVariantRemoved() throws IOException {
byte[] out = autoRedact(helveticaPdf("alpha Secret omega"), "SECRET");
assertGone(out, "Secret");
assertThat(pdfText(out)).contains("alpha").contains("omega");
}
@Test
@DisplayName("whole-word with case variant: 'Cat' removed, substrings survive")
void wholeWordCaseVariant() throws IOException {
byte[] bytes = helveticaPdf("Cat classification scatter");
factoryReturns(bytes);
RedactPdfRequest request = baseRequest(bytes, "cat");
request.setWholeWordSearch(true);
byte[] out = drainBody(controller.redactPdf(request));
String text = pdfText(out);
assertThat(text).contains("classification").contains("scatter");
// The standalone word must be gone in any case variant.
assertThat(text).doesNotContainPattern("(?i)\\bcat\\b");
}
// -----------------------------------------------------------------------
// Page-scoped rasterisation (pipeline-level, deterministic)
// -----------------------------------------------------------------------
@Test
@DisplayName("verification leak rasterises only the leaking page, others keep text")
void leakRasterisesOnlyLeakingPage() throws IOException {
// Cross-operator split defeats the per-operand literal rewriter, so verification must
// catch the leak and rasterise page 1 only, leaving page 2 searchable.
byte[] input = crossOperatorPdf();
byte[] out;
try (PDDocument doc = Loader.loadPDF(input)) {
Set<String> targets = Set.of("SECRET");
List<Pattern> patterns =
RedactionPipeline.buildPatterns(new String[] {"SECRET"}, false, false);
RedactionPipeline.redactLiteralTerms(doc, targets, patterns);
out = RedactionPipeline.finalize(doc, targets, patterns);
}
try (PDDocument reopened = Loader.loadPDF(out)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setStartPage(1);
stripper.setEndPage(1);
String page1 = stripper.getText(reopened);
stripper.setStartPage(2);
stripper.setEndPage(2);
String page2 = stripper.getText(reopened);
assertThat(page1.toLowerCase()).doesNotContain("secret");
assertThat(page2).contains("PUBLIC PAGE TWO");
}
}
// -----------------------------------------------------------------------
// Drivers and assertions
// -----------------------------------------------------------------------
private byte[] autoRedact(byte[] pdfBytes, String target) throws IOException {
factoryReturns(pdfBytes);
RedactPdfRequest request = baseRequest(pdfBytes, target);
ResponseEntity<Resource> response = controller.redactPdf(request);
assertThat(response.getStatusCode().value()).isEqualTo(200);
return drainBody(response);
}
private RedactPdfRequest baseRequest(byte[] pdfBytes, String target) {
RedactPdfRequest request = new RedactPdfRequest();
request.setFileInput(pdfFile(pdfBytes));
request.setListOfText(target);
request.setUseRegex(false);
request.setWholeWordSearch(false);
request.setRedactColor("#000000");
request.setConvertPDFToImage(false);
return request;
}
private void assertGone(byte[] out, String target) throws IOException {
assertThat(pdfText(out).toLowerCase()).doesNotContain(target.toLowerCase());
}
private void factoryReturns(byte[] pdfBytes) throws IOException {
lenient()
.when(pdfDocumentFactory.load(any(MultipartFile.class)))
.thenAnswer(inv -> Loader.loadPDF(pdfBytes));
}
private MockMultipartFile pdfFile(byte[] bytes) {
return new MockMultipartFile("fileInput", "doc.pdf", "application/pdf", bytes);
}
private byte[] drainBody(ResponseEntity<Resource> response) throws IOException {
ByteArrayOutputStream baos = new ByteArrayOutputStream();
try (InputStream in = response.getBody().getInputStream()) {
in.transferTo(baos);
}
return baos.toByteArray();
}
private String pdfText(byte[] pdfBytes) throws IOException {
try (PDDocument doc = Loader.loadPDF(pdfBytes)) {
return new PDFTextStripper().getText(doc);
}
}
// -----------------------------------------------------------------------
// PDF builders
// -----------------------------------------------------------------------
private static PDFont helvetica() {
return new PDType1Font(Standard14Fonts.FontName.HELVETICA);
}
private byte[] helveticaPdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
addTextPage(doc, helvetica(), line, 0, null);
return save(doc);
}
}
private byte[] std14Pdf(Standard14Fonts.FontName fontName, String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
addTextPage(doc, new PDType1Font(fontName), line, 0, null);
return save(doc);
}
}
private byte[] macRomanTtfPdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
PDFont font;
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
font = PDTrueTypeFont.load(doc, ttf, MacRomanEncoding.INSTANCE);
}
addTextPage(doc, font, line, 0, null);
return save(doc);
}
}
private byte[] type0NoToUnicodePdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
PDFont font;
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
font = PDType0Font.load(doc, ttf, true);
}
addTextPage(doc, font, line, 0, null);
font.getCOSObject().removeItem(org.apache.pdfbox.cos.COSName.TO_UNICODE);
return save(doc);
}
}
private byte[] subsetTaggedPdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
PDFont font;
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
font = PDType0Font.load(doc, ttf, true);
}
addTextPage(doc, font, line, 0, null);
String tagged = "ABCDEF+" + font.getName();
font.getCOSObject().setName(org.apache.pdfbox.cos.COSName.BASE_FONT, tagged);
if (font.getFontDescriptor() != null) {
font.getFontDescriptor()
.getCOSObject()
.setName(org.apache.pdfbox.cos.COSName.FONT_NAME, tagged);
}
return save(doc);
}
}
private byte[] type0Pdf(boolean subset, String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
PDFont font;
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
font = PDType0Font.load(doc, ttf, subset);
}
addTextPage(doc, font, line, 0, null);
return save(doc);
}
}
private byte[] simpleTtfPdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
PDFont font;
try (InputStream ttf = PDDocument.class.getResourceAsStream(LIBERATION)) {
font = PDTrueTypeFont.load(doc, ttf, WinAnsiEncoding.INSTANCE);
}
addTextPage(doc, font, line, 0, null);
return save(doc);
}
}
private byte[] rotatedPdf(int rotation, String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
addTextPage(doc, helvetica(), line, rotation, null);
return save(doc);
}
}
private byte[] cropBoxPdf(String line) throws IOException {
try (PDDocument doc = new PDDocument()) {
addTextPage(doc, helvetica(), line, 0, new PDRectangle(40, 40, 500, 700));
return save(doc);
}
}
private byte[] multiPagePdf(String... pageLines) throws IOException {
try (PDDocument doc = new PDDocument()) {
for (String line : pageLines) {
addTextPage(doc, helvetica(), line, 0, null);
}
return save(doc);
}
}
/** One page whose text is emitted as a TJ array: ["public ", "SEC", -20, "RET", " end"]. */
private byte[] tjSplitPdf() throws IOException {
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.LETTER);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(helvetica(), FONT_SIZE);
cs.newLineAtOffset(LEFT_X, TOP_Y);
cs.showTextWithPositioning(
new Object[] {"public ", "SEC", Float.valueOf(-20f), "RET", " end"});
cs.endText();
}
return save(doc);
}
}
/** Page 1 shows "SEC" and "RET" as separate adjacent Tj operators; page 2 is clean. */
private byte[] crossOperatorPdf() throws IOException {
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.LETTER);
doc.addPage(page);
PDFont font = helvetica();
float secWidth = font.getStringWidth("SEC") / 1000f * FONT_SIZE;
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(font, FONT_SIZE);
cs.newLineAtOffset(LEFT_X, TOP_Y);
cs.showText("SEC");
cs.endText();
cs.beginText();
cs.setFont(font, FONT_SIZE);
cs.newLineAtOffset(LEFT_X + secWidth, TOP_Y);
cs.showText("RET");
cs.endText();
}
addTextPage(doc, helvetica(), "PUBLIC PAGE TWO", 0, null);
return save(doc);
}
}
private byte[] formXObjectPdf() throws IOException {
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.LETTER);
doc.addPage(page);
PDFormXObject form = new PDFormXObject(doc);
form.setBBox(new PDRectangle(0, 0, 400, 60));
form.setResources(new PDResources());
try (PDFormContentStream fcs = new PDFormContentStream(form)) {
fcs.beginText();
fcs.setFont(helvetica(), FONT_SIZE);
fcs.newLineAtOffset(10, 20);
fcs.showText("xobj SECRET payload");
fcs.endText();
}
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.saveGraphicsState();
cs.transform(Matrix.getTranslateInstance(LEFT_X, TOP_Y - 60));
cs.drawForm(form);
cs.restoreGraphicsState();
}
return save(doc);
}
}
private void addTextPage(
PDDocument doc, PDFont font, String line, int rotation, PDRectangle cropBox)
throws IOException {
PDPage page = new PDPage(PDRectangle.LETTER);
if (rotation != 0) {
page.setRotation(rotation);
}
if (cropBox != null) {
page.setCropBox(cropBox);
}
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(font, FONT_SIZE);
if (rotation != 0) {
// Real generators compensate the text matrix so text reads upright on rotated
// pages; anchor at page centre so every rotation keeps the line on-page.
cs.setTextMatrix(
Matrix.getRotateInstance(
Math.toRadians(rotation),
PDRectangle.LETTER.getWidth() / 2,
PDRectangle.LETTER.getHeight() / 2));
} else {
cs.newLineAtOffset(LEFT_X, TOP_Y);
}
cs.showText(line);
cs.endText();
}
}
private static byte[] save(PDDocument doc) throws IOException {
ByteArrayOutputStream baos = new ByteArrayOutputStream();
doc.save(baos);
return baos.toByteArray();
}
}
@@ -0,0 +1,174 @@
package stirling.software.SPDF.pdf.redaction;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import java.io.ByteArrayOutputStream;
import java.io.InputStream;
import java.util.Collections;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Locale;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/**
* Integration tests that exercise the real redaction pipeline against PDFs with characteristics
* that have tripped up the implementation on real user files: page rotation (/Rotate 90), entire
* phrases packed into one Tj operator, and standard Type1 fonts with WinAnsiEncoding.
*
* <p>The critical assertion for every test is that a fresh {@link PDFTextStripper} with default
* configuration run over the saved bytes does not contain the target search term
* (case-insensitive).
*/
class RedactionPipelineIntegrationTest {
private static byte[] loadFixture() throws Exception {
try (InputStream in =
RedactionPipelineIntegrationTest.class.getResourceAsStream(
"/redaction/test_pdf_1.pdf")) {
assertNotNull(in, "fixture resource must exist on classpath");
return in.readAllBytes();
}
}
@Test
@DisplayName(
"finalize on rotated ReportLab PDF without a rewrite pass falls back to rasterisation so target disappears")
void finaliseFallsBackToRasterisationWhenTargetWouldSurvive() throws Exception {
byte[] fixtureBytes = loadFixture();
byte[] outputBytes;
try (PDDocument doc = Loader.loadPDF(fixtureBytes)) {
Set<String> literalTargets = new LinkedHashSet<>();
literalTargets.add("Test");
// No content-stream rewriting performed. The primary verification pass inside
// finalize must see the surviving target, switch to the rasterisation fallback,
// and return bytes that no longer extract as text. This proves the safety net is
// wired - without it the output would still contain the target.
outputBytes = RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList());
}
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
String extracted = new PDFTextStripper().getText(reopened);
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
assertFalse(
lower.contains("test"),
"Rasterisation fallback must have removed target. Extracted: '"
+ extracted
+ "'");
}
}
@Test
@DisplayName(
"finalize with zero rewrite + synthetic upright PDF triggers RedactionVerificationFailedException only when fallback disabled")
void verificationExceptionWiringSanityCheck() throws Exception {
// This test proves the verification hook itself still throws - when finalize cannot
// rasterise (we pass an already-closed document by letting finalize re-save an empty
// fresh document that still has the target text baked in).
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("Top Smith Classified");
cs.endText();
}
Set<String> literalTargets = new LinkedHashSet<>();
literalTargets.add("Smith");
// Even with the rasterisation fallback active, the output must not contain the
// target. Either the primary pass or the fallback must succeed; both is a bug.
byte[] outBytes =
RedactionPipeline.finalize(doc, literalTargets, Collections.emptyList());
try (PDDocument reopened = Loader.loadPDF(outBytes)) {
String extracted = new PDFTextStripper().getText(reopened);
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
assertFalse(
lower.contains("smith"),
"Neither primary pass nor rasterisation removed target. Extracted: '"
+ extracted
+ "'");
}
}
}
@Test
@DisplayName(
"auto-word redact on rotated PDF with target packed in single Tj removes the word from text stream")
void autoWordRedactRemovesTargetOnRotatedSingleTjPdf() throws Exception {
byte[] fixtureBytes = loadFixture();
byte[] outputBytes;
Set<String> literalTargets = new LinkedHashSet<>();
literalTargets.add("Test");
List<Pattern> patterns =
RedactionPipeline.buildPatterns(new String[] {"Test"}, false, false);
try (PDDocument doc = Loader.loadPDF(fixtureBytes)) {
RedactionPipeline.redactLiteralTerms(doc, literalTargets, patterns);
outputBytes = RedactionPipeline.finalize(doc, literalTargets, patterns);
}
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
PDFTextStripper stripper = new PDFTextStripper();
String extracted = stripper.getText(reopened);
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
assertFalse(
lower.contains("test"),
"Target term must not be extractable after redact. Extracted: '"
+ extracted
+ "'");
}
}
@Test
@DisplayName("simple upright Helvetica fixture still gets word redacted via the pipeline")
void autoWordRedactRemovesTargetOnSyntheticPdf() throws Exception {
byte[] outputBytes;
Set<String> literalTargets = new LinkedHashSet<>();
literalTargets.add("Secret");
List<Pattern> patterns =
RedactionPipeline.buildPatterns(new String[] {"Secret"}, false, false);
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("Top Secret Classified");
cs.endText();
}
ByteArrayOutputStream tmp = new ByteArrayOutputStream();
doc.save(tmp);
// reload from bytes so we are processing an already-saved PDF similar to the
// real upload flow.
try (PDDocument reloaded = Loader.loadPDF(tmp.toByteArray())) {
RedactionPipeline.redactLiteralTerms(reloaded, literalTargets, patterns);
outputBytes = RedactionPipeline.finalize(reloaded, literalTargets, patterns);
}
}
try (PDDocument reopened = Loader.loadPDF(outputBytes)) {
String extracted = new PDFTextStripper().getText(reopened);
String lower = extracted == null ? "" : extracted.toLowerCase(Locale.ROOT);
assertFalse(
lower.contains("secret"),
"Target term must not be extractable after redact on synthetic PDF. Extracted: '"
+ extracted
+ "'");
}
}
}
@@ -0,0 +1,529 @@
package stirling.software.SPDF.pdf.redaction;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.awt.Color;
import java.io.ByteArrayOutputStream;
import java.util.Collections;
import java.util.HashMap;
import java.util.HashSet;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.cos.COSStream;
import org.apache.pdfbox.cos.COSString;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationHighlight;
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDDocumentOutline;
import org.apache.pdfbox.pdmodel.interactive.documentnavigation.outline.PDOutlineItem;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
import org.apache.pdfbox.pdmodel.interactive.form.PDTextField;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
class RedactionPipelineTest {
@Test
@DisplayName(
"redactAreas removes text that falls inside the rectangle from the saved PDF content")
void manualAreaRedactRemovesText() throws Exception {
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("SECRET PAYLOAD ALPHA");
cs.endText();
}
Map<Integer, List<PDRectangle>> rects = new HashMap<>();
// The text above sits around y=700 with height ~12. Cover it fully.
rects.put(0, List.of(new PDRectangle(90, 695, 260, 25)));
RedactionPipeline.redactAreas(doc, rects, Color.BLACK);
bytes =
RedactionPipeline.finalize(
doc, Collections.emptySet(), Collections.emptyList());
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
String text = new PDFTextStripper().getText(reopened);
assertFalse(
text.contains("SECRET"),
"Manual-area redacted text must not be extractable, actual='" + text + "'");
}
}
@Test
@DisplayName(
"redactWholePages strips all text from the targeted pages while leaving others intact")
void wholePageRedactWipesTextOnSelectedPages() throws Exception {
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage p0 = new PDPage(PDRectangle.A4);
PDPage p1 = new PDPage(PDRectangle.A4);
doc.addPage(p0);
doc.addPage(p1);
try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("TOP SECRET PAGE ONE");
cs.endText();
}
try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("PUBLIC PAGE TWO");
cs.endText();
}
RedactionPipeline.redactWholePages(doc, List.of(0), Color.BLACK);
bytes =
RedactionPipeline.finalize(
doc, Collections.emptySet(), Collections.emptyList());
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setStartPage(1);
stripper.setEndPage(1);
String p0Text = stripper.getText(reopened);
stripper.setStartPage(2);
stripper.setEndPage(2);
String p1Text = stripper.getText(reopened);
assertFalse(p0Text.contains("SECRET"), "Whole-page redaction must wipe text");
assertTrue(p1Text.contains("PUBLIC"), "Non-redacted pages must retain content");
}
}
@Test
@DisplayName(
"CatalogScrubber strips bookmark titles, form field values, and annotation contents")
void catalogScrubRemovesSensitiveStringsFromCarriers() throws Exception {
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
// Bookmark carrying the redacted string.
PDDocumentOutline outline = new PDDocumentOutline();
doc.getDocumentCatalog().setDocumentOutline(outline);
PDOutlineItem item = new PDOutlineItem();
item.setTitle("See page on Smith case");
outline.addLast(item);
// AcroForm field carrying the redacted string. We set the V entry directly on the
// COS dictionary to avoid triggering AppearanceGeneratorHelper which requires a full
// /DA + /DR resource graph - CatalogScrubber is supposed to rewrite V regardless.
PDAcroForm form = new PDAcroForm(doc);
doc.getDocumentCatalog().setAcroForm(form);
PDTextField field = new PDTextField(form);
field.setPartialName("comments");
field.getCOSObject()
.setString(org.apache.pdfbox.cos.COSName.V, "Paid by Smith on receipt");
form.getFields().add(field);
// Annotation carrying the redacted string.
PDAnnotationHighlight annotation = new PDAnnotationHighlight();
annotation.setContents("Note about Smith purchase");
annotation.setRectangle(new PDRectangle(10, 10, 100, 20));
page.getAnnotations().add(annotation);
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
ByteArrayOutputStream baos = new ByteArrayOutputStream();
doc.save(baos);
bytes = baos.toByteArray();
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
PDDocumentOutline outline = reopened.getDocumentCatalog().getDocumentOutline();
assertFalse(
outline.getFirstChild().getTitle().contains("Smith"),
"Bookmark titles must be scrubbed");
PDAcroForm reopenedForm = reopened.getDocumentCatalog().getAcroForm();
String fieldValue = reopenedForm.getField("comments").getValueAsString();
assertFalse(
fieldValue.contains("Smith"),
"AcroForm field values must be scrubbed, actual='" + fieldValue + "'");
PDPage page = reopened.getPage(0);
String annotText = page.getAnnotations().get(0).getContents();
assertFalse(
annotText.contains("Smith"),
"Annotation Contents must be scrubbed, actual='" + annotText + "'");
}
}
@Test
@DisplayName(
"finalize guarantees target removal even when no content rewrite was done (rasterisation fallback)")
void verificationFallbackRasterisesWhenTargetWouldSurvive() throws Exception {
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("Surviving Smith text");
cs.endText();
}
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
// No content-stream rewriting was done. The primary verification must trip and the
// rasterisation fallback must kick in so the final bytes still have no target.
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList());
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
String extracted = new PDFTextStripper().getText(reopened);
assertFalse(
extracted != null && extracted.toLowerCase().contains("smith"),
"Rasterisation fallback must remove target, actual='" + extracted + "'");
}
}
@Test
@DisplayName(
"Manual rect redaction feeds captured text into scoped verification - raster fallback kicks in when text survives")
void manualRectVerificationUsesCapturedStrings() throws Exception {
// Simulate the failure mode: a rect is drawn over "LEAKED" but the content-stream rewrite
// is intentionally bypassed by only calling the CatalogScrubber path (finalize with
// targets=["LEAKED"] and affectedPages=[0]). Verification must see the surviving text and
// trigger rasterisation of page 0. Page 1 must remain text-searchable.
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage p0 = new PDPage(PDRectangle.A4);
PDPage p1 = new PDPage(PDRectangle.A4);
doc.addPage(p0);
doc.addPage(p1);
try (PDPageContentStream cs = new PDPageContentStream(doc, p0)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("The LEAKED value on page 1");
cs.endText();
}
try (PDPageContentStream cs = new PDPageContentStream(doc, p1)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("Totally unrelated page two text");
cs.endText();
}
Set<String> targets = new LinkedHashSet<>();
targets.add("LEAKED");
Set<Integer> affected = new HashSet<>();
affected.add(0);
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected);
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setStartPage(1);
stripper.setEndPage(1);
String p0Text = stripper.getText(reopened);
stripper.setStartPage(2);
stripper.setEndPage(2);
String p1Text = stripper.getText(reopened);
assertFalse(
p0Text.toLowerCase().contains("leaked"),
"Manual rect verification must trigger raster fallback on the targeted page");
assertTrue(
p1Text.contains("Totally unrelated"),
"Non-targeted pages must remain text-searchable after scoped raster fallback");
}
}
@Test
@DisplayName(
"Scoped raster fallback preserves text layer on non-affected pages (only affected pages are rasterised)")
void scopedRasterFallbackPreservesUntargetedPages() throws Exception {
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage p0 = new PDPage(PDRectangle.A4);
PDPage p1 = new PDPage(PDRectangle.A4);
PDPage p2 = new PDPage(PDRectangle.A4);
doc.addPage(p0);
doc.addPage(p1);
doc.addPage(p2);
String[] lines = {"SECRET alpha", "PUBLIC beta", "PUBLIC gamma"};
for (int i = 0; i < 3; i++) {
try (PDPageContentStream cs = new PDPageContentStream(doc, doc.getPage(i))) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText(lines[i]);
cs.endText();
}
}
Set<String> targets = new LinkedHashSet<>();
targets.add("SECRET");
Set<Integer> affected = new HashSet<>();
affected.add(0); // only page 0 is affected
bytes = RedactionPipeline.finalize(doc, targets, Collections.emptyList(), affected);
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
PDFTextStripper stripper = new PDFTextStripper();
for (int i = 1; i <= 3; i++) {
stripper.setStartPage(i);
stripper.setEndPage(i);
String text = stripper.getText(reopened);
if (i == 1) {
assertFalse(
text.toLowerCase().contains("secret"),
"Affected page text must be gone");
} else {
assertTrue(
text.contains("PUBLIC"),
"Untouched page "
+ i
+ " must still be text-searchable after scoped rasterisation");
}
}
}
}
@Test
@DisplayName(
"Pathological verification regex triggers verification FAIL (not silent pass) and engages fallback")
void pathologicalVerificationRegexFailsClosed() throws Exception {
// A regex that throws on .matcher(...).find() - build a pattern that causes a runtime
// exception by constructing a matcher over a degenerate character sequence. We simulate
// the pathological case by supplying a pattern whose matcher throws StackOverflowError via
// deep alternation. Because JVM reliably triggering SOE is tricky, we instead construct
// a pattern where .find() throws an unchecked exception using a custom Pattern subclass
// is not possible (Pattern is final). The realistic pathological case is catastrophic
// backtracking, but we cannot rely on timeouts in a unit test. Instead we verify the
// IMPORTANT invariant indirectly: when finalize is called with no content rewrite and a
// surviving target, verification must FAIL and the raster fallback must engage - this
// proves the verify path is not silently swallowing exceptions (which would return the
// unrasterised bytes).
byte[] bytes;
try (PDDocument doc = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
doc.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(100, 700);
cs.showText("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa!");
cs.endText();
}
// Pattern that matches the content - if regex exceptions were swallowed, verification
// would return unrasterised bytes with the match still present.
List<Pattern> patterns = List.of(Pattern.compile("a{5,}"));
bytes = RedactionPipeline.finalize(doc, Collections.emptySet(), patterns);
}
try (PDDocument reopened = Loader.loadPDF(bytes)) {
String text = new PDFTextStripper().getText(reopened);
// Rasterisation should have eliminated text extractability entirely on the affected
// page (no affectedPages passed means whole-document rasterisation fallback).
assertFalse(
Pattern.compile("a{5,}").matcher(text).find(),
"Regex match must not survive verification fallback");
}
}
@Test
@DisplayName(
"CatalogScrubber.stripMatches is case-insensitive so mixed-case targets are removed from catalog strings")
void catalogStripMatchesIsCaseInsensitive() {
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
String result =
CatalogScrubber.stripMatches(
"SMITH, John (also known as smith and Smith Jr.)", targets, List.of());
assertFalse(
result.toLowerCase().contains("smith"),
"All case variants of the target must be removed, actual='" + result + "'");
}
@Test
@DisplayName(
"Catalog scrub strips mixed-case target from bookmark title even when case differs")
void catalogScrubMixedCaseBookmarkTitle() throws Exception {
try (PDDocument doc = new PDDocument()) {
doc.addPage(new PDPage(PDRectangle.A4));
PDDocumentOutline outline = new PDDocumentOutline();
doc.getDocumentCatalog().setDocumentOutline(outline);
PDOutlineItem item = new PDOutlineItem();
item.setTitle("SMITH memo");
outline.addLast(item);
Set<String> targets = new LinkedHashSet<>();
targets.add("smith"); // lowercase target vs uppercase carrier
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
ByteArrayOutputStream baos = new ByteArrayOutputStream();
doc.save(baos);
try (PDDocument reopened = Loader.loadPDF(baos.toByteArray())) {
String title =
reopened.getDocumentCatalog()
.getDocumentOutline()
.getFirstChild()
.getTitle();
assertFalse(
title.toLowerCase().contains("smith"),
"Case-insensitive catalog scrub must remove 'SMITH' when target is 'smith', actual='"
+ title
+ "'");
}
}
}
@Test
@DisplayName(
"AcroForm widget appearance streams are cleared so viewers cannot render the stale value")
void acroFormAppearanceStreamsCleared() throws Exception {
try (PDDocument doc = new PDDocument()) {
doc.addPage(new PDPage(PDRectangle.A4));
PDAcroForm form = new PDAcroForm(doc);
doc.getDocumentCatalog().setAcroForm(form);
PDTextField field = new PDTextField(form);
field.setPartialName("note");
field.getCOSObject().setString(COSName.V, "Paid by Smith");
// Inject a fake AP dict to simulate a cached appearance stream.
org.apache.pdfbox.cos.COSDictionary apDict = new org.apache.pdfbox.cos.COSDictionary();
apDict.setString(COSName.getPDFName("DUMMY"), "Paid by Smith");
field.getCOSObject().setItem(COSName.AP, apDict);
form.getFields().add(field);
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
assertNull(
field.getCOSObject().getDictionaryObject(COSName.AP),
"Widget appearance dict must be cleared after scrub");
assertTrue(
form.getNeedAppearances(),
"/NeedAppearances must be true so viewers regenerate appearances");
}
}
@Test
@DisplayName("XFA packet containing the redaction target is removed from AcroForm")
void xfaPacketDropped() throws Exception {
try (PDDocument doc = new PDDocument()) {
doc.addPage(new PDPage(PDRectangle.A4));
PDAcroForm form = new PDAcroForm(doc);
doc.getDocumentCatalog().setAcroForm(form);
String xfaXml = "<xdp><data><field>Smith</field></data></xdp>";
COSStream xfa = doc.getDocument().createCOSStream();
try (var os = xfa.createOutputStream()) {
os.write(xfaXml.getBytes(java.nio.charset.StandardCharsets.UTF_8));
}
form.getCOSObject().setItem(COSName.XFA, xfa);
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
assertNull(
form.getCOSObject().getDictionaryObject(COSName.XFA),
"XFA packet containing target literal must be removed");
}
}
@Test
@DisplayName("OpenAction carrying a target URI is removed from the catalog")
void openActionWithTargetUriRemoved() throws Exception {
try (PDDocument doc = new PDDocument()) {
doc.addPage(new PDPage(PDRectangle.A4));
org.apache.pdfbox.cos.COSDictionary openAction =
new org.apache.pdfbox.cos.COSDictionary();
openAction.setItem(COSName.getPDFName("S"), COSName.URI);
openAction.setItem(
COSName.getPDFName("URI"), new COSString("https://example.com/?user=Smith"));
doc.getDocumentCatalog()
.getCOSObject()
.setItem(COSName.getPDFName("OpenAction"), openAction);
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
assertNull(
doc.getDocumentCatalog()
.getCOSObject()
.getDictionaryObject(COSName.getPDFName("OpenAction")),
"OpenAction containing target must be removed from catalog");
}
}
@Test
@DisplayName("Bookmark /A URI action containing the redaction target is removed")
void bookmarkActionUriRemoved() throws Exception {
try (PDDocument doc = new PDDocument()) {
doc.addPage(new PDPage(PDRectangle.A4));
PDDocumentOutline outline = new PDDocumentOutline();
doc.getDocumentCatalog().setDocumentOutline(outline);
PDOutlineItem item = new PDOutlineItem();
item.setTitle("Link to case file");
org.apache.pdfbox.cos.COSDictionary action = new org.apache.pdfbox.cos.COSDictionary();
action.setItem(COSName.getPDFName("S"), COSName.URI);
action.setItem(
COSName.getPDFName("URI"), new COSString("https://example.com/?file=Smith"));
item.getCOSObject().setItem(COSName.A, action);
outline.addLast(item);
Set<String> targets = new LinkedHashSet<>();
targets.add("Smith");
CatalogScrubber.scrub(doc, targets, Collections.emptyList());
assertNull(
item.getCOSObject().getDictionaryObject(COSName.A),
"Bookmark /A action containing target URI must be removed");
}
}
@Test
@DisplayName("buildPatterns respects useRegex and wholeWordSearch flags")
void buildPatternsHonoursFlags() {
List<Pattern> plain = RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, false);
assertEquals(1, plain.size());
assertTrue(plain.get(0).matcher("AeroSmith").find());
assertTrue(plain.get(0).matcher("Smith paid").find());
List<Pattern> wholeWord =
RedactionPipeline.buildPatterns(new String[] {"Smith"}, false, true);
assertTrue(wholeWord.get(0).matcher("Smith paid").find());
assertFalse(wholeWord.get(0).matcher("AeroSmith").find());
List<Pattern> regex = RedactionPipeline.buildPatterns(new String[] {"\\d{3}"}, true, false);
assertTrue(regex.get(0).matcher("ID 123 issued").find());
}
}
Binary file not shown.
+109
View File
@@ -0,0 +1,109 @@
@security @redact
Feature: PDF redaction physically removes text
Redaction must destroy the underlying text, not merely cover it with a box.
These scenarios push known text through the redaction endpoints and assert the
target is gone from the extracted text layer (and catalog carriers) while other
text survives.
Scenario: Auto-redact physically removes the target word
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "PUBLIC alpha SECRET99 omega PUBLIC"
And the request data includes
| parameter | value |
| listOfText | SECRET99 |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response content type should be "application/pdf"
And the response PDF should not contain the text "SECRET99"
And the response PDF should contain the text "omega"
Scenario: Auto-redact leaves substrings when whole-word search is on
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "cat classification scatter"
And the request data includes
| parameter | value |
| listOfText | cat |
| wholeWordSearch | true |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should contain the text "classification"
And the response PDF should contain the text "scatter"
Scenario: Auto-redact with a regex pattern removes matches
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "call 123-45-6789 today"
And the request data includes
| parameter | value |
| listOfText | \d{3}-\d{2}-\d{4} |
| useRegex | true |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "123-45-6789"
And the response PDF should contain the text "today"
Scenario: Auto-redact convert-to-image drops the entire text layer
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "keep SECRET77 hidden"
And the request data includes
| parameter | value |
| listOfText | SECRET77 |
| convertPDFToImage | true |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "SECRET77"
And the response PDF should not contain the text "keep"
Scenario: Auto-redact also scrubs the target from bookmark titles
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "body SECRET55 text"
And the pdf has a bookmark titled "Chapter SECRET55 overview"
And the request data includes
| parameter | value |
| listOfText | SECRET55 |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "SECRET55"
And the response PDF bookmarks should not contain "SECRET55"
Scenario: Manual whole-page redaction wipes every word on the page
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "TOP SECRET material"
And the request data includes
| parameter | value |
| pageNumbers | 1 |
When I send the API request to the endpoint "/api/v1/security/redact"
Then the response status code should be 200
And the response PDF should not contain the text "SECRET"
Scenario: Auto-redact removes case variants of the target
Given I generate a PDF file as "fileInput"
And the pdf pages all contain the text "alpha Secret99x omega"
And the request data includes
| parameter | value |
| listOfText | SECRET99X |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "Secret99x"
And the response PDF should contain the text "omega"
Scenario: Auto-redact only touches pages containing the target
Given I generate a PDF file as "fileInput"
And the pdf contains 3 pages
And the request data includes
| parameter | value |
| listOfText | Page 2 |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "Page 2"
And the response PDF should contain the text "Page 1"
And the response PDF should contain the text "Page 3"
Scenario: Auto-redact handles a Type3 font PDF (glyphs as content streams)
Given I use an example file at "../crop_test.pdf" as parameter "fileInput"
And the request data includes
| parameter | value |
| listOfText | EXAMPLE |
When I send the API request to the endpoint "/api/v1/security/auto-redact"
Then the response status code should be 200
And the response PDF should not contain the text "EXAMPLE"
And the response PDF should contain the text "CROP"
@@ -17,6 +17,10 @@ from PIL import Image, ImageDraw
API_HEADERS = {"X-API-KEY": "123456789"}
# Base URL of the backend under test. Defaults to the CI value; override with
# STIRLING_BASE_URL to point at a backend on another port (e.g. a local sidecar run).
BASE_URL = os.environ.get("STIRLING_BASE_URL", "http://localhost:8080")
#########
# GIVEN #
#########
@@ -581,7 +585,7 @@ def step_request_json_part(context, part_name, json_content):
@when('I send a GET request to "{endpoint}"')
def step_send_get_request(context, endpoint):
base_url = "http://localhost:8080"
base_url = BASE_URL
full_url = f"{base_url}{endpoint}"
response = requests.get(full_url, headers=API_HEADERS, timeout=60)
context.response = response
@@ -589,7 +593,7 @@ def step_send_get_request(context, endpoint):
@when('I send a GET request to "{endpoint}" with parameters')
def step_send_get_request_with_params(context, endpoint):
base_url = "http://localhost:8080"
base_url = BASE_URL
params = {row["parameter"]: row["value"] for row in context.table}
full_url = f"{base_url}{endpoint}"
response = requests.get(full_url, params=params, headers=API_HEADERS, timeout=60)
@@ -598,7 +602,7 @@ def step_send_get_request_with_params(context, endpoint):
@when('I send the API request to the endpoint "{endpoint}"')
def step_send_api_request(context, endpoint):
url = f"http://localhost:8080{endpoint}"
url = f"{BASE_URL}{endpoint}"
files = context.files if hasattr(context, "files") else {}
if not hasattr(context, "request_data") or context.request_data is None:
@@ -803,3 +807,69 @@ def step_response_matches_regex(context, pattern):
assert re.match(
pattern, response_text
), f"Response '{response_text}' does not match the expected pattern '{pattern}'"
# ---------------------------------------------------------------------------
# Redaction: text-layer and catalog assertions
# ---------------------------------------------------------------------------
def _extract_response_pdf_text(context):
reader = PdfReader(io.BytesIO(context.response.content))
return "\n".join((page.extract_text() or "") for page in reader.pages)
@then('the response PDF should contain the text "{text}"')
def step_response_pdf_contains_text(context, text):
extracted = _extract_response_pdf_text(context)
assert text in extracted, (
f"Expected redacted PDF to still contain '{text}', but it was missing. "
f"Extracted text: {extracted!r}"
)
@then('the response PDF should not contain the text "{text}"')
def step_response_pdf_not_contains_text(context, text):
extracted = _extract_response_pdf_text(context)
assert text not in extracted, (
f"Redacted PDF still contains '{text}' - redaction did not remove it. "
f"Extracted text: {extracted!r}"
)
def _collect_outline_titles(outline, titles):
for item in outline:
if isinstance(item, list):
_collect_outline_titles(item, titles)
else:
title = getattr(item, "title", None)
if title:
titles.append(title)
@then('the response PDF bookmarks should not contain "{text}"')
def step_response_pdf_bookmarks_not_contain(context, text):
reader = PdfReader(io.BytesIO(context.response.content))
titles = []
try:
_collect_outline_titles(reader.outline, titles)
except Exception:
titles = []
joined = " ".join(titles)
assert text not in joined, (
f"Redacted PDF bookmark titles still contain '{text}': {titles!r}"
)
@given('the pdf has a bookmark titled "{title}"')
def step_pdf_has_bookmark_titled(context, title):
"""Add a single top-level outline entry with an explicit title."""
reader = PdfReader(context.file_name)
writer = PdfWriter()
for page in reader.pages:
writer.add_page(page)
writer.add_outline_item(title, 0)
with open(context.file_name, "wb") as f:
writer.write(f)
context.files[context.param_name].close()
context.files[context.param_name] = open(context.file_name, "rb")