Feature/pdf ingestion jpdfium (#6525)
This commit is contained in:
-153
@@ -1,153 +0,0 @@
|
||||
package stirling.software.SPDF.pdf.parser;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
import static stirling.software.SPDF.pdf.parser.PdfModels.*;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/**
|
||||
* Unit tests for {@link LineAlignmentTableParser}, focused on the coincident-line merge logic and
|
||||
* column-grid construction.
|
||||
*/
|
||||
class LineAlignmentTableParserTest {
|
||||
|
||||
private final LineAlignmentTableParser parser = new LineAlignmentTableParser();
|
||||
|
||||
// ── mergeCoincidentLines ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_singleLine_unchanged() {
|
||||
var lines = List.of(tokenized(rawLine(10f, 100f, "Revenue")));
|
||||
assertThat(parser.mergeCoincidentLines(lines)).hasSize(1);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_distinctYLines_unchanged() {
|
||||
// Two lines at different y positions — must NOT be merged.
|
||||
var lines =
|
||||
List.of(
|
||||
tokenized(rawLine(10f, 100f, "Revenue")),
|
||||
tokenized(rawLine(10f, 115f, "Cost")));
|
||||
assertThat(parser.mergeCoincidentLines(lines)).hasSize(2);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_sameY_merged() {
|
||||
// Simulates a financial-table row split by LineBuilder at the column gap:
|
||||
// label fragment at x=72 → "Revenue"
|
||||
// value fragment at x=350 → "1,234"
|
||||
// Both have y=100. After merge they should form one TokenizedLine.
|
||||
var label = rawLine(72f, 100f, "Revenue");
|
||||
var value = rawLine(350f, 100f, "1,234");
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(value)));
|
||||
|
||||
assertThat(merged).hasSize(1);
|
||||
// The merged line should contain tokens from both halves.
|
||||
var tokens = merged.get(0).all();
|
||||
assertThat(tokens.stream().map(t -> t.text()).toList())
|
||||
.containsExactlyInAnyOrder("Revenue", "1,234");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_sameY_mergedLineHasCorrectBounds() {
|
||||
var label = rawLine(72f, 100f, "Revenue"); // 7 chars × 6pt = 42pt wide → right = 114
|
||||
var value = rawLine(350f, 100f, "1,234"); // 5 chars × 6pt = 30pt wide → right = 380
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(value)));
|
||||
|
||||
var bounds = merged.get(0).line().bounds();
|
||||
assertThat(bounds.x()).isEqualTo(72f);
|
||||
assertThat(bounds.right()).isEqualTo(380f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_withinTolerance_merged() {
|
||||
// Lines 1.5pt apart (within ROW_MERGE_TOLERANCE_PT = 2pt) should merge.
|
||||
var a = rawLine(10f, 100.0f, "Alpha");
|
||||
var b = rawLine(200f, 101.5f, "99");
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b)));
|
||||
assertThat(merged).hasSize(1);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_beyondTolerance_notMerged() {
|
||||
// Lines 3pt apart (beyond ROW_MERGE_TOLERANCE_PT = 2pt) should NOT merge.
|
||||
var a = rawLine(10f, 100.0f, "Alpha");
|
||||
var b = rawLine(200f, 103.0f, "99");
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b)));
|
||||
assertThat(merged).hasSize(2);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_threeCoincident_allMerged() {
|
||||
// Three fragments at the same y (e.g. wide financial table with two value columns).
|
||||
var a = rawLine(72f, 100f, "Revenue");
|
||||
var b = rawLine(300f, 100f, "1,234");
|
||||
var c = rawLine(400f, 100f, "5,678");
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b), tokenized(c)));
|
||||
assertThat(merged).hasSize(1);
|
||||
assertThat(merged.get(0).all()).hasSize(3);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_coincidentPairFollowedByDistinctLine_twoGroups() {
|
||||
var a = rawLine(72f, 100f, "Revenue");
|
||||
var b = rawLine(350f, 100f, "1,234"); // same y as a → merges with a
|
||||
var c = rawLine(10f, 115f, "Expenses"); // different y → stays separate
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b), tokenized(c)));
|
||||
assertThat(merged).hasSize(2);
|
||||
}
|
||||
|
||||
@Test
|
||||
void mergeCoincidentLines_numericAnchorStatus_correctAfterMerge() {
|
||||
// After merging, the combined line should be an anchor (≥2 numeric tokens).
|
||||
// "Revenue" alone → not an anchor. "1,234 567" alone → anchor.
|
||||
// Merged → anchor with at least 2 numerics.
|
||||
var label = rawLine(72f, 100f, "Revenue");
|
||||
var values = rawLineMultiWord(350f, 100f, "1,234", 30f, "567", 30f);
|
||||
|
||||
var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(values)));
|
||||
|
||||
assertThat(merged).hasSize(1);
|
||||
assertThat(merged.get(0).isAnchor()).isTrue();
|
||||
}
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
/** Creates a RawLine with a single TextFragment of the given text at the given position. */
|
||||
private static RawLine rawLine(float x, float y, String text) {
|
||||
float width = text.length() * 6f; // ~6pt per char — rough but consistent
|
||||
float height = 12f;
|
||||
Bounds bounds = new Bounds(x, y, width, height);
|
||||
TextFragment fragment =
|
||||
new TextFragment("tf-test", text, bounds, y + height, 11f, "Helvetica", false);
|
||||
return new RawLine("ln-test", List.of(fragment), bounds, 1);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a RawLine with two TextFragments representing two words separated by a small gap.
|
||||
* Used to simulate a values-only line with multiple numeric tokens.
|
||||
*/
|
||||
private static RawLine rawLineMultiWord(
|
||||
float x, float y, String word1, float w1, String word2, float w2) {
|
||||
float height = 12f;
|
||||
Bounds b1 = new Bounds(x, y, w1, height);
|
||||
Bounds b2 = new Bounds(x + w1 + 5f, y, w2, height);
|
||||
TextFragment f1 = new TextFragment("tf-1", word1, b1, y + height, 11f, "Helvetica", false);
|
||||
TextFragment f2 = new TextFragment("tf-2", word2, b2, y + height, 11f, "Helvetica", false);
|
||||
Bounds lineBounds = new Bounds(x, y, x + w1 + 5f + w2 - x, height);
|
||||
return new RawLine("ln-test", List.of(f1, f2), lineBounds, 1);
|
||||
}
|
||||
|
||||
/** Tokenises a RawLine via the parser's own tokenise logic (package-private access). */
|
||||
private LineAlignmentTableParser.TokenizedLine tokenized(RawLine line) {
|
||||
return parser.tokenize(line);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,269 @@
|
||||
package stirling.software.common.pdf;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertDoesNotThrow;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.fail;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.junit.jupiter.api.Disabled;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.junit.jupiter.params.ParameterizedTest;
|
||||
import org.junit.jupiter.params.provider.Arguments;
|
||||
import org.junit.jupiter.params.provider.MethodSource;
|
||||
|
||||
import stirling.software.jpdfium.PdfDocument;
|
||||
import stirling.software.jpdfium.text.TextLine;
|
||||
import stirling.software.jpdfium.text.TextWord;
|
||||
|
||||
/**
|
||||
* Accuracy and robustness tests for {@link PdfMarkdownConverter}, comparing conversion output
|
||||
* against hand-authored golden Markdown for a set of owned/synthetic fixtures.
|
||||
*
|
||||
* <p>The {@link #gatedFixtures()} set is enforced in CI: those fixtures currently convert within
|
||||
* the accuracy threshold and guard against regressions. Fixtures still being iterated on live in
|
||||
* {@link #wipFixtures()} under a {@link Disabled} test so the goldens stay in the tree without
|
||||
* breaking the build. Enable the WIP test locally to see per-fixture scores while working on the
|
||||
* converter.
|
||||
*/
|
||||
class PdfMarkdownConverterTest {
|
||||
|
||||
/** Accuracy threshold: output must share at least this fraction of content with the golden. */
|
||||
private static final double THRESHOLD = 0.95;
|
||||
|
||||
@TempDir Path tmp;
|
||||
|
||||
/** Fixtures that meet the accuracy threshold today and therefore gate CI. */
|
||||
static Stream<Arguments> gatedFixtures() {
|
||||
return Stream.of(
|
||||
Arguments.of("multi-column-test_lorem.pdf", "multi-column-test_lorem.md"),
|
||||
Arguments.of("bordered-table-test_widget.pdf", "bordered-table-test_widget.md"),
|
||||
Arguments.of("many-tables-test_stress.pdf", "many-tables-test_stress.md"));
|
||||
}
|
||||
|
||||
/** Fixtures still below the threshold; tracked here, enable locally to iterate. */
|
||||
static Stream<Arguments> wipFixtures() {
|
||||
return Stream.of(
|
||||
Arguments.of(
|
||||
"wrapped-cell-test_expense-report.pdf",
|
||||
"wrapped-cell-test_expense-report.md"));
|
||||
}
|
||||
|
||||
@ParameterizedTest(name = "{0}")
|
||||
@MethodSource("gatedFixtures")
|
||||
void convertMatchesGoldenMarkdown(String pdfName, String mdName) throws IOException {
|
||||
assertConversionMatchesGolden(pdfName, mdName);
|
||||
}
|
||||
|
||||
@Disabled("WIP fixtures below the accuracy threshold; enable locally to iterate")
|
||||
@ParameterizedTest(name = "{0}")
|
||||
@MethodSource("wipFixtures")
|
||||
void convertMatchesGoldenMarkdownWip(String pdfName, String mdName) throws IOException {
|
||||
assertConversionMatchesGolden(pdfName, mdName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Degenerate/extreme geometry must not crash the converter. A crafted or malformed PDF can
|
||||
* position text anywhere via a text matrix, so a row's words can span from near the origin to a
|
||||
* coordinate beyond {@link Integer#MAX_VALUE}. The old column-detection code sized an {@code
|
||||
* int[]} straight from {@code (int) Math.ceil(maxX) - lo}, which either allocated a multi-GB
|
||||
* array (OutOfMemoryError) or overflowed to a negative length (NegativeArraySizeException) —
|
||||
* taking down the request thread. Detection must instead bail out and return no columns.
|
||||
*/
|
||||
@Test
|
||||
void columnDetectionSurvivesDegenerateGeometry() {
|
||||
// x ≈ 2.5e9 is past Integer.MAX_VALUE; combined with a near-origin word it yields an
|
||||
// implausible span that the pre-fix code turned into a fatal array allocation.
|
||||
List<TextLine> rows = new ArrayList<>();
|
||||
for (int r = 0; r < 4; r++) {
|
||||
float y = 400f - r * 12f;
|
||||
TextWord near = new TextWord(List.of(), 50f, y, 30f, 10f);
|
||||
TextWord far = new TextWord(List.of(), 2_500_000_000f, y, 30f, 10f);
|
||||
rows.add(new TextLine(List.of(near, far), 50f, y, 2_499_999_980f, 10f));
|
||||
}
|
||||
|
||||
List<float[]> columns =
|
||||
assertDoesNotThrow(() -> PdfMarkdownConverter.findColumnRangesFromLines(rows));
|
||||
assertTrue(
|
||||
columns.isEmpty(),
|
||||
"implausible page span should disable column detection, not allocate from it");
|
||||
}
|
||||
|
||||
private void assertConversionMatchesGolden(String pdfName, String mdName) throws IOException {
|
||||
Path pdfPath = tmp.resolve(pdfName);
|
||||
try (InputStream in =
|
||||
getClass().getResourceAsStream("/pdf-ingestion-fixtures/" + pdfName)) {
|
||||
if (in == null) {
|
||||
fail("Fixture not found on classpath: /pdf-ingestion-fixtures/" + pdfName);
|
||||
}
|
||||
Files.copy(in, pdfPath);
|
||||
}
|
||||
|
||||
String actual;
|
||||
try (PdfDocument doc = PdfDocument.open(pdfPath)) {
|
||||
actual = new PdfMarkdownConverter().convert(doc);
|
||||
}
|
||||
|
||||
String expected;
|
||||
try (InputStream in = getClass().getResourceAsStream("/pdf-ingestion-fixtures/" + mdName)) {
|
||||
if (in == null) {
|
||||
fail("Golden file not found on classpath: /pdf-ingestion-fixtures/" + mdName);
|
||||
}
|
||||
expected = new String(in.readAllBytes(), StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
// Image placeholders are not scored: their body text is a TODO ("ideally, add the info
|
||||
// available about the image...") rather than real content, so comparing it would penalise
|
||||
// output for matching a placeholder we intend to replace. Drop those lines from both sides.
|
||||
expected = stripImagePlaceholders(expected);
|
||||
actual = stripImagePlaceholders(actual);
|
||||
|
||||
double similarity = similarity(expected, actual);
|
||||
if (similarity < THRESHOLD) {
|
||||
fail(
|
||||
String.format(
|
||||
"Markdown output differs from golden file '%s' by %.1f%% (threshold %.0f%%):%n%s",
|
||||
mdName,
|
||||
(1.0 - similarity) * 100,
|
||||
(1.0 - THRESHOLD) * 100,
|
||||
unifiedDiff(expected, actual)));
|
||||
}
|
||||
}
|
||||
|
||||
/** Substring identifying an image-placeholder line, which is excluded from scoring. */
|
||||
private static final String IMAGE_PLACEHOLDER_MARKER = "Image intentionally redacted";
|
||||
|
||||
/**
|
||||
* Removes non-content lines from the comparison: image placeholders (TODO text we intend to
|
||||
* replace) and GFM table separator rows (the {@code |---|---|} divider, whose exact dash count
|
||||
* is cosmetic — any run of three or more dashes is valid Markdown).
|
||||
*/
|
||||
private static String stripImagePlaceholders(String md) {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
for (String line : md.split("\n", -1)) {
|
||||
if (line.contains(IMAGE_PLACEHOLDER_MARKER)
|
||||
|| line.strip().startsWith("<image redacted")
|
||||
|| isTableSeparatorRow(line)) {
|
||||
continue;
|
||||
}
|
||||
if (sb.length() > 0) {
|
||||
sb.append('\n');
|
||||
}
|
||||
sb.append(line);
|
||||
}
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
/** True for a GFM table separator row, e.g. {@code |---|:--:|---|} (only |, -, :, space). */
|
||||
private static boolean isTableSeparatorRow(String line) {
|
||||
String t = line.strip();
|
||||
if (!t.contains("-")) {
|
||||
return false;
|
||||
}
|
||||
return t.chars().allMatch(c -> c == '|' || c == '-' || c == ':' || c == ' ');
|
||||
}
|
||||
|
||||
/**
|
||||
* Character-level similarity: proportion of expected characters that appear in the LCS. O(n*m)
|
||||
* but golden files are small enough that this is fine.
|
||||
*/
|
||||
private static double similarity(String expected, String actual) {
|
||||
if (expected.isEmpty() && actual.isEmpty()) return 1.0;
|
||||
if (expected.isEmpty() || actual.isEmpty()) return 0.0;
|
||||
// Strip all whitespace for a content-focused comparison
|
||||
String e = expected.replaceAll("\\s+", " ").strip();
|
||||
String a = actual.replaceAll("\\s+", " ").strip();
|
||||
int lcs = lcsLength(e, a);
|
||||
return (double) lcs / Math.max(e.length(), a.length());
|
||||
}
|
||||
|
||||
private static int lcsLength(String a, String b) {
|
||||
// Use two-row DP to keep memory reasonable
|
||||
int m = a.length(), n = b.length();
|
||||
int[] prev = new int[n + 1];
|
||||
int[] curr = new int[n + 1];
|
||||
for (int i = 1; i <= m; i++) {
|
||||
for (int j = 1; j <= n; j++) {
|
||||
if (a.charAt(i - 1) == b.charAt(j - 1)) {
|
||||
curr[j] = prev[j - 1] + 1;
|
||||
} else {
|
||||
curr[j] = Math.max(curr[j - 1], prev[j]);
|
||||
}
|
||||
}
|
||||
int[] tmp = prev;
|
||||
prev = curr;
|
||||
curr = tmp;
|
||||
java.util.Arrays.fill(curr, 0);
|
||||
}
|
||||
return prev[n];
|
||||
}
|
||||
|
||||
private static String unifiedDiff(String expected, String actual) {
|
||||
String[] expectedLines = expected.split("\n", -1);
|
||||
String[] actualLines = actual.split("\n", -1);
|
||||
|
||||
List<String> diff = new ArrayList<>();
|
||||
diff.add("--- expected");
|
||||
diff.add("+++ actual");
|
||||
|
||||
int maxLines = Math.max(expectedLines.length, actualLines.length);
|
||||
int context = 3;
|
||||
boolean inHunk = false;
|
||||
int hunkStart = -1;
|
||||
List<String> hunkLines = new ArrayList<>();
|
||||
|
||||
for (int i = 0; i < maxLines; i++) {
|
||||
String exp = i < expectedLines.length ? expectedLines[i] : null;
|
||||
String act = i < actualLines.length ? actualLines[i] : null;
|
||||
|
||||
boolean changed = exp == null || act == null || !exp.equals(act);
|
||||
if (changed) {
|
||||
if (!inHunk) {
|
||||
inHunk = true;
|
||||
hunkStart = Math.max(0, i - context);
|
||||
// add context lines before change
|
||||
for (int c = hunkStart; c < i; c++) {
|
||||
hunkLines.add(" " + (c < expectedLines.length ? expectedLines[c] : ""));
|
||||
}
|
||||
}
|
||||
if (exp != null) hunkLines.add("-" + exp);
|
||||
if (act != null) hunkLines.add("+" + act);
|
||||
} else {
|
||||
if (inHunk) {
|
||||
hunkLines.add(" " + exp);
|
||||
// check if we're far enough past the last change to close the hunk
|
||||
boolean moreChanges = false;
|
||||
for (int j = i + 1; j < Math.min(i + context, maxLines); j++) {
|
||||
String e2 = j < expectedLines.length ? expectedLines[j] : null;
|
||||
String a2 = j < actualLines.length ? actualLines[j] : null;
|
||||
if (e2 == null || a2 == null || !e2.equals(a2)) {
|
||||
moreChanges = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!moreChanges && (i - hunkStart) >= context) {
|
||||
diff.add("@@ -" + (hunkStart + 1) + " @@");
|
||||
diff.addAll(hunkLines);
|
||||
hunkLines.clear();
|
||||
inHunk = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (inHunk && !hunkLines.isEmpty()) {
|
||||
diff.add("@@ -" + (hunkStart + 1) + " @@");
|
||||
diff.addAll(hunkLines);
|
||||
}
|
||||
|
||||
return String.join("\n", diff);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user