From 837371cff004d383823d783ca2de6e65ddb443aa Mon Sep 17 00:00:00 2001 From: Ludy87 Date: Thu, 11 Jun 2026 20:38:16 +0200 Subject: [PATCH] del --- .../pdf/parser/LineAlignmentTableParser.java | 526 ------------------ .../software/SPDF/pdf/parser/LineBuilder.java | 139 ----- 2 files changed, 665 deletions(-) delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java deleted file mode 100644 index df3feaa08e..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java +++ /dev/null @@ -1,526 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.io.IOException; -import java.util.ArrayList; -import java.util.Arrays; -import java.util.Collections; -import java.util.Comparator; -import java.util.HashMap; -import java.util.List; -import java.util.Map; -import java.util.Optional; -import java.util.TreeMap; -import java.util.regex.Pattern; - -import org.apache.pdfbox.pdmodel.PDDocument; -import org.springframework.stereotype.Service; - -import lombok.extern.slf4j.Slf4j; - -/** - * Fallback {@link TableParser} for borderless financial tables using text geometry. - * - *

Identifies "anchor lines" (≥2 numeric tokens), builds a column grid from their right-edge - * positions, groups vertically proximate anchor lines into table candidates, then scores each group - * on column consistency and anchor density (confidence ceiling 0.85). - */ -@Service -@Slf4j -public class LineAlignmentTableParser implements TableParser { - - /** Width in points of each column position bucket. */ - static final float COLUMN_BUCKET_PT = 5f; - - /** Tolerance in buckets when matching a token's right-edge to a confirmed column position. */ - private static final int COLUMN_MATCH_BUCKETS = 2; - - /** Maximum gap (as a multiple of modal line spacing) before splitting a group. */ - private static final float MAX_GAP_FACTOR = 2.5f; - - /** Minimum anchor rows (numeric-heavy) to form a valid table. */ - static final int MIN_TABLE_ROWS = 3; - - /** Minimum confirmed column positions to form a valid table. */ - static final int MIN_COLUMNS = 2; - - /** - * Min fraction of anchor lines a column must appear on to be confirmed (permissive for N/A - * rows). - */ - private static final double COLUMN_MIN_FREQUENCY = 0.40; - - /** - * Matches financial numeric tokens: integers, decimals, parenthetical negatives, currency, - * percent, nil dashes. - */ - private static final Pattern NUMERIC = - Pattern.compile("^[\\(\\-\\$£€¥]?\\d[\\d,\\.]*[\\)%]?$|^[-–—]$"); - - /** - * Lines within this y-distance are merged into one row (restores rows split by LineBuilder's - * column-gap logic). - */ - static final float ROW_MERGE_TOLERANCE_PT = 2f; - - // ── public API ─────────────────────────────────────────────────────────────────────────────── - - @Override - public List parse(PDDocument document, RawPage rawPage) throws IOException { - List lines = rawPage.lines(); - if (lines.size() < MIN_TABLE_ROWS) return List.of(); - - float modalSpacing = computeModalSpacing(lines); - List tokenized = - mergeCoincidentLines(lines.stream().map(this::tokenize).toList()); - - List anchors = tokenized.stream().filter(TokenizedLine::isAnchor).toList(); - - if (anchors.size() < MIN_TABLE_ROWS) return List.of(); - - List columnGrid = buildColumnGrid(anchors); - if (columnGrid.size() < MIN_COLUMNS) { - log.debug( - "Page {}: LineAlignment — fewer than {} confirmed columns, skipping", - rawPage.pageNumber(), - MIN_COLUMNS); - return List.of(); - } - - List> groups = groupRows(tokenized, columnGrid, modalSpacing); - - List results = new ArrayList<>(); - for (int i = 0; i < groups.size(); i++) { - buildFragment(groups.get(i), columnGrid, rawPage.pageNumber(), i) - .ifPresent(results::add); - } - - log.debug( - "Page {}: LineAlignment detected {} table(s) ({} anchor lines, {} columns)", - rawPage.pageNumber(), - results.size(), - anchors.size(), - columnGrid.size()); - return results; - } - - // ── coincident-line merging ────────────────────────────────────────────────────────────────── - - /** - * Merges tokenised lines sharing the same y-position into one row, rejoining label/value halves - * split by LineBuilder. - */ - List mergeCoincidentLines(List tokenized) { - if (tokenized.size() < 2) return tokenized; - - List result = new ArrayList<>(); - int i = 0; - - while (i < tokenized.size()) { - float baseY = tokenized.get(i).line().bounds().y(); - int j = i + 1; - while (j < tokenized.size() - && Math.abs(tokenized.get(j).line().bounds().y() - baseY) - <= ROW_MERGE_TOLERANCE_PT) { - j++; - } - - if (j == i + 1) { - result.add(tokenized.get(i)); - } else { - result.add(mergeGroup(tokenized.subList(i, j))); - } - i = j; - } - - return result; - } - - private TokenizedLine mergeGroup(List group) { - List mergedFragments = - group.stream() - .flatMap(tl -> tl.line().fragments().stream()) - .sorted(Comparator.comparingDouble(f -> f.bounds().x())) - .toList(); - - Bounds mergedBounds = - group.stream() - .map(tl -> tl.line().bounds()) - .reduce(Bounds::merge) - .orElse(group.getFirst().line().bounds()); - - RawLine mergedLine = - new RawLine( - group.getFirst().line().lineId(), - mergedFragments, - mergedBounds, - group.getFirst().line().pageNumber()); - - return tokenize(mergedLine); - } - - // ── tokenisation ───────────────────────────────────────────────────────────────────────────── - - /** - * Splits fragments into word-level tokens; x-positions are estimated linearly within each - * fragment. - */ - TokenizedLine tokenize(RawLine line) { - List tokens = new ArrayList<>(); - for (TextFragment frag : line.fragments()) { - tokens.addAll(tokensFromFragment(frag)); - } - List numeric = tokens.stream().filter(LineToken::numeric).toList(); - return new TokenizedLine(line, tokens, numeric); - } - - private List tokensFromFragment(TextFragment frag) { - String raw = frag.text(); - if (raw == null || raw.isBlank()) return List.of(); - - float fragX = frag.bounds().x(); - float fragWidth = frag.bounds().width(); - int rawLen = raw.length(); - - List result = new ArrayList<>(); - int offset = 0; - for (String part : raw.split("\\s+")) { - if (part.isEmpty()) { - offset++; - continue; - } - int idx = raw.indexOf(part, offset); - if (idx < 0) idx = offset; - - float tokenX = rawLen > 0 ? fragX + ((float) idx / rawLen) * fragWidth : fragX; - float tokenRight = - rawLen > 0 - ? fragX + ((float) (idx + part.length()) / rawLen) * fragWidth - : fragX + fragWidth; - - result.add(new LineToken(part, tokenX, tokenRight, NUMERIC.matcher(part).matches())); - offset = idx + part.length(); - } - return result; - } - - // ── column grid ────────────────────────────────────────────────────────────────────────────── - - /** - * Returns confirmed column right-edge positions — those appearing on ≥ {@value - * #COLUMN_MIN_FREQUENCY} × N anchor lines. - */ - private List buildColumnGrid(List anchors) { - // bucket → set of line indices that contributed a numeric token to that bucket - Map> bucketLines = new HashMap<>(); - for (int i = 0; i < anchors.size(); i++) { - for (LineToken t : anchors.get(i).numeric()) { - int bucket = bucket(t.right()); - bucketLines.computeIfAbsent(bucket, k -> new ArrayList<>()).add(i); - } - } - - int minHits = - Math.max(MIN_TABLE_ROWS, (int) Math.ceil(anchors.size() * COLUMN_MIN_FREQUENCY)); - - // Confirmed buckets → average right-edge for that bucket - TreeMap confirmed = new TreeMap<>(); - for (Map.Entry> entry : bucketLines.entrySet()) { - // Count distinct lines - long distinctLines = entry.getValue().stream().distinct().count(); - if (distinctLines >= minHits) { - double avg = - entry.getValue().stream() - .distinct() // weight each line equally regardless of token count - .mapToDouble( - lineIdx -> - avgRightEdgeForBucket( - anchors, lineIdx, entry.getKey())) - .average() - .orElse(entry.getKey() * (double) COLUMN_BUCKET_PT); - confirmed.put(entry.getKey(), (float) avg); - } - } - - return new ArrayList<>(confirmed.values()); // already sorted by bucket (left to right) - } - - /** - * Returns the average right-edge position of tokens in {@code line} whose bucket matches {@code - * targetBucket}, falling back to the bucket's nominal centre when no tokens match. - */ - private double avgRightEdgeForBucket( - List anchors, int lineIdx, int targetBucket) { - return anchors.get(lineIdx).numeric().stream() - .filter(t -> bucket(t.right()) == targetBucket) - .mapToDouble(LineToken::right) - .average() - .orElse(targetBucket * (double) COLUMN_BUCKET_PT); - } - - // ── grouping ───────────────────────────────────────────────────────────────────────────────── - - /** - * Groups anchor lines into table candidates, including adjacent label rows; a gap > - * MAX_GAP_FACTOR × modal spacing splits groups. - */ - private List> groupRows( - List all, List columnGrid, float modalSpacing) { - float maxGap = modalSpacing > 0 ? modalSpacing * MAX_GAP_FACTOR : 30f; - - List> groups = new ArrayList<>(); - List current = new ArrayList<>(); - - for (int i = 0; i < all.size(); i++) { - TokenizedLine tl = all.get(i); - boolean fits = tl.isAnchor() && matchesGrid(tl, columnGrid); - - if (current.isEmpty()) { - if (fits) current.add(tl); - continue; - } - - float gap = tl.line().bounds().y() - current.getLast().line().bounds().bottom(); - - if (gap > maxGap) { - groups.add(current); - current = new ArrayList<>(); - if (fits) current.add(tl); - continue; - } - - if (fits) { - current.add(tl); - } else if (!tl.line().text().isBlank()) { - // Include non-anchor lines (labels) only if they have text and are within - // proximity. - current.add(tl); - } - } - - if (!current.isEmpty()) groups.add(current); - - return groups.stream().filter(g -> hasEnoughAnchorRows(g, columnGrid)).toList(); - } - - private boolean hasEnoughAnchorRows(List group, List columnGrid) { - return group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count() - >= MIN_TABLE_ROWS; - } - - /** A line "matches" the grid when ≥ 60 % of its numeric tokens land in confirmed columns. */ - private boolean matchesGrid(TokenizedLine tl, List columnGrid) { - if (tl.numeric().isEmpty()) return false; - long matches = - tl.numeric().stream() - .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0) - .count(); - return (double) matches / tl.numeric().size() >= 0.60; - } - - private boolean hasInconsistentColumnMatch(TokenizedLine tl, List columnGrid) { - if (tl.numeric().isEmpty()) return false; - long hits = - tl.numeric().stream() - .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0) - .count(); - return (double) hits / tl.numeric().size() < 0.60; - } - - // ── fragment assembly ──────────────────────────────────────────────────────────────────────── - - private Optional buildFragment( - List group, List columnGrid, int pageNumber, int tableIndex) { - - long anchorCount = - group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count(); - if (anchorCount < MIN_TABLE_ROWS) return Optional.empty(); - - List warnings = new ArrayList<>(); - List> rawRows = new ArrayList<>(); - List rows = new ArrayList<>(); - - for (int rowIdx = 0; rowIdx < group.size(); rowIdx++) { - TokenizedLine tl = group.get(rowIdx); - List rawRow = buildRawRow(tl, columnGrid); - rawRows.add(Collections.unmodifiableList(rawRow)); - rows.add(buildTableRow(rowIdx, tl, rawRow, columnGrid)); - } - - // Column count = 1 label column + confirmed numeric columns - int colCount = columnGrid.size() + 1; - Bounds bounds = computeGroupBounds(group); - float confidence = computeConfidence(group, columnGrid, warnings); - - return Optional.of( - new TableFragment( - "tbl-la-p" + pageNumber + "-" + tableIndex, - pageNumber, - bounds, - List.of(), - Collections.unmodifiableList(rows), - Collections.unmodifiableList(rawRows), - colCount, - confidence, - Collections.unmodifiableList(warnings), - null)); - } - - /** - * Builds a raw row as a list of strings: index 0 = label text, indices 1..N = column values. - */ - private List buildRawRow(TokenizedLine tl, List columnGrid) { - String[] cells = new String[columnGrid.size() + 1]; - Arrays.fill(cells, ""); - - // Separate label tokens (those not landing in any confirmed column) from column tokens. - List labelParts = new ArrayList<>(); - for (LineToken token : tl.all()) { - int col = nearestColumnIndex(token.right(), columnGrid); - if (col >= 0 && token.numeric()) { - int cellIdx = col + 1; - cells[cellIdx] = - cells[cellIdx].isEmpty() - ? token.text() - : cells[cellIdx] + " " + token.text(); - } else { - labelParts.add(token.text()); - } - } - cells[0] = String.join(" ", labelParts).trim(); - return Arrays.asList(cells); - } - - private TableRow buildTableRow( - int rowIdx, TokenizedLine tl, List rawRow, List columnGrid) { - List cells = new ArrayList<>(rawRow.size()); - - // Label cell: use the line's full bounds as an approximation. - cells.add(TableCell.of(0, rawRow.getFirst(), tl.line().bounds())); - - for (int col = 0; col < columnGrid.size(); col++) { - String text = col + 1 < rawRow.size() ? rawRow.get(col + 1) : ""; - float right = columnGrid.get(col); - float left = col > 0 ? columnGrid.get(col - 1) : right - 50f; - Bounds cellBounds = - new Bounds( - left, - tl.line().bounds().y(), - right - left, - tl.line().bounds().height()); - cells.add(TableCell.of(col + 1, text, cellBounds)); - } - return new TableRow(rowIdx, Collections.unmodifiableList(cells)); - } - - // ── confidence scoring ─────────────────────────────────────────────────────────────────────── - - /** - * Heuristic score in [0.0, 0.85] (ceiling keeps results below Tabula lattice which starts at - * 1.0). Base 0.70; +0.05/col beyond 2 (max +0.10); +0.05 at ≥5 anchors, +0.05 at ≥8; −0.15 if - * >30 % of anchors have inconsistent columns; −0.10 if non-anchors outnumber anchors. - */ - private float computeConfidence( - List group, List columnGrid, List warnings) { - float score = 0.70f; - - long anchorCount = - group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count(); - long totalRows = group.size(); - - // More columns - int extraCols = Math.min(columnGrid.size() - MIN_COLUMNS, 2); - score += extraCols * 0.05f; - - // More anchor rows - if (anchorCount >= 5) score += 0.05f; - if (anchorCount >= 8) score += 0.05f; - - // Inconsistent column matching - long inconsistent = - group.stream() - .filter(TokenizedLine::isAnchor) - .filter(tl -> hasInconsistentColumnMatch(tl, columnGrid)) - .count(); - if (inconsistent > anchorCount * 0.30) { - score -= 0.15f; - warnings.add( - "Column match inconsistent on " - + inconsistent - + "/" - + anchorCount - + " anchor rows"); - } - - // Label-heavy - long nonAnchor = totalRows - anchorCount; - if (nonAnchor > anchorCount) { - score -= 0.10f; - warnings.add( - "Non-anchor rows (" - + nonAnchor - + ") outnumber anchor rows (" - + anchorCount - + ")"); - } - - return Math.max(0f, Math.min(0.85f, score)); - } - - // ── utility ────────────────────────────────────────────────────────────────────────────────── - - /** - * Returns the grid index nearest to {@code rightEdge}, or -1 if none is within {@value - * #COLUMN_MATCH_BUCKETS} buckets. - */ - private int nearestColumnIndex(float rightEdge, List grid) { - int nearest = -1; - float minDist = COLUMN_MATCH_BUCKETS * COLUMN_BUCKET_PT + 1f; - for (int i = 0; i < grid.size(); i++) { - float dist = Math.abs(rightEdge - grid.get(i)); - if (dist < minDist) { - minDist = dist; - nearest = i; - } - } - return nearest; - } - - private Bounds computeGroupBounds(List group) { - return group.stream() - .map(tl -> tl.line().bounds()) - .reduce(Bounds::merge) - .orElse(new Bounds(0, 0, 0, 0)); - } - - /** Modal gap between consecutive line edges, used to calibrate the group-split threshold. */ - private float computeModalSpacing(List lines) { - if (lines.size() < 2) return 0f; - Map freq = new HashMap<>(); - for (int i = 1; i < lines.size(); i++) { - float gap = lines.get(i).bounds().y() - lines.get(i - 1).bounds().bottom(); - if (gap > 0) freq.merge(Math.round(gap / 2f) * 2f, 1L, Long::sum); - } - return freq.entrySet().stream() - .max(Map.Entry.comparingByValue()) - .map(Map.Entry::getKey) - .orElse(0f); - } - - private static int bucket(float x) { - return Math.round(x / COLUMN_BUCKET_PT); - } - - // ── private data types ─────────────────────────────────────────────────────────────────────── - - /** A word-level token with an approximate right-edge x-position. */ - record LineToken(String text, float x, float right, boolean numeric) {} - - /** A {@link RawLine} with tokens pre-computed; an "anchor" has ≥ 2 numeric tokens. */ - record TokenizedLine(RawLine line, List all, List numeric) { - boolean isAnchor() { - return numeric.size() >= 2; - } - } -} diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java deleted file mode 100644 index 35715ef37e..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java +++ /dev/null @@ -1,139 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.util.ArrayList; -import java.util.Comparator; -import java.util.List; - -import org.springframework.stereotype.Service; - -import lombok.extern.slf4j.Slf4j; - -/** - * Groups {@link TextFragment} objects into visual {@link RawLine}s using baseline proximity. - * - *

Fragments are on the same line when their baselines are within a font-size-derived tolerance. - * A new line starts whenever the horizontal gap exceeds an adaptive column-gap threshold ({@code - * max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT)}), splitting two-column text. - */ -@Service -@Slf4j -public class LineBuilder { - - /** Baseline tolerance as a fraction of font size; 0.5 keeps mixed-size text on one line. */ - private static final float BASELINE_TOLERANCE_FACTOR = 0.5f; - - /** Absolute minimum tolerance so tiny font sizes don't collapse multi-line content. */ - private static final float MIN_BASELINE_TOLERANCE = 2f; - - /** - * Column-gap threshold as a fraction of page width; 0.10 clears tab stops but stays below - * two-column gutters. - */ - static final float COLUMN_GAP_RATIO = 0.10f; - - /** Floor for the column-gap threshold so narrow pages don't over-split lines. */ - static final float COLUMN_GAP_MIN_PT = 40f; - - public List build(List fragments, int pageNumber) { - if (fragments.isEmpty()) return List.of(); - - float effectiveWidth = inferEffectiveWidth(fragments); - float columnGapThreshold = Math.max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT); - log.debug( - "LineBuilder page {}: effectiveWidth={:.1f}pt, columnGapThreshold={:.1f}pt", - pageNumber, - effectiveWidth, - columnGapThreshold); - - // Sort top-to-bottom first, then left-to-right within the same baseline band. - List sorted = - fragments.stream() - .sorted( - Comparator.comparingDouble(TextFragment::baseline) - .thenComparingDouble(f -> f.bounds().x())) - .toList(); - - List> groups = groupByBaseline(sorted, columnGapThreshold); - - List lines = new ArrayList<>(groups.size()); - for (int i = 0; i < groups.size(); i++) { - List group = - groups.get(i).stream() - .sorted(Comparator.comparingDouble(f -> f.bounds().x())) - .toList(); - - Bounds lineBounds = - group.stream() - .map(TextFragment::bounds) - .reduce(Bounds::merge) - .orElse(new Bounds(0, 0, 0, 0)); - - lines.add(new RawLine("ln-p" + pageNumber + "-" + i, group, lineBounds, pageNumber)); - } - return lines; - } - - private List> groupByBaseline( - List sorted, float columnGapThreshold) { - List> groups = new ArrayList<>(); - List current = new ArrayList<>(); - float currentBaseline = Float.NaN; - - for (TextFragment fragment : sorted) { - if (current.isEmpty()) { - current.add(fragment); - currentBaseline = fragment.baseline(); - continue; - } - - float maxFontSize = - Math.max( - fragment.fontSize(), - (float) - current.stream() - .mapToDouble(TextFragment::fontSize) - .max() - .orElse(0)); - float tolerance = - Math.max(maxFontSize * BASELINE_TOLERANCE_FACTOR, MIN_BASELINE_TOLERANCE); - - boolean sameBaseline = Math.abs(fragment.baseline() - currentBaseline) <= tolerance; - boolean columnGap = sameBaseline && hasColumnGap(fragment, current, columnGapThreshold); - - if (sameBaseline && !columnGap) { - current.add(fragment); - // Anchor to the weighted mean baseline so long lines stay stable. - currentBaseline = - (currentBaseline * (current.size() - 1) + fragment.baseline()) - / current.size(); - } else { - groups.add(current); - current = new ArrayList<>(); - current.add(fragment); - currentBaseline = fragment.baseline(); - } - } - - if (!current.isEmpty()) groups.add(current); - return groups; - } - - /** - * True when the gap from the rightmost fragment in {@code group} to {@code next} exceeds {@code - * threshold}. - */ - private static boolean hasColumnGap( - TextFragment next, List group, float threshold) { - float lastRight = group.getLast().bounds().right(); - return next.bounds().x() - lastRight > threshold; - } - - /** Infers effective page width from the rightmost fragment right-edge plus a 10 % margin. */ - private static float inferEffectiveWidth(List fragments) { - double maxRight = - fragments.stream().mapToDouble(f -> f.bounds().right()).max().orElse(500.0); - return (float) maxRight * 1.10f; - } -}