diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java
deleted file mode 100644
index df3feaa08e..0000000000
--- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java
+++ /dev/null
@@ -1,526 +0,0 @@
-package stirling.software.SPDF.pdf.parser;
-
-import static stirling.software.SPDF.pdf.parser.PdfModels.*;
-
-import java.io.IOException;
-import java.util.ArrayList;
-import java.util.Arrays;
-import java.util.Collections;
-import java.util.Comparator;
-import java.util.HashMap;
-import java.util.List;
-import java.util.Map;
-import java.util.Optional;
-import java.util.TreeMap;
-import java.util.regex.Pattern;
-
-import org.apache.pdfbox.pdmodel.PDDocument;
-import org.springframework.stereotype.Service;
-
-import lombok.extern.slf4j.Slf4j;
-
-/**
- * Fallback {@link TableParser} for borderless financial tables using text geometry.
- *
- *
Identifies "anchor lines" (≥2 numeric tokens), builds a column grid from their right-edge
- * positions, groups vertically proximate anchor lines into table candidates, then scores each group
- * on column consistency and anchor density (confidence ceiling 0.85).
- */
-@Service
-@Slf4j
-public class LineAlignmentTableParser implements TableParser {
-
- /** Width in points of each column position bucket. */
- static final float COLUMN_BUCKET_PT = 5f;
-
- /** Tolerance in buckets when matching a token's right-edge to a confirmed column position. */
- private static final int COLUMN_MATCH_BUCKETS = 2;
-
- /** Maximum gap (as a multiple of modal line spacing) before splitting a group. */
- private static final float MAX_GAP_FACTOR = 2.5f;
-
- /** Minimum anchor rows (numeric-heavy) to form a valid table. */
- static final int MIN_TABLE_ROWS = 3;
-
- /** Minimum confirmed column positions to form a valid table. */
- static final int MIN_COLUMNS = 2;
-
- /**
- * Min fraction of anchor lines a column must appear on to be confirmed (permissive for N/A
- * rows).
- */
- private static final double COLUMN_MIN_FREQUENCY = 0.40;
-
- /**
- * Matches financial numeric tokens: integers, decimals, parenthetical negatives, currency,
- * percent, nil dashes.
- */
- private static final Pattern NUMERIC =
- Pattern.compile("^[\\(\\-\\$£€¥]?\\d[\\d,\\.]*[\\)%]?$|^[-–—]$");
-
- /**
- * Lines within this y-distance are merged into one row (restores rows split by LineBuilder's
- * column-gap logic).
- */
- static final float ROW_MERGE_TOLERANCE_PT = 2f;
-
- // ── public API ───────────────────────────────────────────────────────────────────────────────
-
- @Override
- public List parse(PDDocument document, RawPage rawPage) throws IOException {
- List lines = rawPage.lines();
- if (lines.size() < MIN_TABLE_ROWS) return List.of();
-
- float modalSpacing = computeModalSpacing(lines);
- List tokenized =
- mergeCoincidentLines(lines.stream().map(this::tokenize).toList());
-
- List anchors = tokenized.stream().filter(TokenizedLine::isAnchor).toList();
-
- if (anchors.size() < MIN_TABLE_ROWS) return List.of();
-
- List columnGrid = buildColumnGrid(anchors);
- if (columnGrid.size() < MIN_COLUMNS) {
- log.debug(
- "Page {}: LineAlignment — fewer than {} confirmed columns, skipping",
- rawPage.pageNumber(),
- MIN_COLUMNS);
- return List.of();
- }
-
- List> groups = groupRows(tokenized, columnGrid, modalSpacing);
-
- List results = new ArrayList<>();
- for (int i = 0; i < groups.size(); i++) {
- buildFragment(groups.get(i), columnGrid, rawPage.pageNumber(), i)
- .ifPresent(results::add);
- }
-
- log.debug(
- "Page {}: LineAlignment detected {} table(s) ({} anchor lines, {} columns)",
- rawPage.pageNumber(),
- results.size(),
- anchors.size(),
- columnGrid.size());
- return results;
- }
-
- // ── coincident-line merging ──────────────────────────────────────────────────────────────────
-
- /**
- * Merges tokenised lines sharing the same y-position into one row, rejoining label/value halves
- * split by LineBuilder.
- */
- List mergeCoincidentLines(List tokenized) {
- if (tokenized.size() < 2) return tokenized;
-
- List result = new ArrayList<>();
- int i = 0;
-
- while (i < tokenized.size()) {
- float baseY = tokenized.get(i).line().bounds().y();
- int j = i + 1;
- while (j < tokenized.size()
- && Math.abs(tokenized.get(j).line().bounds().y() - baseY)
- <= ROW_MERGE_TOLERANCE_PT) {
- j++;
- }
-
- if (j == i + 1) {
- result.add(tokenized.get(i));
- } else {
- result.add(mergeGroup(tokenized.subList(i, j)));
- }
- i = j;
- }
-
- return result;
- }
-
- private TokenizedLine mergeGroup(List group) {
- List mergedFragments =
- group.stream()
- .flatMap(tl -> tl.line().fragments().stream())
- .sorted(Comparator.comparingDouble(f -> f.bounds().x()))
- .toList();
-
- Bounds mergedBounds =
- group.stream()
- .map(tl -> tl.line().bounds())
- .reduce(Bounds::merge)
- .orElse(group.getFirst().line().bounds());
-
- RawLine mergedLine =
- new RawLine(
- group.getFirst().line().lineId(),
- mergedFragments,
- mergedBounds,
- group.getFirst().line().pageNumber());
-
- return tokenize(mergedLine);
- }
-
- // ── tokenisation ─────────────────────────────────────────────────────────────────────────────
-
- /**
- * Splits fragments into word-level tokens; x-positions are estimated linearly within each
- * fragment.
- */
- TokenizedLine tokenize(RawLine line) {
- List tokens = new ArrayList<>();
- for (TextFragment frag : line.fragments()) {
- tokens.addAll(tokensFromFragment(frag));
- }
- List numeric = tokens.stream().filter(LineToken::numeric).toList();
- return new TokenizedLine(line, tokens, numeric);
- }
-
- private List tokensFromFragment(TextFragment frag) {
- String raw = frag.text();
- if (raw == null || raw.isBlank()) return List.of();
-
- float fragX = frag.bounds().x();
- float fragWidth = frag.bounds().width();
- int rawLen = raw.length();
-
- List result = new ArrayList<>();
- int offset = 0;
- for (String part : raw.split("\\s+")) {
- if (part.isEmpty()) {
- offset++;
- continue;
- }
- int idx = raw.indexOf(part, offset);
- if (idx < 0) idx = offset;
-
- float tokenX = rawLen > 0 ? fragX + ((float) idx / rawLen) * fragWidth : fragX;
- float tokenRight =
- rawLen > 0
- ? fragX + ((float) (idx + part.length()) / rawLen) * fragWidth
- : fragX + fragWidth;
-
- result.add(new LineToken(part, tokenX, tokenRight, NUMERIC.matcher(part).matches()));
- offset = idx + part.length();
- }
- return result;
- }
-
- // ── column grid ──────────────────────────────────────────────────────────────────────────────
-
- /**
- * Returns confirmed column right-edge positions — those appearing on ≥ {@value
- * #COLUMN_MIN_FREQUENCY} × N anchor lines.
- */
- private List buildColumnGrid(List anchors) {
- // bucket → set of line indices that contributed a numeric token to that bucket
- Map> bucketLines = new HashMap<>();
- for (int i = 0; i < anchors.size(); i++) {
- for (LineToken t : anchors.get(i).numeric()) {
- int bucket = bucket(t.right());
- bucketLines.computeIfAbsent(bucket, k -> new ArrayList<>()).add(i);
- }
- }
-
- int minHits =
- Math.max(MIN_TABLE_ROWS, (int) Math.ceil(anchors.size() * COLUMN_MIN_FREQUENCY));
-
- // Confirmed buckets → average right-edge for that bucket
- TreeMap confirmed = new TreeMap<>();
- for (Map.Entry> entry : bucketLines.entrySet()) {
- // Count distinct lines
- long distinctLines = entry.getValue().stream().distinct().count();
- if (distinctLines >= minHits) {
- double avg =
- entry.getValue().stream()
- .distinct() // weight each line equally regardless of token count
- .mapToDouble(
- lineIdx ->
- avgRightEdgeForBucket(
- anchors, lineIdx, entry.getKey()))
- .average()
- .orElse(entry.getKey() * (double) COLUMN_BUCKET_PT);
- confirmed.put(entry.getKey(), (float) avg);
- }
- }
-
- return new ArrayList<>(confirmed.values()); // already sorted by bucket (left to right)
- }
-
- /**
- * Returns the average right-edge position of tokens in {@code line} whose bucket matches {@code
- * targetBucket}, falling back to the bucket's nominal centre when no tokens match.
- */
- private double avgRightEdgeForBucket(
- List anchors, int lineIdx, int targetBucket) {
- return anchors.get(lineIdx).numeric().stream()
- .filter(t -> bucket(t.right()) == targetBucket)
- .mapToDouble(LineToken::right)
- .average()
- .orElse(targetBucket * (double) COLUMN_BUCKET_PT);
- }
-
- // ── grouping ─────────────────────────────────────────────────────────────────────────────────
-
- /**
- * Groups anchor lines into table candidates, including adjacent label rows; a gap >
- * MAX_GAP_FACTOR × modal spacing splits groups.
- */
- private List> groupRows(
- List all, List columnGrid, float modalSpacing) {
- float maxGap = modalSpacing > 0 ? modalSpacing * MAX_GAP_FACTOR : 30f;
-
- List> groups = new ArrayList<>();
- List current = new ArrayList<>();
-
- for (int i = 0; i < all.size(); i++) {
- TokenizedLine tl = all.get(i);
- boolean fits = tl.isAnchor() && matchesGrid(tl, columnGrid);
-
- if (current.isEmpty()) {
- if (fits) current.add(tl);
- continue;
- }
-
- float gap = tl.line().bounds().y() - current.getLast().line().bounds().bottom();
-
- if (gap > maxGap) {
- groups.add(current);
- current = new ArrayList<>();
- if (fits) current.add(tl);
- continue;
- }
-
- if (fits) {
- current.add(tl);
- } else if (!tl.line().text().isBlank()) {
- // Include non-anchor lines (labels) only if they have text and are within
- // proximity.
- current.add(tl);
- }
- }
-
- if (!current.isEmpty()) groups.add(current);
-
- return groups.stream().filter(g -> hasEnoughAnchorRows(g, columnGrid)).toList();
- }
-
- private boolean hasEnoughAnchorRows(List group, List columnGrid) {
- return group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count()
- >= MIN_TABLE_ROWS;
- }
-
- /** A line "matches" the grid when ≥ 60 % of its numeric tokens land in confirmed columns. */
- private boolean matchesGrid(TokenizedLine tl, List columnGrid) {
- if (tl.numeric().isEmpty()) return false;
- long matches =
- tl.numeric().stream()
- .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0)
- .count();
- return (double) matches / tl.numeric().size() >= 0.60;
- }
-
- private boolean hasInconsistentColumnMatch(TokenizedLine tl, List columnGrid) {
- if (tl.numeric().isEmpty()) return false;
- long hits =
- tl.numeric().stream()
- .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0)
- .count();
- return (double) hits / tl.numeric().size() < 0.60;
- }
-
- // ── fragment assembly ────────────────────────────────────────────────────────────────────────
-
- private Optional buildFragment(
- List group, List columnGrid, int pageNumber, int tableIndex) {
-
- long anchorCount =
- group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count();
- if (anchorCount < MIN_TABLE_ROWS) return Optional.empty();
-
- List warnings = new ArrayList<>();
- List> rawRows = new ArrayList<>();
- List rows = new ArrayList<>();
-
- for (int rowIdx = 0; rowIdx < group.size(); rowIdx++) {
- TokenizedLine tl = group.get(rowIdx);
- List rawRow = buildRawRow(tl, columnGrid);
- rawRows.add(Collections.unmodifiableList(rawRow));
- rows.add(buildTableRow(rowIdx, tl, rawRow, columnGrid));
- }
-
- // Column count = 1 label column + confirmed numeric columns
- int colCount = columnGrid.size() + 1;
- Bounds bounds = computeGroupBounds(group);
- float confidence = computeConfidence(group, columnGrid, warnings);
-
- return Optional.of(
- new TableFragment(
- "tbl-la-p" + pageNumber + "-" + tableIndex,
- pageNumber,
- bounds,
- List.of(),
- Collections.unmodifiableList(rows),
- Collections.unmodifiableList(rawRows),
- colCount,
- confidence,
- Collections.unmodifiableList(warnings),
- null));
- }
-
- /**
- * Builds a raw row as a list of strings: index 0 = label text, indices 1..N = column values.
- */
- private List buildRawRow(TokenizedLine tl, List columnGrid) {
- String[] cells = new String[columnGrid.size() + 1];
- Arrays.fill(cells, "");
-
- // Separate label tokens (those not landing in any confirmed column) from column tokens.
- List labelParts = new ArrayList<>();
- for (LineToken token : tl.all()) {
- int col = nearestColumnIndex(token.right(), columnGrid);
- if (col >= 0 && token.numeric()) {
- int cellIdx = col + 1;
- cells[cellIdx] =
- cells[cellIdx].isEmpty()
- ? token.text()
- : cells[cellIdx] + " " + token.text();
- } else {
- labelParts.add(token.text());
- }
- }
- cells[0] = String.join(" ", labelParts).trim();
- return Arrays.asList(cells);
- }
-
- private TableRow buildTableRow(
- int rowIdx, TokenizedLine tl, List rawRow, List columnGrid) {
- List cells = new ArrayList<>(rawRow.size());
-
- // Label cell: use the line's full bounds as an approximation.
- cells.add(TableCell.of(0, rawRow.getFirst(), tl.line().bounds()));
-
- for (int col = 0; col < columnGrid.size(); col++) {
- String text = col + 1 < rawRow.size() ? rawRow.get(col + 1) : "";
- float right = columnGrid.get(col);
- float left = col > 0 ? columnGrid.get(col - 1) : right - 50f;
- Bounds cellBounds =
- new Bounds(
- left,
- tl.line().bounds().y(),
- right - left,
- tl.line().bounds().height());
- cells.add(TableCell.of(col + 1, text, cellBounds));
- }
- return new TableRow(rowIdx, Collections.unmodifiableList(cells));
- }
-
- // ── confidence scoring ───────────────────────────────────────────────────────────────────────
-
- /**
- * Heuristic score in [0.0, 0.85] (ceiling keeps results below Tabula lattice which starts at
- * 1.0). Base 0.70; +0.05/col beyond 2 (max +0.10); +0.05 at ≥5 anchors, +0.05 at ≥8; −0.15 if
- * >30 % of anchors have inconsistent columns; −0.10 if non-anchors outnumber anchors.
- */
- private float computeConfidence(
- List group, List columnGrid, List warnings) {
- float score = 0.70f;
-
- long anchorCount =
- group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count();
- long totalRows = group.size();
-
- // More columns
- int extraCols = Math.min(columnGrid.size() - MIN_COLUMNS, 2);
- score += extraCols * 0.05f;
-
- // More anchor rows
- if (anchorCount >= 5) score += 0.05f;
- if (anchorCount >= 8) score += 0.05f;
-
- // Inconsistent column matching
- long inconsistent =
- group.stream()
- .filter(TokenizedLine::isAnchor)
- .filter(tl -> hasInconsistentColumnMatch(tl, columnGrid))
- .count();
- if (inconsistent > anchorCount * 0.30) {
- score -= 0.15f;
- warnings.add(
- "Column match inconsistent on "
- + inconsistent
- + "/"
- + anchorCount
- + " anchor rows");
- }
-
- // Label-heavy
- long nonAnchor = totalRows - anchorCount;
- if (nonAnchor > anchorCount) {
- score -= 0.10f;
- warnings.add(
- "Non-anchor rows ("
- + nonAnchor
- + ") outnumber anchor rows ("
- + anchorCount
- + ")");
- }
-
- return Math.max(0f, Math.min(0.85f, score));
- }
-
- // ── utility ──────────────────────────────────────────────────────────────────────────────────
-
- /**
- * Returns the grid index nearest to {@code rightEdge}, or -1 if none is within {@value
- * #COLUMN_MATCH_BUCKETS} buckets.
- */
- private int nearestColumnIndex(float rightEdge, List grid) {
- int nearest = -1;
- float minDist = COLUMN_MATCH_BUCKETS * COLUMN_BUCKET_PT + 1f;
- for (int i = 0; i < grid.size(); i++) {
- float dist = Math.abs(rightEdge - grid.get(i));
- if (dist < minDist) {
- minDist = dist;
- nearest = i;
- }
- }
- return nearest;
- }
-
- private Bounds computeGroupBounds(List group) {
- return group.stream()
- .map(tl -> tl.line().bounds())
- .reduce(Bounds::merge)
- .orElse(new Bounds(0, 0, 0, 0));
- }
-
- /** Modal gap between consecutive line edges, used to calibrate the group-split threshold. */
- private float computeModalSpacing(List lines) {
- if (lines.size() < 2) return 0f;
- Map freq = new HashMap<>();
- for (int i = 1; i < lines.size(); i++) {
- float gap = lines.get(i).bounds().y() - lines.get(i - 1).bounds().bottom();
- if (gap > 0) freq.merge(Math.round(gap / 2f) * 2f, 1L, Long::sum);
- }
- return freq.entrySet().stream()
- .max(Map.Entry.comparingByValue())
- .map(Map.Entry::getKey)
- .orElse(0f);
- }
-
- private static int bucket(float x) {
- return Math.round(x / COLUMN_BUCKET_PT);
- }
-
- // ── private data types ───────────────────────────────────────────────────────────────────────
-
- /** A word-level token with an approximate right-edge x-position. */
- record LineToken(String text, float x, float right, boolean numeric) {}
-
- /** A {@link RawLine} with tokens pre-computed; an "anchor" has ≥ 2 numeric tokens. */
- record TokenizedLine(RawLine line, List all, List numeric) {
- boolean isAnchor() {
- return numeric.size() >= 2;
- }
- }
-}
diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java
deleted file mode 100644
index 35715ef37e..0000000000
--- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java
+++ /dev/null
@@ -1,139 +0,0 @@
-package stirling.software.SPDF.pdf.parser;
-
-import static stirling.software.SPDF.pdf.parser.PdfModels.*;
-
-import java.util.ArrayList;
-import java.util.Comparator;
-import java.util.List;
-
-import org.springframework.stereotype.Service;
-
-import lombok.extern.slf4j.Slf4j;
-
-/**
- * Groups {@link TextFragment} objects into visual {@link RawLine}s using baseline proximity.
- *
- * Fragments are on the same line when their baselines are within a font-size-derived tolerance.
- * A new line starts whenever the horizontal gap exceeds an adaptive column-gap threshold ({@code
- * max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT)}), splitting two-column text.
- */
-@Service
-@Slf4j
-public class LineBuilder {
-
- /** Baseline tolerance as a fraction of font size; 0.5 keeps mixed-size text on one line. */
- private static final float BASELINE_TOLERANCE_FACTOR = 0.5f;
-
- /** Absolute minimum tolerance so tiny font sizes don't collapse multi-line content. */
- private static final float MIN_BASELINE_TOLERANCE = 2f;
-
- /**
- * Column-gap threshold as a fraction of page width; 0.10 clears tab stops but stays below
- * two-column gutters.
- */
- static final float COLUMN_GAP_RATIO = 0.10f;
-
- /** Floor for the column-gap threshold so narrow pages don't over-split lines. */
- static final float COLUMN_GAP_MIN_PT = 40f;
-
- public List build(List fragments, int pageNumber) {
- if (fragments.isEmpty()) return List.of();
-
- float effectiveWidth = inferEffectiveWidth(fragments);
- float columnGapThreshold = Math.max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT);
- log.debug(
- "LineBuilder page {}: effectiveWidth={:.1f}pt, columnGapThreshold={:.1f}pt",
- pageNumber,
- effectiveWidth,
- columnGapThreshold);
-
- // Sort top-to-bottom first, then left-to-right within the same baseline band.
- List sorted =
- fragments.stream()
- .sorted(
- Comparator.comparingDouble(TextFragment::baseline)
- .thenComparingDouble(f -> f.bounds().x()))
- .toList();
-
- List> groups = groupByBaseline(sorted, columnGapThreshold);
-
- List lines = new ArrayList<>(groups.size());
- for (int i = 0; i < groups.size(); i++) {
- List group =
- groups.get(i).stream()
- .sorted(Comparator.comparingDouble(f -> f.bounds().x()))
- .toList();
-
- Bounds lineBounds =
- group.stream()
- .map(TextFragment::bounds)
- .reduce(Bounds::merge)
- .orElse(new Bounds(0, 0, 0, 0));
-
- lines.add(new RawLine("ln-p" + pageNumber + "-" + i, group, lineBounds, pageNumber));
- }
- return lines;
- }
-
- private List> groupByBaseline(
- List sorted, float columnGapThreshold) {
- List> groups = new ArrayList<>();
- List current = new ArrayList<>();
- float currentBaseline = Float.NaN;
-
- for (TextFragment fragment : sorted) {
- if (current.isEmpty()) {
- current.add(fragment);
- currentBaseline = fragment.baseline();
- continue;
- }
-
- float maxFontSize =
- Math.max(
- fragment.fontSize(),
- (float)
- current.stream()
- .mapToDouble(TextFragment::fontSize)
- .max()
- .orElse(0));
- float tolerance =
- Math.max(maxFontSize * BASELINE_TOLERANCE_FACTOR, MIN_BASELINE_TOLERANCE);
-
- boolean sameBaseline = Math.abs(fragment.baseline() - currentBaseline) <= tolerance;
- boolean columnGap = sameBaseline && hasColumnGap(fragment, current, columnGapThreshold);
-
- if (sameBaseline && !columnGap) {
- current.add(fragment);
- // Anchor to the weighted mean baseline so long lines stay stable.
- currentBaseline =
- (currentBaseline * (current.size() - 1) + fragment.baseline())
- / current.size();
- } else {
- groups.add(current);
- current = new ArrayList<>();
- current.add(fragment);
- currentBaseline = fragment.baseline();
- }
- }
-
- if (!current.isEmpty()) groups.add(current);
- return groups;
- }
-
- /**
- * True when the gap from the rightmost fragment in {@code group} to {@code next} exceeds {@code
- * threshold}.
- */
- private static boolean hasColumnGap(
- TextFragment next, List group, float threshold) {
- float lastRight = group.getLast().bounds().right();
- return next.bounds().x() - lastRight > threshold;
- }
-
- /** Infers effective page width from the rightmost fragment right-edge plus a 10 % margin. */
- private static float inferEffectiveWidth(List fragments) {
- double maxRight =
- fragments.stream().mapToDouble(f -> f.bounds().right()).max().orElse(500.0);
- return (float) maxRight * 1.10f;
- }
-}