From 5fca2f199ae5ba2ccae3ed93a09372d6948a9c59 Mon Sep 17 00:00:00 2001 From: EthanHealy01 <80844253+EthanHealy01@users.noreply.github.com> Date: Wed, 10 Jun 2026 14:51:41 +0100 Subject: [PATCH] Feature/pdf ingestion jpdfium (#6525) --- .../SPDF/pdf/parser/CompositeTableParser.java | 73 -- .../pdf/parser/LineAlignmentTableParser.java | 528 --------- .../software/SPDF/pdf/parser/LineBuilder.java | 139 --- .../software/SPDF/pdf/parser/PdfIngester.java | 79 -- .../pdf/parser/WordExtractingStripper.java | 113 -- .../software/common/pdf/HeadingDetector.java | 191 +++ .../common/pdf/PdfMarkdownConverter.java | 1043 +++++++++++++++++ .../software/common/pdf/TableRenderer.java | 82 ++ .../parser/LineAlignmentTableParserTest.java | 153 --- .../common/pdf/PdfMarkdownConverterTest.java | 269 +++++ .../bordered-table-test_widget.md | 10 + .../bordered-table-test_widget.pdf | 74 ++ .../many-tables-test_stress.md | 222 ++++ .../many-tables-test_stress.pdf | 169 +++ .../multi-column-test_lorem.md | 25 + .../multi-column-test_lorem.pdf | 74 ++ .../wrapped-cell-test_expense-report.md | 62 + .../wrapped-cell-test_expense-report.pdf | Bin 0 -> 95841 bytes .../api/converters/ConvertPDFToMarkdown.java | 32 +- .../converters/ConvertPDFToMarkdownTest.java | 94 +- .../model/api/ai/AiWorkflowOutcome.java | 3 +- .../service/AiWorkflowService.java | 84 +- .../service/PdfContentExtractor.java | 78 -- .../service/AiWorkflowServiceTest.java | 29 + .../PageLayoutArtifactContractTest.java | 66 -- engine/src/stirling/agents/__init__.py | 2 - engine/src/stirling/agents/orchestrator.py | 35 +- .../agents/pdf_to_markdown/__init__.py | 3 - .../stirling/agents/pdf_to_markdown/agent.py | 435 ------- engine/src/stirling/contracts/__init__.py | 22 +- engine/src/stirling/contracts/common.py | 14 + engine/src/stirling/contracts/orchestrator.py | 5 +- .../src/stirling/contracts/pdf_to_markdown.py | 105 -- engine/tests/test_pdf_to_markdown.py | 138 --- testing/cucumber/features/external.feature | 3 +- testing/test.sh | 6 +- 36 files changed, 2439 insertions(+), 2021 deletions(-) delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/CompositeTableParser.java delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/PdfIngester.java delete mode 100644 app/common/src/main/java/stirling/software/SPDF/pdf/parser/WordExtractingStripper.java create mode 100644 app/common/src/main/java/stirling/software/common/pdf/HeadingDetector.java create mode 100644 app/common/src/main/java/stirling/software/common/pdf/PdfMarkdownConverter.java create mode 100644 app/common/src/main/java/stirling/software/common/pdf/TableRenderer.java delete mode 100644 app/common/src/test/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParserTest.java create mode 100644 app/common/src/test/java/stirling/software/common/pdf/PdfMarkdownConverterTest.java create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.md create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.pdf create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.md create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.pdf create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.md create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.pdf create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.md create mode 100644 app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.pdf delete mode 100644 app/proprietary/src/test/java/stirling/software/proprietary/service/PageLayoutArtifactContractTest.java delete mode 100644 engine/src/stirling/agents/pdf_to_markdown/__init__.py delete mode 100644 engine/src/stirling/agents/pdf_to_markdown/agent.py delete mode 100644 engine/src/stirling/contracts/pdf_to_markdown.py delete mode 100644 engine/tests/test_pdf_to_markdown.py diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/CompositeTableParser.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/CompositeTableParser.java deleted file mode 100644 index 429f180f3e..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/CompositeTableParser.java +++ /dev/null @@ -1,73 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.io.IOException; -import java.util.List; - -import org.apache.pdfbox.pdmodel.PDDocument; -import org.springframework.context.annotation.Primary; -import org.springframework.stereotype.Service; - -import lombok.RequiredArgsConstructor; -import lombok.extern.slf4j.Slf4j; - -/** - * Chains table parsers in priority order: Tabula lattice → Tabula stream → {@link - * LineAlignmentTableParser}. The first parser returning a result above {@link - * #TABULA_CONFIDENCE_THRESHOLD} wins; results from different parsers are never mixed on one page. - */ -@Service -@Primary -@RequiredArgsConstructor -@Slf4j -public class CompositeTableParser implements TableParser { - - /** Min Tabula confidence to accept results; below this LineAlignment is tried instead. */ - static final float TABULA_CONFIDENCE_THRESHOLD = 0.5f; - - private final TabulaTableParser tabulaParser; - private final LineAlignmentTableParser lineAlignmentParser; - - @Override - public List parse(PDDocument document, RawPage rawPage) throws IOException { - // Step 1: Tabula lattice mode (ruled/bordered tables). - List latticeResults = filterConfident(tabulaParser.parse(document, rawPage)); - if (!latticeResults.isEmpty()) { - log.debug( - "Page {}: using Tabula lattice ({} table(s))", - rawPage.pageNumber(), - latticeResults.size()); - return latticeResults; - } - - // Step 2: Tabula stream mode (borderless/whitespace-delimited tables). - // parseStream is not on the TableParser interface — this intentionally couples to the - // concrete TabulaTableParser since stream mode is a Tabula-specific concept. - List streamResults = - filterConfident(tabulaParser.parseStream(document, rawPage)); - if (!streamResults.isEmpty()) { - log.debug( - "Page {}: using Tabula stream ({} table(s))", - rawPage.pageNumber(), - streamResults.size()); - return streamResults; - } - - // Step 3: Geometry-based line-alignment fallback. - List lineResults = lineAlignmentParser.parse(document, rawPage); - if (!lineResults.isEmpty()) { - log.debug( - "Page {}: using LineAlignment ({} table(s))", - rawPage.pageNumber(), - lineResults.size()); - return lineResults; - } - - return List.of(); - } - - private List filterConfident(List tables) { - return tables.stream().filter(t -> t.confidence() >= TABULA_CONFIDENCE_THRESHOLD).toList(); - } -} diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java deleted file mode 100644 index b2d8de5167..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParser.java +++ /dev/null @@ -1,528 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.io.IOException; -import java.util.ArrayList; -import java.util.Arrays; -import java.util.Collections; -import java.util.Comparator; -import java.util.HashMap; -import java.util.List; -import java.util.Map; -import java.util.Optional; -import java.util.TreeMap; -import java.util.regex.Pattern; - -import org.apache.pdfbox.pdmodel.PDDocument; -import org.springframework.stereotype.Service; - -import lombok.extern.slf4j.Slf4j; - -/** - * Fallback {@link TableParser} for borderless financial tables using text geometry. - * - *

Identifies "anchor lines" (≥2 numeric tokens), builds a column grid from their right-edge - * positions, groups vertically proximate anchor lines into table candidates, then scores each group - * on column consistency and anchor density (confidence ceiling 0.85). - */ -@Service -@Slf4j -public class LineAlignmentTableParser implements TableParser { - - /** Width in points of each column position bucket. */ - static final float COLUMN_BUCKET_PT = 5f; - - /** Tolerance in buckets when matching a token's right-edge to a confirmed column position. */ - private static final int COLUMN_MATCH_BUCKETS = 2; - - /** Maximum gap (as a multiple of modal line spacing) before splitting a group. */ - private static final float MAX_GAP_FACTOR = 2.5f; - - /** Minimum anchor rows (numeric-heavy) to form a valid table. */ - static final int MIN_TABLE_ROWS = 3; - - /** Minimum confirmed column positions to form a valid table. */ - static final int MIN_COLUMNS = 2; - - /** - * Min fraction of anchor lines a column must appear on to be confirmed (permissive for N/A - * rows). - */ - private static final double COLUMN_MIN_FREQUENCY = 0.40; - - /** - * Matches financial numeric tokens: integers, decimals, parenthetical negatives, currency, - * percent, nil dashes. - */ - private static final Pattern NUMERIC = - Pattern.compile("^[\\(\\-\\$£€¥]?\\d[\\d,\\.]*[\\)%]?$|^[-–—]$"); - - /** - * Lines within this y-distance are merged into one row (restores rows split by LineBuilder's - * column-gap logic). - */ - static final float ROW_MERGE_TOLERANCE_PT = 2f; - - // ── public API ─────────────────────────────────────────────────────────────────────────────── - - @Override - public List parse(PDDocument document, RawPage rawPage) throws IOException { - List lines = rawPage.lines(); - if (lines.size() < MIN_TABLE_ROWS) return List.of(); - - float modalSpacing = computeModalSpacing(lines); - List tokenized = - mergeCoincidentLines(lines.stream().map(this::tokenize).toList()); - - List anchors = tokenized.stream().filter(TokenizedLine::isAnchor).toList(); - - if (anchors.size() < MIN_TABLE_ROWS) return List.of(); - - List columnGrid = buildColumnGrid(anchors); - if (columnGrid.size() < MIN_COLUMNS) { - log.debug( - "Page {}: LineAlignment — fewer than {} confirmed columns, skipping", - rawPage.pageNumber(), - MIN_COLUMNS); - return List.of(); - } - - List> groups = groupRows(tokenized, columnGrid, modalSpacing); - - List results = new ArrayList<>(); - for (int i = 0; i < groups.size(); i++) { - buildFragment(groups.get(i), columnGrid, rawPage.pageNumber(), i) - .ifPresent(results::add); - } - - log.debug( - "Page {}: LineAlignment detected {} table(s) ({} anchor lines, {} columns)", - rawPage.pageNumber(), - results.size(), - anchors.size(), - columnGrid.size()); - return results; - } - - // ── coincident-line merging ────────────────────────────────────────────────────────────────── - - /** - * Merges tokenised lines sharing the same y-position into one row, rejoining label/value halves - * split by LineBuilder. - */ - List mergeCoincidentLines(List tokenized) { - if (tokenized.size() < 2) return tokenized; - - List result = new ArrayList<>(); - int i = 0; - - while (i < tokenized.size()) { - float baseY = tokenized.get(i).line().bounds().y(); - int j = i + 1; - while (j < tokenized.size() - && Math.abs(tokenized.get(j).line().bounds().y() - baseY) - <= ROW_MERGE_TOLERANCE_PT) { - j++; - } - - if (j == i + 1) { - result.add(tokenized.get(i)); - } else { - result.add(mergeGroup(tokenized.subList(i, j))); - } - i = j; - } - - return result; - } - - private TokenizedLine mergeGroup(List group) { - List mergedFragments = - group.stream() - .flatMap(tl -> tl.line().fragments().stream()) - .sorted(Comparator.comparingDouble(f -> f.bounds().x())) - .toList(); - - Bounds mergedBounds = - group.stream() - .map(tl -> tl.line().bounds()) - .reduce(Bounds::merge) - .orElse(group.get(0).line().bounds()); - - RawLine mergedLine = - new RawLine( - group.get(0).line().lineId(), - mergedFragments, - mergedBounds, - group.get(0).line().pageNumber()); - - return tokenize(mergedLine); - } - - // ── tokenisation ───────────────────────────────────────────────────────────────────────────── - - /** - * Splits fragments into word-level tokens; x-positions are estimated linearly within each - * fragment. - */ - TokenizedLine tokenize(RawLine line) { - List tokens = new ArrayList<>(); - for (TextFragment frag : line.fragments()) { - tokens.addAll(tokensFromFragment(frag)); - } - List numeric = tokens.stream().filter(LineToken::numeric).toList(); - return new TokenizedLine(line, tokens, numeric); - } - - private List tokensFromFragment(TextFragment frag) { - String raw = frag.text(); - if (raw == null || raw.isBlank()) return List.of(); - - float fragX = frag.bounds().x(); - float fragWidth = frag.bounds().width(); - int rawLen = raw.length(); - - List result = new ArrayList<>(); - int offset = 0; - for (String part : raw.split("\\s+")) { - if (part.isEmpty()) { - offset++; - continue; - } - int idx = raw.indexOf(part, offset); - if (idx < 0) idx = offset; - - float tokenX = rawLen > 0 ? fragX + ((float) idx / rawLen) * fragWidth : fragX; - float tokenRight = - rawLen > 0 - ? fragX + ((float) (idx + part.length()) / rawLen) * fragWidth - : fragX + fragWidth; - - result.add(new LineToken(part, tokenX, tokenRight, NUMERIC.matcher(part).matches())); - offset = idx + part.length(); - } - return result; - } - - // ── column grid ────────────────────────────────────────────────────────────────────────────── - - /** - * Returns confirmed column right-edge positions — those appearing on ≥ {@value - * #COLUMN_MIN_FREQUENCY} × N anchor lines. - */ - private List buildColumnGrid(List anchors) { - // bucket → set of line indices that contributed a numeric token to that bucket - Map> bucketLines = new HashMap<>(); - for (int i = 0; i < anchors.size(); i++) { - for (LineToken t : anchors.get(i).numeric()) { - int bucket = bucket(t.right()); - bucketLines.computeIfAbsent(bucket, k -> new ArrayList<>()).add(i); - } - } - - int minHits = - Math.max(MIN_TABLE_ROWS, (int) Math.ceil(anchors.size() * COLUMN_MIN_FREQUENCY)); - - // Confirmed buckets → average right-edge for that bucket - TreeMap confirmed = new TreeMap<>(); - for (Map.Entry> entry : bucketLines.entrySet()) { - // Count distinct lines - long distinctLines = entry.getValue().stream().distinct().count(); - if (distinctLines >= minHits) { - double avg = - entry.getValue().stream() - .distinct() // weight each line equally regardless of token count - .mapToDouble( - lineIdx -> - avgRightEdgeForBucket( - anchors, lineIdx, entry.getKey())) - .average() - .orElse(entry.getKey() * (double) COLUMN_BUCKET_PT); - confirmed.put(entry.getKey(), (float) avg); - } - } - - return new ArrayList<>(confirmed.values()); // already sorted by bucket (left to right) - } - - /** - * Returns the average right-edge position of tokens in {@code line} whose bucket matches {@code - * targetBucket}, falling back to the bucket's nominal centre when no tokens match. - */ - private double avgRightEdgeForBucket( - List anchors, int lineIdx, int targetBucket) { - return anchors.get(lineIdx).numeric().stream() - .filter(t -> bucket(t.right()) == targetBucket) - .mapToDouble(LineToken::right) - .average() - .orElse(targetBucket * (double) COLUMN_BUCKET_PT); - } - - // ── grouping ───────────────────────────────────────────────────────────────────────────────── - - /** - * Groups anchor lines into table candidates, including adjacent label rows; a gap > - * MAX_GAP_FACTOR × modal spacing splits groups. - */ - private List> groupRows( - List all, List columnGrid, float modalSpacing) { - float maxGap = modalSpacing > 0 ? modalSpacing * MAX_GAP_FACTOR : 30f; - - List> groups = new ArrayList<>(); - List current = new ArrayList<>(); - - for (int i = 0; i < all.size(); i++) { - TokenizedLine tl = all.get(i); - boolean fits = tl.isAnchor() && matchesGrid(tl, columnGrid); - - if (current.isEmpty()) { - if (fits) current.add(tl); - continue; - } - - float gap = - tl.line().bounds().y() - - current.get(current.size() - 1).line().bounds().bottom(); - - if (gap > maxGap) { - groups.add(current); - current = new ArrayList<>(); - if (fits) current.add(tl); - continue; - } - - if (fits) { - current.add(tl); - } else if (!tl.line().text().isBlank()) { - // Include non-anchor lines (labels) only if they have text and are within - // proximity. - current.add(tl); - } - } - - if (!current.isEmpty()) groups.add(current); - - return groups.stream().filter(g -> hasEnoughAnchorRows(g, columnGrid)).toList(); - } - - private boolean hasEnoughAnchorRows(List group, List columnGrid) { - return group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count() - >= MIN_TABLE_ROWS; - } - - /** A line "matches" the grid when ≥ 60 % of its numeric tokens land in confirmed columns. */ - private boolean matchesGrid(TokenizedLine tl, List columnGrid) { - if (tl.numeric().isEmpty()) return false; - long matches = - tl.numeric().stream() - .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0) - .count(); - return (double) matches / tl.numeric().size() >= 0.60; - } - - private boolean hasInconsistentColumnMatch(TokenizedLine tl, List columnGrid) { - if (tl.numeric().isEmpty()) return false; - long hits = - tl.numeric().stream() - .filter(t -> nearestColumnIndex(t.right(), columnGrid) >= 0) - .count(); - return (double) hits / tl.numeric().size() < 0.60; - } - - // ── fragment assembly ──────────────────────────────────────────────────────────────────────── - - private Optional buildFragment( - List group, List columnGrid, int pageNumber, int tableIndex) { - - long anchorCount = - group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count(); - if (anchorCount < MIN_TABLE_ROWS) return Optional.empty(); - - List warnings = new ArrayList<>(); - List> rawRows = new ArrayList<>(); - List rows = new ArrayList<>(); - - for (int rowIdx = 0; rowIdx < group.size(); rowIdx++) { - TokenizedLine tl = group.get(rowIdx); - List rawRow = buildRawRow(tl, columnGrid); - rawRows.add(Collections.unmodifiableList(rawRow)); - rows.add(buildTableRow(rowIdx, tl, rawRow, columnGrid)); - } - - // Column count = 1 label column + confirmed numeric columns - int colCount = columnGrid.size() + 1; - Bounds bounds = computeGroupBounds(group); - float confidence = computeConfidence(group, columnGrid, warnings); - - return Optional.of( - new TableFragment( - "tbl-la-p" + pageNumber + "-" + tableIndex, - pageNumber, - bounds, - List.of(), - Collections.unmodifiableList(rows), - Collections.unmodifiableList(rawRows), - colCount, - confidence, - Collections.unmodifiableList(warnings), - null)); - } - - /** - * Builds a raw row as a list of strings: index 0 = label text, indices 1..N = column values. - */ - private List buildRawRow(TokenizedLine tl, List columnGrid) { - String[] cells = new String[columnGrid.size() + 1]; - Arrays.fill(cells, ""); - - // Separate label tokens (those not landing in any confirmed column) from column tokens. - List labelParts = new ArrayList<>(); - for (LineToken token : tl.all()) { - int col = nearestColumnIndex(token.right(), columnGrid); - if (col >= 0 && token.numeric()) { - int cellIdx = col + 1; - cells[cellIdx] = - cells[cellIdx].isEmpty() - ? token.text() - : cells[cellIdx] + " " + token.text(); - } else { - labelParts.add(token.text()); - } - } - cells[0] = String.join(" ", labelParts).trim(); - return Arrays.asList(cells); - } - - private TableRow buildTableRow( - int rowIdx, TokenizedLine tl, List rawRow, List columnGrid) { - List cells = new ArrayList<>(rawRow.size()); - - // Label cell: use the line's full bounds as an approximation. - cells.add(TableCell.of(0, rawRow.get(0), tl.line().bounds())); - - for (int col = 0; col < columnGrid.size(); col++) { - String text = col + 1 < rawRow.size() ? rawRow.get(col + 1) : ""; - float right = columnGrid.get(col); - float left = col > 0 ? columnGrid.get(col - 1) : right - 50f; - Bounds cellBounds = - new Bounds( - left, - tl.line().bounds().y(), - right - left, - tl.line().bounds().height()); - cells.add(TableCell.of(col + 1, text, cellBounds)); - } - return new TableRow(rowIdx, Collections.unmodifiableList(cells)); - } - - // ── confidence scoring ─────────────────────────────────────────────────────────────────────── - - /** - * Heuristic score in [0.0, 0.85] (ceiling keeps results below Tabula lattice which starts at - * 1.0). Base 0.70; +0.05/col beyond 2 (max +0.10); +0.05 at ≥5 anchors, +0.05 at ≥8; −0.15 if - * >30 % of anchors have inconsistent columns; −0.10 if non-anchors outnumber anchors. - */ - private float computeConfidence( - List group, List columnGrid, List warnings) { - float score = 0.70f; - - long anchorCount = - group.stream().filter(r -> r.isAnchor() && matchesGrid(r, columnGrid)).count(); - long totalRows = group.size(); - - // More columns - int extraCols = Math.min(columnGrid.size() - MIN_COLUMNS, 2); - score += extraCols * 0.05f; - - // More anchor rows - if (anchorCount >= 5) score += 0.05f; - if (anchorCount >= 8) score += 0.05f; - - // Inconsistent column matching - long inconsistent = - group.stream() - .filter(TokenizedLine::isAnchor) - .filter(tl -> hasInconsistentColumnMatch(tl, columnGrid)) - .count(); - if (inconsistent > anchorCount * 0.30) { - score -= 0.15f; - warnings.add( - "Column match inconsistent on " - + inconsistent - + "/" - + anchorCount - + " anchor rows"); - } - - // Label-heavy - long nonAnchor = totalRows - anchorCount; - if (nonAnchor > anchorCount) { - score -= 0.10f; - warnings.add( - "Non-anchor rows (" - + nonAnchor - + ") outnumber anchor rows (" - + anchorCount - + ")"); - } - - return Math.max(0f, Math.min(0.85f, score)); - } - - // ── utility ────────────────────────────────────────────────────────────────────────────────── - - /** - * Returns the grid index nearest to {@code rightEdge}, or -1 if none is within {@value - * #COLUMN_MATCH_BUCKETS} buckets. - */ - private int nearestColumnIndex(float rightEdge, List grid) { - int nearest = -1; - float minDist = COLUMN_MATCH_BUCKETS * COLUMN_BUCKET_PT + 1f; - for (int i = 0; i < grid.size(); i++) { - float dist = Math.abs(rightEdge - grid.get(i)); - if (dist < minDist) { - minDist = dist; - nearest = i; - } - } - return nearest; - } - - private Bounds computeGroupBounds(List group) { - return group.stream() - .map(tl -> tl.line().bounds()) - .reduce(Bounds::merge) - .orElse(new Bounds(0, 0, 0, 0)); - } - - /** Modal gap between consecutive line edges, used to calibrate the group-split threshold. */ - private float computeModalSpacing(List lines) { - if (lines.size() < 2) return 0f; - Map freq = new HashMap<>(); - for (int i = 1; i < lines.size(); i++) { - float gap = lines.get(i).bounds().y() - lines.get(i - 1).bounds().bottom(); - if (gap > 0) freq.merge(Math.round(gap / 2f) * 2f, 1L, Long::sum); - } - return freq.entrySet().stream() - .max(Map.Entry.comparingByValue()) - .map(Map.Entry::getKey) - .orElse(0f); - } - - private static int bucket(float x) { - return Math.round(x / COLUMN_BUCKET_PT); - } - - // ── private data types ─────────────────────────────────────────────────────────────────────── - - /** A word-level token with an approximate right-edge x-position. */ - record LineToken(String text, float x, float right, boolean numeric) {} - - /** A {@link RawLine} with tokens pre-computed; an "anchor" has ≥ 2 numeric tokens. */ - record TokenizedLine(RawLine line, List all, List numeric) { - boolean isAnchor() { - return numeric.size() >= 2; - } - } -} diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java deleted file mode 100644 index 6831f6d734..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/LineBuilder.java +++ /dev/null @@ -1,139 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.util.ArrayList; -import java.util.Comparator; -import java.util.List; - -import org.springframework.stereotype.Service; - -import lombok.extern.slf4j.Slf4j; - -/** - * Groups {@link TextFragment} objects into visual {@link RawLine}s using baseline proximity. - * - *

Fragments are on the same line when their baselines are within a font-size-derived tolerance. - * A new line starts whenever the horizontal gap exceeds an adaptive column-gap threshold ({@code - * max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT)}), splitting two-column text. - */ -@Service -@Slf4j -public class LineBuilder { - - /** Baseline tolerance as a fraction of font size; 0.5 keeps mixed-size text on one line. */ - private static final float BASELINE_TOLERANCE_FACTOR = 0.5f; - - /** Absolute minimum tolerance so tiny font sizes don't collapse multi-line content. */ - private static final float MIN_BASELINE_TOLERANCE = 2f; - - /** - * Column-gap threshold as a fraction of page width; 0.10 clears tab stops but stays below - * two-column gutters. - */ - static final float COLUMN_GAP_RATIO = 0.10f; - - /** Floor for the column-gap threshold so narrow pages don't over-split lines. */ - static final float COLUMN_GAP_MIN_PT = 40f; - - public List build(List fragments, int pageNumber) { - if (fragments.isEmpty()) return List.of(); - - float effectiveWidth = inferEffectiveWidth(fragments); - float columnGapThreshold = Math.max(effectiveWidth * COLUMN_GAP_RATIO, COLUMN_GAP_MIN_PT); - log.debug( - "LineBuilder page {}: effectiveWidth={:.1f}pt, columnGapThreshold={:.1f}pt", - pageNumber, - effectiveWidth, - columnGapThreshold); - - // Sort top-to-bottom first, then left-to-right within the same baseline band. - List sorted = - fragments.stream() - .sorted( - Comparator.comparingDouble(TextFragment::baseline) - .thenComparingDouble(f -> f.bounds().x())) - .toList(); - - List> groups = groupByBaseline(sorted, columnGapThreshold); - - List lines = new ArrayList<>(groups.size()); - for (int i = 0; i < groups.size(); i++) { - List group = - groups.get(i).stream() - .sorted(Comparator.comparingDouble(f -> f.bounds().x())) - .toList(); - - Bounds lineBounds = - group.stream() - .map(TextFragment::bounds) - .reduce(Bounds::merge) - .orElse(new Bounds(0, 0, 0, 0)); - - lines.add(new RawLine("ln-p" + pageNumber + "-" + i, group, lineBounds, pageNumber)); - } - return lines; - } - - private List> groupByBaseline( - List sorted, float columnGapThreshold) { - List> groups = new ArrayList<>(); - List current = new ArrayList<>(); - float currentBaseline = Float.NaN; - - for (TextFragment fragment : sorted) { - if (current.isEmpty()) { - current.add(fragment); - currentBaseline = fragment.baseline(); - continue; - } - - float maxFontSize = - Math.max( - fragment.fontSize(), - (float) - current.stream() - .mapToDouble(TextFragment::fontSize) - .max() - .orElse(0)); - float tolerance = - Math.max(maxFontSize * BASELINE_TOLERANCE_FACTOR, MIN_BASELINE_TOLERANCE); - - boolean sameBaseline = Math.abs(fragment.baseline() - currentBaseline) <= tolerance; - boolean columnGap = sameBaseline && hasColumnGap(fragment, current, columnGapThreshold); - - if (sameBaseline && !columnGap) { - current.add(fragment); - // Anchor to the weighted mean baseline so long lines stay stable. - currentBaseline = - (currentBaseline * (current.size() - 1) + fragment.baseline()) - / current.size(); - } else { - groups.add(current); - current = new ArrayList<>(); - current.add(fragment); - currentBaseline = fragment.baseline(); - } - } - - if (!current.isEmpty()) groups.add(current); - return groups; - } - - /** - * True when the gap from the rightmost fragment in {@code group} to {@code next} exceeds {@code - * threshold}. - */ - private static boolean hasColumnGap( - TextFragment next, List group, float threshold) { - float lastRight = group.get(group.size() - 1).bounds().right(); - return next.bounds().x() - lastRight > threshold; - } - - /** Infers effective page width from the rightmost fragment right-edge plus a 10 % margin. */ - private static float inferEffectiveWidth(List fragments) { - double maxRight = - fragments.stream().mapToDouble(f -> f.bounds().right()).max().orElse(500.0); - return (float) maxRight * 1.10f; - } -} diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/PdfIngester.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/PdfIngester.java deleted file mode 100644 index a7dc9c282b..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/PdfIngester.java +++ /dev/null @@ -1,79 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.io.IOException; -import java.util.ArrayList; -import java.util.List; - -import org.apache.pdfbox.pdmodel.PDDocument; -import org.apache.pdfbox.pdmodel.PDPage; -import org.apache.pdfbox.pdmodel.common.PDRectangle; -import org.springframework.stereotype.Service; - -import lombok.RequiredArgsConstructor; -import lombok.extern.slf4j.Slf4j; - -/** - * Runs the per-page ingestion pipeline: {@link WordExtractingStripper} → {@link LineBuilder} → - * {@link TableParser}, producing a {@link PdfModels.ParsedPage} per page. The caller owns the - * {@link PDDocument} lifecycle. - */ -@Service -@RequiredArgsConstructor -@Slf4j -public class PdfIngester { - - private final LineBuilder lineBuilder; - private final TableParser tableParser; - - public List parse(PDDocument document) throws IOException { - return parse(document, document.getNumberOfPages()); - } - - public List parse(PDDocument document, int maxPages) throws IOException { - int pageCount = Math.min(document.getNumberOfPages(), maxPages); - List pages = new ArrayList<>(pageCount); - long fragmentsMs = 0; - long tablesMs = 0; - long t0 = System.currentTimeMillis(); - - for (int p = 1; p <= pageCount; p++) { - long ft = System.currentTimeMillis(); - List fragments = extractFragments(document, p); - fragmentsMs += System.currentTimeMillis() - ft; - - PDPage page = document.getPage(p - 1); - PDRectangle mediaBox = page.getMediaBox(); - List lines = lineBuilder.build(fragments, p); - RawPage rawPage = new RawPage(p, mediaBox.getWidth(), mediaBox.getHeight(), lines); - - long tt = System.currentTimeMillis(); - List tables = tableParser.parse(document, rawPage); - tablesMs += System.currentTimeMillis() - tt; - - log.debug( - "Page {}: {} fragments → {} lines, {} table(s)", - p, - fragments.size(), - lines.size(), - tables.size()); - pages.add(new ParsedPage(p, mediaBox.getWidth(), mediaBox.getHeight(), tables, lines)); - } - - log.info( - "[timing] parse pages={} total={}ms fragments={}ms tables={}ms", - pageCount, - System.currentTimeMillis() - t0, - fragmentsMs, - tablesMs); - return pages; - } - - private List extractFragments(PDDocument document, int pageNumber) - throws IOException { - WordExtractingStripper stripper = new WordExtractingStripper(pageNumber); - stripper.getText(document); - return stripper.getFragments(); - } -} diff --git a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/WordExtractingStripper.java b/app/common/src/main/java/stirling/software/SPDF/pdf/parser/WordExtractingStripper.java deleted file mode 100644 index 52ab9d9a18..0000000000 --- a/app/common/src/main/java/stirling/software/SPDF/pdf/parser/WordExtractingStripper.java +++ /dev/null @@ -1,113 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.io.IOException; -import java.util.ArrayList; -import java.util.Collections; -import java.util.List; - -import org.apache.pdfbox.pdmodel.PDPage; -import org.apache.pdfbox.pdmodel.font.PDFont; -import org.apache.pdfbox.text.PDFTextStripper; -import org.apache.pdfbox.text.TextPosition; - -/** - * Extends {@link PDFTextStripper} to capture per-fragment geometry and font metadata. - * - *

Overrides {@link #writeString} to split each content-stream string into word-level {@link - * TextFragment}s with bounding boxes, baseline, font name, and bold flag. Coordinates are in - * PDFTextStripper space: (0,0) top-left, Y increases downward, {@code getY()} is the baseline. - */ -class WordExtractingStripper extends PDFTextStripper { - - private final int targetPage; - private final List fragments = new ArrayList<>(); - private int fragmentIndex = 0; - - WordExtractingStripper(int pageNumber) throws IOException { - this.targetPage = pageNumber; - setStartPage(pageNumber); - setEndPage(pageNumber); - setSortByPosition(true); - } - - @Override - protected void startPage(PDPage page) throws IOException { - super.startPage(page); - fragments.clear(); - fragmentIndex = 0; - } - - @Override - protected void writeString(String text, List textPositions) throws IOException { - if (text == null || text.isBlank()) return; - - // Fast path: no whitespace → emit one fragment (most financial PDFs have each - // number as its own string operation, so this is the common case). - if (text.indexOf(' ') < 0) { - emitFragment(text, textPositions); - return; - } - - // Per-word splitting requires 1:1 text-char to TextPosition correspondence. - // Fall back to one fragment when sizes differ (ligatures, encoding edge cases). - if (textPositions.size() != text.length()) { - emitFragment(text, textPositions); - return; - } - - // Emit one TextFragment per whitespace-delimited word with accurate per-word bounds. - int start = 0; - for (int i = 0; i <= text.length(); i++) { - if (i == text.length() || text.charAt(i) == ' ') { - if (start < i) { - emitFragment(text.substring(start, i), textPositions.subList(start, i)); - } - start = i + 1; - } - } - } - - private void emitFragment(String text, List positions) { - if (positions.isEmpty()) return; - - float minX = Float.MAX_VALUE; - float minY = Float.MAX_VALUE; - float maxRight = -Float.MAX_VALUE; - float maxBaseline = -Float.MAX_VALUE; - TextPosition first = null; - - for (TextPosition tp : positions) { - if (tp == null) continue; - if (first == null) first = tp; - - float x = tp.getX(); - // getY() is the baseline; top of character = getY() - getHeight(). - float top = tp.getY() - tp.getHeight(); - float right = x + tp.getWidth(); - float baseline = tp.getY(); - - minX = Math.min(minX, x); - minY = Math.min(minY, top); - maxRight = Math.max(maxRight, right); - maxBaseline = Math.max(maxBaseline, baseline); - } - - if (first == null) return; - - PDFont font = first.getFont(); - String fontName = font != null ? font.getName() : ""; - boolean bold = fontName != null && fontName.toLowerCase().contains("bold"); - // getHeight() gives the rendered glyph height, which is the most reliable visual size. - float fontSize = first.getHeight(); - - Bounds bounds = new Bounds(minX, minY, maxRight - minX, maxBaseline - minY); - String id = "tf-p" + targetPage + "-" + fragmentIndex++; - fragments.add(new TextFragment(id, text, bounds, maxBaseline, fontSize, fontName, bold)); - } - - List getFragments() { - return Collections.unmodifiableList(fragments); - } -} diff --git a/app/common/src/main/java/stirling/software/common/pdf/HeadingDetector.java b/app/common/src/main/java/stirling/software/common/pdf/HeadingDetector.java new file mode 100644 index 0000000000..0937cef646 --- /dev/null +++ b/app/common/src/main/java/stirling/software/common/pdf/HeadingDetector.java @@ -0,0 +1,191 @@ +package stirling.software.common.pdf; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import stirling.software.jpdfium.text.PageText; +import stirling.software.jpdfium.text.TextChar; +import stirling.software.jpdfium.text.TextLine; +import stirling.software.jpdfium.text.TextWord; + +final class HeadingDetector { + + private HeadingDetector() {} + + /** A heading is at most this many words; longer lines are treated as body text. */ + private static final int MAX_HEADING_WORDS = 12; + + /** + * Returns the Markdown heading prefix for a line. The decision combines several signals, never + * text matching, so a plain line that merely shares text with a heading is never promoted: + * + *

    + *
  • Size — dominant glyph font size vs. the document body median (primary signal). + * Some PDFs encode visual size in the text matrix, so every glyph reports ~1.0; for those + * the line height is used as the proxy instead. + *
  • Brevity — headings are short labels; a line over {@value #MAX_HEADING_WORDS} + * words is body text regardless of size. + *
  • Not a sentence — a line ending in {@code . ! ?} reads as prose, not a heading. + *
+ * + *

Boldness is deliberately not a heading signal — a bold-but-not-larger line is + * emphasis, not a heading (see {@link #isBoldLabel}); promoting it to {@code #}/{@code ##} is + * the main source of false-positive headings. + * + *

    + *
  • size > baseline * 1.4 → {@code "# "} + *
  • size > baseline * 1.2 → {@code "## "} + *
  • otherwise → {@code ""} + *
+ */ + static String headingPrefix(TextLine line, float medianBodySize, float medianBodyHeight) { + String text = line.text().strip(); + if (text.isEmpty() || wordCount(text) > MAX_HEADING_WORDS || endsLikeSentence(text)) { + return ""; + } + + float dominant = dominantFontSize(line); + float value; + float baseline; + if (dominant > 2f && medianBodySize > 2f) { + value = dominant; + baseline = medianBodySize; + } else { + value = line.height(); + baseline = medianBodyHeight; + } + if (baseline <= 0f) { + return ""; + } + + float ratio = value / baseline; + if (ratio > 1.4f) { + return "# "; + } + if (ratio > 1.2f) { + return "## "; + } + return ""; + } + + /** + * True when a line should be emphasised as bold (rendered {@code **like this**}) rather than + * promoted to a heading: it is bold, short, and not a full sentence. Used for bold labels that + * are not large enough to be headings. + */ + static boolean isBoldLabel(TextLine line) { + String text = line.text().strip(); + if (text.isEmpty() || wordCount(text) > MAX_HEADING_WORDS || endsLikeSentence(text)) { + return false; + } + return isBold(line); + } + + private static int wordCount(String text) { + return text.split("\\s+").length; + } + + private static boolean endsLikeSentence(String text) { + char last = text.charAt(text.length() - 1); + return last == '.' || last == '!' || last == '?'; + } + + /** True when the line's dominant font is bold, inferred from PostScript font names. */ + private static boolean isBold(TextLine line) { + Map counts = new HashMap<>(); + for (TextWord word : line.words()) { + for (TextChar ch : word.chars()) { + if (ch.isWhitespace() || ch.isNewline()) { + continue; + } + String name = ch.fontName(); + if (name != null && !name.isBlank()) { + counts.merge(name, 1, Integer::sum); + } + } + } + String dominantFont = ""; + int max = -1; + for (Map.Entry e : counts.entrySet()) { + if (e.getValue() > max) { + max = e.getValue(); + dominantFont = e.getKey(); + } + } + String lower = dominantFont.toLowerCase(java.util.Locale.ROOT); + return lower.contains("bold") + || lower.contains("black") + || lower.contains("heavy") + || lower.contains("semibold"); + } + + /** Computes the median glyph font size across all pages. */ + static float medianFontSize(List allPages) { + List sizes = new ArrayList<>(); + for (PageText page : allPages) { + for (TextChar ch : page.chars()) { + if (!ch.isWhitespace() && !ch.isNewline() && ch.fontSize() > 0f) { + sizes.add(ch.fontSize()); + } + } + } + return median(sizes, 12f); + } + + /** Computes the median TextLine height across all pages. Used when font size is degenerate. */ + static float medianLineHeight(List allPages) { + List heights = new ArrayList<>(); + for (PageText page : allPages) { + for (TextLine line : page.lines()) { + if (line.height() > 0f && !line.text().isBlank()) { + heights.add(line.height()); + } + } + } + return median(heights, 12f); + } + + private static float median(List values, float fallback) { + if (values.isEmpty()) { + return fallback; + } + Collections.sort(values); + int mid = values.size() / 2; + if (values.size() % 2 == 0) { + return (values.get(mid - 1) + values.get(mid)) / 2f; + } + return values.get(mid); + } + + /** + * Returns the font size that appears most often (by character count) in the given line. Ties + * are broken in favour of the larger size. + */ + private static float dominantFontSize(TextLine line) { + Map counts = new HashMap<>(); + for (TextWord word : line.words()) { + for (TextChar ch : word.chars()) { + if (!ch.isWhitespace() && !ch.isNewline() && ch.fontSize() > 0f) { + counts.merge(ch.fontSize(), 1, Integer::sum); + } + } + } + if (counts.isEmpty()) { + return 0f; + } + float dominant = 0f; + int maxCount = -1; + for (Map.Entry entry : counts.entrySet()) { + int count = entry.getValue(); + float size = entry.getKey(); + if (count > maxCount || (count == maxCount && size > dominant)) { + maxCount = count; + dominant = size; + } + } + return dominant; + } +} diff --git a/app/common/src/main/java/stirling/software/common/pdf/PdfMarkdownConverter.java b/app/common/src/main/java/stirling/software/common/pdf/PdfMarkdownConverter.java new file mode 100644 index 0000000000..c19468b5ed --- /dev/null +++ b/app/common/src/main/java/stirling/software/common/pdf/PdfMarkdownConverter.java @@ -0,0 +1,1043 @@ +package stirling.software.common.pdf; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.Comparator; +import java.util.HashSet; +import java.util.List; +import java.util.Set; +import java.util.regex.Pattern; +import java.util.stream.Collectors; + +import stirling.software.jpdfium.PdfDocument; +import stirling.software.jpdfium.PdfPage; +import stirling.software.jpdfium.doc.ExtractedImage; +import stirling.software.jpdfium.doc.PdfImageExtractor; +import stirling.software.jpdfium.model.Rect; +import stirling.software.jpdfium.text.PageText; +import stirling.software.jpdfium.text.PdfTableExtractor; +import stirling.software.jpdfium.text.PdfTextExtractor; +import stirling.software.jpdfium.text.Table; +import stirling.software.jpdfium.text.TextLine; +import stirling.software.jpdfium.text.TextWord; + +/** + * Converts a PDF to Markdown using a TextLine-driven body pipeline. + * + *

Body text is rebuilt from {@link PdfTextExtractor} {@link TextLine}s. TextLines group words + * faithfully and keep paragraph order, so the only pre-processing needed is stitching narrow + * standalone glyph fragments (apostrophes, quotes, asterisks, superscript footnote markers, + * bullets) back into the line they belong to. Column layout and tables are derived from line/word + * geometry directly. + */ +public class PdfMarkdownConverter { + + private static final Pattern SOFT_HYPHEN = Pattern.compile("(\\w+)-\\n([a-z])"); + + /** Width below which a TextLine is treated as a stray glyph fragment to be stitched. */ + private static final float GLYPH_WIDTH = 7.5f; + + public String convert(PdfDocument doc) throws IOException { + List allPageText = PdfTextExtractor.extractAll(doc); + float medianSize = HeadingDetector.medianFontSize(allPageText); + float medianHeight = HeadingDetector.medianLineHeight(allPageText); + + int pageCount = doc.pageCount(); + // Elements are either rendered text (String) or a structured TableBlock. Tables stay + // structured until after the page loop so a table split across a page break can be stitched + // back together before rendering. + List output = new ArrayList<>(); + // Header text of a table that ended the previous page, used to spot a continuation whose + // header repeats at the top of the current page. Null when the previous page did not end in + // a table. + String prevPageTrailingTableHeader = null; + + for (int pageIndex = 0; pageIndex < pageCount; pageIndex++) { + List rawLines = + pageIndex < allPageText.size() ? allPageText.get(pageIndex).lines() : List.of(); + + // Stitch stray glyph fragments (apostrophes, asterisks, superscripts, bullets) into + // their host lines so paragraph assembly sees faithful, complete lines. + List lines = stitchGlyphs(rawLines); + if (lines.isEmpty()) { + emitImages(doc, pageIndex, output); + prevPageTrailingTableHeader = null; + continue; + } + + // Sort top-to-bottom (PDF y=0 is the bottom of the page). + lines.sort(Comparator.comparingDouble((Line l) -> l.y).reversed()); + + // Multi-column guard: only genuine two-column prose should be split. A table's column + // gutters must NOT be mistaken for a page-layout gutter, so this looks at whether row + // lines span the gutter (table) or stay within one side (two-column prose). + // A table that ran to the bottom of the previous page and repeats its header at the top + // of this page is a continuation, not a new two-column layout. Detecting the repeated + // header keeps this page out of the two-column path so the continuation is rebuilt as a + // table and stitched back onto the previous block. + final String continuationHeader = prevPageTrailingTableHeader; + boolean tableContinuation = + continuationHeader != null + && lines.stream() + .anyMatch( + l -> normaliseSpace(l.text).equals(continuationHeader)); + + boolean twoColumn = !tableContinuation && detectsTwoColumns(lines); + + // Tables are detected from text/word geometry (the word-grid detector), which handles + // both ruled and borderless tables and places cells by column alignment. The native + // ruled-line extractor is not used: it both mis-renders cells and double-emits rows. + Set tableRowTexts = new HashSet<>(); + List blocks = twoColumn ? List.of() : findTableBlocks(lines); + Set tableLines = new HashSet<>(); + for (TableBlock b : blocks) { + for (List row : b.rows()) { + for (Line l : row) { + tableLines.add(l); + tableRowTexts.add(repairHyphens(l.text).strip()); + } + } + } + + List pageItems = new ArrayList<>(); + if (twoColumn) { + for (List col : splitIntoColumns(lines)) { + List paras = new ArrayList<>(); + assembleParagraphs(col, medianSize, medianHeight, paras, tableRowTexts); + pageItems.addAll(paras); + } + } else { + // Interleave tables with surrounding text by vertical position. Each block sits in + // its own slot; non-table lines fall into the slot for their y (text above a block, + // between blocks, or below the last). This keeps multiple tables on one page + // separate and in reading order. + List> segments = new ArrayList<>(); + for (int s = 0; s <= blocks.size(); s++) { + segments.add(new ArrayList<>()); + } + for (Line l : lines) { + if (tableLines.contains(l)) { + continue; + } + int slot = 0; + for (TableBlock b : blocks) { + if (b.bottom() > l.y) { + slot++; + } + } + segments.get(slot).add(l); + } + for (int s = 0; s <= blocks.size(); s++) { + List paras = new ArrayList<>(); + assembleParagraphs( + segments.get(s), medianSize, medianHeight, paras, tableRowTexts); + pageItems.addAll(paras); + if (s < blocks.size()) { + pageItems.add(blocks.get(s)); + } + } + } + + emitImages(doc, pageIndex, pageItems); + + if (pageItems.isEmpty()) { + continue; + } + + mergeAcrossPageBoundary(output, pageItems); + output.addAll(pageItems); + prevPageTrailingTableHeader = trailingTableHeader(pageItems); + } + + // Stitch tables split across page breaks, then render every element to Markdown. + List stitched = stitchTables(output); + List rendered = new ArrayList<>(); + for (Object e : stitched) { + rendered.add(e instanceof TableBlock tb ? tb.render() : (String) e); + } + return String.join("\n\n", rendered); + } + + // --- Glyph stitching --------------------------------------------------- + + /** A mutable assembled line: text plus geometry used for ordering and heading detection. */ + private static final class Line { + String text; + float x; + float y; + float width; + float height; + final TextLine source; + + Line(TextLine src) { + this.source = src; + this.text = src.text(); + this.x = src.x(); + this.y = src.y(); + this.width = src.width(); + this.height = src.height(); + } + } + + /** + * Merges narrow glyph fragments (width < {@link #GLYPH_WIDTH}) into the line they belong to. + * + *
    + *
  • A glyph between a left fragment that ends near it and a right fragment that starts near + * it (both on the same baseline) is inserted inline: {@code aren} + {@code '} + {@code t} + * → {@code aren't}. + *
  • A glyph immediately right of a line's end is appended (e.g. superscript footnote marker + * after a number). + *
  • A glyph immediately left of a line's start is prepended (e.g. footnote marker before + * its text). + *
+ */ + private static List stitchGlyphs(List raw) { + List hosts = new ArrayList<>(); + List glyphs = new ArrayList<>(); + for (TextLine l : raw) { + String t = l.text().strip(); + if (t.isEmpty()) { + continue; + } + if (l.width() < GLYPH_WIDTH && t.length() <= 2) { + glyphs.add(l); + } else { + hosts.add(l); + } + } + + List lines = hosts.stream().map(Line::new).collect(Collectors.toList()); + + for (TextLine g : glyphs) { + String gt = g.text().strip(); + if (isBulletGlyph(gt)) { + attachBullet(g, gt, lines); + } else { + attachInlineGlyph(g, gt, lines); + } + } + return lines; + } + + private static boolean isBulletGlyph(String gt) { + return "•".equals(gt) || "▪".equals(gt) || "◦".equals(gt); + } + + /** + * Attaches a bullet glyph to the body line it introduces: the closest line that begins to the + * right of the bullet at roughly the same height or just below it. + */ + private static void attachBullet(TextLine g, String gt, List lines) { + Line best = null; + float bestScore = Float.MAX_VALUE; + for (Line h : lines) { + if (h.x < g.x() - 2f) { + continue; + } + float dy = g.y() - h.y; + if (dy < -4f || dy > 28f) { + continue; + } + float score = Math.abs(dy) + (h.x - g.x()) * 0.2f; + if (score < bestScore) { + bestScore = score; + best = h; + } + } + if (best != null && !best.text.startsWith("•")) { + best.text = "• " + best.text; + best.x = g.x(); + } else { + lines.add(new Line(g)); + } + } + + /** + * Stitches a narrow inline glyph (apostrophe, quote, asterisk, superscript marker) into the + * line it belongs to: inline between two same-baseline fragments, appended to the line that + * ends at it, or prepended to the line that starts at it. + */ + private static void attachInlineGlyph(TextLine g, String gt, List lines) { + Line left = null; + Line right = null; + float lb = 7f; + float rb = 7f; + for (Line h : lines) { + boolean sameBaseline = g.y() >= h.y - 4f && g.y() <= h.y + h.height + 5f; + if (!sameBaseline) { + continue; + } + float rightEdge = h.x + h.width; + float dxLeft = Math.abs(rightEdge - g.x()); + if (dxLeft < lb) { + lb = dxLeft; + left = h; + } + float dxRight = Math.abs(h.x - g.x()); + if (dxRight < rb) { + rb = dxRight; + right = h; + } + } + + if (left != null && right != null && left != right && Math.abs(left.y - right.y) < 6f) { + left.text = left.text + gt + right.text; + left.width = (right.x + right.width) - left.x; + lines.remove(right); + } else if (left != null) { + left.text = left.text + gt; + left.width = Math.max(left.width, g.x() + g.width() - left.x); + } else if (right != null) { + right.text = gt + right.text; + right.x = g.x(); + } else { + lines.add(new Line(g)); + } + } + + // --- Column detection (guard only) ------------------------------------- + + /** + * Returns true when the page is a genuine two-column layout. Uses line/word geometry: body + * blocks (ignoring narrow glyph blocks) and requires a wide horizontal gutter populated on both + * sides, so single apostrophe glyphs cannot create a false second column. + */ + private static boolean detectsTwoColumns(List lines) { + if (lines.size() < 8) { + return false; + } + float minX = Float.MAX_VALUE; + float maxX = -Float.MAX_VALUE; + for (Line l : lines) { + minX = Math.min(minX, l.x); + maxX = Math.max(maxX, l.x + l.width); + } + if (maxX - minX < 200f) { + return false; + } + + // Scan candidate gutter positions across the central band (35%-65% of width) and pick the + // one crossed by the fewest lines. Two-column prose has a gutter that only a handful of + // full-width lines (title, section headings) cross; a table's rows all span the full width, + // so every candidate gutter is crossed by most lines. + float centreLo = minX + (maxX - minX) * 0.35f; + float centreHi = minX + (maxX - minX) * 0.65f; + int bestCrossing = Integer.MAX_VALUE; + int bestLeft = 0; + int bestRight = 0; + for (float gutter = centreLo; gutter <= centreHi; gutter += 2f) { + int crossing = 0; + int leftOnly = 0; + int rightOnly = 0; + for (Line l : lines) { + float lx = l.x; + float rx = l.x + l.width; + if (lx < gutter - 5f && rx > gutter + 5f) { + crossing++; + } else if (rx <= gutter) { + leftOnly++; + } else { + rightOnly++; + } + } + if (crossing < bestCrossing) { + bestCrossing = crossing; + bestLeft = leftOnly; + bestRight = rightOnly; + } + } + + return bestLeft >= 4 && bestRight >= 4 && bestCrossing <= (int) (lines.size() * 0.25f); + } + + private static List> splitIntoColumns(List lines) { + List xs = + lines.stream() + .filter(l -> l.width >= 40f) + .map(l -> l.x) + .sorted() + .collect(Collectors.toList()); + if (xs.isEmpty()) { + return List.of(lines); + } + float minX = xs.get(0); + float maxX = xs.get(xs.size() - 1); + float splitAt = (minX + maxX) / 2f; + float biggestGap = 0; + for (int i = 1; i < xs.size(); i++) { + float gap = xs.get(i) - xs.get(i - 1); + if (gap > biggestGap) { + biggestGap = gap; + splitAt = (xs.get(i - 1) + xs.get(i)) / 2f; + } + } + List left = new ArrayList<>(); + List right = new ArrayList<>(); + for (Line l : lines) { + if (l.x < splitAt) { + left.add(l); + } else { + right.add(l); + } + } + if (left.isEmpty()) { + return List.of(right); + } + if (right.isEmpty()) { + return List.of(left); + } + return List.of(left, right); + } + + // --- Paragraph assembly ------------------------------------------------ + + private static void assembleParagraphs( + List lines, + float medianSize, + float medianHeight, + List out, + Set tableRowTexts) { + StringBuilder para = new StringBuilder(); + float prevBottomY = Float.MAX_VALUE; + float prevHeight = 0f; + + for (Line line : lines) { + String text = repairHyphens(line.text).strip(); + if (text.isEmpty()) { + continue; + } + if (tableRowTexts.contains(text)) { + continue; + } + + float blockTop = line.y + line.height; + float gap = prevBottomY - blockTop; + boolean paragraphBreak = prevHeight > 0f && gap > prevHeight * 0.8f; + + String prefix = HeadingDetector.headingPrefix(line.source, medianSize, medianHeight); + boolean isHeading = !prefix.isEmpty(); + boolean isBullet = startsWithBullet(text); + + if (isHeading) { + flushParagraph(para, out); + out.add(prefix + escapeMarkdown(text)); + } else if (isBullet) { + flushParagraph(para, out); + out.add(escapeMarkdown(text)); + } else if (HeadingDetector.isBoldLabel(line.source)) { + // Bold but not large enough to be a heading → emphasise as bold, don't promote. + flushParagraph(para, out); + out.add("**" + escapeMarkdown(text) + "**"); + } else if (paragraphBreak) { + flushParagraph(para, out); + para.append(text); + } else { + if (!para.isEmpty()) { + char fc = text.charAt(0); + boolean noSpace = fc == '\'' || fc == '’' || fc == '‘' || fc == '"'; + if (!noSpace) { + para.append(' '); + } + } + para.append(text); + } + + prevBottomY = line.y; + prevHeight = line.height; + } + flushParagraph(para, out); + } + + private static boolean startsWithBullet(String text) { + return text.startsWith("•") || text.startsWith("▪") || text.startsWith("◦"); + } + + // --- Word-grid table detection ----------------------------------------- + + /** + * A detected table. Each row is a list of source lines: usually one, but more when a cell wraps + * onto extra lines (those continuation lines are absorbed into the row they belong to). + */ + private record TableBlock(List> rows, float top, float bottom) { + String render() { + return buildTableFromRows(rows); + } + } + + /** + * Detects table blocks on a page. Anchor rows (lines with table-like column gaps) are grouped + * into vertically-contiguous runs separated by large vertical gaps, so multiple separate tables + * on one page stay separate. Non-anchor lines that fall within a run's vertical span are + * treated as wrapped-cell continuations and absorbed into the nearest anchor row above them. + */ + private static List findTableBlocks(List lines) { + List cands = + lines.stream() + .filter(l -> isTableCandidate(l.source)) + .sorted(Comparator.comparingDouble((Line l) -> l.y).reversed()) + .collect(Collectors.toList()); + if (cands.size() < 2) { + return List.of(); + } + + List gaps = new ArrayList<>(); + for (int i = 1; i < cands.size(); i++) { + gaps.add(cands.get(i - 1).y - cands.get(i).y); + } + List sorted = new ArrayList<>(gaps); + sorted.sort(Comparator.naturalOrder()); + float medianGap = sorted.get(sorted.size() / 2); + float splitThreshold = Math.max(medianGap * 2.5f, medianGap + 6f); + + List> anchorGroups = new ArrayList<>(); + List current = new ArrayList<>(); + current.add(cands.get(0)); + for (int i = 1; i < cands.size(); i++) { + float gap = cands.get(i - 1).y - cands.get(i).y; + if (gap > splitThreshold) { + anchorGroups.add(current); + current = new ArrayList<>(); + } + current.add(cands.get(i)); + } + anchorGroups.add(current); + + List nonCandidates = + lines.stream() + .filter(l -> !isTableCandidate(l.source)) + .collect(Collectors.toList()); + + List blocks = new ArrayList<>(); + for (List anchors : anchorGroups) { + if (anchors.size() < 2) { + continue; + } + float top = anchors.get(0).y; + float bottom = anchors.get(anchors.size() - 1).y; + + // Each anchor seeds a row; absorb wrapped continuation lines (non-anchors within the + // run's vertical span, with a little slack below the last row) into the anchor above. + List> rows = new ArrayList<>(); + for (Line a : anchors) { + List row = new ArrayList<>(); + row.add(a); + rows.add(row); + } + for (Line nc : nonCandidates) { + if (nc.y > top || nc.y < bottom - medianGap) { + continue; + } + int owner = 0; + float bestDelta = Float.MAX_VALUE; + for (int i = 0; i < anchors.size(); i++) { + float delta = anchors.get(i).y - nc.y; // positive when anchor is above nc + if (delta >= -1f && delta < bestDelta) { + bestDelta = delta; + owner = i; + } + } + rows.get(owner).add(nc); + } + + if (buildTableFromRows(rows).isBlank()) { + continue; + } + blocks.add(new TableBlock(rows, top, bottom)); + } + return blocks; + } + + private static String buildTableFromRows(List> rowGroups) { + // Detect columns by vertical-whitespace projection across all lines, rather than a 1-D gap + // threshold on pooled word x's. Pooled-gap detection is fragile when numbers are + // right-aligned (a 10-digit value starts well left of a 7-digit one) or when sparse cells + // sit in their own x-band. Projection asks "which x-bands are occupied across many rows", + // which is stable under those conditions. + List flat = rowGroups.stream().flatMap(List::stream).collect(Collectors.toList()); + List columns = findColumnRanges(flat); + if (columns.size() < 2 || columns.size() > 15) { + return ""; + } + + float[] centers = new float[columns.size()]; + for (int i = 0; i < columns.size(); i++) { + centers[i] = (columns.get(i)[0] + columns.get(i)[1]) / 2f; + } + + int cols = centers.length; + List rows = new ArrayList<>(); + for (List rowLines : rowGroups) { + String[] row = new String[cols]; + for (int i = 0; i < cols; i++) { + row[i] = ""; + } + // Top line first so a wrapped cell's words stay in reading order within the cell. + rowLines.sort(Comparator.comparingDouble((Line l) -> l.y).reversed()); + for (Line line : rowLines) { + for (TextWord word : line.source.words()) { + String wt = word.text().strip(); + if (wt.isEmpty()) { + continue; + } + int col = nearestColumn(word.x() + word.width() / 2f, centers); + row[col] = row[col].isEmpty() ? wt : row[col] + " " + wt; + } + } + rows.add(row); + } + + // Guard against false positives while tolerating uneven rows (sparse cells, merged/spanning + // headers). The columns already come from cross-row whitespace alignment, so a stable grid + // exists. Additionally require: at least one "anchor" row that nearly fills the grid (so + // the + // column count is real, not an artefact), and that most rows are genuinely multi-column. + int anchorWidth = Math.max(2, Math.round(cols * 0.6f)); + long anchorRows = rows.stream().filter(r -> filledCells(r) >= anchorWidth).count(); + long multiColumnRows = rows.stream().filter(r -> filledCells(r) >= 2).count(); + if (anchorRows < 1 || multiColumnRows < 2 || multiColumnRows < rows.size() * 0.5) { + return ""; + } + return renderGfm(rows, cols); + } + + /** + * Visible for testing: column detection depends only on word geometry, so tests can drive it + * from synthetic {@link TextLine}s to exercise degenerate-coordinate handling (the crash path + * an extreme text matrix can produce) without needing a binary PDF fixture. + */ + static List findColumnRangesFromLines(List rows) { + return findColumnRanges(rows.stream().map(Line::new).collect(Collectors.toList())); + } + + /** + * Finds column x-ranges by vertical-whitespace projection. Each row contributes coverage for + * the x-bands its words occupy; a column is a contiguous band covered by a sufficient fraction + * of rows, and the gaps between such bands are the gutters. + */ + private static List findColumnRanges(List rows) { + float minX = Float.MAX_VALUE; + float maxX = -Float.MAX_VALUE; + for (Line l : rows) { + for (TextWord w : l.source.words()) { + minX = Math.min(minX, w.x()); + maxX = Math.max(maxX, w.x() + w.width()); + } + } + // Real pages are under ~2000pt wide; anything larger is a malformed/crafted coordinate + // that would allocate a multi-GB array or produce a negative span on overflow. + if (maxX <= minX || (maxX - minX) > 2000f) { + return List.of(); + } + + int lo = (int) Math.floor(minX); + int span = Math.min((int) Math.ceil(maxX) - lo + 1, 2001); + int[] coverage = new int[span]; + for (Line l : rows) { + boolean[] covered = new boolean[span]; + for (TextWord w : l.source.words()) { + int a = Math.max(0, (int) Math.floor(w.x()) - lo); + int b = Math.min(span, (int) Math.ceil(w.x() + w.width()) - lo); + for (int x = a; x < b; x++) { + covered[x] = true; + } + } + for (int x = 0; x < span; x++) { + if (covered[x]) { + coverage[x]++; + } + } + } + + // A column band must be occupied by at least this many rows; below it is gutter. + int support = Math.max(2, Math.round(rows.size() * 0.35f)); + List columns = new ArrayList<>(); + int start = -1; + for (int x = 0; x < span; x++) { + boolean isColumn = coverage[x] >= support; + if (isColumn && start < 0) { + start = x; + } else if (!isColumn && start >= 0) { + columns.add(new float[] {lo + start, lo + x}); + start = -1; + } + } + if (start >= 0) { + columns.add(new float[] {(float) (lo + start), (float) (lo + span)}); + } + + // Merge bands separated by only a narrow gutter. A real column separator is several + // characters wide; the gaps *inside* a multi-word cell (ordinary word spacing) are about + // one character. Without this, a cell like "January 20th, 2026" — whose words align + // vertically across every row — would be split into three spurious columns. + float charWidth = averageCharWidth(rows); + float minGutter = Math.max(10f, charWidth * 2.5f); + List merged = new ArrayList<>(); + for (float[] band : columns) { + if (!merged.isEmpty() && band[0] - merged.get(merged.size() - 1)[1] < minGutter) { + merged.get(merged.size() - 1)[1] = band[1]; + } else { + merged.add(new float[] {band[0], band[1]}); + } + } + return merged; + } + + private static float averageCharWidth(List rows) { + double totalWidth = 0; + int totalChars = 0; + for (Line l : rows) { + for (TextWord w : l.source.words()) { + totalWidth += w.width(); + totalChars += Math.max(1, w.text().strip().length()); + } + } + return totalChars == 0 ? 6f : (float) (totalWidth / totalChars); + } + + private static int nearestColumn(float x, float[] centers) { + int best = 0; + float bestDist = Float.MAX_VALUE; + for (int i = 0; i < centers.length; i++) { + float d = Math.abs(x - centers[i]); + if (d < bestDist) { + bestDist = d; + best = i; + } + } + return best; + } + + private static int filledCells(String[] row) { + int count = 0; + for (String cell : row) { + if (!cell.isEmpty()) { + count++; + } + } + return count; + } + + private static String renderGfm(List rows, int cols) { + if (rows.isEmpty()) { + return ""; + } + int[] widths = new int[cols]; + for (int c = 0; c < cols; c++) { + widths[c] = 3; + } + for (String[] row : rows) { + for (int c = 0; c < cols; c++) { + if (c < row.length) { + widths[c] = Math.max(widths[c], escapeCell(row[c]).length()); + } + } + } + StringBuilder sb = new StringBuilder(); + sb.append(buildGfmRow(rows.get(0), widths, cols)).append('\n'); + sb.append('|'); + for (int c = 0; c < cols; c++) { + sb.append('-').append("-".repeat(widths[c])).append('-').append('|'); + } + for (int r = 1; r < rows.size(); r++) { + sb.append('\n').append(buildGfmRow(rows.get(r), widths, cols)); + } + return sb.toString(); + } + + /** + * A line looks like a table row if it has at least two words separated by a gap far wider than + * normal inter-word spacing. The threshold is derived from the line's own character width + * rather than a document font size, because some PDFs report a unit (matrix-scaled) font size + * that makes absolute thresholds meaningless. (Two-word rows are allowed so two-column tables + * are detected; spurious matches are filtered later by block contiguity and column + * consistency.) + */ + private static boolean isTableCandidate(TextLine line) { + List words = line.words(); + if (words.size() < 2) { + return false; + } + double totalWidth = 0; + int totalChars = 0; + for (TextWord w : words) { + totalWidth += w.width(); + totalChars += Math.max(1, w.text().strip().length()); + } + float charWidth = (float) (totalWidth / Math.max(1, totalChars)); + // A deliberate cell gap is several blank characters wide; ordinary word spaces are ~a third + // of a character. Floor at 8pt so tiny fonts still need a real gap. + float cellGap = Math.max(8f, charWidth * 3f); + for (int i = 1; i < words.size(); i++) { + TextWord prev = words.get(i - 1); + float gap = words.get(i).x() - (prev.x() + prev.width()); + if (gap >= cellGap) { + return true; + } + } + return false; + } + + private static String buildGfmRow(String[] row, int[] widths, int cols) { + StringBuilder sb = new StringBuilder().append('|'); + for (int c = 0; c < cols; c++) { + String cell = c < row.length ? escapeCell(row[c]) : ""; + sb.append(' ').append(padRight(cell, widths[c])).append(' ').append('|'); + } + return sb.toString(); + } + + private static String escapeCell(String cell) { + // Cell content is inline context: escape inline markdown (including the column delimiter) + // but not leading block markers, which have no meaning inside a table cell. + return escapeMarkdownInline(cell); + } + + /** + * Escapes Markdown control characters in body text extracted from the PDF so that literal + * characters (e.g. a line that reads {@code # Heading} or {@code [label](url)}, or an embedded + * {@code }) are emitted as text rather than being reinterpreted as structure or raw HTML. + * Applied to all body text — headings, paragraphs, bold labels, bullets — before emission. + * + *

The generated Markdown should still be treated as untrusted content by any downstream + * renderer: this hardens fidelity and is defence-in-depth, not a substitute for safe rendering. + */ + private static String escapeMarkdown(String text) { + if (text.isEmpty()) { + return text; + } + String inline = escapeMarkdownInline(text); + return escapeLeadingBlockMarker(inline, text); + } + + /** Escapes inline-significant Markdown characters anywhere in the string. */ + private static String escapeMarkdownInline(String text) { + StringBuilder sb = new StringBuilder(text.length() + 8); + for (int i = 0; i < text.length(); i++) { + char c = text.charAt(i); + switch (c) { + case '\\', '`', '*', '_', '[', ']', '<', '>', '|', '~' -> sb.append('\\').append(c); + default -> sb.append(c); + } + } + return sb.toString(); + } + + /** + * Escapes block-level markers that are only significant at the start of a line: ATX headings + * ({@code #}), unordered list / thematic break markers ({@code -}, {@code +}), and ordered list + * markers ({@code 1.} / {@code 1)}). {@code original} carries the unescaped leading characters, + * none of which are altered by inline escaping, so positions line up with {@code escaped}. + */ + private static String escapeLeadingBlockMarker(String escaped, String original) { + char c0 = original.charAt(0); + if (c0 == '#' || c0 == '-' || c0 == '+') { + return "\\" + escaped; + } + int i = 0; + while (i < original.length() && Character.isDigit(original.charAt(i))) { + i++; + } + if (i > 0 && i < original.length()) { + char delim = original.charAt(i); + if (delim == '.' || delim == ')') { + return escaped.substring(0, i) + "\\" + escaped.substring(i); + } + } + return escaped; + } + + private static String padRight(String s, int width) { + return s.length() >= width ? s : s + " ".repeat(width - s.length()); + } + + // --- Page-level emission helpers --------------------------------------- + + private static void emitImages(PdfDocument doc, int pageIndex, List pageItems) + throws IOException { + try (PdfPage page = doc.page(pageIndex)) { + List images = + PdfImageExtractor.extract(page.rawDocHandle(), page.rawHandle(), pageIndex); + for (ExtractedImage img : images) { + pageItems.add(describeImage(img)); + } + } + } + + /** + * Builds an image placeholder annotated with whatever metadata JPDFium exposes: pixel + * dimensions, on-page placement (points), effective DPI, encoded format, colour space and bit + * depth. Missing fields are simply omitted so the line stays valid for any image. + */ + private static String describeImage(ExtractedImage img) { + List parts = new ArrayList<>(); + if (img.width() > 0 && img.height() > 0) { + parts.add(img.width() + "x" + img.height() + "px"); + } + Rect b = img.bounds(); + if (b != null && b.width() > 0 && b.height() > 0) { + parts.add(String.format("%.0fx%.0fpt", b.width(), b.height())); + if (img.width() > 0) { + float dpiX = img.width() / (b.width() / 72f); + float dpiY = img.height() / (b.height() / 72f); + if (Float.isFinite(dpiX) && dpiX > 0) { + parts.add(String.format("~%.0fdpi", (dpiX + dpiY) / 2f)); + } + } + } + String ext = img.suggestedExtension(); + if (ext != null && !ext.isBlank()) { + parts.add(ext.replaceFirst("^\\.", "").toUpperCase(java.util.Locale.ROOT)); + } + if (img.colorSpace() != null) { + parts.add(img.colorSpace().toString()); + } + if (img.bitsPerPixel() > 0) { + parts.add(img.bitsPerPixel() + "bpp"); + } + + StringBuilder sb = new StringBuilder("'); + return sb.toString(); + } + + private static void mergeAcrossPageBoundary(List output, List pageItems) { + if (output.isEmpty() || pageItems.isEmpty()) { + return; + } + // Only merge a sentence continuation between two text paragraphs, never into/out of a + // table. + if (!(output.get(output.size() - 1) instanceof String last) + || !(pageItems.get(0) instanceof String first)) { + return; + } + if (!first.isEmpty() + && Character.isLowerCase(first.charAt(0)) + && !endsWithSentencePunctuation(last)) { + output.set(output.size() - 1, last + " " + first); + pageItems.remove(0); + } + } + + /** + * Joins tables split across a page break. Two consecutive {@link TableBlock}s (no text between + * them — i.e. one ended a page and the next began the following page) are merged when their + * column layouts match; a repeated header row on the continuation is dropped. + */ + private static List stitchTables(List elements) { + List out = new ArrayList<>(); + for (Object e : elements) { + if (e instanceof TableBlock tb + && !out.isEmpty() + && out.get(out.size() - 1) instanceof TableBlock prev + && columnsMatch(flatten(prev.rows()), flatten(tb.rows()))) { + List> merged = new ArrayList<>(prev.rows()); + List> tail = tb.rows(); + if (!tail.isEmpty() + && !prev.rows().isEmpty() + && rowText(tail.get(0)).equals(rowText(prev.rows().get(0)))) { + tail = tail.subList(1, tail.size()); + } + merged.addAll(tail); + out.set(out.size() - 1, new TableBlock(merged, prev.top(), tb.bottom())); + } else { + out.add(e); + } + } + return out; + } + + private static String normaliseSpace(String s) { + return s.strip().replaceAll("\\s+", " "); + } + + private static List flatten(List> rows) { + return rows.stream().flatMap(List::stream).collect(Collectors.toList()); + } + + /** Whitespace-normalised text of a row's lines (top to bottom), for header de-duplication. */ + /** + * Header text of a table at the very bottom of a page, or null if the page does not end in one. + * Trailing image placeholders are skipped; any other text after a table means it did not run to + * the page bottom and so is not a continuation candidate. + */ + private static String trailingTableHeader(List pageItems) { + for (int i = pageItems.size() - 1; i >= 0; i--) { + Object e = pageItems.get(i); + if (e instanceof String s && s.strip().startsWith(" row) { + List ordered = new ArrayList<>(row); + ordered.sort(Comparator.comparingDouble((Line l) -> l.y).reversed()); + StringBuilder sb = new StringBuilder(); + for (Line l : ordered) { + if (sb.length() > 0) { + sb.append(' '); + } + sb.append(l.text); + } + return normaliseSpace(sb.toString()); + } + + /** True when two table blocks have the same number of columns at near-identical x-centres. */ + private static boolean columnsMatch(List a, List b) { + List ca = findColumnRanges(a); + List cb = findColumnRanges(b); + if (ca.size() < 2 || ca.size() != cb.size()) { + return false; + } + for (int i = 0; i < ca.size(); i++) { + float centreA = (ca.get(i)[0] + ca.get(i)[1]) / 2f; + float centreB = (cb.get(i)[0] + cb.get(i)[1]) / 2f; + if (Math.abs(centreA - centreB) > 15f) { + return false; + } + } + return true; + } + + private static void flushParagraph(StringBuilder para, List out) { + if (!para.isEmpty()) { + out.add(escapeMarkdown(para.toString())); + para.setLength(0); + } + } + + private static String repairHyphens(String text) { + return SOFT_HYPHEN.matcher(text).replaceAll("$1$2"); + } + + private static boolean endsWithSentencePunctuation(String s) { + if (s.isEmpty()) { + return false; + } + char last = s.charAt(s.length() - 1); + return last == '.' || last == '?' || last == '!' || last == ':'; + } + + // --- Methods used by other components / tests -------------------------- + + List extractAllPageText(PdfDocument doc) throws IOException { + return PdfTextExtractor.extractAll(doc); + } + + List extractTables(PdfDocument doc, int pageIndex) throws IOException { + return PdfTableExtractor.extract(doc, pageIndex); + } + + List renderTables(List
tables) { + return tables.stream().map(TableRenderer::render).toList(); + } +} diff --git a/app/common/src/main/java/stirling/software/common/pdf/TableRenderer.java b/app/common/src/main/java/stirling/software/common/pdf/TableRenderer.java new file mode 100644 index 0000000000..3f468699fb --- /dev/null +++ b/app/common/src/main/java/stirling/software/common/pdf/TableRenderer.java @@ -0,0 +1,82 @@ +package stirling.software.common.pdf; + +import stirling.software.jpdfium.text.Table; + +final class TableRenderer { + private TableRenderer() {} + + /** Renders a Table as a GitHub-Flavoured Markdown table string. */ + static String render(Table table) { + if (table.rowCount() == 0) { + return ""; + } + + String[][] grid = table.asGrid(); + + if (table.rowCount() < 2) { + // No separator row possible — return plain lines + StringBuilder sb = new StringBuilder(); + for (int c = 0; c < grid[0].length; c++) { + if (c > 0) sb.append('\n'); + sb.append(escape(grid[0][c].trim())); + } + return sb.toString(); + } + + int cols = grid[0].length; + + // Compute column widths: max(3, max content length across all rows) + int[] widths = new int[cols]; + for (int c = 0; c < cols; c++) { + widths[c] = 3; + } + for (String[] row : grid) { + for (int c = 0; c < cols; c++) { + String cell = c < row.length ? row[c].trim() : ""; + widths[c] = Math.max(widths[c], escape(cell).length()); + } + } + + StringBuilder sb = new StringBuilder(); + + // Header row + sb.append(buildRow(grid[0], widths, cols)); + sb.append('\n'); + + // Separator row + sb.append('|'); + for (int c = 0; c < cols; c++) { + sb.append('-').append("-".repeat(widths[c])).append('-').append('|'); + } + sb.append('\n'); + + // Data rows + for (int r = 1; r < grid.length; r++) { + sb.append(buildRow(grid[r], widths, cols)); + if (r < grid.length - 1) { + sb.append('\n'); + } + } + + return sb.toString(); + } + + private static String buildRow(String[] row, int[] widths, int cols) { + StringBuilder sb = new StringBuilder(); + sb.append('|'); + for (int c = 0; c < cols; c++) { + String cell = c < row.length ? escape(row[c].trim()) : ""; + sb.append(' ').append(padRight(cell, widths[c])).append(' ').append('|'); + } + return sb.toString(); + } + + private static String escape(String cell) { + return cell.replace("|", "\\|"); + } + + private static String padRight(String s, int width) { + if (s.length() >= width) return s; + return s + " ".repeat(width - s.length()); + } +} diff --git a/app/common/src/test/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParserTest.java b/app/common/src/test/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParserTest.java deleted file mode 100644 index fbbf5af9cf..0000000000 --- a/app/common/src/test/java/stirling/software/SPDF/pdf/parser/LineAlignmentTableParserTest.java +++ /dev/null @@ -1,153 +0,0 @@ -package stirling.software.SPDF.pdf.parser; - -import static org.assertj.core.api.Assertions.assertThat; -import static stirling.software.SPDF.pdf.parser.PdfModels.*; - -import java.util.List; - -import org.junit.jupiter.api.Test; - -/** - * Unit tests for {@link LineAlignmentTableParser}, focused on the coincident-line merge logic and - * column-grid construction. - */ -class LineAlignmentTableParserTest { - - private final LineAlignmentTableParser parser = new LineAlignmentTableParser(); - - // ── mergeCoincidentLines ───────────────────────────────────────────────────────────────────── - - @Test - void mergeCoincidentLines_singleLine_unchanged() { - var lines = List.of(tokenized(rawLine(10f, 100f, "Revenue"))); - assertThat(parser.mergeCoincidentLines(lines)).hasSize(1); - } - - @Test - void mergeCoincidentLines_distinctYLines_unchanged() { - // Two lines at different y positions — must NOT be merged. - var lines = - List.of( - tokenized(rawLine(10f, 100f, "Revenue")), - tokenized(rawLine(10f, 115f, "Cost"))); - assertThat(parser.mergeCoincidentLines(lines)).hasSize(2); - } - - @Test - void mergeCoincidentLines_sameY_merged() { - // Simulates a financial-table row split by LineBuilder at the column gap: - // label fragment at x=72 → "Revenue" - // value fragment at x=350 → "1,234" - // Both have y=100. After merge they should form one TokenizedLine. - var label = rawLine(72f, 100f, "Revenue"); - var value = rawLine(350f, 100f, "1,234"); - - var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(value))); - - assertThat(merged).hasSize(1); - // The merged line should contain tokens from both halves. - var tokens = merged.get(0).all(); - assertThat(tokens.stream().map(t -> t.text()).toList()) - .containsExactlyInAnyOrder("Revenue", "1,234"); - } - - @Test - void mergeCoincidentLines_sameY_mergedLineHasCorrectBounds() { - var label = rawLine(72f, 100f, "Revenue"); // 7 chars × 6pt = 42pt wide → right = 114 - var value = rawLine(350f, 100f, "1,234"); // 5 chars × 6pt = 30pt wide → right = 380 - - var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(value))); - - var bounds = merged.get(0).line().bounds(); - assertThat(bounds.x()).isEqualTo(72f); - assertThat(bounds.right()).isEqualTo(380f); - } - - @Test - void mergeCoincidentLines_withinTolerance_merged() { - // Lines 1.5pt apart (within ROW_MERGE_TOLERANCE_PT = 2pt) should merge. - var a = rawLine(10f, 100.0f, "Alpha"); - var b = rawLine(200f, 101.5f, "99"); - - var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b))); - assertThat(merged).hasSize(1); - } - - @Test - void mergeCoincidentLines_beyondTolerance_notMerged() { - // Lines 3pt apart (beyond ROW_MERGE_TOLERANCE_PT = 2pt) should NOT merge. - var a = rawLine(10f, 100.0f, "Alpha"); - var b = rawLine(200f, 103.0f, "99"); - - var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b))); - assertThat(merged).hasSize(2); - } - - @Test - void mergeCoincidentLines_threeCoincident_allMerged() { - // Three fragments at the same y (e.g. wide financial table with two value columns). - var a = rawLine(72f, 100f, "Revenue"); - var b = rawLine(300f, 100f, "1,234"); - var c = rawLine(400f, 100f, "5,678"); - - var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b), tokenized(c))); - assertThat(merged).hasSize(1); - assertThat(merged.get(0).all()).hasSize(3); - } - - @Test - void mergeCoincidentLines_coincidentPairFollowedByDistinctLine_twoGroups() { - var a = rawLine(72f, 100f, "Revenue"); - var b = rawLine(350f, 100f, "1,234"); // same y as a → merges with a - var c = rawLine(10f, 115f, "Expenses"); // different y → stays separate - - var merged = parser.mergeCoincidentLines(List.of(tokenized(a), tokenized(b), tokenized(c))); - assertThat(merged).hasSize(2); - } - - @Test - void mergeCoincidentLines_numericAnchorStatus_correctAfterMerge() { - // After merging, the combined line should be an anchor (≥2 numeric tokens). - // "Revenue" alone → not an anchor. "1,234 567" alone → anchor. - // Merged → anchor with at least 2 numerics. - var label = rawLine(72f, 100f, "Revenue"); - var values = rawLineMultiWord(350f, 100f, "1,234", 30f, "567", 30f); - - var merged = parser.mergeCoincidentLines(List.of(tokenized(label), tokenized(values))); - - assertThat(merged).hasSize(1); - assertThat(merged.get(0).isAnchor()).isTrue(); - } - - // ── helpers ────────────────────────────────────────────────────────────────────────────────── - - /** Creates a RawLine with a single TextFragment of the given text at the given position. */ - private static RawLine rawLine(float x, float y, String text) { - float width = text.length() * 6f; // ~6pt per char — rough but consistent - float height = 12f; - Bounds bounds = new Bounds(x, y, width, height); - TextFragment fragment = - new TextFragment("tf-test", text, bounds, y + height, 11f, "Helvetica", false); - return new RawLine("ln-test", List.of(fragment), bounds, 1); - } - - /** - * Creates a RawLine with two TextFragments representing two words separated by a small gap. - * Used to simulate a values-only line with multiple numeric tokens. - */ - private static RawLine rawLineMultiWord( - float x, float y, String word1, float w1, String word2, float w2) { - float height = 12f; - Bounds b1 = new Bounds(x, y, w1, height); - Bounds b2 = new Bounds(x + w1 + 5f, y, w2, height); - TextFragment f1 = new TextFragment("tf-1", word1, b1, y + height, 11f, "Helvetica", false); - TextFragment f2 = new TextFragment("tf-2", word2, b2, y + height, 11f, "Helvetica", false); - Bounds lineBounds = new Bounds(x, y, x + w1 + 5f + w2 - x, height); - return new RawLine("ln-test", List.of(f1, f2), lineBounds, 1); - } - - /** Tokenises a RawLine via the parser's own tokenise logic (package-private access). */ - private LineAlignmentTableParser.TokenizedLine tokenized(RawLine line) { - return parser.tokenize(line); - } -} diff --git a/app/common/src/test/java/stirling/software/common/pdf/PdfMarkdownConverterTest.java b/app/common/src/test/java/stirling/software/common/pdf/PdfMarkdownConverterTest.java new file mode 100644 index 0000000000..b3c104da85 --- /dev/null +++ b/app/common/src/test/java/stirling/software/common/pdf/PdfMarkdownConverterTest.java @@ -0,0 +1,269 @@ +package stirling.software.common.pdf; + +import static org.junit.jupiter.api.Assertions.assertDoesNotThrow; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; + +import java.io.IOException; +import java.io.InputStream; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; +import java.util.stream.Stream; + +import org.junit.jupiter.api.Disabled; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.MethodSource; + +import stirling.software.jpdfium.PdfDocument; +import stirling.software.jpdfium.text.TextLine; +import stirling.software.jpdfium.text.TextWord; + +/** + * Accuracy and robustness tests for {@link PdfMarkdownConverter}, comparing conversion output + * against hand-authored golden Markdown for a set of owned/synthetic fixtures. + * + *

The {@link #gatedFixtures()} set is enforced in CI: those fixtures currently convert within + * the accuracy threshold and guard against regressions. Fixtures still being iterated on live in + * {@link #wipFixtures()} under a {@link Disabled} test so the goldens stay in the tree without + * breaking the build. Enable the WIP test locally to see per-fixture scores while working on the + * converter. + */ +class PdfMarkdownConverterTest { + + /** Accuracy threshold: output must share at least this fraction of content with the golden. */ + private static final double THRESHOLD = 0.95; + + @TempDir Path tmp; + + /** Fixtures that meet the accuracy threshold today and therefore gate CI. */ + static Stream gatedFixtures() { + return Stream.of( + Arguments.of("multi-column-test_lorem.pdf", "multi-column-test_lorem.md"), + Arguments.of("bordered-table-test_widget.pdf", "bordered-table-test_widget.md"), + Arguments.of("many-tables-test_stress.pdf", "many-tables-test_stress.md")); + } + + /** Fixtures still below the threshold; tracked here, enable locally to iterate. */ + static Stream wipFixtures() { + return Stream.of( + Arguments.of( + "wrapped-cell-test_expense-report.pdf", + "wrapped-cell-test_expense-report.md")); + } + + @ParameterizedTest(name = "{0}") + @MethodSource("gatedFixtures") + void convertMatchesGoldenMarkdown(String pdfName, String mdName) throws IOException { + assertConversionMatchesGolden(pdfName, mdName); + } + + @Disabled("WIP fixtures below the accuracy threshold; enable locally to iterate") + @ParameterizedTest(name = "{0}") + @MethodSource("wipFixtures") + void convertMatchesGoldenMarkdownWip(String pdfName, String mdName) throws IOException { + assertConversionMatchesGolden(pdfName, mdName); + } + + /** + * Degenerate/extreme geometry must not crash the converter. A crafted or malformed PDF can + * position text anywhere via a text matrix, so a row's words can span from near the origin to a + * coordinate beyond {@link Integer#MAX_VALUE}. The old column-detection code sized an {@code + * int[]} straight from {@code (int) Math.ceil(maxX) - lo}, which either allocated a multi-GB + * array (OutOfMemoryError) or overflowed to a negative length (NegativeArraySizeException) — + * taking down the request thread. Detection must instead bail out and return no columns. + */ + @Test + void columnDetectionSurvivesDegenerateGeometry() { + // x ≈ 2.5e9 is past Integer.MAX_VALUE; combined with a near-origin word it yields an + // implausible span that the pre-fix code turned into a fatal array allocation. + List rows = new ArrayList<>(); + for (int r = 0; r < 4; r++) { + float y = 400f - r * 12f; + TextWord near = new TextWord(List.of(), 50f, y, 30f, 10f); + TextWord far = new TextWord(List.of(), 2_500_000_000f, y, 30f, 10f); + rows.add(new TextLine(List.of(near, far), 50f, y, 2_499_999_980f, 10f)); + } + + List columns = + assertDoesNotThrow(() -> PdfMarkdownConverter.findColumnRangesFromLines(rows)); + assertTrue( + columns.isEmpty(), + "implausible page span should disable column detection, not allocate from it"); + } + + private void assertConversionMatchesGolden(String pdfName, String mdName) throws IOException { + Path pdfPath = tmp.resolve(pdfName); + try (InputStream in = + getClass().getResourceAsStream("/pdf-ingestion-fixtures/" + pdfName)) { + if (in == null) { + fail("Fixture not found on classpath: /pdf-ingestion-fixtures/" + pdfName); + } + Files.copy(in, pdfPath); + } + + String actual; + try (PdfDocument doc = PdfDocument.open(pdfPath)) { + actual = new PdfMarkdownConverter().convert(doc); + } + + String expected; + try (InputStream in = getClass().getResourceAsStream("/pdf-ingestion-fixtures/" + mdName)) { + if (in == null) { + fail("Golden file not found on classpath: /pdf-ingestion-fixtures/" + mdName); + } + expected = new String(in.readAllBytes(), StandardCharsets.UTF_8); + } + + // Image placeholders are not scored: their body text is a TODO ("ideally, add the info + // available about the image...") rather than real content, so comparing it would penalise + // output for matching a placeholder we intend to replace. Drop those lines from both sides. + expected = stripImagePlaceholders(expected); + actual = stripImagePlaceholders(actual); + + double similarity = similarity(expected, actual); + if (similarity < THRESHOLD) { + fail( + String.format( + "Markdown output differs from golden file '%s' by %.1f%% (threshold %.0f%%):%n%s", + mdName, + (1.0 - similarity) * 100, + (1.0 - THRESHOLD) * 100, + unifiedDiff(expected, actual))); + } + } + + /** Substring identifying an image-placeholder line, which is excluded from scoring. */ + private static final String IMAGE_PLACEHOLDER_MARKER = "Image intentionally redacted"; + + /** + * Removes non-content lines from the comparison: image placeholders (TODO text we intend to + * replace) and GFM table separator rows (the {@code |---|---|} divider, whose exact dash count + * is cosmetic — any run of three or more dashes is valid Markdown). + */ + private static String stripImagePlaceholders(String md) { + StringBuilder sb = new StringBuilder(); + for (String line : md.split("\n", -1)) { + if (line.contains(IMAGE_PLACEHOLDER_MARKER) + || line.strip().startsWith(" 0) { + sb.append('\n'); + } + sb.append(line); + } + return sb.toString(); + } + + /** True for a GFM table separator row, e.g. {@code |---|:--:|---|} (only |, -, :, space). */ + private static boolean isTableSeparatorRow(String line) { + String t = line.strip(); + if (!t.contains("-")) { + return false; + } + return t.chars().allMatch(c -> c == '|' || c == '-' || c == ':' || c == ' '); + } + + /** + * Character-level similarity: proportion of expected characters that appear in the LCS. O(n*m) + * but golden files are small enough that this is fine. + */ + private static double similarity(String expected, String actual) { + if (expected.isEmpty() && actual.isEmpty()) return 1.0; + if (expected.isEmpty() || actual.isEmpty()) return 0.0; + // Strip all whitespace for a content-focused comparison + String e = expected.replaceAll("\\s+", " ").strip(); + String a = actual.replaceAll("\\s+", " ").strip(); + int lcs = lcsLength(e, a); + return (double) lcs / Math.max(e.length(), a.length()); + } + + private static int lcsLength(String a, String b) { + // Use two-row DP to keep memory reasonable + int m = a.length(), n = b.length(); + int[] prev = new int[n + 1]; + int[] curr = new int[n + 1]; + for (int i = 1; i <= m; i++) { + for (int j = 1; j <= n; j++) { + if (a.charAt(i - 1) == b.charAt(j - 1)) { + curr[j] = prev[j - 1] + 1; + } else { + curr[j] = Math.max(curr[j - 1], prev[j]); + } + } + int[] tmp = prev; + prev = curr; + curr = tmp; + java.util.Arrays.fill(curr, 0); + } + return prev[n]; + } + + private static String unifiedDiff(String expected, String actual) { + String[] expectedLines = expected.split("\n", -1); + String[] actualLines = actual.split("\n", -1); + + List diff = new ArrayList<>(); + diff.add("--- expected"); + diff.add("+++ actual"); + + int maxLines = Math.max(expectedLines.length, actualLines.length); + int context = 3; + boolean inHunk = false; + int hunkStart = -1; + List hunkLines = new ArrayList<>(); + + for (int i = 0; i < maxLines; i++) { + String exp = i < expectedLines.length ? expectedLines[i] : null; + String act = i < actualLines.length ? actualLines[i] : null; + + boolean changed = exp == null || act == null || !exp.equals(act); + if (changed) { + if (!inHunk) { + inHunk = true; + hunkStart = Math.max(0, i - context); + // add context lines before change + for (int c = hunkStart; c < i; c++) { + hunkLines.add(" " + (c < expectedLines.length ? expectedLines[c] : "")); + } + } + if (exp != null) hunkLines.add("-" + exp); + if (act != null) hunkLines.add("+" + act); + } else { + if (inHunk) { + hunkLines.add(" " + exp); + // check if we're far enough past the last change to close the hunk + boolean moreChanges = false; + for (int j = i + 1; j < Math.min(i + context, maxLines); j++) { + String e2 = j < expectedLines.length ? expectedLines[j] : null; + String a2 = j < actualLines.length ? actualLines[j] : null; + if (e2 == null || a2 == null || !e2.equals(a2)) { + moreChanges = true; + break; + } + } + if (!moreChanges && (i - hunkStart) >= context) { + diff.add("@@ -" + (hunkStart + 1) + " @@"); + diff.addAll(hunkLines); + hunkLines.clear(); + inHunk = false; + } + } + } + } + + if (inHunk && !hunkLines.isEmpty()) { + diff.add("@@ -" + (hunkStart + 1) + " @@"); + diff.addAll(hunkLines); + } + + return String.join("\n", diff); + } +} diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.md b/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.md new file mode 100644 index 0000000000..4b590e4b63 --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.md @@ -0,0 +1,10 @@ +# Widget Inventory Report + +This report lists current stock levels for each warehouse. + +| Region | Units | Status | +|---|---|---| +| North | 1200 | OK | +| South | 950 | Low | +| East | 1430 | OK | +| West | 875 | Low | diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.pdf b/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.pdf new file mode 100644 index 0000000000..8da041e28d --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/bordered-table-test_widget.pdf @@ -0,0 +1,74 @@ +%PDF-1.4 +%“Œ‹ž ReportLab Generated PDF document (opensource) +1 0 obj +<< +/F1 2 0 R /F2 3 0 R +>> +endobj +2 0 obj +<< +/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font +>> +endobj +3 0 obj +<< +/BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding /Name /F2 /Subtype /Type1 /Type /Font +>> +endobj +4 0 obj +<< +/Contents 8 0 R /MediaBox [ 0 0 612 792 ] /Parent 7 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +5 0 obj +<< +/PageMode /UseNone /Pages 7 0 R /Type /Catalog +>> +endobj +6 0 obj +<< +/Author (\(anonymous\)) /CreationDate (D:20260603003133+01'00') /Creator (\(unspecified\)) /Keywords () /ModDate (D:20260603003133+01'00') /Producer (ReportLab PDF Library - \(opensource\)) + /Subject (\(unspecified\)) /Title (\(anonymous\)) /Trapped /False +>> +endobj +7 0 obj +<< +/Count 1 /Kids [ 4 0 R ] /Type /Pages +>> +endobj +8 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 500 +>> +stream +Gas1[9i&Y\%))C:pc(6t3;pQ8%0T@tD+-Nf,MP8j(>R4tMCHL4NI*\1+UNiI'V9NC%VeJKn/YI0J];XQt&X83?=ihrg<*Mcn1n!1nWcDaQPe\P"9gnJuHl(jf]JQgZ[,&^uobI4QF',k"*^S)3c;)GMWC(T'=",ErnS#U=YCUN0&q4+*KmK1Zd*NI\GQDiZUG7;PTja8lulb"\PWWO#WcfI[ZB:6s*3g$be%?JH(n`oaEJ[XE'%QW=HE04M<,;ERm[MS=uYF=nN3jG'f@#?O48Ia,6Y-3m&tTWVq1?DeiBkp.Ug*;lVZX`Z=P.eklHhNV;!R_?QOuoeJ<0%7idG7GM8boU$^>N.N,2^;25]0Z8M<<]XMCct>noC'Qfb?`*[Mo+,F9#>t~>endstream +endobj +xref +0 9 +0000000000 65535 f +0000000061 00000 n +0000000102 00000 n +0000000209 00000 n +0000000321 00000 n +0000000514 00000 n +0000000582 00000 n +0000000862 00000 n +0000000921 00000 n +trailer +<< +/ID +[] +% ReportLab generated PDF document -- digest (opensource) + +/Info 6 0 R +/Root 5 0 R +/Size 9 +>> +startxref +1511 +%%EOF diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.md b/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.md new file mode 100644 index 0000000000..4b456165df --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.md @@ -0,0 +1,222 @@ +Intro paragraph for section 1. + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | + +# Section 2 Heading + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | +| golf | 301 | india | + +## Section 3 Heading + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | +| golf | 301 | india | 23 | +| juliet | 401 | lima | 33 | + +Intro paragraph for section 4. + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | +| golf | 301 | india | 23 | kilo | +| juliet | 401 | lima | 33 | november | +| mike | 501 | oscar | 43 | alpha | + +# Section 5 Heading + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | +| golf | 301 | +| juliet | 401 | +| mike | 501 | +| papa | 601 | + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | + +# Section 7 Heading + +Intro paragraph for section 7. + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | +| golf | 301 | india | 23 | + +## Section 8 Heading + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | +| golf | 301 | india | 23 | kilo | +| juliet | 401 | lima | 33 | november | + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | +| golf | 301 | +| juliet | 401 | +| mike | 501 | + +# Section 10 Heading + +Intro paragraph for section 10. + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | +| golf | 301 | india | +| juliet | 401 | lima | +| mike | 501 | oscar | +| papa | 601 | bravo | + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | + +# Section 12 Heading + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | +| golf | 301 | india | 23 | kilo | + +## Section 13 Heading + +Intro paragraph for section 13. + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | +| golf | 301 | +| juliet | 401 | + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | +| golf | 301 | india | +| juliet | 401 | lima | +| mike | 501 | oscar | + +# Section 15 Heading + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | +| golf | 301 | india | 23 | +| juliet | 401 | lima | 33 | +| mike | 501 | oscar | 43 | +| papa | 601 | bravo | 53 | + +Intro paragraph for section 16. + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | + +# Section 17 Heading + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | +| golf | 301 | + +## Section 18 Heading + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | +| golf | 301 | india | +| juliet | 401 | lima | + +Intro paragraph for section 19. + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | +| golf | 301 | india | 23 | +| juliet | 401 | lima | 33 | +| mike | 501 | oscar | 43 | + +# Section 20 Heading + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | +| golf | 301 | india | 23 | kilo | +| juliet | 401 | lima | 33 | november | +| mike | 501 | oscar | 43 | alpha | +| papa | 601 | bravo | 53 | delta | + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | + +# Section 22 Heading + +Intro paragraph for section 22. + +| Name | Qty | Price | +|---|---|---| +| alpha | 101 | charlie | +| delta | 201 | foxtrot | +| golf | 301 | india | + +## Section 23 Heading + +| Name | Qty | Price | Region | +|---|---|---|---| +| alpha | 101 | charlie | 3 | +| delta | 201 | foxtrot | 13 | +| golf | 301 | india | 23 | +| juliet | 401 | lima | 33 | + +| Name | Qty | Price | Region | Status | +|---|---|---|---|---| +| alpha | 101 | charlie | 3 | echo | +| delta | 201 | foxtrot | 13 | hotel | +| golf | 301 | india | 23 | kilo | +| juliet | 401 | lima | 33 | november | +| mike | 501 | oscar | 43 | alpha | + +# Section 25 Heading + +Intro paragraph for section 25. + +| Name | Qty | +|---|---| +| alpha | 101 | +| delta | 201 | +| golf | 301 | +| juliet | 401 | +| mike | 501 | +| papa | 601 | diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.pdf b/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.pdf new file mode 100644 index 0000000000..f12925cda3 --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/many-tables-test_stress.pdf @@ -0,0 +1,169 @@ +%PDF-1.4 +%“Œ‹ž ReportLab Generated PDF document (opensource) +1 0 obj +<< +/F1 2 0 R /F2 3 0 R +>> +endobj +2 0 obj +<< +/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font +>> +endobj +3 0 obj +<< +/BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding /Name /F2 /Subtype /Type1 /Type /Font +>> +endobj +4 0 obj +<< +/Contents 13 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +5 0 obj +<< +/Contents 14 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +6 0 obj +<< +/Contents 15 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +7 0 obj +<< +/Contents 16 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +8 0 obj +<< +/Contents 17 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +9 0 obj +<< +/Contents 18 0 R /MediaBox [ 0 0 612 792 ] /Parent 12 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +10 0 obj +<< +/PageMode /UseNone /Pages 12 0 R /Type /Catalog +>> +endobj +11 0 obj +<< +/Author (\(anonymous\)) /CreationDate (D:20260603005358+01'00') /Creator (\(unspecified\)) /Keywords () /ModDate (D:20260603005358+01'00') /Producer (ReportLab PDF Library - \(opensource\)) + /Subject (\(unspecified\)) /Title (\(anonymous\)) /Trapped /False +>> +endobj +12 0 obj +<< +/Count 6 /Kids [ 4 0 R 5 0 R 6 0 R 7 0 R 8 0 R 9 0 R ] /Type /Pages +>> +endobj +13 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 1007 +>> +stream +GauHKgN)$k&:O:SnDZmeMUK]*(BdhV"2[X,j0;*\%_E*0HXJRc`q%4S#/VM`O[Rh4i5T.Phi,$]aT$05!q:PBuCQU/.p2]@1koN*Tb*K9k5G\ht+Dr\K=+8\NZ"alMaOEo**@OK:8-.O1X3-?Gg`@m3%,ti3'">T-&c=M&Wuu?cDbGDp.gOF0!r6&&CLC$tM?fIR"M37["/*k9@YkpKKSS@Np"1/4#R]`I^(g*1Sc,9L3Qt6N(T@F`A>oBGgL-!r#Y4)R\G`D,iVtFeJUE9u4iuUQ?D%C&SB4kkp.>D5tii>nDKJ"Y07jANhOb$R(_$=U7Stjs)-/KZ(6IBm`6u<3;i.Bh4+MFJ"H:.XWUQX6%LU(sg4Tt$_":5`p.2,gZkpUfdg,Hd(qR1)9\ltF2^8b5,,XbY%VfhD52O7A.c.u]dhTc5t%0<0L5E38`Bq+i;"%J7kcc#@i)@okdN-qiaA"33DdPgPrUs7;j%+47_cnGN@&ug9/dqGOAHA4f.N*&guHspf?E;GIc-Wt>:<(m1AmcS_Zc2VlEI'_S_>@#!MF^m1$$endstream +endobj +14 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 987 +>> +stream +GatU3gN&c;&:O:SkV;H,eI!J\f?X"LSPIZ'"4Y.F#oAAY?N.Y?8Iu:s)eV;,aNG1Lea>G$!MSc\rU6m7jF'AOI09V1,^TSTi$A+f4sZVQ%2^>uj$40\AI>WbZo9q!NL^CS9P3E9X.:XEMb$a/X"qH9hN6fY,]D!OEOVT,722%RJnVqK4f'[4d3(/hbJWJRs,>>28jk5A:nCl/%FlC#LntsAC5cE$q`QO_@Pk%Lp>7V23")NNrFkcXfuoI=SJJ-`g)_+K\@,SQ@8ORGg(_(B.[u[WXD%Q8@I";9d(]M62K_)Q+-<\kHY5,)o99%BB2bcQVkd:+MZVH=$Z`S@BhH]-XfPANk@qBN.:Jc'?.Kn\p7(aPIQOI(du7U]WDrNNPW%d"8P_j^Y8d%G,'JJLXrV.UD!ah19&9b#*fgR2o730<9)QX)X0"EK6u;mBJs3/fl\d&K$B3bXI5R>F*E,^\%+b?0o8s$o.CuX4uDAT_?G0(p4N3La,8qfi'j?e`e88KrZ8*NDgc:62'icCb],4B]Z"SO&e/BRC*C$2PmbSKAjc\$FLD8?&qHVH3G4p834/77n[%()-p/JDRBapcO2b\,3dW%#U9u!ilSkd%2'rr&g[u".1O]W%0J<@#(ptUD;%0:^.^ls:.#.V6NgR[`2\RUW!k$-*GZIf3DOBiB_Mu"=LVGlO\1Z]0pF(@UaiVQ=dh?h5kfQ_gj>$jHW$8TCt:C.jVU98ne.XoiMtoV;q8XhnlOE;3a]=dGJ8KL&+Khj"&P4Q=P-:nHsDprY"B4+#A3dF,;b&gf(sT\Ts*iH)f!KhfHV;\mMSHej.9.XrBG41_<:b)MG;-Y(:&nCKS0F]r&CGl>EfMc0td\0GHScl3Lr;?OUoo6619Qdr)S8S);]7L5rmWW&(5t4Dil"USDhhY/Y=!=!X:85**$:~>endstream +endobj +15 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 976 +>> +stream +GatU3gN&c;'R]XVkgAc"70atdYFXp#3h7VV#7/=-%N$"G?N.YO&iX#cVsS_l_5i8)eum:1-)(7`mXJ7LnhqY0hGZal,We=q^e"$M]Lu;_=4AoIN$QOjOF.@i*Sg!_rLL=-'?kk;m;M&R`$D>'HCL7]$A,-S`H1I/1T]4\B3=gQ9:I4j9f]+af*Z7*GH5H+;j73Z-T#DAcgpOW(/'b?XF9+X+'-Dqi"+1U7?6=ZRJNBSJ6ojDTa+]o4RF^26C+i0s(9G2o;$@RNbs\(f75],L(SVpI0M!'7JVhZ"$t/%kb1+)F\p54/@jpVJF2+)7^F=gVcp8(h*m%F#'p9c%9Ep9?b7g^F5>(>I3t^d_"f'aue%^E\X4kN:g\P[;sC=g3h,nD/n9*`94uoM?sBCL4;XilfMPT4]&FC.U%DI(c6uf3*XPJP2MRdq@Ag3uYF().nqLFAE2Nih'(^p\5FnSNa>n+cI?U65A61_N1<9ZXH4Xr)Fgq(6md3H9[MQ3Q^%aq>E?XM:O:RKGjc+QV2MB!.I.17QpSt]X[C*6ap.`#'o;8F(ebZ.VMM6k`BkZB\I,2MNoX]J"k]Qd";,@($aWX&29q"6k;>Let)@]QHm:i/lf[DB_&U@ROq-0MLQp;@j+]?$"E)brXVun52AVQo$"[a7@!?!LT*>$&:5X_.R;9)'$l-S.>WjLQIVTo.JU%(H\,-*pl_kP9~>endstream +endobj +16 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 1024 +>> +stream +GauHKflEQ9'Rf^WgrHc4<#\/Qm7c-rFIIq+TFSD%Z#L'6o(Nl&^_!k0ds=.%-s&pt#i0QBKE91*<^/MJJCcfoHA_e;aL?\&_BAj]Dt;HW$6(8Ql\%O7-5%uT80ZIL#\l)(+[ABOTNKmq!2F,S:.F93Z6QIu[nf\Q[qCE`Y-=.8k$67iCR!7=El+50a@f&b^6'Yg'mqlcRQ-5j:%[K@'5!A_m0L)&VlW1T50^X)"3Ma3,U1P8Di$>uPZ)Zj_igBc0lm]WE0#e>*M53(Yh6\9jHYh6B+3bM\b[Q9j'O_9:!:W0$Ijm,#(bKn_D?UYMGJ7LLqC0\[$l<8U*O5^GnN;!N6YB'?d[d8DAPt^>+E:r0<2$'VL/TtuNg;C@5iM#%KdS-GU->0]`/20eLXVW#qnjsU.P:Xj&=enVE[mpqDCn2!qDRuQhI/#cZ-#m6`U9'SI4"9\"r:4p.\W>IK_$#C,EkU#uO1g6*HoUK':XiJjtQfI'Sdc)B3J1eqgGJ+l?"$//&C$^f'TA!2:K!0pd;^Zif)c[:f!C/8hNKc7+'+1WbOFKr9T7c_G$s*O634O2Zs:"iN>j,pWU+:kEQ>5YbF=$27*59iJEA//7(KsL\pUG.G[+JB-r8t;Sih+(W.A'8Q*LZK*I9WjCLjEt>lT%2RFJZg0Ps:5c,HdK#tSNI:#Rp!]O$Ic=T%K0[._E_Fcjn>OcM,iSAf9iF7,,+Qe:@o'Y4o_:54t*l+TAgendstream +endobj +17 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 1075 +>> +stream +GauHKgN&c;&:O:SkV;H,6&QZ[5)3-pL]0bpl':D91fR,o"F49.1/bfmFt32ljt6eO[j7!K]\nZ:^QQASu[l@3no(6*HL'@3I!Qfi7%,.N\.>A91O)D\n5*o@JWSOI@ng5p/"_I`g*Y$j0+5PWaso'25ltcglFd,QIaO'dPO0Yr\h5/,=m\PiP:P+Gg0&N;_5iLN$]BD.#VL+h,tSe\fL&,;+2fg&%\pAJiV#eVhq@ro%%/`[(&8n29YiYd>bDF^Bg[AVf/n>[FE(au"ZAe&>%Q[7/n3-QhsPXuPbp#<$d"UR8u-nBnr!Q$Wjd1;9WN/,KG8])Ca5rrDrI'Ue'*I0C%pn+f`Hd"4sB_&2*)S$pX=F/6"DC"TU`bMtd/m7u]?]=J5L[SNBuE:6Ri,NmW7.*qUp)97Tt+a84b\os`Ti8/uRDlhrl*Xrn(a0\,%4%*k^rL\k3%6ako`&A3MoN^CXV77D!Qg%=j9Ge+0B@dZDHELb2fDU(,iiXXRl^*U,U#Fld!R72UIpKGWu#-DXi653a?U$fqCs18gt(5/#U%!u,gkMK;>OT_/='S:kn,iD@u]<8-rlTRu'+\7L$U7-#!5-5\WQn(n:q@-p24u!~>endstream +endobj +18 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 590 +>> +stream +GasbXgMWc?&;KY!MRgt!"n`^Bf\9$c*`Q-Sb6tfd8P(q-dT1eng=QV-!OG*\kiWo2AGBdKLqI^%h,XNB)-gDk+GFVBa?g*a%:!J\97Vc8-jH8>8P>0S@Wr)o*"Hu<6qPp`EF:M3'l4@S\c2]`&[J%dO@5p$<(([H6:SlB;FS8L`&pO%6Y\V"/O7F]E%L@d.%>L@"4bK_XXs&";n_.W^Nr847HH,X@$.pC"4bYC?9RH,dN?h8a+8/!?Q=D\F41Nc@'B`';4XgMeh!;U23C+>AbMQD-o--lJ/":m#(Yt,X5(`1KL,GTMBLlr6=-1mi#k.-Ou\\T4$Pun2dEU(\?$&;1@T4dm^t!KuOD-U@N_AMr"&uVpsGm,+8I7B*f!%9.o4cC1a[CZ12(hd>1*0bU`k2-MXo1[Gor#kmXGIM'#R49X#NSAOpdf'0ilUH4:M(^Snc3;m/+>endstream +endobj +xref +0 19 +0000000000 65535 f +0000000061 00000 n +0000000102 00000 n +0000000209 00000 n +0000000321 00000 n +0000000516 00000 n +0000000711 00000 n +0000000906 00000 n +0000001101 00000 n +0000001296 00000 n +0000001491 00000 n +0000001561 00000 n +0000001842 00000 n +0000001932 00000 n +0000003031 00000 n +0000004109 00000 n +0000005176 00000 n +0000006292 00000 n +0000007459 00000 n +trailer +<< +/ID +[] +% ReportLab generated PDF document -- digest (opensource) + +/Info 11 0 R +/Root 10 0 R +/Size 19 +>> +startxref +8140 +%%EOF diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.md b/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.md new file mode 100644 index 0000000000..5c35de111f --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.md @@ -0,0 +1,25 @@ +# Lorem Ipsum in Two Columns + +## 1. Origins + +Lorem ipsum dolor sit amet consectetur adipiscing elit. Sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. + +## 2. Structure + +Ut enim ad minim veniam quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat. + +## 3. Usage + +Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur. + +## 4. Variations + +Excepteur sint occaecat cupidatat non proident sunt in culpa qui officia deserunt mollit anim id est laborum. + +## 5. Typography + +Curabitur pretium tincidunt lacus. Nulla gravida orci a odio. Nullam various turpis et commodo pharetra est. + +## 6. Conclusion + +Nunc nonummy metus. Vestibulum volutpat pretium libero. Cras id dui. Aenean ut eros et nisl sagittis vestibulum. diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.pdf b/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.pdf new file mode 100644 index 0000000000..36dc3a1a65 --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/multi-column-test_lorem.pdf @@ -0,0 +1,74 @@ +%PDF-1.4 +%“Œ‹ž ReportLab Generated PDF document (opensource) +1 0 obj +<< +/F1 2 0 R /F2 3 0 R +>> +endobj +2 0 obj +<< +/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font +>> +endobj +3 0 obj +<< +/BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding /Name /F2 /Subtype /Type1 /Type /Font +>> +endobj +4 0 obj +<< +/Contents 8 0 R /MediaBox [ 0 0 612 792 ] /Parent 7 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +5 0 obj +<< +/PageMode /UseNone /Pages 7 0 R /Type /Catalog +>> +endobj +6 0 obj +<< +/Author (\(anonymous\)) /CreationDate (D:20260603021636+01'00') /Creator (\(unspecified\)) /Keywords () /ModDate (D:20260603021636+01'00') /Producer (ReportLab PDF Library - \(opensource\)) + /Subject (\(unspecified\)) /Title (\(anonymous\)) /Trapped /False +>> +endobj +7 0 obj +<< +/Count 1 /Kids [ 4 0 R ] /Type /Pages +>> +endobj +8 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 861 +>> +stream +Gat=i>Ak00&;B$5/*8Q/Z)fmNp(^2^OK'MC[dT6#Oq.&b^4c4;1No6>+D%TC.n+fice+l9R0[L%'(osQ?dQP@AUa.A:T=#%ZIIl18hB7ZfI(hVU+?`]UK%;aabAH:q>9NUGQq^.=u])Wt-ETI))C'a[FE@elj!RSrf2Q'F>URAO.C,!DneTPqrj#4e2kb9%1"4qfZ)"#&^0j9nHQ?9nF!j7mVPP5\*Uq'_jMVS]9%`kQB\8*AF_bpr/hGj;HCUOSQU-%5:6S79Ud\b!*tPbr_'pCr$Ea#(FYP31NFhSX.-("1M:$cgH#hX8L(2]R3Q>'BYHCS%pI!;=WdJp,'ii[`QPZ_9mcd\baZ2U(_;c\-p,8EoIEpQ*lstL>]LE;C#\dLnT2R:)BM-fTc['3_He[U,k'!Bo".uERd>SkhRj^J+koSIrZ_dEf_5L'/1h.`+DTK(R:P,WH)h5\se=SZ"L/5b8b..,e/E\o+4YQ+*im^C>AERG/TieEK\)#>U@HXnJ,H0A9-MqhhkDp8%.6Lr,OrK*lih;B<-opZ8%EU?,$r^jmCDAQ`-0/-8/`[p]7Fm%0f:E&S*FV)DX2>#q\bRqA=^_`43#8EA$u%8r6F5`rc9>K@q>E4q^*~>endstream +endobj +xref +0 9 +0000000000 65535 f +0000000061 00000 n +0000000102 00000 n +0000000209 00000 n +0000000321 00000 n +0000000514 00000 n +0000000582 00000 n +0000000862 00000 n +0000000921 00000 n +trailer +<< +/ID +[<21a9fbd0a0991a91b6e6e2db0856056e><21a9fbd0a0991a91b6e6e2db0856056e>] +% ReportLab generated PDF document -- digest (opensource) + +/Info 6 0 R +/Root 5 0 R +/Size 9 +>> +startxref +1872 +%%EOF diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.md b/app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.md new file mode 100644 index 0000000000..8008a3b970 --- /dev/null +++ b/app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.md @@ -0,0 +1,62 @@ +# Employee Expense Report + +Reimbursement Request + +EMP-1047 + +**Report Header** + +| Employee Name | Michael Tran | +|---|---| +| Employee ID | EMP-1047 | +| Department | Client Services | +| Report Date | January 20th, 2026 | +| Reporting Period | January 5th–16th, 2026 | +| Manager Approver | Laura Simmons | + +**Company Information** + +| Company | Summit Consulting Partners | +|---|---| +| Company Address | 88 Riverside Plaza, Suite 1400, New York, NY 10069 | +| Accounting Department Email | expenses@example.com | + +**Trip Purpose** + +The trip was undertaken for client onsite meetings with Atlantic Energy Solutions in Boston, MA. + +**Expense Details** + +| Description | Amount | Date | Category | +|---|---|---|---| +| Flight (NYC to Boston roundtrip) | $325.40 | January 5th, 2026 | Airline ticket | +| Hotel (3 nights at Harborview Hotel) | $822.75 | January 5th–8th, 2026 | Lodging | +| Taxi from airport to hotel | $48.00 | January 5th, 2026 | Ground transportation | +| Client dinner (3 attendees) | $186.20 | January 6th, 2026 | Meals | +| Parking at JFK Airport | $72.00 | January 5th–8th, 2026 | Parking | +| Breakfast (per diem not used) | $18.50 | January 7th, 2026 | Meals | + +| Description | Amount | Date | Category | +|---|---|---|---| +| Uber to client office | $22.10 | January 7th, 2026 | Ground transportation | +| Printing + presentation materials | $46.90 | January 8th, 2026 | Materials | +| Lunch with client | $39.75 | January 8th, 2026 | Meals | +| Office supplies (notebooks, pens) | $27.60 | January 10th, 2026 | Supplies | +| Mileage reimbursement (client visit in NJ, 42 miles @ $0.67/mile) | $28.14 | January 14th, 2026 | Mileage | +| Team lunch meeting (internal) | $64.30 | January 15th, 2026 | Meals | + +Total Expenses $1,701.64 + +Reimbursement Method + +Reimbursement method Direct deposit + +Notes + +All receipts are attached. Expenses are business-related and comply with company travel policy. + +**Approval** + +Michael Tran, Employee + +Laura Simmons, Manager diff --git a/app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.pdf b/app/common/src/test/resources/pdf-ingestion-fixtures/wrapped-cell-test_expense-report.pdf new file mode 100644 index 0000000000000000000000000000000000000000..95a0b2e07a20b9c76d6f6901dc7ae55fdbf83b8a GIT binary patch literal 95841 zcmY!laBiE*l$t=c3falFa-(m&B4(4HqjT10yp7 zGXrBIV`DQDV{HQibpr!+O^B-eA`RdCs?406M14~|1BDn3-^Aq1ypsIl3~L3?ypq%$ z1>eME1^-}$2n9ofctg{8^B7G9$AW^K)bP|K@5~Z?Q)6>IV>5*q4R4PSUmqQXoXqT0 z1^3kC?EDx_1&`Fkl$_M!V&{ya{M=N1LlZMS0}xaQPE1QI%7kjx)c4I#L3OK%fdRy= zh7h*~6y>LsCZ`r@1eatMjt>EDU=!M85-&tXzGV#mgJ;r6vN~SQqpuSO)Sj} z%?%7pjLgjqjSbB$3{CYiN^)~F?d*8DQu9*4VP%9IX!&_1sd**E3WgxjAO(Hj)RfFb zr~Ha&1p@^G1yf5?JxfDNOA7@H6GI3qRzD!IC^fG{!O&E}Kp{v$KPa_0zqBYhwb;f+ z-?gH|J-8&XB-O@7-`!Zj7_7q1PTws*uf)bi-_6iK!4PC_kb=IOp`n5yDE>ff69p5n zQhhg51*oxZW(sCdF>?iTD9b{@0?M*fumrR0?DPYQ@{@y8OQQ7yT-@|SQY%XIJ#!P& zQ=K4`GlcStwX@R?E-5NaE`hkeSiwL)Br&O2KUhDcvLIDIATd1^Ig~(d2M3ppjlPqw zzF&S(Zeospa-xDEvMP|X;i?jgQ$ZfpcLW1%kJOy9)RN5PM8DM1R9&b1oD_YR)Z*mS zyp+Ve5|F~;Xayq^NN~jJyXGb5r)1`(>wBi8=9OfYRO))@2bU(5fZPUhpn-l!erR51 za(+sxf{_U{;6M`Kz(sZwC;&0s=$t&H-5&qyPy>kk62v1Byiq=QtDT z97uGN>KsVY!0sGSQo?YK3l`_#iZN42P9f1vrl9nMD~Le8#Bh@*AJ?>cCgDc+5$V%E~_>wlb{K9aQ2Np+R3nWmi zlN@Jepm@O*XW&8{<{r1qoRZWceYc#%l2n&eP`RP+lbV-alA&O1Y-nU>S6ot*nwZPW zRWavn?Co_rTLo+9uYd8}!%#1LYV0(DOZ#Leu`4-loASJcAxNY{;mQBS?_QS0-^o5^ zTm9ewVErcGXdPJ)XLX?f#Oap|2G?XX115 z&soRH!gzxB)den_Vw9?KKX5^x`@1jo@8%m8R;uYd|6Wn&znJg*q^a)bbc&vDI)81a zjtE=Ps`l17`~N?_A1~TfvP|Q$gP<=Kp|F}QhUhL!Xz(4)#m-pZE6kpW6vHI^Xm-L=j|3#C& z|9bl}`TgO`>UUr7_nQ*qA+=V%`fvL8jo-BnI|#U+ihN`<;at3nxDNZpqa6Io7b7+&L_{rsQM^Bx2Z1Cw*RgvP%lD?TcUu$l1 z)Lz{lp?x_>d#iF`V6}gy^wlc{tJK6-?O~a(&FXP3NMQesW05m&u!=5~*uKl>;L7Q9 zQm*+}u20lUd$uLz@ns8#SDUkLeDazjxm0Cp$K=NoHXrqh&nhwueZlSOV%z89>XjC` z%}V8*`qV8K7A#(~b#YCTv7(u7h_A-+bDb)$=KR|H`nKKSt~77gZrRq}rlq{!RFC`I zEJ;i-+UP!|C~UU-r5~PNvwYM0Ciw_;Xg;^OZGMxZ%P{a=@(!a37Y-;(RcOjot?oYI zEvK5(b7bcxH`BmPUeRYcJ6l!XESD?kI?3|xsvlipNDQmTnj%} z%oe*wMw_co!duSfd^{?l^wn|BlQT83N>dI#nj)}svWG^r*vyra$}_aK{E%IFDBkA& zc^l`&m2C6tPCkoFl7HwU6SQ{GO>duV3%;@H@Ts1-Y^M`EzbWpn`5jgjF+-PU4<7Jd zIKaLzQSmcBtJH-@#|0l4HXS{wH~a1T51Ru#Gh5}aX=c?j{1aR3>1k5aWtqin-w&^4}M9=@`RC`SK*d?Q)w89tJdsnYw z-Xtfp$yaO3*M!Afi5tUzS1>3vl%HcSXWZVKAtf8VwexjduJ*pi9X;iYD{HOx96S8? zf%>*X{!jRM>Y93gE2!xwDo!|l+K{8?+L6@2g?9u5>=b66UsTIHVd-LruomwN_ZKnd zgh+l6zxCO;VbW~f6!xhHZY|CzZ%Wzntn0K21_pP_pCV;^3_=O0(RU^`!WgqIi z62}b*{ipe#3LU(E-a2Y^`5BW+_ZOjZ-&i~5CFfgZ-WFT6Airk4DRYdhO_WZf;lYnl zC%3vRvVFH>Va%^B2D3yTC9S$}ZsDIFa)~Z>Qf?yJ2ReS_IjQQ;oPIFbNs4{B)tpHW zR$dTS^ktA!PVct9yY`vvqZlUjnhs5aDGnhMmD-t=1D_P6gsnCIwxB6L$H_4I<&CrK z!cE?8g`P!Pa~3`jN`IPqL-P4$wu_lprevRcyQT92!>cvFJ{q!j&dgc8zLY(5-f{g^ zZJ+K~Nk=A!UFW>Px_C_p>lEw17FY7L9VY3tY+1!Fa_IfT0A^FG5Wj8jF3L4}$K=g7 z*tzo!&t96^^*vW?J_2Y>@9v)-SalaeD?DF); z<8v>(?iGmd_FP*geE8U1w&$DE{jLA42z9^6ZgAKB&PV&Ww&UwK_!b!1tXt}1HEE`a z_ve52GR~A;nUOwovGPF+md4UMhcBMtbxjG;-naC3Wpef>y_s3B3q8L+)w%g3XR&$e z+ASfmpA=bF7i6k)dbfT*a<16pP5R27iOyx3kqYb1&3q$svsJ|Bx#{^i+^;4bjamHR z_LcdylYjm8bJb4E`@p%3v%Q*cMO?I6?8`|_~u#>cUR zdmOLqvyf}Oqfo?Px+AxEM_1X5mp`sZ#QZM!(z^T?Yg<>o<*G&Zl%4rHR)n5h#~HWc z=);364fyM4H5(l~B)3{cUEru^TmhHj4c3xxx$U)^e<7UZ&f%fr~Z#P>09$2?k*u-__GCS&vamKo-`%Gifvv{xP zUX75-7r(l_@zT=OPAk?2^Ht{R9guyi(ff6y^7MHLi`logC44%Vv0zE~SAB7XoolMN z56AwDJ^1BsSg+&m6LrxlDaz}sb0-Nb-gw`!{^E&{`R-Mfttmy)GuOO{Fpd_U>)&zi zWKFQy!W|1fGu8&al5q^$AgkCW;e5y2Qz&~yUd%O}L&8V@c7BoaUfR2KQt{bK?bY4; z{w=aNaCYB;vWX0f_8()IY5!6;Tj9T__*MGn zA9AcYaiSYK*Uhe5^7g8|LtiYX9Cb{XXrlyC(Gh{@Dcvk!kxao*I_E_1tqw zsYT0OsrmVgZ|h=m_BZ^xblLQ83iH?4mLvNfZVsrsn7}P8X!w(H#ivxR1ydi&J}p#; z-n3_jLx*RxS+kfv>lEFNsZX}sl0bhwgmM=19ZQ!9UNu zr`Ev8fTy{D;nxH4FGsg(O*#=M|EKy`_2%<6`jYE27N0jM{d=Bqor_IqhVCT2)2kx% zE*}-TnzEABY@>Rm9Q!1dIHk`Y|E&2F5Or$W;&)rvZr#1{@lTq)?e2rBiTqW`VtsS} z&;E0UH8n?PgC^^Crr8ERFTBuQuXpB8&COq>!PA%cu?cG}t211>pHVYW^vS#^vrAjm zAA0{ZihC3FG*;?#(X5+m_X>R5eO%V4WWB)D^~-kjs!n^P-1DdXS=@ZV(DJhNVxN8& ztnV$IJ>kgG6%)mk+OA%c<~n8Yv%#xzFXt*hxxO9!;;$Aw`YV(9_2`|_mp_jzRXM!# zXW0F~jS_|o#cYcIugdsIzs`}`Bh7Sxm4kDW{~pE@E-S9^UpKoD@onbkO_l67>?Ncc z{xK>_mdS)Syf`#X`tv)7sX>vq3)9=bv)n?@?@r$SrmpaxX=|1auV&!?{XvGmw`yEz z)iuB7wag^to zo{;FBia(lVivyk>Hhx&FaJ}NnjGpNh7XL3jdT`B{N!jM>eEGRy z`+w$5b&4Os17Wbv4{S`#5Y#<~^_-mHgJO^_;{Xkc(R={R5bpq(A;|NvAi>?4G%`0e zM(Iq(MEBp_CRk^`->E;hvB1(;W)iam+qY+1_%$D-G}i*An!r_07*^AuLc2?&)BI)gZtA9r( z-TCl+_j|r8Q|;&1-;%Ab_<#1{v*4Heetuur)8T04eP!?eS0BDUeEYE7*>=WW|8rNY zJ}%OW(3}7FDgWK?1sDA-s^i5m$5n031L+)Is&oA}$IJQmU;g7jK_cO2T(|x+}EGLDFW3M7>utEQt4WZ{z4m!3vov1a=JGTNEcxSi>^+xjZS*PU;F&&4=lU$2 zyQao`zR%Mdqfc*Bmgd?;{=b{E=~9{9O{UZ>4wAY7ZpEi=6dOlo&i21nZ$5j?mbw1R zxF^3nl(NfU<~g;=tfx{MjUpHLoYN>isZ^@h?PaLVx3|``x4HK3MEA#!PVP`Cem`ls z<@5McUs5~DUY<6nU1@kdep|tX{=c8?_y2!%__zH1T8ZTsE{l}=x6d{YVR^qujOEnb z;F-(XmnN;(%Sw5BW=Zw0cOOnw@2kFR`}c8zpHZdn%FlN%-`^E}ul-KB%5lTAkBM(C zJnneVws|k3Vt-N^TF-)bhKclCShxZer!_a+m2k>|LTWw`a7>l`!V(EWY~i zH=m}m#VTLfmnNY?Pj~1k+62bT^9Wa&wawFH##W)dd$&EhE%>l-rtSgVN%dc}lTO?| zvcS^f^urh)_q@&BowesgO&7jbC{moYvMuxzyXFf2NzbfuX1!;dCsX-u+5TTQs;@QO zy?5@%ve517=N}%~e&M98N#)a9wniUjPZZR@84`T@#~x;X(~Mc3!CY3?OG*xYz`xl{9Tz5f3CudkkcyW1|m|IfFVtN+)$sQdo)dj0>}@4rgJr(AJR zn6DDM#rxjWl5Xug?Qhl{3tE(0w(zLlIh*U3}1MNS57YyS=IdS zZpL2jki$o>6)`>PotF40&GF#uHv%dIRqb$G{IeB$PK?cDhl`a<7N^Vcu?@=FjsMi}hV#cp4yb;G@XH7oU|r z__NG-n-pP`xa7d5qo?{eguQbso6d1Rx4uGg2K(K0JE!Fpubr^Rd!PQQkab&i{~LATVx|L4Q3ryVYOh$*q6k#Yt^b9<}J&=3h-IWeQPWDcAoEHWpBJInZJv$wA?o1^UO)9>rWNfwA^SCY1o*0!)T>zU+7w%=ltTOKfUk%TbEgu zyw~%l_3Q6>Q4a8C8rYg?@Cf6IQ>v|Tcu!l&JT zsIr;$v>Gg}ym-0wucSgBzy0N}EHT@vuJv~Dx<6O+;8ByFakDmOgVf%X*QvWdK3Kfz zNBoh4?Qiyqm>D)W3;Hxa;E?FLAbBdQ(dyoF&)okjHYX%(ZjiS+;LiGi?e~tmcgvb(_jN~Ty6au&@@#Ux_U^p$VGe0| z)+I0g+%ej4_V9}RzniT1&KC6km7er!>cMO7bsqE5x1P8X7$K5gDPfv8pLfGLOO2;) zcW<(C%`81QD>!YPaLd7xBZX?gtNhK}E-!tu=$FRXrD>%niy!F}KlasA=i8@wMJJ$* zv%&ex6+vH>RyNNofnOedZn061;W($xJ6SF@>UNUKT_?S$M^hH)2#CJ^eY3Xf&eYkz zuUTrA1@SD)+>>CC{`K^oiC<2ZR*G30&-gU$^O=NGJ3SI|FCShM$EC7%CySwwU+=aX zKSgx2Os^I+&vEKFBl?Ozdj`MZ(`Ua{Se?=eEXsCXGIOsE!-F|7KB7_!J60%4xF!ch za_FASp7cs=jZgP%aerU6OL~3RBqrZ!TgA~Lz3$^Z&vUCd+I+f}9+@8VX~W4atHfIz zYc}oVEUdIW>pfFFa@&S4ch+9CGHMU(`@C|&x?`vJ6f+(^dfZDWz;pE-#j4{HXO?_i z8Kw}KK*f1+c>-5&d7-V^Mv8iEs7^8b+Pa5}^7aks~zp|O|c z6C*?Nf1e4w88*9p6Zaqc&ipi@X~LZ1v#gJ*eXcye;$&7=<+I|csOik)p+H@LJaB(1#LJ7Mw@=V@Bp zo(Y92cRnsjTAe&cZ~gSwDRu(-GTqZfH|bv4*)?m$qL)19Q@exnCWbt?BX&4krIgKm zwn?76vXjaLE0Y=PRhlxZBI2sgo@^^#=&iZz(uAaU{e>E57#1{j*`5rG^@$8t)snie zTmR*q;*Yhv*{{~jE16tww_&}a)!cvft0tX(KToi8>xr&wZd0fI=2#ZPzCFZO$m2Byh-^A@VlJ8{Y~GHjmn0jBot$!D)g zEiyD$W_3L2Bi)|w+QZ?%?3Z4stNpm}Vh!sO?Gvo&OC~d#bzE7MeNx=2?P##k>C(HR zSx*wJ-X$M6-DmqVH9#cjx%2;HKDjkYKY8j!H~HQ>CB8RQV|U3PRo$uAon|-b8Ohn5 z&wYM2F4@^kH6itlr1MV8)QxxE&D-Y8`9;DhJK0&jX%$E0y7#<^Dvi_nT$prie{oN% ztE*gbpyFKiG1csa3O`HY-k#BGDJjbo`iVqeY)cv%mp^Fv|!!5lOO}z#0+enCl0?!dyRqWH*5#0b!PrF?hKH z$d`~>x)`&32&`>`>0cFBcd`R*gFWr zOyKFBR7Zg{BFr%~)enFL6&a2~OZO&_l%-$-iA*xW9&7@_c_yIb3rZCV&;pX^um@>G zm}3a4*}-Om3?sD`1$zf!mq#n;dU~y@N0ex^51XXhHrbz7hg!M3`d;s@1_} zljSJ1bZ-hx_oj$+4{huc9cCaC;LeLyFa@RGAO!_OP#%Idtx@tP{w54qJKQ)0BT$tK zbtH-@L=`mq zLBR#fq(p~3NF%}=BTzE|>N?U35U_U;hM7ZZC{TiclpMr53ak-fjuEI*2b&G@A*uBz zxU_;B7Oh|oP50)YgaWBQNi1`~CcsTpFb5@Hkjr2NBhlUjX+)S~1ZpOL%?24pvNu8A zK^SHZO1>a-NOlxRBf=abP^AtHDl!rTDBj_QMJt$tQY~oA2$AkdtUtjfAe#tEzo0<{ zP#%H~NujtCfBk6=(hfIH!3fk^09yeHI#Q#+0^}=%VHWV@4|7+regM8fDzFBGeq&H& z4mOi4$DkGW7SLpG0ZJ#3WKUwygH3=tFIvF@lzc&*Ur^9PM*&F)da!o5aSFzu8W(H@ zS-u7N3T{}mf(0lQ2Pr66g2X}1D+8!w4WVPN3Wm_ZNd-gbXn=wtxE}{n3~j_J7(!bB z3WneU93%zJR|U%3#Kne@}09b5-HeY$*bs9(q%xT`y3dTmT5HJOK5Jyyi{6kin z0|h(GHW~$lIVd13LFEd_g+vDgSO?5$pnx!j1%w%%fB^XiWDTk1o+T*V!EB>ZKv;kS z0-8jLPZ5?N9WbYX0s@{V%y9<<*gs?i1SsRdY@<;?Sb_rrQs$75HNd45%xR#2fR|Dh zcme|CAF=`hRH`7^W(g~fK^YWOj2b|aw1S}lD1so>5wt`F<#N0Q5ZEx7;owjP?V(LA zDvm}iLShl+3u$GMw}K_COn_A&AV<02D|)~>V2<)uFa|YQ!44%W_&~t~GYqE>%{3|U zAuO{K??aFdn4`QEj6tncOOQj!@u49&0l-Yd=|u}lyl7|uDwja6BifG;Eih+!D;R@X zzYw>A%ptX4grshmaX5WxNtrKUnUn-yg0#S#<*i^0YV|_gN}ex4r4ih?Xaz$9Q1%CB zb4Wf?Ff;%q9*_ul6C`Q*9bz2Je4L?ZMp-CAN+{5rB($kWFv}Ynfb_uJ<*i@>YOg~a zOI|R70t;pwPH$3_>kXlE^rU$cl*wW4@>VbbHPj6aK#nEPo1mlsGY+RWDa-lb{l6gB zDHswRn_xXKcX=zAfCk%O=@wK)LMkd^`}T(5gbp(fr#C6f`rsYBqb zZ4k$T3?jLN1GlN+#ziX_f)|2=b2=z@fEl3l17d&|PLP=Q!N$SN#~F;2<$c)M6afmb z-XY<<57q;7m$!lmXm}9jcXCn-D6nA0;q)eDc^}r@C&`YP~@m3o?l0OaV^lFynA~ld`-IUOfbI z9SPn9>w&q;Tfqd>gfj#?7Gx0V-b5)%q7@9G%Z?NbVM~1sp=*5<48dzmNX+|S<6zFk z8H|+WefX*&_<#`Mybsm`bCG`78G8jXA0z^1ZQkg zmiNI6tw63LAvVE!kh}?Q#X)l=8Lh$D7DS2~KZP znD;^D3ncFopJ+jPkYW?uii74#GQ0_%CxIImtzZaVvj)oh(8XH{hS2p~3Wm@%R3znn zuyHW+aRwu0c^|Q^4mRyWI8%W2z})4nUxD^M@mE?F6r6`G3Fa$3G1m%73!aFbnlzzYrPzaOQ<_8-`Ftt#Y z_uy@WFbSP#ry-r&6gFt39`5j+e6Za)yU(#{YZSTN&orWVTbK72JHDc(dK z83r%mhd36J_eu6AC@m25CuNx*yap2FJW~9LTqt=ffOiELA}yRG*`J`KPSBr}<$mzu zNmBiZTq=1hfVTx0!j?M)q{DWi5k0C2UJwEECup-6bit*9A$YAKDAz+*?kN~TSId!9 z^@ELrnU6DHP?r7SD=6XXI|yfgupXGZycNJZ84SUW1w{zy;Rp^bn0Yw;Nm>4fF53ll zZ@|$8E}la0cW=OYVD9o(0B>c0MJ+ky95^w+%){wV%JM&A-6?Fn4iWK*TrPPlfOkh2 zf*ng{d?J@8IQ>ak{x<`aF(BuWl5RnIVBznr0Nx$}%a~-vCrWt|tzZaUE2>}!U7x97 z2wraqDhQzKB1y{sVB=u!#Tkwi=6_HV1`>|YWnDz17myy9yFgoTLAeU-Sdc-a=YMc$ z!OX+yPs;K?coi+kdGO_Hg#8KD19O+R0(eUYENaO~FW|%gGY_XfDa-%xC9|aX6Rn*C zZWV(a3konY;uE<%!Rb%R@;`hPE-C&*E|ENaQ|Cpa;{%){wV%JM&a=`kt(M47sRY&kImJC?ls4_>%T(4Um$ zfAE57kn>3KCvv&utpMJ|VhDCD$RN_oIq(`im_I@7fAE@QaBc_X5D){r_7uzjg)xcw zA8Z`Ve4OcpvixrWYQuoM4`0MhIRAt7z})4n0Ny5J2zD&UATq)c6k0IzaQc(N{0}N) zK+Yq@pCCOjcX=y-citGn7h;p{PjF&@nTOM#l;wZ;a$OSq3DyI1m$w3V>y07Uv7i7W z!=IqE4l@s@KPk)q&}IFgmM1tH64Rdp>w&q;TLHYA#}K?InoNJ9lqb;&hVTs*;03VY zY!6C8pdC5T<(DM2|G~z=+>0|DDa-%xWx4RZ8-(*eSP#ry-jIDlFu#*iFM>l0W*$y| zQkMVWi*rfwCvv&utpMJFWC(UFndt?T7zp~4qWo_JUP=sd9`XK!=z+P*8?yPx&=TZW zkU=EpeS)3DSe)Ptf!?EF#G$)r`Qai{a*hQU@eQDHwrQ!$MLAY#AbvM+O*y zmyRPF3SOuSs$f7h3&c+JXu-ef(+F$=+(-pu(0n+=9(03b7A1V)BY@z^O+7YCn0ABk5qM`E8X;cO9oG9o(8c+=b zqM`1Co^7H4o|^z2KLnMB4&ExjMxLR=ckpuwpu-jl&~Ci~wA-ryZ9Ky^c`87g{R)uH zf1oZbw4tZ~+2I8i0i{L|1Ke}~Ge9L0m;v>MA+-2WFod2`s$d8UNkee408$JKOG9Y# zQ!s=^wgO~jDM$@??IoB24Om0)L3JPzSkM|m*L5ivLRUU17(&+u#e$Ed1*ZTb#9CFv zfe5h3hBbE-U=@V|teArByT=)BhTyd@Am>0=m?#)RmtZIuLf0rL7=kCvL2AG=%3uaG z+66i^}u6dd4qva!+kE6pti-Jxp)%UgyDkk|v2>(E1jLD%Sl&hJG& z9@hw*R$!+1Waeg;pznf4-V+V#VfZEHrlPA+fQBHlT60YW3yAZPMJ+THpka$FYN-i3 zb0QYib)Xy!a~<>qWUz~a6wnlc0vx6geps+?Vo|ndURplr?1IFi?9>#6lA==R2^las z(2c%{>8Zu=3x**_XMh3U}0=- zhI}W=*~sb~>64ZJ-|9?%e_T&<(xe|zlY}%s>IN>-jhvJk&Hwt%mG#!nR>wDf-L!O0 z%F0PKcA}O?9BK>w9ar&w;I?5W_uq9IX=+6eRgEx5#oGvoGbUiB{F5>&g zzk3TP`_tC#Hs2ZFnKJy=KN~-<_Q&nr@0kvDH>`X4>CC64t^Izs40|jOd|A4( z^X7G{Utj(+->_rw2@amTSeU==-)g~z*Q@~xWDiCEVLR}X(ZuJh-d?d;3Vsdo>^r25 ziytrknDe=uQHklt?VA^`KRX%T!qdRZ@M-GBnyI+b*vIf9ZpQrRm|MTZq8MwO zo_^+@{8|2g!3Q&ke~bs11Ma(hG@W6f>8qGX&pZ|Yh zzT^4(#mj#d-t4;1eaH5Zn*AgH&&z-6zn-(p@a^_hzol7CJ=bx_x>QWMBkR$j+)`zc z$ojzlUZb$4N!XVi7Xl2~$`joAdM7-*EBVAhdXlKEl8|UufD3QQ2UXD7 zMy5VHleX*8)lOl+ZxhK^Z7BVtcC?4cV7kc05EXUdMuyro4f`g2{87 z8he!8js+(C&0pZ~O5lN>Q~bi!d`eER?LEK!+HLbk-0e^QZT;@gw|?*a=iL20OGICkPdJq|VdV^7P1av7 zQH_CDi_$9I?#f?Vo|5XCJu#x|BqQJbwMYu>R9-z^Peew*-I zw-r77^{V8zr6N~?+x}YfeE)SV;M|9=$2pD`FSiMBvy3$oSfVaCm-B+@ffZLddl>kV z`HW{A5ImD8cRJ#LiC!|7_k@crg(l8!+*TXpbv@o*i9Y@6=cX$5l%7XiYFp;<`BuqX z-qB>KlPLLl#+9mL8yW(25+se-47&|wT6-*Y9FCbx+i+q_>>*X3WVXu^M}^d8zB~SP zX`#sE83%b%dXo7z&q%jeX4vVPqVRZvTJ4kjO4)%wRm47MU7ph@nYO1xv#qeHcd3Wl z&ATGcRn{4B3rkIk>`9Ud<~hG)S)quq#8HW~i1SL%3OK!Y9IIJsJKaS7`Njh#u}3X- z|2Im}$=nuy#b)=BFaAbbrZg9}B?NBX*J61h;+V`jqjN7bPMzLrZ{l)y;JeRf4>^^Grf1QCE`{}fuCjL)vtSIl!x#zuj#k^Er>E1OP zW!IguTwr%`_qwcqOSf1?9I;88(H8VxXLtH6SGT7p3eP0uwaZU^t}2!;a#Qu$!zQ)8 z$1eCAew%hmV)l`29`V%LiM`PY%F|8Q+zi)0tefh3>~lg$?;ij2hyGgMyw(^KX?y;Q zPlDg$Z#>gBo)9>1B;U4WN9+Nc-}+C3D)Z*7JJEZ2cTw}SCq;XZh>5eQGP* z9%Fs)g$`RYqm^y#vJ0;hYt^JhvwW+QE=2aQF|#zM9ZoZCusv)OE*-Y`BF90G`SW<# zEZ!VT-;k)6A5nJNbhFj&lPOK}ZW>4_{(EP(#(z_ynREW4%gIJgMKcb}GtQ<-gh;FHQ2A_pV?{vrec@(>$hBb-Iwe(S(ssm(h0I$qKSb+_)wd%GJJy&< z_n6E{*tpwQY%Bmnd+jtesg+n%8-@=sv6Jt2wLU<~;44Wosz$+QqZ! zsN?eEiS7&TedIBVdCL1eZ|AdF8}Gd=$ue_cf3hILboQ$0TwCwmEjsHTykXfgwy=q9 z4JwKo;Dk**T$O=jycJ zuR#mVcBr!zuUWTGZqanN?bn-q4td|dYMZ(;K>D8U`f&YiE$832Z|-Z0jMa_0TpDV; zt!mwbv%Ww-LkDe2C;MZI>_(JE4{<}RB}M2<{+cOrS^*Cms5@}GP=^59KgUY=VmGjC&? zM&H+?tI~45UJAOW@uu_O)#>L_1Tq#WUJyAx^Ks`Y$>_ZyE8gDHy&m&fCjaDf#pRhC zOD6w#GxLn!iQ1}_0o4`T_r&h4IREDMetBM96Q1az85Y;Me$2miY~iP47w>-J`x;dL z&Y|4=5vx>NtJ$w3YK_nB)V}!6V|;KZ=!R;m>bqyp^mpDWo*h|~DE#EkVZ~3$zy58N zS9ttHPUnR9O{3%LqBC4rca>f+pCa=kWn%7|5QdE$$}C~#T|335i{$D5{pDHkx%qK7GL&p>uy zawt6K|C!L#V8xNq*1*Rh(bOQwF{7;^m7}6K`ro&0f9B@?e4G0-mSe-*@U2nTL^&2Q z6~;Es&{b;TU^#h{k@eXcrtoQ2r!&uq-As%5r0){$dGhAO?82?jqix?w)lQlGDQMr- z$Gk=NYTEkKe^gZSU%xj^@nZjdT@kLK+cb>9*`Cn&l^a^?J;Wg;Nn)A2xoHGH$N zu*_V}UrN);P9zEPy@{UE`@qb43^Hu61 zX6}plHf0rOzdg!_xU#C+B|j#;`ef^k z`d8-Ic`er4_BH#GZ%lz=-~QkWQ=>F%t8+=|~=tDolzp4A&tk$Ph6*xCWo-HwccR1!$iSerh zH{&xe8YGdG!8zib+nXou*BsU@T&S9`l`Tq2s)+G*R!+j2-PJV;V$u@t-^odL6slaA z%E@9ByYNEen)~x@1zydR=sC)!UR9;W{`liTPXUo_eG~1D#W9tOEZO8*@i0wckBR&X zsVSioghWqBbIGi_&v&PN@e}jX1C41G7Z};w9vtA1nbF34y5IqdLmwt+%Ur$Z!fYGu zVtnVrrauezc9ybs>WJ*vZ1V3ckB@Tn!y0St7+u?Y8}}?|G2dj7)sx~;Etx%MO8$BF zo$gC6Zg|A!Qp7mhbV9G&CdaC~e{O26w4eIAV&&oY?`7Vdo4ZE+`QMl8Civ_;UDoOp zn`!_3Z#(n;&ZEKlF*&!*D{X|mWv+c{{O$axNm|#ntY*Q3JJui1KibE!d)nQyiFMnI z4#jm3;-;VrqHT&ao z2Y>Mmyj%UkpVVGdh+kwh@1ggL&yxGL*i|>jXh!U>o~+V8$9H+0)txuzWjI)^FV|gs zz5Up?&*r6PZPkT3C+{uPviLXYvUu5!$Bx&(-~J`+r+IOHeg@}t37#xPu zOYkW%`L@k%lj3y8DXHOe4=0qhA9URRwZYo=A5W>p;=143xk@dE;=^ChE6tUQx%Kab z(94+)XN_gQ7RJV|ka^7|6Iput>$=NxjbHWc`aEgd$1R(Cm&cykem~x}dYfIX!W%!! zvvS$aZ$$+6eeC6)Eq|=|tJ#U;E>d|7W-J*Vp}5e*byr z(yRVb((11FFZ2IXq;G$*RLSpoUh+ii;r8n%{qf3P_Q&GQ!uqORxmR(1Y-jQV++ez1xv$d1!GgASOrUv21CnO)KMGo)>l}};~wlV zvOu}=HFlzJ-eCoS*6(wYrkzx3vRIg@ayrO6@OHS|osd^d z7m}aV_Q=JAPCs>%oamk5@`X2|c76lN7l=-y(ui6MR$w}{Kd zq>HWsPZsijb1qL_xvSib%U5dU8Y$K)c7^kMpPwr~|MQ$}x?cMOhKNFTn`LqP=A7KS zucVaW!9U&u(%19zJ@;w3hNRz!#axda$+?9gMQ3w9cZ`ioY zWdE;k$M651yPsKSKC|25ONZY4xp&}-KjS)<1v|EEFN z@ssF*+_h_B=l8Y?=`u*NefYk5bN>3t_pg^RR&Y4%vfRC^;H|ayeKwx^%qGFFvu8*D zu5Xh$uyaLSZF1sY+uc*xca$^kdBPMk<47pOqkow*I>p33F(|D#Z(PCersK%Wbn6xa zldpP3!GoI|-~P8G*)J_*(0{LfG3&?0aN+9mFZPKYUvt}f3l6-k|M9dn``bbHIVbA< zMD6Xqv(5g{?f>pCNSWeae-l;)x2s$y3RHJ;ad6mpzEl1n;iNx_;~?h()6RZ}4~Hf@ zi)JivRM2dmd~Dy|H_}rRy-qnyViJ0%{4O?wv8L@j!@9RGtlQ%Q=Wo3d&D5Lp{`%Xa zo!$DuVTIlwr%wNPzTIk(D*ueWDBFMab$|cN)_HK2;iS}G8pRb(wum1iC+n7J`cmH#plRa6t#eU(R@SdyXKQnnIi?qFdm-{^KkmNf> zr#lQiA?nO4JtdE=ywq>;j{lU=_W!Y!A_spPH0}RZwVg*#)ie3@;@x+o?;3e{u}$YF zKm8#1($Cm^kN+$U+M@H;!&>;W^`*+p%(A-h%Ln)APhG@q{j_IG_vxjnnd0GI5AQxK zdf*YGG2OrX^o0N!-orOFy`L{N4{b4*JmhEjc}H}cnW^U6Pa9WW(q4MkdU503hZl-IGCHnZr>Gk%q9ZJmV zT-h1ab1g7y-SR7wCMBj#iJC9Gblc;AEPJt~hc0W11{<+>t$h4)irfCi3yT{cE;p!g zPyBH%Sfb4C;D*<;CCjcGzwta-VBEI%d}reQLWytk-4CZNHnHzKOcXOWT;KO)^)2y@J+mVYi*4K3$6A`gC!O;)*JDRR z=$S+rp=%o$IZxj>q~;ubL}}W_rqnY@+}u371=mYIjyS*-rtr|}+J>&uwBtr%8#+x? zdRE+8|Mzx^?W*_#YnR%?D zg<_v)9F+;XaYF6dh7R35M{AZUcdzPyeIvnRd%@hh^~^%su54YuVxF$iAJ#P8rw^<) zCj`gVa%WH2&}(`pY0kySDbu&VnlL5w+RW+EXV|m)k4QwP)vZbB%go>4&$;+Ug4tiO z&v~2A?VsIqlDjm}t|xeJ?v>rXV&4x(ilj$x?R~dqr}Da2;IW;&mH(Drk;qA!`}We? z#TP@%&Kup9^gp+!Pjrp!$HgD^9Q`J_HD_Ywj)htGldkC-e%tm+BKpAXGkT|fsmGMwXdi(BV{P7}RYI9)!;m?1K&77U%`CmlLWa_bUSby=ORm7R|(({`F z1NOSDzj)Zo^=#*d-m|Y?zW%_=b3e~?j>!Ehd*(SFUB19=tz&jWg6TYw`*(W$V;3ce zOnD@&c&j&If7!gfky5{wt^cUG!ZZDn)I*Uo>PNSmeL2gODl)%mR%zN?$-H(~&flBO zqn+nChBV(mwJUCvydqj^00;IprLvLd@G z8TX&f{i<~LW$w8tk4;xDln;nmlv1*dfpPcsYfUkS?t_$0Y^#%wZh9KANHjw@x+P4+ zm@)OXUG?jX7v5Z{Q{MPUD_BMyHJfveeYWw(inW^;$}>7;MV|6@I3ii%VNq^%XHMtw z?3d5JzIe^#b?hbIg3H2%U=Qz8OJ5%BIt*euc<@=>JOcQ=MKljD=O{x)nzE2zP1z62E{%5u2g0C(=l6n6! zE?6;bY54Dk#sa@ri;qvWv3O_v$^XH^<4m>pH4aah*x19Znd?2P{d*jj#H zFgO{}%#*i9#d$L41;@()k+K&qG3uI>I4Iv)-oB{aXbTgoB7gO!m}#z582p z`uX>F72ezbOD_6eX#dk);8$JO9&b4ZC!u<+#n!=&Q|3ueGFsHc6EyXr?g@@-T0K7& zym&EPaD8vaNBxe++(#VK+?WiM`0O9OaFY4kk<3&$Mew-MqZ@8A?`2-RmwRDvapAuG zh59awHO5z#_wYn7V^TOJ)*9f*Vl!EpDRzpxOUak-DRYzJ&OE=>vAy^Ev&%I$XOD9` z$2<(tIpxm%bI$R%nZfs`aljqk|<{4bD^^mjC zD~c>PxqK<;%#kyFXI3V8d;9MRb(bktUVrX=?T`4yM=Q)S7G1R3{=cJSGW!R`)@q|m zGr6o!1U;_Nvi=esytHuQM6r|`S*KGY(Psv;-R=*OXo0K- z-dG8j2OT=LN3mVe(D3?gYx9|1>Xx7H&Ad?&uM)FIAlv@F^u(@Yg^c_4E)`0LS|s9D zA7~XYKCT)PadF|M=moWsF;m;3+Lme^p5(|{wq-MCnOL#x`80C$U;yhzN1Tl)>~KBI-|uXS9S=gbhlKYyNDvA4Hzf4%RPFS7p)4(#siV6j`X;j-o4 zM)8!0Y-O8u2f6anT8_)haa~%t;WUdR)7^Mev!ly1*2zKz8l}#rX|qh*S~B(As-AN` ztY7kF=k>C&NUppUsCMc@x>m=*13cZFUXev%qLHh#Dp$OdHy7=8pS0=hr5KY%{bFs) z-#&D5-*7m2j-0)8S2fS&fA2Pb{8Q3hurd9siQ=u}`=9MgeebuWb74#Ud%5px4;H#> zJiNX8_}{Yh`rR}9zLZRh3z2-hPeo~u#Om}lOwwn3mPxwR7Kf+qFcu9fUYWnje;UtO z9_QLLjc?U`J!#LZlK+0pxxZzG#Myi?k>9r#Gi%1jpT8Gh5H6Iw^V)U&#ek&YQEsb@l(ww8?#6UVaF*QW1vWzGn1ApxbTj11%QV)%cldv=@!$HF z@*(WM{9kNLNOL*;@54|2sB`HlGgu_=t>Vddnyu$mrY~v>)Ew9b6#ri>fR^CNAKRu0_!#E-^#xGx{4>e?dWo*#ZuCXqqeSC*gC^c zUU$ySvZcp9Cx<)hNbNtfHza4T7u)%DL2oYnQ+}83awPMQZ|mR6W9r%yu5>ALilzuQ z9Vx3YJG?4iEMB2HqIUBKWcr%xnFNA4+Ju)li2{|65B{~i8oZd&te zR<_qwuPZ4#vW?C(^&FJa(pV+BS|j9wJl*^;;aSfWvm3%io`uVIYd=!FT(jhVYe>o(^I5ZXe4o|G zYKD88z59K7UU4j6b&kg6iQK!tRIN6CQTX$%)ne^8=jXiL(0y@V#X5bJUCN2^9v5cr zDJ=NH&LFN*&!@lkMz3h!JhzNJUqY9@U#q!saaDL|?|j*5i%+c$ET7-@D|YE}fmzem zNbOquZqJJ>W#*Vh+omHm50777I!Sj1XaAAdy4l~@F4q13J-J$b+wz)Yp7)+?C{~`V z^Z&biVZ4xTO~q0s?ln1#-4}A!emb)D05?;g>%6{SFO02%!?}41Hr?iN@;B3Qb3InR z#zEDT|IVH1U#~9s@6{miHqKX*?au3Jd!L@COm}Ukf4HC1J|zrCPY|@01=bUU4YZgV z;v8r(!aC4mX=(@^XR)*}S1>d%H$xd`0e3fp6u`p|;BgiwEaNO#hgd93A?M?RMkNpf z7pNmF$ZKBUBP>{LpyA+(fdSZe2sbJinSg?psKFJ`fC*?&1vHjoW?>quU~U`@9y$Rn zelrE1K?h9F~xT`E&x@p4k={&%57?!XIQEfcL{I-kC|`YylFdT9A~nf1>v z?aH=VYjdGI-KAg7ZRyg(N7wi%@K|X2ZjxCdRJVLb_snRg*Xv$xKl=F2%Zn2>u`H0c zSb20eFW>n~JQXTm&U%)3`4{ZC`gh`mx~<=&)33-$tkjxdAr!U0;P;TGI!?U6r@NRz7Y0H}fXCEPOuU zt*yXDtM*&R8IK4Y`abp8@18Xq9YwG4ho0|V$;*|Rmx3`?30^HnICRYnjm_bMDTaoo zCI-l(DQly$3!;yg{wr(j{{A>_Pdtl02a|$>fC2*(%MT&Vi9(r2jlM2$J+UywYmtlD zq?HpTZu`vcS)O|*A~$OL%w3_Eu5Sr8&wDg2H~o@!$)$>0R%@rEwAtTU#_tzD^UK0C zqn9D?l=W67eUEdPaCXbThj+any|FAmZ(03q&fU%24GbF|v%WC+f8=D%$2YHLGBfwK7;eb+lJncK_G>oqNT5>Q7jfO-t=lJmB#^eP87{%Wq;6Oa7nXh`0D+ zQvXmozV2`Ef8*`fbN@Kk&Xn}N%*18$cL^gGSG@gCp^_N~41QhYHE-~+-^bE*T+Y>W zLA%_;I#!o?KVL*8L^UprXZ$7U?6uD&u8rZr>K|GTi{74B+IRYx&W4LYFG6eOr}3YW zt!|1Go&74Jl2RI@ zr&g6V-f>@PyLDe}`X5{IM~6K>eQd9D-hASMj$io}W!Vq^{(bFpvj6yCckT~fnZ@D- zZ>smrK5BS3^^Erb9@$K71Ab>_trwH8saCu(PiC{%|uR@Qyo&hhi(E+@H~n3-ske;XMv3*O&7aOL7SBJ`{`Bh8?&x1E|7|u( z#P5tee>&P?=MJ^}hgXtvvNy@)HEVzUb)tdE_VME9HF?*b9{%(_Q)XX(@s7yz@A)(Y zHn~65jV-GC_~X$v?#mC?o_twlAvQUE+r1g_cdC-DWL95i7m>I%>r|G-Dl6&DTs$)w zS=FYVK6!Ml_@U^@UuTDOj6*6Pbh z`TVRZV$$Bx!~d9TTA^(FhQ90@M@rW`Gz#0uC|SFs!6Zc_z*&?v<#xeFX1&q_rY9nf z8Shw;m?ia|Z2`gX?S9r{Htx)v(4!P`zgS<}^lD!t&$=v6?m-Iz6=J@70#e8BB z1zf9Rj(7iwU3Svw{n;-u2aRJ6x%y6C9dV3xO;K%_N#E%m9jSAUi}vkoER8wDHt%tA z>J0B&66+sYJ^zvz^Wd+(^RYIy-&4#sdC&RJb$!~+2Ta!=w5ioHzt(wuz{>7S?y-AG zWtHoiCu=>Gl7Gr3&wSbanOOAW$2*!n XEf4X7s>zL*IA1~g0A#vaM&LnrMw4J8@ zb9ZgLc=y5QdsW>ZZ(b=$o>#grceS7N2E&bU^4E?{zxwO}+qy#iyE)10KiOs9F_x?S z%(O0HZ?62+bJBOt&q$m8gJXU6mFGLYt-ezwEh`p%@N6pi>Yj}!h-H$tN`ftCe_uP$z zk0#!*__*!{0~3Q@l&;&BPyfmij|(+38Zfn)UE-M0qm{8?g|BnEI`eurX4Vxq6AYa4 zFBqz^-B|PcqyWR>iJuL(OXus+!|F)k^yj|aJ2y}ET&t#bIL~Od z;tZCloi7aD>mSYuE3I{Db7ei-=kBsCWk#gwYsS5I*PJpamYj5%r_Sh=PE+92#CKB? zS#xwH&$-W%+T`4JIcMX61i=usypy~3^Q@lwdE>T;W(V>vAMwn(eCy`PL#wZzO6dq& zEy}jKk@pVM`qS55F32*R!7#h`^Tw?c`3_`WKeDspa^mYZDNnDa2{Rl_-F!1)i&SpR z>4x+7UfN_C&SKm>SN{9k+$9_yexh5dD=)u%@h0W@)x>n8nzD9gwb>`nZRKH?nR{-r z*ksWHE5@%EySNLywO;UiSmyoa&66jgC7ULEuyhqn+dVV$p+oqEo%?wLpX{` zm+RqpFQ+r-q(G$T%$^1f&kYu5m97W96SAJ`uA3(PplIc;Z4=`Tyua8c^3Zqv>dljn ztiFBp(7Mx4iwv(o-MX+x4o1s(HP6r1otX9Vk*QQ_ z;`IKq(*>?eGG<7YF1d15k|j&m7VTwsbn?U4tgU%JSj|Z8K0NVG0^`IIjp>O|bp>|!UoObv z{K~XcXn|dnB=fRWS`o9JpS@U=zMN_OQ=#3WbM>dTU!40|t58>eo?e;NiYO(`1yQV5 z+_i%bJS-GlGw<2i=B?-aMB}nUY_0G1F?DQT_|EhKf4OH`i9aQ1{WT9Pgob< z=$o=hRARfR#An`pp*K{*d396Sxewo)bl$P|9;cC?U!Tyyhtu8Ku3gWaxZ!@l6u&EB z4gp8DpY$|hlwFhYVRmTj|Ad2Yd#84*xT!sR?l`gXgV@q9zCQvFg^Tb!6`k^#tIdVA zF_)E5EiYZ`OnKRSse&`b1ri4)Gw3l^Y}@&J%1`NvW>_-kZRcAQGYq@=m>u)|*U1ajK^Og5e&e}?i z<^A6+zS#Sm30iaGc65(tno`Sgt*IL~6u9ZTwa=4SCnw$2RV6A{*K{yNv*f{n8xAT) z24xQ%R=hv-a^uF!;z28eKf7z)y*pL;^KuEB=!s8Hz4@alP{jj-rx7kK`9J#ByT}1uf zjO`(MQ%|1EukSwR|Gsu2&w2a5`A=5vjaRMT6>jzM)oS;h_Kk0X(*C_;)48=mCnw~Y z&F^gv#|1-VbS4y?kuHAGvy@Zz=3VW{y{|qcS01)hG%uePH+#1H-;MMC-Yi8aIo9+aR?5nDDJ0r<6aeFI2;GHwdVyZtt7Ymi$j>=Vs6Ozwg}J`?PYB z^|o!_^;KNu?*Eqkq`iCU`ue%|SW~~B`*Qhoa(ic#m38GA&4RAx*2I+|JwfY_FYJH4 zB&e`yTKlAr*Xv(Cm>Fd-M>Cx5=-utF>?>Bi-(FujFHOijOpRa9_scHUUlZ?oeZ12@ z@8_TN$UmDd`drw(VbPTbo#M%PVmGIJ(odMNh0UI8i;02bgAJWWTSTroHq3Z$Iu|GPeK0Uw{6}0`;=`=Kpu^{?&QXm|h8NU}Kyu@F8e&-D=PJwVvtL2V^-m=&y`fdpkh+ z!kWl4I%YpUdd(HL>soaFwc!#D>Av-c<}NJ!^Wx&;>i;G0)~{Q$K3-A!NI^zX(X0vT z<=RW8$Ly{soEvHJ@qzUhroY)&KAdD*bnBD2sIRRfo+EcK>V_IhKbH z^RGQPC}XqYaIC+l-92k`{ylushr-;9dzr|>i;+=w{m}0<&~F{=(H>A`}6&Ulc|V~ZJ?&l(&oJIaJ`9IWjAeRu54Uc z{5Y`wWyRVW3-NCr3rlq7_1{RFG4;mL;}(0lm4oKiov`hDnS13+n)U{!yW5_HKb`U2 z`uV=Szt#Spc-Q?!xu*S1?FX@~0i4Yi9(E^;+U`4xTHO$Q@!vLOLGsVK{Y?8;INV4( zXQO(sL91qkQm2;Iibn>kDuhB~oJ7wr;u5|sY2T++Cl$KSIPB={2@7}E&p!LS_Sjo9 zHd7DB#Arvw^$N;bGxE}vIHy&pBxI#4O;KDitFxe`Q_+X#VA4#bgVX=FR7eY6_#J$v zRe8rzDeqt*pALlvj$ZM6|21@MqMIMUn ziBEp?-p+hdAZcmSKF8{s_{787YmNsoH_uYz6}e^AyLs&if%l%%WLvDyU0{)&cVTLK zk;%PegJbF?H6N$g>c&%w&`k{~5iyme?bkM0wFnH9v)OKp*U8c9P z)i%=eSALVQ4SViexWa7G>q&?9&N~xv$@OigkDB5)p5>da?l_%ba;v$~qyO}Tb-Z#u zyLIkfZQT36!>z@BTV!vVztgv}oj(66ZXK;z{rzC{lieXsGP8MI5}vLT5v-gh(#S8U zBIw8}DC5TBuhit-@gUOWL(=4bUX%Z2P5#%a_V;whmQ71I_MKE`;{CBz_Q-i(ehVC*`Mae7vFtc?xugIV2+ur*?CM z`cMvu@6Vtif%d0d7O1|TU1C$z}>Kb zr?UlIr~XzuT($k@n`aN4`8=NlIP$emJUsctL!qJyv7Or{@4vOAexc`oqoTDp&pw-+Hp=f2OqPeo5yuX@6{ZByICC#zaqr^!#-@@InXn@QU?@QVor-H|>&@!U%Nt;KgN zwnwdc%WiXbK}!C-AO0eIwHu^fNL3asiTEC^RG zZF;_X-rLj(uauANFxmOQ?^;mntEMS4^?q_Q?-e^WiQ$s8{8_eD&DHyT{8H1p1iQ_} z6ofSXw;ottxh~V?-`(Q#Ldl2x8*SS2KHOilLZ3(M&eIC<+iUFiZg4oiex2UieZH(_ z@7ittzQ2`uWwX@fy0pf0^>S11mwQ5-*!JJP>D*YG?$utbb3Jm_rHATK4R@HfSuA*_ z-6;Fd%VFu9Y{nw)8;=^AKKL;g-cT%>pB(gSo9LIPGou-;`cfkc%G@vCm>4c_#(ct+ zG$qYd%UQo~<2rp?_VdAOPbQg!WIDg9IxBNULf>D{BKfIH_@VT;8MYVZewr!P#LOPF zv~@y7&WBeVOAo!f{bWJ0g7DN`#s~LaKY4ItWO%vShVNEC|Lu!fKjk#fH{H4i3-dic zBtN^jX4B>EB{qIndCo8I<^5fGZQs8mSKoeECY>6(zkJ7~o5E|q8oH?Nn3%Nu*}gqZ zC2@}O(ZbtBEACwnHa`EnOeEq;&AF!P4O>20ayM;I;N{NRkiaK)CGnt?&W)UT7f?$1 z4J*<`)SKR#TCJNa@aWaboqKaKmOl&obE})-jgT3qub(@=^m6g@?5*6zGtb;%4Unla@M2%l!mz0#)=}^N zgZO$wRi<0N1THL^D9cu?wA(PCa9=L#zCBj=9k0qc-f`ZL%KtUbq*oz}BZ<%Pir^8c zrW-B-`lc(l&-!wG)|cD9zhqe|>rPu7-!6LOb5z|6xA zpPl39pKm|Q_pAIp?{i_*NmG?i-aCBQwx~6I)8jYg-dA|rXPunXzWV4krfTox&Cl+p zg?g{^wmf-7I{rgSy0>xquATP(lRtj+dcq@XT`=|IySfizKZ~^zS>|WQ3YaQn@7}Ak ziD#{}QM@u&W!_t+cN@3(ZY^VU6q)cWeu|huJJTzT1YXvhMGezueTnw{mF@fMt=eBc zt~O@`b4hi=JPsAM+5q2sX5!L>g=-UShlT_$FK9!yg}AJi7R(gVJkG z(V)h7Nt-EUmXn2lx)?q)lzRJ4Z@&5qlNq)f6aG0I(mc9)WzyE@Pv4F(-CcC>k^jfX zDjwp0RewZo`*JR}^{(ctEw^p1eS5#6*!@S=!n5;(y*(pbUTy7o5^nRx?8YB`$hBi z6WdobFx;M-yCM5~&--g(y#~`-f+m#xyL@kT^lD9O-|g>2lY(L%ee;_pdF7PXiS1U` zFW*{eu=LQqzFl)BZLXew{*1_B=4SO7Wqb{Pltq6`J$=}k@#`_hnBId)2WCtmNZT)VEyY^h|m#zw&X*lT?m;cBkEYOg9J% zS-0p+<`xP@2GNie%q9llwa*HmnK*L;6VT%3SOrs~Xax&% z!&n7VQw38C(AsDPQ%mTIXmcY4b2HEaYz1=*sE~z$f`zdHXhzS%BwE43)Hqhb0<<35 z+yp$KXP{tgYz$s;ZER>73)%>8X$In=t(6Atr6)Y|WQ;o16g$bcyV*dX?fp+t%RK(0 z8C&EPoU*kmZt!1w%fF$pggKkX(DuiB@$@a+)d3Zg_M9=E$x>OV+564rM?v$d1E0fY zA3JDaGl!|J-A`J~ZO_G1`YrWWuTNvGtX#S5qI3MLvV_c2l^;0g%I%%~FDEFGzL#7j=94E_S(uoM^LzJ>M>Wn-*TLzE6$Y z#)GM$1^O#Hc zBHN%6hc0K8o$g*5;+OCBeS0K4QRP&p(2}_hx8yJ8eY98;c3#ct5?Au3(B#)zcT&oli z=BdE7W-{h^lXrDKGRd4{k$nA&iSM}?*Ul}T+W zGu!2UcsJ}~`1AJ76X)eGA0Ja{NM$&6^5o5zyX`CYb41wjNaWepzIt=#RqzX@4`!3+ z+ZosXx+-3_hvPy$!w0b$`ICRJ=*(zk6j^hGk;_?{MSx=u>+1Q>lK1QuaQJaAOU&@` zr`yYZH6J@&bL5%P$!Ly%>c=1X<;wW>I658DpYdSz?f>cKyZ^npectxtww3b?;>_kS zH@a?TiaYZtu!d(o+k->vLfLz+G`eaTUzB-wob}9Otib>|4d7Zi%huiW>-|%`Z_0#B~}y#M&T z#($gk@Eh$dp8fxRa#VKiap~UTvEOmOGY=4pB6lg6uoD&@*sowpUO{0DJc^jB=T-BxO~<> z+{?EmFXQ8aRA>1;JnW}3TAL13sjKgrc>ls9r@X@sD-LoqD^-e?rG9xh^Faq!=T@PB z8#6+i!YVnLZY@w?F<(>gJiduhw?mOZQs_?Q9+oDqA`c}7_V*4Gow)=n5|U-J3=T0^ zu4m%)J<^~BfPujHGY`WLObzy5;akrTPQ-js!K9)0E(A11DsXq>&=ePZSvcRAOoZ34>Q z4;@KOvy*ppW)Ka(AJo_vtEE$YJYFHC&PjE3|0?Bb4koC%TiCt;68spU5TilrS~hUeqD0$j6Cq~qhu;$ zd5^kyo4V5Lt1mk{&$5cLne{dH7iT(22HP#@|G;Q>XldsqFUQu6$pk$+zzuLC4QJyxODb5qrHY z#MyO*qD-uNg~g_Zv=;eF_5~H^75hyc&pEL@%27?)z|bZ(oehd6x_t^FRF zoO{fZRv_P(%*S=BK;&(}g-*L1Cv~f9uVa0`dIj(0=Ta*a3a;{=uXndsa?TOWGY@V2 zlB4o1n~Y~1k(|@sBs=4%;2kE`gxW@f21e-Kjx8ppe(NFcu3abT<)AlwKG*G|H(^x zZsJ$O8@x+(^Nb^bGS9#HyjJ#0;+kyHB)sRTj^9IxXCD;4Pu(hWxz?uFxMoSQ#i{rE z`MI8z+%11GPxkR+|Ke;(Hus&Z>7{RrEm&Ef7jiwfNLwR*`02rA6HRoN@0?zIhdr(T zkl^fxb#o2~>h10SA8ycOd}rD3o#pAX|5%@#c<17mDbnXx-ZXo5H}c;d>Ag2Bo|oR1 z*xa`!`Dzd2=gzyOe&5$S*R!P;N~!p!f!W8L@94GtioExp=e^}}WrNPkyYBKAr0o&!$U3w8qt2ZzcDH{8 zrTZau(r+D zM_|qigKnjsMF-E6-pEz@A@{89=%KUiN33qEL?k@C_<0qd@tH-9zcV+?JewwQ*3I}b z_wvIvldV(^ocb4$-YzP%#wT|&uULeR+q>ll6E{XA7)k4<%{W_NnIG0ZbvD0rw3k)x zhC|1$n7y2!5PMb8(;(7E%=mD4S=55H4*4ND4R^V2?Bx5syVOHS<>Zl#CKjrmO&Nxs zXUr>7UaFPeU~K0t*l=zWbD71DdoNbaH}g|iEogAcY&P$r&@G00KUA)1c$rpW!65!S z*~&)G;=!FA!W(W{HcCu$>pz>I0J<|AXWf-ro-z`QN4OeA5;^O~F_;ar^DFB|GL8SS~ygd*j6C z*P`|UUl<*)S&FsrGcNmxyzF3q=w7gx zuU|Z_U-Vw7Y}w6cQy)&dIO|h_^7-}QP7W#@E)t2sf-M4xS?k5(85rmD)-6+h$zw^6}Bf*Ye3dbm6*`M!0C3whbi!gVBR$L!yN3g9G(qRI9m@Fu+CAmXQ~q^X1sH! zFPt%`P~oW6^B@m}&Wz1AGOi|XT#oSFD4v)h@@|5*kc&pgI;Rf*_0rQtF05i$yQX-d zsApZnr8ntJoilnYE(dm~3w?08^U**lF~nN=hQ;)C;Zcp-OfDSjW46A1p-E@b*-tJ) zCtIYumqEK+znMf` zx?wtrvBrGyVy2u|EM^;>=L>G*U-9ZS)7~fD8qb_xd|;S2(P?8xMD&gcuF+z@%b#zU z`24zd?STMM2`H$FL)Um|L&b_ws2|iOWnnrdsgn=4K{f3fwj?%(T^G4 z?GwFS$jxT{!?NSxK9ygG9e%v~chE6q^~-;GA-iw(`L;R4Eq$Zn&*Bl}SM_9xkKVO< z$ENb@-d_r2zr33|_e(ZIYI^J!?GizzH5sj3&YrCsmmCUOwQ(sYOIgkHRgIet+uXbx z{;p?2m%GB$dnpT?{(TqN`}*|P_}Vn#jbZlj_01KXEB~qeeC73ZZii1tmT2DYMfkT-mWBq+)t#$LZ+aAL1Pqyh<+?t`*$J!|mu|?efL(h+oHyh7QRN z3x2`=wI}DtOt#;wR$u5_$Ex(E{XkCHy4x~LPrpuBednfj_^wA+o|j%QnK(JtdCT(M z9+%h8ySHcgo9=Sn<*)v`?D{@;|K7Eq&#}AfPsmqmf4An_`_(^Q-tgM`{T}!G**gsG zF&_GB=_Q`%y7bs&Ch2>Rf)XQ`tk0_od@|klB>Tnzzk3mP-)NOrJxTF@Z0z;;YyX7Z zuly%}E-74E_3$UZZ6)Kbw(~{2)>%16kLGXNm1Xdfd1lO;-CVnt9C{Iy7}vx#C8R@D zw8eg}lnTe5%}2I|Jy{Yn@p4g+++Lqa)Ag^{9}1cD?pOKCq`&qZ(bv1fE%}v9_m!k= z=hhQFmYVT+(UM-%s+o7s_C5X?dq^U-MY!DT{!tzC_0i`Ztq(2xzA#Zfd3xIBtxFcN z-Mo8zL)I0UD&NZyA46aLvNwnvnJ!Y#W7aj;)b?}_zibL;KePms9>`>A2<;y$&itgNW z?mxflE~7?=W!gdipFY*k*}vRg=;i)&>z^0+&ostpP34|(^4?S}$Fhq0eaZI=!oI(c z+{IzMslw0I`uq7!@8WlRyJpSY*!Da9NVb95A7dpJomaf}9Eo6t6smg?=z~m_*K@T-Ip7i)~GUD;=j$?nkH<WM;GylF|y zwj1x=KK10$)S&XMKHJuZcrK_8{x@%malGECiCK!7%MO0e&%7V#+hAXCp}9m*Y13Jm zCo##vkumd6_4FK?bZ-5;CGXRx*Khb_YPBtBak98xgHvSfT(*_$l|0Jl?@3R!-p-}5 zd1q0^>V~)_(o+*BIN#7(&{a3Vd7fgUebLR={7br3UMsu|nfR@Qf99q$m#290Sg>4F z{IRF-)XM-dD>uF;o2$1!J#*Z#swma@tdQrDlyId7 z1uUYL;a{XrZd{j}yD9Xg_SJZQ{@po>)mE?0<>bA7;eB$_NwE*?N`A@SdB<`LC0QIN z9gg$Ak=d1YJ7>X)b^H9bUvRCMaCm!2$%hM1FFQ}~;+xKu8p#<+f?uxGDtjqVw7 z&02;srWp(d%n$xVG1TSw{(Ca(&rH*w-PxbCbMBXWNSkh0q7<;IF_&57=H2C6x9P6T z(9ZmNXHu6;H_ttx>Q2#zY?862TVCBe}REqu#;;nnKV`AO2PH)y}Ay01Zi3ysR zB&V0(dw_+r&f|oGlc>-!g@+0%GR-aQ3J(`&fAU{*a(>+8`cmJ2xgHzjZ)F{^cRF-l zu~F`l8^7L4j%iz}Dm-0hJQRDdBlUV=_sbneV@z8fy1bc{!CGUpPt4Qm=>jRk)q9_C zZ_Ch}tD`FOEy8lH-m#o7k8F6~>3@@b(|z>UG4t>|e!*s=fUaX+%!Nz$OZ1BUiofe7 z{yIN>Q){&P?$yEDMWc#&R^NE_dP{3nf*V(K#EB2WL|yqq?AZFuiuP{yDx7^ zJaE$K(4TJYIm_2z|BN-8j#pj1V=_BK^3%t9W0YH%z*1xwSIk)wDPFoDcq)@Gor69XIiZ#~=3Iys_oT zp;=t#C&_uo=v;pux+BJP-|dvr_1>>lbTUgHPjbt?J&7y2^J{2gd{Cmk=_9{g%Xde; zPIu>5uT#O3_zQ#w8{D)sbllVwZHm&$CpU19xL zrsUk!gv$>%?KI)rG?i<2)l5s(+O3C*)`hHcyjD0Z*7Qe0)1G;`GhOTCSDUE$mF}x9 zUYZ+s?(LiG-O5TIp1i(eyeoCbKA*Nz)m6f8S(Dno)qSz9(N$g)UcLElMa`7;SGVl) z+#hv=>-Y&_$uE66(Mytb0&CLi=E+%<=e<7_edWZwGgmj}Z}GcjQ(roFi=@D}Abzfj z9c@i@;Rlb(&9D9Y@u=D*<1)!FXt+pR;-m2BxvoU1MOqwsd` zPABhM-xChszCTIb*4IW;ribIsrScC?#KicGR1$Itr#Y85dv>X133J*7${+yisDgD|L{dFhTe+b|aTkQ1OP-|^I z|D~SYz0X%2zn-IfygYm3EA4F&JC`msPTPFcDmvHp zS=iZ_q!T9Nq4LZ7V#Z2)LzAt_lW!WDnwacWo_*KwZ1HzJt3wx0KTOpvXa90X_ChmX zbclJ;j^NL?JFh>;`EgdJ;M13z?CNEQIyo&2nbpr7?mT|&iT}Ej^CKtk-=Vhu5MLmF z=Hl#|vzErmGu70w%=)}-{gEH~YoBMmu#~B~*EVmvs%`#uvsA^lTU$Ss@xSz+np_sW zW{Tg`M{lD3KeXY!ANS7sf_QJ9oV?C^-pkGx4>>p`-k8cJ)Y<4Xr6x1=O|Q?k$JeTN zS1&IYH{O~1t43kJ_f60A^j`US29IlA|GA#L+j;(7KDpax3rsB5&wR!ICFREBCq3mV zt0SvJCdp1**Ei)z`g8ru%&i+7^d(|9etWQ2LFfO(?JL=9XRQ4wAHVTzmW<&#m38m^ z)|}Q^lW|)v*YC$PQ_o6I*PTngzKM6^wf!~c&BH(3um4=Wwl(6~E1mFf%N{#1pUMf$ zh=|wZ>OCXbF_|s${~x8Z`SuYnEHQ~MH|U{GX7rQ;Lg!`o zZCS7M+Pg_^pUR7g+P3S@%G*fjeNnHK%{!T&%_+D1n1Ai7g1^-_i*7izPM*RR%~6pT zTRYcnMPT>zH>rKwl(+PPkC%@4HU5xm7W8TBELPWCp1Qs>dUD>3@Q)Wy zSTm>n-5;?-MfYp|wC`Vz-R9cfu%TQ#WP#g0-`sn~*SExf%6wdXL(cZCV!6%#jgeNr zkLw2{|J5|wx#;gDK@*qW4NBZ^w>Vw8&e#^z`~Ozplg*bFT|YHfZs8^A6&!mr4QH>r zI?t-`QL%pg&rh-C^?f^D{r~y%yzu$;dT}g~wb`ZOsr&7|D6(y{P5FIrwYnl(Mak`+ zUAZ~@8ncWd+igmFaAnDF|q21Cc6=n;f3Om z760T7UH9C;G5Tw$V4wgS{e{ipQ2hiO@E9}lB#sM~NgT>fumPW}4f6Tb1mow8BhljmEsJ4B5m@`%FHIB&^6apR{Pflv=0!We?JKRbG6|yRBt!HSfI} z^YYehx+!$+&fMRPj5!PDOi{ghK~OC|`ujCLhsY^!vv|#B7`$J%>F4Snvt$DV3!5&l z2t52Gign*s+X+p@=e(mrLPMvrtoc_yfAOnpE8kx`{%Xq-*5^&rxrL5*Y*JbDMIf-R z{3Yi^hZ46bNA*_UfBEpAj9bFY2`^?RPS;vAMY+=DFwaSulin(uOXYI*9^bg>^J0}r zFJ^}JM=E#DvurzUcsx#N=QD?&jE|pa++X~sASO7m@2R=o>$-OyclWYOKbqZ99ch-N zV>7>3`tG8reeo=hRxH4gn~^se;*Jqhb7Q2r977{hBji0^N254%WUg2KyLWr8ZQA;o zdv`v6qqo!hov+@`J(^xhD=%M5)m$QabD4`*(jx6$Ijn`*2R4Ojb|f5GA!HsnXHl>4 zc1|DVPUCxKg%eoZ*06CzZ^@ixp!wih?i(go>FW+zk*D8P8+%XQlAzttxR2E$|Gxd_ z`1im2E4>96_-vS%%ii4xezohrKVOZU!e<6Ki;{n9K0QePo6qn>k)gKTU)Hkj_1o}$ zd>r2yj(t-;t|#O7|91I*5soR0@B91C$<)7yx?ik#;A8TEsxL3zd^vMAexDudgg}PS zZ3;ElHgHVfXt=ic_<6f6g@=AvAK1V!q5HU>+^o7E>*ZuwSFk*AKfYYv%5}a#7lRn1 zhWfJp^Y8xOIexa`Yj8v1>y?w!&*|Cr>M^`G__gx#=H+pLA=l(0!A2aOBpRBPn`u~}i^YbU&pLG9H^B0BM z*MIHWBo`;xsLp8ld#G;m`H~A?SpNT6YME}A|1ol3=FeqM?boKSu@{j|YOLp5_2HQZ zuban*=O?@;Omvh{+>yvGaB?q;`KOHNt&bdEud4naG)H4GFBj8k^PLxZ@2nE#ycyaP z&@ArOQmiF&>c`>YgNC9?(W(nJ7_a5I^`zQhfhowX@Rs_K|lduS_X&-MK0CS(nj5 znYr_3ns58~`rXMZZ>FxCWYljvx%iaLyO};``u$p`FW(#-%*-n4cI^0xpD{Co)TZ>m zl(9Qz88cJsuGR{HPq(L3I#)e^vhL(h_hW}XFHR1A=;90Zxs#>W))qQ z>MP2dGV9#4JzT-t793hsc+=SOzehz~$*WVMf2DoZpY5-I>h;4b%PZPbHH4|wk#+y^ z3ufykU7S0|WR_RUO((61UOuhD!O5(m*Zyp?o%}fYp-WSy(!S$MC88e&-S(P(>$6tQ zouz&JCzqbHE?oJ#FGlO`y`@EQtw&;B?c3`e{QmKUr>z%nSNsq>d$|A3vBUF1<-YCP z-(IV2`{w-W4~6{fb}Ibr?~feTEr0Ux!dqLJbE`QF`Q4}4+DOei?!|Av>Da;_7h>gN z9~Mp16G)!hagj?RI`QIj5itho0-@su&Ak?#!8;l*&p0e7b2L&Tel}I~bDhSFW1MGRX#ce;wSHE>q+Tc{Ucj~a*D1$2$4vYlGl}FJDwH{rreV=} zIl!QOv&G?^vd0q7Jd}85(7SoY(>>QeX(t$`74jTk@Ib)tk%eEO?K6XJV-2T+Rdau> zt4;r5?l%8u=A7qqx&;&;$;mvmI`^S^rrPB{naiH*_!V#(?@~K%*z0U@>g&v_mVU`> zmn}MyYo7A>Ju*1=$T3el+H%e93a;gr%Zn|J%(rJ^O)tIq{lzrdXMfUnZac%?zJo2j z~4wq+`IC4BH5?cZoR~QM(x`b%e(vb>|;+a z;5u&6{_IuZzcVuK$6i_UKa}v(`|bYub7Y_KvYUx{4>RZfQn>z4+cg z`)d4W-{n8&Ht%9zR_Jy8w~hO;_J@JImGUyJwWlv$>;KeM+4_0rLC?F}s|y&@cN{yu zlQ+8L&b(uq|F%ndYbPgvUv=Zg-+4C@lsOWPr5P{$8Ta|$VW#e*JRihaFESaQx72Ps znPW6N!AzQQ@!1@+!x=`id>c5w-FI%gnsqVt#+o;;vv$qwUB=UWljmU0mI6t>-I3LA znA=4c-I$Uno><$@(|t3k?dYMD8@giLEZqlq%nv`^a%UlrpxNvdrweY~xNd5-fpPWJ zlLebro>Ve8tUDp5Tr@ zznPZS*8QO}C+GG11Pet^zt}eSCgF9aWma$atYym74kwpeBy61h*6i8i1^3Tw`^VSD zvOW5d1lylHkN5k+Us~;$qo*F@r*`;x$^+#@7 z|IID@*^6I(Q|_v;IrEss{L%9l*?aVs~P>ZI^4^;~b_G2YQ9+U4qU+qsoz!ohiOBiHoaxZPjZ(-TrThvBx{h3+^~=|r z_uXRO@b6clz5Drg{l6ymOSWOtX1!m! zRm^NRF0hFno5Zzn$)TuK8?UjjmDQ+Ub-0;0@69{?@;1+-$qG~NxqxP481{bGf3^Q_ zko?Bb|7-qHY3?ffSsDYIUYO*&nC^0oFvC!V?^6Wn-0 zj~Fycd{b;Y-0{#RsbY`g{%Z^DFE7}CVZr_$2^scl+&|cle2_|7z#ibtt#WRY>#Bn- zBJ*c2DN0)pZ4~@ zyH}7@KK)R%PWFXGZ)!V}t>BhP#aD)x_ZD1Ci;?p~wK zr=OQz5}7C*tGs3TZja#g^Q3MBemlC$ZSvQ8mAkw3-(VWV>A9In|Rk1Ow~?fzYAC)FisFS>2gGTW4L`^!5m zi(eW0CLIm_81kfRj?ogH$Nra|>)E)^|8sTvi}0CpPuA}0-4knmQMy$ACx5K^uhe&| zb61xdy|$XE`b$dz5AC;p%r%)99ays_gaoOTNx8zy?PyZT}^?OchEy|uCJ~iZry~~r#m{{(BhdNguJn#s3I3a7#kIW2>@~Wz^1sjyY zrxpGaNsH1 z-<52CZ?gS8%Jx^3FM`M-(Bo~8YMVGy4rj$PIxJ)ILmrp+}g0!U*E^?ywc!o$)jMwionTh5xu!ylLE z9A)-P516>wb-x2&d2Yt^T(A0-|I)HTHf^^E5-hJ;V0(Z56xYeN5BOVRowCc$C7DX{ zvIX*0+A2=hoRqn}dC{&fYO(J58=i^UhfLJp=)$x3o92g>uo7a5fBsCUL`P5+jgo% zWZ$GEC+AE$`T1OgW2bnsU-}w}a4y62zZ<+&dwlQFuIA*GUv7LkFtPjG>1X%67D;Toz4+*px=o9-z0(bLzddo`z!#TiHzpR< zuQ3tmFsT$?U60=N|yuzTRS*A zZm7OQE#SzXai^JEt4TFiE-FAUDTAZHBxTw~ zfvXnHY ze4Ua1_^N|zXi#9BPWiUAH;TjLnTt+sPrI&dw_kIrPkN9cvp1*88|Daw}SSjFW6%4;Q2nF%HdCs;`TcCs#nRipVXCq-t*Y9Tv_7OckPK@ z)dgLDURNm2y<4(i_49wevNQWO-aGs8+n-%!{*^WAr};x0+FtVh*tq)G{nJ|{RvlcI zzd!a;QkPyv_I+p8*Uon=l}eE{-xSi!uXf_}T%Yo(SDnlTQ%oYHs&A_| zWWIghIeDI_h2|S>=^e|m3pPcbmdd}BIM3ARhxA&v@+l_lSrJeFI?kAX%V_CMM=ooH z84Mfv49t7vdsj&7J1G`F5-M zgREO;c?v$NEtl8JY2*$rNaWLtY2Zd_zdU#}H!Q(&`=*qa&JD-i6H?2cY8(FPJ}oQy z#gDtHoYnSnS5^5|v%~k+?S7@$e6r|DV*j+DlR-VPukH7HIqyF8xB9p8#Kw~kKb;e> z70~24D*9~qr2redkd6uK@6M{wo_jLCcERrQvTDP*cNZLIuAH*z@~d-F_6*6Veop&% zzVQ4#gRL*hyw=Q#e2|kA7kpLzh3DqCPddugR!O?MOqu&|T}{xFi$ANfRtOkf=x!2u z%VEcG=#^zDOmcKeX6p;pj-xI4fs3bzhfMLlHMhBY#X|S~ zlX89gC-m)46JLHd^6mSptA&bFmt|yR?O$}9`-q3|n)PpwUHv#Tq*FDr*v-P^G%KYoS5j1N!mF1mkD%Eqfz zzvkh;U;f5VCrp*{3y{iD-7)`R@=IT-WlhucUH5($Tl)RgDWQGqm(<2j)%Sg)(qDds zC&9=^#+uVowc0mk_LMz$tS)@G)e`wXf0~|W_Fwj=vU@jvJCr8ytysh~pi!Q2jZOcG z%>TVBKYrivxqnLgozHjD`z}AO|6?Qi%7609XmiJDd3w2v-`LyMb~No(+FbrfP5#*7 z#h(RA|x~6%*?`-#d*rEUW>+?U+W&M{v{+fQ`VATy(_DgI@ zOP;S-_^bX0CdWqG-j{bs8;ato%&U$2YWZ(I&5=yFH>Ff!HBY+9cEfN zE%~NpeVv&FxxgiVqUSHU4lXJ2k!f^UrL&7h4bT@&C4F_&c69>X2$z-wTC&( z3Ql%4WxtrI5pVVVR__7UX|hthW-~U(tV{j<{6~z;2acZ}nOBx1md$0lf0HpnZ|M}) z^RIWU<7Nx{|9f8G)wSXAbKT2hKE#~z5Yu+`_c$r2In5<~k#LaLxgS0vim6X*<`zD? z_Sf*Wpzf{<8z%iac16>%D_BTdWU?-YrSYlBHviQeKDH!zv3qF7hE3YBbhSp>f@^(@jn; zb#a|`z05`T`lP*HDYvFP^6t|xzGSsUXFc15RPCU$+)F`CJA`(oEKEASs#L~p>sG%V zMSF}6_>$7(wueJBZQiZv@2kFDxnC!( zP|x>ZXYzEtw!ObUMl!r+*rR@KuFd9;@7|tgzSYn4s&Ddg{rNw99(-Tq@Z+cWI9)$x z&wuR~OdXRSK0l{FcXt~1bcRn{4eE0ngm-V?aM*L7UHIw&pZ5+dvp-KS;ANZBBqg$@ zg@LP-Jwzekqu8DQ={axOMVMA(eK+QKRsZ_^GV}N=wR8J#MD05g%~9~F+h5iuN4lcD zW7Yno1KE$yf0W+$``GILU*0b>zg-$1KAllQL@S`#*g4G6@!`1%_uILgUpTV8EnwhC z`|x`~MN;);tt$b_6IM31sMpurv<_@G3}!aUaqy~}zlVvpA^(9i1E=Zw*ox_ABM+x} zFRT;1oxZWU>U?m|>cn=N@KE0W3mXG@_Z~Lz`+83PWihAz!|NZiQ_L;>@2S4Ol=w^c zs_E7JLI0)Cx<}dHvcI%%_3tjbwf}eg%RYO4_pfr^y2FNtm-Egy+f;OSis@N1d-+@c z4C{{wZTd6EH2ZJMSK)1%xAkxBOZz?R?YsoRJdXTM`OU}Q82*m?{aw~t=DN&!%X?*P zbHnGD|DXS}X67B|l`GrcwP!u=kLi&9eWLy1(RDGaB9~-}v1f|KJuRP*rg%|D$fR3u zmMi;NrC;BBmvbrm#_#&CS1VfmLSETdV8?b+{MDrEen|D($Pzmsb8 z-~ShSRI*~DR@AS3Uk>NqE|5nD%cw$IqLs zQak@Yyv1@WETQ4mhs|?MC*)Yz?6fKDwz(e3eD<2drHSXHLT-Mzq8nP1^?uJ5fjY}S zpT2yYemMAHuzTPApyNMF-|yM-<&%?L4fpS7(+_XoP;Xb1ouYVQjh?Q~m(pts(jKq) zX`!#*vi#8dK(`8h`G-$i|G1^|&U&-%gKlV2T3y+{Yu&<)(aOTyjUFmCm9JMDf8M^& zO0Yvmr{vm#+#soGzMFhD>qR%4-HmLHlS{q-XRSE%b~mxOy2zid!V{z8YkxjHaQafy z9VOQ+7B6SsKXRcpo7~*eEKHPKnbtUVc}#3|NJ(6@na#QE`o5nhI~`OGX&C+!@HV`7 zC@5Il`|M3ApE*mXxgQC7-G3c9QW#kI ztu8f9{l^dQIZ?-&ljk&XpSrC7NaxUwp3N1FpL^~e{qbqvw24oa&w0Fg&xz*;UT_xo z>3*)9b1blL-jn}rhK--+Ec^cHW^BgH`=94bwvJkIXF{-cZcdr_jyuxj8w>Tcw-}za zy7eSfoBzg~i1p%Aex`~_#vHI*w{-gKwdob}Q`bq}es9xu+2Zz-{xi?JW6n#Ys{i0Q zul?e;#n*{j!f zIeG6ra@?|SGgtoihemn@UFSbo`e;{G%RT?Gzjy1EU1s0%qNS_rGjbb#@7|wn^y)&# z-&w9p!yH#%by|Kkv(#$m*%g;xf8DdL_1cPdzN)>`{)QB6sE!op(lyQ4YPDjk*4nW3 zSF_~W-yf07HM8w0>*cbV+gDrqi|uVy^Fqhn=Q0}PPKA8&QlIOy+mChasUuaZOS9%) zJ5cdqM#gr7TUYrRvU;Ny#@=6YcZs(F~#V_kypBE+q8QAEDwCS#QX*Cnr^PDsja1# zUzk?y^}Ijh&tl!4 zxFRuxd(HN1OH3u#cBR?AI&|aNrrA$jZ%1vZ(mj)A{Cg9(e$&;o&5;KrbGO}2XuWz$ zarLE>DQTZgW}6yhzF9N-+!Ev2EWyqN{o4$-RouEEyMfQ5~&pvx> zpV?~PyyDxke(fCRGG<@rw3aFJ+ZOvWML4I4x5<}#^|f4K76~1;*;g~SM#-_i?a9rx zyL~uQoTU+DXaC`sMMYM*0yTOT3w(}VTo5bba?O={_BHJer5>~9gyP#dVK!}tETr;; zpSa$R;z?tEQ)d19*4ssQo|W&kk(<`m#!-ENS^SQko#^ySo+TZ8f_%IA7Ogv*SAIZ6 z;%!s7NKSuq*|CcrvwsNO(lvU^=eGOM^@h`Lb1&U{*2*n6eeu+?-Iw2OG0!ZwNEGTV zyQtp9BMc zUd1CtmJ4D}QrM^X<;_3-yzzQ+!|e$lDrz_E(vb_l@wUj0@0R54B3ZdIxh;Dm-iZD9 zTlXzhJnqrLZypvL=Q8TuSrl8Iu56j&x359}f%Ad4kXHzLxvXk0xu_m03*VKf$(q>x~$*7AM6mtxgLAX5X1MHR+^F%he{1r;{fLOkb{~ zIrr9~+=pf@PH8M>e+E=bd|M#FwWNvh|KkO73%9JlvTduWS=R$W0ls!Y&i2C(JMK*L zUvf#)FP-CIkQ8UhiC|BSxwjU-y;$aQSgb{7X@Lrt`)u1g(|EVC%Co=iS^T(m;@Y=M z17+&iAAPj=eN4D;=h{gxb)MV|RC!zx&nWrCXwI9SOPcotUw0@9Zg<@D^{B9HfqU%7 z49~8%j}lzgEjLYP-7vb?`zWheaju{8)HK&TX^m-GsbbHRI=^JFx=!hu^JdW{&8crs zlqpOWn<(ObX{Jt1VZK$MuUZl)scfsY4*s+#S9jh9_MI_(wZ(HScjoR1l6~>mJx=mT z8Q+4YtfdF)!lwAmJ$f!loy*eN=$q_LC&7bL?5E3}TVf4L-iyzlc{20qjkrIC_FhYc zyq4;fJxr}isL0te{VJEn*`7p(!++mw5oh*Vp5(uAc}LO7&ZgpJA@3Zk*R7p0TkFXt zkRvWWX5T&I>5aIilL}SU)i$2kDtH%a+~^(k>lIldb#++kuxstJhibtdijSIN*C)TEL zXwBGg>b_b)Dwk`!cF;Q|K_e66!}It*JYOvk@${wqx0&Xh0h zi(mcN{DwDg=^x%_>M~MSKTK!2my%->;vBg`absL_X>CZeeXD4BBCD;5R0Ct!j|5k) zu#m)~CTk)ZL{D#cWVB{S)baJ#Kh7?#S=x8MO>~-bL@V>#nmE2h(YY5sRxJ^CpIbT96v;pX{Q%fk*m zJaum7Uj++}36H;&N+_;v+OS#vn_1g|Eq&fKQ=f8FF80$9QC^f1>bY)`M_>OdAzR^% zQnh;*Tc^yr5z=gMeCi(8cy~7W^$U&npNc;{;dpZJTE$&)&7RLzaF!=#6oFYuU=<>YCQDMzD4>hs+FB zS@^Wy-$-w|kn@%+qKkUYTCMgH*K}_Sv2|WmRNm(msD1XFVDuIl{<5QY=bh_3_rt~Y z(uDfmKRI(ZrRxYuYHjgk-n{eP&wmO#h2`F5ZFm2i`|+NW-HUfp(GvGWHB%@1g|FLX ze1Ge15sx1G!`EsA7HIvl3fYiUHdTF%gk@IhRLN^kc6sjP%2}KJ$yCDf^%t(!H$E9S zDc@Rcy5f3Dabldd$m2e}J;G-&3V>etmJX-S{8m!v=2xnIpczpm}i@3WWg zNt>57I_&z-cXNUEi3g#PMtK!KKIAQA+ocwN^TDeI(DD4uCPx;AE_vwKa&uy62gfJ&Zl~2%Ym^qQ z>1a8ScU@8O*6LOzP4|#_ikBux$IAtIguaqmwCT>4T`BxOuRU9Sd|B<{ml5^19{YZ6 zIrMZ%`?-lXF689BxDZgClTmMH_I4rXuRrU~hF@8@!~C-T()P=G?Nj(g%BC}&+_>}H z!R74EcJ_bSCPzQ|XC2}m-`*luQ!eV58p^`Dtt0r9(z9u59Og|P3LRWcGn@q21&(mC z6mmMotbBa^)sJ4Qe^Ija%m>?+az659kqnk>58lxo(tKF+$Am}owSs22adCSjN@~Wq zae>oB6(mh~eB7EKkfpKKcvsenb;+WZW!q+~R8%)O+O5%b{F+>dd3o?o5y=cQc!iu7$)whJseA|=0i zN}HX)>LV`hYjiwq3zsVgE|%P(WQ4}Y0sqSt1`w+24kH}R&5^Sr%@%O3HDJoxpnD8oyssuQ&vidA+VAgoR!nLD?Y2B5N*`vjq zeHpBk=}D8WJ=Xie>-qX&vz+FOz*f1-7m^vPbe#K7UwE7)yFc!D{ z;Cc7rwtY78oB7KxUYFR+Z}&y9)h^^>d*+u!#@YxsdA-H_;NXX5$rq1r3CE?+dg~Uk zB|KbFbF0*`nQFYs2O^Di)+jOau1{|Xoe}V?R5)HNb;gu)MZNE{w_JHzX1Vp#x@X@a zPEKo6o#ebEvFfR{bkwvTQ;a8lF75B1`e)~g;*FOq4TR+9>WlL89B+5CI(_fby?s~K z?R&98{pw{ot@4VPlSLIdeNjxZj@kZ8jS_g%4z)(^etn@Jb=xEUZ)ZySZQqvIg?98! zyR~=c$Fu$0%8$ydi`b}pMYP;^^H;gO|M_k%S-)6T^^c^nz0&-MsD#<^5jX#2eK>8z z^H{ICKHyoogk8(e#DuBxFIzVr=ubHOr{F7cI3+IA@Hu>`p!w*;%#aDkL@rr0nD{C5 zdV4(flHDJ6Jpana>#u%DdzhQ(^qNn6X2-R7vA)nn!AO9^{*CV2H9_ga$Cv_+>_ZGv|3FK=%9pH4+N`&O4EXxs|B8~<|N_n@!y zY?rEcT%EbX<(7Bhblo7oBU**WpU;b|*}Lt_<=wXzHy1B`_fv7@9nbn_%6p%9RGU8j zyi_}PVq?vP`!|o*?%e(|{_^_-^LeWBv96+=ei@+=N0of`?9(tfZGCUco|c)B>ylQb zon3ilN6fK8@yEP7dk(i3E-~oSI%e2gsFNUbTH<(W4@zozT$3wmpx1v~C8Tw>a9HX@ z>16>&8AA%0K&PFCzxuJ&s%~-L{;=cs9kwbi)Vk*~RWRu=r_LiGu|n4F9c`|64k*PP zR=RgUsifUa_rf*Ca#@cnjb5@d4k__nvCeq9Q0$|c*v<{}lbhbTwS8V39pTF2yhlzj zY@;yK`u8GX3y)6Tv+k))wNTVeHcnHSm~6UJJ_@rp=nTc2S*56R(^LX0a92mYwrmw#u#N^rZZ+=g; z+wClN{#N$>Tk9P@U(3eW`l0N(nu_NicJ557dw+A{r-^sx{_9(Ge06nP#^;DVFYkFQ zdf@wASeuit{Mha^zY_Ofm-bDu7xH^w_x{C^{|ad%@p~Ifc58mHXNkXZ@86M(W$Lms<-8j2rN0;4@AUqIhbrIAzlm}Zx6>C? zs0iGcx43bZv4(>epQz?n7aP^0-$u3fpPRjxo2j_d_Vn!kttuxc&))A_SG)L=di|l} zcY@9J^ee3zDvCh?{GQN)OS|u~u__!}5N?OIV@ge8v z3x#h}zyIEm6a3TtY50xnGuLM{&Fw$E=go^K-m-;lGQlS_&eRA6?vtPJtnNUOFYmpb zkxl;h1W#>RmZ7TTHYIk!nQe_OM$0FK_V+Y9gY)E_ta+4}y~o|`UPy}YDFSN?J-dLyY_`fhdDyaLHysZE7X^L-lUm>4iN zo%y@{<^Bct9c9wP1cOurCVl@YaEC*&Q9z3$kxNm=fkR#3kZMZ-q>c6JM|SBS-WI8B zn-+6Hrqsf@{*g6y`+lgTek@$?b#>Q-h$mNK4DzfBBku2gUh)3k{28a0eXA?De>Hxa zTv+R=%ReWk|7t$?Z@GKfT*d!;t2XFgseklpo~^#@`nca7W##_WiQmp7ykB|qP~NtD z!@s{&_8w-wB)?wqz+tv%<40T@eq`)ryZdYLMFY9n+`c<+iB#T>n4_m<=AgUMCAE2a z+~2DkkFL_6pcl?{=$JLP{nq`Dx#!PZ`0_{J(uo_LoekUD>Jk=TD*cfzX8&LM%zS54Nt{${vLtiwlOvNnE4hB8 zt<1lse*k z>Mf4Ft$E8!=mg`c!rs=6x+aEICAVJ~<}GiU;^H~w#N_TNdkV59=IuM8DRJkA)!xk? zMXl~#_`diH-`n@$Xx${dk6*M&%L`f zyX>>Q{J+bW)6dUS6YngmNl9s%z$`5)r2M>3-v0Is{@b;OPCS_YWJgR;g_dTaW%tqB z*TO3=ZR@@LV2RCcm3K-vXT%EGsZ4Yd3RN-eRS}zT#Awov9+h?1KE5vflO`L#^6~bG zDXSx=^_uR#8|6OrMOS2reMm>1=L&5XTg9o%rv$4OYaK1{)>?UHdH&LBzW-CFe4Bk$ z%B%F)_ZvA?UZHCKn`*cGKh$|f`o6S0M+u+7%t?#9PyPNowdPRD`SS2T|D23gEd924 zF?)Z0-OGG+t&>T!C%%mO^Y8W-iNAh#*Z-2(eEV(J-L&6xCSMO``hE9|hGn7nIX3q> z9}b>;zvfV9ro~1E^=S`}q*@ei?TcUk*!=YmFWLRP>eI`w@3P7IZ?*mPMzP)ZJ{#;_ zmwW8!rLY%kLUn9@uJ1P4`uOI&M+;x(6qUTcbMIoxj%t%T8+U8zK1;uE9bjV?9CV9$ zZtX&KzRCOF&VT*&x0iot*J922hwn}6R^NT=%(*Wa&tE(|IAOU-w66bCCclgGyF7j_ zs1&+W(8zbm`hwz4-Tqq}lC4_vDl=DI6N#Lryly-H{;eOa{vJ3c`ipVi#HFWhU2C2l zscXJnYntEGwcUF9**#YKPi?-@lHJzmp{69b(Irmd(aZYpYvq;C9aeG6%c-09=6=kb znDi>S-T(Pag9KT7CZalD(9R>gy}R~`svU07pSY^deb9k=6RH%nEZ zqyDlB-OMkJa(;OvRJAbu_lNe2-Sge`%O_N`iY2amx7NUS=W(}aN5_>{17cP$oRMxA z%PV|#PSu>H7am_QFMc(}G{AECIq5HJ9M4P5U|rtv*;GP2EaKagT_5740}y^})h*S&HaUHQ#0()fK66-=`(t^U2RD6*~R2WrcT&bcFw)8 z6E*~U+sy6bzBn&sTll8zQmgLs(ml7fuFOi*Nj2a0y)3WJUgXN3#+yrwEgmO^&yIaR zF{o^r;N$Q27tZ~{x97yBo)o?Pr|wr}s4ScD^z*h9m13sTM^5MV_N)(GaZr7Ijr_Yq z*CT$~{Y$);Cu-+&xSjRK1|E66!|kg}zkoAt_^TheKYpA_SN6^}^{7lO=hKw-?npPc znzBSb-uCIGZ#+v>>?F+l&j)5}O)vkV|7CmYFL5@lZauGGtr7KR)>F20cQ*QD8N5t5 zR&eQZ-u30n`@emWxjWT2ts}G8VB=i7fe=UH@;ezIOcE z_poo@42>i0MNOf=J1=Zn$J!pwO*<+Ipe4 zXWR~{boOjh7x|Zx{kvZ-Kk@aLrqJr;*RwpHM;_AV5|58xV|u!V@mbl82_ZG{(`qkj z+??*TF+JL3&SI%&mvho%PJUHXG@Fna%JPZp`t&)QL*vclFEraPEjQc$KC}FVaZL2Q z6KY1mT+c&7S_C<5&OWNHoL|DV`cK2&gO1Asm?I~jmpzkhI^(8gr|->&IW_C2pZEH6 zN4Cst?kb7b2cF1J z%;g#PRy=*8Q`C4ISOvXhucRc1j^?0hzZ+Kr}fyx@DJ70A**XbAgx*Roq zSlH}cr7(TdUxCGwY7`os%+fp3^9%pXjLVO{-?(obtGTFfZDzQhq(bqQ8sVe5MTOgipcyI zdHmClFWNXawDTv=oBS*L-QOQL@3-_%*EEBJ(W18x<*bh1c=+Gi$)-2E)9o58&#zgu zuX5hY+s{sGcgBZJE_G4wR4(pwF^zmPSNO89s{cmgv?X^Wdp!R}=I1AVzMN8DwMjdp z#C4Ag%hUSTzdYx;UR3;lcvHGH^TX%Q{+wC!Rj2c*l{!YLuPllzHHh(e}*=1}0r2nM)iyr66#WuTkF7|MKx=88LkD%;+DV^7gwPeyR-dY$aTDLBc z(_CwNg==5sp4WfG*7wNdg&gWVr5m?Wal^SJ9&VO(*WYN$W}B8gnfZ!I>)xqPY&XBA z)dV!_->-5|Vz1~k-62f6*k4@&n zR-bg3!=2eT@6I;0kbk-E=;C=_rFu5KPd*YWA-%6(&`x4vlVGTXA+NGn!V#mM9X!g& zHF)Bb%~obVX8mqh@+NK41&f23mQ{Dl>@t?he@gwWJn`R}^@XuK{WjJXB^fF*a+W40 zKi+uExbUZB`u+Q+bI<#W|NZxGIUl_k(ep%4R|3UE+o<#rM zAGSZBbBWvs*3+~8+r|fM4pcn$_S}Oj&3X)%|2lpP|96FPwaoj8y4f4UnwL6mbiA0? z{Ai<5)G1E4dk^Lq@7`>(^TxTaKlOLdTf-v$^1ug4v$EfR%TH`y(P3A!?xWN>twUYz zf?^R{F0PxyBleW1Sg<^jL#cQ-hkTdENBes1Bq>*&1s$m(g_5qwBTb7>?97dtzIW1> zD&BRaH|6*@*lt@?p(0v0>1qDF6jPgE%L$LvqkIKbidV7Sw~gk@{(N;-x6ZtqlRfQB z=j5E`Z*yjDElam}S+I0bp4IZMu=yw4)T@`Qi~YQN*Pf4SF30;+)$cW!x9PT!^UsR# zBP%aQ%H-xPKflb>LV4exf;(zndrz*rS9Vdfx8gSI+KZE)XlhO{`?dOmLI0bBbG!Gx zTpDz5W72-x*PGsOY!~bLlhbuO`SJ4av({dBi>NrmD}DcI%$rb|^p01@AW7FLJtycXx=}?%tkvkHo(Gvwl(?!mR(9L2e%B)RmLN zCAbAO_dpLD{XH1#q#!lLRa0Kqx!$-F}uPW z)0$suR}Zv(x<6ZV>hX#3g_E6UG)%)PWT_l~x%#9_^0jZDXs@61^RWJ$ zA1C?eDIR)u`~EMnk9)pa?>~HxX;=J{NkxyjPR}th5?4Cy-Jv^^TRgj9tBJ;X_N>Tf zO;7DzFUF<5f1)ADZfs+q%xPSeaExz8p+MT47RmR`J0}(zJeI2e+J02@6G!W!HrZsA zNvU-`Ha;OXwLU)E%ru3!w60lZwm9C%<9^LM@fXL%)rId0&W_sbzW8$`U(hLuW1l}T z>~uCYP-wVj^W$6j*{Ya{MJ@XuJXDLnQ`5_nW9Of=c=7HN-L<=){Qvk|x9ENQJtL9W ziALG;W-@-TXRJ!HUCfJpsUq@$*080D)V_@vd_n{AN<}v;D;3GTjo1vd)DE^z3%*7X zs}0VcF2R+>C8@cdd1?7JHv0ZWDXB%7dFdLSDXDoSnI)B)`a!AbnZ+eVl^TvI`AMmo z`oX0I1v#m?sd*&|26lG(&YmtI`R<-BzKI3;FirX{;h-CUF%QZB1#OT5(FbK<9(=9< zUaV+r8LePpU;sWY!_2}|!Q9A50e%R&324osp(Xq@bg(Q$t$_kaHE3a@skuq4f|*G) z$cR`4bI@5D#^#VSF_35j1+Y2`gJ|$6>ELCFV3vg`p|jGXK_{gvn44NCm>ODu*(MeS zu?mJ3Agc@v&@M+dLq4bl&*DZCLzKmhh{HB=_$JCbWT>=EnV?ko!g=XWh9t4B5FVp* zjr-T%XbpMA7;SuS=H1lrl%q>G%zggoFz+&!^Rv%g<6HLd9nS~pEpt=7KJPNFbFG-Q zPip3gE^kYJxnGCOn7xJ09aZ0Q_sQKe^ZmXSG|LLVU&nU+<}~}*%|*vuD&>#Mzc^T> z7dh*bl$6@0>sd9cs~#|={$9lP+S>m8`pBQ+A5HrYm?pViow2Yja}CdqQ`-&*X)g7t zobHq^R7x&AqODM0dQZ#I$R+6b?7m>m2=Pj`Cz6xWH_V#oy=3LH=ej97 z`MOR_U#qKALnTowgFf&3rM+bNGm>C-y==-GRrI%zFDi|7?qh4jbHmbTHCcN_h+~V)= z?oKazzwds2X87(~)3@BZ8a*x7vret-&5a07rHQ&H#Wp5C4o|u@MP!b|>|=)y_;|?h z$YwCVJkT?vv0$$+C-YpMvV|Rf-pm~T8q&D({%5>43ii+`J-@SjY3%VOFMZyfdt>{( z=6PNJ{h#N)*FJyK#E@}#vcQJw>-Un*JBKfu&wN6up`CxOt!(w1;CZtdDrFr0|LHt@ zJn#R%U*QZN>m06E{&=yo@^5+dIi`C;4}Shu-h5o{|FQV}Rg4+G5Te;7XovX~1fh$&>MSx=u`|9~mZSK_xI{bK;C6!qCQ}^$w3a{iXl4{Pq71|CwJll{o%l`>!KeCLL$M zClznLFRy6UJmbH&?kuk5(=I5TC{;4y;Y+3Uhx$wwaU5XeZV0%(x|M9 zLL&m5*poi2B`DgWr({ z?aXNwyY{il-V$sRFPrh;#i8vE0d8W26^T!oH8y!DaH#9#IPfZ!Y$(@sa2DF-@K5@i z>G!X9uCBhWH>*0`>SKb-ZvO1YcTL*g`o2Hy`}bw}vt@ffU%h(&^KWU7i{=I&^X^qj z_tyv9U)=s`;_Lsv-)SU$5m;3sqWRWx&BsUQTaq>yPTSh*EwSYL!c#Ba*xvguvg%5| zL8U~MTlPc!wfj~w?3Cr|+09?mJK=ONS7fc1r27x49s#*u4snbpk0uFoKkk0=a1Yb0 zV24c~_b@4n&yEqXQu{9B;dfN;s9x^uy&|)FWM|Hdy;Az&_dMp4JH3OC-cb4`7tx{n z=pXaPqZ?N~h$_y@$}(8*aQw#$i}~Kc8&BR~H|3LmBBSRkn*Ut1W2fv!SH-Qr57e@Y z&6nZV>+AJref;spkw@Fh?_4v=IV0nz9*SjZhUCo942%38^7fg-&C7({Jt-GsByQl z%o$A%gBIs(j(%40i#;DEmos#Im-$p6{6r^4;gFW#T)_)l4=>oN$-`jw(4sF%Y<8iH zdI6vM#)iWdO_w_s3b#*i^@2&oR?W_T z{vPglxJB)E-Pui*=j!{iPa9YqF|}CgY2W*}IqUX; zFP())#CFKK7fuM~|NLI!SmC7CUpThE3z=NoU!=0#HixmKC~p+?{Ohl?N8%=#^PIcHvU zMbKiqV_WSHU9`U2azjS$d6nT6|Br?0KTlkiVluW^__B6d^|9*46|XA)f7f~wADzpr zVY>D7obx(;**D%DwVd-vO8&7~&gDPfRQx`?KNgi0CtV$N?by8SYkA9lGMt%pw#ww2 z>4ofa*^&#d83iZRtqn3fkdUF-%QIn?{MSixRUSFY3yr-evCNHAFI+awUd4G511H_3`)x2@4lo8J)mcUDr7MIPsS@yLD-rUp7qwKTHI%UfZRmmAF%+nJMC3pmQ zm82C8rkSoeoneq-Bw^qtasB3u1hefY3>F{NTYT|E+ykREZyfUldSeQAM!GdG?=aq_0nt=j22q$NVd+joz3k|LFVh&dBGO+O*uYqMv1reAI-?%hV1Z zlT%rI?}%dQE#KEhQtu5Uc_fOBZE`n6MW;kuOqzZ(DQ%lk+M>Hj8y$?i#m-%MYqI`# zaI@m%gAE%^4kx$jY(KGK>pK>mT(LCm=?}MT*?KVHq0yNSJ9S?d-%U6nGneu5HJ*N+ z_QkGkvfIQq=Y&NhJ~p}&Hk;&K={NzS^}dyY~CB z(ha94B&MI?sk;d}9AuK^<$?p-^cEkSV9CMkme+Qib^C7DHfi0mo3dp$9zHN);%jd{ zSx~TSV@Hpveg2E1DFz8L1?x6{Jn%ew;zQT&Sib4Zd^;qJlaD+tI;b=|c21^EQvCFb z2^Haz_pDQYM5S89`n5gZ)nm5Xb?QaI@^!!01n)LBPd~6@_QQ|L5;+qETeh=Wt(VTL z*x9q=`0LiISD!uf*{u0}m%Zt?ySW+7*H4>#UAvi6St@s;pyPt0Ya00NSY{krHpO0` z;my{cmY1Ebb(`9MH$Tb2a-%q5t}jogOLvcj)7k#U00zbuiOt(`3`Cj^>$6)fv90{S z>%7^9gg!YgM#XwGowwy>{*YFLukV@NDB>B_Bge(1mVB)60w|wh<=j`xwQ#o`Y z;r^|wjpxpZNoF3IEx-7q#-a}6*0pyxR>-`YV0?@($>4@Gn^gCLh2qgZ8zq|CYqP`E zckPNwNtu|_s3++z)mprNzpGp5YVIwoS6^t= z6Pa=7+{c=Q4F%~6=N0enHtsgHKc8K$zehu4#;c>HX3P?@xf2b0in=483VCU4W_@t; zW~`p_H2!kqn-%}>F!m=jSiJ|HSmtD{d40l10+Z zH+t$e zrLS%dah7Be+nt=SxMTjJuGuS^{PKAYb!h!rq13F!7I4%o#IRLMEa0eRNa04m`h9-& zEB*F|Esu8>&AqJkf1xU?X5_)P)X>eHsTv((q8&%2lnRQcDsO7C2zPLqvikY6_0x9! zo+S9Fbh`Fpr={B+F0)Qr_oZUb)20dQRPKfyeA^YEFCD0G&#PreY?#&#?j@J3z2;XQ z`=YMiR%T=KU}Dhj?a_ayxpc;e1s&e~$@%B+vM8WEWOMf|N8x#yODg|-|r~5Xn&tpZT7rbP3w}pr1{$)+vfQF zp0v-bOMlkw$M06$KC&z|<*VZ#k?;i9oyVLNj4x);!QM_e5Co_{&lIXG-N)}3+8VS;GK z=?dxNr#@WO4qw-BXm3?_)l1%ri>>>ECunP(5Ep%t!n#kDb>AW;oqV28uR2rbi^Tsw zpwjS{;Thwcwc%F&*I%B$_ELY-e2rIMu1B!Vyr=PKKEs}ad@9RCpQN&Orm|j3edix` zr)BFy|F9p4u54jE2fA`MByol59Ox?E@F;2SznZ!Kg67t(_1n*gI5Ig;Q2f%~6=JHso~QiG8tu|L*?gDmHKD{=I+yrHehY zKlpAwL!$ciCy&2~&X};~qkQrVDVH`T5zz@rz4;a7*{J;8^ z=v)f7TK&=N(}`0G-ONS~&R;}1HYB9dgtsf+V=(YeEjQEQafkm-Hobz#~p9s{IBTg3HOUn z#Qk_4mFGyhS>E0hp=(_eqN9*|cN zBbmS8jIExswnW;$d?)MhTPoKD9PIPipGb8{?QorGt6iXbz2ceXEm`KJd6$G2^71DM zE?FNVFys4<>w7e(>P<*-(qHP{TGZd}(lpayw&@f;md%N?O|N{t$o=cehrK^#-OXHU z7Q2^ywcTJS$>I9#`RVU%nj76tt?ZNg)Ukc>i|?YhZ0GeY(htZfkJrt6%a;@NOanOfG@n|U7u31? zah29ocUG-+Dksj}wBKaozKHqvi)H-t?YDpGUOhSf$DR3&o9_F)xc-ut!(cHpSBdM@ z6ZS_PD;VFmntl<63e@{QI<%@0Yz~uHX78e*SCqXRpM=>%3!J_J6AW^y%%Z z=l1^j4GLR-$$ts`WxS-Yr?q0U*rT61?hhPUG+hE*StfTR*elI&XGs@4c)9e~=F(r? zrN2&>{>tt!NwaJ+7hp^cJitG5^Ow`6FD5?8wcY%$@8>J;D9>cQ)3$Brj`Oe|FUl(W@`;Y}c!NwUpVFLr_a|P7I8W{TN$;Tb;Jq zgVXeP(8GsbJKoO|+F7voMzrq4v)fl6xqbWo*T$B$t3x)hul~8C@z=fmikp{8vCdDp z@b?Rw%97c}dRIEuUk~ez)85yzG-&%|&Ct%>^`&2rZ{P2!o`3ztlb@d#bgaF%$nKSk z_^}H|>r(Yf^F;!h7*nbiH`u92_Zm*f%n=DVDsW=UWY@VIrMF|{@;5nL%D%VO|Fg~V z>3hN)gSwUE)_XbS)x?KaeP6cb;1|!i9zyT+OM9j^U3|xzym+a=#-ESR<#>i0vn}S_ zE%P;M?Yzz>?sYZ)5Hsm$d1UXwLmQ*Hk7unFjDA+i_F3(Ln!flB^Od(R z^4wQsxcy4?oJa#J!}C|Se|DGt`ds?UyY$yqtG`zIRw3)|sj}Wn;qB1REpS@XkW>`(`iwRbNjY!Rw>`ACRi(aSDTfrh+L?tZ1AIVZ;YnZAVp{qsMX?mzlzuHlwjQcfo~ z&llJ<`@5)_jEw)XW9e-C^S8%M`E}}}(}g5XaUuRmv9`B&)!vkCZ)Z)m3x!B2w!d!0F0vuOLD;vzTkTLlG|FPuqPmHg+yiAMWA8^zu--u{V>(OM7R{(tge z@$b_&b2m31-W+{-_ww4<#X9?zhq-6WvZ^%wqRi^-`YpdlC;ZUSZK^f;OBKSN7*0If z#li2gL}3w^;ADY~@=DJXAL+Q*#4fjwUcNtK`Tm&Y`;Q$~m@g`r{~=+O!s%6uRKLEv z_^x-B&F*KXS17K}{rstj*KtOL^$F#a<54YN4z?_Kc3k#W+j=c+4c+OxtY_8COyyf| z`_gaIrZm}On{F@mR||TX;*yZ(cj;ryg*#&Jx>UFoyLE$pJY|*Iu{`GPZTWBE@8xyg zeGa;0K1a&LR=Fhi&eU1I688T+yf7wSBj)4}rk#xuEAG6Ut@Ph$qJQdkpYI`Zw~e9M|I2>*u_eK4xzwUW&v3_Vz1$eKK3VT=g%(LxbQ^+5kDm>7o>ZiyEKl63AGFDRE3IColfQoUgf$`h z-}$Cl>)+mdo9)x@W~S>`wq++BzH!P(Qupnv?zB(QSGQJn_kGwpGtD|e?5?q^s6(m# z$GCS1kxzvq!xDmur*fJnW~r0=WZPD+Rn|AQidJryeREy+VNi8-`Hdvb^6lHFADVW0<;iAc<#x5; zl_#B*mD$yUTin{Y9~LB9NXcroIUQa2;>DY3Az7#2ym)bET1nMune%6@Us$;7cZXlS z^X6*N=ILwORZFEWoVaLqHBUP4(6tS-B@-UMEMzdtY32>ykjQ8@r9@(Jww#LXP0&%<`S9uM!?)*5zw7$!|FW$yJi6}Ar%zvaG}H6+>GES6cRzi~ zCFwQgQ0}w>!&BWMtA}poz4A+qp5Dp+xBhBlSZ--_>NFFc4Q4fQp;sTT zU2>zOsDJyh)i1YAIGM-$?TL`uKCAl4GUwP?s^X>`mb`yhK_`3C_7%d+Ir>*EzIJCE zS9o)y=cvSIBW~^9$+p|n?wrU+QOk?q;6b%a7K5 zz2cjZaX&=1;_73wXKU&U8_RO<%|HM4_7qOhwnY;U)_8nt(^*;fuu0E&m-g(tw$svD z=KhvBxZ~{HK8yW#tS6S<*ev|?+q)g!lXT4H2baxiDOk2eQT4Xfu_~dXD}pB_8!o@E zHuFxQz>DzvD%CIMAMW3qf1ba0&x6QI&knA9D}KXsqF;3Sn-&)DtOZFcuZ4wdSGcD- z&-#vM`OS5y2e|ZAw7wNRefIS4Z|0v1cUV6+sMLA8PdTLKVQYI>-3&d}o{)+uLGjiH zozK{ZxGClp!9?O6j zj+Y=INn= zxI7hKrc`Ts&iC4V(`wbEQ?KX$6R`a}Gyg@iV)?bFFF)nnHg;1FvvcQL$C6u?@1 z)yDPp$FKdR6MQmoOn(?F7|{M20q`UX@|37MmMKxnt`RT;oezUCF^bzB`p%A;fr6QY z*-^ydEryl~MuwoNQ3Vt9V_aetOwFTVQ=o>1(78@?@P*N!*-ZuT9H)W-=mKd4$dv)0 z*<+-u1Hdw7hM*ZrV+%_K&}{**`vX8K4X~WmVg@>f1{Th^=gW+ZEm5X3QO|1mY+xef z*HIH4GcsvsOx+w&4D^?~rrwO73W@GqSJa(ju@)~x>`^SfE4*!wQ@ z&Jj7YZSn&vV+Ku&=X;j8#$GS=Jg}nvtH!Zy(fa$X#GiP2a-U0%W2m=kM zt`efQqGvwS(!d2zBs`9ItPok@v)W;%!!CuVjV$g~Y^NrCO6J&Pd9LoS{U^7|39c^` z4$54Vw14#3vrWxkg5mEHPOBSpl64O=Fg0`6SRMY8y9M%eR?M& zA^%P7Oy1mCW*;JSt#5A4wzc2Ce0?U4+-!!Fn{h{up@E@=5qvh&*vQz>3^}si)^eA` z9Iw1*zWM$4y}#=c=ii8ZleCdvr)*=l`40xheJpkkOj9&ER6LmYG&iZZcJ?jS@N=5! zbxFhX?3A)cTRfGgZ=7nBxkzWij*lCaEmPhIhxop^bu3e8wO91D(yP-pMQg{DrN6s- zbdpl&{019ezYqGIbFKgH{eR}q=kxz(-Ha7rP}{@6swRJX^4q#k&fi`${Pbq8RME~;&m(q_UtKkhW6#3cCLK0{=a!b zx9p3zGwiD})?KszDcaz}ta0|dyzP=o@86jXUl=mx+gHBZ+28ZxAI}CKrd@r)%ll-4 z`s@DdELhKU>a;q)ZFy9s<5ZRcZig2yoz?r#+1dYBZm5zA@SA5>`t8o#ct>-GH$3&f z&Ro1aU3VdqLmbbUtpe6B=w?l#9ZrAAxiE>^61{@3`r}#7+7CbQB z@!7lQTWteF*Q>dHKeq1wYA(E0+$285bgjeoPf|>KzI{6P-zrbb8EWp zf4nBW{@1%Z}loaU)wC>52JlN6C6J8mA;>z9JcXdkVZVi36Ikf z+x*|u^b~%zZ8ZsV6|KJ~Gr#Lzg2e@`3HQQU^NN0L=lFTtIJD@w`s&#iwsFXcZ1yh; zb-c9j!n0Q2>nDO^Lobwws))W@}v`$ak zw0-xgfTi18*J(bQe0le>^?Ns*EYD53+@mkUm%H!V&yz2+=ih(Aef~^I=)$~jCMp{P zUu{;4ZMC`@+8VZKQ~Ix4-21YXjrI4c#9yrVaqD?})yeNcC2p%a+g7s_xr+XgbImE9 z+T&An#lz%HvgpME6Pw4}=W|}Ia+-c`TYA-JwRnp&Rg3E7y*I9GUD_5lP3m%}goU>+ zf9KL?@-0iJ$#*P0^fyv#*=hj^@3r64GrurP+*%{ClfONFg2kUhwH7(MyAOZkf3{)q z)WhdWMETAoUUz=9;Wzuk4bRyhZTS9A=G)TMAB^~Z&z6z<80)Ov9>1e~VZ{rb{Om-_ z>xPWWVy9X;+~nIFv&?|`M(<*azVrg7_5wEPLayb7jOiO19?v*#*>a&#*+Z$tMoUAv zzhv{ByFq$erf73K6qt8xPE7E`jkh;8EHC7GUck7#Fl3!&kMW+j+#HGAevb@}2_Dj% z_wG&pGhheoLD7y|C~8>8hd*`?J3)Mf^XDJ$o{5&JjhKgEsHa2hMz? zc;=x1-(!~bmfgV?ZN(MS`lnu%nS8uLD&1mfdO?eQy^P|V$+z!m_rH-?UO(mgw2u!s z-XDzI`L)=hRY327$h?O($>oOoY;HSwW<52Le`;yRc5S}E^BYX^$2n!5-FW;bpJR4G z+xM64KVF>M*7H8|+(}*C&39%jzqW7V#ko&E-`mlBai_xfUe9X})z(K<9yoSnUX`0% zeAK+=%@${uesS9TO{G4q(fiola=C+?bFx0UKl*H`m%b+b_z#ZrCvK!yzJG4{{oK}q zHyZOE|4oXqKmFeN@OvBe;s*kLr~lfWSACpV5He}c|LbabLRHy z0><>pe*aH~J|~K**_Qvje=I61R(kjQDCynzx0^M5{JqyBqVBK2PuqW0v-kLBEpc0a z$v*T=IG>jNs!uEStqaR?Ion$u_uu_D`!1Dw-uzh4jPU*~JM_Zc4bP*n)IGv-}ZB(_5J1RlhkX2x%eyZ z3Y1-Xzi{%t&bfc0nAQKPlwC3@)9Oo}w69b4OTDt6z1qX*d5gdJ=Iotar#z|F;kVqD z%ds~!_HW-@eu+_8*2&s8ndLv@?5B1)X{kp=+#+m`zdz)5u3uI{WxIgN-wTE^J6VnY z>bb0H>$}tRt$F7@zIthe?f>e(C&*q)1fhTXnw2<|HMkU>oc_|kuz}4z^4@N)+-+Pb zt1obcu9&_$AYI&X;f_$&7FUf*&+ScGC!W33ni46$`U?Nq6QWvS3!5f|YP7mObGx(3 zN!38)f9IdY-lE-3`&n3R%I#czyUbKBHajkgc*I%b z`dr9l&C_#QCWnN2PJG-Hm~cHbV*1kZ^G4a5xbA-vI;8VaKT12aWphw&%s%a{ds_Si zBwVvoLt66{+Gn3=l#}`D*e+vpLa|eZf4TaE!#qV14t(Mh4llL(=OtUewJ(0n2mH$@^4G?y!cWp#bc2~=?qc6>V{oMcnU?Cx!Ujgi1zg?x)EGeMH{M=!({^4+z?~nxt?6HPmuoxkz5RJ6 zZ)U&Ok98F(Ret*mo0YEYUaM_;MbhNzzUB8V|IWMi%zsPkCdbCSy(>3XCEfF~=h#&K ztFyCCSWG6RT7k{ZVyVHaW+^X=g)TB)Qx+VNm~w+%wa>BhbHJk{+4`M*^-KHe*Y?#j zd(GhMTlL|2z@yx4nk53E89OYVeag9VbY4*Y6`va$_S@`AEl8*-d7gYO<^JpR{a^p^ z|9?93@%uUVKiruatWdtMe7Su6vEL;fUr%0nAv`-*IPcJv8M(qX28x`*S_X+sYG)3# zu+Au8REt^lc>R@+;je!jwfbk}Q?t$0Y`)db_*sVp5{)N!uyq_>cdhq>-q!veUcNnb zfA-J6|KZNf;HR7OZT~;q&vbB}wf|#o3s&V_Gej0^9-YAII63ywI^)TTCexDi?!VTs zI2-akhi$S@bk z9XM+BuSoVk*X^j)kMAe!o_WcJDc_j2U_+Nubfb3YjPM^iGYW;D9n+cduu;-;Mqs0) zut72#=*R$WgU3O#^-KHqM;+I{_VIdR+e?LM3Cb8Q(N zHK$g~C#M!&u(q+e@br5*FAsmUb>AC#i5vAQ6K`1WTq?pdZJW=#)TbYVG(x22-{Efh z%P@2CdguNZLL9fxCR8^uU7lWJHdE{JsZ~1D`MU!*E4|X_NXZKket-8$+|u>sUs!l% zXL$+iVEiQe%gX9wacD{E^dD`@T>5#pcAfANW$RqbcBf_G!9Zpk0|icIErSFmwlfDB zSS1P=*-#>!CuW_pljympT5}>-vU+lDU@@18;Z$ERtu$OOU;9XV#zJcc$=Rl9t!gnm z{F+s&rlP$wGOM~Gl#b5WZ~d&KH1}0ku9j=)6aVc0E^}Yp72xB4dTSB$Zlk9!>+E+5 zO1)qyOc=EZ$FLCZx4wgJ$maqTgk@Wn8pgU#HV}lm2PwsG;uHf)w zu3(GmkCbV@mYA+P>YE|G=fWXo-ANY~G3lx;ILfmnqCrf2!BOPW!I}H=K~)yTL&rl4 zS)$@rDNTxXl2T%+n|dzn;#Q@?wHjv~1o%CJm2WQA}NK{+u-2I_7owB`@2*3=`ZBxihCz~prWaSvUVLT7{0x4>%|lSJr}RX=OXT&hGvFi+SIrb&pO=PRqJI>vVs5%*4jMtqGP-SBCs~X~uoK{S$MDV|+VP+MA>( zsZG-mQ0bhoJdPdDXK!1mz?qwg`$!7&0oSB^+Vu*}AboDmeHti@r-ym|B>$*VX+~m9#kxMe0 zPljd-?R~O(Yy4eNfW6Kd1BipjL^KXshV8^;%jvxv4kKrKgKZ zHgd*Ah}kTAeyj4P?(ejSlTSSroo`qr`W=s5BO4^LYwj$uXX4du8b7a0DXXjceW$_w zP35bNz5U;JG4I{YRcp-Xb!H>)(nA;8Z+pL8w05sV@3&)bwpDa;U1oQ;E6H5+=gu9E z>(BOf{<*g>!^(fM|2wU3wHI&Sl;;b(B-}zI_uDcraS9IPwxD*aBbw*HCo0WRcvhyZ~ni$|IbY3>UDcReQscq zEx)(a>z9?ziKHnqQBHng_Lh2eQIUUs>@f=AD|`~*#Fu;`QNGBcP3D|(dw1!dr&j-_ zTK(H<^^fn-312kak6!1mec*)9}SyXn`m!?fenQY|whPpP*4&1n%6_LPUc z6PcFaFmXC>O{7TT1o>%RcibmVSr_#wV2<9&wYeHELjUUAy&3X&{+4#*dGmUux$gNK zsatTx)IdmjuHKXHyvN(zc6Gfw^lsmiZTn8F(7q|nBVrxX=VG`ev1?Y1Y+Z5wN|S_d zqet-F<^uv$Kekbodk)inFT+W)mB^!3%eU@>4JL|6GKF)3TlsBE_ zUDU8Y;g$QQU7C%SMv6Blt~{iDHj%^pW~WN$eZ>laN6}M7_dS$2#B*PP@8jD0D5Xya zC)eZ+B6B@J+phrYWLdBf|>w+c-eog*^S3JmYXb=|c4e5CREYWMn@Y6ByS z&=)i1y2N(B@ZTGu*4{MLuU|bhi<>{?@4a<)7p&j7+&KM(O~gQf?c^JS)5ouwr+70o zt}PXv5;Lo2NkZ@hr~DhC3fJ{7q}vDYuFRN}Ql@w1AGhZO*{$`yb#KdFX2(A^DZi%v zm9>hws%4^Di=>qEsovd!vCmFUjekJW7P zYE~OKS>~4CTe#nGf2V|>o}$8e2Zu}R87%}IWEiIi9FSnT(azAP*zlR-0ZKvB*!G3} zgBg=fR+_Np3E`bATW#{T9#}hbQRtS)FBhj|s^2wP{r0A1_`Sc2gI_***Z*ec7tPOW zlmh0Q=(D-_{`+@(yG?IZe#TGzTif_Ad3xEqKR05Ye(HYtcJIME>n1lp{TaLS?HN16 zpIw)vFL*BGnjzqc5+qHPE-`%skcwXbJ z>Hm~@zr5Gxx3LrX^j%)&?Tw(m?6j|YclUkw{re&$J?+=elMnmPoSE=k?Q`(qw>OXX zP1jamUX}Cl-;)nM^Yec=+ixy@v$**0PqzE@P4^lYqa)NWzF89WZpN(2%bVO+3$41a z>1RTx-Lyo-qB1Z2^X>Z#<&zwHx)KYY>s02IZTn`J>6X>~h3)jcjb(Go?I7pna({j)Eazv-2edoi&^PV|K#qix!P<4j+6 z1bknrkK910(!89~*lv7M$w1yO$}e!?!=qbU;>(Y6r2U%wuTkCIeZI^G^Ly#m-}f4r zem%7B+%ap*sHhv!8gF7XOz)}3&5@UxQ(u_$=;+Ud3xA3`?lh5GX1<`6clxS=l<3c1 zp%xV-F|A%w>$%JqX7z-#@`rdhwQ_~bXy7sqNMzKCY2bp@HS?anS?_%Gt@gu0^Cc0{ z>fXEiq63nRCuaz&{x@};Azp|F>cSExKyOFGKwa!9z3hUn={b3_mKm8b15{$2X(;cfAk?dmgq zZ&W>({bl$slWE@--c7Hn=D(Zhwx})jD&MJJ>TxNo>cPk5XV0zuvod+}{F;|X=c`Zo zQ@Hlem5Iv!8^6#0`_a2-z5b`j7b!gfpIc@$b*|acCLI+rGxpJL&6f27vYk>hj%iLv z@2GfmzwWEN;BTEw_4QZmyuV%JyApQeRmT3$dRK!5Eprt7H`va&?=HzTwP`sI zns#sInjX;F_43`D6&$9g3KsQPoYGj>V>o3af2xeL<=a;P3`+<+>dQ!7ygjICZ}_tnft?oAlYhoNA%{u z6ONrlB|Ez^!nQRsMl!Ni)ZA8FUaO)Vy~CPaRP=rC95cVbVx!4hzn=fHo%L5bn-{yx zlrP+G|GxYZa7gPIyVc}oPHAK1*|CnR%}fh-Z2$gYTgjry5?Re3KNhLP+!D%{f8E>S zHQVvVFU9-Il@{$d@F|eZQvPc}Q9wkK_qV07cV=>#Zxs%U__UMj_o=^svMxOepBz6k z?C9_ECG7Jhcv#rw-`f{m+Y`S#b;siMu_8&%GcR$>Je%WVZM9wY?Q3fjvHA5`3C=5a zth6-@?3*Gbb;u|GK$}Tj{HHqEq93m^-mvIw{`2MVmc{3mnJY|{u6FX<7a8fQbvj5+ z(Bj9hq^+50#xp86DO}t#J$KXIoaEj9-*0$}ewKNc7IX09o(pytZfbEnN!8PrtX{q6 z%Daip`4^D&u5&mWEEZLF4M1nP%mHdChIXr6U&V0!Y?1Ze5}dReU^737xbKdmMDSAEOVD9RWFw@+eEjNO-N~~jD zr+o?8K5@n46IbrUsAPPQ>T!I0<4E165NT!a+sBmiT8mZ(Sz7H$o~^PlGH9`ql|jS# zjV!A-&P<5idNfESG;ZCZGqPS?M$=`(UPnADn%yWYd1Rs^r(};}bM1)-5=IvNKEWqp zO-LW-JmZS!YyT1@CT70-(;l{DT6;v=f%YAHe2>W7clzw$7}V!8DJbPp$F^lPGkxw& zjnh|+Dv@-acuS@5?YaD#MQ`$6-@bm`_VJRWJ^D{isMMr%mGR_d|39(w{gP9HnHlT(Z@D?c40{U;Uf=y! zQ|Ei@nx4-zIkB()-OQs(F;R*~XRqEZqa!Swn|$tx`m$v&9{&87Ky z$8yO_eBA2h_5Dcgwx&hqs|3yceg9qSoVtF_Pc!Mm^0{mO`Wt;@zppfZ+N}w{-&W1{ z*Lr)Mo!`}k7k*bQJ8W{NblIKB)_{?eiRryyy926t4+97bNvS6@=g6g zSKod=!R8&VrN9pP>b@ z3NrVTdehR~$K7>T{VTIu^{TA%#rv|Ledb+h&o^lR|Sh z#-Uq^2MN~x zIPh=EBp2b z9gko8Sp4;m*C&eeq*Lsbw+6K)yIwlZeCF7+6#tb+zDMjiYdfv&xDIPY&HAO5tej;z zZ)-g-J!70#taI+x&6yQB78j?<^02ZMIaX^V7ON=xSuO1d*|to1cJ<=Bx0h`%-JP7d z+kAe(|ERp=OIe-#smXT}ce^ET+G_b$t~)B7zk1VQ!)^&(^D<9kyhr^4fjVT+6C;%AYo`ZkSV4s#C1}Jb}IL4dX5U zu9zoBoIY+{`7*OO+RSFwhUzn`9*2BA>7TdyZ(Q}QwYQgDh}KOhk-om5{GRIlWm-FC zvKRc1S$)MivvuQz)=w@iA-BJhh_3?Yu&_ zih9nSq@F$BH}~DQ(x+vc@6BZVbUoy#dDpvbr{5H`PvPghy{K!2QkL?`^Jo9?{M~u; zd~j&4_|Nd|_n8|xE^J@PB_<+2LwM>7kM<`^CjE@NHOKCB@1=?H^X=UIW#%kDZ}an% z@aK}oRawSQF2DKz_WKF-Z1<;@`+Y+V6cnEqd?|lg$YHWabZ*S4r@38D>$`UgcfOck z^YF)M{rP_$-?#ggEN|)Aa?3jYuU290^YZ%XbNzQtes28qi0ibAnJFPI)AR+SjkNf} z9%ihZ(0_haq+9FL_bwNs5|_I^@#zrMlbF~bxYr| za9=dx^y(XK{x*S6q~{!VIXLzVem$%k#1~WZrW|Pgylr ztM~ZU`9a4F{@F_X;M?ii_G4d)^Tj8h7AeQ1T%D}*^6VqwU%vl#F4N&k2|Shd(V5d` z_S7e3^QT@4SkV1^tJLbQz0qM(7aNwu#(c6q&7t~qudLYHXVX@DoUUBGWy@vl&;DnB zEYC|^F*(#CTITIuw_o>w~bu!-mNo3XGCi;tx>uVG%-U@>g6*FNq^s@Wp9{l^ee~)CdTk~~+oe`1AVvb^Y>z<;*3!9oIWP-kq@Itqbc8iZQEMB6)&{KJn)OxLGX8(Yigkyj9MVl}9HyBVWk_#YboC`e zuOIu?u*KJ3?vQvHQTsSi&D*QjjdAOt1s5|GT+uMiy1Pr~uQTJ`&XbdZ_9Z1P3}~un z4RN0IZlda5KB>9K7F^yRKkdd|ueENBypvURr>pkvTX|hh&qiw1>kVudqJ3_KMkXy< zI3?9nn|ZRr%dowS7bJNuyxOYP?0EiNx}@RWp6jo+Sgj3O9qO>!VB_(SsHtJu7wl&B zC$FzR@%z1u1ncH^|28eT{IVqK+j2$is>Ks;x+rg9n|m|k$)=P&-qT8~=6Wp;c3N%n zQYG?IX7B4wU-(&l7P+$5{o&jcp%MBf>S&7R`G1m9o01G9XZo5rrC*5bQCl}@hK^*) z9>4XMy*Dihjw-3Ly;3#RFFotk)^_gxwpUhb7=6qxZqV}f(Fxh4Bbj1gmU1-1jG1lo z%%2g9Zk{=M;!MAs=K}K=yj5EdE@+9Up7Pc;j9?m zt{D+4uDprjO%_gBax*E>>>^Y1&5W%aF+DnK;*Lr8uq}4XHCdGq5P36Ms`*%k08bmc z$zHPxlaT9)yb3$!F1xy*L-u5<;hwhIiiMoL(X4_7(XJA#lbyG4`SiGi-LJTjV>H_& zZ8DeJ)XOPm&e>OH^;}z~V;2K4AmRJYv#a}bf{mq2cV9b{v~h;+pZ=aLx85{7*l;*! z>yEd4+QQ1;zTUic>~NFoh78tjskE(^l{hZUED_s$H>b|wa(1)3bE)h#hI3~9siD(m z%t|a2pRBr&)96xKIm4pMjNQH;EatrVeZsZu!fCa~x{DOJtQ-RlFDm0+#Ie^^Jz|d7 zY35mR(vVb8(0Z%o>VgxeoU6;$HebyWUCZMl`LbXK!|L0t7ZbnkzT?xgMlVeNvFq#% z;Zn2JY>Sqi&DwfU;lk}XHh z+-3RJvu_`2oxH2$9k%;w-c|FIlWFaaSE60}HS7?vk*|{y%e&#N&0gMrYp?V1m9_6x z)XrPaU+|-LL#2%3{Q2Dz{ckBvlPkM5x2*Z_!d74D5{o$oQN4V#@26EGf}%0x&Q2*2 z!^uaJDwM7D9~bSsX}{gu?bj`)25XUq%kR5bonm^J*2EpNw${ClZqI5ZftknEOC!Tcu(B=2llEjAhc^!ZUAdmHfwM^2JZ%65p1i zw;%HUzWny1!0!EX{~wsla<4)E!R9SLc|S-6{oBUivi`yMBYWo|650ff)Y{dIWev-> z-`vXD=%8@n*#8qJGOC$4)Pz55XW!JBBI00ha6f}<)3$)<6}}uSJs+euHzsg|?U(qV zI(yTPf`{PPaY&C}m}`0iDQw=Tzeh^sXv7BHP1({(Mn{s3Hf`2s zKf-P4vOB4EC)4k*-4%Qp|9ITiEZ)2Lu3Uelz}7EEAITQD$!zpc{MBf`{NwgENV;8` zyZGbP1sX>$-+pteF+^*s*Cu;S$(j078qRrfcD}okv2s#+(h<9P)2DwtS>nFy@XeV* z69wFZ=4MtTy$jhhsRNOf8yamNpS-%D<8PE?_rusXyIoRVoH}tLD3##^FaLtg=Eimg zXO3u?E7aea>=?dJ`~K6DscT+r$vgSv(+BmvGp`h+zhd-V&nw4!wKRe`rzXjW+vHH) zx7R^BZQl=dGqt~STKI8tF!N#QhDSHpp0IIGmOeH^LezXimE`>QXTLwIJ=!E|`8#Fd zjqZeHYbQvaPDP}m14r)s%iHjN|FeL!g?p#R{tnD*>XDI^X4krX{Nm<~d+KX`Yc!m) zSUvsY(tknqngN>D7a!k=U-w1%j;vOB{`;N({nOTqLR{tZFTOEh}v2h!^N zBQww1(FZo=O1CUfzB&@K|xF23SUzNLo)>bAD43gCLKpm+8d@rtnVH8bm>NYZSeP4vh0VbCFCfOrsUdEiGeX?d-rQ61;HO$P9E=6qspa zqmO%;s<8>mvea0VBawP5RXjK}VMih{ac>RjNu9%7AAi$ZXeE2j*^2b?X|Z9~o=16l z-&{U(H=B;^l2^PspU;*!tT*s-Wpb>(d6W0&=Gjrc5s~v}mu@t;nl;07THvXD2~)aF zwQDWEeikijshuI$6QW$ZbIRGi4bPT+)%RJT^Hk-0CO>oH)9BUrZ*yvV-c)Fgx#E;>BF{WueLwdWz2#&k79Da}wODGmOPN7TIV{ z@DWUW;+PYgDF04;cWA|#^}8jRPyMR>Xk2`G5|^Cz()70>?|w7?d%yN8IAV%Ric%AE z!Ax+(R1~GA@p2g`7?~UJav8vZf|;qQv8e(`3Myu3t^j5!iih)i_ z18GLmYhi$?*V4oiU9X{mr4fdhkqL$!hQ<~a80t)o4Kc*bEHL$27@44}Gc_{AP-kjl zVT!KK!q^bQeHO;%7-3~$Vu2}UW`bd_g}EU{xLBBDhJl3zXdeV>m|9q3hKr>Eru!@n zFvH5y&>X|RmPVNIVQFlH5eAkfnBi<`YG{leR+gri;b&HB!_dIc5bR;2aI2#)nWB9|^5TmR!G&aO2uMCaB^*Ne(#wM8NnP8?}V-rmG8JmJy zkZ9(CTQz86=9pn>Y+-k3fT!pPI6=H?jgGqp6rC=bjGFv}M+ zL(HVhTVUoRb1dn=!oUQ>%@&3jb(x`s5oWwu7@J{) zvxO;Uda*FI#Bj5PIcB|JVS!n`SXg4#0hR`sWr?K`W_fREgjwEOVz$i;Eln}nDu$M3 zh8S+PG{>wjEiEw0T_Xd`v}cHZm{;9m*%7Ob+ zy{VxkMtqnWV}`vcX1mGA)D+Wire>I7Z)%R2u1qa4!`{>qqdqq>GsFygGb4<0$jHnX zGwjWP91TL(H%@H^R)9=Ej&|Z;mDHnww&l zFXrYLeFGzNbBz9*k-4QYdV9#o!T_V4V`O27(Y7$MFvK*^!U&`PW@KTEQCApQm|%=O z7+IKNmYo)u?QA0pGgI_-laYnFF{YRWM%%*3!V;t1X=Gt(f!-E1vNXWVTb72H@oQ;> z(cd?+#8M7fVkw6#O)&a?MwX@+eI_GIGmJitktOJ;bx`&-fRD)-S(;<<2bS{660^KA zHZZ{$k1#f{z!(QHHn7CVAE53BdKqDCXo%6aH8#X7H;hdUF~Y#u)EJ|TFgC>;_c1mz zz(~8smSz}nWNc}H(VsCfFvN&A69W^B{)~y48Af}^#LU71J*}CT8)1wEnV4ht2ThHQ zFvf07vGlD>jV&+T6edqaAB*V1`ld znj0Eo>V8-36v3krq`1`0t6`a${mB??BMR0QG$XI7;u7?~UEd%8s1SUMRP8k?D0 z8k$==TbLLbIvblB7#O)Y8JL)xTeumx7?_%Y`m2UcW){vSX3nlgZe}K~F0L+4j-aNj zo1vYJo4K2%qou2}xsjQRg`ts=i=%;og_*I1lZ%C!nW?#<9eAw0xFoTt1Tu(gX<%ex O!ONwp>gw;t%LM?zslovO literal 0 HcmV?d00001 diff --git a/app/core/src/main/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdown.java b/app/core/src/main/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdown.java index ce5a610789..42ebd51ab3 100644 --- a/app/core/src/main/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdown.java +++ b/app/core/src/main/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdown.java @@ -1,11 +1,13 @@ package stirling.software.SPDF.model.api.converters; -import org.springframework.core.io.Resource; +import java.nio.charset.StandardCharsets; + import org.springframework.http.MediaType; import org.springframework.http.ResponseEntity; import org.springframework.web.bind.annotation.ModelAttribute; import org.springframework.web.multipart.MultipartFile; +import io.github.pixee.security.Filenames; import io.swagger.v3.oas.annotations.Operation; import lombok.RequiredArgsConstructor; @@ -15,8 +17,11 @@ import stirling.software.common.annotations.AutoJobPostMapping; import stirling.software.common.annotations.api.ConvertApi; import stirling.software.common.enumeration.ResourceWeight; import stirling.software.common.model.api.PDFFile; -import stirling.software.common.util.PDFToFile; +import stirling.software.common.pdf.PdfMarkdownConverter; +import stirling.software.common.util.TempFile; import stirling.software.common.util.TempFileManager; +import stirling.software.common.util.WebResponseUtils; +import stirling.software.jpdfium.PdfDocument; @ConvertApi @RequiredArgsConstructor @@ -33,10 +38,27 @@ public class ConvertPDFToMarkdown { summary = "Convert PDF to Markdown", description = "This endpoint converts a PDF file to Markdown format. Input:PDF Output:Markdown Type:SISO") - public ResponseEntity processPdfToMarkdown(@ModelAttribute PDFFile file) + public ResponseEntity processPdfToMarkdown(@ModelAttribute PDFFile file) throws Exception { MultipartFile inputFile = file.getFileInput(); - PDFToFile pdfToFile = new PDFToFile(tempFileManager); - return pdfToFile.processPdfToMarkdown(inputFile); + + String originalName = Filenames.toSimpleFileName(inputFile.getOriginalFilename()); + String baseName = + originalName.contains(".") + ? originalName.substring(0, originalName.lastIndexOf('.')) + : originalName; + + String markdown; + try (TempFile tempInput = new TempFile(tempFileManager, ".pdf")) { + inputFile.transferTo(tempInput.getFile()); + try (PdfDocument doc = PdfDocument.open(tempInput.getPath())) { + markdown = new PdfMarkdownConverter().convert(doc); + } + } + + return WebResponseUtils.bytesToWebResponse( + markdown.getBytes(StandardCharsets.UTF_8), + baseName + ".md", + MediaType.valueOf("text/markdown")); } } diff --git a/app/core/src/test/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdownTest.java b/app/core/src/test/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdownTest.java index b63e58b524..3bd6b7fadb 100644 --- a/app/core/src/test/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdownTest.java +++ b/app/core/src/test/java/stirling/software/SPDF/model/api/converters/ConvertPDFToMarkdownTest.java @@ -1,16 +1,17 @@ package stirling.software.SPDF.model.api.converters; -import static org.junit.jupiter.api.Assertions.assertEquals; import static org.mockito.ArgumentMatchers.any; import static org.mockito.Mockito.*; import static org.springframework.test.web.servlet.request.MockMvcRequestBuilders.multipart; import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.*; +import java.io.File; import java.nio.charset.StandardCharsets; +import java.nio.file.Path; import org.junit.jupiter.api.Test; -import org.mockito.ArgumentCaptor; import org.mockito.MockedConstruction; +import org.mockito.MockedStatic; import org.mockito.Mockito; import org.springframework.core.io.ByteArrayResource; import org.springframework.core.io.Resource; @@ -21,9 +22,10 @@ import org.springframework.test.web.servlet.MockMvc; import org.springframework.test.web.servlet.setup.MockMvcBuilders; import org.springframework.web.bind.annotation.ExceptionHandler; import org.springframework.web.bind.annotation.RestControllerAdvice; -import org.springframework.web.multipart.MultipartFile; -import stirling.software.common.util.PDFToFile; +import stirling.software.common.pdf.PdfMarkdownConverter; +import stirling.software.common.util.TempFile; +import stirling.software.jpdfium.PdfDocument; class ConvertPDFToMarkdownTest { @@ -47,68 +49,68 @@ class ConvertPDFToMarkdownTest { @Test void pdfToMarkdownReturnsMarkdownBytes() throws Exception { byte[] md = "# heading\n\ncontent\n".getBytes(StandardCharsets.UTF_8); + String expectedMd = "# heading\n\ncontent\n"; - try (MockedConstruction construction = - Mockito.mockConstruction( - PDFToFile.class, - (mock, ctx) -> { - when(mock.processPdfToMarkdown(any(MultipartFile.class))) - .thenAnswer( - inv -> - ResponseEntity.ok() - .header("Content-Type", "text/markdown") - .body(new ByteArrayResource(md))); - })) { + File tmpFile = File.createTempFile("test", ".pdf"); + tmpFile.deleteOnExit(); - MockMvc mvc = mockMvc(); + try (MockedConstruction tempMock = + Mockito.mockConstruction( + TempFile.class, + (mock, ctx) -> { + when(mock.getFile()).thenReturn(tmpFile); + when(mock.getPath()).thenReturn(tmpFile.toPath()); + }); + MockedStatic docStatic = Mockito.mockStatic(PdfDocument.class); + MockedConstruction converterMock = + Mockito.mockConstruction( + PdfMarkdownConverter.class, + (mock, ctx) -> when(mock.convert(any())).thenReturn(expectedMd))) { + + PdfDocument mockDoc = Mockito.mock(PdfDocument.class); + docStatic.when(() -> PdfDocument.open(any(Path.class))).thenReturn(mockDoc); MockMultipartFile file = new MockMultipartFile( - "fileInput", // must match the field name in PDFFile - "input.pdf", - "application/pdf", - new byte[] {1, 2, 3}); + "fileInput", "input.pdf", "application/pdf", new byte[] {1, 2, 3}); - // ResponseEntity is written synchronously on the request thread, - // so there is no async dispatch to wait for (unlike the old StreamingResponseBody - // path). - mvc.perform(multipart("/api/v1/convert/pdf/markdown").file(file)) + mockMvc() + .perform(multipart("/api/v1/convert/pdf/markdown").file(file)) .andExpect(status().isOk()) .andExpect(header().string("Content-Type", "text/markdown")) .andExpect(content().bytes(md)); - - // Verify that exactly one instance was created - assert construction.constructed().size() == 1; - - // And that the uploaded file was passed to processPdfToMarkdown() - PDFToFile created = construction.constructed().get(0); - ArgumentCaptor captor = ArgumentCaptor.forClass(MultipartFile.class); - verify(created, times(1)).processPdfToMarkdown(captor.capture()); - MultipartFile passed = captor.getValue(); - - // Minimal plausibility checks - assertEquals("input.pdf", passed.getOriginalFilename()); - assertEquals("application/pdf", passed.getContentType()); } } @Test void pdfToMarkdownWhenServiceThrowsReturns500() throws Exception { - try (MockedConstruction ignored = - Mockito.mockConstruction( - PDFToFile.class, - (mock, ctx) -> { - when(mock.processPdfToMarkdown(any(MultipartFile.class))) - .thenThrow(new RuntimeException("boom")); - })) { + File tmpFile = File.createTempFile("test", ".pdf"); + tmpFile.deleteOnExit(); - MockMvc mvc = mockMvc(); + try (MockedConstruction tempMock = + Mockito.mockConstruction( + TempFile.class, + (mock, ctx) -> { + when(mock.getFile()).thenReturn(tmpFile); + when(mock.getPath()).thenReturn(tmpFile.toPath()); + }); + MockedStatic docStatic = Mockito.mockStatic(PdfDocument.class); + MockedConstruction converterMock = + Mockito.mockConstruction( + PdfMarkdownConverter.class, + (mock, ctx) -> + when(mock.convert(any())) + .thenThrow(new RuntimeException("boom")))) { + + PdfDocument mockDoc = Mockito.mock(PdfDocument.class); + docStatic.when(() -> PdfDocument.open(any(Path.class))).thenReturn(mockDoc); MockMultipartFile file = new MockMultipartFile( "fileInput", "x.pdf", "application/pdf", new byte[] {0x01}); - mvc.perform(multipart("/api/v1/convert/pdf/markdown").file(file)) + mockMvc() + .perform(multipart("/api/v1/convert/pdf/markdown").file(file)) .andExpect(status().isInternalServerError()); } } diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/model/api/ai/AiWorkflowOutcome.java b/app/proprietary/src/main/java/stirling/software/proprietary/model/api/ai/AiWorkflowOutcome.java index a7239e8f90..2bed56f0f0 100644 --- a/app/proprietary/src/main/java/stirling/software/proprietary/model/api/ai/AiWorkflowOutcome.java +++ b/app/proprietary/src/main/java/stirling/software/proprietary/model/api/ai/AiWorkflowOutcome.java @@ -21,7 +21,8 @@ public enum AiWorkflowOutcome { COMPLETED("completed"), UNSUPPORTED_CAPABILITY("unsupported_capability"), CANNOT_CONTINUE("cannot_continue"), - GENERATE_FILE("generate_file"); + GENERATE_FILE("generate_file"), + CONVERT_MARKDOWN("convert_markdown"); private final String value; diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/service/AiWorkflowService.java b/app/proprietary/src/main/java/stirling/software/proprietary/service/AiWorkflowService.java index 5683d634c1..8f82a36581 100644 --- a/app/proprietary/src/main/java/stirling/software/proprietary/service/AiWorkflowService.java +++ b/app/proprietary/src/main/java/stirling/software/proprietary/service/AiWorkflowService.java @@ -66,6 +66,7 @@ import tools.jackson.databind.ObjectMapper; public class AiWorkflowService { private static final String DOCUMENTS_ENDPOINT = "/api/v1/documents"; + private static final String PDF_TO_MARKDOWN_ENDPOINT = "/api/v1/convert/pdf/markdown"; private final CustomPDFDocumentFactory pdfDocumentFactory; private final AiEngineClient aiEngineClient; @@ -194,6 +195,7 @@ public class AiWorkflowService { return switch (response.getOutcome()) { case NEED_CONTENT -> onNeedContent(response, filesById, request, listener); case NEED_INGEST -> onNeedIngest(response, filesById, request, listener); + case CONVERT_MARKDOWN -> onConvertMarkdown(response, filesById, listener); case TOOL_CALL -> onToolCall(response, filesById, listener); case PLAN -> onPlan(response, filesById, request, listener); case ANSWER -> onAnswer(response, filesById, request, listener); @@ -330,6 +332,77 @@ public class AiWorkflowService { return new WorkflowState.Pending(nextRequest); } + /** + * Deterministically convert each requested PDF to Markdown via the {@code + * /convert/pdf/markdown} endpoint (backed by {@code PdfMarkdownConverter}) and return the + * {@code .md} file(s) as a completed result. No AI resume — the conversion output is the final + * answer. + */ + private WorkflowState onConvertMarkdown( + AiWorkflowResponse response, + Map filesById, + ProgressListener listener) { + List filesToConvert = response.getFilesToIngest(); + if (filesToConvert == null || filesToConvert.isEmpty()) { + return new WorkflowState.Terminal( + cannotContinue( + "AI engine requested markdown conversion without listing any files.")); + } + + try { + List resultFiles = new ArrayList<>(); + List inputNames = new ArrayList<>(); + for (int i = 0; i < filesToConvert.size(); i++) { + AiFile file = filesToConvert.get(i); + MultipartFile multipartFile = filesById.get(file.getId()); + if (multipartFile == null) { + return new WorkflowState.Terminal( + cannotContinue( + "AI engine requested markdown conversion for unknown file: " + + file.getName())); + } + listener.onProgress( + AiWorkflowProgressEvent.executingTool( + PDF_TO_MARKDOWN_ENDPOINT, i + 1, filesToConvert.size())); + Resource input = toResource(multipartFile); + PipelineDefinition definition = + new PipelineDefinition( + "convert-markdown", + List.of(new PipelineStep(PDF_TO_MARKDOWN_ENDPOINT, Map.of())), + null); + PolicyExecutionResult result = + policyExecutor.execute( + definition, + PolicyInputs.of(List.of(input)), + PolicyProgressListener.NOOP); + resultFiles.addAll(result.files()); + inputNames.add(multipartFile.getOriginalFilename()); + } + return new WorkflowState.Terminal( + buildCompletedResponse(null, resultFiles, inputNames, null)); + } catch (InternalApiTimeoutException e) { + log.error("PDF to Markdown conversion timed out: {}", e.getMessage()); + return new WorkflowState.Terminal( + cannotContinue(toolTimeoutMessage(PDF_TO_MARKDOWN_ENDPOINT, e))); + } catch (Exception e) { + log.error("Failed to convert PDF to Markdown: {}", e.getMessage(), e); + return new WorkflowState.Terminal( + cannotContinue(toolFailureMessage(PDF_TO_MARKDOWN_ENDPOINT, e))); + } + } + + private Resource toResource(MultipartFile file) throws IOException { + TempFile tempFile = tempFileManager.createManagedTempFile("ai-workflow"); + file.transferTo(tempFile.getPath()); + final String originalName = Filenames.toSimpleFileName(file.getOriginalFilename()); + return new FileSystemResource(tempFile.getFile()) { + @Override + public String getFilename() { + return originalName; + } + }; + } + private void ingestFile(AiFile file, MultipartFile multipartFile) throws IOException { List pages = new ArrayList<>(); try (PDDocument document = pdfDocumentFactory.load(multipartFile, true)) { @@ -551,16 +624,7 @@ public class AiWorkflowService { private List toResources(Map filesById) throws IOException { List resources = new ArrayList<>(); for (MultipartFile file : filesById.values()) { - TempFile tempFile = tempFileManager.createManagedTempFile("ai-workflow"); - file.transferTo(tempFile.getPath()); - final String originalName = Filenames.toSimpleFileName(file.getOriginalFilename()); - resources.add( - new FileSystemResource(tempFile.getFile()) { - @Override - public String getFilename() { - return originalName; - } - }); + resources.add(toResource(file)); } return resources; } diff --git a/app/proprietary/src/main/java/stirling/software/proprietary/service/PdfContentExtractor.java b/app/proprietary/src/main/java/stirling/software/proprietary/service/PdfContentExtractor.java index c06007f318..9dccb91f38 100644 --- a/app/proprietary/src/main/java/stirling/software/proprietary/service/PdfContentExtractor.java +++ b/app/proprietary/src/main/java/stirling/software/proprietary/service/PdfContentExtractor.java @@ -30,11 +30,7 @@ import lombok.RequiredArgsConstructor; import lombok.extern.slf4j.Slf4j; import stirling.software.SPDF.pdf.parser.PageImageLocator; -import stirling.software.SPDF.pdf.parser.PdfIngester; -import stirling.software.SPDF.pdf.parser.PdfModels.ParsedPage; -import stirling.software.SPDF.pdf.parser.PdfModels.RawLine; import stirling.software.SPDF.pdf.parser.PdfModels.TableFragment; -import stirling.software.SPDF.pdf.parser.PdfModels.TextFragment; import stirling.software.SPDF.pdf.parser.TabulaTableParser; import stirling.software.common.util.ExceptionUtils; import stirling.software.common.util.PdfUtils; @@ -50,7 +46,6 @@ import stirling.software.proprietary.model.api.ai.FolioType; public class PdfContentExtractor { private final TabulaTableParser tabulaTableParser; - private final PdfIngester pdfIngester; private static final int MAX_CHARACTERS_PER_PAGE = 4_000; @@ -196,8 +191,6 @@ public class PdfContentExtractor { case PAGE_TEXT, FULL_TEXT -> Optional.ofNullable( extractText(lf, fileReq, remainingPages, remainingCharacters)); - case PAGE_LAYOUT -> - Optional.ofNullable(extractPageLayout(lf, remainingPages)); default -> { log.warn( "Content type {} not yet implemented, skipping for {}", @@ -222,35 +215,6 @@ public class PdfContentExtractor { return extracted.isEmpty() ? null : buildExtractedFileText(lf.fileName(), extracted); } - private PageLayoutFileResult extractPageLayout(LoadedFile lf, int maxPages) throws IOException { - List parsedPages = pdfIngester.parse(lf.document(), maxPages); - List pages = new ArrayList<>(); - for (ParsedPage pp : parsedPages) { - if (pp.layoutLines().isEmpty()) continue; - List lines = new ArrayList<>(); - for (RawLine rawLine : pp.layoutLines()) { - List fragments = new ArrayList<>(); - for (TextFragment tf : rawLine.fragments()) { - fragments.add( - new LayoutFragment( - tf.text(), - tf.bounds().x(), - tf.bounds().y(), - tf.bounds().width(), - tf.fontSize(), - tf.bold())); - } - lines.add(new LayoutLine(rawLine.bounds().y(), fragments)); - } - pages.add(new LayoutPage(pp.pageNumber(), lines)); - } - if (pages.isEmpty()) return null; - PageLayoutFileResult result = new PageLayoutFileResult(); - result.setFileName(lf.fileName()); - result.setPages(pages); - return result; - } - private WorkflowArtifact buildArtifact(ArtifactKind kind, List results) { return switch (kind) { case EXTRACTED_TEXT -> { @@ -258,11 +222,6 @@ public class PdfContentExtractor { artifact.setFiles(results.stream().map(ExtractedFileText.class::cast).toList()); yield artifact; } - case PAGE_LAYOUT -> { - PageLayoutArtifact artifact = new PageLayoutArtifact(); - artifact.setFiles(results.stream().map(PageLayoutFileResult.class::cast).toList()); - yield artifact; - } case TOOL_REPORT -> throw new IllegalArgumentException( "TOOL_REPORT artifacts are not produced by PdfContentExtractor"); @@ -569,7 +528,6 @@ public class PdfContentExtractor { */ enum ArtifactKind { EXTRACTED_TEXT("extracted_text"), - PAGE_LAYOUT("page_layout"), TOOL_REPORT("tool_report"); private final String value; @@ -633,40 +591,4 @@ public class PdfContentExtractor { this.report = report; } } - - // Serialization contract with the Python engine — see PageLayoutArtifactContractTest. - - /** One text fragment with its bounding-box geometry and font properties. */ - record LayoutFragment( - String text, float x, float y, float width, float fontSize, boolean bold) {} - - /** A visual line on the page: y-coordinate and all fragments on that line. */ - record LayoutLine(float y, List fragments) {} - - /** All layout lines for a single page. */ - record LayoutPage(int pageNumber, List lines) {} - - /** Page layout data for one file, as a PdfContentResult. */ - @Data - static final class PageLayoutFileResult implements PdfContentResult { - private String fileName; - private List pages = new ArrayList<>(); - - @Override - public ArtifactKind getArtifactKind() { - return ArtifactKind.PAGE_LAYOUT; - } - - @Override - public int pagesConsumed() { - return pages.size(); - } - } - - /** Artifact carrying full spatial page layout for all input files. */ - @Data - static final class PageLayoutArtifact implements WorkflowArtifact { - private final ArtifactKind kind = ArtifactKind.PAGE_LAYOUT; - private List files = new ArrayList<>(); - } } diff --git a/app/proprietary/src/test/java/stirling/software/proprietary/service/AiWorkflowServiceTest.java b/app/proprietary/src/test/java/stirling/software/proprietary/service/AiWorkflowServiceTest.java index 81d1ccdfa4..9e611cf32e 100644 --- a/app/proprietary/src/test/java/stirling/software/proprietary/service/AiWorkflowServiceTest.java +++ b/app/proprietary/src/test/java/stirling/software/proprietary/service/AiWorkflowServiceTest.java @@ -439,6 +439,35 @@ class AiWorkflowServiceTest { verify(internalApiClient, never()).post(anyString(), any()); } + @Test + void convertMarkdownRunsDeterministicConversionAndReturnsMdFile() throws IOException { + MockMultipartFile input = pdf("multi-column-test_lorem.pdf", "pdf-bytes"); + when(fileIdStrategy.idFor(any())).thenReturn("doc-1"); + stubOrchestrator( + """ + { + "outcome":"convert_markdown", + "reason":"PDF to Markdown requested.", + "filesToIngest":[{"id":"doc-1","name":"multi-column-test_lorem.pdf"}] + } + """); + when(toolMetadataService.shouldUnpackZipResponse("/api/v1/convert/pdf/markdown")) + .thenReturn(false); + stubEndpoint( + "/api/v1/convert/pdf/markdown", + pdfResource("# Title", "multi-column-test_lorem.md")); + AtomicInteger ids = stubFileStorage(); + + AiWorkflowResponse result = service.orchestrate(requestFor(input, "convert to markdown")); + + assertEquals(AiWorkflowOutcome.COMPLETED, result.getOutcome()); + assertEquals(1, result.getResultFiles().size()); + // Extension changes (pdf -> md), so the converter's response filename wins. + assertEquals("multi-column-test_lorem.md", result.getResultFiles().get(0).getFileName()); + assertEquals(1, ids.get()); + verify(internalApiClient, times(1)).post(eq("/api/v1/convert/pdf/markdown"), any()); + } + @Test void toolCallWithoutEndpointFallsBackToCannotContinue() throws IOException { MockMultipartFile input = pdf("input.pdf", "bytes"); diff --git a/app/proprietary/src/test/java/stirling/software/proprietary/service/PageLayoutArtifactContractTest.java b/app/proprietary/src/test/java/stirling/software/proprietary/service/PageLayoutArtifactContractTest.java deleted file mode 100644 index ae853b2e6a..0000000000 --- a/app/proprietary/src/test/java/stirling/software/proprietary/service/PageLayoutArtifactContractTest.java +++ /dev/null @@ -1,66 +0,0 @@ -package stirling.software.proprietary.service; - -import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertTrue; - -import java.util.List; - -import org.junit.jupiter.api.Test; - -import stirling.software.proprietary.service.PdfContentExtractor.LayoutFragment; -import stirling.software.proprietary.service.PdfContentExtractor.LayoutLine; -import stirling.software.proprietary.service.PdfContentExtractor.LayoutPage; -import stirling.software.proprietary.service.PdfContentExtractor.PageLayoutArtifact; -import stirling.software.proprietary.service.PdfContentExtractor.PageLayoutFileResult; - -import tools.jackson.databind.JsonNode; -import tools.jackson.databind.json.JsonMapper; - -/** - * Contract test: verifies that {@link PageLayoutArtifact} serializes to the JSON field names that - * the Python engine expects in {@code engine/src/stirling/contracts/pdf_to_markdown.py}. - * - *

The companion Python test in {@code tests/test_pdf_to_markdown.py} deserializes the same JSON - * literal and asserts field values. If either side renames a field, one of these tests fails. - */ -class PageLayoutArtifactContractTest { - - static final String CONTRACT_JSON = - """ - {"kind":"page_layout","files":[{"fileName":"test.pdf","pages":[{"pageNumber":1,"lines":[{"y":10.0,"fragments":[{"text":"Hello","x":1.0,"y":2.0,"width":30.0,"fontSize":12.0,"bold":true}]}]}]}]}"""; - - @Test - void pageLayoutArtifact_serialisesToExpectedJson() throws Exception { - LayoutFragment fragment = new LayoutFragment("Hello", 1.0f, 2.0f, 30.0f, 12.0f, true); - LayoutLine line = new LayoutLine(10.0f, List.of(fragment)); - LayoutPage page = new LayoutPage(1, List.of(line)); - - PageLayoutFileResult fileResult = new PageLayoutFileResult(); - fileResult.setFileName("test.pdf"); - fileResult.setPages(List.of(page)); - - PageLayoutArtifact artifact = new PageLayoutArtifact(); - artifact.setFiles(List.of(fileResult)); - - JsonNode json = new JsonMapper().valueToTree(artifact); - - assertEquals("page_layout", json.get("kind").asText()); - - JsonNode file = json.get("files").get(0); - assertEquals("test.pdf", file.get("fileName").asText()); - - JsonNode pg = file.get("pages").get(0); - assertEquals(1, pg.get("pageNumber").asInt()); - - JsonNode ln = pg.get("lines").get(0); - assertEquals(10.0, ln.get("y").asDouble(), 0.001); - - JsonNode frag = ln.get("fragments").get(0); - assertEquals("Hello", frag.get("text").asText()); - assertEquals(1.0, frag.get("x").asDouble(), 0.001); - assertEquals(2.0, frag.get("y").asDouble(), 0.001); - assertEquals(30.0, frag.get("width").asDouble(), 0.001); - assertEquals(12.0, frag.get("fontSize").asDouble(), 0.001); - assertTrue(frag.get("bold").asBoolean()); - } -} diff --git a/engine/src/stirling/agents/__init__.py b/engine/src/stirling/agents/__init__.py index cddd0275c3..5410ac098a 100644 --- a/engine/src/stirling/agents/__init__.py +++ b/engine/src/stirling/agents/__init__.py @@ -5,7 +5,6 @@ from .orchestrator import OrchestratorAgent from .pdf_edit import PdfEditAgent, PdfEditParameterSelector, PdfEditPlanSelection from .pdf_questions import PdfQuestionAgent from .pdf_review import PdfReviewAgent -from .pdf_to_markdown import PdfToMarkdownAgent from .user_spec import UserSpecAgent __all__ = [ @@ -16,6 +15,5 @@ __all__ = [ "PdfEditPlanSelection", "PdfQuestionAgent", "PdfReviewAgent", - "PdfToMarkdownAgent", "UserSpecAgent", ] diff --git a/engine/src/stirling/agents/orchestrator.py b/engine/src/stirling/agents/orchestrator.py index 4dbf0b65ab..d2a0b4a19b 100644 --- a/engine/src/stirling/agents/orchestrator.py +++ b/engine/src/stirling/agents/orchestrator.py @@ -11,14 +11,13 @@ from pydantic_ai.tools import RunContext from stirling.agents.pdf_edit import PdfEditAgent from stirling.agents.pdf_questions import PdfQuestionAgent from stirling.agents.pdf_review import PdfReviewAgent -from stirling.agents.pdf_to_markdown import PdfToMarkdownAgent from stirling.agents.user_spec import UserSpecAgent from stirling.contracts import ( AgentDraftWorkflowResponse, + ConvertMarkdownResponse, ExtractedTextArtifact, OrchestratorRequest, OrchestratorResponse, - PageLayoutArtifact, PdfEditResponse, PdfQuestionOrchestrateResponse, PdfReviewOrchestrateResponse, @@ -27,7 +26,6 @@ from stirling.contracts import ( format_conversation_history, format_file_names, ) -from stirling.contracts.pdf_to_markdown import PdfToMarkdownOrchestrateResponse from stirling.services import AppRuntime logger = logging.getLogger(__name__) @@ -72,9 +70,11 @@ class OrchestratorAgent: ), ), ToolOutput( - self.delegate_pdf_to_markdown, - name="delegate_pdf_to_markdown", - description=("Delegate requests to reconstruct a PDF as a Markdown document."), + self.delegate_pdf_ingest, + name="delegate_pdf_ingest", + description=( + "Delegate requests to convert a PDF to Markdown or extract its content as readable text." + ), ), ToolOutput( self.unsupported_capability, @@ -92,8 +92,8 @@ class OrchestratorAgent: "Use delegate_pdf_review when the user wants the PDF returned with review" " comments attached — anything like 'review this', 'annotate with comments'," " 'leave feedback on the PDF'. " - "Use delegate_pdf_to_markdown for any request to convert a PDF to Markdown " - "or reconstruct its content as readable text. " + "Use delegate_pdf_ingest for any request to convert a PDF to Markdown " + "or extract its content as readable text. " "Use unsupported_capability when the user asks about the assistant itself " "or when none of the other outputs fit; supply a helpful message." ), @@ -133,13 +133,12 @@ class OrchestratorAgent: return await self._run_pdf_edit(request) case SupportedCapability.AGENT_DRAFT: return await self._run_agent_draft(request) - case SupportedCapability.PDF_TO_MARKDOWN: - return await self._run_pdf_to_markdown(request) case ( SupportedCapability.ORCHESTRATE | SupportedCapability.AGENT_REVISE | SupportedCapability.AGENT_NEXT_ACTION | SupportedCapability.MATH_AUDITOR_AGENT + | SupportedCapability.PDF_TO_MARKDOWN ): raise ValueError(f"Cannot resume orchestrator with capability: {capability}") case _ as unreachable: @@ -163,11 +162,12 @@ class OrchestratorAgent: async def _run_agent_draft(self, request: OrchestratorRequest) -> AgentDraftWorkflowResponse: return await UserSpecAgent(self.runtime).orchestrate(request) - async def delegate_pdf_to_markdown(self, ctx: RunContext[OrchestratorDeps]) -> PdfToMarkdownOrchestrateResponse: - return await self._run_pdf_to_markdown(ctx.deps.request) - - async def _run_pdf_to_markdown(self, request: OrchestratorRequest) -> PdfToMarkdownOrchestrateResponse: - return await PdfToMarkdownAgent(self.runtime).orchestrate(request) + async def delegate_pdf_ingest(self, ctx: RunContext[OrchestratorDeps]) -> ConvertMarkdownResponse: + request = ctx.deps.request + return ConvertMarkdownResponse( + reason="PDF to Markdown requested — Java converts deterministically.", + files_to_ingest=request.files, + ) async def delegate_pdf_review(self, ctx: RunContext[OrchestratorDeps]) -> PdfReviewOrchestrateResponse: return await self._run_pdf_review(ctx.deps.request) @@ -204,10 +204,5 @@ class OrchestratorAgent: file_names = [f.file_name for f in artifact.files] descriptions.append(f"- extracted_text: {total_pages} pages from {file_names}") continue - if isinstance(artifact, PageLayoutArtifact): - total_pages = sum(len(f.pages) for f in artifact.files) - file_names = [f.file_name for f in artifact.files] - descriptions.append(f"- page_layout: {total_pages} pages from {file_names}") - continue descriptions.append("- unknown artifact") return "\n".join(descriptions) diff --git a/engine/src/stirling/agents/pdf_to_markdown/__init__.py b/engine/src/stirling/agents/pdf_to_markdown/__init__.py deleted file mode 100644 index d35ae05c7c..0000000000 --- a/engine/src/stirling/agents/pdf_to_markdown/__init__.py +++ /dev/null @@ -1,3 +0,0 @@ -from .agent import PdfToMarkdownAgent - -__all__ = ["PdfToMarkdownAgent"] diff --git a/engine/src/stirling/agents/pdf_to_markdown/agent.py b/engine/src/stirling/agents/pdf_to_markdown/agent.py deleted file mode 100644 index 8c0d7d8ee5..0000000000 --- a/engine/src/stirling/agents/pdf_to_markdown/agent.py +++ /dev/null @@ -1,435 +0,0 @@ -"""PDF to Markdown Agent. - -Converts a parsed PDF document into a single clean Markdown document, preserving -headings, paragraphs, and tables in reading order. -""" - -from __future__ import annotations - -import asyncio -import logging -import re -import time - -from pydantic import BaseModel, Field -from pydantic_ai import Agent -from pydantic_ai.output import NativeOutput - -from stirling.contracts import ( - EditCannotDoResponse, - GenerateFileResponse, - NeedContentFileRequest, - NeedContentResponse, - OrchestratorRequest, - PdfContentType, - SupportedCapability, - format_conversation_history, -) -from stirling.contracts.pdf_to_markdown import ( - PageLayout, - PageLayoutArtifact, - PdfToMarkdownCannotDoResponse, - PdfToMarkdownOrchestrateResponse, - PdfToMarkdownRequest, - PdfToMarkdownResponse, - PdfToMarkdownSuccessResponse, -) -from stirling.services import AppRuntime - -logger = logging.getLogger(__name__) - - -# Warn when output tokens are close to the typical model output limit (~8192 for most -# configurations). The actual limit is model-specific; this threshold catches likely truncation. -_OUTPUT_TOKEN_TRUNCATION_THRESHOLD = 7500 - -# Chunking limits — keep each LLM call to a manageable payload size. -# Fragment count is the primary driver of JSON payload size (each fragment carries x/y/width/ -# fontSize/bold metadata beyond its text). Page cap prevents low-text pages accumulating. -_MAX_CHUNK_FRAGMENTS = 1_000 -_MAX_CHUNK_PAGES = 10 - -# Max concurrent LLM calls — limits API rate pressure on large documents. -_MAX_PARALLEL_CHUNKS = 3 - -# ── LLM output model ──────────────────────────────────────────────────────────────────────────── - - -class _ReconstructionOutput(BaseModel): - markdown: str = Field(description="Full document reconstructed as clean Markdown.") - - -# ── Agent ──────────────────────────────────────────────────────────────────────────────────────── - - -class PdfToMarkdownAgent: - def __init__(self, runtime: AppRuntime) -> None: - self.runtime = runtime - self._sem = asyncio.Semaphore(_MAX_PARALLEL_CHUNKS) - self._reconstruct_agent = Agent( - model=runtime.smart_model, - output_type=NativeOutput(_ReconstructionOutput), - system_prompt=( - "You reconstruct PDF pages into clean Markdown from spatial fragment data.\n" - "Input: PAGE LAYOUT — per-fragment x/y/font data for structural analysis.\n\n" - "COLUMN DETECTION (for tables in page_layout):\n" - "- Look at the x-positions of fragments across 3+ consecutive lines.\n" - "- If fragments cluster at the same x-positions across multiple lines, those are table columns.\n" - "- Each distinct x-cluster is one column." - " Name them from the header row (the first line in the cluster).\n" - "- Do NOT merge values from different x-columns into one cell.\n\n" - "ROW DETECTION:\n" - "- Each unique y-coordinate (or group within 3pt) is one table row.\n" - "- Every line of layout data is its own row — do not merge rows.\n" - "- If a column has no fragment on a given y-row, that cell is empty.\n\n" - "TABLE RENDERING:\n" - "- Render as: | col1 | col2 | col3 |\n" - " | --- | --- | --- |\n" - " | val | val | val |\n" - "- One source row = one table row. Never collapse multiple rows into one.\n" - "- Preserve numeric values exactly (no rounding, no formatting changes).\n" - "- Bold cells: wrap with ** in the Markdown cell.\n" - "- CRITICAL: the separator row `| --- | --- |` appears EXACTLY ONCE per table, immediately\n" - " after the header row. NEVER put `| --- |` after a data row or between data rows.\n" - " NEVER put a blank line inside a table. All rows (header + data) must be consecutive.\n" - "- Do NOT produce a header-only table followed by a second table with the data rows.\n" - " One logical table = one markdown table block, with header, one separator, then all data.\n\n" - "GROUP HEADERS (label-only rows inside a table):\n" - "- A row is a group header when: the first column has text AND every numeric column is empty.\n" - "- Do NOT render group headers as table rows with empty cells.\n" - "- Break the table, emit the label as **bold text** on its own line," - " then start a new table for the rows that follow.\n" - "- Example labels: 'Policy functions', 'Non-current assets'.\n\n" - "TOTAL AND SUBTOTAL ROWS:\n" - "- Detect rows whose first cell contains (case-insensitive):" - " total, subtotal, surplus, balance, net, sum.\n" - "- These rows have numeric content — they are NOT group headers.\n" - "- Render the entire row in bold: | **Total income** | **1,234** | **5,678** |\n" - "- Keep total rows attached to the group they summarise.\n\n" - "MULTI-LEVEL TABLES (year or period as a row label):\n" - "- Detect when a row contains only a single label (a year like '2010' or period like 'Q1 2023')" - " with no numeric content, followed by repeated metric rows.\n" - "- Do NOT render the year as a table row.\n" - "- Normalise: add 'Year' as the first column, 'Metric' as the second," - " and repeat the year value on each metric row.\n\n" - "PROSE REGIONS:\n" - "- Lines where x-positions vary across lines (not repeating columns) are prose.\n" - "- Merge lines at the same x-level into paragraphs. Separate indented lines.\n\n" - "HEADINGS:\n" - "- A line is a heading when it is bold OR font_size ≥2pt above body.\n" - " CRITICAL EXCEPTION: a bold fragment is a TABLE HEADER CELL, not a document heading, when\n" - " the same y-row in page_layout contains other fragments at different x-positions.\n" - " Only classify a bold line as a document heading when it is the SOLE fragment on its y-row.\n" - " Example: 'Non-current assets' at y=120 with '2010'@x=350, '2009'@x=420, '2008'@x=490\n" - " → this is a table header row, NOT a heading. Render it as the first cell of the table.\n" - "- Use ## for section headings, ### for sub-headings. Use # only for the document title.\n\n" - "ORDERING:\n" - "- Process content top-to-bottom as it appears on the page.\n" - "- Interleave prose blocks and table blocks in page order.\n" - "- Do not move text that appears before a table to after it, or vice versa.\n\n" - "FIDELITY:\n" - "- Do NOT invent, summarise, or omit any content.\n" - "- Do NOT add commentary, metadata, or JSON — output Markdown only." - ), - model_settings={ - **runtime.smart_model_settings, - "temperature": 0.0, - "max_tokens": _OUTPUT_TOKEN_TRUNCATION_THRESHOLD, - }, - ) - - async def orchestrate(self, request: OrchestratorRequest) -> PdfToMarkdownOrchestrateResponse: - """Entry point for the orchestrator delegate. - - First turn: requests PAGE_LAYOUT extraction from Java via NeedContentResponse. - Resume turn: runs the LLM reconstruction and returns a write-file plan step. - """ - layout_artifact = next( - (a for a in request.artifacts if isinstance(a, PageLayoutArtifact)), - None, - ) - if layout_artifact is None: - return NeedContentResponse( - resume_with=SupportedCapability.PDF_TO_MARKDOWN, - reason="Page layout data is required to reconstruct the document.", - files=[ - NeedContentFileRequest(file=f, content_types=[PdfContentType.PAGE_LAYOUT]) for f in request.files - ], - max_pages=self.runtime.settings.max_pages, - max_characters=self.runtime.settings.max_characters, - ) - - page_layout = [page for entry in layout_artifact.files for page in entry.pages] - file_names = [f.name for f in request.files] - result = await self.handle( - PdfToMarkdownRequest( - user_message=request.user_message, - file_names=file_names, - conversation_history=request.conversation_history, - page_layout=page_layout, - ) - ) - if isinstance(result, PdfToMarkdownCannotDoResponse): - return EditCannotDoResponse(reason=result.reason) - - base = file_names[0].rsplit(".", 1)[0] if file_names else "document" - return GenerateFileResponse( - content=result.markdown, - filename=f"{base}-reconstruction.md", - summary="Reconstructed the document as a Markdown file.", - ) - - async def handle(self, request: PdfToMarkdownRequest) -> PdfToMarkdownResponse: - total_fragments = sum(len(line.fragments) for page in request.page_layout for line in page.lines) - logger.info( - "[pdf-to-markdown] received layout-pages=%d fragments=%d", - len(request.page_layout), - total_fragments, - ) - - if not request.page_layout: - logger.warning("[pdf-to-markdown] no content extracted from document; returning cannot_do") - return PdfToMarkdownCannotDoResponse( - reason=( - "No content was extracted from the document. " - "The file may be a scanned image PDF with no readable text. " - "Try running OCR on the document first." - ) - ) - - chunks = _build_page_chunks(request.page_layout) - logger.info("[pdf-to-markdown] chunks=%d (max %d in parallel)", len(chunks), _MAX_PARALLEL_CHUNKS) - - if len(chunks) == 1: - return await self._reconstruct_chunk(request, chunks[0], chunk_num=1, total_chunks=1) - - total = len(chunks) - results = await asyncio.gather( - *( - self._reconstruct_chunk(request, chunk, chunk_num=i + 1, total_chunks=total) - for i, chunk in enumerate(chunks) - ) - ) - - markdown_parts: list[str] = [] - for result in results: - if isinstance(result, PdfToMarkdownSuccessResponse) and result.markdown: - markdown_parts.append(result.markdown) - elif isinstance(result, PdfToMarkdownCannotDoResponse): - logger.warning("[pdf-to-markdown] chunk dropped: %s", result.reason) - - if not markdown_parts: - return PdfToMarkdownCannotDoResponse(reason="The document could not be reconstructed. All chunks failed.") - - logger.info("[pdf-to-markdown] assembly: %d/%d chunks produced output", len(markdown_parts), len(chunks)) - return PdfToMarkdownSuccessResponse(markdown="\n\n".join(markdown_parts)) - - async def _reconstruct_chunk( - self, - request: PdfToMarkdownRequest, - pages: list[PageLayout], - chunk_num: int, - total_chunks: int, - ) -> PdfToMarkdownResponse: - chunk_request = PdfToMarkdownRequest( - user_message=request.user_message, - file_names=request.file_names, - conversation_history=request.conversation_history, - page_layout=pages, - ) - try: - async with self._sem: - return await self._reconstruct_document(chunk_request, chunk_num, total_chunks) - except Exception as e: - logger.error("[pdf-to-markdown] chunk %d/%d failed: %s", chunk_num, total_chunks, e, exc_info=True) - return PdfToMarkdownCannotDoResponse( - reason="The document could not be reconstructed. The AI model failed to process it." - ) - - async def _reconstruct_document( - self, request: PdfToMarkdownRequest, chunk_num: int = 1, total_chunks: int = 1 - ) -> PdfToMarkdownSuccessResponse: - content = _build_reconstruction_prompt(request) - logger.info("[timing] chunk %d/%d llm-call prompt-chars=%d", chunk_num, total_chunks, len(content)) - t0 = time.monotonic() - result = await self._reconstruct_agent.run([content]) - llm_ms = int((time.monotonic() - t0) * 1000) - output: _ReconstructionOutput = result.output - usage = result.usage() - logger.info( - "[timing] chunk %d/%d llm-done ms=%d input-tokens=%s output-tokens=%s markdown-chars=%d", - chunk_num, - total_chunks, - llm_ms, - usage.input_tokens, - usage.output_tokens, - len(output.markdown), - ) - if usage.output_tokens and usage.output_tokens >= _OUTPUT_TOKEN_TRUNCATION_THRESHOLD: - logger.warning( - "[timing] chunk %d/%d output likely truncated (output-tokens=%d)", - chunk_num, - total_chunks, - usage.output_tokens, - ) - markdown = _remove_extra_separators(_fix_markdown_tables(_merge_orphaned_table_rows(output.markdown))) - return PdfToMarkdownSuccessResponse(markdown=markdown) - - -# ── Chunking ──────────────────────────────────────────────────────────────────────────────────── - - -def _build_page_chunks(pages: list[PageLayout]) -> list[list[PageLayout]]: - chunks: list[list[PageLayout]] = [] - current: list[PageLayout] = [] - current_fragments = 0 - for page in pages: - page_fragments = sum(len(line.fragments) for line in page.lines) - fragment_full = current and current_fragments + page_fragments > _MAX_CHUNK_FRAGMENTS - page_full = len(current) >= _MAX_CHUNK_PAGES - if fragment_full or page_full: - chunks.append(current) - current = [] - current_fragments = 0 - current.append(page) - current_fragments += page_fragments - if current: - chunks.append(current) - return chunks - - -# ── Prompt builders (module-level, no state) ──────────────────────────────────────────────────── - - -def _build_reconstruction_prompt(request: PdfToMarkdownRequest) -> str: - history = format_conversation_history(request.conversation_history) - file_names = ", ".join(request.file_names) if request.file_names else "Unknown files" - layout_section = _format_layout(request.page_layout) - - return ( - f"Files: {file_names}\n\n" - f"User request: {request.user_message}\n\n" - f"Conversation history:\n{history}\n\n" - "PAGE LAYOUT (structural source — x/y fragment positions):\n" - "Each line is: y=NNN | text@(x,y) fs=N text@(x,y) fs=N ...\n" - "- y=NNN is the vertical position (row). Lines close in y are the same visual row.\n" - "- x=NNN is the horizontal position (column). Consistent x across rows = a column.\n" - "- fs=N is font size. Larger = likely a heading.\n" - "- **bold** markers indicate bold text.\n\n" - f"{layout_section}" - ) - - -# ── LLM output post-processing ────────────────────────────────────────────────────────────────── - - -def _fix_markdown_tables(markdown: str) -> str: - """Remove blank lines between table rows produced by the LLM.""" - lines = markdown.split("\n") - result: list[str] = [] - i = 0 - while i < len(lines): - result.append(lines[i]) - if lines[i].strip().startswith("|"): - j = i + 1 - while j < len(lines) and lines[j].strip() == "": - j += 1 - if j < len(lines) and lines[j].strip().startswith("|"): - i = j - continue - i += 1 - return "\n".join(result) - - -_SEP_CELL = re.compile(r"^:?-+:?$") - - -def _is_sep_row(line: str) -> bool: - """Return True when a pipe row is a Markdown table separator (| --- | --- |).""" - stripped = line.strip() - if not stripped.startswith("|"): - return False - cells = [c.strip() for c in stripped.split("|") if c.strip()] - return bool(cells) and all(_SEP_CELL.match(c) for c in cells) - - -def _merge_orphaned_table_rows(markdown: str) -> str: - """Merge pipe-row blocks that lack a separator into the preceding table. - - When the LLM incorrectly breaks a table (e.g. on a false group-header), it emits - orphaned pipe rows with no header or separator. These are invalid markdown and get - merged back into the preceding table, discarding the intervening non-table content. - """ - lines = markdown.split("\n") - - segments: list[tuple[str, list[str]]] = [] - i = 0 - while i < len(lines): - if lines[i].strip().startswith("|"): - block: list[str] = [] - while i < len(lines) and lines[i].strip().startswith("|"): - block.append(lines[i]) - i += 1 - has_sep = any(_is_sep_row(row) for row in block) - segments.append(("table" if has_sep else "orphan", block)) - else: - block = [] - while i < len(lines) and not lines[i].strip().startswith("|"): - block.append(lines[i]) - i += 1 - segments.append(("prose", block)) - - result: list[tuple[str, list[str]]] = [] - last_table_idx: int | None = None - for seg_type, seg_lines in segments: - if seg_type == "orphan": - if last_table_idx is not None: - result = result[: last_table_idx + 1] - result[-1] = ("table", result[-1][1] + seg_lines) - else: - result.append((seg_type, seg_lines)) - else: - if seg_type == "table": - last_table_idx = len(result) - result.append((seg_type, seg_lines)) - - return "\n".join(line for _, seg_lines in result for line in seg_lines) - - -def _remove_extra_separators(markdown: str) -> str: - """Within each contiguous table block, keep only the first separator row.""" - lines = markdown.split("\n") - result: list[str] = [] - seen_sep = False - - for line in lines: - if not line.strip().startswith("|"): - seen_sep = False - result.append(line) - continue - if _is_sep_row(line): - if seen_sep: - continue - seen_sep = True - result.append(line) - - return "\n".join(result) - - -# ── Formatting helpers (module-level, no state) ────────────────────────────────────────────────── - - -def _format_layout(pages: list[PageLayout]) -> str: - if not pages: - return "None" - parts: list[str] = [] - for page in pages: - line_strs: list[str] = [] - for line in page.lines: - frags = " ".join( - f"{'**' if f.bold else ''}{f.text}{'**' if f.bold else ''}@({f.x:.0f},{f.y:.0f}) fs={f.font_size:.0f}" - for f in line.fragments - ) - line_strs.append(f"y={line.y:.0f} | {frags}") - parts.append(f"--- Page {page.page_number} ---\n" + "\n".join(line_strs)) - return "\n\n".join(parts) diff --git a/engine/src/stirling/contracts/__init__.py b/engine/src/stirling/contracts/__init__.py index 696749d7d7..4bc4febcf5 100644 --- a/engine/src/stirling/contracts/__init__.py +++ b/engine/src/stirling/contracts/__init__.py @@ -13,6 +13,7 @@ from .common import ( AiFile, ArtifactKind, ConversationMessage, + ConvertMarkdownResponse, ExtractedFileText, GenerateFileResponse, MathAuditorToolReportArtifact, @@ -96,17 +97,6 @@ from .pdf_questions import ( PdfQuestionTerminalResponse, ) from .pdf_review import PdfReviewOrchestrateResponse -from .pdf_to_markdown import ( - LayoutFragment, - LayoutLine, - PageLayout, - PageLayoutArtifact, - PageLayoutFileEntry, - PdfToMarkdownCannotDoResponse, - PdfToMarkdownOrchestrateResponse, - PdfToMarkdownRequest, - PdfToMarkdownResponse, -) from .progress import ( ProgressEvent, WholeDocCompressionRound, @@ -139,10 +129,6 @@ __all__ = [ "ConversationMessage", "DeleteDocumentResponse", "PurgeOwnerResponse", - "PdfToMarkdownCannotDoResponse", - "PdfToMarkdownOrchestrateResponse", - "PdfToMarkdownRequest", - "PdfToMarkdownResponse", "Discrepancy", "DiscrepancyKind", "EditCannotDoResponse", @@ -166,15 +152,11 @@ __all__ = [ "NeedContentFileRequest", "NeedContentResponse", "NeedIngestResponse", + "ConvertMarkdownResponse", "NextExecutionAction", "OrchestratorRequest", "OrchestratorResponse", - "LayoutFragment", - "LayoutLine", "Page", - "PageLayout", - "PageLayoutArtifact", - "PageLayoutFileEntry", "PageRange", "PageText", "PdfCommentInstruction", diff --git a/engine/src/stirling/contracts/common.py b/engine/src/stirling/contracts/common.py index 05103b1a4a..b8030c58b6 100644 --- a/engine/src/stirling/contracts/common.py +++ b/engine/src/stirling/contracts/common.py @@ -62,6 +62,7 @@ class WorkflowOutcome(StrEnum): CANNOT_CONTINUE = "cannot_continue" UNSUPPORTED_CAPABILITY = "unsupported_capability" GENERATE_FILE = "generate_file" + CONVERT_MARKDOWN = "convert_markdown" class ArtifactKind(StrEnum): @@ -183,6 +184,19 @@ class NeedIngestResponse(ApiModel): content_types: list[PdfContentType] = Field(default_factory=list) +class ConvertMarkdownResponse(ApiModel): + """Terminal signal: convert the listed files to Markdown deterministically. + + This is a deterministic, non-AI conversion. Java runs the PDF→Markdown converter + (``PdfMarkdownConverter``) on each file and returns the resulting ``.md`` file(s) as a + completed result. There is no resume turn — the conversion output is the final answer. + """ + + outcome: Literal[WorkflowOutcome.CONVERT_MARKDOWN] = WorkflowOutcome.CONVERT_MARKDOWN + reason: str + files_to_ingest: list[AiFile] + + class ToolOperationStep(ApiModel): kind: Literal[StepKind.TOOL] = StepKind.TOOL tool: AnyToolId diff --git a/engine/src/stirling/contracts/orchestrator.py b/engine/src/stirling/contracts/orchestrator.py index 1bf0f6eb36..8b916ccaff 100644 --- a/engine/src/stirling/contracts/orchestrator.py +++ b/engine/src/stirling/contracts/orchestrator.py @@ -11,6 +11,7 @@ from .common import ( AiFile, ArtifactKind, ConversationMessage, + ConvertMarkdownResponse, ExtractedFileText, GenerateFileResponse, NeedContentResponse, @@ -23,7 +24,6 @@ from .common import ( from .execution import NextExecutionAction from .pdf_edit import PdfEditTerminalResponse from .pdf_questions import PdfQuestionTerminalResponse -from .pdf_to_markdown import PageLayoutArtifact class ExtractedTextArtifact(ApiModel): @@ -32,7 +32,7 @@ class ExtractedTextArtifact(ApiModel): WorkflowArtifact = Annotated[ - ExtractedTextArtifact | PageLayoutArtifact | ToolReportArtifact, + ExtractedTextArtifact | ToolReportArtifact, Field(discriminator="kind"), ] @@ -61,6 +61,7 @@ type OrchestratorResponse = Annotated[ | GenerateFileResponse | NeedContentResponse | NeedIngestResponse + | ConvertMarkdownResponse | AgentDraftResponse | NextExecutionAction | UnsupportedCapabilityResponse, diff --git a/engine/src/stirling/contracts/pdf_to_markdown.py b/engine/src/stirling/contracts/pdf_to_markdown.py deleted file mode 100644 index 4d272e6e2a..0000000000 --- a/engine/src/stirling/contracts/pdf_to_markdown.py +++ /dev/null @@ -1,105 +0,0 @@ -"""Contracts for the PDF to Markdown Agent. - -The agent accepts a parsed document and returns a single Markdown document that -faithfully reconstructs the PDF content — headings, paragraphs, and tables in -reading order, using page_layout as the primary source of truth for structure. - -Java extracts page layout via PdfIngester and returns it as a PageLayoutArtifact -through the orchestrator resume_with pattern. -""" - -from __future__ import annotations - -from typing import Annotated, Literal - -from pydantic import Field - -from stirling.models import ApiModel - -from .common import ArtifactKind, ConversationMessage, GenerateFileResponse, NeedContentResponse -from .pdf_edit import EditCannotDoResponse - -# ── Input: layout models (mirror Java's RawLine / TextFragment geometry) ──────────────────────── - - -class LayoutFragment(ApiModel): - """One text fragment with its bounding-box geometry and font properties.""" - - text: str - x: float - y: float - width: float - font_size: float - bold: bool - - -class LayoutLine(ApiModel): - """A visual line on the page: one y-coordinate and all fragments on that line.""" - - y: float - fragments: list[LayoutFragment] - - -class PageLayout(ApiModel): - """All layout lines for a single page, in top-to-bottom order.""" - - page_number: int - lines: list[LayoutLine] - - -# ── Artifact: page layout (produced by Java, consumed by orchestrate()) ────────────────────────── - - -class PageLayoutFileEntry(ApiModel): - """Page layout data for one file, as extracted by Java's PdfIngester.""" - - file_name: str - pages: list[PageLayout] = Field(default_factory=list) - - -class PageLayoutArtifact(ApiModel): - """Artifact carrying full spatial page layout for all input files.""" - - kind: Literal[ArtifactKind.PAGE_LAYOUT] = ArtifactKind.PAGE_LAYOUT - files: list[PageLayoutFileEntry] = Field(default_factory=list) - - -# ── Input: full request ────────────────────────────────────────────────────────────────────────── - - -class PdfToMarkdownRequest(ApiModel): - """Request sent by Java after PdfIngester has parsed the document. - - page_layout: per-fragment positional data from the original (y-sorted) line order. - Each fragment carries its x/y position, width, font size, and bold flag. - This is the primary source of truth for column detection and heading hierarchy. - """ - - user_message: str - file_names: list[str] = Field(default_factory=list) - conversation_history: list[ConversationMessage] = Field(default_factory=list) - page_layout: list[PageLayout] = Field(default_factory=list) - - -# ── Output: response variants ──────────────────────────────────────────────────────────────────── - - -class PdfToMarkdownSuccessResponse(ApiModel): - outcome: Literal["document_reconstructed"] = "document_reconstructed" - markdown: str - - -class PdfToMarkdownCannotDoResponse(ApiModel): - outcome: Literal["cannot_do"] = "cannot_do" - reason: str - - -type PdfToMarkdownResponse = Annotated[ - PdfToMarkdownSuccessResponse | PdfToMarkdownCannotDoResponse, - Field(discriminator="outcome"), -] - -type PdfToMarkdownOrchestrateResponse = Annotated[ - GenerateFileResponse | EditCannotDoResponse | NeedContentResponse, - Field(discriminator="outcome"), -] diff --git a/engine/tests/test_pdf_to_markdown.py b/engine/tests/test_pdf_to_markdown.py deleted file mode 100644 index 32870a9459..0000000000 --- a/engine/tests/test_pdf_to_markdown.py +++ /dev/null @@ -1,138 +0,0 @@ -"""Tests for PDF to Markdown agent. - -Two cases: -1. Narrative-only page: request validates and routes to reconstruction. -2. Mixed text + table page: layout with table region validates correctly. -""" - -from __future__ import annotations - -from stirling.contracts.pdf_to_markdown import ( - LayoutFragment, - LayoutLine, - PageLayout, - PageLayoutArtifact, - PdfToMarkdownRequest, - PdfToMarkdownSuccessResponse, -) - - -def _frag(text: str, x: float, y: float, font_size: float = 10.0, bold: bool = False) -> LayoutFragment: - return LayoutFragment(text=text, x=x, y=y, width=float(len(text) * 6), font_size=font_size, bold=bold) - - -def _line(y: float, *frags: LayoutFragment) -> LayoutLine: - return LayoutLine(y=y, fragments=list(frags)) - - -# ── Test 1: Narrative-only reconstruction ──────────────────────────────────────────────────────── - - -# ── Contract test: Java serialization ↔ Python deserialization ────────────────────────────────── -# This JSON is also asserted field-by-field in PageLayoutArtifactContractTest.java. -# If either side renames a field, one of these tests fails. -_CONTRACT_JSON = ( - '{"kind":"page_layout","files":[{"fileName":"test.pdf","pages":' - '[{"pageNumber":1,"lines":[{"y":10.0,"fragments":' - '[{"text":"Hello","x":1.0,"y":2.0,"width":30.0,"fontSize":12.0,"bold":true}]}]}]}]}' -) - - -def test_page_layout_artifact_deserialises_java_json() -> None: - artifact = PageLayoutArtifact.model_validate_json(_CONTRACT_JSON) - - assert artifact.kind == "page_layout" - assert artifact.files[0].file_name == "test.pdf" - page = artifact.files[0].pages[0] - assert page.page_number == 1 - line = page.lines[0] - assert line.y == 10.0 - frag = line.fragments[0] - assert frag.text == "Hello" - assert frag.x == 1.0 - assert frag.y == 2.0 - assert frag.width == 30.0 - assert frag.font_size == 12.0 - assert frag.bold is True - - -def test_narrative_reconstruction_request_validates() -> None: - """A prose-only page with no tables produces a valid PdfToMarkdownRequest.""" - layout = PageLayout( - page_number=1, - lines=[ - _line(72.0, _frag("Annual Report 2023", x=72.0, y=72.0, font_size=18.0, bold=True)), - _line(100.0, _frag("Our revenue grew significantly", x=72.0, y=100.0)), - _line(114.0, _frag("during the fiscal year ended", x=72.0, y=114.0)), - _line(128.0, _frag("December 31, 2023.", x=72.0, y=128.0)), - ], - ) - request = PdfToMarkdownRequest( - user_message="reconstruct this document", - page_layout=[layout], - ) - - assert len(request.page_layout) == 1 - assert len(request.page_layout[0].lines) == 4 - assert request.page_layout[0].lines[0].fragments[0].bold is True - assert request.page_layout[0].lines[0].fragments[0].font_size == 18.0 - - -def test_narrative_reconstruction_response_validates() -> None: - """PdfToMarkdownSuccessResponse accepts markdown and returns document_reconstructed outcome.""" - response = PdfToMarkdownSuccessResponse( - markdown="# Annual Report 2023\n\nOur revenue grew significantly during the fiscal year.", - ) - - assert response.outcome == "document_reconstructed" - assert response.markdown.startswith("#") - - -# ── Test 2: Mixed text + table reconstruction ───────────────────────────────────────────────────── - - -def test_mixed_page_layout_validates() -> None: - """A page with both prose lines and a table region produces a valid request.""" - layout = PageLayout( - page_number=1, - lines=[ - # Prose heading - _line(50.0, _frag("Projects in Development", x=72.0, y=50.0, font_size=14.0, bold=True)), - # Table header row - _line( - 80.0, - _frag("Project Name", x=72.0, y=80.0, bold=True), - _frag("Location", x=200.0, y=80.0, bold=True), - _frag("Size (MW)", x=290.0, y=80.0, bold=True), - ), - # Table data rows - _line( - 95.0, - _frag("Chaplin Wind 1", x=72.0, y=95.0), - _frag("Saskatchewan", x=200.0, y=95.0), - _frag("177", x=290.0, y=95.0), - ), - _line( - 110.0, - _frag("Amherst Island 2", x=72.0, y=110.0), - _frag("Ontario", x=200.0, y=110.0), - _frag("75", x=290.0, y=110.0), - ), - # Prose after table - _line(140.0, _frag("Notes:", x=72.0, y=140.0, bold=True)), - _line(154.0, _frag("1 PPA signed", x=85.0, y=154.0)), - ], - ) - request = PdfToMarkdownRequest( - user_message="markdown", - page_layout=[layout], - ) - - assert len(request.page_layout[0].lines) == 6 - # Header line has 3 fragments at distinct x-positions (column detection) - header_line = request.page_layout[0].lines[1] - xs = [f.x for f in header_line.fragments] - assert xs == [72.0, 200.0, 290.0] - # Data rows have matching x-positions - data_row = request.page_layout[0].lines[2] - assert [f.x for f in data_row.fragments] == [72.0, 200.0, 290.0] diff --git a/testing/cucumber/features/external.feature b/testing/cucumber/features/external.feature index bf392d7206..e8e8d85272 100644 --- a/testing/cucumber/features/external.feature +++ b/testing/cucumber/features/external.feature @@ -233,8 +233,7 @@ Feature: API Validation When I send the API request to the endpoint "/api/v1/convert/pdf/markdown" Then the response status code should be 200 And the response file should have size greater than 100 - And the response file should have extension ".zip" - And the response ZIP should contain 4 files + And the response file should have extension ".md" @positive @pdftocsv diff --git a/testing/test.sh b/testing/test.sh index 0aa30294a2..6c7c62c4ab 100644 --- a/testing/test.sh +++ b/testing/test.sh @@ -911,7 +911,7 @@ main() { CUCUMBER_JUNIT_DIR="$PROJECT_ROOT/testing/cucumber/junit" mkdir -p "$CUCUMBER_JUNIT_DIR" cd "testing/cucumber" - start_test_timer "Stirling-PDF-Regression" + start_test_timer "Stirling-PDF-Regression $CONTAINER_NAME" # Snapshot docker log line count before behave so we can extract only behave-window logs DOCKER_LOG_BEFORE=$(docker logs "$CONTAINER_NAME" 2>&1 | wc -l) @@ -999,7 +999,7 @@ main() { # Save docker logs from the behave window to a dedicated file local cucumber_log="$REPORT_DIR/cucumber-docker-context.log" docker logs "$CONTAINER_NAME" 2>&1 | tail -n +"$((DOCKER_LOG_BEFORE + 1))" > "$cucumber_log" 2>/dev/null || true - test_failure_logs["Stirling-PDF-Regression"]="$cucumber_log" + test_failure_logs["Stirling-PDF-Regression $CONTAINER_NAME"]="$cucumber_log" gha_group "Docker logs during behave run: $CONTAINER_NAME" tail -100 "$cucumber_log" @@ -1012,7 +1012,7 @@ main() { capture_file_list "$CONTAINER_NAME" "$AFTER_FILE" compare_file_lists "$BEFORE_FILE" "$AFTER_FILE" "$DIFF_FILE" "$CONTAINER_NAME" fi - stop_test_timer "Stirling-PDF-Regression" + stop_test_timer "Stirling-PDF-Regression $CONTAINER_NAME" fi # `down` with the override removes the agent bind-mount cleanly. The # SIGTERM that `down` sends is what triggers dumponexit=true in the