Files
silo-server/web/vendor/foliate-js/pdf.js
T
8b70357703 feat(ebooks): first-class ebook libraries, scanner, and reader (#124)
* docs: define ebook architecture matching audiobooks

* docs: plan ebook audiobook-parity implementation

* feat: add ebook scanner parser foundation

* fix: harden ebook scanner foundation

* fix: handle ebook isbn labels

* fix: guard ebook subtree scans

* feat: scan ebook libraries in core

* fix: preserve ebook scan people credits

* fix: refresh ebook scan metadata safely

* feat: persist ebook series membership

* test: cover ebook series persistence decisions

* fix: address ebook scanner PR review

* docs: clarify ebook foundation PR scope

* feat: add ebook metadata enricher

* fix: harden ebook poster cache

* feat: wire ebook metadata sync task

* feat: expose ebook library metadata setup

* feat: add ebook catalog scope support

* feat: add ebook detail view

* feat: label ebook file versions by format

* feat: use file-size copy for downloads

* feat: use file language in download dialog

* test: cover ebook detail authors and downloads

* fix: drop narrator credits from ebook scanner merges

* fix: align ebook collection filters with book media

* fix: drop asin provider ids from ebook enrichment

* fix: force ebook people refresh for stale narrators

* chore: omit ebook planning docs from branch

* feat: add ebook detail related content

* feat: add ebook reader file entrypoint

* feat: render ebooks with foliate reader

* feat: persist ebook reader progress

* feat: add ebook reader controls

* feat: extract ebook pdf metadata

* feat: favor scanner isbn during ebook enrichment

* feat: extract fbz ebook metadata

* feat: count cbz ebook pages

* feat: show ebook file page counts

* feat: show ebook download summaries

* feat: switch ebook reader files

* feat: prefer epub for ebook read action

* feat: surface ebook reader progress

* feat: sync ebook reader progress cache

* feat: hide ebook read action for unsupported files

* feat: filter ebook reader file selector

* fix: serve fbz ebook archives with reader mime type

* fix: detect fbz ebooks from compound filename

* fix: authorize fbz ebooks from compound filename

* fix: scope ebook catalog facets

* fix: reject narrator queries for ebooks

* fix: build ebook recommendation text from authors

* fix: include ebooks in embedding eligibility

* fix: include ebooks in recommendation media mix

* fix: include ebooks in recently added recommendations

* feat: include ebook progress in recommendation signals

* feat: include ebooks in continue watching sections

* feat: include ebooks in catalog progress metrics

* fix: read ebook isbn from epub metadata

* fix: filter ebook asin provider aliases

* fix: fall back from unsupported ebook reader files

* fix: sort ebook catalogs by reader progress

* fix: filter ebook catalogs by reader progress

* fix: include ebooks in last watched catalog filters

* feat: reflect ebook reader progress in item user state

* feat: share ebook progress state across item surfaces

* feat: report ebook scan progress

* fix: include ebook activity in recommendations

* fix: expose ebook reader progress on item detail

* fix: support ebook subtree scans

* fix: honor profile header for ebook item progress

* fix: add ebook library default sections

* fix: route ebook continue cards to reader

* fix: hide watched toggle for ebooks

* fix: route ebook watch tonight cards to reader

* fix: route ebook hero actions to reader

* fix: detect archive ebook reader formats by filename

* feat: cache embedded ebook covers during scan

* fix: encode ebook hero reader links

* fix: persist non-epub ebook reader progress

* fix: scope narrator catalog badges to audiobooks

* fix: merge ebook reader progress during item repair

* fix: label ebook progress filters as read

* fix: show ebook related rails as book covers

* fix: remove txt ebook reader support

* fix: reject txt ebook reader files

* fix: label ebook advanced filters as read

* fix: label ebook personalized sorts as read

* fix: remove plain text reader loader path

* test: cover ebook unread catalog rules

* fix: preserve ebook reader library context

* fix: link ebook genres with library scope

* fix: encode related rail item links

* fix: encode catalog card item links

* fix: encode hero and continue item links

* fix: encode watch tonight item links

* fix: encode recommendation and search item links

* test: cover ebook scan format set

* fix: label ebook search results clearly

* fix: make global search prompt media neutral

* fix: encode catalog read API ids

* fix: encode item API ids

* fix: include ebook reader vendor in docker build

* fix: make ebook reader build clean

* fix: clean ebook embedded descriptions

* docs: plan ebook reader shell parity

* feat: add ebook reader shell controls

* fix: widen ebook scrolled reader flow

* fix: remove scrolled reader content width cap

* docs: plan ebook reader full parity

* feat: persist ebook reader config

* feat: add ebook annotations and bookmarks

* feat: add ebook reader tools and aids

* feat: add ebook advanced reader settings

* fix: keep ebook reader panel in viewport

* fix: use foliate sizing units for ebook scroll flow

* fix: keep ebook settings controls readable

* fix: simplify ebook reader settings controls

* feat(ebooks): extract local covers during scan (#98)

* feat(ebooks): extract local covers during scan

* fix(ebooks): read nullable poster paths during cover scan

* fix(catalog): coalesce nullable media artwork fields

* fix(ebooks): group sibling formats by book identity

* fix(ebooks): tolerate legacy ebook metadata encodings

* fix(ebooks): decode PDF hex metadata strings

* fix(ebooks): harden local cover extraction and format grouping

Address review findings on the local cover scan:

- Restrict generic sidecar covers (cover.jpg, folder.png, ...) to
  single-book directories, always accept images named after the book
  file, and apply exactly one cover per reconcile with sidecar taking
  precedence over the embedded cover.
- Replace the read-then-write poster update with an atomic conditional
  UPDATE (ItemRepository.SetLocalPoster) so provider/admin artwork is
  never clobbered by concurrent writers, and refresh locally owned
  posters when the extracted cover bytes change (thumbhash compare).
- Preserve UTF-8 PDF Info strings (including a UTF-8 BOM) instead of
  forcing everything through Windows-1252; the cp1252 fallback now only
  applies to non-UTF-8 bytes.
- Select EPUB covers by manifest media-type with properties="cover-image"
  outranking the EPUB2 meta name="cover" id, so XHTML cover pages no
  longer shadow the real image.
- Order CBZ pages naturally (2.jpg before 10.jpg, ch2/ before ch10/)
  when picking the cover page, via a single O(n) min-scan.
- Bump the ebook content group key scheme to version 2 and reprocess
  rows written under older versions so pre-existing libraries gain
  sibling-format grouping instead of accumulating duplicates.
- Group different formats only (a same-format sibling with colliding
  sparse metadata stays a separate item) and stop a joining sibling's
  embedded metadata from overwriting a provider-matched item.
- Decode any IANA-labelled OPF/FB2 XML charset (windows-1251, koi8-r,
  shift_jis, ...) via x/net/html/charset, and wire the charset reader
  into FB2 parsing which previously had none.
- Strip the full .fb2.zip double extension from filename-derived titles
  and group keys.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>

* feat(ebooks): add reader profiles and ruler (#99)

* feat(ebooks): extract local covers during scan

* fix(ebooks): read nullable poster paths during cover scan

* fix(catalog): coalesce nullable media artwork fields

* fix(ebooks): group sibling formats by book identity

* fix(ebooks): tolerate legacy ebook metadata encodings

* fix(ebooks): decode PDF hex metadata strings

* feat(ebooks): add reader profiles and ruler

* fix(ebooks): address reader ruler and profile review findings

- skip renderer setStyles/render when computed styles and attributes are
  unchanged, so ruler position updates no longer re-style the book view
- drag the ruler via a local draft that commits on release, with the
  surface rect cached at pointer-down
- migrate font values persisted before the generic stacks (Inter,
  Georgia, Merriweather, legacy serif) so the font select never renders
  blank, with a Custom fallback option for unknown values
- make the ruler band click-through and move dragging to a dedicated
  keyboard-accessible slider handle so links and text selection keep
  working under the band
- share font stacks between options and profiles via READER_FONT_STACKS
- surface the active reading profile, move presets to the top of the
  settings panel, and drop the redundant profile button aria-labels

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ebooks): resolve prefer-const lint error in readest document lib

`pnpm run lint` failed on the branch because `direction` is never
reassigned in getDirection; split the destructure so only the
reassigned `writingMode` stays mutable.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>

* Merge branch 'main' into work/ebooks-reader-base

Brings the ebook integration branch up to date with main (audiobook
library redesign, continue-watching rework and card affordances,
quic-go bump, jellycompat fixes). Conflict resolutions favor main's
generalized mechanisms and register ebooks with them:

- media scope validation goes through IsValidMediaScope (now including
  "ebook" alongside main's "video" group scope), in Go and in the web
  filter/search types
- continue-watching uses main's typed rails; reading-type sections pull
  resume points from ebook_reader_progress and the ebook library default
  section is wired to ContinueTypeConfig(ContinueTypeReading)
- item_repo keeps main's derived select-list machinery (itemColumnExpr)
  and both poster accessors (GetPoster/SetLocalPoster for ebook covers,
  GetPosterPath for audiobook covers)
- web cards/hero/watch-tonight adopt main's buildMediaPlayHref helpers,
  which now route ebooks to /reader/ebook and encode content ids;
  ebook affordances (BookOpen icon, Read verb, percent-read subtitle)
  carry over onto main's reworked components
- LibraryForm ebook support ported into main's refactored
  useLibraryForm/libraryTypes modules

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(docker): copy foliate-js vendor into Dockerfile.dev frontend stage

foliate-js is a file:vendor/foliate-js dependency, so pnpm install needs
the vendor directory before the lockfile install layer. The production
Dockerfile already copies it; the dev image was missed, breaking
make dev-deploy with ENOENT on /app/web/vendor/foliate-js.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(ebooks): render Continue Reading sections as upright poster cards

All-ebook continue sections previously fell through to the horizontal
16:9 wide card; include ebooks in the poster-variant check so book
covers render in their natural 2:3 framing.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ui): stop related-rail highlight ring clipping on detail pages

Move the current-item ring onto the cover artwork with a themed
ring-offset color (matching the sidebar profile highlight) and give
the scroll container top headroom so the ring is not cut off by
overflow-x-auto. Applies to both ebook and audiobook detail rails.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(scanner): harden ebook scanning against data loss and bad metadata

- Reconcile missing ebook files like video/audio, with real per-root walk
  failure tracking (failed/unmounted roots are excluded from deletion),
  symlinked-root support via the shared logical walker, and the empty-root
  cleanup allowance before any destructive reconciliation.
- Create ebook items as 'pending' so enrichment can promote them to
  'matched' (backfill migration included), and protect matched items from
  re-scan clobbering: title/year skipped, people/series fill-empty only.
- PDF metadata: scan head + tail windows (non-linearized PDFs keep the Info
  dict at the end), require proper key delimiters, head values win.
- Cap plain .fb2 reads like .fbz entries; drop .md as an ebook format.
- gofmt internal/scanner/audiobook.go (pre-existing drift).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ebooks): make enrichment failures non-terminal with dedicated backoff state

- Provider errors now record a failure (capped retries) instead of stamping
  last_refreshed, which permanently excluded items after transient outages.
- Unconfigured metadata chains and the scan-window membership race skip the
  item without stamping or burning a retry.
- Failure tracking moves to a new ebook_enrichment_state table, decoupling
  it from media_items.refresh_failures (shared with metadata refresh debt).
- Preserve non-author people credits when persisting enrichment results.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(catalog): gate ebook progress on hidden history and centralize threshold

- Apply user_history_hidden_items gating (video semantics) to the ebook
  watched/in-progress filters, progress sort plan, and Continue Reading.
- Continue Reading pages past dismissed items via the shared collector and
  dedupes items across pages (also fixes the video path's latent exposure).
- Centralize the 0.9 finished threshold as models.EbookFinishedProgressThreshold
  with a single SQL-interpolated mirror in catalog.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(recommendations): correct watcher counting and wire ebook taste signals

- itemWatchersQuery dedupes to distinct (watcher, item) rows so one
  binge-watcher can no longer satisfy minWatchers; the eligibility floor
  now counts distinct accounts rather than profiles.
- Hidden-history gating on GetEbookReaderProgressForUser (signal reader).
- Ebook reading produces canonical implicit taste signals (weighted like
  the equivalent movie progress ratio); ebooks join taste-seed candidates.
- Stale GetRecentlyAddedItems doc comment corrected.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(api): harden ebook reader endpoints and serve a Content-Security-Policy

- Serve a CSP on all SPA HTML responses: blob/srcdoc book iframes inherit
  it, so script-src 'self' 'wasm-unsafe-eval' blocks script execution from
  malicious book content (sandbox alone is defeated by the WebKit
  allow-scripts requirement). Threat model documented on the constant.
- X-Content-Type-Options: nosniff on frontend, jellycompat, and ebook file
  responses; MIME resolution can no longer fall through to octet-stream
  for an admitted ebook file.
- Annotation PATCH: presence-aware field semantics (absent keeps, present
  sets/clears), invariant re-validation on the merged row, and an atomic
  SELECT ... FOR UPDATE read-merge-write.
- Request size caps (413) on progress/config/annotation writes;
  Content-Disposition via mime.FormatMediaType; hidden-history gating in
  the shared ebook progress lister; FK-cascade indexes for reader tables.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(api): native read-state endpoints for ebooks

- POST/DELETE /watched/{id} accepts ebook content IDs: mark read upserts
  progress 1.0 preserving the reader's file/location (or picks the
  preferred reader file for never-opened books); mark unread mirrors video
  unwatch semantics and deletes the progress row.
- /history/remove accepts ebooks: hides via user_history_hidden_items
  without touching the reading position (hidden != unread; next reading
  activity resurfaces the book, mirroring video re-watch).
- Access-filter checks match the video branch; shared logic lives in
  ebook_read_state.go. Sort metrics/user-state thresholds use the shared
  constant; profile-header fallback deduplicated.

Clients: response is {type: "ebook", affected_count: 1, played: bool};
the existing watched SSE event fires.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(web): harden the ebook reader UI

- Open-flow race: cancellation checked after every await with full stale-run
  teardown (no wrong-file progress saves, no leaked views/blob URLs);
  book.destroy() on cleanup.
- Progress: monotonic stale-response guard; visibilitychange flush uses the
  refresh-capable client, pagehide uses keepalive; per-book cross-format
  progress documented as deliberate.
- Settings: side effects out of the setState updater; local edits no longer
  clobbered by late server config; pending saves flushed on unmount/pagehide.
- TTS: generation token so Stop actually stops (Chromium/Firefox synthetic
  events); Media Session uninstalled on unmount.
- External book links: http(s) only, opened with noopener,noreferrer.
- apiBlob 512 MiB guard with a user-facing error; fraction bookmarks
  navigable; search-result key collisions fixed; dead e-ink code removed;
  getLibrarySortRelevanceScope deduplicated; md format dropped.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(web): mark read/unread affordances for ebooks

- Item detail gets a Mark Read/Unread button; card menus drop the ebook
  gate and share type-aware labels/toasts (also dedupes audiobook wording).
- Watched-state invalidation includes the reader progress query key so the
  Continue button and percent refresh after toggling.
- Continue Reading dismiss copy for ebooks; dismissal path now URL-encodes
  item IDs (ebook content IDs can contain reserved characters).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* docs: record the PR #124 review and hardening pass

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-06-10 08:18:35 -04:00

514 lines
18 KiB
JavaScript

const pdfjsPath = path => `/vendor/pdfjs/${path}`
import '@pdfjs/pdf.min.mjs'
const pdfjsLib = globalThis.pdfjsLib
pdfjsLib.GlobalWorkerOptions.workerSrc = pdfjsPath('pdf.worker.min.mjs')
const fetchText = async url => await (await fetch(url)).text()
let textLayerBuilderCSS = null
let annotationLayerBuilderCSS = null
// Track active render tasks per iframe document to cancel superseded renders
const activeRenderTasks = new WeakMap()
// Generation counter per document to detect stale renders after async gaps
const renderGenerations = new WeakMap()
// Set up panning and selection event handlers once per iframe document
const setupPanningEvents = (doc) => {
if (doc._readestEventsInitialized) return
doc._readestEventsInitialized = true
const container = doc.querySelector('.textLayer')
if (!container) return
let isPanning = false
let startX = 0
let startY = 0
let scrollLeft = 0
let scrollTop = 0
let scrollParent = null
const findScrollableParent = (element) => {
let current = element
while (current) {
if (current !== document.body && current.nodeType === 1) {
const style = window.getComputedStyle(current)
const overflow = style.overflow + style.overflowY + style.overflowX
if (/(auto|scroll)/.test(overflow)) {
if (current.scrollHeight > current.clientHeight ||
current.scrollWidth > current.clientWidth) {
return current
}
}
}
if (current.parentElement) {
current = current.parentElement
} else if (current.parentNode && current.parentNode.host) {
current = current.parentNode.host
} else {
break
}
}
return window
}
container.onpointerdown = (e) => {
const selection = doc.getSelection()
const hasTextSelection = selection && selection.toString().length > 0
const elementUnderCursor = doc.elementFromPoint(e.clientX, e.clientY)
const hasTextUnderneath = elementUnderCursor &&
(elementUnderCursor.tagName === 'SPAN' || elementUnderCursor.tagName === 'P') &&
elementUnderCursor.textContent.trim().length > 0
if (!hasTextUnderneath && !hasTextSelection) {
isPanning = true
startX = e.screenX
startY = e.screenY
const iframe = doc.defaultView?.frameElement
if (iframe) {
scrollParent = findScrollableParent(iframe)
if (scrollParent === window) {
scrollLeft = window.scrollX || window.pageXOffset
scrollTop = window.scrollY || window.pageYOffset
} else {
scrollLeft = scrollParent.scrollLeft
scrollTop = scrollParent.scrollTop
}
container.style.cursor = 'grabbing'
}
} else {
container.classList.add('selecting')
}
}
container.onpointermove = (e) => {
if (isPanning && scrollParent) {
e.preventDefault()
const dx = e.screenX - startX
const dy = e.screenY - startY
if (scrollParent === window) {
window.scrollTo(scrollLeft - dx, scrollTop - dy)
} else {
scrollParent.scrollLeft = scrollLeft - dx
scrollParent.scrollTop = scrollTop - dy
}
}
}
container.onpointerup = () => {
if (isPanning) {
isPanning = false
scrollParent = null
container.style.cursor = 'grab'
} else {
container.classList.remove('selecting')
}
}
container.onpointerleave = () => {
if (isPanning) {
isPanning = false
scrollParent = null
container.style.cursor = 'grab'
}
}
doc.addEventListener('selectionchange', () => {
const selection = doc.getSelection()
if (selection && selection.toString().length > 0) {
container.style.cursor = 'text'
} else if (!isPanning) {
container.style.cursor = 'grab'
}
})
container.style.cursor = 'grab'
}
const render = async (page, doc, zoom, pageColors) => {
if (!doc) return
// Increment generation to invalidate any in-progress render for this doc
const generation = (renderGenerations.get(doc) || 0) + 1
renderGenerations.set(doc, generation)
// Cancel any in-progress render task for this document
const existingTask = activeRenderTasks.get(doc)
if (existingTask) {
existingTask.cancel()
activeRenderTasks.delete(doc)
}
const scale = zoom * devicePixelRatio
doc.documentElement.style.transform = `scale(${1 / devicePixelRatio})`
doc.documentElement.style.transformOrigin = 'top left'
doc.documentElement.style.setProperty('--total-scale-factor', scale)
doc.documentElement.style.setProperty('--user-unit', '1')
doc.documentElement.style.setProperty('--scale-round-x', '1px')
doc.documentElement.style.setProperty('--scale-round-y', '1px')
const viewport = page.getViewport({ scale })
// the canvas must be in the `PDFDocument`'s `ownerDocument`
// (`globalThis.document` by default); that's where the fonts are loaded
const canvas = document.createElement('canvas')
canvas.height = viewport.height
canvas.width = viewport.width
const canvasContext = canvas.getContext('2d')
const renderTask = page.render({ canvasContext, viewport, pageColors })
activeRenderTasks.set(doc, renderTask)
try {
await renderTask.promise
} catch {
// Render was cancelled or failed — release canvas bitmap memory
canvas.width = 0
canvas.height = 0
return
} finally {
if (activeRenderTasks.get(doc) === renderTask) {
activeRenderTasks.delete(doc)
}
}
// Bail out if a newer render has started or iframe was removed
if (renderGenerations.get(doc) !== generation || !doc.defaultView) {
canvas.width = 0
canvas.height = 0
return
}
const canvasElement = doc.querySelector('#canvas')
if (!canvasElement) {
canvas.width = 0
canvas.height = 0
return
}
// Release old canvas bitmap memory before replacing
const oldCanvas = canvasElement.querySelector('canvas')
if (oldCanvas) {
oldCanvas.width = 0
oldCanvas.height = 0
}
canvasElement.replaceChildren(doc.adoptNode(canvas))
// Clear text layer before re-rendering to prevent DOM accumulation
const container = doc.querySelector('.textLayer')
container.replaceChildren()
const textLayer = new pdfjsLib.TextLayer({
textContentSource: await page.streamTextContent(),
container, viewport,
})
await textLayer.render()
// Bail out if superseded after async text layer render
if (renderGenerations.get(doc) !== generation) return
// hide "offscreen" canvases appended to document when rendering text layer
// https://github.com/mozilla/pdf.js/blob/642b9a5ae67ef642b9a8808fd9efd447e8c350e2/web/pdf_viewer.css#L51-L58
for (const hiddenCanvas of document.querySelectorAll('.hiddenCanvasElement'))
Object.assign(hiddenCanvas.style, {
position: 'absolute',
top: '0',
left: '0',
width: '0',
height: '0',
display: 'none',
})
// fix text selection
// https://github.com/mozilla/pdf.js/blob/642b9a5ae67ef642b9a8808fd9efd447e8c350e2/web/text_layer_builder.js#L105-L107
const endOfContent = document.createElement('div')
endOfContent.className = 'endOfContent'
container.append(endOfContent)
// Set up panning/selection event handlers once per document
setupPanningEvents(doc)
// Clear annotation layer before re-rendering to prevent DOM accumulation
const div = doc.querySelector('.annotationLayer')
div.replaceChildren()
const linkService = {
goToDestination: () => {},
getDestinationHash: dest => JSON.stringify(dest),
addLinkAttributes: (link, url) => link.href = url,
}
await new pdfjsLib.AnnotationLayer({ page, viewport, div, linkService }).render({
annotations: await page.getAnnotations(),
})
}
const renderPage = async (page, getImageBlob) => {
const viewport = page.getViewport({ scale: 1 })
if (getImageBlob) {
const canvas = document.createElement('canvas')
canvas.height = viewport.height
canvas.width = viewport.width
const canvasContext = canvas.getContext('2d')
await page.render({ canvasContext, viewport }).promise
return new Promise(resolve => canvas.toBlob(blob => {
// Release canvas bitmap memory after extracting the blob
canvas.width = 0
canvas.height = 0
resolve(blob)
}))
}
// https://github.com/mozilla/pdf.js/blob/642b9a5ae67ef642b9a8808fd9efd447e8c350e2/web/text_layer_builder.css
if (textLayerBuilderCSS == null) {
textLayerBuilderCSS = await fetchText(pdfjsPath('text_layer_builder.css'))
}
// https://github.com/mozilla/pdf.js/blob/642b9a5ae67ef642b9a8808fd9efd447e8c350e2/web/annotation_layer_builder.css
if (annotationLayerBuilderCSS == null) {
annotationLayerBuilderCSS = await fetchText(pdfjsPath('annotation_layer_builder.css'))
}
const data = `
<!DOCTYPE html>
<html lang="en">
<meta charset="utf-8">
<meta name="viewport" content="width=${viewport.width}, height=${viewport.height}">
<style>
html, body {
margin: 0;
padding: 0;
}
${textLayerBuilderCSS}
${annotationLayerBuilderCSS}
</style>
<div id="canvas"></div>
<div class="textLayer"></div>
<div class="annotationLayer"></div>
`
const src = URL.createObjectURL(new Blob([data], { type: 'text/html' }))
const onZoom = ({ doc, scale, pageColors }) => render(page, doc, scale, pageColors)
return { src, data, onZoom }
}
const makeTOCItem = async (item, pdf) => {
let pageIndex = undefined
if (item.dest) {
try {
const dest = typeof item.dest === 'string'
? await pdf.getDestination(item.dest)
: item.dest
if (dest?.[0]) {
pageIndex = await pdf.getPageIndex(dest[0])
}
} catch (e) {
console.warn('Failed to get page index for TOC item:', item.title, e)
}
}
return {
label: item.title,
href: item.dest ? JSON.stringify(item.dest) : '',
index: pageIndex,
subitems: item.items?.length
? await Promise.all(item.items.map(i => makeTOCItem(i, pdf)))
: null,
}
}
const MAX_CACHED_PAGES = 8
const CALIBRE_NS = 'http://calibre-ebook.com/xmp-namespace'
const CALIBRE_SI_NS = 'http://calibre-ebook.com/xmp-namespace-series-index'
const RDF_NS = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#'
// Calibre writes series metadata into the XMP packet as
// <calibre:series rdf:parseType="Resource">
// <rdf:value>Name</rdf:value>
// <calibreSI:series_index>1.00</calibreSI:series_index>
// </calibre:series>
const parseCalibreSeriesFromXMP = raw => {
if (!raw || typeof raw !== 'string') return null
let doc
try {
doc = new DOMParser().parseFromString(raw, 'application/xml')
} catch {
return null
}
if (!doc || doc.getElementsByTagName('parsererror').length) return null
const seriesEls = doc.getElementsByTagNameNS(CALIBRE_NS, 'series')
const seriesEl = seriesEls.item(0)
if (!seriesEl) return null
const valueEl = seriesEl.getElementsByTagNameNS(RDF_NS, 'value').item(0)
const name = valueEl?.textContent?.trim()
if (!name) return null
const indexEl = seriesEl.getElementsByTagNameNS(CALIBRE_SI_NS, 'series_index').item(0)
const position = indexEl?.textContent?.trim()
return position ? { name, position } : { name }
}
export const makePDF = async file => {
const transport = new pdfjsLib.PDFDataRangeTransport(file.size, [])
transport.requestDataRange = (begin, end) => {
file.slice(begin, end).arrayBuffer().then(chunk => {
transport.onDataRange(begin, chunk)
})
}
const pdf = await pdfjsLib.getDocument({
range: transport,
wasmUrl: pdfjsPath(''),
cMapUrl: pdfjsPath('cmaps/'),
standardFontDataUrl: pdfjsPath('standard_fonts/'),
isEvalSupported: false,
}).promise
// Get viewport dimensions from first page for fixed-layout rendering
const firstPage = await pdf.getPage(1)
const firstViewport = firstPage.getViewport({ scale: 1 })
const book = { rendition: {
layout: 'pre-paginated',
viewport: { width: firstViewport.width, height: firstViewport.height },
} }
const { metadata, info } = await pdf.getMetadata() ?? {}
// TODO: for better results, parse `metadata.getRaw()`
book.metadata = {
title: metadata?.get('dc:title') ?? info?.Title,
author: metadata?.get('dc:creator') ?? info?.Author,
contributor: metadata?.get('dc:contributor'),
description: metadata?.get('dc:description') ?? info?.Subject,
language: metadata?.get('dc:language'),
publisher: metadata?.get('dc:publisher'),
subject: metadata?.get('dc:subject'),
identifier: metadata?.get('dc:identifier'),
source: metadata?.get('dc:source'),
rights: metadata?.get('dc:rights'),
}
const calibreSeries = parseCalibreSeriesFromXMP(metadata?.getRaw?.())
if (calibreSeries) book.metadata.belongsTo = { series: calibreSeries }
const outline = await pdf.getOutline()
book.toc = outline ? await Promise.all(outline.map(item => makeTOCItem(item, pdf))) : null
const cache = new Map()
const pageCache = new Map()
const getPage = async (i) => {
const cached = pageCache.get(i)
if (cached) {
// Move to end for LRU ordering
pageCache.delete(i)
pageCache.set(i, cached)
return cached
}
const page = await pdf.getPage(i + 1)
pageCache.set(i, page)
// Evict oldest pages when over limit, freeing internal page data
while (pageCache.size > MAX_CACHED_PAGES) {
const oldestKey = pageCache.keys().next().value
const oldPage = pageCache.get(oldestKey)
pageCache.delete(oldestKey)
oldPage?.cleanup()
}
return page
}
book.sections = Array.from({ length: pdf.numPages }).map((_, i) => ({
id: i,
load: async () => {
const cached = cache.get(i)
if (cached) {
// Move to end for LRU ordering
cache.delete(i)
cache.set(i, cached)
return cached
}
const url = await renderPage(await getPage(i))
cache.set(i, url)
// Evict oldest render results when over limit
while (cache.size > MAX_CACHED_PAGES) {
const oldestKey = cache.keys().next().value
const oldEntry = cache.get(oldestKey)
cache.delete(oldestKey)
if (oldEntry?.src) URL.revokeObjectURL(oldEntry.src)
}
return url
},
createDocument: async () => {
const page = await getPage(i)
const doc = document.implementation.createHTMLDocument('')
const canvas = doc.createElement('div')
canvas.id = 'canvas'
doc.body.appendChild(canvas)
const textLayer = doc.createElement('div')
textLayer.className = 'textLayer'
doc.body.appendChild(textLayer)
const annotationLayer = doc.createElement('div')
annotationLayer.className = 'annotationLayer'
doc.body.appendChild(annotationLayer)
// TextLayer requires canvas 2d context for font metrics;
// fall back to manual span construction when unavailable
const probe = doc.createElement('canvas')
if (probe.getContext?.('2d')) {
const textLayerInstance = new pdfjsLib.TextLayer({
textContentSource: await page.streamTextContent(),
container: textLayer, viewport: page.getViewport({ scale: 1 }),
})
await textLayerInstance.render()
} else {
const content = await page.getTextContent()
for (const item of content.items) {
if (item.str) {
const span = doc.createElement('span')
span.textContent = item.str
textLayer.appendChild(span)
}
}
}
return doc
},
size: 1000,
}))
book.isExternal = uri => /^\w+:/i.test(uri)
book.resolveHref = async href => {
const parsed = JSON.parse(href)
const dest = typeof parsed === 'string'
? await pdf.getDestination(parsed) : parsed
const index = await pdf.getPageIndex(dest[0])
return { index }
}
book.splitTOCHref = async href => {
if (!href) return [null, null]
const parsed = JSON.parse(href)
const dest = typeof parsed === 'string'
? await pdf.getDestination(parsed) : parsed
try {
const index = await pdf.getPageIndex(dest[0])
return [index, null]
} catch (e) {
console.warn('Error getting page index for href', href, e)
return [null, null]
}
}
book.getTOCFragment = doc => doc.documentElement
book.getCover = async () => renderPage(await pdf.getPage(1), true)
book.destroy = () => {
// Clean up all cached canvases and revoke blob URLs
for (const [, entry] of cache) {
if (entry?.src) URL.revokeObjectURL(entry.src)
}
cache.clear()
for (const [, page] of pageCache) {
page?.cleanup()
}
pageCache.clear()
pdf.destroy()
}
return book
}