summaryrefslogtreecommitdiff
path: root/src/pdf_bridge.c
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-14 22:38:50 -0300
committerGabriel Schneider <[email protected]>2026-08-15 11:58:29 -0300
commitbe2a9957708cbf0c478ca861c4a1f0f227bbfe10 (patch)
tree1abf5cb65057048ebfbce7fc427289a437a693f4 /src/pdf_bridge.c
parentf67fec978a9296c651ec06bd2f43686d34ff86ee (diff)
downloadpardes-be2a9957708cbf0c478ca861c4a1f0f227bbfe10.tar.gz
pardes-be2a9957708cbf0c478ca861c4a1f0f227bbfe10.zip
pdf: continuous scroll bench harness and per-frame render path
Diffstat (limited to 'src/pdf_bridge.c')
-rw-r--r--src/pdf_bridge.c66
1 files changed, 59 insertions, 7 deletions
diff --git a/src/pdf_bridge.c b/src/pdf_bridge.c
index 25870f20..bd72a54a 100644
--- a/src/pdf_bridge.c
+++ b/src/pdf_bridge.c
@@ -1,6 +1,7 @@
#include "pdf_bridge.h"
#include <mupdf/fitz.h>
+#include <mupdf/pdf.h>
#include <math.h>
#include <limits.h>
@@ -380,6 +381,22 @@ pardes_pdf_close(pardes_pdf_document *document)
atomic_fetch_sub(&pardes_pdf_active_documents, 1);
}
+/*
+ * A page's SIZE, without building a page.
+ *
+ * fz_bound_page(fz_load_page(n)) is the obvious spelling and it is what this
+ * used to do, but fz_load_page builds a whole pdf_page: it resolves the page
+ * dictionary, then loads and parses every link annotation on it. Laying out a
+ * 5363-page manual's strip asks for 5363 sizes, and profiling an open showed
+ * pdf_load_link_annots alone at 29% of the run — parsing links for pages
+ * nobody has looked at yet, to answer a question about their height.
+ *
+ * The page OBJECT answers it directly, and identically: fz_bound_page on a PDF
+ * is pdf_bound_page(FZ_CROP_BOX), which is pdf_page_obj_transform_box on
+ * page->obj followed by fz_transform_rect — exactly the two calls below, from
+ * exactly the same object. Non-PDF documents (cbz, xps, svg) have no page
+ * objects and keep the loading path.
+ */
int
pardes_pdf_get_page_size(
pardes_pdf_document *document,
@@ -388,6 +405,7 @@ pardes_pdf_get_page_size(
{
fz_context *ctx;
fz_rect bounds;
+ pdf_document *pdf;
if (document == NULL || out == NULL || page_number < 0 ||
page_number >= document->page_count)
@@ -397,8 +415,18 @@ pardes_pdf_get_page_size(
ctx = document->ctx;
fz_try(ctx)
{
- pardes_pdf_cache_page(document, page_number, 0);
- bounds = document->cached_bounds;
+ pdf = pdf_specifics(ctx, document->doc);
+ if (pdf != NULL) {
+ fz_matrix page_ctm;
+ fz_rect cropbox;
+ pdf_obj *page_obj = pdf_lookup_page_obj(ctx, pdf, page_number);
+ pdf_page_obj_transform_box(ctx, page_obj, &cropbox, &page_ctm,
+ FZ_CROP_BOX);
+ bounds = fz_transform_rect(cropbox, page_ctm);
+ } else {
+ pardes_pdf_cache_page(document, page_number, 0);
+ bounds = document->cached_bounds;
+ }
out->width = bounds.x1 - bounds.x0;
out->height = bounds.y1 - bounds.y0;
if (!(out->width > 0.0f) || !(out->height > 0.0f))
@@ -526,7 +554,9 @@ pardes_pdf_render_into(
size_t samples_len,
int width,
int height,
- int stride)
+ int stride,
+ int band_y,
+ int band_height)
{
fz_context *ctx;
fz_pixmap *pixmap = NULL;
@@ -542,9 +572,10 @@ pardes_pdf_render_into(
(highlight_count != 0 && highlights == NULL) ||
highlight_count > PARDES_PDF_MAX_RESULT_QUADS || samples == NULL ||
width < 1 || height < 1 || stride < 1 || width > INT_MAX / 4 ||
- stride != width * 4 ||
- (size_t)height > SIZE_MAX / (size_t)stride ||
- samples_len != (size_t)stride * (size_t)height)
+ stride != width * 4 || band_y < 0 || band_height < 1 ||
+ band_y > height - band_height ||
+ (size_t)band_height > SIZE_MAX / (size_t)stride ||
+ samples_len != (size_t)stride * (size_t)band_height)
return PARDES_PDF_ERROR;
ctx = document->ctx;
@@ -559,17 +590,38 @@ pardes_pdf_render_into(
fz_irect_height(bbox) != height)
fz_throw(ctx, FZ_ERROR_ARGUMENT,
"PDF raster layout changed between measure and render");
+ /*
+ * The BAND: rows [band_y, band_y + band_height) of the page raster,
+ * and nothing else. A pixmap's bbox IS the draw device's clip, so the
+ * page runs under the same CTM it would for a full raster and MuPDF
+ * discards everything outside these rows. Every row inside them is
+ * therefore bit-identical to the same row of the full-page raster —
+ * "band rows equal full-page rows" in pdf.zig proves it, because the
+ * whole point of a band is that the reader cannot tell.
+ */
+ bbox.y0 += band_y;
+ bbox.y1 = bbox.y0 + band_height;
/* External samples are never marked FZ_PIXMAP_FLAG_FREE_SAMPLES. */
pixmap = fz_new_pixmap_with_bbox_and_data(
ctx, fz_device_rgb(ctx), bbox, NULL, 1, samples);
if (fz_pixmap_samples(ctx, pixmap) != samples ||
fz_pixmap_width(ctx, pixmap) != width ||
- fz_pixmap_height(ctx, pixmap) != height ||
+ fz_pixmap_height(ctx, pixmap) != band_height ||
fz_pixmap_stride(ctx, pixmap) != stride ||
fz_pixmap_components(ctx, pixmap) != 4)
fz_throw(ctx, FZ_ERROR_FORMAT,
"MuPDF wrapped an unexpected RGBA layout");
+ /*
+ * ponytail: this memset is ~9% of a fast scroll's profile and it has
+ * twice measured as unremovable. Filling with 32-byte vector stores
+ * instead (a memset of a multi-megabyte raster goes out through
+ * non-temporal stores, 7.4 GB/s against 41.7 for a store loop) changed
+ * a fling's median frame by nothing at all, and skipping the fill
+ * ENTIRELY changed it by 1-2%: the cache misses it is blamed for are
+ * paid either way by the glyph spans and the tint pass that walk the
+ * same buffer immediately afterwards. Leave it alone.
+ */
fz_clear_pixmap_with_value(ctx, pixmap, 0xFF);
if (document->cached_display_list != NULL ||