diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-14 22:38:50 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-15 11:58:29 -0300 |
| commit | be2a9957708cbf0c478ca861c4a1f0f227bbfe10 (patch) | |
| tree | 1abf5cb65057048ebfbce7fc427289a437a693f4 /src/pdf_bridge.c | |
| parent | f67fec978a9296c651ec06bd2f43686d34ff86ee (diff) | |
| download | pardes-be2a9957708cbf0c478ca861c4a1f0f227bbfe10.tar.gz pardes-be2a9957708cbf0c478ca861c4a1f0f227bbfe10.zip | |
pdf: continuous scroll bench harness and per-frame render path
Diffstat (limited to 'src/pdf_bridge.c')
| -rw-r--r-- | src/pdf_bridge.c | 66 |
1 files changed, 59 insertions, 7 deletions
diff --git a/src/pdf_bridge.c b/src/pdf_bridge.c index 25870f20..bd72a54a 100644 --- a/src/pdf_bridge.c +++ b/src/pdf_bridge.c @@ -1,6 +1,7 @@ #include "pdf_bridge.h" #include <mupdf/fitz.h> +#include <mupdf/pdf.h> #include <math.h> #include <limits.h> @@ -380,6 +381,22 @@ pardes_pdf_close(pardes_pdf_document *document) atomic_fetch_sub(&pardes_pdf_active_documents, 1); } +/* + * A page's SIZE, without building a page. + * + * fz_bound_page(fz_load_page(n)) is the obvious spelling and it is what this + * used to do, but fz_load_page builds a whole pdf_page: it resolves the page + * dictionary, then loads and parses every link annotation on it. Laying out a + * 5363-page manual's strip asks for 5363 sizes, and profiling an open showed + * pdf_load_link_annots alone at 29% of the run — parsing links for pages + * nobody has looked at yet, to answer a question about their height. + * + * The page OBJECT answers it directly, and identically: fz_bound_page on a PDF + * is pdf_bound_page(FZ_CROP_BOX), which is pdf_page_obj_transform_box on + * page->obj followed by fz_transform_rect — exactly the two calls below, from + * exactly the same object. Non-PDF documents (cbz, xps, svg) have no page + * objects and keep the loading path. + */ int pardes_pdf_get_page_size( pardes_pdf_document *document, @@ -388,6 +405,7 @@ pardes_pdf_get_page_size( { fz_context *ctx; fz_rect bounds; + pdf_document *pdf; if (document == NULL || out == NULL || page_number < 0 || page_number >= document->page_count) @@ -397,8 +415,18 @@ pardes_pdf_get_page_size( ctx = document->ctx; fz_try(ctx) { - pardes_pdf_cache_page(document, page_number, 0); - bounds = document->cached_bounds; + pdf = pdf_specifics(ctx, document->doc); + if (pdf != NULL) { + fz_matrix page_ctm; + fz_rect cropbox; + pdf_obj *page_obj = pdf_lookup_page_obj(ctx, pdf, page_number); + pdf_page_obj_transform_box(ctx, page_obj, &cropbox, &page_ctm, + FZ_CROP_BOX); + bounds = fz_transform_rect(cropbox, page_ctm); + } else { + pardes_pdf_cache_page(document, page_number, 0); + bounds = document->cached_bounds; + } out->width = bounds.x1 - bounds.x0; out->height = bounds.y1 - bounds.y0; if (!(out->width > 0.0f) || !(out->height > 0.0f)) @@ -526,7 +554,9 @@ pardes_pdf_render_into( size_t samples_len, int width, int height, - int stride) + int stride, + int band_y, + int band_height) { fz_context *ctx; fz_pixmap *pixmap = NULL; @@ -542,9 +572,10 @@ pardes_pdf_render_into( (highlight_count != 0 && highlights == NULL) || highlight_count > PARDES_PDF_MAX_RESULT_QUADS || samples == NULL || width < 1 || height < 1 || stride < 1 || width > INT_MAX / 4 || - stride != width * 4 || - (size_t)height > SIZE_MAX / (size_t)stride || - samples_len != (size_t)stride * (size_t)height) + stride != width * 4 || band_y < 0 || band_height < 1 || + band_y > height - band_height || + (size_t)band_height > SIZE_MAX / (size_t)stride || + samples_len != (size_t)stride * (size_t)band_height) return PARDES_PDF_ERROR; ctx = document->ctx; @@ -559,17 +590,38 @@ pardes_pdf_render_into( fz_irect_height(bbox) != height) fz_throw(ctx, FZ_ERROR_ARGUMENT, "PDF raster layout changed between measure and render"); + /* + * The BAND: rows [band_y, band_y + band_height) of the page raster, + * and nothing else. A pixmap's bbox IS the draw device's clip, so the + * page runs under the same CTM it would for a full raster and MuPDF + * discards everything outside these rows. Every row inside them is + * therefore bit-identical to the same row of the full-page raster — + * "band rows equal full-page rows" in pdf.zig proves it, because the + * whole point of a band is that the reader cannot tell. + */ + bbox.y0 += band_y; + bbox.y1 = bbox.y0 + band_height; /* External samples are never marked FZ_PIXMAP_FLAG_FREE_SAMPLES. */ pixmap = fz_new_pixmap_with_bbox_and_data( ctx, fz_device_rgb(ctx), bbox, NULL, 1, samples); if (fz_pixmap_samples(ctx, pixmap) != samples || fz_pixmap_width(ctx, pixmap) != width || - fz_pixmap_height(ctx, pixmap) != height || + fz_pixmap_height(ctx, pixmap) != band_height || fz_pixmap_stride(ctx, pixmap) != stride || fz_pixmap_components(ctx, pixmap) != 4) fz_throw(ctx, FZ_ERROR_FORMAT, "MuPDF wrapped an unexpected RGBA layout"); + /* + * ponytail: this memset is ~9% of a fast scroll's profile and it has + * twice measured as unremovable. Filling with 32-byte vector stores + * instead (a memset of a multi-megabyte raster goes out through + * non-temporal stores, 7.4 GB/s against 41.7 for a store loop) changed + * a fling's median frame by nothing at all, and skipping the fill + * ENTIRELY changed it by 1-2%: the cache misses it is blamed for are + * paid either way by the glyph spans and the tint pass that walk the + * same buffer immediately afterwards. Leave it alone. + */ fz_clear_pixmap_with_value(ctx, pixmap, 0xFF); if (document->cached_display_list != NULL || |
