summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-10-01 01:25:06 -0300
committerGabriel Schneider <[email protected]>2026-10-01 01:35:12 -0300
commitb3bbad85c61f81c01fb983813ccc28ceb6361469 (patch)
tree5cdf4fa745d5ae2c098db463bfabddd48abd6d0e
parentea28dbf4ce827ee5544562bd19453c68599d912f (diff)
downloadpardes-b3bbad85c61f81c01fb983813ccc28ceb6361469.tar.gz
pardes-b3bbad85c61f81c01fb983813ccc28ceb6361469.zip
A PDF's body puts each line of the page's text on a line of its own, not a heading and its paragraph run together with a space
MuPDF's text buffer joins the lines of one block with spaces, so a typst page's heading and the text under it read as one line ("Page one Hello normal PDF alpha."). The bridge now walks the structured text itself, structure blocks included, and ends each line with a newline. Co-Authored-By: Claude Opus 5.5 <[email protected]>
-rw-r--r--src/pdf_bridge.c28
-rw-r--r--test/pdf.zig13
2 files changed, 40 insertions, 1 deletions
diff --git a/src/pdf_bridge.c b/src/pdf_bridge.c
index 6fdab46c..594db0f9 100644
--- a/src/pdf_bridge.c
+++ b/src/pdf_bridge.c
@@ -940,6 +940,31 @@ pardes_pdf_highlight_rows(
return PARDES_PDF_OK;
}
+/* A page's text, a line of it a line of the result: MuPDF's own buffer
+ * joins a block's lines with spaces, which ran a heading into the
+ * paragraph under it. Structure blocks are walked into. */
+static void
+pardes_append_stext_blocks(fz_context *ctx, fz_buffer *buffer, fz_stext_block *block)
+{
+ for (; block != NULL; block = block->next)
+ {
+ if (block->type == FZ_STEXT_BLOCK_STRUCT)
+ {
+ if (block->u.s.down != NULL)
+ pardes_append_stext_blocks(ctx, buffer, block->u.s.down->first_block);
+ continue;
+ }
+ if (block->type != FZ_STEXT_BLOCK_TEXT)
+ continue;
+ for (fz_stext_line *line = block->u.t.first_line; line != NULL; line = line->next)
+ {
+ for (fz_stext_char *ch = line->first_char; ch != NULL; ch = ch->next)
+ fz_append_rune(ctx, buffer, ch->c);
+ fz_append_byte(ctx, buffer, '\n');
+ }
+ }
+}
+
int
pardes_pdf_page_text(
pardes_pdf_document *document,
@@ -961,7 +986,8 @@ pardes_pdf_page_text(
fz_try(ctx)
{
pardes_pdf_cache_page(document, page_number, 1);
- buffer = fz_new_buffer_from_stext_page(ctx, document->cached_text);
+ buffer = fz_new_buffer(ctx, 1024);
+ pardes_append_stext_blocks(ctx, buffer, document->cached_text->first_block);
len = fz_buffer_storage(ctx, buffer, &data);
}
fz_catch(ctx)
diff --git a/test/pdf.zig b/test/pdf.zig
index d91f50a7..727a1515 100644
--- a/test/pdf.zig
+++ b/test/pdf.zig
@@ -352,6 +352,19 @@ test "a +PdfSections row's first number is the page: a section not there, or on
}
}
+test "a PDF's body puts each line of its text on a line of its own, never a heading run into the text under it" {
+ if (!pdf_enabled or platform == .web) return;
+ var tmp = std.testing.tmpDir(.{});
+ defer tmp.cleanup();
+ const p = try PdfLinkTests.init(tmp);
+ defer p.deinit();
+ const text = p.panes[0].?.pdf.?.ensureText(p.pdf_gpa);
+ // The page's three text runs, each its own line.
+ try std.testing.expect(std.mem.indexOf(u8, text, "target.txt\n") != null);
+ try std.testing.expect(std.mem.indexOf(u8, text, "https://example.com/same\n") != null);
+ try std.testing.expect(std.mem.indexOf(u8, text, "label\n") != null);
+}
+
test "PdfSections Look follows the exact owning PDF, not an equal path" {
if (!pdf_enabled or platform == .web) return;
const gpa = std.testing.allocator;