From 5961587b227e5fa097fb033e32c29da08a23fb18 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sun, 2 Aug 2026 17:17:35 -0300 Subject: tty/image: big harness + golden coverage pass --- mupdf.zig | 291 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 291 insertions(+) create mode 100644 mupdf.zig (limited to 'mupdf.zig') diff --git a/mupdf.zig b/mupdf.zig new file mode 100644 index 00000000..c80f678d --- /dev/null +++ b/mupdf.zig @@ -0,0 +1,291 @@ +const std = @import("std"); + +/// MuPDF 1.27.0's release archive does not publish a Zig build. Keep its +/// upstream source-of-truth (`Makelists`) and compile the selected C sources +/// as an ordinary Zig static library. This is intentionally a PDF-only first +/// slice: the application needs rendering and structured text, not MuPDF's +/// other document handlers, tools, Javascript engine, OCR, or writers. +pub const Result = struct { + dependency: *std.Build.Dependency, + library: *std.Build.Step.Compile, + + /// Give a Zig module the same preprocessor view as the library, expose the + /// public C headers to @cImport, and propagate the static link. + pub fn linkTo(self: Result, module: *std.Build.Module) void { + configureModule(module, self.dependency); + // MuPDF's public context headers select the lock-debugging contract + // from NDEBUG. Every consumer must therefore agree with the archive, + // even when its own Zig optimization/safety mode differs. + module.addCMacro("NDEBUG", "1"); + module.linkLibrary(self.library); + } +}; + +pub fn add(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency, options: struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, +}) Result { + const module = b.createModule(.{ + .target = options.target, + .optimize = options.optimize, + .link_libc = true, + }); + configureModule(module, dependency); + // MuPDF's upstream release build defines NDEBUG independently of the + // optimizer. Without it, an optimized Zig build still enables Fitz's + // debug-locking/assertion paths. linkTo applies the same public-header + // contract to consumers; keeping it here also covers every archive source. + module.addCMacro("NDEBUG", "1"); + + // MuPDF itself uses wildcards for these two directories. The dependency + // is immutable and content-addressed, so discovering the flat C lists at + // configure time is deterministic; sorting prevents filesystem order from + // perturbing Zig's cache key. + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = flatCSources(b, io, dependency, "source/fitz"), + // MuPDF's C uses intentional patterns that Zig/Clang's trap-mode + // UBSan rejects in ReleaseSafe (opening any document traps in + // fz_drop_archive). Keep ReleaseSafe's optimization while leaving + // undefined-behaviour instrumentation off for this upstream layer. + .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined" }, + }); + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = flatCSources(b, io, dependency, "source/pdf"), + .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined" }, + }); + + // TOFU + TOFU_CJK leaves the Base-14 PDF fonts enabled. The official + // source release ships their generated C blobs, so no host-side objcopy or + // shell generator is involved. + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = flatCSources(b, io, dependency, "generated/resources/fonts/urw"), + // These are large byte arrays; optimization only wastes build time. + // This matches MuPDF's own generated-resource compilation rule. + .flags = &.{ "-std=gnu11", "-O0" }, + }); + + // Parse only the upstream variables we actually select. These lists are + // the same ones MuPDF's Makefile places in libmupdf-third.a; keeping the + // names in Makelists avoids a large, easy-to-stale duplicate in this file. + const makelists = readMakelists(b, io, dependency); + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = makeSources(b, makelists, "FREETYPE_SRC"), + .flags = &.{ + "-std=gnu11", + "-DFT_CONFIG_MODULES_H=\"slimftmodules.h\"", + "-DFT_CONFIG_OPTIONS_H=\"slimftoptions.h\"", + "-DFT2_BUILD_LIBRARY", + }, + }); + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = makeSources(b, makelists, "LIBJPEG_SRC"), + .flags = &.{"-std=gnu11"}, + }); + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = makeSources(b, makelists, "ZLIB_SRC"), + .flags = &.{ "-std=gnu11", "-DHAVE_UNISTD_H", "-DHAVE_STDARG_H" }, + }); + module.addCSourceFiles(.{ + .root = dependency.path(""), + .files = makeSources(b, makelists, "JBIG2DEC_SRC"), + .flags = &.{ + "-std=gnu11", + "-DHAVE_STDINT_H", + "-DJBIG_EXTERNAL_MEMENTO_H=\"mupdf/memento.h\"", + }, + }); + + const library = b.addLibrary(.{ + .name = "mupdf", + .linkage = .static, + .root_module = module, + }); + if (options.target.result.os.tag != .windows) + module.linkSystemLibrary("m", .{}); + + return .{ .dependency = dependency, .library = library }; +} + +/// Add a link-complete smoke test. Unlike merely producing an archive, this +/// exercises document registration, page loading, rasterization and search, +/// so an omitted source or codec symbol fails the build before application +/// integration starts. +pub fn addProbe(b: *std.Build, result: Result, options: struct { + target: std.Build.ResolvedTarget, + pdf: std.Build.LazyPath, +}) *std.Build.Step { + const generated = b.addWriteFiles(); + const source = generated.add("mupdf-probe.c", + \\#include + \\#include + \\ + \\int main(int argc, char **argv) + \\{ + \\ fz_context *ctx = NULL; + \\ fz_document *doc = NULL; + \\ fz_page *page = NULL; + \\ fz_pixmap *pix = NULL; + \\ fz_stext_page *text = NULL; + \\ fz_search *search = NULL; + \\ fz_search_result search_result; + \\ int pages = 0, hit_count = 0, failed = 0; + \\ + \\ if (argc != 2) return 2; + \\ ctx = fz_new_context(NULL, NULL, FZ_STORE_DEFAULT); + \\ if (!ctx) return 3; + \\ fz_var(doc); + \\ fz_var(page); + \\ fz_var(pix); + \\ fz_var(text); + \\ fz_var(search); + \\ fz_try(ctx) + \\ { + \\ fz_register_document_handlers(ctx); + \\ doc = fz_open_document(ctx, argv[1]); + \\ pages = fz_count_pages(ctx, doc); + \\ if (pages < 1) fz_throw(ctx, FZ_ERROR_FORMAT, "empty document"); + \\ page = fz_load_page(ctx, doc, 0); + \\ pix = fz_new_pixmap_from_page(ctx, page, fz_scale(0.25f, 0.25f), fz_device_rgb(ctx), 0); + \\ text = fz_new_stext_page_from_page(ctx, page, NULL); + \\ search = fz_new_search(ctx); + \\ fz_search_set_options(ctx, search, FZ_SEARCH_IGNORE_CASE, "Pardes"); + \\ fz_feed_search(ctx, search, fz_keep_stext_page(ctx, text), 0); + \\ for (;;) + \\ { + \\ search_result = fz_search_forwards(ctx, search); + \\ if (search_result.reason == FZ_SEARCH_COMPLETE) break; + \\ if (search_result.reason == FZ_SEARCH_MORE_INPUT) + \\ fz_feed_search(ctx, search, NULL, search_result.u.more_input.seq_needed); + \\ else if (search_result.reason == FZ_SEARCH_MATCH) + \\ ++hit_count; + \\ else + \\ fz_throw(ctx, FZ_ERROR_FORMAT, "invalid iterative search state"); + \\ } + \\ if (!pix || pix->w < 1 || pix->h < 1 || hit_count < 1) + \\ fz_throw(ctx, FZ_ERROR_FORMAT, "render/search probe failed"); + \\ printf("MuPDF probe: %d pages, %d hit(s), first page %dx%d pixels\n", pages, hit_count, pix->w, pix->h); + \\ } + \\ fz_catch(ctx) + \\ { + \\ fz_report_error(ctx); + \\ failed = 1; + \\ } + \\ fz_drop_search(ctx, search); + \\ fz_drop_stext_page(ctx, text); + \\ fz_drop_pixmap(ctx, pix); + \\ fz_drop_page(ctx, page); + \\ fz_drop_document(ctx, doc); + \\ fz_drop_context(ctx); + \\ return failed; + \\} + ); + + const probe = b.addExecutable(.{ + .name = "pardes-mupdf-probe", + .root_module = b.createModule(.{ + .target = options.target, + .optimize = .ReleaseSafe, + .link_libc = true, + }), + }); + result.linkTo(probe.root_module); + probe.root_module.addCSourceFile(.{ .file = source, .flags = &.{"-std=gnu11"} }); + + const run = b.addRunArtifact(probe); + run.addFileArg(options.pdf); + run.expectExitCode(0); + return &run.step; +} + +fn configureModule(module: *std.Build.Module, dependency: *std.Build.Dependency) void { + module.addIncludePath(dependency.path("include")); + module.addIncludePath(dependency.path("source/fitz")); + module.addIncludePath(dependency.path("source/pdf")); + module.addIncludePath(dependency.path("thirdparty/freetype/include")); + module.addIncludePath(dependency.path("scripts/freetype")); + module.addIncludePath(dependency.path("thirdparty/libjpeg")); + module.addIncludePath(dependency.path("scripts/libjpeg")); + module.addIncludePath(dependency.path("thirdparty/zlib")); + module.addIncludePath(dependency.path("thirdparty/jbig2dec")); + // encode-jpx.c includes this header before its FZ_ENABLE_JPX guard. + module.addIncludePath(dependency.path("thirdparty/openjpeg/src/lib/openjp2")); + + inline for (.{ + .{ "FZ_ENABLE_PDF", "1" }, + .{ "FZ_ENABLE_XPS", "0" }, + .{ "FZ_ENABLE_SVG", "0" }, + .{ "FZ_ENABLE_CBZ", "0" }, + .{ "FZ_ENABLE_IMG", "0" }, + .{ "FZ_ENABLE_HTML", "0" }, + .{ "FZ_ENABLE_FB2", "0" }, + .{ "FZ_ENABLE_MOBI", "0" }, + .{ "FZ_ENABLE_EPUB", "0" }, + .{ "FZ_ENABLE_OFFICE", "0" }, + .{ "FZ_ENABLE_TXT", "0" }, + .{ "FZ_ENABLE_HTML_ENGINE", "0" }, + .{ "FZ_ENABLE_OCR_OUTPUT", "0" }, + .{ "FZ_ENABLE_ODT_OUTPUT", "0" }, + .{ "FZ_ENABLE_DOCX_OUTPUT", "0" }, + .{ "FZ_ENABLE_JS", "0" }, + .{ "FZ_ENABLE_BARCODE", "0" }, + .{ "FZ_ENABLE_BROTLI", "0" }, + .{ "FZ_ENABLE_ICC", "0" }, + .{ "FZ_ENABLE_JPX", "0" }, + }) |macro| module.addCMacro(macro[0], macro[1]); + module.addCMacro("OCR_DISABLED", "1"); + module.addCMacro("TOFU", "1"); + module.addCMacro("TOFU_CJK", "1"); +} + +fn flatCSources(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency, relative_dir: []const u8) []const []const u8 { + var dir = std.Io.Dir.cwd().openDir(io, dependency.path(relative_dir).getPath(b), .{ .iterate = true }) catch + std.debug.panic("MuPDF source directory is missing: {s}", .{relative_dir}); + defer dir.close(io); + + var files: std.ArrayList([]const u8) = .empty; + var it = dir.iterate(); + while (it.next(io) catch @panic("read MuPDF source directory")) |entry| { + // Some filesystems report d_type as unknown. The selected directories + // are flat and immutable, so the suffix is the reliable predicate. + if (!std.mem.endsWith(u8, entry.name, ".c")) continue; + files.append(b.allocator, b.fmt("{s}/{s}", .{ relative_dir, entry.name })) catch @panic("OOM"); + } + std.mem.sort([]const u8, files.items, {}, lessThan); + if (files.items.len == 0) std.debug.panic("MuPDF source directory has no C files: {s}", .{relative_dir}); + return files.items; +} + +fn readMakelists(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency) []const u8 { + return std.Io.Dir.cwd().readFileAlloc( + io, + dependency.path("Makelists").getPath(b), + b.allocator, + .limited(1024 * 1024), + ) catch @panic("read MuPDF Makelists"); +} + +fn makeSources(b: *std.Build, makelists: []const u8, variable: []const u8) []const []const u8 { + var files: std.ArrayList([]const u8) = .empty; + var lines = std.mem.splitScalar(u8, makelists, '\n'); + while (lines.next()) |raw_line| { + const line = std.mem.trim(u8, raw_line, " \t\r"); + if (!std.mem.startsWith(u8, line, variable)) continue; + const assignment = std.mem.trimStart(u8, line[variable.len..], " \t"); + if (!std.mem.startsWith(u8, assignment, "+=")) continue; + const path = std.mem.trim(u8, assignment[2..], " \t\r"); + if (!std.mem.endsWith(u8, path, ".c")) continue; + files.append(b.allocator, b.dupe(path)) catch @panic("OOM"); + } + if (files.items.len == 0) std.debug.panic("MuPDF Makelists variable has no C sources: {s}", .{variable}); + return files.items; +} + +fn lessThan(_: void, a: []const u8, b: []const u8) bool { + return std.mem.order(u8, a, b) == .lt; +} -- cgit v1.3