const std = @import("std"); /// MuPDF 1.27.0's release archive does not publish a Zig build. Keep its /// upstream source-of-truth (`Makelists`) and compile the selected C sources /// as an ordinary Zig static library. This is intentionally a PDF-only first /// slice: the application needs rendering and structured text, not MuPDF's /// other document handlers, tools, Javascript engine, OCR, or writers. pub const Result = struct { dependency: *std.Build.Dependency, library: *std.Build.Step.Compile, /// Whether this archive carries the JPEG 2000 decoder. Remembered rather /// than re-derived because `linkTo` has to hand consumers the SAME /// preprocessor view the archive was built with — `FZ_ENABLE_JPX` reaches /// the public headers, and a consumer that disagrees is an ODR bug that /// only shows up as a wrong struct layout at runtime. jpx: bool, /// Give a Zig module the same preprocessor view as the library, expose the /// public C headers to @cImport, and propagate the static link. pub fn linkTo(self: Result, module: *std.Build.Module) void { configureModule(module, self.dependency, self.jpx); // MuPDF's public context headers select the lock-debugging contract // from NDEBUG. Every consumer must therefore agree with the archive, // even when its own Zig optimization/safety mode differs. module.addCMacro("NDEBUG", "1"); module.linkLibrary(self.library); } }; pub fn add(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency, options: struct { target: std.Build.ResolvedTarget, optimize: std.builtin.OptimizeMode, /// Compile openjpeg and let MuPDF decode JPEG 2000. See the OPENJPEG /// block below for why this is a switch rather than always-on. jpx: bool = true, /// The FreeType the rest of the build links, which MuPDF is compiled /// against and links instead of the slim one it bundles (FREETYPE_SRC, /// configured by scripts/freetype). Both define the same symbols and a /// linker handed two keeps one of each, so the gui's MuPDF ran the full /// FreeType's code compiled against the slim one's headers, while the tty /// ran the slim one. MuPDF asks nothing of FreeType the full /// configuration lacks. freetype: *std.Build.Step.Compile, /// The zlib FreeType links, which MuPDF links instead of compiling the /// one it bundles (ZLIB_SRC): the two define the same symbols, as the /// FreeTypes do. zlib: *std.Build.Step.Compile, }) Result { const module = b.createModule(.{ .target = options.target, .optimize = options.optimize, .link_libc = true, }); configureModule(module, dependency, options.jpx); // MuPDF's upstream release build defines NDEBUG independently of the // optimizer. Without it, an optimized Zig build still enables Fitz's // debug-locking/assertion paths. linkTo applies the same public-header // contract to consumers; keeping it here also covers every archive source. module.addCMacro("NDEBUG", "1"); // The build's own FreeType and zlib, headers and all; see above. module.linkLibrary(options.freetype); module.linkLibrary(options.zlib); // MuPDF itself uses wildcards for these two directories. The dependency // is immutable and content-addressed, so discovering the flat C lists at // configure time is deterministic; sorting prevents filesystem order from // perturbing Zig's cache key. module.addCSourceFiles(.{ .root = dependency.path(""), .files = flatCSources(b, io, dependency, "source/fitz"), // MuPDF's C uses intentional patterns that Zig/Clang's trap-mode // UBSan rejects in ReleaseSafe (opening any document traps in // fz_drop_archive). Keep ReleaseSafe's optimization while leaving // undefined-behaviour instrumentation off for this upstream layer. .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined" }, }); module.addCSourceFiles(.{ .root = dependency.path(""), .files = flatCSources(b, io, dependency, "source/pdf"), .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined" }, }); // TOFU + TOFU_CJK leaves the Base-14 PDF fonts enabled. The official // source release ships their generated C blobs, so no host-side objcopy or // shell generator is involved. module.addCSourceFiles(.{ .root = dependency.path(""), .files = flatCSources(b, io, dependency, "generated/resources/fonts/urw"), // These are large byte arrays; optimization only wastes build time. // This matches MuPDF's own generated-resource compilation rule. .flags = &.{ "-std=gnu11", "-O0" }, }); // Parse only the upstream variables we actually select. These lists are // the same ones MuPDF's Makefile places in libmupdf-third.a; keeping the // names in Makelists avoids a large, easy-to-stale duplicate in this file. const makelists = readMakelists(b, io, dependency); module.addCSourceFiles(.{ .root = dependency.path(""), .files = makeSources(b, makelists, "LIBJPEG_SRC"), // libjpeg calls back into source/fitz (filter-dct.c's source // manager), which is built without UBSan: -fsanitize=function at // such a call finds no type hash on the callee and traps, so any // PDF with a DCT image crashed ReleaseSafe and Debug builds in // jpeg_fill_bit_buffer. The same holds for jbig2dec below. .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined" }, }); module.addCSourceFiles(.{ .root = dependency.path(""), .files = makeSources(b, makelists, "JBIG2DEC_SRC"), .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined", "-DHAVE_STDINT_H", "-DJBIG_EXTERNAL_MEMENTO_H=\"mupdf/memento.h\"", }, }); // --- OPENJPEG --- // // The JPEG 2000 decoder, and the reason a scanned PDF is not a blank page: // a scan is typically one /JPXDecode image per page, so without this MuPDF // raises "JPX support disabled" for every one of them and hands back a // page with nothing drawn on it. Everything else still works — the pane // opens, the page count is right — which makes it look like a renderer bug // rather than a missing codec. // // A switch rather than always-on because it is 31 files of third-party C // that exists to parse untrusted input, and openjpeg has the CVE history // to match. Default on: a PDF viewer that cannot open scans is the more // surprising default, and the ones that matter are exactly the documents // nobody can retypeset. // // Flags are MuPDF's own OPENJPEG_CFLAGS plus OPENJPEG_BUILD_CFLAGS from // Makelists; the include path is already on the module (configureModule // adds it unconditionally, because encode-jpx.c wants the header even when // the codec is off). -fno-sanitize=undefined for the reason source/fitz // needs it: this is upstream C full of deliberate wrapping arithmetic, and // ReleaseSafe's trap-mode UBSan would turn a legal shift into a crash. if (options.jpx) module.addCSourceFiles(.{ .root = dependency.path(""), .files = makeSources(b, makelists, "OPENJPEG_SRC"), .flags = &.{ "-std=gnu11", "-fno-sanitize=undefined", "-DOPJ_STATIC", "-DOPJ_HAVE_INTTYPES_H", "-DOPJ_HAVE_STDINT_H", "-DMUTEX_pthread=0", }, }); const library = b.addLibrary(.{ .name = "mupdf", .linkage = .static, .root_module = module, }); if (options.target.result.os.tag != .windows) module.linkSystemLibrary("m", .{}); return .{ .dependency = dependency, .library = library, .jpx = options.jpx }; } /// Add a link-complete smoke test. Unlike merely producing an archive, this /// exercises document registration, page loading, rasterization and search, /// so an omitted source or codec symbol fails the build before application /// integration starts. pub fn addProbe(b: *std.Build, result: Result, options: struct { target: std.Build.ResolvedTarget, pdf: std.Build.LazyPath, }) *std.Build.Step { const generated = b.addWriteFiles(); const source = generated.add("mupdf-probe.c", \\#include \\#include \\ \\int main(int argc, char **argv) \\{ \\ fz_context *ctx = NULL; \\ fz_document *doc = NULL; \\ fz_page *page = NULL; \\ fz_pixmap *pix = NULL; \\ fz_stext_page *text = NULL; \\ fz_search *search = NULL; \\ fz_search_result search_result; \\ int pages = 0, hit_count = 0, failed = 0; \\ \\ if (argc != 2) return 2; \\ ctx = fz_new_context(NULL, NULL, FZ_STORE_DEFAULT); \\ if (!ctx) return 3; \\ fz_var(doc); \\ fz_var(page); \\ fz_var(pix); \\ fz_var(text); \\ fz_var(search); \\ fz_try(ctx) \\ { \\ fz_register_document_handlers(ctx); \\ doc = fz_open_document(ctx, argv[1]); \\ pages = fz_count_pages(ctx, doc); \\ if (pages < 1) fz_throw(ctx, FZ_ERROR_FORMAT, "empty document"); \\ page = fz_load_page(ctx, doc, 0); \\ pix = fz_new_pixmap_from_page(ctx, page, fz_scale(0.25f, 0.25f), fz_device_rgb(ctx), 0); \\ text = fz_new_stext_page_from_page(ctx, page, NULL); \\ search = fz_new_search(ctx); \\ fz_search_set_options(ctx, search, FZ_SEARCH_IGNORE_CASE, "Pardes"); \\ fz_feed_search(ctx, search, fz_keep_stext_page(ctx, text), 0); \\ for (;;) \\ { \\ search_result = fz_search_forwards(ctx, search); \\ if (search_result.reason == FZ_SEARCH_COMPLETE) break; \\ if (search_result.reason == FZ_SEARCH_MORE_INPUT) \\ fz_feed_search(ctx, search, NULL, search_result.u.more_input.seq_needed); \\ else if (search_result.reason == FZ_SEARCH_MATCH) \\ ++hit_count; \\ else \\ fz_throw(ctx, FZ_ERROR_FORMAT, "invalid iterative search state"); \\ } \\ if (!pix || pix->w < 1 || pix->h < 1 || hit_count < 1) \\ fz_throw(ctx, FZ_ERROR_FORMAT, "render/search probe failed"); \\ printf("MuPDF probe: %d pages, %d hit(s), first page %dx%d pixels\n", pages, hit_count, pix->w, pix->h); \\ } \\ fz_catch(ctx) \\ { \\ fz_report_error(ctx); \\ failed = 1; \\ } \\ fz_drop_search(ctx, search); \\ fz_drop_stext_page(ctx, text); \\ fz_drop_pixmap(ctx, pix); \\ fz_drop_page(ctx, page); \\ fz_drop_document(ctx, doc); \\ fz_drop_context(ctx); \\ return failed; \\} ); const probe = b.addExecutable(.{ .name = "pardes-mupdf-probe", .root_module = b.createModule(.{ .target = options.target, .optimize = .ReleaseSafe, .link_libc = true, }), }); result.linkTo(probe.root_module); probe.root_module.addCSourceFile(.{ .file = source, .flags = &.{"-std=gnu11"} }); const run = b.addRunArtifact(probe); run.addFileArg(options.pdf); run.expectExitCode(0); return &run.step; } fn configureModule(module: *std.Build.Module, dependency: *std.Build.Dependency, jpx: bool) void { module.addIncludePath(dependency.path("include")); module.addIncludePath(dependency.path("source/fitz")); module.addIncludePath(dependency.path("source/pdf")); module.addIncludePath(dependency.path("thirdparty/libjpeg")); module.addIncludePath(dependency.path("scripts/libjpeg")); module.addIncludePath(dependency.path("thirdparty/jbig2dec")); // encode-jpx.c includes this header before its FZ_ENABLE_JPX guard. module.addIncludePath(dependency.path("thirdparty/openjpeg/src/lib/openjp2")); inline for (.{ .{ "FZ_ENABLE_PDF", "1" }, .{ "FZ_ENABLE_XPS", "0" }, .{ "FZ_ENABLE_SVG", "0" }, .{ "FZ_ENABLE_CBZ", "0" }, .{ "FZ_ENABLE_IMG", "0" }, .{ "FZ_ENABLE_HTML", "0" }, .{ "FZ_ENABLE_FB2", "0" }, .{ "FZ_ENABLE_MOBI", "0" }, .{ "FZ_ENABLE_EPUB", "0" }, .{ "FZ_ENABLE_OFFICE", "0" }, .{ "FZ_ENABLE_TXT", "0" }, .{ "FZ_ENABLE_HTML_ENGINE", "0" }, .{ "FZ_ENABLE_OCR_OUTPUT", "0" }, .{ "FZ_ENABLE_ODT_OUTPUT", "0" }, .{ "FZ_ENABLE_DOCX_OUTPUT", "0" }, .{ "FZ_ENABLE_JS", "0" }, .{ "FZ_ENABLE_BARCODE", "0" }, .{ "FZ_ENABLE_BROTLI", "0" }, .{ "FZ_ENABLE_ICC", "0" }, }) |macro| module.addCMacro(macro[0], macro[1]); // The one codec that is a build option rather than a fixed answer, and it // reaches the public headers — so consumers get it through linkTo from the // archive's own `jpx`, never re-derived. See the OPENJPEG block in add(). module.addCMacro("FZ_ENABLE_JPX", if (jpx) "1" else "0"); module.addCMacro("OCR_DISABLED", "1"); module.addCMacro("TOFU", "1"); module.addCMacro("TOFU_CJK", "1"); } fn flatCSources(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency, relative_dir: []const u8) []const []const u8 { var dir = std.Io.Dir.cwd().openDir(io, dependency.path(relative_dir).getPath(b), .{ .iterate = true }) catch std.debug.panic("MuPDF source directory is missing: {s}", .{relative_dir}); defer dir.close(io); const max_flat_sources = 4096; var bounded: [max_flat_sources][]const u8 = undefined; var count: usize = 0; var it = dir.iterate(); while (it.next(io) catch @panic("read MuPDF source directory")) |entry| { // Some filesystems report d_type as unknown. The selected directories // are flat and immutable, so the suffix is the reliable predicate. if (!std.mem.endsWith(u8, entry.name, ".c")) continue; if (count == bounded.len) std.debug.panic("MuPDF source directory exceeds {d} C files: {s}", .{ bounded.len, relative_dir }); bounded[count] = b.fmt("{s}/{s}", .{ relative_dir, entry.name }); count += 1; } if (count == 0) std.debug.panic("MuPDF source directory has no C files: {s}", .{relative_dir}); const files = b.allocator.alloc([]const u8, count) catch @panic("OOM"); @memcpy(files, bounded[0..count]); std.mem.sort([]const u8, files, {}, lessThan); return files; } fn readMakelists(b: *std.Build, io: std.Io, dependency: *std.Build.Dependency) []const u8 { return std.Io.Dir.cwd().readFileAlloc( io, dependency.path("Makelists").getPath(b), b.allocator, .limited(1024 * 1024), ) catch @panic("read MuPDF Makelists"); } fn makeSources(b: *std.Build, makelists: []const u8, variable: []const u8) []const []const u8 { var count: usize = 0; var lines = std.mem.splitScalar(u8, makelists, '\n'); while (lines.next()) |raw_line| { const line = std.mem.trim(u8, raw_line, " \t\r"); if (!std.mem.startsWith(u8, line, variable)) continue; const assignment = std.mem.trimStart(u8, line[variable.len..], " \t"); if (!std.mem.startsWith(u8, assignment, "+=")) continue; const path = std.mem.trim(u8, assignment[2..], " \t\r"); if (std.mem.endsWith(u8, path, ".c")) count += 1; } if (count == 0) std.debug.panic("MuPDF Makelists variable has no C sources: {s}", .{variable}); const files = b.allocator.alloc([]const u8, count) catch @panic("OOM"); lines = std.mem.splitScalar(u8, makelists, '\n'); var file_i: usize = 0; while (lines.next()) |raw_line| { const line = std.mem.trim(u8, raw_line, " \t\r"); if (!std.mem.startsWith(u8, line, variable)) continue; const assignment = std.mem.trimStart(u8, line[variable.len..], " \t"); if (!std.mem.startsWith(u8, assignment, "+=")) continue; const path = std.mem.trim(u8, assignment[2..], " \t\r"); if (!std.mem.endsWith(u8, path, ".c")) continue; files[file_i] = b.dupe(path); file_i += 1; } return files; } fn lessThan(_: void, a: []const u8, b: []const u8) bool { return std.mem.order(u8, a, b) == .lt; }