//! Dependency-free tree-sitter grammar metadata shared by build.zig and the //! runtime syntax registry. Keep target symbols and query values out of this //! file: the build runner imports it as host code. const std = @import("std"); pub const Tier = enum { zig, minimal, full }; pub const Grammar = struct { name: []const u8, dep: []const u8, exts: []const []const u8, tier: Tier, src: []const u8 = "src", scanner: bool = false, query: []const u8 = "queries/highlights.scm", }; pub const all = [_]Grammar{ .{ .name = "ada", .dep = "ts_ada", .exts = &.{ ".adb", ".ads", ".ada" }, .tier = .full }, .{ .name = "bash", .dep = "ts_bash", .exts = &.{ ".sh", ".bash", ".zsh" }, .tier = .full, .scanner = true }, .{ .name = "c", .dep = "ts_c", .exts = &.{ ".c", ".h" }, .tier = .minimal }, .{ .name = "c_sharp", .dep = "ts_c_sharp", .exts = &.{ ".cs", ".csx" }, .tier = .full, .scanner = true }, .{ .name = "clojure", .dep = "ts_clojure", .exts = &.{ ".clj", ".cljs", ".cljc", ".edn" }, .tier = .full }, .{ .name = "cpp", .dep = "ts_cpp", .exts = &.{ ".cpp", ".cc", ".cxx", ".hpp", ".hh", ".hxx" }, .tier = .minimal, .scanner = true }, .{ .name = "css", .dep = "ts_css", .exts = &.{".css"}, .tier = .full, .scanner = true }, .{ .name = "elixir", .dep = "ts_elixir", .exts = &.{ ".ex", ".exs" }, .tier = .full, .scanner = true }, .{ .name = "erlang", .dep = "ts_erlang", .exts = &.{ ".erl", ".hrl" }, .tier = .full, .scanner = true }, .{ .name = "fortran", .dep = "ts_fortran", .exts = &.{ ".f", ".for", ".ftn", ".f90", ".f95", ".f03", ".f08" }, .tier = .full, .scanner = true }, .{ .name = "go", .dep = "ts_go", .exts = &.{".go"}, .tier = .full }, .{ .name = "haskell", .dep = "ts_haskell", .exts = &.{ ".hs", ".lhs" }, .tier = .full, .scanner = true }, .{ .name = "html", .dep = "ts_html", .exts = &.{ ".html", ".htm" }, .tier = .full, .scanner = true }, .{ .name = "java", .dep = "ts_java", .exts = &.{".java"}, .tier = .full }, .{ .name = "javascript", .dep = "ts_javascript", .exts = &.{ ".js", ".jsx", ".mjs", ".cjs" }, .tier = .full, .scanner = true }, .{ .name = "json", .dep = "ts_json", .exts = &.{".json"}, .tier = .full }, .{ .name = "kotlin", .dep = "ts_kotlin", .exts = &.{ ".kt", ".kts" }, .tier = .full, .scanner = true }, .{ .name = "ocaml", .dep = "ts_ocaml", .exts = &.{ ".ml", ".mli" }, .tier = .full, .src = "grammars/ocaml/src", .scanner = true }, .{ .name = "markdown", .dep = "ts_markdown", .exts = &.{ ".md", ".markdown" }, .tier = .full, .src = "tree-sitter-markdown/src", .scanner = true, .query = "tree-sitter-markdown/queries/highlights.scm" }, // Second grammar off the SAME dep: the split_parser markdown repo ships a // block parser and an inline parser, and bold/italic/code-span live only in // the inline one. `.exts` is empty because no filename ever selects it — the // block grammar's `inline` nodes are re-parsed with it by name (forLang). .{ .name = "markdown_inline", .dep = "ts_markdown", .exts = &.{}, .tier = .full, .src = "tree-sitter-markdown-inline/src", .scanner = true, .query = "tree-sitter-markdown-inline/queries/highlights.scm" }, .{ .name = "pascal", .dep = "ts_pascal", .exts = &.{ ".pas", ".pp", ".p" }, .tier = .full }, .{ .name = "php", .dep = "ts_php", .exts = &.{ ".php", ".phtml", ".php3", ".php4", ".php5" }, .tier = .full, .src = "php/src", .scanner = true }, .{ .name = "powershell", .dep = "ts_powershell", .exts = &.{ ".ps1", ".psm1", ".psd1" }, .tier = .full, .scanner = true }, .{ .name = "python", .dep = "ts_python", .exts = &.{ ".py", ".pyw" }, .tier = .full, .scanner = true }, .{ .name = "ruby", .dep = "ts_ruby", .exts = &.{ ".rb", ".rake" }, .tier = .full, .scanner = true }, .{ .name = "rust", .dep = "ts_rust", .exts = &.{".rs"}, .tier = .full, .scanner = true }, .{ .name = "scala", .dep = "ts_scala", .exts = &.{ ".scala", ".sc" }, .tier = .full, .scanner = true }, .{ .name = "typst", .dep = "ts_typst", .exts = &.{ ".typ", ".typst" }, .tier = .full, .scanner = true, .query = "queries/typst/highlights.scm" }, .{ .name = "zig", .dep = "ts_zig", .exts = &.{ ".zig", ".zon" }, .tier = .zig }, }; /// The names a language goes by besides its grammar's: a fence's `py`, /// a `Repl sh`. pub const aliases = [_]struct { []const u8, []const u8 }{ .{ "js", "javascript" }, .{ "jsx", "javascript" }, .{ "mjs", "javascript" }, .{ "py", "python" }, .{ "sh", "bash" }, .{ "shell", "bash" }, .{ "zsh", "bash" }, .{ "bash", "bash" }, .{ "rs", "rust" }, .{ "c++", "cpp" }, .{ "cxx", "cpp" }, .{ "cc", "cpp" }, .{ "cs", "c_sharp" }, .{ "kt", "kotlin" }, .{ "rb", "ruby" }, .{ "ml", "ocaml" }, .{ "hs", "haskell" }, }; /// The grammar a language name means: its own name or an alias, in any case. pub fn byName(name: []const u8) ?usize { var canonical = name; for (aliases) |a| if (std.ascii.eqlIgnoreCase(name, a[0])) { canonical = a[1]; break; }; for (all, 0..) |g, i| if (std.ascii.eqlIgnoreCase(canonical, g.name)) return i; return null; } /// The grammar a path's extension selects, in any case. pub fn byPath(path: []const u8) ?usize { const ext = std.fs.path.extension(path); if (ext.len == 0) return null; for (all, 0..) |g, i| for (g.exts) |e| if (std.ascii.eqlIgnoreCase(ext, e)) return i; return null; }