From 87f69a12ee28445cd95e39060fd626ccfc27b5e0 Mon Sep 17 00:00:00 2001 From: Mitchell Hashimoto Date: Fri, 14 Aug 2026 09:10:46 -0700 Subject: [PATCH] libghostty: much faster vt_write on wasm targets This makes `ghostty_terminal_vt_write` on wasm32-freestanding anywhere from 1.4x to 13x faster depending on the input, measured in V8 via Node for Chrome as well as `jsc` for Safari. ## Changes * stream: the batched parse path (bulk UTF-8 decode, print_slice runs) is used even when `build_options.simd` is false. The per-byte loop is now debug-only. * simd/vt: the scalar `utf8DecodeUntilControlSeq` gets a vectorized ASCII bulk path that is compatible with wasm simd128. * style: on wasm, `Style.eql` compares canonical `PackedStyle` forms which is faster by like 11%. On native its slower so we only do this for wasm. * build: wasm targets now default to the `simd128` CPU feature since every browser engine has supported it for years. Opt out with `-Dcpu=generic`. * PACKAGING.md documents the wasm build, including `wasm-opt` notes. ## Benchmarks | Workload | Before | After | Speedup | |---|---|---|---| | ascii | 85 MB/s | 1070 MB/s | 12.5x | | ascii-wrap | 84 MB/s | 1103 MB/s | 13.1x | | clear-redraw | 85 MB/s | 913 MB/s | 10.7x | | scroll | 79 MB/s | 304 MB/s | 3.8x | | cursor | 120 MB/s | 255 MB/s | 2.1x | | utf8 | 99 MB/s | 169 MB/s | 1.7x | | sgr16 | 81 MB/s | 133 MB/s | 1.6x | | sgr-truecolor | 62 MB/s | 88 MB/s | 1.4x | End result: wasm at roughly 50-85% of the native ReleaseFast+SIMD build on the same workloads. Plain ASCII was at 6% of native before. **AI usage:** Lots of Fable help. As always, the human language stuff like this commit and comments were rewritten by me. --- PACKAGING.md | 32 +++++++++++++ nix/devShell.nix | 8 ++++ src/build/Config.zig | 19 ++++++++ src/simd/vt.zig | 99 ++++++++++++++++++++--------------------- src/terminal/stream.zig | 9 ++-- src/terminal/style.zig | 11 +++++ 6 files changed, 124 insertions(+), 54 deletions(-) diff --git a/PACKAGING.md b/PACKAGING.md index 1483b8591..ec2ef3dee 100644 --- a/PACKAGING.md +++ b/PACKAGING.md @@ -122,3 +122,35 @@ relevant to package maintainers: often necessary for system packages to specify a specific minimum Linux version, glibc, etc. Run `zig targets` to a get a full list of available targets. + +## WebAssembly (libghostty-vt) + +libghostty-vt can be built for WebAssembly for use in browsers and other +wasm runtimes: + +```sh +zig build -Demit-lib-vt -Dtarget=wasm32-freestanding -Doptimize=ReleaseSmall +``` + +This produces `zig-out/bin/ghostty-vt.wasm`. + +Some notes for packaging the wasm module: + +- The build enables the `simd128` feature by default. Every browser engine + has supported it for years (Chrome 91, Firefox 89, Safari 16.4) and it is + a large performance win for VT parsing. If you target an unusual runtime + without SIMD support, opt out with `-Dcpu=generic`. + +- `ReleaseSmall` is the recommended optimization mode for the web. Running + the result through [Binaryen's](https://github.com/WebAssembly/binaryen) + `wasm-opt -O3` shrinks it by roughly a further 10% without hurting + performance. + +- `ReleaseFast` measures 10-20% faster than `ReleaseSmall` on escape-heavy + terminal workloads, but the artifact is dominated by DWARF debug info. + If you want the speed, strip it: `wasm-opt -O3 --strip-dwarf` reduces a + ReleaseFast build from over 5MB to roughly 1.1MB (versus roughly 0.8MB + for ReleaseSmall). When invoking `wasm-opt`, pass the feature flags for + what the module uses, e.g. `--enable-simd --enable-bulk-memory +--enable-sign-ext --enable-nontrapping-float-to-int --enable-multivalue +--enable-reference-types`. diff --git a/nix/devShell.nix b/nix/devShell.nix index 609a8a31f..c94892414 100644 --- a/nix/devShell.nix +++ b/nix/devShell.nix @@ -24,7 +24,11 @@ #, vulkan-loader # unused vttest, wabt, + wasm-tools, wasmtime, + binaryen, + twiggy, + wizer, wraptest, zig, zip, @@ -127,8 +131,12 @@ in kaitai-struct-compiler # wasm + binaryen + twiggy wabt + wasm-tools wasmtime + wizer # Localization gettext diff --git a/src/build/Config.zig b/src/build/Config.zig index 07e047ef5..09ab1a915 100644 --- a/src/build/Config.zig +++ b/src/build/Config.zig @@ -106,6 +106,25 @@ pub fn init(b: *std.Build, appVersion: []const u8, libVersion: []const u8) !Conf result = b.resolveTargetQuery(query); } + // On wasm, default to enabling the simd128 feature. Every + // browser engine has supported it since 2023 or earlier + // (Chrome 91, Firefox 89, Safari 16.4) and it is a large + // performance win for the terminal hot paths (50%+ on + // print-heavy VT streams). Only apply when no explicit CPU + // was requested; targets for exotic non-browser runtimes + // can opt out with `-Dcpu=generic`. + if (result.result.cpu.arch.isWasm() and + result.query.cpu_model == .determined_by_arch_os and + result.query.cpu_features_add.isEmpty() and + result.query.cpu_features_sub.isEmpty()) + { + var query = result.query; + query.cpu_features_add.addFeature( + @intFromEnum(std.Target.wasm.Feature.simd128), + ); + result = b.resolveTargetQuery(query); + } + // The full Ghostty build no longer supports iOS; Fail early // with a clear message rather than partway through the build. if (result.result.os.tag == .ios and !emit_lib_vt) { diff --git a/src/simd/vt.zig b/src/simd/vt.zig index 807285f71..ded4ab3f5 100644 --- a/src/simd/vt.zig +++ b/src/simd/vt.zig @@ -53,11 +53,29 @@ fn utf8DecodeUntilControlSeqScalar( while (decode_offset < decode.len) { const b0 = decode[decode_offset]; - // ASCII fast path + // ASCII fast path. Use vectorization if it is available. This + // path is only run when simd=false, but that only controls our C++ + // simd builds. We can still rely on Zig intrinsics for platforms + // like wasm32+simd128. if (b0 < 0x80) { - output[decode_count] = b0; - decode_count += 1; - decode_offset += 1; + if (comptime std.simd.suggestVectorLength(u8)) |vl| { + const V = @Vector(vl, u8); + while (decode_offset + vl <= decode.len) { + const v: V = decode[decode_offset..][0..vl].*; + if (@reduce(.Or, v >= @as(V, @splat(0x80)))) break; + const w: @Vector(vl, u32) = @intCast(v); + output[decode_count..][0..vl].* = w; + decode_count += vl; + decode_offset += vl; + } + } + while (decode_offset < decode.len) { + const b = decode[decode_offset]; + if (b >= 0x80) break; + output[decode_count] = b; + decode_count += 1; + decode_offset += 1; + } continue; } @@ -70,15 +88,27 @@ fn utf8DecodeUntilControlSeqScalar( continue; } - // Multi-byte sequence. Determine expected length and the valid - // range for each continuation byte per Unicode Table 3-7. - const seq = utf8SeqInfo(b0); + // Multi-byte sequence. Only the first continuation byte has a + // lead-dependent valid range per Unicode Table 3-7; later + // continuation bytes are always 80-BF. Range validity per + // Table 3-7 excludes overlong, surrogate, and out-of-range + // encodings, so a fully valid sequence can be decoded by + // direct bit assembly with no further checks. + const seq_len: usize = if (b0 < 0xE0) 2 else if (b0 < 0xF0) 3 else 4; + const cb1_lo: u8, const cb1_hi: u8 = switch (b0) { + 0xE0 => .{ 0xA0, 0xBF }, + 0xED => .{ 0x80, 0x9F }, + 0xF0 => .{ 0x90, 0xBF }, + 0xF4 => .{ 0x80, 0x8F }, + else => .{ 0x80, 0xBF }, + }; // Check how many continuation bytes form a valid prefix (the - // maximal subpart). We check each byte against its specific - // valid range. + // maximal subpart), accumulating codepoint bits as we go. The + // lead byte contributes its low 7-len bits. + var cp: u32 = b0 & (@as(u32, 0x7F) >> @intCast(seq_len)); var valid: usize = 1; // lead byte is valid - for (0..seq.len - 1) |ci| { + while (valid < seq_len) { if (decode_offset + valid >= decode.len) { // The sequence is cut off by the end of the decode // region. If the region ends at the true end of the @@ -94,29 +124,17 @@ fn utf8DecodeUntilControlSeqScalar( break; } const cb = decode[decode_offset + valid]; - if (cb < seq.ranges[ci][0] or cb > seq.ranges[ci][1]) { - // Byte doesn't match expected range. The maximal - // subpart ends here. - break; - } + const lo: u8 = if (valid == 1) cb1_lo else 0x80; + const hi: u8 = if (valid == 1) cb1_hi else 0xBF; + if (cb < lo or cb > hi) break; + cp = (cp << 6) | (cb & 0x3F); valid += 1; } - if (valid == seq.len) { - // Full sequence present and structurally valid. Decode it. - // (Structural validity per Table 3-7 guarantees decode success.) - const cp_bytes = decode[decode_offset..][0..seq.len]; - if (std.unicode.utf8Decode(cp_bytes)) |cp| { - output[decode_count] = @intCast(cp); - decode_count += 1; - decode_offset += seq.len; - } else |_| { - // Should not happen given Table 3-7 validation, but - // be safe: emit FFFD for the lead byte. - output[decode_count] = 0xFFFD; - decode_count += 1; - decode_offset += 1; - } + if (valid == seq_len) { + output[decode_count] = cp; + decode_count += 1; + decode_offset += seq_len; } else { // Incomplete/ill-formed: the maximal subpart (valid bytes) // maps to a single FFFD. @@ -132,27 +150,6 @@ fn utf8DecodeUntilControlSeqScalar( }; } -const Utf8SeqInfo = struct { - len: u3, - ranges: [3][2]u8, -}; - -/// Returns the expected byte count and valid continuation byte ranges -/// for a UTF-8 sequence based on its lead byte, per Unicode Table 3-7. -fn utf8SeqInfo(lead: u8) Utf8SeqInfo { - return switch (lead) { - 0xC2...0xDF => .{ .len = 2, .ranges = .{ .{ 0x80, 0xBF }, .{ 0, 0 }, .{ 0, 0 } } }, - 0xE0 => .{ .len = 3, .ranges = .{ .{ 0xA0, 0xBF }, .{ 0x80, 0xBF }, .{ 0, 0 } } }, - 0xE1...0xEC => .{ .len = 3, .ranges = .{ .{ 0x80, 0xBF }, .{ 0x80, 0xBF }, .{ 0, 0 } } }, - 0xED => .{ .len = 3, .ranges = .{ .{ 0x80, 0x9F }, .{ 0x80, 0xBF }, .{ 0, 0 } } }, - 0xEE...0xEF => .{ .len = 3, .ranges = .{ .{ 0x80, 0xBF }, .{ 0x80, 0xBF }, .{ 0, 0 } } }, - 0xF0 => .{ .len = 4, .ranges = .{ .{ 0x90, 0xBF }, .{ 0x80, 0xBF }, .{ 0x80, 0xBF } } }, - 0xF1...0xF3 => .{ .len = 4, .ranges = .{ .{ 0x80, 0xBF }, .{ 0x80, 0xBF }, .{ 0x80, 0xBF } } }, - 0xF4 => .{ .len = 4, .ranges = .{ .{ 0x80, 0x8F }, .{ 0x80, 0xBF }, .{ 0x80, 0xBF } } }, - else => unreachable, - }; -} - // Differential test: the SIMD implementation must agree with the // scalar implementation on any input. Exercises random mixtures of // ASCII, escapes, controls, valid and invalid UTF-8, at various diff --git a/src/terminal/stream.zig b/src/terminal/stream.zig index 729b6fc04..afbfa719c 100644 --- a/src/terminal/stream.zig +++ b/src/terminal/stream.zig @@ -646,9 +646,12 @@ pub fn Stream(comptime H: type) type { } inline fn nextSliceUntracked(self: *Self, input: []const u8) void { - // Disable SIMD optimizations if build requests it or if our - // manual debug mode is on. - if (comptime debug or !build_options.simd) { + // Byte-at-a-time parsing in debug mode only. Even without + // SIMD support (build_options.simd == false, e.g. wasm), the + // batched path below is much faster than per-byte dispatch + // because it decodes UTF-8 in bulk (via scalar fallbacks) and + // hands printable runs to the handler as print_slice actions. + if (comptime debug) { for (input) |c| self.nextUntracked(c); return; } diff --git a/src/terminal/style.zig b/src/terminal/style.zig index ee4c0e562..ce20e4fac 100644 --- a/src/terminal/style.zig +++ b/src/terminal/style.zig @@ -1,4 +1,5 @@ const std = @import("std"); +const builtin = @import("builtin"); const assert = @import("../quirks.zig").inlineAssert; const fastprint = @import("../fastprint.zig"); const color = @import("color.zig"); @@ -94,6 +95,16 @@ pub const Style = struct { /// True if the style is equal to another style. pub fn eql(self: Style, other: Style) bool { + // On wasm, comparing packed forms wins (~13% on SGR-heavy streams). + // On native, the branchy early-exit field compare below measures + // ~5% faster (unequal styles usually differ in the first + // field or two), so each target keeps its own strategy. + if (comptime builtin.cpu.arch.isWasm()) { + const a: u128 = @bitCast(PackedStyle.fromStyle(self)); + const b: u128 = @bitCast(PackedStyle.fromStyle(other)); + return a == b; + } + return self.flags == other.flags and self.fg_color.eql(other.fg_color) and self.bg_color.eql(other.bg_color) and