Sample code for 30+ languages & platforms
Zig Requires Chilkat v11.1.0+

Unicode Escape and Unescape Text in StringBuilder

Demonstrates options for unicode escaping non-us-ascii chars and emojis.

Chilkat Zig Downloads

Zig
const std = @import("std");
const chilkat = @import("chilkat");

pub fn main(init: std.process.Init) !void {
    const alloc = init.arena.allocator();

    const sb_original = try chilkat.StringBuilder.init();
    defer sb_original.deinit();
    sb_original.loadFile("qa_data/txt/utf16_emojis_accented_jap.txt", "utf-16") catch {
        std.debug.print("{s}\n", .{try sb_original.getLastErrorText(alloc)});
        return;
    };

    // The above file contains the following text, which includes some emoji's,
    // Japanese chars, and accented chars.

    const sb = try chilkat.StringBuilder.init();
    defer sb.deinit();
    sb.appendSb(sb_original) catch {};

    // Charset is not used for unicode escaping.  Set it to "utf-8", but it means nothing.
    const charset_not_used = "utf-8";

    // Indicate the desired format/style of Unicode escaping.
    // Choose JSON-style (JavaScript-style) Unicode escape sequences by using "unicodeescape"
    var encoding: [:0]const u8 = "unicodeescape";

    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // \ud83e\udde0
    // \ud83d\udd10
    // \u2705
    // \u26a0\ufe0f
    // \u274c
    // \u2713
    // \u4e2d
    // \u00e9 xyz \u00e0
    // abc \u79c1 \u306f \u3093 ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    // Do the same, but use uppercase letters (A-F) in the hex values.
    encoding = "unicodeescape-upper";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // \uD83E\uDDE0
    // \uD83D\uDD10
    // \u2705
    // \u26A0\uFE0F
    // \u274C
    // \u2713
    // \u4E2D
    // \u00E9 xyz \u00E0
    // abc \u79C1 \u306F \u3093 ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    //  ECMAScript (JavaScript) �code point escape� syntax

    encoding = "unicodeescape-curly";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // \u{d83e}\u{dde0}
    // \u{d83d}\u{dd10}
    // \u{2705}
    // \u{26a0}\u{fe0f}
    // \u{274c}
    // \u{2713}
    // \u{4e2d}
    // \u{00e9} xyz \u{00e0}
    // abc \u{79c1} \u{306f} \u{3093} ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    // Do the same, but use uppercase letters (A-F) in the hex values.
    encoding = "unicodeescape-curly-upper";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // \u{D83E}\u{DDE0}
    // \u{D83D}\u{DD10}
    // \u{2705}
    // \u{26A0}\u{FE0F}
    // \u{274C}
    // \u{2713}
    // \u{4E2D}
    // \u{00E9} xyz \u{00E0}
    // abc \u{79C1} \u{306F} \u{3093} ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    // HTML hexadecimal character reference

    encoding = "unicodeescape-htmlhex";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // 🧠
    // 🔐
    // ✅
    // ⚠️
    // ❌
    // ✓
    // 中
    // é xyz à
    // abc 私 は ん ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    // HTML decimal character reference

    encoding = "unicodeescape-htmldec";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // 🧠
    // 🔐
    // ✅
    // ⚠️
    // ❌
    // ✓
    // 中
    // é xyz à
    // abc 私 は ん ghi

    // Revert back to the unescaped chars:
    sb.decode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // -----------------------------------------------------------------------------------------
    // Unicode code point notation or U+ notation

    encoding = "unicodeescape-plus";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // u+1f9e0
    // u+1f510
    // u+2705
    // u+26a0u+fe0f
    // u+274c
    // u+2713
    // u+4e2d
    // u+00e9 xyz u+00e0
    // abc u+79c1 u+306f u+3093 ghi

    // Chilkat cannot unescape the Unicode code point notation or U+ notation.
    // For this style, Chilkat only goes in one direction, which is to escape.

    // To emit uppercase hex, specify unicodeescape-plus-upper
    encoding = "unicodeescape-plus-upper";
    // ...
    // ...

    sb.clear();
    sb.appendSb(sb_original) catch {};

    // -----------------------------------------------------------------------------------------
    // Hex in Angle Brackets

    encoding = "unicodeescape-angle";
    sb.encode(encoding, charset_not_used) catch {};
    std.debug.print("{s}\n", .{try sb.getAsString(alloc)});

    // Output:
    // <1f9e0>
    // <1f510>
    // <2705>
    // <26a0><fe0f>
    // <274c>
    // <2713>
    // <4e2d>
    // <e9> xyz <e0>
    // abc <79c1> <306f> <3093> ghi

    // Chilkat cannot unescape the angle bracket notation.
    // For this style, Chilkat only goes in one direction, which is to escape.

    sb.clear();
    sb.appendSb(sb_original) catch {};
}