Android™
Android™
Unicode Escape
Demonstrates options for unicode escaping non-us-ascii chars and emojis.Chilkat Android™ Downloads
// Important: Don't forget to include the call to System.loadLibrary
// as shown at the bottom of this code sample.
package com.test;
import android.app.Activity;
import com.chilkatsoft.*;
import android.widget.TextView;
import android.os.Bundle;
public class SimpleActivity extends Activity {
private static final String TAG = "Chilkat";
// Called when the activity is first created.
@Override
public void onCreate(Bundle savedInstanceState) {
super.onCreate(savedInstanceState);
boolean success = false;
CkStringBuilder sb = new CkStringBuilder();
success = sb.LoadFile("qa_data/txt/utf16_emojis_accented_jap.txt","utf-16");
if (success == false) {
Log.i(TAG, sb.lastErrorText());
return;
}
String original = sb.getAsString();
// The above file contains the following text, which includes some emoji's,
// Japanese chars, and accented chars.
// 🧠
// 🔐
// ✅
// ⚠️
// ❌
// ✓
// 中
// é xyz à
// abc 私 は ん ghi
CkCrypt2 crypt = new CkCrypt2();
// Charset is not used for unicode escaping. Set it to "utf-8", but it means nothing.
String charsetNotUsed = "utf-8";
// Indicate the desired format/style of Unicode escaping.
// Choose JSON-style (JavaScript-style) Unicode escape sequences by using "unicodeescape"
String encoding = "unicodeescape";
String escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// \ud83e\udde0
// \ud83d\udd10
// \u2705
// \u26a0\ufe0f
// \u274c
// \u2713
// \u4e2d
// \u00e9 xyz \u00e0
// abc \u79c1 \u306f \u3093 ghi
// Revert back to the unescaped chars:
String unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// Do the same, but use uppercase letters (A-F) in the hex values.
encoding = "unicodeescape-upper";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// \uD83E\uDDE0
// \uD83D\uDD10
// \u2705
// \u26A0\uFE0F
// \u274C
// \u2713
// \u4E2D
// \u00E9 xyz \u00E0
// abc \u79C1 \u306F \u3093 ghi
// Revert back to the unescaped chars:
unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// ECMAScript (JavaScript) “code point escape” syntax
encoding = "unicodeescape-curly";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// \u{d83e}\u{dde0}
// \u{d83d}\u{dd10}
// \u{2705}
// \u{26a0}\u{fe0f}
// \u{274c}
// \u{2713}
// \u{4e2d}
// \u{00e9} xyz \u{00e0}
// abc \u{79c1} \u{306f} \u{3093} ghi
// Revert back to the unescaped chars:
unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// Do the same, but use uppercase letters (A-F) in the hex values.
encoding = "unicodeescape-curly-upper";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// \u{D83E}\u{DDE0}
// \u{D83D}\u{DD10}
// \u{2705}
// \u{26A0}\u{FE0F}
// \u{274C}
// \u{2713}
// \u{4E2D}
// \u{00E9} xyz \u{00E0}
// abc \u{79C1} \u{306F} \u{3093} ghi
// Revert back to the unescaped chars:
unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// Unicode code point notation or U+ notation
encoding = "unicodeescape-plus";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// u+1f9e0
// u+1f510
// u+2705
// u+26a0u+fe0f
// u+274c
// u+2713
// u+4e2d
// u+00e9 xyz u+00e0
// abc u+79c1 u+306f u+3093 ghi
// Chilkat cannot unescape the Unicode code point notation or U+ notation.
// For this style, Chilkat only goes in one direction, which is to escape.
// To emit uppercase hex, specify unicodeescape-plus-upper
encoding = "unicodeescape-plus-upper";
// ...
// ...
// -----------------------------------------------------------------------------------------
// HTML hexadecimal character reference
encoding = "unicodeescape-htmlhex";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// 🧠
// 🔐
// ✅
// ⚠️
// ❌
// ✓
// 中
// é xyz à
// abc 私 は ん ghi
// Revert back to the unescaped chars:
unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// HTML decimal character reference
encoding = "unicodeescape-htmldec";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// 🧠
// 🔐
// ✅
// ⚠️
// ❌
// ✓
// 中
// é xyz à
// abc 私 は ん ghi
// Revert back to the unescaped chars:
unescaped = crypt.decodeString(escaped,charsetNotUsed,encoding);
Log.i(TAG, unescaped);
// -----------------------------------------------------------------------------------------
// Hex in Angle Brackets
encoding = "unicodeescape-angle";
escaped = crypt.encodeString(original,charsetNotUsed,encoding);
Log.i(TAG, escaped);
// Output:
// <1f9e0>
// <1f510>
// <2705>
// <26a0><fe0f>
// <274c>
// <2713>
// <4e2d>
// <e9> xyz <e0>
// abc <79c1> <306f> <3093> ghi
// Chilkat cannot unescape the angle bracket notation.
// For this style, Chilkat only goes in one direction, which is to escape.
}
static {
System.loadLibrary("chilkat");
// Note: If the incorrect library name is passed to System.loadLibrary,
// then you will see the following error message at application startup:
//"The application <your-application-name> has stopped unexpectedly. Please try again."
}
}