diff --git a/.gitignore b/.gitignore
index d6737749..a40407fa 100644
--- a/.gitignore
+++ b/.gitignore
@@ -82,7 +82,9 @@ dasher.log
# Training files written by the engine to CWD when contexts are destroyed.
# This is a real bug (Tier 1 item: library should not write to CWD) — for
# now we ignore the leaked files so they don't pollute git status.
-training_*.txt
+# Root-anchored: Data/training/ holds the SHIPPED corpora and must stay
+# trackable (training_emoji.txt was silently ignored by the bare pattern).
+/training_*.txt
build-san/
build-tidy/
diff --git a/Data/alphabets/alphabet.emoji.xml b/Data/alphabets/alphabet.emoji.xml
new file mode 100644
index 00000000..9d56826a
--- /dev/null
+++ b/Data/alphabets/alphabet.emoji.xml
@@ -0,0 +1,354 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/Data/alphabets/alphabet_index.json b/Data/alphabets/alphabet_index.json
index a7efdfde..3e34e3b1 100644
--- a/Data/alphabets/alphabet_index.json
+++ b/Data/alphabets/alphabet_index.json
@@ -1,7 +1,7 @@
{
"generator": "generate-alphabet-index.py",
- "generated_utc": "2026-08-29T14:22:40Z",
- "count": 474,
+ "generated_utc": "2026-09-18T16:45:26Z",
+ "count": 475,
"summaries": {
"by_script": {
"Latn": 317,
@@ -12,7 +12,7 @@
"Ethi": 8,
"Hani": 7,
"Beng": 4,
- "unknown": 3,
+ "unknown": 4,
"Kana": 3,
"Orya": 3,
"Mymr": 2,
@@ -54,14 +54,14 @@
"Runr": 1
},
"by_orientation": {
- "ltr": 451,
+ "ltr": 452,
"rtl": 22,
"ttb": 1
},
"with_lang_code": 456,
- "without_lang_code": 18,
- "declaring_training": 239,
- "training_available": 150,
+ "without_lang_code": 19,
+ "declaring_training": 240,
+ "training_available": 151,
"training_missing": 89,
"no_training_declared": 235
},
@@ -2380,6 +2380,32 @@
],
"source": "worldalphabets"
},
+ {
+ "id": "Emoji",
+ "file": "alphabet.emoji.xml",
+ "orientation": "ltr",
+ "lang": null,
+ "script": null,
+ "script_name": null,
+ "training": "training_emoji.txt",
+ "training_available": true,
+ "palette": "Default",
+ "conversion": "none",
+ "chars": 308,
+ "groups": [
+ "Smileys and faces",
+ "Gestures and hands",
+ "Hearts and celebration",
+ "People and family",
+ "Animals and nature",
+ "Food and drink",
+ "Travel and places",
+ "Objects and activities",
+ "Communication and symbols",
+ "Separators"
+ ],
+ "source": "maintained"
+ },
{
"id": "English (WorldAlphabets)",
"file": "autoConverted/alphabet.wa.english.en-Latn.xml",
diff --git a/Data/training/Makefile.am b/Data/training/Makefile.am
index a60e6fd0..baed9f7b 100644
--- a/Data/training/Makefile.am
+++ b/Data/training/Makefile.am
@@ -1,4 +1,5 @@
-dist_pkgdata_DATA = \
+ dist_pkgdata_DATA = \
+ training_emoji.txt \
training_english_GB.txt \
training_wa_af_Latn.txt \
training_wa_am_Ethi.txt \
diff --git a/Data/training/training_emoji.txt b/Data/training/training_emoji.txt
new file mode 100644
index 00000000..0040047a
--- /dev/null
+++ b/Data/training/training_emoji.txt
@@ -0,0 +1,320 @@
+😀 😃 😄 😊 🙂 😉
+😍 💕 🥰 😘
+👍 👍 👏 🙌 🙏
+😂 🤣 😅 😆 😅
+😊 😍 😊
+👋 😊 🤗 💛
+🎉 🎊 🥳 ✨ 🎁
+🎂 🎉 🍰 🥳 🎁
+😥 😢 😭 💔 🥺
+😎 😎 😎 😎
+🤔 🤔 🤨 💭
+😊 👍 ✅ 💯
+😍 💖 💘 💝
+🌷 🌹 🌸 💐 🌻
+🐶 🐱 🐰 🐹 🦊
+🌈 ⭐ 🌟 ✨
+☕ 🍰 🍫 ☕ 🍰
+🍕 🍔 🍟 🌭 🍕
+🍎 🍌 🍇 🍓 🍊
+🥗 🍎 💪 ✅ 👍
+🚗 🚌 🚕 🚲 🚗
+📱 💻 📱
+📞 📧 📧 💬
+💭 💬 💭
+🔥 🔥 💯 🔥
+🧡 💛 💚 💙 💜 🖤
+💔 😢 💔 😭
+😊 😇 🙂 😊
+😴 😪 🥱 😴
+🤒 🤕 😷 🤒
+🤢 🤮 🤧 🤒
+😱 😨 😳 😱
+😡 😠 😤 😡
+😂 😂 😂 🤣
+😉 😜 🤪 😜
+🕺 💃 🎧 🎧 🎉
+🎁 🎁 🎁 🎁
+✅ ❌ ❓ ❗
+⭐ ⭐ ⭐ ✨
+👍 👎 👌 🤝
+🙏 🙏 🙏
+👪 💕 👪
+👶 👧 👦 💕
+👵 👴 💕
+🤦 🤷 💭 🤦
+💪 💪 💪 💪
+🌸 🌸 🌸 ✨
+⚡ ⚡ 🔥 ⚡
+🌙
+🌞 🌞 🌞
+🐳
+🍕 🍕 🍕
+🍫 🍬 🍰 🍦
+🍵 ☕ 🧃 🥤
+🍷 🥂 🍺 🎉
+🍞 🧀 🥚 🍳
+🐶 🐱 🐦 🐬
+🌸 🌸 🌸 💖
+😊 😊
+👍 🙏 💯 ✅
+😃 😋
+😂 😅 🤭
+🤯 😃
+🥰 😢
+😠 🤗
+🥰 🤯 😴 😀
+😥 🧐 🙂
+😅 😎 🤓
+😳 😴 🤔
+🤕 😎
+🤤 🥵 🥴
+🤗 😄 🥰
+🥲 😪 🥴
+😋 🥶 🤓 😎
+😨 🤗 😨
+😳 😉
+😉 😎 🥲
+😥 😪 😮
+😴 🥱 🥳
+🥰 😷
+🥰 😢 🤢
+🙂 😂 🤭 😤
+🤮 🥳 🥰 🥲
+😅 😄 🤣 🥵
+🤯 😎 🤯 🙁
+🤕 🥴 😀
+😨 😤
+🙄 🧐 😜
+😔 🤒 🥴
+🤕 🙄 😢
+🥴 🤪
+🙂 🙄 😤
+😳 🥺
+🤧 🤢
+😍 😍 😡
+😆 🥺
+🤕 😤
+😂 🥺 😠
+😤 😳 🤒
+🤗 🥳 😨
+👉 👇 👌
+👎 👍 👋
+👍 🖖 👎
+🤝 🤲
+👇 🤘
+🤞 👌 🙏 👋
+🤟 👆 👈 👌
+🤙 👉 🤝 👎
+👎 👏
+🤟 🤞
+🤌 🤞 👆 🤝
+👉 🤲
+👎 💪
+💪 🙏
+🤌 👆 🤞
+🤝 🤌 👍 👈
+🤘 🖖 👏 🙌
+🤌 🤘 💪 👋
+🎊 💛
+💝 ✅ 💜 ❓
+✅ 💚
+🔥 🤎
+❓ ⭐
+🌟 💗
+❓ 🤍 🎉
+🤎 💙 💫
+💕 🎊 💚 💖
+💚 🤍
+💙 💞 💚
+💓 💜 ✅
+🌟 💯 💫
+🔥 ✨
+❓ 💜
+❌ 🎊
+💔 🌟 🎉
+🔥 ✅ 💝
+❗ ✅ 💚
+🎁 💛 💞 💙
+💜 💖 🎉
+💙 ❓
+🦉 🌸 🦋
+🐔 🐱 🐸
+🌈 🐼
+🐙 🐷 🦋
+⚡ 🐳 🌞 🐨
+⚡ 🌷 💐
+🐻 🌞
+🌹 💐 🌷 🐯
+🐸 🐱
+🐨 🌷 🐧 🦁
+🐨 🐢 🐵
+💐 🌙 🦁
+🐙 🌸 🐮 🐶
+🐻 🌙 🐯
+🐝 🐷
+🐔 ⚡ 🐰
+🐯 🐬 🐶
+🐸 🐭 🐢
+🐞 🐢 🐰
+🦆 🐙 🐮
+🦉 🐻 🐢
+🐙 🌈 🐞 🦁
+🦁 🐼
+🐧 🐔 🐼
+🌹 🌙 🐳 🐢
+🦆 🐞 🐷
+🐙 🐨 🐼
+🐱 ⚡
+🧀 🍵
+🍚 🌽 🥕
+🥗 🍏 🍦
+🍰 🍒
+🍐 🍑 🍬 🧀
+🍜 🥂 🌭 🍺
+🌮 🍚 🥓 🍫
+🧃 🍛 🥚 🥭
+🍬 🍍 🍰
+🌯 🍍 🍊 🍍
+🍅 🍵 🍊
+🍑 🍣 🍣
+🥕 🥝
+🥕 🫐 🍞 🧃
+☕ 🍦
+🥚 🍺 🥥 🌽
+🍔 🍛 🎂 🧃
+🍒 🍞 🍎 🍅
+🍚 🍬 🥤 🥂
+🧃 🌭
+🍋 🍞
+🍓 🥭 🥝 🧀
+🍦 🌽 🍦
+🥭 🍊 🍏 🍔
+🍒 🍊 🍺
+🍦 🍺
+🍬 🌮 🍑
+🍜 🍕 🍒 🍣
+🍓 🌮 🍛
+🍰 🍇
+🚕 🚗 🚚
+🛵 ⛵ 🚕 ⛵
+🚕 🚤 🗼
+🗽 🚀
+🚜 🛵
+🚕 🚁 🚗
+🗽 🚕 🌋 🚁
+⛵ 🛵 🚤 🌋
+🚁 🏰 🚤
+🛵 🚤 🚒 🗼
+🚚 🚲
+🚀 ⛵ 🚜 🚲
+🚑 🚑 🗽
+🚚 🚀 🗼
+🚨 🚑
+⛵ 🚲 🚤
+🔧 🔒 🎤 📷
+💳 🔌 🔧
+🔑 🔓 💎
+💵 💎 🔦
+📷 📞 📷
+📞 🎨 💡
+🎲 🎧 💡
+🎥 ⌚ 🔓
+📱 📱 💡
+🔧 🎨 📷
+📌 🔑
+💵 🎮 🔌 🎸
+🎧 💳
+📍 🔨
+📸 📌 🔦
+📷 🎬 📏
+🎬 💰 🎹
+🎸 🎲 🔦
+📏 🎨
+📏 🔌 💻
+🎥 💻 ⌚
+🔨 🔑 📱 💡
+🎯 💎 🔨
+🎬 📏 🔧
+📰 💭 🔱
+🚫 📧 📤
+💭 🚫 🔞
+📨 🔱 ⭕ 📵
+📰 💭 🔞 ⭕
+📧 📤 🔱
+📵 🔱
+💭 ⭕ 📤
+🔞 📤 💭 🔱
+📵 💬 📰 📦
+🔞 📦
+📥 📵 🚫 💭
+😭 🖤
+🤓 ✨
+🤓 🤎
+🥴 ✅
+🤕 👌
+🤧 🧡
+🤭 🤞
+😭 ❌
+😁 💗
+🤯 🎈
+😥 🤲
+😠 👆
+😢 🤲
+😁 💖
+🤪 🤟
+🙂 🌟
+😢 ✋
+😠 🤎
+😃 👆
+😨 💪
+😋 🖖
+😊 💫
+🥵 🧡
+🥵 ❌
+😅 💘
+😷 🤎
+😭 🤎
+🤔 💕
+😠 🖖
+🤨 👌
+🥱 ⭐
+🙁 👍
+😤 💝
+😭 🤟
+😉 🎁
+🥰 👈
+😡 🙌
+🥲 💔
+🤕 💜
+🤪 🤝
+🤢 ❌
+😎 🙏
+😡 ⭐
+🤧 🧡
+🤨 👏
+🧐 🤙
+😱 👋
+😠 💕
+😢 🤞
+😤 🎊
+🔌 🥵
+🔧 😒
+🔒 🤓
+📵 😮
+💻 🥶
+💵 😘
+💳 😄
+🔋 🤤
+🔓 🥱
+💵 😒
+🎹 🤢
+🎸 😉
+📌 😄
+🔱 😡
+🎨 😎
+💬 🤕
+🔑 🤣
+📏 😭
+📧 😴
+⭕ 🙁
diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp
index b12d06b5..793925d8 100644
--- a/tests/test_alphabet_xml.cpp
+++ b/tests/test_alphabet_xml.cpp
@@ -1,6 +1,9 @@
// Alphabet XML parsing tests: verify alphabet loading, switching, and structure
#include "test_common.h"
+#include
+#include
+
TEST(alphabet_default_loaded) {
dasher_ctx* ctx = create_isolated_context();
ASSERT(ctx);
@@ -100,6 +103,296 @@ TEST(alphabet_symbol_out_of_range_returns_error) {
dasher_destroy(ctx);
}
+// ── Emoji alphabet (Dasher-Android #61, option 2) ─────────────────────────
+// Multi-codepoint nodes: ZWJ sequences (family: 5 codepoints / 18 bytes) and
+// skin-tone modifiers must survive the XML round-trip whole — ReadCharAttributes
+// stores the full label as Text and TextOutputAction commits it atomically.
+
+TEST(alphabet_emoji_loads_and_has_groups) {
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+
+ dasher_set_alphabet_id(ctx, "Emoji");
+ const char* loaded = dasher_get_alphabet_id(ctx);
+ printf(" Switched to: '%s'\n", loaded);
+ ASSERT_STR_EQ(loaded, "Emoji");
+
+ int sym_count = dasher_get_alphabet_symbol_count(ctx);
+ printf(" Emoji symbol count: %d\n", sym_count);
+ // Emoji alphabet loads (expect ~310 symbols: 9 groups, ~307 nodes + control).
+ ASSERT(sym_count > 150);
+
+ dasher_destroy(ctx);
+}
+
+TEST(alphabet_emoji_zwj_sequence_roundtrip) {
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+
+ dasher_set_alphabet_id(ctx, "Emoji");
+ ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji");
+
+ const char* family =
+ "\xF0\x9F\x91\xA8\xE2\x80\x8D\xF0\x9F\x91\xA9\xE2\x80\x8D\xF0\x9F\x91\xA7"; // U+1F468 ZWJ U+1F469 ZWJ U+1F467
+ const char* toned = "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD"; // U+1F44D U+1F3FD
+ bool found_family = false, found_toned = false, found_space = false;
+ int sym_count = dasher_get_alphabet_symbol_count(ctx);
+ for (int i = 1; i < sym_count; i++) {
+ char buf[128];
+ if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue;
+ if (strcmp(buf, family) == 0) found_family = true;
+ if (strcmp(buf, toned) == 0) found_toned = true;
+ if (strcmp(buf, " ") == 0) found_space = true;
+ }
+ printf(" ZWJ family: %d, skin-tone: %d, space: %d\n", found_family, found_toned, found_space);
+ ASSERT(found_family);
+ ASSERT(found_toned);
+ ASSERT(found_space); // separator node: display ␣, text " "
+
+ dasher_destroy(ctx);
+}
+
+TEST(alphabet_emoji_training_file_present) {
+ // The engine tolerates a missing training file, but we ship one for mild
+ // ordering priors — assert the shipped tree still has it next to the
+ // alphabet so release packaging doesn't silently drop it.
+ std::error_code ec;
+ bool ok =
+ std::filesystem::exists(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec);
+ ASSERT(ok);
+}
+
+TEST(alphabet_emoji_corpus_tokens_are_nodes) {
+ // Greptile P2: corpus tokens that are not alphabet symbols train
+ // UNKNOWN_SYMBOL observations — pure noise. Every whitespace-separated
+ // token must be a whole node text, and (trainer limitation) must be a
+ // SINGLE code point: the trainer looks up one code point per symbol,
+ // so multi-codepoint tokens (ZWJ, VS16) can never match and are
+ // silently split.
+ std::vector nodes;
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ dasher_set_alphabet_id(ctx, "Emoji");
+ ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji");
+ int sym_count = dasher_get_alphabet_symbol_count(ctx);
+ for (int i = 1; i < sym_count; i++) {
+ char buf[128];
+ if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') nodes.push_back(buf);
+ }
+ dasher_destroy(ctx);
+
+ std::ifstream in(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt");
+ ASSERT(in.is_open());
+ std::string line;
+ int tokens = 0;
+ while (std::getline(in, line)) {
+ size_t start = 0;
+ while (start < line.size()) {
+ size_t end = line.find(' ', start);
+ if (end == std::string::npos) end = line.size();
+ if (end > start) {
+ std::string tok = line.substr(start, end - start);
+ tokens++;
+ bool known = false;
+ for (const auto& n : nodes) {
+ if (n == tok) {
+ known = true;
+ break;
+ }
+ }
+ if (!known) {
+ printf(" unknown corpus token (%zu bytes):", tok.size());
+ for (unsigned char ch : tok)
+ printf(" %02x", ch);
+ printf("\n");
+ ASSERT(false);
+ }
+ // Greptile P2: membership alone is insufficient — a
+ // multi-codepoint NODE (👨👩👧) would pass even though the
+ // trainer looks up one code point per symbol and can never
+ // match it. Enforce single-codepoint tokens explicitly:
+ // count UTF-8 lead bytes (non-continuation).
+ int codepoints = 0;
+ for (unsigned char ch : tok)
+ if ((ch & 0xC0) != 0x80) codepoints++;
+ if (codepoints != 1) {
+ printf(" multi-codepoint corpus token (%d codepoints):", codepoints);
+ for (unsigned char ch : tok)
+ printf(" %02x", ch);
+ printf("\n");
+ ASSERT(false);
+ }
+ }
+ start = end + 1;
+ }
+ }
+ printf(" %d corpus tokens, all valid single-codepoint nodes\n", tokens);
+ ASSERT(tokens > 100);
+}
+
+TEST(alphabet_emoji_output_segments_into_whole_nodes) {
+ // Greptile P2: prove the OUTPUT path commits multi-codepoint nodes
+ // atomically — at EVENT level, not byte level. A concatenated-bytes
+ // check would pass even if 👨👩👧 arrived as five separate events
+ // (👨, ZWJ, 👩, ZWJ, 👧); the output callback sees each insert as its
+ // own event, so requiring every event's text to be a WHOLE node text
+ // closes that hole. We also require at least one multi-codepoint
+ // event (>4 bytes ⇒ ZWJ or VS16 carrier) so the multi-byte path is
+ // actually exercised, not vacuously green.
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ dasher_set_alphabet_id(ctx, "Emoji");
+ ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji");
+ dasher_set_speed_percent(ctx, 300);
+
+ std::vector symbols;
+ int sym_count = dasher_get_alphabet_symbol_count(ctx);
+ for (int i = 1; i < sym_count; i++) {
+ char buf[128];
+ if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') symbols.push_back(buf);
+ }
+ ASSERT(symbols.size() > 100);
+
+ std::vector events;
+ dasher_set_output_callback(
+ ctx,
+ [](int event_type, const char* text, void* user_data) {
+ if (event_type == DASHER_EVENT_OUTPUT) {
+ static_cast*>(user_data)->push_back(text);
+ }
+ },
+ &events);
+
+ // Sweep sy across varied bands with the button held. Several passes
+ // with different y-bands and x-depths; every committed event must be a
+ // WHOLE node text (byte-level concatenation can't prove atomicity —
+ // five separate events for 👨👩👧 produce identical bytes).
+ const struct {
+ int y0, y1, x;
+ } passes[] = {
+ {100, 500, 700}, // full safe band (same as test_spell_word)
+ {150, 350, 720}, // upper-half dwell
+ {300, 540, 680}, // lower-half dwell
+ {200, 460, 740}, // deeper zoom
+ };
+ int frame = 0;
+ for (const auto& p : passes) {
+ dasher_mouse_down(ctx);
+ const int span = p.y1 - p.y0;
+ for (int i = 0; i < 500; i++) {
+ int sy = p.y0 + (i % span);
+ dasher_mouse_move(ctx, static_cast(p.x), static_cast(sy));
+ int* c = nullptr;
+ int cc = 0;
+ char** s = nullptr;
+ int sc = 0;
+ dasher_frame(ctx, 1000 + (frame++) * 16, &c, &cc, &s, &sc);
+ }
+ dasher_mouse_up(ctx);
+ if (events.size() > 0) break;
+ }
+
+ size_t total = 0;
+ for (const auto& ev : events) {
+ total += ev.size();
+ bool whole = false;
+ for (const auto& sym : symbols) {
+ if (sym == ev) {
+ whole = true;
+ break;
+ }
+ }
+ if (!whole) {
+ printf(" NON-ATOMIC event (%zu bytes):", ev.size());
+ for (unsigned char ch : ev)
+ printf(" %02x", ch);
+ printf("\n");
+ }
+ ASSERT(whole);
+ }
+ printf(" %zu output events, %zu bytes — every event a whole node text\n", events.size(), total);
+ ASSERT(events.size() > 0);
+
+ dasher_destroy(ctx);
+}
+
+TEST(alphabet_multicodepoint_commit_is_atomic) {
+ // Deterministic multi-codepoint coverage: a purpose-built 3-node test
+ // alphabet (ZWJ sequence, VS16 carrier, plain emoji) where EVERY node
+ // but one is multi-codepoint and each holds ~1/3 of the tree mass —
+ // navigation cannot avoid committing them. Complements the sweep test
+ // above, whose mass distribution follows the training priors and may
+ // not reach the shipped alphabet's low-mass multi-codepoint nodes.
+ ScopedTempDir dataRoot;
+ const std::string data_dir = build_data_dir(dataRoot);
+ // No training file: uniform-ish symbol ordering (the engine's
+ // documented no-training fallback).
+ std::string xml = std::string("\n") +
+ "\n" +
+ "\n" +
+ " \n" +
+ " \n" +
+ " \n" +
+ " \n" + " \n" + "\n";
+ ASSERT(write_data_file(data_dir, "alphabets", "alphabet.zwjtest.xml", xml));
+
+ dasher_ctx* ctx = dasher_create(data_dir.c_str(), dataRoot.c_str(), nullptr);
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ printf(" custom dir alphabets: %d\n", dasher_get_alphabet_count(ctx));
+ dasher_set_alphabet_id(ctx, "ZWJ Test");
+ ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "ZWJ Test");
+
+ const std::string family = "\xF0\x9F\x91\xA8\xE2\x80\x8D\xF0\x9F\x91\xA9\xE2\x80\x8D\xF0\x9F\x91\xA7";
+ const std::string plane = "\xE2\x9C\x88\xEF\xB8\x8F";
+ const std::string grin = "\xF0\x9F\x98\x80";
+
+ std::vector events;
+ dasher_set_output_callback(
+ ctx,
+ [](int event_type, const char* text, void* user_data) {
+ if (event_type == DASHER_EVENT_OUTPUT) {
+ static_cast*>(user_data)->emplace_back(text);
+ }
+ },
+ &events);
+
+ dasher_set_speed_percent(ctx, 300);
+ dasher_mouse_down(ctx);
+ for (int i = 0; i < 600; i++) {
+ int sy = 150 + (i % 300);
+ dasher_mouse_move(ctx, 700.0f, static_cast(sy));
+ int* c = nullptr;
+ int cc = 0;
+ char** s = nullptr;
+ int sc = 0;
+ dasher_frame(ctx, 1000 + i * 16, &c, &cc, &s, &sc);
+ }
+ dasher_mouse_up(ctx);
+
+ size_t multi = 0;
+ for (const auto& ev : events) {
+ bool known = (ev == family || ev == plane || ev == grin);
+ if (!known) {
+ printf(" NON-ATOMIC event (%zu bytes):", ev.size());
+ for (unsigned char ch : ev)
+ printf(" %02x", ch);
+ printf("\n");
+ }
+ ASSERT(known);
+ if (ev.size() > 4) multi++;
+ }
+ printf(" %zu events, %zu multi-codepoint commits\n", events.size(), multi);
+ ASSERT(events.size() > 0);
+ ASSERT(multi > 0);
+
+ dasher_destroy(ctx);
+}
+
TEST(alphabet_switch_changes_probabilities) {
dasher_ctx* ctx = create_isolated_context();
ASSERT(ctx);
@@ -181,7 +474,7 @@ TEST(alphabet_v6_space_character_resolves_to_space) {
// Valid symbol indices are 1..sym_count inclusive (index 0 is the sentinel).
bool found_space = false;
- for (int i = 1; i <= sym_count; i++) {
+ for (int i = 1; i < sym_count; i++) {
char buf[128];
if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && strcmp(buf, " ") == 0) {
found_space = true;
@@ -226,7 +519,7 @@ TEST(alphabet_v6_paragraph_outputs_newline) {
ASSERT(sym_count > 0);
bool found_paragraph_display = false, paragraph_is_newline = false;
- for (int i = 1; i <= sym_count; i++) {
+ for (int i = 1; i < sym_count; i++) {
char disp[128], text[128];
if (dasher_get_alphabet_symbol_display(ctx, i, disp, sizeof(disp)) != 0) continue;
if (strcmp(disp, "\xc2\xb6") != 0) continue; // UTF-8 pilcrow
@@ -320,7 +613,7 @@ TEST(alphabet_v5_symbols_have_correct_text) {
ASSERT(sym_count >= 3);
bool found_x = false, found_space = false, found_emoji = false;
- for (int i = 1; i <= sym_count; i++) {
+ for (int i = 1; i < sym_count; i++) {
char buf[128];
if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue;
std::string s(buf);
@@ -425,7 +718,7 @@ TEST(alphabet_v5_special_chars_as_direct_children) {
// Scan all symbols for the expected text values.
bool found_letter_a = false, found_space = false, found_newline = false;
- for (int i = 1; i <= sym_count; i++) {
+ for (int i = 1; i < sym_count; i++) {
char buf[128];
if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue;
std::string s(buf);