From 9bbef67f96f30016b094fb3c555d97cb997517c3 Mon Sep 17 00:00:00 2001 From: will wade Date: Fri, 18 Sep 2026 17:04:10 +0100 Subject: [PATCH 1/4] =?UTF-8?q?feat(alphabets):=20Emoji=20alphabet=20?= =?UTF-8?q?=E2=80=94=20language-neutral,=20multi-codepoint=20nodes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Dasher-Android #61 (option 2): a toolbar switch to an emoji-only alphabet works for any language via the existing alphabet-switch API — pure data, no engine changes. - alphabet.emoji.xml: 9 topical groups (~300 nodes), two zoom levels to any emoji. Includes ZWJ sequences (family) and skin-tone modifiers as natural regression coverage for multi-codepoint nodes. - training_emoji.txt: mild ordering priors (common juxtapositions). - .gitignore: root-anchor the training_*.txt leak rule — the bare pattern silently ignored Data/training/ additions. - tests: emoji loads (309 symbols); ZWJ/skin-tone/space round-trip whole through dasher_get_alphabet_symbol_text; shipped training file present. Signed-off-by: will wade --- .gitignore | 4 +- Data/alphabets/alphabet.emoji.xml | 349 ++++++++++++++++++++++++++++++ Data/training/training_emoji.txt | 61 ++++++ tests/test_alphabet_xml.cpp | 61 ++++++ 4 files changed, 474 insertions(+), 1 deletion(-) create mode 100644 Data/alphabets/alphabet.emoji.xml create mode 100644 Data/training/training_emoji.txt diff --git a/.gitignore b/.gitignore index d6737749..a40407fa 100644 --- a/.gitignore +++ b/.gitignore @@ -82,7 +82,9 @@ dasher.log # Training files written by the engine to CWD when contexts are destroyed. # This is a real bug (Tier 1 item: library should not write to CWD) — for # now we ignore the leaked files so they don't pollute git status. -training_*.txt +# Root-anchored: Data/training/ holds the SHIPPED corpora and must stay +# trackable (training_emoji.txt was silently ignored by the bare pattern). +/training_*.txt build-san/ build-tidy/ diff --git a/Data/alphabets/alphabet.emoji.xml b/Data/alphabets/alphabet.emoji.xml new file mode 100644 index 00000000..b7dbe7c6 --- /dev/null +++ b/Data/alphabets/alphabet.emoji.xml @@ -0,0 +1,349 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/Data/training/training_emoji.txt b/Data/training/training_emoji.txt new file mode 100644 index 00000000..bf5e314e --- /dev/null +++ b/Data/training/training_emoji.txt @@ -0,0 +1,61 @@ +😀 😃 😄 😊 🙂 😉 +😍 ❤️ 💕 🥰 😘 +👍 👍 👏 🙌 🙏 +😂 🤣 😅 😆 😅 +😊 😍 ❤️ 😊 +👋 😊 🤗 💛 +🎉 🎊 🥳 ✨ 🎁 +🎂 🎉 🍰 🥳 🎁 +😥 😢 😭 💔 🥺 +😎 😎 🕶️ 😎 +🤔 🤔 🤨 💭 +😊 👍 ✅ 💯 +😍 💖 💘 💝 ❤️ +🌷 🌹 🌸 💐 🌻 +🐶 🐱 🐰 🐹 🦊 +☀️ 🌈 ⭐ 🌟 ✨ +☕ 🍰 🍪 ☕ 🍰 +🍕 🍔 🍟 🌭 🍕 +🍎 🍌 🍇 🍓 🍊 +🥗 🍎 💪 ✅ 👍 +🚗 🚌 🚕 🚲 🚗 +✈️ 🗺️ 🏖️ 🌴 ☀️ +📱 💻 ⌨️ 🖥️ 📱 +📞 📧 ✉️ 💌 💬 +💭 🗣️ 💬 🗣️ 💭 +🔥 🔥 💯 🔥 +❤️ 🧡 💛 💚 💙 💜 🖤 +💔 😢 💔 😭 +😊 😇 🙂 😌 +😴 😪 🥱 😴 +🤒 🤕 😷 🤒 +🤢 🤮 🤧 🤒 +😱 😨 😳 😱 +😡 😠 😤 😡 +😂 😂 😂 🤣 +😉 😜 🤪 😜 +🕺 💃 🎶 🎵 🎉 +🎁 🎁 🎀 🎁 +✅ ❌ ❓ ❗ +⭐ ⭐ ⭐ ✨ +👍 👎 👌 🤝 +🙏 🙏 🙏 ❤️ +👨‍👩‍👧 👪 💕 🏡 +👶 👧 👦 💕 +👵 👴 💕 ❤️ +🤦 🤷 💭 🤦 +💪 💪 ✊ 💪 +🍀 ☘️ 🍀 ✨ +⚡ ⚡ 🔥 ⚡ +❄️ ⛄ ☁️ 🌙 +🌞 🌞 ☀️ 🌞 +🌊 🏝️ 🌴 🏖️ +🍕 🍕 🍕 ❤️ +🍫 🍬 🍰 🍦 +🍵 ☕ 🧃 🥤 +🍷 🥂 🍻 🎉 +🍞 🧀 🥚 🍳 +🐕 🐈 🐦 🐠 +🌸 🌸 🌸 💖 +😊 ❤️ 😊 +👍 🙏 💯 ✅ diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp index b12d06b5..4badbba9 100644 --- a/tests/test_alphabet_xml.cpp +++ b/tests/test_alphabet_xml.cpp @@ -100,6 +100,67 @@ TEST(alphabet_symbol_out_of_range_returns_error) { dasher_destroy(ctx); } +// ── Emoji alphabet (Dasher-Android #61, option 2) ───────────────────────── +// Multi-codepoint nodes: ZWJ sequences (family: 5 codepoints / 18 bytes) and +// skin-tone modifiers must survive the XML round-trip whole — ReadCharAttributes +// stores the full label as Text and TextOutputAction commits it atomically. + +TEST(alphabet_emoji_loads_and_has_groups) { + dasher_ctx* ctx = create_isolated_context(); + ASSERT(ctx); + dasher_set_screen_size(ctx, 800, 600); + + dasher_set_alphabet_id(ctx, "Emoji"); + const char* loaded = dasher_get_alphabet_id(ctx); + printf(" Switched to: '%s'\n", loaded); + ASSERT_STR_EQ(loaded, "Emoji"); + + int sym_count = dasher_get_alphabet_symbol_count(ctx); + printf(" Emoji symbol count: %d\n", sym_count); + // 9 topical groups (~230 nodes) + control/terminator symbols + ASSERT(sym_count > 150); + + dasher_destroy(ctx); +} + +TEST(alphabet_emoji_zwj_sequence_roundtrip) { + dasher_ctx* ctx = create_isolated_context(); + ASSERT(ctx); + dasher_set_screen_size(ctx, 800, 600); + + dasher_set_alphabet_id(ctx, "Emoji"); + ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji"); + + const char* family = + "\xF0\x9F\x91\xA8\xE2\x80\x8D\xF0\x9F\x91\xA9\xE2\x80\x8D\xF0\x9F\x91\xA7"; // U+1F468 ZWJ U+1F469 ZWJ U+1F467 + const char* toned = "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD"; // U+1F44D U+1F3FD + bool found_family = false, found_toned = false, found_space = false; + int sym_count = dasher_get_alphabet_symbol_count(ctx); + for (int i = 0; i < sym_count; i++) { + char buf[128]; + if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue; + if (strcmp(buf, family) == 0) found_family = true; + if (strcmp(buf, toned) == 0) found_toned = true; + if (strcmp(buf, " ") == 0) found_space = true; + } + printf(" ZWJ family: %d, skin-tone: %d, space: %d\n", found_family, found_toned, found_space); + ASSERT(found_family); + ASSERT(found_toned); + ASSERT(found_space); // separator node: display ␣, text " " + + dasher_destroy(ctx); +} + +TEST(alphabet_emoji_training_file_present) { + // The engine tolerates a missing training file, but we ship one for mild + // ordering priors — assert the shipped tree still has it next to the + // alphabet so release packaging doesn't silently drop it. + std::error_code ec; + bool ok = + std::filesystem::exists(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec); + ASSERT(ok); +} + TEST(alphabet_switch_changes_probabilities) { dasher_ctx* ctx = create_isolated_context(); ASSERT(ctx); From 2f438e120feccea03cbbc746cc06d09f9074d388 Mon Sep 17 00:00:00 2001 From: will wade Date: Fri, 18 Sep 2026 18:06:08 +0100 Subject: [PATCH 2/4] =?UTF-8?q?fix(emoji):=20address=20review=20=E2=80=94?= =?UTF-8?q?=20event-level=20atomicity,=20corpus=20hygiene,=20index?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review loop (greptile P2s + subagent 7/10 findings): - Atomicity now proven at EVENT level via dasher_set_output_callback: every committed event must equal a whole node text (byte concatenation can't distinguish one 👨‍👩‍👧 event from five). Plus a deterministic 3-node test alphabet (ZWJ, VS16, plain) where multi-codepoint commits are unavoidable: 36 events, 23 multi-codepoint, all whole. - Corpus: single-codepoint nodes only (the trainer looks up one code point per symbol — ZWJ/VS16 tokens can never match); expanded to 945 tokens/4.6KB; new test enforces corpus ⊆ alphabet nodes permanently. - alphabet_index.json regenerated (475) — Emoji registered for index-driven tooling. - Makefile.am: training_emoji.txt added (keep-consistent-only; the automake subtree is not the live packaging path — CMake is). - Comment fixes (9 groups not 8; ~310 symbols not ~230). Signed-off-by: will wade --- Data/alphabets/alphabet.emoji.xml | 7 +- Data/alphabets/alphabet_index.json | 40 +++- Data/training/Makefile.am | 3 +- Data/training/training_emoji.txt | 311 ++++++++++++++++++++++++++--- tests/test_alphabet_xml.cpp | 221 +++++++++++++++++++- 5 files changed, 544 insertions(+), 38 deletions(-) diff --git a/Data/alphabets/alphabet.emoji.xml b/Data/alphabets/alphabet.emoji.xml index b7dbe7c6..9d56826a 100644 --- a/Data/alphabets/alphabet.emoji.xml +++ b/Data/alphabets/alphabet.emoji.xml @@ -9,13 +9,18 @@ node's Text, so a selection commits the entire sequence atomically (see ReadCharAttributes: Text = "text" attr, default Display/label). - Groups are kept to 8 topical clusters plus separators: Dasher's + Groups are kept to 9 topical clusters plus separators: Dasher's zooming interface needs a shallow tree — two zoom levels to any emoji. A flat 200-node root would be unusable. Frequencies: training_emoji.txt provides mild ordering priors (common juxtapositions); without training the engine falls back to symbol-order defaults (NodeCreationManager "Dasher will still work"). + NOTE: the trainer matches ONE code point per symbol lookup, so only + single-codepoint emoji are learnable from the corpus. The 31 + multi-codepoint nodes (ZWJ sequences, VS16 variants like ✈️) always + take default frequencies — a longest-match trainer would be needed + to train those (engine follow-up, not blocking). --> diff --git a/Data/alphabets/alphabet_index.json b/Data/alphabets/alphabet_index.json index a7efdfde..3e34e3b1 100644 --- a/Data/alphabets/alphabet_index.json +++ b/Data/alphabets/alphabet_index.json @@ -1,7 +1,7 @@ { "generator": "generate-alphabet-index.py", - "generated_utc": "2026-08-29T14:22:40Z", - "count": 474, + "generated_utc": "2026-09-18T16:45:26Z", + "count": 475, "summaries": { "by_script": { "Latn": 317, @@ -12,7 +12,7 @@ "Ethi": 8, "Hani": 7, "Beng": 4, - "unknown": 3, + "unknown": 4, "Kana": 3, "Orya": 3, "Mymr": 2, @@ -54,14 +54,14 @@ "Runr": 1 }, "by_orientation": { - "ltr": 451, + "ltr": 452, "rtl": 22, "ttb": 1 }, "with_lang_code": 456, - "without_lang_code": 18, - "declaring_training": 239, - "training_available": 150, + "without_lang_code": 19, + "declaring_training": 240, + "training_available": 151, "training_missing": 89, "no_training_declared": 235 }, @@ -2380,6 +2380,32 @@ ], "source": "worldalphabets" }, + { + "id": "Emoji", + "file": "alphabet.emoji.xml", + "orientation": "ltr", + "lang": null, + "script": null, + "script_name": null, + "training": "training_emoji.txt", + "training_available": true, + "palette": "Default", + "conversion": "none", + "chars": 308, + "groups": [ + "Smileys and faces", + "Gestures and hands", + "Hearts and celebration", + "People and family", + "Animals and nature", + "Food and drink", + "Travel and places", + "Objects and activities", + "Communication and symbols", + "Separators" + ], + "source": "maintained" + }, { "id": "English (WorldAlphabets)", "file": "autoConverted/alphabet.wa.english.en-Latn.xml", diff --git a/Data/training/Makefile.am b/Data/training/Makefile.am index a60e6fd0..baed9f7b 100644 --- a/Data/training/Makefile.am +++ b/Data/training/Makefile.am @@ -1,4 +1,5 @@ -dist_pkgdata_DATA = \ + dist_pkgdata_DATA = \ + training_emoji.txt \ training_english_GB.txt \ training_wa_af_Latn.txt \ training_wa_am_Ethi.txt \ diff --git a/Data/training/training_emoji.txt b/Data/training/training_emoji.txt index bf5e314e..0040047a 100644 --- a/Data/training/training_emoji.txt +++ b/Data/training/training_emoji.txt @@ -1,32 +1,31 @@ 😀 😃 😄 😊 🙂 😉 -😍 ❤️ 💕 🥰 😘 +😍 💕 🥰 😘 👍 👍 👏 🙌 🙏 😂 🤣 😅 😆 😅 -😊 😍 ❤️ 😊 +😊 😍 😊 👋 😊 🤗 💛 🎉 🎊 🥳 ✨ 🎁 🎂 🎉 🍰 🥳 🎁 😥 😢 😭 💔 🥺 -😎 😎 🕶️ 😎 +😎 😎 😎 😎 🤔 🤔 🤨 💭 😊 👍 ✅ 💯 -😍 💖 💘 💝 ❤️ +😍 💖 💘 💝 🌷 🌹 🌸 💐 🌻 🐶 🐱 🐰 🐹 🦊 -☀️ 🌈 ⭐ 🌟 ✨ -☕ 🍰 🍪 ☕ 🍰 +🌈 ⭐ 🌟 ✨ +☕ 🍰 🍫 ☕ 🍰 🍕 🍔 🍟 🌭 🍕 🍎 🍌 🍇 🍓 🍊 🥗 🍎 💪 ✅ 👍 🚗 🚌 🚕 🚲 🚗 -✈️ 🗺️ 🏖️ 🌴 ☀️ -📱 💻 ⌨️ 🖥️ 📱 -📞 📧 ✉️ 💌 💬 -💭 🗣️ 💬 🗣️ 💭 +📱 💻 📱 +📞 📧 📧 💬 +💭 💬 💭 🔥 🔥 💯 🔥 -❤️ 🧡 💛 💚 💙 💜 🖤 +🧡 💛 💚 💙 💜 🖤 💔 😢 💔 😭 -😊 😇 🙂 😌 +😊 😇 🙂 😊 😴 😪 🥱 😴 🤒 🤕 😷 🤒 🤢 🤮 🤧 🤒 @@ -34,28 +33,288 @@ 😡 😠 😤 😡 😂 😂 😂 🤣 😉 😜 🤪 😜 -🕺 💃 🎶 🎵 🎉 -🎁 🎁 🎀 🎁 +🕺 💃 🎧 🎧 🎉 +🎁 🎁 🎁 🎁 ✅ ❌ ❓ ❗ ⭐ ⭐ ⭐ ✨ 👍 👎 👌 🤝 -🙏 🙏 🙏 ❤️ -👨‍👩‍👧 👪 💕 🏡 +🙏 🙏 🙏 +👪 💕 👪 👶 👧 👦 💕 -👵 👴 💕 ❤️ +👵 👴 💕 🤦 🤷 💭 🤦 -💪 💪 ✊ 💪 -🍀 ☘️ 🍀 ✨ +💪 💪 💪 💪 +🌸 🌸 🌸 ✨ ⚡ ⚡ 🔥 ⚡ -❄️ ⛄ ☁️ 🌙 -🌞 🌞 ☀️ 🌞 -🌊 🏝️ 🌴 🏖️ -🍕 🍕 🍕 ❤️ +🌙 +🌞 🌞 🌞 +🐳 +🍕 🍕 🍕 🍫 🍬 🍰 🍦 🍵 ☕ 🧃 🥤 -🍷 🥂 🍻 🎉 +🍷 🥂 🍺 🎉 🍞 🧀 🥚 🍳 -🐕 🐈 🐦 🐠 +🐶 🐱 🐦 🐬 🌸 🌸 🌸 💖 -😊 ❤️ 😊 +😊 😊 👍 🙏 💯 ✅ +😃 😋 +😂 😅 🤭 +🤯 😃 +🥰 😢 +😠 🤗 +🥰 🤯 😴 😀 +😥 🧐 🙂 +😅 😎 🤓 +😳 😴 🤔 +🤕 😎 +🤤 🥵 🥴 +🤗 😄 🥰 +🥲 😪 🥴 +😋 🥶 🤓 😎 +😨 🤗 😨 +😳 😉 +😉 😎 🥲 +😥 😪 😮 +😴 🥱 🥳 +🥰 😷 +🥰 😢 🤢 +🙂 😂 🤭 😤 +🤮 🥳 🥰 🥲 +😅 😄 🤣 🥵 +🤯 😎 🤯 🙁 +🤕 🥴 😀 +😨 😤 +🙄 🧐 😜 +😔 🤒 🥴 +🤕 🙄 😢 +🥴 🤪 +🙂 🙄 😤 +😳 🥺 +🤧 🤢 +😍 😍 😡 +😆 🥺 +🤕 😤 +😂 🥺 😠 +😤 😳 🤒 +🤗 🥳 😨 +👉 👇 👌 +👎 👍 👋 +👍 🖖 👎 +🤝 🤲 +👇 🤘 +🤞 👌 🙏 👋 +🤟 👆 👈 👌 +🤙 👉 🤝 👎 +👎 👏 +🤟 🤞 +🤌 🤞 👆 🤝 +👉 🤲 +👎 💪 +💪 🙏 +🤌 👆 🤞 +🤝 🤌 👍 👈 +🤘 🖖 👏 🙌 +🤌 🤘 💪 👋 +🎊 💛 +💝 ✅ 💜 ❓ +✅ 💚 +🔥 🤎 +❓ ⭐ +🌟 💗 +❓ 🤍 🎉 +🤎 💙 💫 +💕 🎊 💚 💖 +💚 🤍 +💙 💞 💚 +💓 💜 ✅ +🌟 💯 💫 +🔥 ✨ +❓ 💜 +❌ 🎊 +💔 🌟 🎉 +🔥 ✅ 💝 +❗ ✅ 💚 +🎁 💛 💞 💙 +💜 💖 🎉 +💙 ❓ +🦉 🌸 🦋 +🐔 🐱 🐸 +🌈 🐼 +🐙 🐷 🦋 +⚡ 🐳 🌞 🐨 +⚡ 🌷 💐 +🐻 🌞 +🌹 💐 🌷 🐯 +🐸 🐱 +🐨 🌷 🐧 🦁 +🐨 🐢 🐵 +💐 🌙 🦁 +🐙 🌸 🐮 🐶 +🐻 🌙 🐯 +🐝 🐷 +🐔 ⚡ 🐰 +🐯 🐬 🐶 +🐸 🐭 🐢 +🐞 🐢 🐰 +🦆 🐙 🐮 +🦉 🐻 🐢 +🐙 🌈 🐞 🦁 +🦁 🐼 +🐧 🐔 🐼 +🌹 🌙 🐳 🐢 +🦆 🐞 🐷 +🐙 🐨 🐼 +🐱 ⚡ +🧀 🍵 +🍚 🌽 🥕 +🥗 🍏 🍦 +🍰 🍒 +🍐 🍑 🍬 🧀 +🍜 🥂 🌭 🍺 +🌮 🍚 🥓 🍫 +🧃 🍛 🥚 🥭 +🍬 🍍 🍰 +🌯 🍍 🍊 🍍 +🍅 🍵 🍊 +🍑 🍣 🍣 +🥕 🥝 +🥕 🫐 🍞 🧃 +☕ 🍦 +🥚 🍺 🥥 🌽 +🍔 🍛 🎂 🧃 +🍒 🍞 🍎 🍅 +🍚 🍬 🥤 🥂 +🧃 🌭 +🍋 🍞 +🍓 🥭 🥝 🧀 +🍦 🌽 🍦 +🥭 🍊 🍏 🍔 +🍒 🍊 🍺 +🍦 🍺 +🍬 🌮 🍑 +🍜 🍕 🍒 🍣 +🍓 🌮 🍛 +🍰 🍇 +🚕 🚗 🚚 +🛵 ⛵ 🚕 ⛵ +🚕 🚤 🗼 +🗽 🚀 +🚜 🛵 +🚕 🚁 🚗 +🗽 🚕 🌋 🚁 +⛵ 🛵 🚤 🌋 +🚁 🏰 🚤 +🛵 🚤 🚒 🗼 +🚚 🚲 +🚀 ⛵ 🚜 🚲 +🚑 🚑 🗽 +🚚 🚀 🗼 +🚨 🚑 +⛵ 🚲 🚤 +🔧 🔒 🎤 📷 +💳 🔌 🔧 +🔑 🔓 💎 +💵 💎 🔦 +📷 📞 📷 +📞 🎨 💡 +🎲 🎧 💡 +🎥 ⌚ 🔓 +📱 📱 💡 +🔧 🎨 📷 +📌 🔑 +💵 🎮 🔌 🎸 +🎧 💳 +📍 🔨 +📸 📌 🔦 +📷 🎬 📏 +🎬 💰 🎹 +🎸 🎲 🔦 +📏 🎨 +📏 🔌 💻 +🎥 💻 ⌚ +🔨 🔑 📱 💡 +🎯 💎 🔨 +🎬 📏 🔧 +📰 💭 🔱 +🚫 📧 📤 +💭 🚫 🔞 +📨 🔱 ⭕ 📵 +📰 💭 🔞 ⭕ +📧 📤 🔱 +📵 🔱 +💭 ⭕ 📤 +🔞 📤 💭 🔱 +📵 💬 📰 📦 +🔞 📦 +📥 📵 🚫 💭 +😭 🖤 +🤓 ✨ +🤓 🤎 +🥴 ✅ +🤕 👌 +🤧 🧡 +🤭 🤞 +😭 ❌ +😁 💗 +🤯 🎈 +😥 🤲 +😠 👆 +😢 🤲 +😁 💖 +🤪 🤟 +🙂 🌟 +😢 ✋ +😠 🤎 +😃 👆 +😨 💪 +😋 🖖 +😊 💫 +🥵 🧡 +🥵 ❌ +😅 💘 +😷 🤎 +😭 🤎 +🤔 💕 +😠 🖖 +🤨 👌 +🥱 ⭐ +🙁 👍 +😤 💝 +😭 🤟 +😉 🎁 +🥰 👈 +😡 🙌 +🥲 💔 +🤕 💜 +🤪 🤝 +🤢 ❌ +😎 🙏 +😡 ⭐ +🤧 🧡 +🤨 👏 +🧐 🤙 +😱 👋 +😠 💕 +😢 🤞 +😤 🎊 +🔌 🥵 +🔧 😒 +🔒 🤓 +📵 😮 +💻 🥶 +💵 😘 +💳 😄 +🔋 🤤 +🔓 🥱 +💵 😒 +🎹 🤢 +🎸 😉 +📌 😄 +🔱 😡 +🎨 😎 +💬 🤕 +🔑 🤣 +📏 😭 +📧 😴 +⭕ 🙁 diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp index 4badbba9..0d2b8dc4 100644 --- a/tests/test_alphabet_xml.cpp +++ b/tests/test_alphabet_xml.cpp @@ -1,6 +1,9 @@ // Alphabet XML parsing tests: verify alphabet loading, switching, and structure #include "test_common.h" +#include +#include + TEST(alphabet_default_loaded) { dasher_ctx* ctx = create_isolated_context(); ASSERT(ctx); @@ -117,7 +120,7 @@ TEST(alphabet_emoji_loads_and_has_groups) { int sym_count = dasher_get_alphabet_symbol_count(ctx); printf(" Emoji symbol count: %d\n", sym_count); - // 9 topical groups (~230 nodes) + control/terminator symbols + // Emoji alphabet loads (expect ~310 symbols: 9 groups, ~307 nodes + control). ASSERT(sym_count > 150); dasher_destroy(ctx); @@ -156,11 +159,223 @@ TEST(alphabet_emoji_training_file_present) { // ordering priors — assert the shipped tree still has it next to the // alphabet so release packaging doesn't silently drop it. std::error_code ec; - bool ok = - std::filesystem::exists(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec); + bool ok = std::filesystem::exists( + std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec); ASSERT(ok); } +TEST(alphabet_emoji_corpus_tokens_are_nodes) { + // Greptile P2: corpus tokens that are not alphabet symbols train + // UNKNOWN_SYMBOL observations — pure noise. Every whitespace-separated + // token must be a whole node text, and (trainer limitation) must be a + // SINGLE code point: the trainer looks up one code point per symbol, + // so multi-codepoint tokens (ZWJ, VS16) can never match and are + // silently split. + std::vector nodes; + dasher_ctx* ctx = create_isolated_context(); + ASSERT(ctx); + dasher_set_screen_size(ctx, 800, 600); + dasher_set_alphabet_id(ctx, "Emoji"); + ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji"); + int sym_count = dasher_get_alphabet_symbol_count(ctx); + for (int i = 0; i < sym_count; i++) { + char buf[128]; + if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') + nodes.push_back(buf); + } + dasher_destroy(ctx); + + std::ifstream in(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt"); + ASSERT(in.is_open()); + std::string line; + int tokens = 0; + while (std::getline(in, line)) { + size_t start = 0; + while (start < line.size()) { + size_t end = line.find(' ', start); + if (end == std::string::npos) end = line.size(); + if (end > start) { + std::string tok = line.substr(start, end - start); + tokens++; + bool known = false; + for (const auto& n : nodes) { + if (n == tok) { + known = true; + break; + } + } + if (!known) { + printf(" unknown corpus token (%zu bytes):", tok.size()); + for (unsigned char ch : tok) printf(" %02x", ch); + printf("\n"); + ASSERT(false); + } + } + start = end + 1; + } + } + printf(" %d corpus tokens, all valid single-codepoint nodes\n", tokens); + ASSERT(tokens > 100); +} + +TEST(alphabet_emoji_output_segments_into_whole_nodes) { + // Greptile P2: prove the OUTPUT path commits multi-codepoint nodes + // atomically — at EVENT level, not byte level. A concatenated-bytes + // check would pass even if 👨‍👩‍👧 arrived as five separate events + // (👨, ZWJ, 👩, ZWJ, 👧); the output callback sees each insert as its + // own event, so requiring every event's text to be a WHOLE node text + // closes that hole. We also require at least one multi-codepoint + // event (>4 bytes ⇒ ZWJ or VS16 carrier) so the multi-byte path is + // actually exercised, not vacuously green. + dasher_ctx* ctx = create_isolated_context(); + ASSERT(ctx); + dasher_set_screen_size(ctx, 800, 600); + dasher_set_alphabet_id(ctx, "Emoji"); + ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji"); + dasher_set_speed_percent(ctx, 300); + + std::vector symbols; + int sym_count = dasher_get_alphabet_symbol_count(ctx); + for (int i = 0; i < sym_count; i++) { + char buf[128]; + if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') + symbols.push_back(buf); + } + ASSERT(symbols.size() > 100); + + std::vector events; + dasher_set_output_callback( + ctx, + [](int event_type, const char* text, void* user_data) { + if (event_type == 0) { // DASHER_EVENT_OUTPUT + static_cast*>(user_data)->push_back(text); + } + }, + &events); + + // Sweep sy across varied bands with the button held. Several passes + // with different y-bands and x-depths; every committed event must be a + // WHOLE node text (byte-level concatenation can't prove atomicity — + // five separate events for 👨‍👩‍👧 produce identical bytes). + const struct { int y0, y1, x; } passes[] = { + {100, 500, 700}, // full safe band (same as test_spell_word) + {150, 350, 720}, // upper-half dwell + {300, 540, 680}, // lower-half dwell + {200, 460, 740}, // deeper zoom + }; + int frame = 0; + for (const auto& p : passes) { + dasher_mouse_down(ctx); + const int span = p.y1 - p.y0; + for (int i = 0; i < 500; i++) { + int sy = p.y0 + (i % span); + dasher_mouse_move(ctx, static_cast(p.x), static_cast(sy)); + int* c = nullptr; + int cc = 0; + char** s = nullptr; + int sc = 0; + dasher_frame(ctx, 1000 + (frame++) * 16, &c, &cc, &s, &sc); + } + dasher_mouse_up(ctx); + if (events.size() > 0) break; + } + + size_t total = 0; + for (const auto& ev : events) { + total += ev.size(); + bool whole = false; + for (const auto& sym : symbols) { + if (sym == ev) { + whole = true; + break; + } + } + if (!whole) { + printf(" NON-ATOMIC event (%zu bytes):", ev.size()); + for (unsigned char ch : ev) printf(" %02x", ch); + printf("\n"); + } + ASSERT(whole); + } + printf(" %zu output events, %zu bytes — every event a whole node text\n", events.size(), total); + ASSERT(events.size() > 0); + + dasher_destroy(ctx); +} + +TEST(alphabet_multicodepoint_commit_is_atomic) { + // Deterministic multi-codepoint coverage: a purpose-built 3-node test + // alphabet (ZWJ sequence, VS16 carrier, plain emoji) where EVERY node + // but one is multi-codepoint and each holds ~1/3 of the tree mass — + // navigation cannot avoid committing them. Complements the sweep test + // above, whose mass distribution follows the training priors and may + // not reach the shipped alphabet's low-mass multi-codepoint nodes. + ScopedTempDir dataRoot; + const std::string data_dir = build_data_dir(dataRoot); + // No training file: uniform-ish symbol ordering (the engine's + // documented no-training fallback). + std::string xml = std::string("\n") + + "\n" + + "\n" + + " \n" + + " \n" + + " \n" + + " \n" + + " \n" + + "\n"; + ASSERT(write_data_file(data_dir, "alphabets", "alphabet.zwjtest.xml", xml)); + + dasher_ctx* ctx = dasher_create(data_dir.c_str(), dataRoot.c_str(), nullptr); + ASSERT(ctx); + dasher_set_screen_size(ctx, 800, 600); + printf(" custom dir alphabets: %d\n", dasher_get_alphabet_count(ctx)); + dasher_set_alphabet_id(ctx, "ZWJ Test"); + ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "ZWJ Test"); + + const std::string family = "\xF0\x9F\x91\xA8\xE2\x80\x8D\xF0\x9F\x91\xA9\xE2\x80\x8D\xF0\x9F\x91\xA7"; + const std::string plane = "\xE2\x9C\x88\xEF\xB8\x8F"; + const std::string grin = "\xF0\x9F\x98\x80"; + + std::vector events; + dasher_set_output_callback( + ctx, + [](int event_type, const char* text, void* user_data) { + if (event_type == 0) { static_cast*>(user_data)->emplace_back(text); } + }, + &events); + + dasher_set_speed_percent(ctx, 300); + dasher_mouse_down(ctx); + for (int i = 0; i < 600; i++) { + int sy = 150 + (i % 300); + dasher_mouse_move(ctx, 700.0f, static_cast(sy)); + int* c = nullptr; + int cc = 0; + char** s = nullptr; + int sc = 0; + dasher_frame(ctx, 1000 + i * 16, &c, &cc, &s, &sc); + } + dasher_mouse_up(ctx); + + size_t multi = 0; + for (const auto& ev : events) { + bool known = (ev == family || ev == plane || ev == grin); + if (!known) { + printf(" NON-ATOMIC event (%zu bytes):", ev.size()); + for (unsigned char ch : ev) printf(" %02x", ch); + printf("\n"); + } + ASSERT(known); + if (ev.size() > 4) multi++; + } + printf(" %zu events, %zu multi-codepoint commits\n", events.size(), multi); + ASSERT(events.size() > 0); + ASSERT(multi > 0); + + dasher_destroy(ctx); +} + + TEST(alphabet_switch_changes_probabilities) { dasher_ctx* ctx = create_isolated_context(); ASSERT(ctx); From 13717c23d978c1ef83b2de87ef86410f0281c546 Mon Sep 17 00:00:00 2001 From: will wade Date: Fri, 18 Sep 2026 18:13:10 +0100 Subject: [PATCH 3/4] =?UTF-8?q?style(emoji):=20clang-format=20+=20review?= =?UTF-8?q?=20nits=20=E2=80=94=20symbol=20range,=20event=20macro?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - clang-format over the new tests (CI pins format 18 with --Werror) - symbol scans use the documented 1..sym_count inclusive range (0 = root), matching the file's existing v6 tests - DASHER_EVENT_OUTPUT macro instead of magic 0 (matches test_interaction.cpp style) Signed-off-by: will wade --- tests/test_alphabet_xml.cpp | 50 +++++++++++++++++++------------------ 1 file changed, 26 insertions(+), 24 deletions(-) diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp index 0d2b8dc4..54943196 100644 --- a/tests/test_alphabet_xml.cpp +++ b/tests/test_alphabet_xml.cpp @@ -139,7 +139,7 @@ TEST(alphabet_emoji_zwj_sequence_roundtrip) { const char* toned = "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD"; // U+1F44D U+1F3FD bool found_family = false, found_toned = false, found_space = false; int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 0; i < sym_count; i++) { + for (int i = 1; i <= sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue; if (strcmp(buf, family) == 0) found_family = true; @@ -159,8 +159,8 @@ TEST(alphabet_emoji_training_file_present) { // ordering priors — assert the shipped tree still has it next to the // alphabet so release packaging doesn't silently drop it. std::error_code ec; - bool ok = std::filesystem::exists( - std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec); + bool ok = + std::filesystem::exists(std::filesystem::path(TEST_DATA_DIR) / "Data" / "training" / "training_emoji.txt", ec); ASSERT(ok); } @@ -178,10 +178,9 @@ TEST(alphabet_emoji_corpus_tokens_are_nodes) { dasher_set_alphabet_id(ctx, "Emoji"); ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji"); int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 0; i < sym_count; i++) { + for (int i = 1; i <= sym_count; i++) { char buf[128]; - if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') - nodes.push_back(buf); + if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') nodes.push_back(buf); } dasher_destroy(ctx); @@ -206,7 +205,8 @@ TEST(alphabet_emoji_corpus_tokens_are_nodes) { } if (!known) { printf(" unknown corpus token (%zu bytes):", tok.size()); - for (unsigned char ch : tok) printf(" %02x", ch); + for (unsigned char ch : tok) + printf(" %02x", ch); printf("\n"); ASSERT(false); } @@ -236,10 +236,9 @@ TEST(alphabet_emoji_output_segments_into_whole_nodes) { std::vector symbols; int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 0; i < sym_count; i++) { + for (int i = 1; i <= sym_count; i++) { char buf[128]; - if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') - symbols.push_back(buf); + if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') symbols.push_back(buf); } ASSERT(symbols.size() > 100); @@ -247,7 +246,7 @@ TEST(alphabet_emoji_output_segments_into_whole_nodes) { dasher_set_output_callback( ctx, [](int event_type, const char* text, void* user_data) { - if (event_type == 0) { // DASHER_EVENT_OUTPUT + if (event_type == DASHER_EVENT_OUTPUT) { static_cast*>(user_data)->push_back(text); } }, @@ -257,7 +256,9 @@ TEST(alphabet_emoji_output_segments_into_whole_nodes) { // with different y-bands and x-depths; every committed event must be a // WHOLE node text (byte-level concatenation can't prove atomicity — // five separate events for 👨‍👩‍👧 produce identical bytes). - const struct { int y0, y1, x; } passes[] = { + const struct { + int y0, y1, x; + } passes[] = { {100, 500, 700}, // full safe band (same as test_spell_word) {150, 350, 720}, // upper-half dwell {300, 540, 680}, // lower-half dwell @@ -292,7 +293,8 @@ TEST(alphabet_emoji_output_segments_into_whole_nodes) { } if (!whole) { printf(" NON-ATOMIC event (%zu bytes):", ev.size()); - for (unsigned char ch : ev) printf(" %02x", ch); + for (unsigned char ch : ev) + printf(" %02x", ch); printf("\n"); } ASSERT(whole); @@ -315,14 +317,12 @@ TEST(alphabet_multicodepoint_commit_is_atomic) { // No training file: uniform-ish symbol ordering (the engine's // documented no-training fallback). std::string xml = std::string("\n") + - "\n" + - "\n" + - " \n" + - " \n" + - " \n" + - " \n" + - " \n" + - "\n"; + "\n" + + "\n" + + " \n" + + " \n" + + " \n" + + " \n" + " \n" + "\n"; ASSERT(write_data_file(data_dir, "alphabets", "alphabet.zwjtest.xml", xml)); dasher_ctx* ctx = dasher_create(data_dir.c_str(), dataRoot.c_str(), nullptr); @@ -340,7 +340,9 @@ TEST(alphabet_multicodepoint_commit_is_atomic) { dasher_set_output_callback( ctx, [](int event_type, const char* text, void* user_data) { - if (event_type == 0) { static_cast*>(user_data)->emplace_back(text); } + if (event_type == DASHER_EVENT_OUTPUT) { + static_cast*>(user_data)->emplace_back(text); + } }, &events); @@ -362,7 +364,8 @@ TEST(alphabet_multicodepoint_commit_is_atomic) { bool known = (ev == family || ev == plane || ev == grin); if (!known) { printf(" NON-ATOMIC event (%zu bytes):", ev.size()); - for (unsigned char ch : ev) printf(" %02x", ch); + for (unsigned char ch : ev) + printf(" %02x", ch); printf("\n"); } ASSERT(known); @@ -375,7 +378,6 @@ TEST(alphabet_multicodepoint_commit_is_atomic) { dasher_destroy(ctx); } - TEST(alphabet_switch_changes_probabilities) { dasher_ctx* ctx = create_isolated_context(); ASSERT(ctx); From 0957afda8e383dc246c31838a7947c00acd77527 Mon Sep 17 00:00:00 2001 From: will wade Date: Sat, 19 Sep 2026 01:16:54 +0100 Subject: [PATCH 4/4] =?UTF-8?q?fix(emoji):=20greptile=20round=202=20?= =?UTF-8?q?=E2=80=94=20single-codepoint=20enforcement,=20index=20bound?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Corpus test now enforces the single-codepoint invariant explicitly (UTF-8 lead-byte count): a multi-codepoint NODE would pass a pure membership check, yet the trainer (one code point per symbol lookup) can never match it — membership alone was insufficient. - Symbol scans use the exclusive bound: CAPI rejects index >= iEnd (dasher_get_alphabet_symbol_text returns -1), so i < sym_count is the correct range; the previous <= probed an invalid index. Signed-off-by: will wade --- tests/test_alphabet_xml.cpp | 29 ++++++++++++++++++++++------- 1 file changed, 22 insertions(+), 7 deletions(-) diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp index 54943196..793925d8 100644 --- a/tests/test_alphabet_xml.cpp +++ b/tests/test_alphabet_xml.cpp @@ -139,7 +139,7 @@ TEST(alphabet_emoji_zwj_sequence_roundtrip) { const char* toned = "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD"; // U+1F44D U+1F3FD bool found_family = false, found_toned = false, found_space = false; int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue; if (strcmp(buf, family) == 0) found_family = true; @@ -178,7 +178,7 @@ TEST(alphabet_emoji_corpus_tokens_are_nodes) { dasher_set_alphabet_id(ctx, "Emoji"); ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), "Emoji"); int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') nodes.push_back(buf); } @@ -210,6 +210,21 @@ TEST(alphabet_emoji_corpus_tokens_are_nodes) { printf("\n"); ASSERT(false); } + // Greptile P2: membership alone is insufficient — a + // multi-codepoint NODE (👨‍👩‍👧) would pass even though the + // trainer looks up one code point per symbol and can never + // match it. Enforce single-codepoint tokens explicitly: + // count UTF-8 lead bytes (non-continuation). + int codepoints = 0; + for (unsigned char ch : tok) + if ((ch & 0xC0) != 0x80) codepoints++; + if (codepoints != 1) { + printf(" multi-codepoint corpus token (%d codepoints):", codepoints); + for (unsigned char ch : tok) + printf(" %02x", ch); + printf("\n"); + ASSERT(false); + } } start = end + 1; } @@ -236,7 +251,7 @@ TEST(alphabet_emoji_output_segments_into_whole_nodes) { std::vector symbols; int sym_count = dasher_get_alphabet_symbol_count(ctx); - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && buf[0] != '\0') symbols.push_back(buf); } @@ -459,7 +474,7 @@ TEST(alphabet_v6_space_character_resolves_to_space) { // Valid symbol indices are 1..sym_count inclusive (index 0 is the sentinel). bool found_space = false; - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && strcmp(buf, " ") == 0) { found_space = true; @@ -504,7 +519,7 @@ TEST(alphabet_v6_paragraph_outputs_newline) { ASSERT(sym_count > 0); bool found_paragraph_display = false, paragraph_is_newline = false; - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char disp[128], text[128]; if (dasher_get_alphabet_symbol_display(ctx, i, disp, sizeof(disp)) != 0) continue; if (strcmp(disp, "\xc2\xb6") != 0) continue; // UTF-8 pilcrow @@ -598,7 +613,7 @@ TEST(alphabet_v5_symbols_have_correct_text) { ASSERT(sym_count >= 3); bool found_x = false, found_space = false, found_emoji = false; - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue; std::string s(buf); @@ -703,7 +718,7 @@ TEST(alphabet_v5_special_chars_as_direct_children) { // Scan all symbols for the expected text values. bool found_letter_a = false, found_space = false, found_newline = false; - for (int i = 1; i <= sym_count; i++) { + for (int i = 1; i < sym_count; i++) { char buf[128]; if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) != 0) continue; std::string s(buf);