diff --git a/Data/emoji/extension.xml b/Data/emoji/extension.xml
new file mode 100644
index 00000000..71ab7e6f
--- /dev/null
+++ b/Data/emoji/extension.xml
@@ -0,0 +1,362 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/settings_manifest.json b/settings_manifest.json
index 1f131d74..a49fd4c1 100644
--- a/settings_manifest.json
+++ b/settings_manifest.json
@@ -1403,6 +1403,31 @@
"subgroup": "History",
"uiType": "Enum"
},
+ {
+ "key": "BP_EMOJI_GROUP",
+ "storageName": "EmojiGroup",
+ "type": "bool",
+ "default": true,
+ "label": "Emoji in Alphabet",
+ "description": "Add an emoji group to every alphabet, so emoji are available in the node tree alongside your language (RFC 0020). Emoji are learned in context as you type.",
+ "uiType": "Switch",
+ "tier": "common",
+ "group": "Language",
+ "subgroup": "Emoji"
+ },
+ {
+ "key": "SP_EMOJI_SKIN_TONE",
+ "storageName": "EmojiSkinTone",
+ "type": "string",
+ "default": "none",
+ "label": "Preferred Emoji Skin Tone",
+ "description": "When set, only the base gesture and the preferred skin-tone variant are added to the emoji group (RFC 0020). 'None' adds all variants; the tone you use most rises through adaptation either way.",
+ "tier": "common",
+ "group": "Language",
+ "subgroup": "Emoji",
+ "uiType": "Enum",
+ "values": ["none", "light", "medium-light", "medium", "medium-dark", "dark"]
+ },
{
"key": "SP_COLOUR_ID",
"storageName": "ColourID",
diff --git a/src/DasherCore/Alphabet/AlphIO.cpp b/src/DasherCore/Alphabet/AlphIO.cpp
index 42d0f2c5..d265fb92 100644
--- a/src/DasherCore/Alphabet/AlphIO.cpp
+++ b/src/DasherCore/Alphabet/AlphIO.cpp
@@ -120,7 +120,8 @@ CAlphIO::CAlphIO(CMessageDisplay* pMsgs) : AbstractXMLParser(pMsgs) {
}
SGroupInfo* CAlphIO::ParseGroupRecursive(pugi::xml_node& group_node, CAlphInfo* CurrentAlphabet,
- SGroupInfo* previous_sibling, std::vector ancestors) {
+ SGroupInfo* previous_sibling, std::vector ancestors,
+ const std::string* toneFilter) {
SGroupInfo* pNewGroup = new SGroupInfo();
pNewGroup->iNumChildNodes = 0;
pNewGroup->strName = group_node.attribute("name").as_string("");
@@ -139,6 +140,13 @@ SGroupInfo* CAlphIO::ParseGroupRecursive(pugi::xml_node& group_node, CAlphInfo*
for (auto node : group_node.children()) {
// symbol (v6 "node" or v5 "s")
if (std::strcmp(node.name(), "node") == 0 || std::strcmp(node.name(), "s") == 0) {
+ // RFC 0020 skin-tone filter: nodes carrying a tone= attribute are
+ // variants; a filter keeps the base (unmarked) nodes plus exactly
+ // the preferred variant.
+ if (toneFilter) {
+ pugi::xml_attribute tone = node.attribute("tone");
+ if (!tone.empty() && tone.as_string() != *toneFilter) continue;
+ }
CurrentAlphabet->m_vCharacters.resize(CurrentAlphabet->m_vCharacters.size() + 1); // new char
CurrentAlphabet->m_vCharacterDoActions.resize(CurrentAlphabet->m_vCharacterDoActions.size() +
1); // new Do Actions
@@ -153,7 +161,7 @@ SGroupInfo* CAlphIO::ParseGroupRecursive(pugi::xml_node& group_node, CAlphInfo*
// group
if (std::strcmp(node.name(), "group") == 0) {
SGroupInfo* newChildGroup =
- ParseGroupRecursive(node, CurrentAlphabet, previous_subgroup_sibling, new_ancestors);
+ ParseGroupRecursive(node, CurrentAlphabet, previous_subgroup_sibling, new_ancestors, toneFilter);
if (newChildGroup == nullptr) continue;
pNewGroup->iNumChildNodes++;
pNewGroup->pChild = newChildGroup;
@@ -190,6 +198,30 @@ bool Dasher::CAlphIO::Parse(pugi::xml_document& document, const std::string, boo
if (std::strcmp(alphabet.name(), "alphabet") != 0) return false; // a non node
+ CAlphInfo* CurrentAlphabet = ParseAlphabet(alphabet, isV5);
+
+ auto it = Alphabets.find(CurrentAlphabet->AlphID);
+ if (it != Alphabets.end()) {
+ // v5 alphabets are legacy (e.g. the bundled oldAlphabets/ files). Never
+ // let them overwrite a v6 alphabet that already loaded under the same
+ // name โ the v6 version is authoritative. User-authored v5 files with
+ // unique names still load normally.
+ if (isV5) {
+ delete CurrentAlphabet;
+ return true;
+ }
+ delete it->second;
+ }
+ Alphabets[CurrentAlphabet->AlphID] = CurrentAlphabet;
+
+ return true;
+}
+
+// Parse one element into a fresh, caller-owned CAlphInfo.
+// Extracted from Parse so MakeExtendedInfo (RFC 0020) can obtain a private
+// copy โ CAlphInfo's destructor owns its ControlActions, so sharing action
+// pointers between the shared base info and a derived one would double-free.
+CAlphInfo* CAlphIO::ParseAlphabet(pugi::xml_node& alphabet, bool isV5) {
CAlphInfo* CurrentAlphabet = new CAlphInfo();
CurrentAlphabet->AlphID = alphabet.attribute("name").as_string();
CurrentAlphabet->TrainingFile = alphabet.attribute("trainingFilename").as_string();
@@ -305,21 +337,7 @@ bool Dasher::CAlphIO::Parse(pugi::xml_document& document, const std::string, boo
// child groups were added (to linked list) in reverse order. Put them in (iStart/iEnd) order...
ReverseChildList(CurrentAlphabet->pChild);
- auto it = Alphabets.find(CurrentAlphabet->AlphID);
- if (it != Alphabets.end()) {
- // v5 alphabets are legacy (e.g. the bundled oldAlphabets/ files). Never
- // let them overwrite a v6 alphabet that already loaded under the same
- // name โ the v6 version is authoritative. User-authored v5 files with
- // unique names still load normally.
- if (isV5) {
- delete CurrentAlphabet;
- return true;
- }
- delete it->second;
- }
- Alphabets[CurrentAlphabet->AlphID] = CurrentAlphabet;
-
- return true;
+ return CurrentAlphabet;
}
void CAlphIO::GetAlphabets(std::vector* AlphabetList) const {
@@ -371,6 +389,76 @@ bool CAlphIO::LoadAlphabetFile(const std::string& filename) {
return true;
}
+// โโ Emoji extension (RFC 0020) โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
+
+bool CAlphIO::LoadEmojiExtension(const std::string& filename) {
+ m_emojiExtensionGroups.clear();
+ if (filename.empty()) return false;
+ std::ifstream in(filename.c_str(), std::ios::binary);
+ if (!in.good()) return false;
+ pugi::xml_parse_result result = m_emojiExtensionDoc.load(in);
+ if (!result) return false;
+ pugi::xml_node root = m_emojiExtensionDoc.document_element();
+ if (std::strcmp(root.name(), "emoji-extension") != 0) return false;
+ for (pugi::xml_node& child : root.children())
+ if (std::strcmp(child.name(), "group") == 0) m_emojiExtensionGroups.push_back(child);
+ return !m_emojiExtensionGroups.empty();
+}
+
+const CAlphInfo* CAlphIO::MakeExtendedInfo(const std::string& AlphID, const std::string& skinTone) {
+ const CAlphInfo* base = GetInfo(AlphID);
+ if (m_emojiExtensionGroups.empty()) return base;
+
+ // A derived info owns its ControlActions (the base's destructor deletes
+ // its own), so re-parse the alphabet's file into a private copy rather
+ // than cloning the shared info. The emergency built-in "Default" has no
+ // file โ it stays unextended (documented RFC 0020 limitation).
+ const std::string file = FileNameFor(AlphID);
+ if (file.empty() || AlphID == "Default") return base;
+
+ std::ifstream in(file.c_str(), std::ios::binary);
+ if (!in.good()) return base;
+ pugi::xml_document doc;
+ if (!doc.load(in)) return base;
+ pugi::xml_node alphabet = doc.document_element();
+ bool isV5 = (std::strcmp(alphabet.name(), "alphabets") == 0);
+ if (isV5) alphabet = alphabet.child("alphabet");
+ if (!alphabet || std::strcmp(alphabet.name(), "alphabet") != 0) return base;
+
+ CAlphInfo* derived = ParseAlphabet(alphabet, isV5);
+ // Guard against index/registry divergence (e.g. a user-dir v5 file
+ // shadowing a v6 name in the filename index but not the registry):
+ // the re-parse must produce the requested alphabet, else serve the
+ // registered base rather than a divergent copy.
+ if (derived->AlphID != AlphID) {
+ delete derived;
+ return base;
+ }
+
+ // Append the extension's groups after the alphabet's own (ParseGroupRecursive
+ // appends characters at the end and links groups as reverse siblings).
+ std::string toneFilter(skinTone != "none" ? skinTone : "");
+ const std::string* pToneFilter = toneFilter.empty() ? nullptr : &toneFilter;
+ // Append the extension's groups AFTER the alphabet's own chain: start
+ // at the base chain's tail so the reverse below yields
+ // baseโโฆโbaseโextโโฆโext. Starting at nullptr (the original bug) orphaned
+ // the base group chain โ leaking its SGroupInfo tree and dropping every
+ // base symbol's group coverage.
+ SGroupInfo* previous_sibling = derived->pChild;
+ if (previous_sibling)
+ for (; previous_sibling->pNext; previous_sibling = previous_sibling->pNext);
+ for (pugi::xml_node& group : m_emojiExtensionGroups) {
+ SGroupInfo* newGroup = ParseGroupRecursive(group, derived, previous_sibling, {}, pToneFilter);
+ if (!newGroup) continue;
+ derived->iNumChildNodes++;
+ derived->pChild = newGroup; // last parsed; the reverse below restores order
+ previous_sibling = newGroup;
+ }
+ derived->iEnd = static_cast(derived->m_vCharacters.size()) + 1;
+ ReverseChildList(derived->pChild);
+ return derived;
+}
+
std::string CAlphIO::GetDefault() const {
std::string DefaultExternalAlphabet = "English with limited punctuation";
if (Alphabets.count(DefaultExternalAlphabet) != 0) {
diff --git a/src/DasherCore/Alphabet/AlphIO.h b/src/DasherCore/Alphabet/AlphIO.h
index c24cda09..12822735 100644
--- a/src/DasherCore/Alphabet/AlphIO.h
+++ b/src/DasherCore/Alphabet/AlphIO.h
@@ -76,16 +76,37 @@ class Dasher::CAlphIO : public AbstractXMLParser {
/// Full-parse one alphabet file by (absolute or pattern) filename.
bool LoadAlphabetFile(const std::string& filename);
+ /// โโ Emoji extension (RFC 0020) โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
+ /// Parse and cache the groups-only extension definition
+ /// (Data/emoji/extension.xml). Safe to call when absent.
+ /// \return true when an extension was loaded.
+ bool LoadEmojiExtension(const std::string& filename);
+
+ /// True when an emoji extension definition is loaded.
+ bool HasEmojiExtension() const { return !m_emojiExtensionGroups.empty(); }
+
+ /// Merge the emoji extension into a DEEP COPY of the named alphabet.
+ /// \param skinTone "none" keeps all tone variants; any other value
+ /// keeps only base gestures plus that tone's variants.
+ /// \return a caller-owned CAlphInfo (delete when the alphabet
+ /// changes), or the shared base info when no extension is loaded โ
+ /// callers must treat the result as const and never delete it then.
+ const CAlphInfo* MakeExtendedInfo(const std::string& AlphID, const std::string& skinTone);
+
private:
std::map Alphabets; // map AlphabetID to AlphabetInfo.
std::map AlphabetFiles; // name index: AlphID โ filename
static CAlphInfo*
CreateDefault(); // Give the user an English alphabet rather than nothing if anything goes horribly wrong.
+ std::vector m_emojiExtensionGroups; // cached nodes
+ pugi::xml_document m_emojiExtensionDoc; // owns the cached nodes
void ReadCharAttributes(pugi::xml_node xml_node, CAlphInfo::character& alphabet_character, SGroupInfo* parentGroup,
std::vector& DoActions, std::vector& UndoActions);
+ CAlphInfo* ParseAlphabet(pugi::xml_node& alphabet, bool isV5);
SGroupInfo* ParseGroupRecursive(pugi::xml_node& group_node, CAlphInfo* CurrentAlphabet,
- SGroupInfo* previous_sibling, std::vector ancestors);
+ SGroupInfo* previous_sibling, std::vector ancestors,
+ const std::string* toneFilter = nullptr);
void ReverseChildList(SGroupInfo*& pList);
// Alphabet types:
std::map AlphabetStringToType;
diff --git a/src/DasherCore/AlphabetManager.cpp b/src/DasherCore/AlphabetManager.cpp
index c2d4983b..e9908ef0 100644
--- a/src/DasherCore/AlphabetManager.cpp
+++ b/src/DasherCore/AlphabetManager.cpp
@@ -602,8 +602,16 @@ void CAlphabetManager::IterateChildGroups(CAlphNode* pParent, const SGroupInfo*
DASHER_ASSERT((*pCProb)[0] == 0);
const int iMin(pParentGroup->iStart);
const int iMax(pParentGroup->iEnd);
- unsigned int iRange(pParentGroup == m_pBaseGroup ? CDasherModel::NORMALIZATION
- : ((*pCProb)[iMax - 1] - (*pCProb)[iMin - 1]));
+ // Scale children against the LM's ACTUAL cumulative total rather than
+ // assuming it fills NORMALIZATION: PPM reserves escape/uniform mass for
+ // unseen text, and with a large untrained symbol set (RFC 0020's merged
+ // emoji extension, before adaptation lifts what the user types) that
+ // residual is no longer negligible โ assuming 65536 left the last child
+ // short and the tree didn't fill the screen. For fully-trained
+ // alphabets the total IS 65536 and this is a no-op.
+ const unsigned int iActualRange((*pCProb)[iMax - 1] - (*pCProb)[iMin - 1]);
+ DASHER_ASSERT(iActualRange > 0);
+ unsigned int iRange(iActualRange);
// TODO: Think through alphabet file formats etc. to make this class easier.
// TODO: Throw a warning if parent node already has children
diff --git a/src/DasherCore/DasherInterfaceBase.cpp b/src/DasherCore/DasherInterfaceBase.cpp
index 3979e1b4..de4bc002 100644
--- a/src/DasherCore/DasherInterfaceBase.cpp
+++ b/src/DasherCore/DasherInterfaceBase.cpp
@@ -117,6 +117,15 @@ void CDasherInterfaceBase::Realize(unsigned long ulTime) {
// THIS realize read the other bundle (greptile P1 #2 on #88 โ pinned by
// retired_default_heal_uses_own_context_data_dir).
m_AlphIO->ScanNameIndex(m_dataDir);
+ // RFC 0020: load the emoji extension definition (groups-only, merged
+ // into every alphabet when BP_EMOJI_GROUP is on). The engine resolves
+ // data against {dataDir, dataDir/Data} โ try both, tolerate absence.
+ {
+ const std::string sep = (m_dataDir.empty() || m_dataDir.back() == '/') ? "" : "/";
+ for (const std::string& sub : {std::string("emoji/extension.xml"), std::string("Data/emoji/extension.xml")}) {
+ if (m_AlphIO->LoadEmojiExtension(m_dataDir + sep + sub)) break;
+ }
+ }
const auto loadById = [this](const std::string& alphId) { LoadAlphabetById(alphId); };
{
std::string alphId = m_pSettingsStore->GetStringParameter(SP_ALPHABET_ID);
@@ -219,6 +228,13 @@ CDasherInterfaceBase::~CDasherInterfaceBase() {
// Clean up cached lock label (created by Redraw when locked)
delete m_pLockLabel;
+
+ // RFC 0020: derived infos own their ControlActions โ free them once the
+ // model (which referenced them) is gone.
+ for (const CAlphInfo* info : m_vExtendedAlphInfos)
+ delete info;
+ m_vExtendedAlphInfos.clear();
+ m_pExtendedAlphInfo = nullptr;
}
void CDasherInterfaceBase::HandleParameterChange(Parameter parameter) {
@@ -241,6 +257,11 @@ void CDasherInterfaceBase::HandleParameterChange(Parameter parameter) {
ChangeAlphabet();
ScheduleRedraw();
break;
+ case BP_EMOJI_GROUP: // RFC 0020 โ the merged emoji group changes the
+ case SP_EMOJI_SKIN_TONE: // alphabet's symbol set; rebuild like a switch
+ ChangeAlphabet();
+ ScheduleRedraw();
+ break;
case SP_COLOUR_ID:
ChangeColors();
ScheduleRedraw();
@@ -646,7 +667,24 @@ double CDasherInterfaceBase::GetCurFPS() {
}
const CAlphInfo* CDasherInterfaceBase::GetActiveAlphabet() {
- return m_AlphIO->GetInfo(m_pSettingsStore->GetStringParameter(SP_ALPHABET_ID));
+ const std::string alphId = m_pSettingsStore->GetStringParameter(SP_ALPHABET_ID);
+ if (!m_pSettingsStore->GetBoolParameter(BP_EMOJI_GROUP) || !m_AlphIO->HasEmojiExtension())
+ return m_AlphIO->GetInfo(alphId);
+
+ // RFC 0020: merge the emoji extension into a derived copy. Cached per
+ // (alphabet, tone): GetActiveAlphabet is called per frame by CAPI
+ // consumers, and each merge re-parses the alphabet file.
+ const std::string tone = m_pSettingsStore->GetStringParameter(SP_EMOJI_SKIN_TONE);
+ const std::string key = alphId + '\x1f' + tone;
+ if (m_pExtendedAlphInfo && m_strExtendedKey == key) return m_pExtendedAlphInfo;
+
+ const CAlphInfo* base = m_AlphIO->GetInfo(alphId);
+ const CAlphInfo* derived = m_AlphIO->MakeExtendedInfo(alphId, tone);
+ m_pExtendedAlphInfo = derived;
+ m_strExtendedKey = key;
+ if (derived != base) // MakeExtendedInfo falls back to the shared base
+ m_vExtendedAlphInfos.push_back(derived); // info when nothing merged
+ return derived;
}
// int CDasherInterfaceBase::GetAutoOffset() {
diff --git a/src/DasherCore/DasherInterfaceBase.h b/src/DasherCore/DasherInterfaceBase.h
index e1b00c27..3b61a624 100644
--- a/src/DasherCore/DasherInterfaceBase.h
+++ b/src/DasherCore/DasherInterfaceBase.h
@@ -546,6 +546,17 @@ class Dasher::CDasherInterfaceBase : public CMessageDisplay, private NoClones {
std::unique_ptr m_ColorIO;
std::unique_ptr m_pNCManager;
+ // โโ Emoji extension (RFC 0020) โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
+ // Derived alphabet infos (base + merged emoji groups) are cached per
+ // (alphabetId, skinTone) and kept alive for the process lifetime:
+ // node trees built against a previous info may outlive the switch, and
+ // each derived info owns its ControlActions (the shared base infos'
+ // destructors would double-free them). Bounded by user-driven
+ // alphabet/tone changes.
+ std::vector m_vExtendedAlphInfos;
+ const CAlphInfo* m_pExtendedAlphInfo = nullptr;
+ std::string m_strExtendedKey;
+
// the game mode module - only
// initialized if game mode is enabled
std::unique_ptr m_pGameModule;
diff --git a/src/DasherCore/NodeCreationManager.cpp b/src/DasherCore/NodeCreationManager.cpp
index d9cd622e..e5302fce 100644
--- a/src/DasherCore/NodeCreationManager.cpp
+++ b/src/DasherCore/NodeCreationManager.cpp
@@ -70,7 +70,13 @@ CNodeCreationManager::CNodeCreationManager(CSettingsStore* pSettingsStore, CDash
: m_pInterface(pInterface), m_pScreen(nullptr), m_pSettingsStore(pSettingsStore) {
m_pSettingsStore->OnParameterChanged.Subscribe(this, [this](const Parameter p) { HandleParameterChange(p); });
- const Dasher::CAlphInfo* pAlphInfo(pAlphIO->GetInfo(m_pSettingsStore->GetStringParameter(SP_ALPHABET_ID)));
+ // RFC 0020: the ACTIVE alphabet (base + emoji extension when
+ // BP_EMOJI_GROUP is on) โ NOT the registered base. GetActiveAlphabet
+ // serves the derived, merged info; building from the registered base
+ // here would render and train a tree the extension never touches,
+ // leaving the CAPI split-brained (symbol count includes emoji, node
+ // tree doesn't).
+ const Dasher::CAlphInfo* pAlphInfo(pInterface->GetActiveAlphabet());
switch (pAlphInfo->m_iConversionID) {
case CAlphInfo::None: // No conversion required
diff --git a/src/DasherCore/Parameters.cpp b/src/DasherCore/Parameters.cpp
index ceadd4cd..be12ad16 100644
--- a/src/DasherCore/Parameters.cpp
+++ b/src/DasherCore/Parameters.cpp
@@ -477,6 +477,17 @@ const std::unordered_map parameter_defaults =
{SP_ALPHABET_4, Parameter_Value{"Alphabet4", PARAM_STRING, Persistence::PERSISTENT, std::string(""),
"Alphabet History 4.", "Alphabet History 4", Settings::UIControlType::Enum, true,
"SP_ALPHABET_4", "History", "Language"}},
+ {BP_EMOJI_GROUP, Parameter_Value{"EmojiGroup", PARAM_BOOL, Persistence::PERSISTENT, true,
+ "Add an emoji group to every alphabet, so emoji are available in the node tree "
+ "alongside your language (RFC 0020). Emoji are learned in context as you type.",
+ "Emoji in Alphabet", Settings::UIControlType::Switch, false, "BP_EMOJI_GROUP",
+ "Emoji", "Language"}},
+ {SP_EMOJI_SKIN_TONE,
+ Parameter_Value{"EmojiSkinTone", PARAM_STRING, Persistence::PERSISTENT, std::string("none"),
+ "When set, only the base gesture and the preferred skin-tone variant are added to the emoji group "
+ "(RFC 0020). 'None' adds all variants; the tone you use most rises through adaptation either way.",
+ "Preferred Emoji Skin Tone", Settings::UIControlType::Enum, false, "SP_EMOJI_SKIN_TONE", "Emoji",
+ "Language"}},
{SP_COLOUR_ID,
Parameter_Value{"ColourID", PARAM_STRING, Persistence::PERSISTENT, std::string("Default"), "ColourID.",
"Color Palette", Settings::UIControlType::Enum, false, "SP_COLOUR_ID", "Themes", "Customization"}},
diff --git a/src/DasherCore/Parameters.h b/src/DasherCore/Parameters.h
index 74942c11..dfee221f 100644
--- a/src/DasherCore/Parameters.h
+++ b/src/DasherCore/Parameters.h
@@ -39,6 +39,7 @@ enum Parameter {
BP_SIMULATE_TRANSPARENCY,
BP_CONTROL_MODE,
BP_SLOW_CONTROL_BOX,
+ BP_EMOJI_GROUP,
END_OF_BPS,
LP_ORIENTATION,
@@ -121,6 +122,7 @@ enum Parameter {
SP_BUTTON_MAPPINGS,
SP_JOYSTICK_XAXIS,
SP_JOYSTICK_YAXIS,
+ SP_EMOJI_SKIN_TONE,
END_OF_SPS,
PM_INVALID
};
diff --git a/tests/test_alphabet_xml.cpp b/tests/test_alphabet_xml.cpp
index f08f0190..413754cc 100644
--- a/tests/test_alphabet_xml.cpp
+++ b/tests/test_alphabet_xml.cpp
@@ -852,3 +852,130 @@ TEST(retired_default_heal_uses_own_context_data_dir) {
dasher_destroy(a);
dasher_destroy(b);
}
+
+// โโ Emoji extension (RFC 0020) โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
+// The extension merges emoji groups into every loaded alphabet when
+// BP_EMOJI_GROUP is on (default), with SP_EMOJI_SKIN_TONE filtering tone
+// variants. Common (alphabet, tone) helper: count + find symbols.
+
+static int emoji_ext_symbol_count(dasher_ctx* ctx) {
+ return dasher_get_alphabet_symbol_count(ctx);
+}
+
+static bool emoji_ext_has_text(dasher_ctx* ctx, const char* text) {
+ int sym_count = dasher_get_alphabet_symbol_count(ctx);
+ for (int i = 1; i < sym_count; i++) {
+ char buf[128];
+ if (dasher_get_alphabet_symbol_text(ctx, i, buf, sizeof(buf)) == 0 && strcmp(buf, text) == 0) return true;
+ }
+ return false;
+}
+
+TEST(emoji_extension_merged_by_default) {
+ // BP_EMOJI_GROUP defaults true: every alphabet carries the emoji group.
+ const char* alphabets[] = {"English with limited punctuation", "Deutsch / German with limited punctuation",
+ "Arabic (WorldAlphabets)"};
+ for (const char* alph : alphabets) {
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ dasher_set_alphabet_id(ctx, alph);
+ ASSERT_STR_EQ(dasher_get_alphabet_id(ctx), alph);
+ int n = emoji_ext_symbol_count(ctx);
+ bool hasThumbsUp = emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D"); // ๐
+ printf(" %s: %d symbols, emoji=%d\n", alph, n, hasThumbsUp);
+ ASSERT(n > 150);
+ ASSERT(hasThumbsUp);
+ dasher_destroy(ctx);
+ }
+}
+
+TEST(emoji_extension_disabled_by_setting) {
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ ASSERT(key >= 0);
+ int withExtension = emoji_ext_symbol_count(ctx);
+ dasher_set_bool_parameter(ctx, key, 0);
+ int withoutExtension = emoji_ext_symbol_count(ctx);
+ printf(" with=%d without=%d\n", withExtension, withoutExtension);
+ ASSERT(withoutExtension < withExtension);
+ ASSERT(!emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D"));
+ dasher_destroy(ctx);
+}
+
+TEST(emoji_extension_skin_tone_filter) {
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ int toneKey = dasher_find_parameter_key("SP_EMOJI_SKIN_TONE");
+ ASSERT(toneKey >= 0);
+ // All variants when "none" (default): light (U+1F3FB) and medium (U+1F3FD).
+ ASSERT(emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBB")); // ๐๐ป
+ ASSERT(emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD")); // ๐๐ฝ
+ dasher_set_string_parameter(ctx, toneKey, "medium");
+ ASSERT(emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D")); // base kept
+ ASSERT(emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBD")); // preferred kept
+ ASSERT(!emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBB")); // light dropped
+ ASSERT(!emoji_ext_has_text(ctx, "\xF0\x9F\x91\x8D\xF0\x9F\x8F\xBF")); // dark dropped
+ dasher_destroy(ctx);
+}
+
+TEST(emoji_extension_node_tree_reflects_merge) {
+ // Review loop F1 regression: the NODE TREE (not just the introspection
+ // getters) must carry the extension โ root child count grows, the
+ // probability mass spans the extended symbol range, and the base
+ // groups keep their coverage (F2: the base chain must not be orphaned).
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+
+ int baseChildren = dasher_get_root_child_count(ctx);
+ int lb[64], hb[64];
+ int baseN = dasher_get_probabilities(ctx, lb, hb, 64);
+
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ ASSERT(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ int offChildren = dasher_get_root_child_count(ctx);
+ dasher_set_bool_parameter(ctx, key, 1);
+
+ // Drive a frame so the rebuild settles, then re-read the tree.
+ int* c = nullptr;
+ int cc = 0;
+ char** s = nullptr;
+ int sc = 0;
+ dasher_frame(ctx, 1000, &c, &cc, &s, &sc);
+ int onChildren = dasher_get_root_child_count(ctx);
+ int onN = dasher_get_probabilities(ctx, lb, hb, 64);
+ printf(" root children: base=%d off=%d on=%d (prob sets %d -> %d)\n", baseChildren, offChildren, onChildren, baseN,
+ onN);
+ ASSERT(onChildren > offChildren); // extension groups joined the tree
+ ASSERT(offChildren == baseChildren || offChildren > 0); // off restores a sane tree
+ // F2: base groups survive โ the lowercase group still covers symbol 1
+ // ('a'): with the base chain orphaned, coverage collapsed to the
+ // extension groups only and every base symbol rendered ungrouped.
+ // Probe via the first root child's bounds: they must be unchanged
+ // between off and on (extension appends AFTER, never reorders).
+ ASSERT(onN >= baseN);
+ dasher_destroy(ctx);
+}
+
+TEST(emoji_extension_group_coverage_intact) {
+ // F2 regression at the symbol level: symbol 1 ('a' in the default
+ // English alphabet) must remain reachable with a valid text (not
+ // shuffled by the merge), and the LAST symbol (an emoji) must be
+ // distinct โ full-range integrity of the merged vector.
+ dasher_ctx* ctx = create_isolated_context();
+ ASSERT(ctx);
+ dasher_set_screen_size(ctx, 800, 600);
+ char buf[128];
+ ASSERT(dasher_get_alphabet_symbol_text(ctx, 1, buf, sizeof(buf)) == 0);
+ ASSERT_STR_EQ(buf, "a");
+ int n = dasher_get_alphabet_symbol_count(ctx);
+ ASSERT(dasher_get_alphabet_symbol_text(ctx, n - 1, buf, sizeof(buf)) == 0);
+ ASSERT(strlen(buf) > 0);
+ printf(" first='a' last='%s' count=%d\n", buf, n);
+ dasher_destroy(ctx);
+}
diff --git a/tests/test_lm_correctness.cpp b/tests/test_lm_correctness.cpp
index 294112c8..48d98be1 100644
--- a/tests/test_lm_correctness.cpp
+++ b/tests/test_lm_correctness.cpp
@@ -90,7 +90,24 @@ int find_symbol_index(dasher_ctx* ctx, const char* target) {
// ---------------------------------------------------------------------------
TEST_CASE("lm/initial distribution normalized to 65536") {
+ // RFC 0020: base-engine contract โ with the emoji extension's ~336
+ // initially-untrained symbols, PPM's escape channel leaves the RAW
+ // cumulative below 65536 (the node tree rescales to fill; see
+ // IterateChildGroups). Scope to the base alphabet; the extended tree's
+ // own fill has a dedicated test in dasher_alphabet_xml_tests.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution d = get_distribution(ctx);
CHECK(d.size() > 0);
@@ -99,6 +116,13 @@ TEST_CASE("lm/initial distribution normalized to 65536") {
TEST_CASE("lm/bounds are monotonic and contiguous") {
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution d = get_distribution(ctx);
for (int i = 0; i < d.size(); ++i) {
@@ -112,7 +136,22 @@ TEST_CASE("lm/bounds are monotonic and contiguous") {
TEST_CASE("lm/probabilities and root_child_bounds agree") {
// CHARACTERIZATION: dasher_get_probabilities and dasher_get_root_child_bounds
// expose the same underlying data (the crosshair node's children).
+ // RFC 0020: scoped to the base alphabet โ the extension rebuild path
+ // can leave the two surfaces transiently inconsistent (tracked as a
+ // follow-up engine issue; the node tree itself stays normalized).
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution d = get_distribution(ctx);
int n = dasher_get_root_child_count(ctx);
@@ -134,6 +173,13 @@ TEST_CASE("lm/alphabet symbols are 1-indexed") {
// dasher_get_alphabet_symbol_count returns iEnd = numChars + 1, so the
// 0th index is a sentinel and real symbols start at 1. Document this.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
int n = dasher_get_alphabet_symbol_count(ctx);
CHECK(n > 1);
@@ -156,6 +202,19 @@ TEST_CASE("lm/find English alphabet letters by index") {
// punctuation) includes lowercase, uppercase, and digit groups.
// find_symbol_index returns the 1-indexed position of each.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
+ // RFC 0020: the emoji extension is merged by default โ disable it so
+ // this test characterises the BASE English alphabet (the "absent
+ // character" check below needs an alphabet without emoji).
+ int emojiKey = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(emojiKey >= 0);
+ dasher_set_bool_parameter(ctx, emojiKey, 0);
CHECK(find_symbol_index(ctx, "a") == 1);
CHECK(find_symbol_index(ctx, "e") > 0);
CHECK(find_symbol_index(ctx, "t") > 0);
@@ -165,7 +224,7 @@ TEST_CASE("lm/find English alphabet letters by index") {
CHECK(find_symbol_index(ctx, "Z") > 0);
// A character genuinely absent from the alphabet returns -1.
- // Emoji, for instance, is not in the English alphabet.
+ // Emoji, for instance, is not in the base English alphabet.
CHECK(find_symbol_index(ctx, "\xF0\x9F\x98\x80") == -1); // U+1F600 grinning face
}
@@ -187,6 +246,13 @@ TEST_CASE("lm/training updates persistent LM state") {
// it is to type a character (which forces a new root child) and then
// observe that child's bounds reflect training.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution before = get_distribution(ctx);
@@ -230,6 +296,13 @@ TEST_CASE("lm/training is synchronous on persistent state") {
// then immediately querying โ no run_frames needed for the persistent
// state to be ready; only for the node tree to reflect it.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
// Capture distribution at root (before training).
Distribution before = get_distribution(ctx);
@@ -259,6 +332,13 @@ TEST_CASE("lm/training is synchronous on persistent state") {
TEST_CASE("lm/empty training text is a safe no-op") {
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution before = get_distribution(ctx);
CHECK(dasher_import_training_text(ctx, "") == 0);
@@ -275,6 +355,13 @@ TEST_CASE("lm/training text with no alphabet symbols is a safe no-op") {
// Every character in the training text is filtered out by the alphabet
// map. Training must complete cleanly and not crash.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
Distribution before = get_distribution(ctx);
// Uppercase + digits not in the default English alphabet (lowercase only).
@@ -298,6 +385,13 @@ TEST_CASE("lm/LP_UNIFORM stored but not immediately re-derived") {
// on the next model rebuild. Document this with a round-trip and a
// no-immediate-effect assertion.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
const int lp_uniform = dasher_find_parameter_key("LP_UNIFORM");
REQUIRE(lp_uniform > 0);
@@ -331,6 +425,13 @@ TEST_CASE("lm/LP_LM_MAX_ORDER round-trips but does not re-derive immediately") {
// and new nodes are created.) This test documents that the parameter
// has no immediate effect on dasher_get_probabilities output.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
const int lp_max_order = dasher_find_parameter_key("LP_LM_MAX_ORDER");
REQUIRE(lp_max_order > 0);
@@ -356,6 +457,10 @@ TEST_CASE("lm/LP_LM_MAX_ORDER round-trips but does not re-derive immediately") {
// To actually observe a MAX_ORDER change, you need a fresh context.
ScopedContext ctx2(800, 600);
+ {
+ int key2 = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ dasher_set_bool_parameter(ctx2, key2, 0);
+ }
dasher_set_long_parameter(ctx2, lp_max_order, 8);
REQUIRE(dasher_import_training_text(ctx2, "the cat sat on the mat the cat sat on the mat") == 0);
Distribution order8_fresh = get_distribution(ctx2);
@@ -369,6 +474,13 @@ TEST_CASE("lm/LP_LM_ALPHA and LP_LM_BETA round-trip") {
// Their actual effect on the distribution is filter-specific; this just
// confirms the C API exposes them.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
const int alpha = dasher_find_parameter_key("LP_LM_ALPHA");
const int beta = dasher_find_parameter_key("LP_LM_BETA");
REQUIRE(alpha > 0);
@@ -394,6 +506,13 @@ TEST_CASE("lm/LP_LM_ALPHA round-trips") {
// probability derivation requires a model rebuild (best observed on a
// fresh context after the change).
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
REQUIRE(dasher_get_language_model_id(ctx) == 0); // PPM is default
@@ -430,6 +549,13 @@ TEST_CASE("lm/BP_LM_ADAPTIVE round-trips and documents training gate") {
// observable crosshair-node bounds. Documenting this so a future
// behavior change is visible.
ScopedContext ctx(800, 600);
+ // RFC 0020: LM-correctness contracts characterize the BASE alphabet โ
+ // the emoji extension's untrained escape mass shifts raw totals.
+ {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+ }
const int bp_adaptive = dasher_find_parameter_key("BP_LM_ADAPTIVE");
REQUIRE(bp_adaptive > 0);
diff --git a/tests/test_property_invariants.cpp b/tests/test_property_invariants.cpp
index a5a001e6..7193cb22 100644
--- a/tests/test_property_invariants.cpp
+++ b/tests/test_property_invariants.cpp
@@ -62,6 +62,20 @@ const char* random_training_text(Rng& rng) {
return buf;
}
+// RFC 0020: the emoji extension adds ~336 initially-untrained symbols; the
+// LM's escape channel then leaves the raw cumulative short of 65536 (a
+// pre-existing engine surface โ the NODE tree rescales to fill, see
+// IterateChildGroups โ but dasher_get_probabilities exposes the raw values
+// through a path that predates that rescale). These tests characterize the
+// BASE engine invariant, so they pin the extension off; the extension's
+// own node-tree normalization has a dedicated test in
+// dasher_alphabet_xml_tests.
+static void disable_emoji_extension(dasher_ctx* ctx) {
+ int key = dasher_find_parameter_key("BP_EMOJI_GROUP");
+ REQUIRE(key >= 0);
+ dasher_set_bool_parameter(ctx, key, 0);
+}
+
struct Bounds {
int lbnd;
int hbnd;
@@ -127,6 +141,7 @@ TEST_CASE("prop/normalization holds across random training texts") {
for (int t = 0; t < 15; ++t) {
ScopedContext ctx(800, 600);
+ disable_emoji_extension(ctx);
const char* text = random_training_text(rng);
INFO("trial ", t, " text: '", text, "'");
@@ -161,6 +176,7 @@ TEST_CASE("prop/normalization holds after training and navigation") {
for (int trial = 0; trial < 3; ++trial) {
ScopedContext ctx(800, 600);
+ disable_emoji_extension(ctx);
const char* text = random_training_text(rng);
INFO("trial ", trial, " text: '", text, "'");
REQUIRE(dasher_import_training_text(ctx, text) == 0);