Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -395,6 +395,18 @@ dasher_add_test(dasher_text_metrics_tests test_text_metrics.cpp)
dasher_add_test(dasher_benchmark_tests test_benchmarks.cpp)
dasher_add_test(dasher_property_invariant_tests test_property_invariants.cpp)

# Longest-match symbol lookup (RFC 0020) — internal CAlphabetMap unit
# tests; header-only usage against the DasherCore static library.
add_executable(dasher_alphabet_map_longest_tests
${CMAKE_CURRENT_LIST_DIR}/tests/test_alphabet_map_longest.cpp)
target_include_directories(dasher_alphabet_map_longest_tests PRIVATE
${CMAKE_CURRENT_LIST_DIR}/src/
${CMAKE_CURRENT_LIST_DIR}/tests/
${DOCTEST_INCLUDE_DIR})
target_link_libraries(dasher_alphabet_map_longest_tests PRIVATE DasherCore)
add_test(NAME dasher_alphabet_map_longest_tests COMMAND dasher_alphabet_map_longest_tests)
set_tests_properties(dasher_alphabet_map_longest_tests PROPERTIES TIMEOUT ${DASHER_TEST_TIMEOUT})

# Control action system tests — needs internal DasherCore classes (ActionRegistry)
# AND C API functions, so we compile CAPI.cpp directly and link DasherCore
add_executable(dasher_control_action_tests
Expand Down
149 changes: 110 additions & 39 deletions src/DasherCore/Alphabet/AlphabetMap.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -79,20 +79,23 @@ void CAlphabetMap::SymbolStream::readMore() {
}
}

inline void CAlphabetMap::SymbolStream::ensureLookahead(size_t want) {
if (pos + want > len) {
if (pos) {
// shift remaining bytes to beginning
len -= pos; // len of them
memmove(buf, &buf[pos], len);
bytesRead(pos);
pos = 0;
}
// and look for more
readMore();
}
}

inline int CAlphabetMap::SymbolStream::findNext() {
for (;;) {
if (pos + m_utf8_count_array.max_length > len) {
// may need more bytes for next char
if (pos) {
// shift remaining bytes to beginning
len -= pos; // len of them
memmove(buf, &buf[pos], len);
bytesRead(pos);
pos = 0;
}
// and look for more
readMore();
}
ensureLookahead(m_utf8_count_array.max_length);
// if still don't have any chars after attempting to read more...EOF!
if (pos == len) {
if (m_skippedInvalid && m_pMsgs)
Expand Down Expand Up @@ -120,60 +123,96 @@ inline int CAlphabetMap::SymbolStream::findNext() {
}
}

std::string CAlphabetMap::SymbolStream::peekAhead() {
std::string CAlphabetMap::SymbolStream::peekAhead(const CAlphabetMap* map) {
int numChars = findNext();
if (numChars == 0) return "";

// RFC 0020 / greptile P1: the peek must agree with what the next
// next(map) call will consume. Annotation readers (Routing/Mandarin
// conversion trainers, CTrainer::readEscape) record the peeked token
// and then advance via next() — peeking only the FIRST codepoint of a
// multi-codepoint key would record a token that never matches the
// route/pronunciation table while next() skips the whole key.
if (map && map->MaxKeyLen() > 0) {
ensureLookahead(map->MaxKeyLen());
size_t matched = 0;
if (map->LongestMatch(&buf[pos], len - pos, matched) != UNKNOWN_SYMBOL) return std::string(&buf[pos], matched);
Comment thread
willwade marked this conversation as resolved.
}
return std::string(&buf[pos], numChars);
}

std::string CAlphabetMap::SymbolStream::peekBack() {
bool bSeenHighBit = false;
for (int i = pos - 1; i >= 0; i--) {
if (buf[i] & 0x80) {
// multibyte character...
bSeenHighBit = true;
if (buf[i] & 0x40) {
// START of multibyte character
int numChars = m_utf8_count_array[buf[i]];
if (i + numChars > pos) {
// last (attempt to read a) symbol was an incomplete UTF8 character (!).
// We'll have reported an error already when we saw it the first time, so for now just:
return "";
}
DASHER_ASSERT(i + numChars == pos);
return std::string(&buf[i], numChars);
}
// in middle of multibyte, keep going back...
} else {
// high bit not set -> single-byte char
if (bSeenHighBit)
return ""; // followed by a "continuation of multibyte char" without a "first byte of multibyte char"
// before it. (Malformed!)
return std::string(&buf[i], 1);
// RFC 0020: the previous symbol may have been a longest-match key
// spanning multiple codepoints (or "\r\n"), which a backward buffer
// walk could never reconstruct — it would return only the final
// codepoint. next() records exactly what it consumed; replay that.
// (The read window may have shifted between the calls, so the copy —
// not a buffer slice — is the only safe source. Callers that respect
// the documented precondition — no peekAhead() since the last next()
// — see the symbol text as consumed; "" before the first next().)
return m_lastConsumed;
}

symbol CAlphabetMap::SymbolStream::next(const CAlphabetMap* map) {
int numChars = findNext();
if (numChars == 0) return -1; // EOF

// RFC 0020 clause 4 — longest match first: multi-codepoint symbols
// (digraph outputs, ZWJ sequences, skin-tone modifiers) must train as
// one symbol. Probe before the single-character path so a multi-
// codepoint key that STARTS with a known single character (❤️ over ❤)
// still wins.
if (map->MaxKeyLen() > 0) {
ensureLookahead(map->MaxKeyLen());
size_t matched = 0;
symbol sym = map->LongestMatch(&buf[pos], len - pos, matched);
Comment thread
greptile-apps[bot] marked this conversation as resolved.
if (sym != UNKNOWN_SYMBOL) {
m_lastConsumed.assign(&buf[pos], matched);
pos += matched;
return sym;
Comment thread
greptile-apps[bot] marked this conversation as resolved.
}
}
// fail...relatively gracefully ;-)
return "";
return nextCharLocked(map, numChars);
}

symbol CAlphabetMap::SymbolStream::next(const CAlphabetMap* map) {
symbol CAlphabetMap::SymbolStream::nextRaw(const CAlphabetMap* map) {
// Structural parsing (annotations, escape delimiters): one codepoint
// per call, exactly the pre-RFC behaviour — a multi-codepoint key
// sharing a prefix with a grammar delimiter must not shadow it.
int numChars = findNext();
if (numChars == 0) return -1; // EOF
return nextCharLocked(map, numChars);
}

// Shared single-codepoint consumption tail (paragraph special case, then
// direct/hash lookup). pos and m_lastConsumed advance by exactly one
// codepoint ('\r\n' paragraph: two bytes).
symbol CAlphabetMap::SymbolStream::nextCharLocked(const CAlphabetMap* map, int numChars) {
if (numChars == 1) {
if (map->m_ParagraphSymbol != UNKNOWN_SYMBOL && buf[pos] == '\r') {
DASHER_ASSERT(pos + 1 < len || len < 1024); // there are more characters (we should have read
// utf8...max_length), or else input is exhausted
if (pos + 1 < len && buf[pos + 1] == '\n') {
m_lastConsumed.assign("\r\n");
pos += 2;
return map->m_ParagraphSymbol;
}
}
m_lastConsumed.assign(1, buf[pos]);
return map->GetSingleChar(buf[pos++]);
}
int sym = map->Get(std::string(&buf[pos], numChars));
m_lastConsumed.assign(&buf[pos], numChars);
pos += numChars;
return sym;
}

std::string CAlphabetMap::SymbolStream::peekAheadRaw() {
int numChars = findNext();
if (numChars == 0) return "";
return std::string(&buf[pos], numChars);
}

void CAlphabetMap::GetSymbols(std::vector<symbol>& Symbols, const std::string& Input) const {
std::istringstream in(Input);
SymbolStream syms(in);
Expand Down Expand Up @@ -247,6 +286,38 @@ void CAlphabetMap::Add(const std::string& Key, symbol Value) {

Entries.push_back(Entry(Key, Value, HashEntry));
HashEntry = &Entries.back();

// RFC 0020 clause 4: register multi-codepoint keys for longest-match
// probing. A key qualifies when it is longer than its lead byte's
// UTF-8 length — i.e. more than one codepoint, unreachable by the
// single-character path in next(). (Single-codepoint multi-byte keys
// like a 4-byte 😀 already match there.) Copies, not pointers: the
// Entries vector reallocates as it grows. Keep sorted longest-first.
// Keys of STREAM_WINDOW bytes or more can never be fully buffered for
// probing (greptile P2) — leave them unregistered rather than
// pretending they train; such keys were equally dead through the
// per-codepoint path, so no behaviour regresses.
if (Key.length() > static_cast<size_t>(m_utf8_count_array[static_cast<unsigned char>(Key[0])]) &&
Key.length() < STREAM_WINDOW) {
auto it = m_vMultiCharKeys.begin();
while (it != m_vMultiCharKeys.end() && it->first.length() >= Key.length())
++it;
m_vMultiCharKeys.insert(it, {Key, Value});
m_iMaxKeyLen = std::max(m_iMaxKeyLen, Key.length());
}
}

symbol CAlphabetMap::LongestMatch(const char* at, size_t avail, size_t& matchedLen) const {
// m_vMultiCharKeys is sorted longest-first, so the first byte-exact
// match is the longest possible.
for (const auto& [key, sym] : m_vMultiCharKeys) {
if (key.length() > avail) continue;
if (std::memcmp(at, key.data(), key.length()) == 0) {
matchedLen = key.length();
return sym;
}
}
return UNKNOWN_SYMBOL;
}

symbol CAlphabetMap::Get(const std::string& Key) const {
Expand Down
71 changes: 62 additions & 9 deletions src/DasherCore/Alphabet/AlphabetMap.h
Original file line number Diff line number Diff line change
Expand Up @@ -34,12 +34,14 @@ class CAlphabetMap;
/// Ian clearly had reservations about this system, as follows; and I'd add
/// that much of the fun comes from supporting single unicode characters
/// which are multiple octets, as we use std::string (which works in octets)
/// for everything...note that we do *not* support multi-unicode-character
/// symbols (such as the "asdf" suggested below) except in the case of "\r\n"
/// for the paragraph symbol.
/// for everything. Since RFC 0020 the map also supports MULTI-unicode-
/// character symbols (digraph outputs, ZWJ emoji sequences, VS16 skin
/// tones) via longest-match probing in SymbolStream::next — see
/// LongestMatch(). Keys longer than the stream's 1024-byte window can
/// never match and are silently unregistered.
///
/// Note that in 2010 we did indeed tailor this to the alphabet more closely,
/// fast-casing single-octet characters to avoid using a hash etc. - this makes
/// fast-casing single-octet characters to avoid using a hash etc. - which makes
/// many common alphabets substantially faster!
///
/// Anyway, Ian writes:
Expand Down Expand Up @@ -77,10 +79,27 @@ class Dasher::CAlphabetMap {
public:
~CAlphabetMap();

/// Read-window size of SymbolStream: multi-codepoint keys of this
/// length or longer can never be fully buffered for probing and are
/// not registered (RFC 0020 — documented at Add).
static constexpr size_t STREAM_WINDOW = 1024;

// Return the symbol associated with Key or Undefined.
symbol Get(const std::string& Key) const;
symbol GetSingleChar(char key) const;

/// Longest-match support (RFC 0020 clause 4): probe multi-codepoint
/// keys (digraph outputs, ZWJ/VS16 emoji) at a buffer position, longest
/// first, before the single-character path in SymbolStream::next().
/// \param at buffer position; \param avail bytes readable from it
/// \param matchedLen set to the matched key's byte length on success
/// \return the symbol, or UNKNOWN_SYMBOL (0) when no key matches.
symbol LongestMatch(const char* at, size_t avail, size_t& matchedLen) const;

/// Longest multi-codepoint key in the map (0 when none) — the
/// lookahead SymbolStream must keep buffered.
size_t MaxKeyLen() const { return m_iMaxKeyLen; }

class SymbolStream {
public:
virtual ~SymbolStream() = default;
Expand All @@ -91,18 +110,39 @@ class Dasher::CAlphabetMap {
/// \return 0 for unknown symbol (not in map); -1 for EOF; else symbol#.
symbol next(const CAlphabetMap* map);

/// RFC 0020 / greptile P1: raw (single-codepoint) variants for
/// STRUCTURAL parsing — conversion annotations (<route>, pinyin) and
/// context-escape delimiters are grammar, not symbol content: a
/// multi-codepoint key sharing a prefix with a delimiter must never
/// shadow it. Longest-match applies to symbol training only.
symbol nextRaw(const CAlphabetMap* map);
/// Single-codepoint peek, ignoring longest-match (see nextRaw).
std::string peekAheadRaw();

/// Shared single-codepoint consumption tail of next/nextRaw.
inline symbol nextCharLocked(const CAlphabetMap* map, int numChars);

/// Finds the next complete character in the stream, but does not advance past it.
/// Hence, repeated calls will return the same string. (Always constructs a string,
/// which next() avoids for single-octet chars, so may be slower)
std::string peekAhead();
/// RFC 0020: when a multi-codepoint key starts at the current position,
/// returns the WHOLE key — exactly the bytes the next next() call would
/// consume — so annotation readers (Routing/Mandarin escape and route
/// parsing) never record a different token than they advance past.
std::string peekAhead(const CAlphabetMap* map);

/// Returns the string representation of the previous symbol (i.e. that returned
/// by the previous call to next()). Undefined if next() has not been called, or
/// if peekAhead() has been called since the last call to next(). Does not change
/// the stream position. (Always constructs a string, which next() avoids for
/// single-octet chars, so may be slower.)
/// the stream position. Returns the full multi-codepoint key when the previous
/// symbol matched one (longest-match, RFC 0020) — not just its final codepoint.
std::string peekBack();

/// Bytes consumed by the last next() call — the source of
/// peekBack's answer, kept as a copy because the read window can
/// shift (ensureLookahead) between the two calls.
std::string m_lastConsumed;

protected:
/// Called periodically to indicate some number of bytes have been read.
/// Default implementation does nothing; subclasses may override for e.g. logging.
Expand All @@ -116,8 +156,13 @@ class Dasher::CAlphabetMap {
/// \return the number of octets representing the next character, or 0 for EOF
/// (inc. where the file ends with an incomplete character)
inline int findNext();

/// Ensure at least `want` bytes are buffered past pos (shifting the
/// remaining window to the front and reading more; at EOF the buffer
/// simply holds what's left). findNext's refill logic, parameterised.
inline void ensureLookahead(size_t want);
void readMore();
char buf[1024];
char buf[STREAM_WINDOW];
off_t pos, len;
std::istream& in;
CMessageDisplay* const m_pMsgs;
Expand Down Expand Up @@ -175,7 +220,15 @@ class Dasher::CAlphabetMap {
std::vector<Entry*> HashTable;
symbol* m_pSingleChars;
/// both "\r\n" and "\n" are mapped to this (if not Undefined).
/// This is the only case where >1 character can map to a symbol.
/// (Historically the only multi-character mapping; multi-codepoint
/// keys via Add() now exist too — see LongestMatch.)
symbol m_ParagraphSymbol;

/// Multi-codepoint keys (copies, with their symbols), sorted longest
/// first — copies because Entries vector growth relocates its strings.
/// Only keys that the single-character path can never match (more than
/// one codepoint) belong here.
std::vector<std::pair<std::string, symbol>> m_vMultiCharKeys;
size_t m_iMaxKeyLen = 0;
};
/// \}
4 changes: 2 additions & 2 deletions src/DasherCore/MandarinAlphMgr.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -214,8 +214,8 @@ void CMandarinAlphMgr::CMandarinTrainer::Train(CAlphabetMap::SymbolStream& syms)
strPy.c_str());
strPy.clear();
bHavePy = true;
for (std::string s; (s = syms.peekAhead()).length(); strPy += s) {
syms.next(m_pAlphabet);
for (std::string s; (s = syms.peekAheadRaw()).length(); strPy += s) {
syms.nextRaw(m_pAlphabet); // structural: annotation text, no longest-match
if (s == m_pInfo->m_strConversionTrainStop) break;
}
continue; // read next, hopefully a CH (!)
Expand Down
4 changes: 2 additions & 2 deletions src/DasherCore/RoutingAlphMgr.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -145,8 +145,8 @@ void CRoutingAlphMgr::CRoutingTrainer::Train(CAlphabetMap::SymbolStream& syms) {
strRoute.c_str());
strRoute.clear();
bHaveRoute = true;
for (std::string s; (s = syms.peekAhead()).length(); strRoute += s) {
syms.next(m_pAlphabet);
for (std::string s; (s = syms.peekAheadRaw()).length(); strRoute += s) {
syms.nextRaw(m_pAlphabet); // structural: annotation text, no longest-match
if (s == m_pInfo->m_strConversionTrainStop) break;
}
continue; // read next, hopefully a CH (!)
Expand Down
6 changes: 3 additions & 3 deletions src/DasherCore/Trainer.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -47,8 +47,8 @@ bool CTrainer::readEscape(CLanguageModel::Context& sContext, symbol sym, CAlphab

// Yes, found escape character....

std::string delim = syms.peekAhead();
syms.next(m_pAlphabet); // peekAhead doesn't read
std::string delim = syms.peekAheadRaw();
syms.nextRaw(m_pAlphabet); // structural: escape delimiters are grammar, not symbols

// A double escape character means an actual occurrence of the character is wanted...
if (delim == m_pInfo->GetContextEscapeChar()) {
Expand All @@ -63,7 +63,7 @@ bool CTrainer::readEscape(CLanguageModel::Context& sContext, symbol sym, CAlphab
for (std::vector<symbol>::iterator it = defCtx.begin(); it != defCtx.end(); it++)
m_pLanguageModel->EnterSymbol(sContext, *it);
// and read the first delimiter; everything until the second occurrence of this, is _context_ only.
for (symbol s; (s = syms.next(m_pAlphabet)) != -1;) {
for (symbol s; (s = syms.nextRaw(m_pAlphabet)) != -1;) {
Comment thread
willwade marked this conversation as resolved.
if (syms.peekBack() == delim) break;
m_pLanguageModel->EnterSymbol(sContext, s);
}
Expand Down
Loading
Loading