diff --git a/dChatFilter/ChatFilterCore.h b/dChatFilter/ChatFilterCore.h new file mode 100644 index 000000000..b6acbb609 --- /dev/null +++ b/dChatFilter/ChatFilterCore.h @@ -0,0 +1,342 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/** + * The chat filter's words, without the server around them (pure, unit tested; dChatFilter and the dashboard both use it). + * + * A word is compared lower case (ASCII only, so every platform agrees) without ! ? ; . , and a phrase is its words + * joined by one space. Entries are stored and compared by ChatFilterWords::Hash: 64-bit FNV-1a over the entry's bytes, + * the same on every compiler, standard library and platform. + */ +namespace ChatFilterWords { + // ASCII lower case; other bytes (UTF-8) stay as they are + inline std::string AsciiLower(std::string text) { + for (auto& c : text) if (c >= 'A' && c <= 'Z') c = static_cast(c - 'A' + 'a'); + return text; + } + + // A word as the filter compares it: lower case, without ! ? ; . , + inline std::string NormalizeWord(std::string word) { + std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; }); + return AsciiLower(std::move(word)); + } + + // A word or phrase as the filter stores it: each word normalized, words that end up empty dropped, joined by one space + inline std::string NormalizeEntry(std::string_view text) { + std::string entry; + size_t start = 0; + while (start < text.size()) { + auto end = text.find_first_of(" \t\r\n", start); + if (end == std::string_view::npos) end = text.size(); + const auto word = NormalizeWord(std::string(text.substr(start, end - start))); + if (!word.empty()) { + if (!entry.empty()) entry += ' '; + entry += word; + } + start = end + 1; + } + return entry; + } + + // How many words an entry has (1 for a word, 0 for an empty entry) + inline uint32_t WordCount(std::string_view entry) { + return entry.empty() ? 0 : static_cast(std::count(entry.begin(), entry.end(), ' ')) + 1; + } + + // 64-bit FNV-1a: offset basis 0xcbf29ce484222325, prime 0x100000001b3, one byte at a time + constexpr uint64_t Hash(std::string_view entry) { + uint64_t hash = 0xcbf29ce484222325ULL; + for (const char c : entry) { + hash ^= static_cast(c); + hash *= 0x100000001b3ULL; + } + return hash; + } + + // One piece of a message between spaces: where it is in the message and the word the filter compares + struct Token { + uint32_t position{}; + uint32_t length{}; + std::string word; + }; + + // A message split at each space, the way the filter checks it: two spaces in a row give an empty piece, a trailing space none + inline std::vector Tokenize(std::string_view message) { + std::vector tokens; + size_t start = 0; + while (start < message.size()) { + auto end = message.find(' ', start); + if (end == std::string_view::npos) end = message.size(); + tokens.push_back({ static_cast(start), static_cast(end - start), NormalizeWord(std::string(message.substr(start, end - start))) }); + start = end + 1; + } + return tokens; + } + + // A run of tokens [first, last] that matched a blocked entry, and the longest entry that matched where the run starts + struct Match { + size_t first{}; + size_t last{}; + std::string entry; + }; + + /** + * The blocked words and phrases in a message: at each word, the longest run of up to maxWords consecutive words + * (tokens with no word are skipped) whose entry isBlocked accepts. Runs that share a word are merged into one. + */ + inline std::vector FindBlocked(const std::vector& tokens, uint32_t maxWords, const std::function& isBlocked) { + std::vector words; + for (size_t i = 0; i < tokens.size(); i++) if (!tokens[i].word.empty()) words.push_back(i); + + std::vector matches; + for (size_t i = 0; i < words.size(); i++) { + std::string entry; + std::string best; + size_t bestLength = 0; + for (size_t n = 1; n <= maxWords && i + n <= words.size(); n++) { + if (n > 1) entry += ' '; + entry += tokens[words[i + n - 1]].word; + if (isBlocked(entry)) { + best = entry; + bestLength = n; + } + } + if (bestLength == 0) continue; + const size_t first = words[i]; + const size_t last = words[i + bestLength - 1]; + if (!matches.empty() && first <= matches.back().last) { + matches.back().last = std::max(matches.back().last, last); + } else { + matches.push_back({ first, last, std::move(best) }); + } + } + return matches; + } + + // Hashes of words or phrases, and the most words any of them has + struct WordList { + std::unordered_set hashes; + uint32_t maxWords{}; + + void AddEntry(std::string_view entry) { + if (entry.empty()) return; + hashes.insert(Hash(entry)); + maxWords = std::max(maxWords, WordCount(entry)); + } + bool Contains(std::string_view entry) const { return hashes.contains(Hash(entry)); } + bool Empty() const { return hashes.empty(); } + size_t Size() const { return hashes.size(); } + }; + + // Everything the filter checks a message against + struct Lists { + WordList approved; // chatplus_en_us.txt and approved character names: whitelist chat, one word at a time + WordList denied; // blocklist.dcf: best friends' free chat + WordList customAllowed; // allowed on the dashboard + WordList customBlocked; // blocked on the dashboard: stopped in every kind of chat + }; + + /** + * The pieces of a message the filter stops, as (position, length) in the message. Blocked words and phrases (the + * dashboard's always, blocklist.dcf's in free chat) are stopped as one span each. In whitelist chat (allowList) every + * other piece must be an allowed word, one at a time, as the client checks words. In free chat without a block list + * the whole message is stopped. + */ + inline std::set> CheckMessage(std::string_view message, bool allowList, const Lists& lists) { + if (message.empty()) return {}; + if (!allowList && lists.denied.Empty()) return { { 0, static_cast(message.length()) } }; + + const auto tokens = Tokenize(message); + const uint32_t maxWords = std::max(lists.customBlocked.maxWords, allowList ? 0u : lists.denied.maxWords); + const auto matches = FindBlocked(tokens, maxWords, [&](const std::string& entry) { + return lists.customBlocked.Contains(entry) || (!allowList && lists.denied.Contains(entry)); + }); + + std::set> bad; + std::vector covered(tokens.size(), false); + for (const auto& match : matches) { + const auto& first = tokens[match.first]; + const auto& last = tokens[match.last]; + bad.emplace(static_cast(first.position), static_cast(last.position + last.length - first.position)); + for (size_t i = match.first; i <= match.last; i++) covered[i] = true; + } + + if (allowList) { + for (size_t i = 0; i < tokens.size(); i++) { + if (covered[i]) continue; + const auto hash = Hash(tokens[i].word); + if (!lists.approved.hashes.contains(hash) && !lists.customAllowed.hashes.contains(hash)) { + bad.emplace(static_cast(tokens[i].position), static_cast(tokens[i].length)); + } + } + } + return bad; + } +} + +/** + * The chat filter's word list files (.dcf). These are DLU's own files: the client reads no .dcf and hashes no chat + * words (it keeps its lists as plain text). Layout, little-endian: + * uint32 magic 'DCFB' | uint32 version (3) | uint32 most words in one entry | uint64 count | count x uint64 ChatFilterWords::Hash + * Version 2 (older DLU) stored std::hash values, which differ between compilers and platforms; those can't be read. + */ +namespace dChatFilterDCF { + constexpr uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24)); + constexpr uint32_t formatVersion = 3; + constexpr uint32_t oldFormatVersion = 2; + constexpr size_t headerSize = 4 + 4 + 4 + 8; + + // The block list's plain source (one word or phrase per line) and the .dcf built from it, next to the servers + constexpr const char* BLOCK_LIST_TEXT = "blocklist.txt"; + constexpr const char* BLOCK_LIST_FILE = "blocklist.dcf"; + + enum class eStatus : uint8_t { + OK, + MISSING, // no file + NOT_DCF, // not a .dcf file + OLD_FORMAT, // version 2: platform-dependent hashes, rebuild it from the plain word list + UNKNOWN, // a version this server doesn't know + TRUNCATED, // shorter than its count says + }; + + inline const char* StatusText(eStatus status) { + switch (status) { + case eStatus::OK: return "ok"; + case eStatus::MISSING: return "missing"; + case eStatus::NOT_DCF: return "not a .dcf file"; + case eStatus::OLD_FORMAT: return "old format (version 2, platform-dependent hashes)"; + case eStatus::UNKNOWN: return "unknown version"; + case eStatus::TRUNCATED: return "truncated"; + } + return "unknown"; + } + + struct ParseResult { + eStatus status{ eStatus::NOT_DCF }; + uint32_t version{}; + ChatFilterWords::WordList list; + }; + + namespace detail { + inline uint64_t ReadLE(std::string_view bytes, size_t offset, size_t size) { + uint64_t value = 0; + for (size_t i = 0; i < size; i++) value |= static_cast(static_cast(bytes[offset + i])) << (8 * i); + return value; + } + inline void WriteLE(std::string& out, uint64_t value, size_t size) { + for (size_t i = 0; i < size; i++) out.push_back(static_cast((value >> (8 * i)) & 0xFF)); + } + } + + inline ParseResult Parse(std::string_view bytes) { + ParseResult result; + if (bytes.size() < 8 || detail::ReadLE(bytes, 0, 4) != header) return result; + result.version = static_cast(detail::ReadLE(bytes, 4, 4)); + if (result.version == oldFormatVersion) { + result.status = eStatus::OLD_FORMAT; + return result; + } + if (result.version != formatVersion) { + result.status = eStatus::UNKNOWN; + return result; + } + result.status = eStatus::TRUNCATED; + if (bytes.size() < headerSize) return result; + const auto maxWords = static_cast(detail::ReadLE(bytes, 8, 4)); + const auto count = detail::ReadLE(bytes, 12, 8); + if (count > (bytes.size() - headerSize) / 8) return result; + result.list.maxWords = maxWords; + result.list.hashes.reserve(count); + for (uint64_t i = 0; i < count; i++) result.list.hashes.insert(detail::ReadLE(bytes, headerSize + i * 8, 8)); + result.status = eStatus::OK; + return result; + } + + // A list as a .dcf file; hashes sorted, so the same words always give the same bytes + inline std::string Serialize(const ChatFilterWords::WordList& list) { + std::vector hashes(list.hashes.begin(), list.hashes.end()); + std::sort(hashes.begin(), hashes.end()); + std::string out; + out.reserve(headerSize + hashes.size() * 8); + detail::WriteLE(out, header, 4); + detail::WriteLE(out, formatVersion, 4); + detail::WriteLE(out, list.maxWords, 4); + detail::WriteLE(out, hashes.size(), 8); + for (const auto hash : hashes) detail::WriteLE(out, hash, 8); + return out; + } + + // A plain block list: one word or phrase per line, normalized (ChatFilterWords::NormalizeEntry); empty lines skipped + inline ChatFilterWords::WordList BlockListFromText(std::string_view text) { + ChatFilterWords::WordList list; + size_t start = 0; + while (start < text.size()) { + auto end = text.find('\n', start); + if (end == std::string_view::npos) end = text.size(); + list.AddEntry(ChatFilterWords::NormalizeEntry(text.substr(start, end - start))); + start = end + 1; + } + return list; + } + + // A plain allow list (chatplus_en_us.txt): one word per line, lower case, compared whole (as the filter always has) + inline ChatFilterWords::WordList AllowListFromText(std::string_view text) { + ChatFilterWords::WordList list; + size_t start = 0; + while (start < text.size()) { + auto end = text.find('\n', start); + if (end == std::string_view::npos) end = text.size(); + std::string line(text.substr(start, end - start)); + std::erase(line, '\r'); + line = ChatFilterWords::AsciiLower(std::move(line)); + list.hashes.insert(ChatFilterWords::Hash(line)); + list.maxWords = std::max(list.maxWords, 1u); + start = end + 1; + } + return list; + } + + inline std::optional ReadBytes(const std::filesystem::path& path) { + std::ifstream in(path, std::ios::binary); + if (!in) return std::nullopt; + return std::string((std::istreambuf_iterator(in)), std::istreambuf_iterator()); + } + + inline ParseResult ReadFile(const std::filesystem::path& path) { + const auto bytes = ReadBytes(path); + if (!bytes) return { eStatus::MISSING }; + return Parse(*bytes); + } + + // Writes the file whole or not at all (a temporary file renamed over it), so servers starting together don't read half a file + inline bool WriteFile(const std::filesystem::path& path, const ChatFilterWords::WordList& list) { + auto temp = path; + temp += "." + std::to_string(std::random_device{}()) + ".tmp"; + { + std::ofstream out(temp, std::ios::binary | std::ios::trunc); + if (!out) return false; + const auto bytes = Serialize(list); + out.write(bytes.data(), static_cast(bytes.size())); + if (!out) return false; + } + std::error_code error; + std::filesystem::rename(temp, path, error); + if (!error) return true; + std::filesystem::remove(temp, error); + return false; + } +} diff --git a/dChatFilter/dChatFilter.cpp b/dChatFilter/dChatFilter.cpp index 83b837ece..fcbb4e92b 100644 --- a/dChatFilter/dChatFilter.cpp +++ b/dChatFilter/dChatFilter.cpp @@ -1,15 +1,8 @@ #include "dChatFilter.h" -#include "BinaryIO.h" -#include -#include -#include -#include -#include -#include -#include "dCommonVars.h" +#include + #include "Logger.h" -#include "dConfig.h" #include "Database.h" #include "Game.h" #include "eGameMasterLevel.h" @@ -19,154 +12,96 @@ using namespace dChatFilterDCF; dChatFilter::dChatFilter(const std::string& filepath, bool dontGenerateDCF) { m_DontGenerateDCF = dontGenerateDCF; - if (!BinaryIO::DoesFileExist(filepath + ".dcf") || m_DontGenerateDCF) { - ReadWordlistPlaintext(filepath + ".txt", true); - if (!m_DontGenerateDCF) ExportWordlistToDCF(filepath + ".dcf", true); - } else if (!ReadWordlistDCF(filepath + ".dcf", true)) { - ReadWordlistPlaintext(filepath + ".txt", true); - ExportWordlistToDCF(filepath + ".dcf", true); - } + LoadAllowList(filepath); + LoadBlockList(); - if (BinaryIO::DoesFileExist("blocklist.dcf")) { - ReadWordlistDCF("blocklist.dcf", false); - } - - //Read player names that are ok as well: - auto approvedNames = Database::Get()->GetApprovedCharacterNames(); - for (auto& name : approvedNames) { - std::transform(name.begin(), name.end(), name.begin(), ::tolower); //Transform to lowercase - m_ApprovedWords.push_back(CalculateHash(name)); + // Approved character names count as allowed words + for (const auto& name : Database::Get()->GetApprovedCharacterNames()) { + m_Lists.approved.hashes.insert(ChatFilterWords::Hash(ChatFilterWords::AsciiLower(name))); } ReloadCustomWords(); } -void dChatFilter::ReloadCustomWords() { - m_CustomAllowedWords.clear(); - m_CustomBlockedWords.clear(); - // Words remembered as not allowed may be allowed now - m_UserUnapprovedWordCache.clear(); - for (const auto& word : Database::Get()->GetChatFilterWords()) { - (word.allowed ? m_CustomAllowedWords : m_CustomBlockedWords).insert(CalculateHash(NormalizeWord(word.word))); - } -} - -dChatFilter::~dChatFilter() { - m_ApprovedWords.clear(); - m_DeniedWords.clear(); -} - -void dChatFilter::ReadWordlistPlaintext(const std::string& filepath, bool allowList) { - std::ifstream file(filepath); - if (file) { - std::string line; - while (std::getline(file, line)) { - line.erase(std::remove(line.begin(), line.end(), '\r'), line.end()); - std::transform(line.begin(), line.end(), line.begin(), ::tolower); //Transform to lowercase - if (allowList) m_ApprovedWords.push_back(CalculateHash(line)); - else m_DeniedWords.push_back(CalculateHash(line)); +void dChatFilter::LoadAllowList(const std::string& filepath) { + const std::string dcf = filepath + ".dcf"; + const std::string txt = filepath + ".txt"; + if (!m_DontGenerateDCF) { + auto cached = ReadFile(dcf); + if (cached.status == eStatus::OK) { + m_Lists.approved = std::move(cached.list); + return; } + if (cached.status != eStatus::MISSING) LOG("%s is %s; building it again from %s", dcf.c_str(), StatusText(cached.status), txt.c_str()); } + + const auto text = ReadBytes(txt); + if (!text) { + LOG("Could not read the chat filter's allowed words (%s)", txt.c_str()); + return; + } + m_Lists.approved = AllowListFromText(*text); + if (!m_DontGenerateDCF && !WriteFile(dcf, m_Lists.approved)) LOG("Could not write %s", dcf.c_str()); } -bool dChatFilter::ReadWordlistDCF(const std::string& filepath, bool allowList) { - std::ifstream file(filepath, std::ios::binary); - if (file) { - fileHeader hdr; - BinaryIO::BinaryRead(file, hdr); - if (hdr.header != header) { - file.close(); - return false; +void dChatFilter::LoadBlockList() { + std::error_code error; + const bool hasText = std::filesystem::exists(BLOCK_LIST_TEXT, error); + if (hasText) { + const auto text = ReadBytes(BLOCK_LIST_TEXT); + const auto existing = ReadFile(BLOCK_LIST_FILE); + // Rebuilt when the .dcf is missing, unreadable or older than the text + bool stale = existing.status != eStatus::OK; + if (!stale) { + std::error_code textError, fileError; + const auto textTime = std::filesystem::last_write_time(BLOCK_LIST_TEXT, textError); + const auto fileTime = std::filesystem::last_write_time(BLOCK_LIST_FILE, fileError); + stale = textError || fileError || fileTime < textTime; } - - if (hdr.formatVersion == formatVersion) { - size_t wordsToRead = 0; - BinaryIO::BinaryRead(file, wordsToRead); - if (allowList) m_ApprovedWords.reserve(wordsToRead); - else m_DeniedWords.reserve(wordsToRead); - - size_t word = 0; - for (size_t i = 0; i < wordsToRead; ++i) { - BinaryIO::BinaryRead(file, word); - if (allowList) m_ApprovedWords.push_back(word); - else m_DeniedWords.push_back(word); + if (text && (m_DontGenerateDCF || stale)) { + m_Lists.denied = BlockListFromText(*text); + if (m_DontGenerateDCF) { + LOG("Loaded %zu blocked words and phrases from %s", m_Lists.denied.Size(), BLOCK_LIST_TEXT); + return; } - - return true; - } else { - file.close(); - return false; + if (WriteFile(BLOCK_LIST_FILE, m_Lists.denied)) { + LOG("Built %s from %s (%zu words and phrases)", BLOCK_LIST_FILE, BLOCK_LIST_TEXT, m_Lists.denied.Size()); + } else { + LOG("Could not write %s", BLOCK_LIST_FILE); + } + return; } } - return false; + auto blocked = ReadFile(BLOCK_LIST_FILE); + switch (blocked.status) { + case eStatus::OK: + m_Lists.denied = std::move(blocked.list); + break; + case eStatus::MISSING: + LOG("No %s: best friends' free chat stops every message. Put the blocked words in %s next to the servers (one word or phrase per line) and start the servers again.", + BLOCK_LIST_FILE, BLOCK_LIST_TEXT); + break; + case eStatus::OLD_FORMAT: + LOG("%s is in the old format (version 2), whose hashes depend on the compiler and platform, so it can't be read; best friends' free chat stops every message. " + "Put the plain word list in %s next to the servers (one word or phrase per line) and start the servers again to rebuild it.", + BLOCK_LIST_FILE, BLOCK_LIST_TEXT); + break; + default: + LOG("%s is %s and can't be read; best friends' free chat stops every message. Rebuild it from %s.", BLOCK_LIST_FILE, StatusText(blocked.status), BLOCK_LIST_TEXT); + break; + } } -void dChatFilter::ExportWordlistToDCF(const std::string& filepath, bool allowList) { - std::ofstream file(filepath, std::ios::binary | std::ios_base::out); - if (file) { - BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::header)); - BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::formatVersion)); - BinaryIO::BinaryWrite(file, size_t(allowList ? m_ApprovedWords.size() : m_DeniedWords.size())); - - for (size_t word : allowList ? m_ApprovedWords : m_DeniedWords) { - BinaryIO::BinaryWrite(file, word); - } - - file.close(); +void dChatFilter::ReloadCustomWords() { + m_Lists.customAllowed = {}; + m_Lists.customBlocked = {}; + for (const auto& word : Database::Get()->GetChatFilterWords()) { + (word.allowed ? m_Lists.customAllowed : m_Lists.customBlocked).AddEntry(ChatFilterWords::NormalizeEntry(word.word)); } } std::set> dChatFilter::IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList) { if (gmLevel > eGameMasterLevel::FORUM_MODERATOR) return { }; //If anything but a forum mod, return true. - if (message.empty()) return { }; - if (!allowList && m_DeniedWords.empty()) return { { 0, message.length() } }; - - std::stringstream sMessage(message); - std::string segment; - - std::set> listOfBadSegments; - - uint32_t position = 0; - - while (std::getline(sMessage, segment, ' ')) { - std::string originalSegment = segment; - - segment = NormalizeWord(segment); - - size_t hash = CalculateHash(segment); - - // Blocked on the dashboard: stopped in every kind of chat - if (m_CustomBlockedWords.contains(hash)) { - listOfBadSegments.emplace(position, originalSegment.length()); - position += originalSegment.length() + 1; - continue; - } - - if (std::find(m_UserUnapprovedWordCache.begin(), m_UserUnapprovedWordCache.end(), hash) != m_UserUnapprovedWordCache.end() && allowList) { - listOfBadSegments.emplace(position, originalSegment.length()); - } - - if (std::find(m_ApprovedWords.begin(), m_ApprovedWords.end(), hash) == m_ApprovedWords.end() && !m_CustomAllowedWords.contains(hash) && allowList) { - m_UserUnapprovedWordCache.push_back(hash); - listOfBadSegments.emplace(position, originalSegment.length()); - } - - if (std::find(m_DeniedWords.begin(), m_DeniedWords.end(), hash) != m_DeniedWords.end() && !allowList) { - m_UserUnapprovedWordCache.push_back(hash); - listOfBadSegments.emplace(position, originalSegment.length()); - } - - position += originalSegment.length() + 1; - } - - return listOfBadSegments; -} - -size_t dChatFilter::CalculateHash(const std::string& word) { - std::hash hash{}; - - size_t value = hash(word); - - return value; + return ChatFilterWords::CheckMessage(message, allowList, m_Lists); } diff --git a/dChatFilter/dChatFilter.h b/dChatFilter/dChatFilter.h index cd8dac353..e6a221df5 100644 --- a/dChatFilter/dChatFilter.h +++ b/dChatFilter/dChatFilter.h @@ -1,70 +1,52 @@ #pragma once #include #include +#include +#include #include -#include -#include #include +#include +#include "ChatFilterCore.h" #include "dCommonVars.h" enum class eGameMasterLevel : uint8_t; -namespace dChatFilterDCF { - static const uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24)); - static const uint32_t formatVersion = 2; - - struct fileHeader { - uint32_t header; - uint32_t formatVersion; - }; -}; class dChatFilter { public: + /** + * Loads the allow list (filepath + ".txt", cached as filepath + ".dcf") and the block list (blocklist.dcf next to the + * servers, rebuilt from blocklist.txt there when that file is newer). dontGenerateDCF: read the plain lists only and + * write no .dcf files. + */ dChatFilter(const std::string& filepath, bool dontGenerateDCF); - ~dChatFilter(); + ~dChatFilter() = default; - void ReadWordlistPlaintext(const std::string& filepath, bool allowList); - bool ReadWordlistDCF(const std::string& filepath, bool allowList); - void ExportWordlistToDCF(const std::string& filepath, bool allowList); std::set> IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList = true); // Whether a deny list is loaded (without one, IsSentenceOkay(..., false) refuses every message) - bool HasDenyList() const { return !m_DeniedWords.empty(); } + bool HasDenyList() const { return !m_Lists.denied.Empty(); } /** * Load the words staff added on the dashboard (chat_filter_words) again, replacing the ones loaded before. - * Allowed words are accepted in whitelisted chat; blocked words are always stopped, even when a file allows them. + * Allowed words are accepted in whitelisted chat; blocked words and phrases are always stopped, even when a file allows them. */ void ReloadCustomWords(); // A word as the filter compares it: lower case, without ! ? ; . , - static std::string NormalizeWord(std::string word) { - std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; }); - std::transform(word.begin(), word.end(), word.begin(), ::tolower); //Transform to lowercase - return word; - } + static std::string NormalizeWord(std::string word) { return ChatFilterWords::NormalizeWord(std::move(word)); } // A message split into words (at spaces) the way the filter checks it static std::vector Words(const std::string& message) { std::vector words; - size_t start = 0; - while (start <= message.size()) { - const auto end = std::min(message.find(' ', start), message.size()); - words.push_back(NormalizeWord(message.substr(start, end - start))); - start = end + 1; - } + for (auto& token : ChatFilterWords::Tokenize(message)) words.push_back(std::move(token.word)); return words; } private: - bool m_DontGenerateDCF; - std::vector m_DeniedWords; - std::vector m_ApprovedWords; - std::vector m_UserUnapprovedWordCache; - std::unordered_set m_CustomAllowedWords; - std::unordered_set m_CustomBlockedWords; + void LoadAllowList(const std::string& filepath); + void LoadBlockList(); - //Private functions: - size_t CalculateHash(const std::string& word); + bool m_DontGenerateDCF; + ChatFilterWords::Lists m_Lists; }; diff --git a/dDashboardServer/routes/ChatFilterWords.h b/dDashboardServer/routes/ChatFilterWords.h index 29b07cea8..00dcb4cb4 100644 --- a/dDashboardServer/routes/ChatFilterWords.h +++ b/dDashboardServer/routes/ChatFilterWords.h @@ -8,23 +8,26 @@ #include #include -#include "dChatFilter.h" +#include "ChatFilterCore.h" // Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h) namespace ModerationTools { constexpr size_t MAX_FILTER_WORD = 64; - // A word staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,); nullopt if it isn't - // one word of 1-64 characters. Pure; unit tested. + // A word or phrase staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,, words joined by + // one space); nullopt if it isn't 1-64 characters or has no word. Pure; unit tested. inline std::optional FilterWord(std::string text) { text.erase(0, text.find_first_not_of(" \t\r\n")); text.erase(text.find_last_not_of(" \t\r\n") + 1); - if (text.empty() || text.size() > MAX_FILTER_WORD || text.find_first_of(" \t\r\n") != std::string::npos) return std::nullopt; - auto word = dChatFilter::NormalizeWord(text); - if (word.empty()) return std::nullopt; - return word; + if (text.empty() || text.size() > MAX_FILTER_WORD) return std::nullopt; + auto entry = ChatFilterWords::NormalizeEntry(text); + if (entry.empty()) return std::nullopt; + return entry; } + // Whether an entry is a phrase (more than one word). Phrases can only be blocked: whitelist chat checks one word at a time. + inline bool IsPhrase(const std::string& entry) { return ChatFilterWords::WordCount(entry) > 1; } + // The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once inline std::vector FileWords(const std::string& text) { std::vector words; @@ -33,7 +36,7 @@ namespace ModerationTools { const auto end = std::min(text.find('\n', start), text.size()); auto line = text.substr(start, end - start); std::erase(line, '\r'); - std::transform(line.begin(), line.end(), line.begin(), ::tolower); + line = ChatFilterWords::AsciiLower(std::move(line)); if (!line.empty()) words.push_back(std::move(line)); start = end + 1; } @@ -42,28 +45,12 @@ namespace ModerationTools { return words; } - // The hashes of a .dcf word list (blocklist.dcf), as dChatFilter::ReadWordlistDCF reads them; nullopt if it isn't one - inline std::optional> DcfHashes(const std::string& bytes) { - dChatFilterDCF::fileHeader header{}; - size_t count = 0; - if (bytes.size() < sizeof(header) + sizeof(count)) return std::nullopt; - std::memcpy(&header, bytes.data(), sizeof(header)); - if (header.header != dChatFilterDCF::header || header.formatVersion != dChatFilterDCF::formatVersion) return std::nullopt; - std::memcpy(&count, bytes.data() + sizeof(header), sizeof(count)); - const size_t offset = sizeof(header) + sizeof(count); - if (count > (bytes.size() - offset) / sizeof(size_t)) return std::nullopt; - std::vector hashes(count); - if (count) std::memcpy(hashes.data(), bytes.data() + offset, count * sizeof(size_t)); - return hashes; - } - - // A word's hash as the filter stores it (dChatFilter::CalculateHash) - inline size_t WordHash(const std::string& word) { return std::hash{}(word); } - - // Whether a message contains `word` as one of the words the chat filter checks - inline bool HasFilterWord(const std::string& message, const std::string& word) { - const auto words = dChatFilter::Words(message); - return std::find(words.begin(), words.end(), word) != words.end(); + // Whether a message contains a word or phrase (FilterWord) as the chat filter reads it: whole words, in a row, skipping + // pieces that are only punctuation + inline bool HasFilterWord(const std::string& message, const std::string& entry) { + const auto tokens = ChatFilterWords::Tokenize(message); + const auto words = ChatFilterWords::WordCount(entry); + return !ChatFilterWords::FindBlocked(tokens, words, [&entry](const std::string& run) { return run == entry; }).empty(); } // What the filter decides about one word of a message, and why @@ -72,6 +59,7 @@ namespace ModerationTools { std::string word; // as the filter compares it bool stopped{}; std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file + std::string phrase; // the blocked phrase this word is part of (blocked_here or block_file), when it was a phrase }; // Where the filter finds its words (callbacks keep this pure; the route reads the files and the database) @@ -81,33 +69,49 @@ namespace ModerationTools { std::function characterName; // approved character names count as allowed words std::function blockFile; // blocklist.dcf (by hash) bool blockFileLoaded{}; + uint32_t maxWords{ 1 }; // the longest blocked phrase, in words (here or in the file) }; /** * Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher - * levels skip the filter). Normal chat (allowList) needs every word allowed; best friends' free chat (!allowList) stops - * only blocked words, or every word when there is no blocked words file. Words are split at spaces as the filter - * splits them. Pure; unit tested. + * levels skip the filter). Blocked words and phrases (here always, the block file's in free chat) are stopped; a phrase + * stops each of its words. Normal chat (allowList) needs every other word allowed, one at a time; best friends' free + * chat stops only blocked ones, or every word when there is no blocked words file. Words are split at spaces as the + * filter splits them (ChatFilterWords::CheckMessage). Pure; unit tested. */ inline std::vector ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) { + const auto tokens = ChatFilterWords::Tokenize(message); std::vector verdicts; - std::stringstream stream(message); - std::string segment; - while (std::getline(stream, segment, ' ')) { - WordVerdict verdict{ segment, dChatFilter::NormalizeWord(segment) }; - const auto here = sources.dashboard(verdict.word); - if (!allowList && !sources.blockFileLoaded) { + for (const auto& token : tokens) verdicts.push_back({ message.substr(token.position, token.length), token.word }); + if (!allowList && !sources.blockFileLoaded) { + for (auto& verdict : verdicts) { verdict.stopped = true; verdict.reason = "no_block_file"; - } else if (here && !*here) { - verdict.stopped = true; - verdict.reason = "blocked_here"; - } else if (!allowList) { - verdict.stopped = sources.blockFile(verdict.word); - verdict.reason = verdict.stopped ? "block_file" : "not_in_block_file"; + } + return verdicts; + } + + const auto blockedHere = [&sources](const std::string& entry) { const auto here = sources.dashboard(entry); return here && !*here; }; + const auto matches = ChatFilterWords::FindBlocked(tokens, std::max(sources.maxWords, 1u), [&](const std::string& entry) { + return blockedHere(entry) || (!allowList && sources.blockFile(entry)); + }); + for (const auto& match : matches) { + const auto reason = blockedHere(match.entry) ? "blocked_here" : "block_file"; + for (size_t i = match.first; i <= match.last; i++) { + verdicts[i].stopped = true; + verdicts[i].reason = reason; + if (ChatFilterWords::WordCount(match.entry) > 1) verdicts[i].phrase = match.entry; + } + } + + for (auto& verdict : verdicts) { + if (verdict.stopped) continue; + const auto here = sources.dashboard(verdict.word); + if (!allowList) { + verdict.reason = "not_in_block_file"; } else if (sources.allowFile(verdict.word)) { verdict.reason = "allow_file"; - } else if (here) { + } else if (here && *here) { verdict.reason = "allowed_here"; } else if (sources.characterName(verdict.word)) { verdict.reason = "character_name"; @@ -115,7 +119,6 @@ namespace ModerationTools { verdict.stopped = true; verdict.reason = "not_allowed"; } - verdicts.push_back(std::move(verdict)); } return verdicts; } diff --git a/dDashboardServer/routes/ModerationTools.cpp b/dDashboardServer/routes/ModerationTools.cpp index 517288fad..31c63e9de 100644 --- a/dDashboardServer/routes/ModerationTools.cpp +++ b/dDashboardServer/routes/ModerationTools.cpp @@ -31,7 +31,8 @@ namespace { // The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked // words (only their hashes) next to the servers constexpr const char* ALLOW_FILE = "chatplus_en_us.txt"; - constexpr const char* BLOCK_FILE = "blocklist.dcf"; + constexpr const char* BLOCK_FILE = dChatFilterDCF::BLOCK_LIST_FILE; + constexpr const char* BLOCK_TEXT = dChatFilterDCF::BLOCK_LIST_TEXT; constexpr uint32_t FILE_WORDS_PAGE = 200; // Recent chat searched when checking what a word would change constexpr uint32_t CHECK_MESSAGES = 1000; @@ -216,11 +217,8 @@ namespace { return text ? ModerationTools::FileWords(*text) : std::vector{}; } - std::optional> BlockFileHashes() { - std::ifstream in(BLOCK_FILE, std::ios::binary); - if (!in) return std::nullopt; - return ModerationTools::DcfHashes(std::string((std::istreambuf_iterator(in)), std::istreambuf_iterator())); - } + // blocklist.dcf as the servers read it (an old-format file is refused, as they refuse it) + dChatFilterDCF::ParseResult BlockFile() { return dChatFilterDCF::ReadFile(BLOCK_FILE); } // The dashboard's own lists by word: true allowed, false blocked std::map DashboardWords() { @@ -258,25 +256,27 @@ namespace { const auto it = dashboard.find(word); words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} }); } - const auto blocked = BlockFileHashes(); + const auto blocked = BlockFile(); uint32_t imported = 0; for (const auto& word : all) if (dashboard.contains(word)) imported++; JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched}, {"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported}, - {"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.has_value()}, {"blockTotal", blocked ? blocked->size() : 0} }); + {"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK}, {"blockTotal", blocked.list.Size()}, + {"blockMaxWords", blocked.list.maxWords}, {"blockFileStatus", dChatFilterDCF::StatusText(blocked.status)}, + {"blockFileOld", blocked.status == dChatFilterDCF::eStatus::OLD_FORMAT}, {"blockText", BLOCK_TEXT} }); }); Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"), - "Where a word stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word", + "Where a word or phrase stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word", [](HTTPReply& reply, const HTTPContext& context) { const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word")); - if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); + if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters"); const auto all = AllowFileWords(); - const auto blocked = BlockFileHashes(); + const auto blocked = BlockFile(); const auto dashboard = DashboardWords(); const auto it = dashboard.find(*word); - JsonSuccess(reply, { {"word", *word}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)}, - {"inBlockFile", blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(*word)) != blocked->end()}, + JsonSuccess(reply, { {"word", *word}, {"phrase", ModerationTools::IsPhrase(*word)}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)}, + {"inBlockFile", blocked.list.Contains(*word)}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} }); }); @@ -293,7 +293,7 @@ namespace { for (const auto& word : all) { // Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only) const auto filterWord = ModerationTools::FilterWord(word); - if (!filterWord || *filterWord != word || dashboard.contains(word)) continue; + if (!filterWord || *filterWord != word || ModerationTools::IsPhrase(word) || dashboard.contains(word)) continue; Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now }); added++; } @@ -305,13 +305,17 @@ namespace { }); Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"), - "Allow or block a word (or move it to the other list); running worlds pick it up at once. Body: {word, allowed: bool}", + "Allow or block a word, or block a phrase (or move it to the other list); running worlds pick it up at once. Phrases can't be allowed: " + "whitelist chat checks each word on its own, as the client does. Body: {word, allowed: bool}", [](HTTPReply& reply, const HTTPContext& context) { const auto body = ParseBody(context); if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON"); const auto word = ModerationTools::FilterWord(body->value("word", "")); - if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); + if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters"); const bool allowed = body->value("allowed", false); + if (allowed && ModerationTools::IsPhrase(*word)) { + return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead"); + } Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast(std::time(nullptr)) }); Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat"); BroadcastTableChanged("chat_filter"); @@ -345,12 +349,11 @@ namespace { if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters"); const bool allowList = QueryValue(context.queryString, "chat") != "free"; const auto all = AllowFileWords(); - const auto blocked = BlockFileHashes(); + const auto blocked = BlockFile(); const auto dashboard = DashboardWords(); std::set names; for (auto name : Database::Get()->GetApprovedCharacterNames()) { - std::transform(name.begin(), name.end(), name.begin(), ::tolower); - names.insert(std::move(name)); + names.insert(ChatFilterWords::AsciiLower(std::move(name))); } ModerationTools::WordSources sources; sources.dashboard = [&dashboard](const std::string& word) -> std::optional { @@ -359,18 +362,18 @@ namespace { }; sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); }; sources.characterName = [&names](const std::string& word) { return names.contains(word); }; - sources.blockFile = [&blocked](const std::string& word) { - return blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(word)) != blocked->end(); - }; - sources.blockFileLoaded = blocked && !blocked->empty(); + sources.blockFile = [&blocked](const std::string& entry) { return blocked.list.Contains(entry); }; + sources.blockFileLoaded = !blocked.list.Empty(); + sources.maxWords = blocked.list.maxWords; + for (const auto& [entry, allowed] : dashboard) if (!allowed) sources.maxWords = std::max(sources.maxWords, ChatFilterWords::WordCount(entry)); nlohmann::json words = nlohmann::json::array(); bool stopped = false; for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) { stopped |= verdict.stopped; - words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason} }); + words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason}, {"phrase", verdict.phrase} }); } JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words}, - {"allowFileFound", !all.empty()}, {"blockFileFound", blocked.has_value()} }); + {"allowFileFound", !all.empty()}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK} }); }); Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"), @@ -379,7 +382,7 @@ namespace { [](HTTPReply& reply, const HTTPContext& context) { if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission"); const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word")); - if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); + if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters"); const bool allowed = QueryValue(context.queryString, "allowed") == "1"; IChatLog::ChatQuery query; query.search = *word; diff --git a/tests/dCommonTests/CMakeLists.txt b/tests/dCommonTests/CMakeLists.txt index bca761760..772126890 100644 --- a/tests/dCommonTests/CMakeLists.txt +++ b/tests/dCommonTests/CMakeLists.txt @@ -39,6 +39,7 @@ set(DCOMMONTEST_SOURCES "FdbReaderTests.cpp" "FdbSnapshotTests.cpp" "WorldFileWatchTests.cpp" + "ChatFilterCoreTests.cpp" ) add_subdirectory(dEnumsTests) @@ -59,6 +60,8 @@ endif() target_link_libraries(dCommonTests ${COMMON_LIBRARIES} MD5 GTest::gtest_main) # SpareBackoff.h (header only) target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dMasterServer") +# ChatFilterCore.h (header only) +target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dChatFilter") # Copy test files to testing directory add_subdirectory(TestBitStreams) diff --git a/tests/dCommonTests/ChatFilterCoreTests.cpp b/tests/dCommonTests/ChatFilterCoreTests.cpp new file mode 100644 index 000000000..50b66590b --- /dev/null +++ b/tests/dCommonTests/ChatFilterCoreTests.cpp @@ -0,0 +1,149 @@ +#include + +#include +#include + +#include "ChatFilterCore.h" + +using namespace ChatFilterWords; +using Spans = std::set>; + +// The hash is a compile-time constant: the same value on every compiler, standard library and platform +static_assert(Hash("") == 0xcbf29ce484222325ULL); +static_assert(Hash("a") == 0xaf63dc4c8601ec8cULL); +static_assert(Hash("foobar") == 0x85944171f73967e8ULL); + +TEST(ChatFilterCoreTest, HashIsFnv1a64) { + // The standard FNV-1a 64 test vectors + EXPECT_EQ(Hash(""), 0xcbf29ce484222325ULL); + EXPECT_EQ(Hash("a"), 0xaf63dc4c8601ec8cULL); + EXPECT_EQ(Hash("foobar"), 0x85944171f73967e8ULL); + // Words of the client's chatplus_en_us.txt, as the filter hashes them (lower case) + EXPECT_EQ(Hash("hello"), 0xa430d84680aabd0bULL); + EXPECT_EQ(Hash("brick"), 0xf9236d1e24832c9aULL); + EXPECT_EQ(Hash(AsciiLower("Brick")), Hash("brick")); + // A phrase is its words joined by one space + EXPECT_EQ(Hash("bad phrase"), 0x72e9fce49c1b0875ULL); + // Bytes, not chars: a high byte hashes as 0x80-0xFF whether char is signed or not + EXPECT_EQ(Hash("\xC3\xA9"), 0x0ac21707b7181e01ULL); +} + +TEST(ChatFilterCoreTest, NormalizeEntry) { + EXPECT_EQ(NormalizeWord("Hello!?"), "hello"); + EXPECT_EQ(NormalizeEntry(" Bad \t PHRASE! "), "bad phrase"); + EXPECT_EQ(NormalizeEntry("word , here"), "word here"); + EXPECT_EQ(NormalizeEntry("..."), ""); + EXPECT_EQ(WordCount("bad phrase here"), 3u); + EXPECT_EQ(WordCount("word"), 1u); + EXPECT_EQ(WordCount(""), 0u); + // ASCII only: other bytes are left alone whatever the locale + EXPECT_EQ(AsciiLower("\xC3\x89Z"), "\xC3\x89z"); +} + +TEST(ChatFilterCoreTest, DcfBytesAreFixed) { + WordList list; + list.AddEntry("a"); + list.AddEntry("bad phrase"); + const auto bytes = dChatFilterDCF::Serialize(list); + // magic DCFB, version 3, 2 words at most, 2 hashes, then the hashes sorted, all little-endian + const std::string expected( + "DCFB" "\x03\x00\x00\x00" "\x02\x00\x00\x00" "\x02\x00\x00\x00\x00\x00\x00\x00" + "\x75\x08\x1b\x9c\xe4\xfc\xe9\x72" "\x8c\xec\x01\x86\x4c\xdc\x63\xaf", 4 + 4 + 4 + 8 + 16); + EXPECT_EQ(bytes, expected); + + const auto parsed = dChatFilterDCF::Parse(bytes); + ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK); + EXPECT_EQ(parsed.list.maxWords, 2u); + EXPECT_EQ(parsed.list.hashes, list.hashes); +} + +TEST(ChatFilterCoreTest, DcfRejectsOldAndBadFiles) { + // Version 2 (std::hash, platform dependent) is refused, not guessed at + const std::string old("DCFB" "\x02\x00\x00\x00" "\x01\x00\x00\x00\x00\x00\x00\x00" "\x11\x22\x33\x44\x55\x66\x77\x88", 24); + EXPECT_EQ(dChatFilterDCF::Parse(old).status, dChatFilterDCF::eStatus::OLD_FORMAT); + EXPECT_EQ(dChatFilterDCF::Parse("XCFB\x03\x00\x00\x00").status, dChatFilterDCF::eStatus::NOT_DCF); + EXPECT_EQ(dChatFilterDCF::Parse(std::string("DCFB\x09\x00\x00\x00", 8)).status, dChatFilterDCF::eStatus::UNKNOWN); + WordList list; + list.AddEntry("word"); + auto bytes = dChatFilterDCF::Serialize(list); + bytes.pop_back(); + EXPECT_EQ(dChatFilterDCF::Parse(bytes).status, dChatFilterDCF::eStatus::TRUNCATED); + EXPECT_EQ(dChatFilterDCF::ReadFile("no such blocklist.dcf").status, dChatFilterDCF::eStatus::MISSING); +} + +TEST(ChatFilterCoreTest, BlockListFromText) { + const auto list = dChatFilterDCF::BlockListFromText("Badword\r\n\r\n Very BAD phrase!\nbadword\n...\n"); + EXPECT_EQ(list.Size(), 2u); + EXPECT_TRUE(list.Contains("badword")); + EXPECT_TRUE(list.Contains("very bad phrase")); + EXPECT_EQ(list.maxWords, 3u); + // Written and read back, the same list + const auto parsed = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(list)); + ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK); + EXPECT_EQ(parsed.list.hashes, list.hashes); + EXPECT_EQ(parsed.list.maxWords, 3u); +} + +namespace { + Lists FreeChatLists(std::string_view blockText) { + Lists lists; + // Through the file format, as the servers load it + lists.denied = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(dChatFilterDCF::BlockListFromText(blockText))).list; + lists.approved = dChatFilterDCF::AllowListFromText("hello\nthere\nfriend\nbad\nphrase\n"); + return lists; + } +} + +TEST(ChatFilterCoreTest, BlockedWordFromFileIsStopped) { + const auto lists = FreeChatLists("badword\n"); + EXPECT_EQ(CheckMessage("hello badword there", false, lists), (Spans{ { 6, 7 } })); + EXPECT_EQ(CheckMessage("hello BadWord!", false, lists), (Spans{ { 6, 8 } })); + EXPECT_TRUE(CheckMessage("hello there", false, lists).empty()); + // Whitelist chat doesn't use the block list: the word is simply not allowed there + EXPECT_EQ(CheckMessage("hello badword", true, lists), (Spans{ { 6, 7 } })); + // No block list: free chat stops everything + EXPECT_EQ(CheckMessage("hello", false, Lists{}), (Spans{ { 0, 5 } })); +} + +TEST(ChatFilterCoreTest, PhrasesAtStartMiddleEnd) { + const auto lists = FreeChatLists("bad phrase\n"); + EXPECT_EQ(CheckMessage("bad phrase hello", false, lists), (Spans{ { 0, 10 } })); + EXPECT_EQ(CheckMessage("hello bad phrase there", false, lists), (Spans{ { 6, 10 } })); + EXPECT_EQ(CheckMessage("hello bad phrase", false, lists), (Spans{ { 6, 10 } })); + // The words alone are fine + EXPECT_TRUE(CheckMessage("bad hello phrase", false, lists).empty()); + EXPECT_TRUE(CheckMessage("phrase bad", false, lists).empty()); +} + +TEST(ChatFilterCoreTest, PhrasesWithExtraSpacesAndPunctuation) { + const auto lists = FreeChatLists("bad phrase\n"); + // Two spaces: the span covers both words and the gap + EXPECT_EQ(CheckMessage("hi Bad Phrase!", false, lists), (Spans{ { 4, 12 } })); + // A piece that is only punctuation is skipped + EXPECT_EQ(CheckMessage("bad ... phrase", false, lists), (Spans{ { 0, 14 } })); + EXPECT_EQ(CheckMessage("bad, phrase.", false, lists), (Spans{ { 0, 12 } })); +} + +TEST(ChatFilterCoreTest, PhrasesOverlapWithWords) { + const auto lists = FreeChatLists("bad phrase\nphrase here\nbadword\nfriend\n"); + // Two phrases sharing a word become one span + EXPECT_EQ(CheckMessage("a bad phrase here b", false, lists), (Spans{ { 2, 15 } })); + // A blocked word inside a blocked phrase: one span for the phrase + EXPECT_EQ(CheckMessage("bad phrase", false, FreeChatLists("bad phrase\nphrase\n")), (Spans{ { 0, 10 } })); + // A blocked word right after a phrase: its own span + EXPECT_EQ(CheckMessage("bad phrase badword", false, lists), (Spans{ { 0, 10 }, { 11, 7 } })); + // Longest match wins where it starts + EXPECT_EQ(CheckMessage("friend bad phrase", false, FreeChatLists("friend\nfriend bad\nbad phrase\n")), (Spans{ { 0, 17 } })); +} + +TEST(ChatFilterCoreTest, DashboardPhrasesInWhitelistChat) { + auto lists = FreeChatLists(""); + lists.customBlocked.AddEntry(NormalizeEntry("Bad Phrase")); + // Blocked on the dashboard: stopped in whitelist chat too, even though each word is allowed + EXPECT_EQ(CheckMessage("hello bad phrase", true, lists), (Spans{ { 6, 10 } })); + EXPECT_TRUE(CheckMessage("hello bad there phrase", true, lists).empty()); + // Whitelist chat checks one word at a time, as the client does + EXPECT_EQ(CheckMessage("hello stranger", true, lists), (Spans{ { 6, 8 } })); + lists.customAllowed.AddEntry("stranger"); + EXPECT_TRUE(CheckMessage("hello stranger", true, lists).empty()); +} diff --git a/tests/dWebTests/ModerationToolsTests.cpp b/tests/dWebTests/ModerationToolsTests.cpp index 960f832c1..ccc2b980c 100644 --- a/tests/dWebTests/ModerationToolsTests.cpp +++ b/tests/dWebTests/ModerationToolsTests.cpp @@ -43,7 +43,10 @@ TEST(ChatFilterWordsTest, FilterWord) { EXPECT_EQ(ModerationTools::FilterWord(" Hello! "), "hello"); EXPECT_EQ(ModerationTools::FilterWord("W.o,r;d?"), "word"); EXPECT_FALSE(ModerationTools::FilterWord("")); - EXPECT_FALSE(ModerationTools::FilterWord("two words")); + // Phrases: words normalized and joined by one space + EXPECT_EQ(ModerationTools::FilterWord(" Two Words! "), "two words"); + EXPECT_TRUE(ModerationTools::IsPhrase("two words")); + EXPECT_FALSE(ModerationTools::IsPhrase("word")); EXPECT_FALSE(ModerationTools::FilterWord("!!!")); EXPECT_FALSE(ModerationTools::FilterWord(std::string(65, 'a'))); } @@ -53,6 +56,11 @@ TEST(ChatFilterWordsTest, HasFilterWord) { EXPECT_TRUE(ModerationTools::HasFilterWord("bad", "bad")); EXPECT_FALSE(ModerationTools::HasFilterWord("badger badminton", "bad")); EXPECT_FALSE(ModerationTools::HasFilterWord("", "bad")); + // Phrases: the words in a row, whatever the spaces and punctuation between them + EXPECT_TRUE(ModerationTools::HasFilterWord("well, Bad Phrase!", "bad phrase")); + EXPECT_TRUE(ModerationTools::HasFilterWord("bad ... phrase", "bad phrase")); + EXPECT_FALSE(ModerationTools::HasFilterWord("bad other phrase", "bad phrase")); + EXPECT_FALSE(ModerationTools::HasFilterWord("phrase bad", "bad phrase")); } TEST(ChatFilterWordsTest, FileWords) { @@ -61,27 +69,6 @@ TEST(ChatFilterWordsTest, FileWords) { ASSERT_TRUE(ModerationTools::FileWords("").empty()); } -TEST(ChatFilterWordsTest, DcfHashes) { - const std::vector hashes{ ModerationTools::WordHash("badword"), 42 }; - std::string bytes(sizeof(dChatFilterDCF::fileHeader) + sizeof(size_t) * (hashes.size() + 1), '\0'); - const dChatFilterDCF::fileHeader header{ dChatFilterDCF::header, dChatFilterDCF::formatVersion }; - const size_t count = hashes.size(); - std::memcpy(bytes.data(), &header, sizeof(header)); - std::memcpy(bytes.data() + sizeof(header), &count, sizeof(count)); - std::memcpy(bytes.data() + sizeof(header) + sizeof(count), hashes.data(), sizeof(size_t) * count); - ASSERT_EQ(ModerationTools::DcfHashes(bytes), hashes); - - // Wrong header, other version, or fewer hashes than it says - auto wrong = bytes; - wrong[0] = 'X'; - ASSERT_FALSE(ModerationTools::DcfHashes(wrong).has_value()); - auto version = bytes; - version[sizeof(uint32_t)] = 9; - ASSERT_FALSE(ModerationTools::DcfHashes(version).has_value()); - ASSERT_FALSE(ModerationTools::DcfHashes(bytes.substr(0, bytes.size() - sizeof(size_t) * 2)).has_value()); - ASSERT_FALSE(ModerationTools::DcfHashes("DCFB").has_value()); -} - namespace { ModerationTools::WordSources Sources(bool blockFileLoaded = true) { ModerationTools::WordSources sources; @@ -121,3 +108,22 @@ TEST(ChatFilterWordsTest, ExplainFreeChat) { EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("zzz darn", false, Sources(false))), (std::vector{ "x no_block_file", "x no_block_file" })); } + +TEST(ChatFilterWordsTest, ExplainPhrases) { + auto sources = Sources(); + sources.dashboard = [](const std::string& w) -> std::optional { + if (w == "no way") return false; + return std::nullopt; + }; + sources.allowFile = [](const std::string& w) { return w == "hello" || w == "no" || w == "way" || w == "rude"; }; + sources.blockFile = [](const std::string& w) { return w == "very rude"; }; + sources.maxWords = 2; + // A phrase blocked here stops each of its words (and the empty piece between two spaces inside it), in normal chat too + auto verdicts = ModerationTools::ExplainMessage("hello No way!", true, sources); + EXPECT_EQ(Reasons(verdicts), (std::vector{ "ok allow_file", "x blocked_here", "x blocked_here", "x blocked_here" })); + EXPECT_EQ(verdicts[1].phrase, "no way"); + EXPECT_EQ(verdicts[3].text, "way!"); + // The block file's phrases only in free chat + EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("very rude", false, sources)), (std::vector{ "x block_file", "x block_file" })); + EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("rude very", false, sources)), (std::vector{ "ok not_in_block_file", "ok not_in_block_file" })); +}