fix(chat-filter): portable .dcf hashing, block list phrases

The filter stored and compared words by std::hash<std::string> in size_t,
which differs between standard libraries and platforms, so a .dcf made on
one system never matched on another and the block list never worked
there (issue 215).

- ChatFilterCore.h: 64-bit FNV-1a over the entry's bytes, ASCII lower
  case, fixed-width uint64_t everywhere stored or compared.
- .dcf version 3: little-endian magic, version, longest entry in words,
  uint64 count and sorted uint64 hashes. Version 2 files are refused:
  the allowed words cache is rebuilt from its .txt, an old
  blocklist.dcf is logged as unreadable.
- The servers build blocklist.dcf from a plain blocklist.txt next to
  them (one word or phrase per line) when it is newer.
- Blocked entries can be phrases: runs of consecutive words up to the
  longest entry, the whole run marked. Whitelist chat still checks one
  word at a time, as the client does.
- Dashboard: the chat filter API reads blocklist.dcf the same way
  (status, phrase length), accepts blocked phrases, refuses allowed
  ones, and explains phrase matches in its message test.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Aaron Kimbrell
2026-09-30 08:05:00 -05:00
parent 3018312f29
commit c1bcda8dd1
8 changed files with 691 additions and 268 deletions

View File

@@ -0,0 +1,342 @@
#pragma once
#include <algorithm>
#include <cstdint>
#include <filesystem>
#include <fstream>
#include <functional>
#include <iterator>
#include <optional>
#include <random>
#include <set>
#include <string>
#include <string_view>
#include <unordered_set>
#include <utility>
#include <vector>
/**
* The chat filter's words, without the server around them (pure, unit tested; dChatFilter and the dashboard both use it).
*
* A word is compared lower case (ASCII only, so every platform agrees) without ! ? ; . , and a phrase is its words
* joined by one space. Entries are stored and compared by ChatFilterWords::Hash: 64-bit FNV-1a over the entry's bytes,
* the same on every compiler, standard library and platform.
*/
namespace ChatFilterWords {
// ASCII lower case; other bytes (UTF-8) stay as they are
inline std::string AsciiLower(std::string text) {
for (auto& c : text) if (c >= 'A' && c <= 'Z') c = static_cast<char>(c - 'A' + 'a');
return text;
}
// A word as the filter compares it: lower case, without ! ? ; . ,
inline std::string NormalizeWord(std::string word) {
std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; });
return AsciiLower(std::move(word));
}
// A word or phrase as the filter stores it: each word normalized, words that end up empty dropped, joined by one space
inline std::string NormalizeEntry(std::string_view text) {
std::string entry;
size_t start = 0;
while (start < text.size()) {
auto end = text.find_first_of(" \t\r\n", start);
if (end == std::string_view::npos) end = text.size();
const auto word = NormalizeWord(std::string(text.substr(start, end - start)));
if (!word.empty()) {
if (!entry.empty()) entry += ' ';
entry += word;
}
start = end + 1;
}
return entry;
}
// How many words an entry has (1 for a word, 0 for an empty entry)
inline uint32_t WordCount(std::string_view entry) {
return entry.empty() ? 0 : static_cast<uint32_t>(std::count(entry.begin(), entry.end(), ' ')) + 1;
}
// 64-bit FNV-1a: offset basis 0xcbf29ce484222325, prime 0x100000001b3, one byte at a time
constexpr uint64_t Hash(std::string_view entry) {
uint64_t hash = 0xcbf29ce484222325ULL;
for (const char c : entry) {
hash ^= static_cast<uint8_t>(c);
hash *= 0x100000001b3ULL;
}
return hash;
}
// One piece of a message between spaces: where it is in the message and the word the filter compares
struct Token {
uint32_t position{};
uint32_t length{};
std::string word;
};
// A message split at each space, the way the filter checks it: two spaces in a row give an empty piece, a trailing space none
inline std::vector<Token> Tokenize(std::string_view message) {
std::vector<Token> tokens;
size_t start = 0;
while (start < message.size()) {
auto end = message.find(' ', start);
if (end == std::string_view::npos) end = message.size();
tokens.push_back({ static_cast<uint32_t>(start), static_cast<uint32_t>(end - start), NormalizeWord(std::string(message.substr(start, end - start))) });
start = end + 1;
}
return tokens;
}
// A run of tokens [first, last] that matched a blocked entry, and the longest entry that matched where the run starts
struct Match {
size_t first{};
size_t last{};
std::string entry;
};
/**
* The blocked words and phrases in a message: at each word, the longest run of up to maxWords consecutive words
* (tokens with no word are skipped) whose entry isBlocked accepts. Runs that share a word are merged into one.
*/
inline std::vector<Match> FindBlocked(const std::vector<Token>& tokens, uint32_t maxWords, const std::function<bool(const std::string&)>& isBlocked) {
std::vector<size_t> words;
for (size_t i = 0; i < tokens.size(); i++) if (!tokens[i].word.empty()) words.push_back(i);
std::vector<Match> matches;
for (size_t i = 0; i < words.size(); i++) {
std::string entry;
std::string best;
size_t bestLength = 0;
for (size_t n = 1; n <= maxWords && i + n <= words.size(); n++) {
if (n > 1) entry += ' ';
entry += tokens[words[i + n - 1]].word;
if (isBlocked(entry)) {
best = entry;
bestLength = n;
}
}
if (bestLength == 0) continue;
const size_t first = words[i];
const size_t last = words[i + bestLength - 1];
if (!matches.empty() && first <= matches.back().last) {
matches.back().last = std::max(matches.back().last, last);
} else {
matches.push_back({ first, last, std::move(best) });
}
}
return matches;
}
// Hashes of words or phrases, and the most words any of them has
struct WordList {
std::unordered_set<uint64_t> hashes;
uint32_t maxWords{};
void AddEntry(std::string_view entry) {
if (entry.empty()) return;
hashes.insert(Hash(entry));
maxWords = std::max(maxWords, WordCount(entry));
}
bool Contains(std::string_view entry) const { return hashes.contains(Hash(entry)); }
bool Empty() const { return hashes.empty(); }
size_t Size() const { return hashes.size(); }
};
// Everything the filter checks a message against
struct Lists {
WordList approved; // chatplus_en_us.txt and approved character names: whitelist chat, one word at a time
WordList denied; // blocklist.dcf: best friends' free chat
WordList customAllowed; // allowed on the dashboard
WordList customBlocked; // blocked on the dashboard: stopped in every kind of chat
};
/**
* The pieces of a message the filter stops, as (position, length) in the message. Blocked words and phrases (the
* dashboard's always, blocklist.dcf's in free chat) are stopped as one span each. In whitelist chat (allowList) every
* other piece must be an allowed word, one at a time, as the client checks words. In free chat without a block list
* the whole message is stopped.
*/
inline std::set<std::pair<uint8_t, uint8_t>> CheckMessage(std::string_view message, bool allowList, const Lists& lists) {
if (message.empty()) return {};
if (!allowList && lists.denied.Empty()) return { { 0, static_cast<uint8_t>(message.length()) } };
const auto tokens = Tokenize(message);
const uint32_t maxWords = std::max(lists.customBlocked.maxWords, allowList ? 0u : lists.denied.maxWords);
const auto matches = FindBlocked(tokens, maxWords, [&](const std::string& entry) {
return lists.customBlocked.Contains(entry) || (!allowList && lists.denied.Contains(entry));
});
std::set<std::pair<uint8_t, uint8_t>> bad;
std::vector<bool> covered(tokens.size(), false);
for (const auto& match : matches) {
const auto& first = tokens[match.first];
const auto& last = tokens[match.last];
bad.emplace(static_cast<uint8_t>(first.position), static_cast<uint8_t>(last.position + last.length - first.position));
for (size_t i = match.first; i <= match.last; i++) covered[i] = true;
}
if (allowList) {
for (size_t i = 0; i < tokens.size(); i++) {
if (covered[i]) continue;
const auto hash = Hash(tokens[i].word);
if (!lists.approved.hashes.contains(hash) && !lists.customAllowed.hashes.contains(hash)) {
bad.emplace(static_cast<uint8_t>(tokens[i].position), static_cast<uint8_t>(tokens[i].length));
}
}
}
return bad;
}
}
/**
* The chat filter's word list files (.dcf). These are DLU's own files: the client reads no .dcf and hashes no chat
* words (it keeps its lists as plain text). Layout, little-endian:
* uint32 magic 'DCFB' | uint32 version (3) | uint32 most words in one entry | uint64 count | count x uint64 ChatFilterWords::Hash
* Version 2 (older DLU) stored std::hash values, which differ between compilers and platforms; those can't be read.
*/
namespace dChatFilterDCF {
constexpr uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24));
constexpr uint32_t formatVersion = 3;
constexpr uint32_t oldFormatVersion = 2;
constexpr size_t headerSize = 4 + 4 + 4 + 8;
// The block list's plain source (one word or phrase per line) and the .dcf built from it, next to the servers
constexpr const char* BLOCK_LIST_TEXT = "blocklist.txt";
constexpr const char* BLOCK_LIST_FILE = "blocklist.dcf";
enum class eStatus : uint8_t {
OK,
MISSING, // no file
NOT_DCF, // not a .dcf file
OLD_FORMAT, // version 2: platform-dependent hashes, rebuild it from the plain word list
UNKNOWN, // a version this server doesn't know
TRUNCATED, // shorter than its count says
};
inline const char* StatusText(eStatus status) {
switch (status) {
case eStatus::OK: return "ok";
case eStatus::MISSING: return "missing";
case eStatus::NOT_DCF: return "not a .dcf file";
case eStatus::OLD_FORMAT: return "old format (version 2, platform-dependent hashes)";
case eStatus::UNKNOWN: return "unknown version";
case eStatus::TRUNCATED: return "truncated";
}
return "unknown";
}
struct ParseResult {
eStatus status{ eStatus::NOT_DCF };
uint32_t version{};
ChatFilterWords::WordList list;
};
namespace detail {
inline uint64_t ReadLE(std::string_view bytes, size_t offset, size_t size) {
uint64_t value = 0;
for (size_t i = 0; i < size; i++) value |= static_cast<uint64_t>(static_cast<uint8_t>(bytes[offset + i])) << (8 * i);
return value;
}
inline void WriteLE(std::string& out, uint64_t value, size_t size) {
for (size_t i = 0; i < size; i++) out.push_back(static_cast<char>((value >> (8 * i)) & 0xFF));
}
}
inline ParseResult Parse(std::string_view bytes) {
ParseResult result;
if (bytes.size() < 8 || detail::ReadLE(bytes, 0, 4) != header) return result;
result.version = static_cast<uint32_t>(detail::ReadLE(bytes, 4, 4));
if (result.version == oldFormatVersion) {
result.status = eStatus::OLD_FORMAT;
return result;
}
if (result.version != formatVersion) {
result.status = eStatus::UNKNOWN;
return result;
}
result.status = eStatus::TRUNCATED;
if (bytes.size() < headerSize) return result;
const auto maxWords = static_cast<uint32_t>(detail::ReadLE(bytes, 8, 4));
const auto count = detail::ReadLE(bytes, 12, 8);
if (count > (bytes.size() - headerSize) / 8) return result;
result.list.maxWords = maxWords;
result.list.hashes.reserve(count);
for (uint64_t i = 0; i < count; i++) result.list.hashes.insert(detail::ReadLE(bytes, headerSize + i * 8, 8));
result.status = eStatus::OK;
return result;
}
// A list as a .dcf file; hashes sorted, so the same words always give the same bytes
inline std::string Serialize(const ChatFilterWords::WordList& list) {
std::vector<uint64_t> hashes(list.hashes.begin(), list.hashes.end());
std::sort(hashes.begin(), hashes.end());
std::string out;
out.reserve(headerSize + hashes.size() * 8);
detail::WriteLE(out, header, 4);
detail::WriteLE(out, formatVersion, 4);
detail::WriteLE(out, list.maxWords, 4);
detail::WriteLE(out, hashes.size(), 8);
for (const auto hash : hashes) detail::WriteLE(out, hash, 8);
return out;
}
// A plain block list: one word or phrase per line, normalized (ChatFilterWords::NormalizeEntry); empty lines skipped
inline ChatFilterWords::WordList BlockListFromText(std::string_view text) {
ChatFilterWords::WordList list;
size_t start = 0;
while (start < text.size()) {
auto end = text.find('\n', start);
if (end == std::string_view::npos) end = text.size();
list.AddEntry(ChatFilterWords::NormalizeEntry(text.substr(start, end - start)));
start = end + 1;
}
return list;
}
// A plain allow list (chatplus_en_us.txt): one word per line, lower case, compared whole (as the filter always has)
inline ChatFilterWords::WordList AllowListFromText(std::string_view text) {
ChatFilterWords::WordList list;
size_t start = 0;
while (start < text.size()) {
auto end = text.find('\n', start);
if (end == std::string_view::npos) end = text.size();
std::string line(text.substr(start, end - start));
std::erase(line, '\r');
line = ChatFilterWords::AsciiLower(std::move(line));
list.hashes.insert(ChatFilterWords::Hash(line));
list.maxWords = std::max(list.maxWords, 1u);
start = end + 1;
}
return list;
}
inline std::optional<std::string> ReadBytes(const std::filesystem::path& path) {
std::ifstream in(path, std::ios::binary);
if (!in) return std::nullopt;
return std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>());
}
inline ParseResult ReadFile(const std::filesystem::path& path) {
const auto bytes = ReadBytes(path);
if (!bytes) return { eStatus::MISSING };
return Parse(*bytes);
}
// Writes the file whole or not at all (a temporary file renamed over it), so servers starting together don't read half a file
inline bool WriteFile(const std::filesystem::path& path, const ChatFilterWords::WordList& list) {
auto temp = path;
temp += "." + std::to_string(std::random_device{}()) + ".tmp";
{
std::ofstream out(temp, std::ios::binary | std::ios::trunc);
if (!out) return false;
const auto bytes = Serialize(list);
out.write(bytes.data(), static_cast<std::streamsize>(bytes.size()));
if (!out) return false;
}
std::error_code error;
std::filesystem::rename(temp, path, error);
if (!error) return true;
std::filesystem::remove(temp, error);
return false;
}
}

View File

@@ -1,15 +1,8 @@
#include "dChatFilter.h"
#include "BinaryIO.h"
#include <fstream>
#include <string>
#include <functional>
#include <algorithm>
#include <sstream>
#include <regex>
#include "dCommonVars.h"
#include <system_error>
#include "Logger.h"
#include "dConfig.h"
#include "Database.h"
#include "Game.h"
#include "eGameMasterLevel.h"
@@ -19,154 +12,96 @@ using namespace dChatFilterDCF;
dChatFilter::dChatFilter(const std::string& filepath, bool dontGenerateDCF) {
m_DontGenerateDCF = dontGenerateDCF;
if (!BinaryIO::DoesFileExist(filepath + ".dcf") || m_DontGenerateDCF) {
ReadWordlistPlaintext(filepath + ".txt", true);
if (!m_DontGenerateDCF) ExportWordlistToDCF(filepath + ".dcf", true);
} else if (!ReadWordlistDCF(filepath + ".dcf", true)) {
ReadWordlistPlaintext(filepath + ".txt", true);
ExportWordlistToDCF(filepath + ".dcf", true);
}
LoadAllowList(filepath);
LoadBlockList();
if (BinaryIO::DoesFileExist("blocklist.dcf")) {
ReadWordlistDCF("blocklist.dcf", false);
}
//Read player names that are ok as well:
auto approvedNames = Database::Get()->GetApprovedCharacterNames();
for (auto& name : approvedNames) {
std::transform(name.begin(), name.end(), name.begin(), ::tolower); //Transform to lowercase
m_ApprovedWords.push_back(CalculateHash(name));
// Approved character names count as allowed words
for (const auto& name : Database::Get()->GetApprovedCharacterNames()) {
m_Lists.approved.hashes.insert(ChatFilterWords::Hash(ChatFilterWords::AsciiLower(name)));
}
ReloadCustomWords();
}
void dChatFilter::ReloadCustomWords() {
m_CustomAllowedWords.clear();
m_CustomBlockedWords.clear();
// Words remembered as not allowed may be allowed now
m_UserUnapprovedWordCache.clear();
for (const auto& word : Database::Get()->GetChatFilterWords()) {
(word.allowed ? m_CustomAllowedWords : m_CustomBlockedWords).insert(CalculateHash(NormalizeWord(word.word)));
}
}
dChatFilter::~dChatFilter() {
m_ApprovedWords.clear();
m_DeniedWords.clear();
}
void dChatFilter::ReadWordlistPlaintext(const std::string& filepath, bool allowList) {
std::ifstream file(filepath);
if (file) {
std::string line;
while (std::getline(file, line)) {
line.erase(std::remove(line.begin(), line.end(), '\r'), line.end());
std::transform(line.begin(), line.end(), line.begin(), ::tolower); //Transform to lowercase
if (allowList) m_ApprovedWords.push_back(CalculateHash(line));
else m_DeniedWords.push_back(CalculateHash(line));
void dChatFilter::LoadAllowList(const std::string& filepath) {
const std::string dcf = filepath + ".dcf";
const std::string txt = filepath + ".txt";
if (!m_DontGenerateDCF) {
auto cached = ReadFile(dcf);
if (cached.status == eStatus::OK) {
m_Lists.approved = std::move(cached.list);
return;
}
if (cached.status != eStatus::MISSING) LOG("%s is %s; building it again from %s", dcf.c_str(), StatusText(cached.status), txt.c_str());
}
const auto text = ReadBytes(txt);
if (!text) {
LOG("Could not read the chat filter's allowed words (%s)", txt.c_str());
return;
}
m_Lists.approved = AllowListFromText(*text);
if (!m_DontGenerateDCF && !WriteFile(dcf, m_Lists.approved)) LOG("Could not write %s", dcf.c_str());
}
bool dChatFilter::ReadWordlistDCF(const std::string& filepath, bool allowList) {
std::ifstream file(filepath, std::ios::binary);
if (file) {
fileHeader hdr;
BinaryIO::BinaryRead(file, hdr);
if (hdr.header != header) {
file.close();
return false;
void dChatFilter::LoadBlockList() {
std::error_code error;
const bool hasText = std::filesystem::exists(BLOCK_LIST_TEXT, error);
if (hasText) {
const auto text = ReadBytes(BLOCK_LIST_TEXT);
const auto existing = ReadFile(BLOCK_LIST_FILE);
// Rebuilt when the .dcf is missing, unreadable or older than the text
bool stale = existing.status != eStatus::OK;
if (!stale) {
std::error_code textError, fileError;
const auto textTime = std::filesystem::last_write_time(BLOCK_LIST_TEXT, textError);
const auto fileTime = std::filesystem::last_write_time(BLOCK_LIST_FILE, fileError);
stale = textError || fileError || fileTime < textTime;
}
if (hdr.formatVersion == formatVersion) {
size_t wordsToRead = 0;
BinaryIO::BinaryRead(file, wordsToRead);
if (allowList) m_ApprovedWords.reserve(wordsToRead);
else m_DeniedWords.reserve(wordsToRead);
size_t word = 0;
for (size_t i = 0; i < wordsToRead; ++i) {
BinaryIO::BinaryRead(file, word);
if (allowList) m_ApprovedWords.push_back(word);
else m_DeniedWords.push_back(word);
if (text && (m_DontGenerateDCF || stale)) {
m_Lists.denied = BlockListFromText(*text);
if (m_DontGenerateDCF) {
LOG("Loaded %zu blocked words and phrases from %s", m_Lists.denied.Size(), BLOCK_LIST_TEXT);
return;
}
return true;
} else {
file.close();
return false;
if (WriteFile(BLOCK_LIST_FILE, m_Lists.denied)) {
LOG("Built %s from %s (%zu words and phrases)", BLOCK_LIST_FILE, BLOCK_LIST_TEXT, m_Lists.denied.Size());
} else {
LOG("Could not write %s", BLOCK_LIST_FILE);
}
return;
}
}
return false;
auto blocked = ReadFile(BLOCK_LIST_FILE);
switch (blocked.status) {
case eStatus::OK:
m_Lists.denied = std::move(blocked.list);
break;
case eStatus::MISSING:
LOG("No %s: best friends' free chat stops every message. Put the blocked words in %s next to the servers (one word or phrase per line) and start the servers again.",
BLOCK_LIST_FILE, BLOCK_LIST_TEXT);
break;
case eStatus::OLD_FORMAT:
LOG("%s is in the old format (version 2), whose hashes depend on the compiler and platform, so it can't be read; best friends' free chat stops every message. "
"Put the plain word list in %s next to the servers (one word or phrase per line) and start the servers again to rebuild it.",
BLOCK_LIST_FILE, BLOCK_LIST_TEXT);
break;
default:
LOG("%s is %s and can't be read; best friends' free chat stops every message. Rebuild it from %s.", BLOCK_LIST_FILE, StatusText(blocked.status), BLOCK_LIST_TEXT);
break;
}
}
void dChatFilter::ExportWordlistToDCF(const std::string& filepath, bool allowList) {
std::ofstream file(filepath, std::ios::binary | std::ios_base::out);
if (file) {
BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::header));
BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::formatVersion));
BinaryIO::BinaryWrite(file, size_t(allowList ? m_ApprovedWords.size() : m_DeniedWords.size()));
for (size_t word : allowList ? m_ApprovedWords : m_DeniedWords) {
BinaryIO::BinaryWrite(file, word);
}
file.close();
void dChatFilter::ReloadCustomWords() {
m_Lists.customAllowed = {};
m_Lists.customBlocked = {};
for (const auto& word : Database::Get()->GetChatFilterWords()) {
(word.allowed ? m_Lists.customAllowed : m_Lists.customBlocked).AddEntry(ChatFilterWords::NormalizeEntry(word.word));
}
}
std::set<std::pair<uint8_t, uint8_t>> dChatFilter::IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList) {
if (gmLevel > eGameMasterLevel::FORUM_MODERATOR) return { }; //If anything but a forum mod, return true.
if (message.empty()) return { };
if (!allowList && m_DeniedWords.empty()) return { { 0, message.length() } };
std::stringstream sMessage(message);
std::string segment;
std::set<std::pair<uint8_t, uint8_t>> listOfBadSegments;
uint32_t position = 0;
while (std::getline(sMessage, segment, ' ')) {
std::string originalSegment = segment;
segment = NormalizeWord(segment);
size_t hash = CalculateHash(segment);
// Blocked on the dashboard: stopped in every kind of chat
if (m_CustomBlockedWords.contains(hash)) {
listOfBadSegments.emplace(position, originalSegment.length());
position += originalSegment.length() + 1;
continue;
}
if (std::find(m_UserUnapprovedWordCache.begin(), m_UserUnapprovedWordCache.end(), hash) != m_UserUnapprovedWordCache.end() && allowList) {
listOfBadSegments.emplace(position, originalSegment.length());
}
if (std::find(m_ApprovedWords.begin(), m_ApprovedWords.end(), hash) == m_ApprovedWords.end() && !m_CustomAllowedWords.contains(hash) && allowList) {
m_UserUnapprovedWordCache.push_back(hash);
listOfBadSegments.emplace(position, originalSegment.length());
}
if (std::find(m_DeniedWords.begin(), m_DeniedWords.end(), hash) != m_DeniedWords.end() && !allowList) {
m_UserUnapprovedWordCache.push_back(hash);
listOfBadSegments.emplace(position, originalSegment.length());
}
position += originalSegment.length() + 1;
}
return listOfBadSegments;
}
size_t dChatFilter::CalculateHash(const std::string& word) {
std::hash<std::string> hash{};
size_t value = hash(word);
return value;
return ChatFilterWords::CheckMessage(message, allowList, m_Lists);
}

View File

@@ -1,70 +1,52 @@
#pragma once
#include <algorithm>
#include <cctype>
#include <cstdint>
#include <filesystem>
#include <set>
#include <unordered_set>
#include <vector>
#include <string>
#include <vector>
#include "ChatFilterCore.h"
#include "dCommonVars.h"
enum class eGameMasterLevel : uint8_t;
namespace dChatFilterDCF {
static const uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24));
static const uint32_t formatVersion = 2;
struct fileHeader {
uint32_t header;
uint32_t formatVersion;
};
};
class dChatFilter
{
public:
/**
* Loads the allow list (filepath + ".txt", cached as filepath + ".dcf") and the block list (blocklist.dcf next to the
* servers, rebuilt from blocklist.txt there when that file is newer). dontGenerateDCF: read the plain lists only and
* write no .dcf files.
*/
dChatFilter(const std::string& filepath, bool dontGenerateDCF);
~dChatFilter();
~dChatFilter() = default;
void ReadWordlistPlaintext(const std::string& filepath, bool allowList);
bool ReadWordlistDCF(const std::string& filepath, bool allowList);
void ExportWordlistToDCF(const std::string& filepath, bool allowList);
std::set<std::pair<uint8_t, uint8_t>> IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList = true);
// Whether a deny list is loaded (without one, IsSentenceOkay(..., false) refuses every message)
bool HasDenyList() const { return !m_DeniedWords.empty(); }
bool HasDenyList() const { return !m_Lists.denied.Empty(); }
/**
* Load the words staff added on the dashboard (chat_filter_words) again, replacing the ones loaded before.
* Allowed words are accepted in whitelisted chat; blocked words are always stopped, even when a file allows them.
* Allowed words are accepted in whitelisted chat; blocked words and phrases are always stopped, even when a file allows them.
*/
void ReloadCustomWords();
// A word as the filter compares it: lower case, without ! ? ; . ,
static std::string NormalizeWord(std::string word) {
std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; });
std::transform(word.begin(), word.end(), word.begin(), ::tolower); //Transform to lowercase
return word;
}
static std::string NormalizeWord(std::string word) { return ChatFilterWords::NormalizeWord(std::move(word)); }
// A message split into words (at spaces) the way the filter checks it
static std::vector<std::string> Words(const std::string& message) {
std::vector<std::string> words;
size_t start = 0;
while (start <= message.size()) {
const auto end = std::min(message.find(' ', start), message.size());
words.push_back(NormalizeWord(message.substr(start, end - start)));
start = end + 1;
}
for (auto& token : ChatFilterWords::Tokenize(message)) words.push_back(std::move(token.word));
return words;
}
private:
bool m_DontGenerateDCF;
std::vector<size_t> m_DeniedWords;
std::vector<size_t> m_ApprovedWords;
std::vector<size_t> m_UserUnapprovedWordCache;
std::unordered_set<size_t> m_CustomAllowedWords;
std::unordered_set<size_t> m_CustomBlockedWords;
void LoadAllowList(const std::string& filepath);
void LoadBlockList();
//Private functions:
size_t CalculateHash(const std::string& word);
bool m_DontGenerateDCF;
ChatFilterWords::Lists m_Lists;
};

View File

@@ -8,23 +8,26 @@
#include <string>
#include <vector>
#include "dChatFilter.h"
#include "ChatFilterCore.h"
// Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h)
namespace ModerationTools {
constexpr size_t MAX_FILTER_WORD = 64;
// A word staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,); nullopt if it isn't
// one word of 1-64 characters. Pure; unit tested.
// A word or phrase staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,, words joined by
// one space); nullopt if it isn't 1-64 characters or has no word. Pure; unit tested.
inline std::optional<std::string> FilterWord(std::string text) {
text.erase(0, text.find_first_not_of(" \t\r\n"));
text.erase(text.find_last_not_of(" \t\r\n") + 1);
if (text.empty() || text.size() > MAX_FILTER_WORD || text.find_first_of(" \t\r\n") != std::string::npos) return std::nullopt;
auto word = dChatFilter::NormalizeWord(text);
if (word.empty()) return std::nullopt;
return word;
if (text.empty() || text.size() > MAX_FILTER_WORD) return std::nullopt;
auto entry = ChatFilterWords::NormalizeEntry(text);
if (entry.empty()) return std::nullopt;
return entry;
}
// Whether an entry is a phrase (more than one word). Phrases can only be blocked: whitelist chat checks one word at a time.
inline bool IsPhrase(const std::string& entry) { return ChatFilterWords::WordCount(entry) > 1; }
// The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once
inline std::vector<std::string> FileWords(const std::string& text) {
std::vector<std::string> words;
@@ -33,7 +36,7 @@ namespace ModerationTools {
const auto end = std::min(text.find('\n', start), text.size());
auto line = text.substr(start, end - start);
std::erase(line, '\r');
std::transform(line.begin(), line.end(), line.begin(), ::tolower);
line = ChatFilterWords::AsciiLower(std::move(line));
if (!line.empty()) words.push_back(std::move(line));
start = end + 1;
}
@@ -42,28 +45,12 @@ namespace ModerationTools {
return words;
}
// The hashes of a .dcf word list (blocklist.dcf), as dChatFilter::ReadWordlistDCF reads them; nullopt if it isn't one
inline std::optional<std::vector<size_t>> DcfHashes(const std::string& bytes) {
dChatFilterDCF::fileHeader header{};
size_t count = 0;
if (bytes.size() < sizeof(header) + sizeof(count)) return std::nullopt;
std::memcpy(&header, bytes.data(), sizeof(header));
if (header.header != dChatFilterDCF::header || header.formatVersion != dChatFilterDCF::formatVersion) return std::nullopt;
std::memcpy(&count, bytes.data() + sizeof(header), sizeof(count));
const size_t offset = sizeof(header) + sizeof(count);
if (count > (bytes.size() - offset) / sizeof(size_t)) return std::nullopt;
std::vector<size_t> hashes(count);
if (count) std::memcpy(hashes.data(), bytes.data() + offset, count * sizeof(size_t));
return hashes;
}
// A word's hash as the filter stores it (dChatFilter::CalculateHash)
inline size_t WordHash(const std::string& word) { return std::hash<std::string>{}(word); }
// Whether a message contains `word` as one of the words the chat filter checks
inline bool HasFilterWord(const std::string& message, const std::string& word) {
const auto words = dChatFilter::Words(message);
return std::find(words.begin(), words.end(), word) != words.end();
// Whether a message contains a word or phrase (FilterWord) as the chat filter reads it: whole words, in a row, skipping
// pieces that are only punctuation
inline bool HasFilterWord(const std::string& message, const std::string& entry) {
const auto tokens = ChatFilterWords::Tokenize(message);
const auto words = ChatFilterWords::WordCount(entry);
return !ChatFilterWords::FindBlocked(tokens, words, [&entry](const std::string& run) { return run == entry; }).empty();
}
// What the filter decides about one word of a message, and why
@@ -72,6 +59,7 @@ namespace ModerationTools {
std::string word; // as the filter compares it
bool stopped{};
std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file
std::string phrase; // the blocked phrase this word is part of (blocked_here or block_file), when it was a phrase
};
// Where the filter finds its words (callbacks keep this pure; the route reads the files and the database)
@@ -81,33 +69,49 @@ namespace ModerationTools {
std::function<bool(const std::string&)> characterName; // approved character names count as allowed words
std::function<bool(const std::string&)> blockFile; // blocklist.dcf (by hash)
bool blockFileLoaded{};
uint32_t maxWords{ 1 }; // the longest blocked phrase, in words (here or in the file)
};
/**
* Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher
* levels skip the filter). Normal chat (allowList) needs every word allowed; best friends' free chat (!allowList) stops
* only blocked words, or every word when there is no blocked words file. Words are split at spaces as the filter
* splits them. Pure; unit tested.
* levels skip the filter). Blocked words and phrases (here always, the block file's in free chat) are stopped; a phrase
* stops each of its words. Normal chat (allowList) needs every other word allowed, one at a time; best friends' free
* chat stops only blocked ones, or every word when there is no blocked words file. Words are split at spaces as the
* filter splits them (ChatFilterWords::CheckMessage). Pure; unit tested.
*/
inline std::vector<WordVerdict> ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) {
const auto tokens = ChatFilterWords::Tokenize(message);
std::vector<WordVerdict> verdicts;
std::stringstream stream(message);
std::string segment;
while (std::getline(stream, segment, ' ')) {
WordVerdict verdict{ segment, dChatFilter::NormalizeWord(segment) };
const auto here = sources.dashboard(verdict.word);
if (!allowList && !sources.blockFileLoaded) {
for (const auto& token : tokens) verdicts.push_back({ message.substr(token.position, token.length), token.word });
if (!allowList && !sources.blockFileLoaded) {
for (auto& verdict : verdicts) {
verdict.stopped = true;
verdict.reason = "no_block_file";
} else if (here && !*here) {
verdict.stopped = true;
verdict.reason = "blocked_here";
} else if (!allowList) {
verdict.stopped = sources.blockFile(verdict.word);
verdict.reason = verdict.stopped ? "block_file" : "not_in_block_file";
}
return verdicts;
}
const auto blockedHere = [&sources](const std::string& entry) { const auto here = sources.dashboard(entry); return here && !*here; };
const auto matches = ChatFilterWords::FindBlocked(tokens, std::max(sources.maxWords, 1u), [&](const std::string& entry) {
return blockedHere(entry) || (!allowList && sources.blockFile(entry));
});
for (const auto& match : matches) {
const auto reason = blockedHere(match.entry) ? "blocked_here" : "block_file";
for (size_t i = match.first; i <= match.last; i++) {
verdicts[i].stopped = true;
verdicts[i].reason = reason;
if (ChatFilterWords::WordCount(match.entry) > 1) verdicts[i].phrase = match.entry;
}
}
for (auto& verdict : verdicts) {
if (verdict.stopped) continue;
const auto here = sources.dashboard(verdict.word);
if (!allowList) {
verdict.reason = "not_in_block_file";
} else if (sources.allowFile(verdict.word)) {
verdict.reason = "allow_file";
} else if (here) {
} else if (here && *here) {
verdict.reason = "allowed_here";
} else if (sources.characterName(verdict.word)) {
verdict.reason = "character_name";
@@ -115,7 +119,6 @@ namespace ModerationTools {
verdict.stopped = true;
verdict.reason = "not_allowed";
}
verdicts.push_back(std::move(verdict));
}
return verdicts;
}

View File

@@ -31,7 +31,8 @@ namespace {
// The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked
// words (only their hashes) next to the servers
constexpr const char* ALLOW_FILE = "chatplus_en_us.txt";
constexpr const char* BLOCK_FILE = "blocklist.dcf";
constexpr const char* BLOCK_FILE = dChatFilterDCF::BLOCK_LIST_FILE;
constexpr const char* BLOCK_TEXT = dChatFilterDCF::BLOCK_LIST_TEXT;
constexpr uint32_t FILE_WORDS_PAGE = 200;
// Recent chat searched when checking what a word would change
constexpr uint32_t CHECK_MESSAGES = 1000;
@@ -216,11 +217,8 @@ namespace {
return text ? ModerationTools::FileWords(*text) : std::vector<std::string>{};
}
std::optional<std::vector<size_t>> BlockFileHashes() {
std::ifstream in(BLOCK_FILE, std::ios::binary);
if (!in) return std::nullopt;
return ModerationTools::DcfHashes(std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>()));
}
// blocklist.dcf as the servers read it (an old-format file is refused, as they refuse it)
dChatFilterDCF::ParseResult BlockFile() { return dChatFilterDCF::ReadFile(BLOCK_FILE); }
// The dashboard's own lists by word: true allowed, false blocked
std::map<std::string, bool> DashboardWords() {
@@ -258,25 +256,27 @@ namespace {
const auto it = dashboard.find(word);
words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
}
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
uint32_t imported = 0;
for (const auto& word : all) if (dashboard.contains(word)) imported++;
JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched},
{"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported},
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.has_value()}, {"blockTotal", blocked ? blocked->size() : 0} });
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK}, {"blockTotal", blocked.list.Size()},
{"blockMaxWords", blocked.list.maxWords}, {"blockFileStatus", dChatFilterDCF::StatusText(blocked.status)},
{"blockFileOld", blocked.status == dChatFilterDCF::eStatus::OLD_FORMAT}, {"blockText", BLOCK_TEXT} });
});
Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"),
"Where a word stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
"Where a word or phrase stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
[](HTTPReply& reply, const HTTPContext& context) {
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const auto all = AllowFileWords();
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
const auto dashboard = DashboardWords();
const auto it = dashboard.find(*word);
JsonSuccess(reply, { {"word", *word}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
{"inBlockFile", blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(*word)) != blocked->end()},
JsonSuccess(reply, { {"word", *word}, {"phrase", ModerationTools::IsPhrase(*word)}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
{"inBlockFile", blocked.list.Contains(*word)},
{"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
});
@@ -293,7 +293,7 @@ namespace {
for (const auto& word : all) {
// Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only)
const auto filterWord = ModerationTools::FilterWord(word);
if (!filterWord || *filterWord != word || dashboard.contains(word)) continue;
if (!filterWord || *filterWord != word || ModerationTools::IsPhrase(word) || dashboard.contains(word)) continue;
Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now });
added++;
}
@@ -305,13 +305,17 @@ namespace {
});
Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"),
"Allow or block a word (or move it to the other list); running worlds pick it up at once. Body: {word, allowed: bool}",
"Allow or block a word, or block a phrase (or move it to the other list); running worlds pick it up at once. Phrases can't be allowed: "
"whitelist chat checks each word on its own, as the client does. Body: {word, allowed: bool}",
[](HTTPReply& reply, const HTTPContext& context) {
const auto body = ParseBody(context);
if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON");
const auto word = ModerationTools::FilterWord(body->value("word", ""));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = body->value("allowed", false);
if (allowed && ModerationTools::IsPhrase(*word)) {
return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead");
}
Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast<int64_t>(std::time(nullptr)) });
Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat");
BroadcastTableChanged("chat_filter");
@@ -345,12 +349,11 @@ namespace {
if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters");
const bool allowList = QueryValue(context.queryString, "chat") != "free";
const auto all = AllowFileWords();
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
const auto dashboard = DashboardWords();
std::set<std::string> names;
for (auto name : Database::Get()->GetApprovedCharacterNames()) {
std::transform(name.begin(), name.end(), name.begin(), ::tolower);
names.insert(std::move(name));
names.insert(ChatFilterWords::AsciiLower(std::move(name)));
}
ModerationTools::WordSources sources;
sources.dashboard = [&dashboard](const std::string& word) -> std::optional<bool> {
@@ -359,18 +362,18 @@ namespace {
};
sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); };
sources.characterName = [&names](const std::string& word) { return names.contains(word); };
sources.blockFile = [&blocked](const std::string& word) {
return blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(word)) != blocked->end();
};
sources.blockFileLoaded = blocked && !blocked->empty();
sources.blockFile = [&blocked](const std::string& entry) { return blocked.list.Contains(entry); };
sources.blockFileLoaded = !blocked.list.Empty();
sources.maxWords = blocked.list.maxWords;
for (const auto& [entry, allowed] : dashboard) if (!allowed) sources.maxWords = std::max(sources.maxWords, ChatFilterWords::WordCount(entry));
nlohmann::json words = nlohmann::json::array();
bool stopped = false;
for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) {
stopped |= verdict.stopped;
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason} });
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason}, {"phrase", verdict.phrase} });
}
JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words},
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.has_value()} });
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK} });
});
Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"),
@@ -379,7 +382,7 @@ namespace {
[](HTTPReply& reply, const HTTPContext& context) {
if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission");
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = QueryValue(context.queryString, "allowed") == "1";
IChatLog::ChatQuery query;
query.search = *word;

View File

@@ -39,6 +39,7 @@ set(DCOMMONTEST_SOURCES
"FdbReaderTests.cpp"
"FdbSnapshotTests.cpp"
"WorldFileWatchTests.cpp"
"ChatFilterCoreTests.cpp"
)
add_subdirectory(dEnumsTests)
@@ -59,6 +60,8 @@ endif()
target_link_libraries(dCommonTests ${COMMON_LIBRARIES} MD5 GTest::gtest_main)
# SpareBackoff.h (header only)
target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dMasterServer")
# ChatFilterCore.h (header only)
target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dChatFilter")
# Copy test files to testing directory
add_subdirectory(TestBitStreams)

View File

@@ -0,0 +1,149 @@
#include <gtest/gtest.h>
#include <string>
#include <vector>
#include "ChatFilterCore.h"
using namespace ChatFilterWords;
using Spans = std::set<std::pair<uint8_t, uint8_t>>;
// The hash is a compile-time constant: the same value on every compiler, standard library and platform
static_assert(Hash("") == 0xcbf29ce484222325ULL);
static_assert(Hash("a") == 0xaf63dc4c8601ec8cULL);
static_assert(Hash("foobar") == 0x85944171f73967e8ULL);
TEST(ChatFilterCoreTest, HashIsFnv1a64) {
// The standard FNV-1a 64 test vectors
EXPECT_EQ(Hash(""), 0xcbf29ce484222325ULL);
EXPECT_EQ(Hash("a"), 0xaf63dc4c8601ec8cULL);
EXPECT_EQ(Hash("foobar"), 0x85944171f73967e8ULL);
// Words of the client's chatplus_en_us.txt, as the filter hashes them (lower case)
EXPECT_EQ(Hash("hello"), 0xa430d84680aabd0bULL);
EXPECT_EQ(Hash("brick"), 0xf9236d1e24832c9aULL);
EXPECT_EQ(Hash(AsciiLower("Brick")), Hash("brick"));
// A phrase is its words joined by one space
EXPECT_EQ(Hash("bad phrase"), 0x72e9fce49c1b0875ULL);
// Bytes, not chars: a high byte hashes as 0x80-0xFF whether char is signed or not
EXPECT_EQ(Hash("\xC3\xA9"), 0x0ac21707b7181e01ULL);
}
TEST(ChatFilterCoreTest, NormalizeEntry) {
EXPECT_EQ(NormalizeWord("Hello!?"), "hello");
EXPECT_EQ(NormalizeEntry(" Bad \t PHRASE! "), "bad phrase");
EXPECT_EQ(NormalizeEntry("word , here"), "word here");
EXPECT_EQ(NormalizeEntry("..."), "");
EXPECT_EQ(WordCount("bad phrase here"), 3u);
EXPECT_EQ(WordCount("word"), 1u);
EXPECT_EQ(WordCount(""), 0u);
// ASCII only: other bytes are left alone whatever the locale
EXPECT_EQ(AsciiLower("\xC3\x89Z"), "\xC3\x89z");
}
TEST(ChatFilterCoreTest, DcfBytesAreFixed) {
WordList list;
list.AddEntry("a");
list.AddEntry("bad phrase");
const auto bytes = dChatFilterDCF::Serialize(list);
// magic DCFB, version 3, 2 words at most, 2 hashes, then the hashes sorted, all little-endian
const std::string expected(
"DCFB" "\x03\x00\x00\x00" "\x02\x00\x00\x00" "\x02\x00\x00\x00\x00\x00\x00\x00"
"\x75\x08\x1b\x9c\xe4\xfc\xe9\x72" "\x8c\xec\x01\x86\x4c\xdc\x63\xaf", 4 + 4 + 4 + 8 + 16);
EXPECT_EQ(bytes, expected);
const auto parsed = dChatFilterDCF::Parse(bytes);
ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK);
EXPECT_EQ(parsed.list.maxWords, 2u);
EXPECT_EQ(parsed.list.hashes, list.hashes);
}
TEST(ChatFilterCoreTest, DcfRejectsOldAndBadFiles) {
// Version 2 (std::hash, platform dependent) is refused, not guessed at
const std::string old("DCFB" "\x02\x00\x00\x00" "\x01\x00\x00\x00\x00\x00\x00\x00" "\x11\x22\x33\x44\x55\x66\x77\x88", 24);
EXPECT_EQ(dChatFilterDCF::Parse(old).status, dChatFilterDCF::eStatus::OLD_FORMAT);
EXPECT_EQ(dChatFilterDCF::Parse("XCFB\x03\x00\x00\x00").status, dChatFilterDCF::eStatus::NOT_DCF);
EXPECT_EQ(dChatFilterDCF::Parse(std::string("DCFB\x09\x00\x00\x00", 8)).status, dChatFilterDCF::eStatus::UNKNOWN);
WordList list;
list.AddEntry("word");
auto bytes = dChatFilterDCF::Serialize(list);
bytes.pop_back();
EXPECT_EQ(dChatFilterDCF::Parse(bytes).status, dChatFilterDCF::eStatus::TRUNCATED);
EXPECT_EQ(dChatFilterDCF::ReadFile("no such blocklist.dcf").status, dChatFilterDCF::eStatus::MISSING);
}
TEST(ChatFilterCoreTest, BlockListFromText) {
const auto list = dChatFilterDCF::BlockListFromText("Badword\r\n\r\n Very BAD phrase!\nbadword\n...\n");
EXPECT_EQ(list.Size(), 2u);
EXPECT_TRUE(list.Contains("badword"));
EXPECT_TRUE(list.Contains("very bad phrase"));
EXPECT_EQ(list.maxWords, 3u);
// Written and read back, the same list
const auto parsed = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(list));
ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK);
EXPECT_EQ(parsed.list.hashes, list.hashes);
EXPECT_EQ(parsed.list.maxWords, 3u);
}
namespace {
Lists FreeChatLists(std::string_view blockText) {
Lists lists;
// Through the file format, as the servers load it
lists.denied = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(dChatFilterDCF::BlockListFromText(blockText))).list;
lists.approved = dChatFilterDCF::AllowListFromText("hello\nthere\nfriend\nbad\nphrase\n");
return lists;
}
}
TEST(ChatFilterCoreTest, BlockedWordFromFileIsStopped) {
const auto lists = FreeChatLists("badword\n");
EXPECT_EQ(CheckMessage("hello badword there", false, lists), (Spans{ { 6, 7 } }));
EXPECT_EQ(CheckMessage("hello BadWord!", false, lists), (Spans{ { 6, 8 } }));
EXPECT_TRUE(CheckMessage("hello there", false, lists).empty());
// Whitelist chat doesn't use the block list: the word is simply not allowed there
EXPECT_EQ(CheckMessage("hello badword", true, lists), (Spans{ { 6, 7 } }));
// No block list: free chat stops everything
EXPECT_EQ(CheckMessage("hello", false, Lists{}), (Spans{ { 0, 5 } }));
}
TEST(ChatFilterCoreTest, PhrasesAtStartMiddleEnd) {
const auto lists = FreeChatLists("bad phrase\n");
EXPECT_EQ(CheckMessage("bad phrase hello", false, lists), (Spans{ { 0, 10 } }));
EXPECT_EQ(CheckMessage("hello bad phrase there", false, lists), (Spans{ { 6, 10 } }));
EXPECT_EQ(CheckMessage("hello bad phrase", false, lists), (Spans{ { 6, 10 } }));
// The words alone are fine
EXPECT_TRUE(CheckMessage("bad hello phrase", false, lists).empty());
EXPECT_TRUE(CheckMessage("phrase bad", false, lists).empty());
}
TEST(ChatFilterCoreTest, PhrasesWithExtraSpacesAndPunctuation) {
const auto lists = FreeChatLists("bad phrase\n");
// Two spaces: the span covers both words and the gap
EXPECT_EQ(CheckMessage("hi Bad Phrase!", false, lists), (Spans{ { 4, 12 } }));
// A piece that is only punctuation is skipped
EXPECT_EQ(CheckMessage("bad ... phrase", false, lists), (Spans{ { 0, 14 } }));
EXPECT_EQ(CheckMessage("bad, phrase.", false, lists), (Spans{ { 0, 12 } }));
}
TEST(ChatFilterCoreTest, PhrasesOverlapWithWords) {
const auto lists = FreeChatLists("bad phrase\nphrase here\nbadword\nfriend\n");
// Two phrases sharing a word become one span
EXPECT_EQ(CheckMessage("a bad phrase here b", false, lists), (Spans{ { 2, 15 } }));
// A blocked word inside a blocked phrase: one span for the phrase
EXPECT_EQ(CheckMessage("bad phrase", false, FreeChatLists("bad phrase\nphrase\n")), (Spans{ { 0, 10 } }));
// A blocked word right after a phrase: its own span
EXPECT_EQ(CheckMessage("bad phrase badword", false, lists), (Spans{ { 0, 10 }, { 11, 7 } }));
// Longest match wins where it starts
EXPECT_EQ(CheckMessage("friend bad phrase", false, FreeChatLists("friend\nfriend bad\nbad phrase\n")), (Spans{ { 0, 17 } }));
}
TEST(ChatFilterCoreTest, DashboardPhrasesInWhitelistChat) {
auto lists = FreeChatLists("");
lists.customBlocked.AddEntry(NormalizeEntry("Bad Phrase"));
// Blocked on the dashboard: stopped in whitelist chat too, even though each word is allowed
EXPECT_EQ(CheckMessage("hello bad phrase", true, lists), (Spans{ { 6, 10 } }));
EXPECT_TRUE(CheckMessage("hello bad there phrase", true, lists).empty());
// Whitelist chat checks one word at a time, as the client does
EXPECT_EQ(CheckMessage("hello stranger", true, lists), (Spans{ { 6, 8 } }));
lists.customAllowed.AddEntry("stranger");
EXPECT_TRUE(CheckMessage("hello stranger", true, lists).empty());
}

View File

@@ -43,7 +43,10 @@ TEST(ChatFilterWordsTest, FilterWord) {
EXPECT_EQ(ModerationTools::FilterWord(" Hello! "), "hello");
EXPECT_EQ(ModerationTools::FilterWord("W.o,r;d?"), "word");
EXPECT_FALSE(ModerationTools::FilterWord(""));
EXPECT_FALSE(ModerationTools::FilterWord("two words"));
// Phrases: words normalized and joined by one space
EXPECT_EQ(ModerationTools::FilterWord(" Two Words! "), "two words");
EXPECT_TRUE(ModerationTools::IsPhrase("two words"));
EXPECT_FALSE(ModerationTools::IsPhrase("word"));
EXPECT_FALSE(ModerationTools::FilterWord("!!!"));
EXPECT_FALSE(ModerationTools::FilterWord(std::string(65, 'a')));
}
@@ -53,6 +56,11 @@ TEST(ChatFilterWordsTest, HasFilterWord) {
EXPECT_TRUE(ModerationTools::HasFilterWord("bad", "bad"));
EXPECT_FALSE(ModerationTools::HasFilterWord("badger badminton", "bad"));
EXPECT_FALSE(ModerationTools::HasFilterWord("", "bad"));
// Phrases: the words in a row, whatever the spaces and punctuation between them
EXPECT_TRUE(ModerationTools::HasFilterWord("well, Bad Phrase!", "bad phrase"));
EXPECT_TRUE(ModerationTools::HasFilterWord("bad ... phrase", "bad phrase"));
EXPECT_FALSE(ModerationTools::HasFilterWord("bad other phrase", "bad phrase"));
EXPECT_FALSE(ModerationTools::HasFilterWord("phrase bad", "bad phrase"));
}
TEST(ChatFilterWordsTest, FileWords) {
@@ -61,27 +69,6 @@ TEST(ChatFilterWordsTest, FileWords) {
ASSERT_TRUE(ModerationTools::FileWords("").empty());
}
TEST(ChatFilterWordsTest, DcfHashes) {
const std::vector<size_t> hashes{ ModerationTools::WordHash("badword"), 42 };
std::string bytes(sizeof(dChatFilterDCF::fileHeader) + sizeof(size_t) * (hashes.size() + 1), '\0');
const dChatFilterDCF::fileHeader header{ dChatFilterDCF::header, dChatFilterDCF::formatVersion };
const size_t count = hashes.size();
std::memcpy(bytes.data(), &header, sizeof(header));
std::memcpy(bytes.data() + sizeof(header), &count, sizeof(count));
std::memcpy(bytes.data() + sizeof(header) + sizeof(count), hashes.data(), sizeof(size_t) * count);
ASSERT_EQ(ModerationTools::DcfHashes(bytes), hashes);
// Wrong header, other version, or fewer hashes than it says
auto wrong = bytes;
wrong[0] = 'X';
ASSERT_FALSE(ModerationTools::DcfHashes(wrong).has_value());
auto version = bytes;
version[sizeof(uint32_t)] = 9;
ASSERT_FALSE(ModerationTools::DcfHashes(version).has_value());
ASSERT_FALSE(ModerationTools::DcfHashes(bytes.substr(0, bytes.size() - sizeof(size_t) * 2)).has_value());
ASSERT_FALSE(ModerationTools::DcfHashes("DCFB").has_value());
}
namespace {
ModerationTools::WordSources Sources(bool blockFileLoaded = true) {
ModerationTools::WordSources sources;
@@ -121,3 +108,22 @@ TEST(ChatFilterWordsTest, ExplainFreeChat) {
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("zzz darn", false, Sources(false))),
(std::vector<std::string>{ "x no_block_file", "x no_block_file" }));
}
TEST(ChatFilterWordsTest, ExplainPhrases) {
auto sources = Sources();
sources.dashboard = [](const std::string& w) -> std::optional<bool> {
if (w == "no way") return false;
return std::nullopt;
};
sources.allowFile = [](const std::string& w) { return w == "hello" || w == "no" || w == "way" || w == "rude"; };
sources.blockFile = [](const std::string& w) { return w == "very rude"; };
sources.maxWords = 2;
// A phrase blocked here stops each of its words (and the empty piece between two spaces inside it), in normal chat too
auto verdicts = ModerationTools::ExplainMessage("hello No way!", true, sources);
EXPECT_EQ(Reasons(verdicts), (std::vector<std::string>{ "ok allow_file", "x blocked_here", "x blocked_here", "x blocked_here" }));
EXPECT_EQ(verdicts[1].phrase, "no way");
EXPECT_EQ(verdicts[3].text, "way!");
// The block file's phrases only in free chat
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("very rude", false, sources)), (std::vector<std::string>{ "x block_file", "x block_file" }));
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("rude very", false, sources)), (std::vector<std::string>{ "ok not_in_block_file", "ok not_in_block_file" }));
}