Merge exp/dcf-hash: portable chat filter hash, blocklist from text, blocked phrases

This commit is contained in:
Aaron Kimbrell
2026-09-30 08:06:05 -05:00
14 changed files with 759 additions and 289 deletions

View File

@@ -139,6 +139,10 @@ locally. See [docs/UgcServer.md](docs/UgcServer.md).
* **World hot reload:** worlds report the zone files they loaded (`.luz`, `.lvl`, triggers, terrain, navmesh); when one * **World hot reload:** worlds report the zone files they loaded (`.luz`, `.lvl`, triggers, terrain, navmesh); when one
changes on disk, or on `/reloadworld` or the dashboard's Reload, master replaces those instances with new ones and changes on disk, or on `/reloadworld` or the dashboard's Reload, master replaces those instances with new ones and
moves their players over; properties are kept until empty instead ([docs/WorldHotReload.md](docs/WorldHotReload.md)). moves their players over; properties are kept until empty instead ([docs/WorldHotReload.md](docs/WorldHotReload.md)).
* **Chat filter:** the block list and the allowed words cache (`.dcf`) are hashed with 64-bit FNV-1a, so a list works
on every platform (before, `std::hash` values made on one system never matched on another); old files are refused
with a log line. Servers build `blocklist.dcf` from a plain `blocklist.txt` next to them, and blocked entries can be
phrases. See "Block list file" in [docs/Dashboard.md](docs/Dashboard.md).
* The chat server's old web API is removed; the dashboard's API covers online players, teams and announcements. * The chat server's old web API is removed; the dashboard's API covers online players, teams and announcements.
## License ## License
@@ -377,7 +381,7 @@ All listed files are required for a server to start.
* masterconfig.ini * masterconfig.ini
* WorldServer(.exe) * WorldServer(.exe)
* worldconfig.ini * worldconfig.ini
* blocklist.dcf * blocklist.dcf (or blocklist.txt, one blocked word or phrase per line, which the servers build it from)
* migrations * migrations
* vanity * vanity
* navmeshes * navmeshes

View File

@@ -0,0 +1,342 @@
#pragma once
#include <algorithm>
#include <cstdint>
#include <filesystem>
#include <fstream>
#include <functional>
#include <iterator>
#include <optional>
#include <random>
#include <set>
#include <string>
#include <string_view>
#include <unordered_set>
#include <utility>
#include <vector>
/**
* The chat filter's words, without the server around them (pure, unit tested; dChatFilter and the dashboard both use it).
*
* A word is compared lower case (ASCII only, so every platform agrees) without ! ? ; . , and a phrase is its words
* joined by one space. Entries are stored and compared by ChatFilterWords::Hash: 64-bit FNV-1a over the entry's bytes,
* the same on every compiler, standard library and platform.
*/
namespace ChatFilterWords {
// ASCII lower case; other bytes (UTF-8) stay as they are
inline std::string AsciiLower(std::string text) {
for (auto& c : text) if (c >= 'A' && c <= 'Z') c = static_cast<char>(c - 'A' + 'a');
return text;
}
// A word as the filter compares it: lower case, without ! ? ; . ,
inline std::string NormalizeWord(std::string word) {
std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; });
return AsciiLower(std::move(word));
}
// A word or phrase as the filter stores it: each word normalized, words that end up empty dropped, joined by one space
inline std::string NormalizeEntry(std::string_view text) {
std::string entry;
size_t start = 0;
while (start < text.size()) {
auto end = text.find_first_of(" \t\r\n", start);
if (end == std::string_view::npos) end = text.size();
const auto word = NormalizeWord(std::string(text.substr(start, end - start)));
if (!word.empty()) {
if (!entry.empty()) entry += ' ';
entry += word;
}
start = end + 1;
}
return entry;
}
// How many words an entry has (1 for a word, 0 for an empty entry)
inline uint32_t WordCount(std::string_view entry) {
return entry.empty() ? 0 : static_cast<uint32_t>(std::count(entry.begin(), entry.end(), ' ')) + 1;
}
// 64-bit FNV-1a: offset basis 0xcbf29ce484222325, prime 0x100000001b3, one byte at a time
constexpr uint64_t Hash(std::string_view entry) {
uint64_t hash = 0xcbf29ce484222325ULL;
for (const char c : entry) {
hash ^= static_cast<uint8_t>(c);
hash *= 0x100000001b3ULL;
}
return hash;
}
// One piece of a message between spaces: where it is in the message and the word the filter compares
struct Token {
uint32_t position{};
uint32_t length{};
std::string word;
};
// A message split at each space, the way the filter checks it: two spaces in a row give an empty piece, a trailing space none
inline std::vector<Token> Tokenize(std::string_view message) {
std::vector<Token> tokens;
size_t start = 0;
while (start < message.size()) {
auto end = message.find(' ', start);
if (end == std::string_view::npos) end = message.size();
tokens.push_back({ static_cast<uint32_t>(start), static_cast<uint32_t>(end - start), NormalizeWord(std::string(message.substr(start, end - start))) });
start = end + 1;
}
return tokens;
}
// A run of tokens [first, last] that matched a blocked entry, and the longest entry that matched where the run starts
struct Match {
size_t first{};
size_t last{};
std::string entry;
};
/**
* The blocked words and phrases in a message: at each word, the longest run of up to maxWords consecutive words
* (tokens with no word are skipped) whose entry isBlocked accepts. Runs that share a word are merged into one.
*/
inline std::vector<Match> FindBlocked(const std::vector<Token>& tokens, uint32_t maxWords, const std::function<bool(const std::string&)>& isBlocked) {
std::vector<size_t> words;
for (size_t i = 0; i < tokens.size(); i++) if (!tokens[i].word.empty()) words.push_back(i);
std::vector<Match> matches;
for (size_t i = 0; i < words.size(); i++) {
std::string entry;
std::string best;
size_t bestLength = 0;
for (size_t n = 1; n <= maxWords && i + n <= words.size(); n++) {
if (n > 1) entry += ' ';
entry += tokens[words[i + n - 1]].word;
if (isBlocked(entry)) {
best = entry;
bestLength = n;
}
}
if (bestLength == 0) continue;
const size_t first = words[i];
const size_t last = words[i + bestLength - 1];
if (!matches.empty() && first <= matches.back().last) {
matches.back().last = std::max(matches.back().last, last);
} else {
matches.push_back({ first, last, std::move(best) });
}
}
return matches;
}
// Hashes of words or phrases, and the most words any of them has
struct WordList {
std::unordered_set<uint64_t> hashes;
uint32_t maxWords{};
void AddEntry(std::string_view entry) {
if (entry.empty()) return;
hashes.insert(Hash(entry));
maxWords = std::max(maxWords, WordCount(entry));
}
bool Contains(std::string_view entry) const { return hashes.contains(Hash(entry)); }
bool Empty() const { return hashes.empty(); }
size_t Size() const { return hashes.size(); }
};
// Everything the filter checks a message against
struct Lists {
WordList approved; // chatplus_en_us.txt and approved character names: whitelist chat, one word at a time
WordList denied; // blocklist.dcf: best friends' free chat
WordList customAllowed; // allowed on the dashboard
WordList customBlocked; // blocked on the dashboard: stopped in every kind of chat
};
/**
* The pieces of a message the filter stops, as (position, length) in the message. Blocked words and phrases (the
* dashboard's always, blocklist.dcf's in free chat) are stopped as one span each. In whitelist chat (allowList) every
* other piece must be an allowed word, one at a time, as the client checks words. In free chat without a block list
* the whole message is stopped.
*/
inline std::set<std::pair<uint8_t, uint8_t>> CheckMessage(std::string_view message, bool allowList, const Lists& lists) {
if (message.empty()) return {};
if (!allowList && lists.denied.Empty()) return { { 0, static_cast<uint8_t>(message.length()) } };
const auto tokens = Tokenize(message);
const uint32_t maxWords = std::max(lists.customBlocked.maxWords, allowList ? 0u : lists.denied.maxWords);
const auto matches = FindBlocked(tokens, maxWords, [&](const std::string& entry) {
return lists.customBlocked.Contains(entry) || (!allowList && lists.denied.Contains(entry));
});
std::set<std::pair<uint8_t, uint8_t>> bad;
std::vector<bool> covered(tokens.size(), false);
for (const auto& match : matches) {
const auto& first = tokens[match.first];
const auto& last = tokens[match.last];
bad.emplace(static_cast<uint8_t>(first.position), static_cast<uint8_t>(last.position + last.length - first.position));
for (size_t i = match.first; i <= match.last; i++) covered[i] = true;
}
if (allowList) {
for (size_t i = 0; i < tokens.size(); i++) {
if (covered[i]) continue;
const auto hash = Hash(tokens[i].word);
if (!lists.approved.hashes.contains(hash) && !lists.customAllowed.hashes.contains(hash)) {
bad.emplace(static_cast<uint8_t>(tokens[i].position), static_cast<uint8_t>(tokens[i].length));
}
}
}
return bad;
}
}
/**
* The chat filter's word list files (.dcf). These are DLU's own files: the client reads no .dcf and hashes no chat
* words (it keeps its lists as plain text). Layout, little-endian:
* uint32 magic 'DCFB' | uint32 version (3) | uint32 most words in one entry | uint64 count | count x uint64 ChatFilterWords::Hash
* Version 2 (older DLU) stored std::hash values, which differ between compilers and platforms; those can't be read.
*/
namespace dChatFilterDCF {
constexpr uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24));
constexpr uint32_t formatVersion = 3;
constexpr uint32_t oldFormatVersion = 2;
constexpr size_t headerSize = 4 + 4 + 4 + 8;
// The block list's plain source (one word or phrase per line) and the .dcf built from it, next to the servers
constexpr const char* BLOCK_LIST_TEXT = "blocklist.txt";
constexpr const char* BLOCK_LIST_FILE = "blocklist.dcf";
enum class eStatus : uint8_t {
OK,
MISSING, // no file
NOT_DCF, // not a .dcf file
OLD_FORMAT, // version 2: platform-dependent hashes, rebuild it from the plain word list
UNKNOWN, // a version this server doesn't know
TRUNCATED, // shorter than its count says
};
inline const char* StatusText(eStatus status) {
switch (status) {
case eStatus::OK: return "ok";
case eStatus::MISSING: return "missing";
case eStatus::NOT_DCF: return "not a .dcf file";
case eStatus::OLD_FORMAT: return "old format (version 2, platform-dependent hashes)";
case eStatus::UNKNOWN: return "unknown version";
case eStatus::TRUNCATED: return "truncated";
}
return "unknown";
}
struct ParseResult {
eStatus status{ eStatus::NOT_DCF };
uint32_t version{};
ChatFilterWords::WordList list;
};
namespace detail {
inline uint64_t ReadLE(std::string_view bytes, size_t offset, size_t size) {
uint64_t value = 0;
for (size_t i = 0; i < size; i++) value |= static_cast<uint64_t>(static_cast<uint8_t>(bytes[offset + i])) << (8 * i);
return value;
}
inline void WriteLE(std::string& out, uint64_t value, size_t size) {
for (size_t i = 0; i < size; i++) out.push_back(static_cast<char>((value >> (8 * i)) & 0xFF));
}
}
inline ParseResult Parse(std::string_view bytes) {
ParseResult result;
if (bytes.size() < 8 || detail::ReadLE(bytes, 0, 4) != header) return result;
result.version = static_cast<uint32_t>(detail::ReadLE(bytes, 4, 4));
if (result.version == oldFormatVersion) {
result.status = eStatus::OLD_FORMAT;
return result;
}
if (result.version != formatVersion) {
result.status = eStatus::UNKNOWN;
return result;
}
result.status = eStatus::TRUNCATED;
if (bytes.size() < headerSize) return result;
const auto maxWords = static_cast<uint32_t>(detail::ReadLE(bytes, 8, 4));
const auto count = detail::ReadLE(bytes, 12, 8);
if (count > (bytes.size() - headerSize) / 8) return result;
result.list.maxWords = maxWords;
result.list.hashes.reserve(count);
for (uint64_t i = 0; i < count; i++) result.list.hashes.insert(detail::ReadLE(bytes, headerSize + i * 8, 8));
result.status = eStatus::OK;
return result;
}
// A list as a .dcf file; hashes sorted, so the same words always give the same bytes
inline std::string Serialize(const ChatFilterWords::WordList& list) {
std::vector<uint64_t> hashes(list.hashes.begin(), list.hashes.end());
std::sort(hashes.begin(), hashes.end());
std::string out;
out.reserve(headerSize + hashes.size() * 8);
detail::WriteLE(out, header, 4);
detail::WriteLE(out, formatVersion, 4);
detail::WriteLE(out, list.maxWords, 4);
detail::WriteLE(out, hashes.size(), 8);
for (const auto hash : hashes) detail::WriteLE(out, hash, 8);
return out;
}
// A plain block list: one word or phrase per line, normalized (ChatFilterWords::NormalizeEntry); empty lines skipped
inline ChatFilterWords::WordList BlockListFromText(std::string_view text) {
ChatFilterWords::WordList list;
size_t start = 0;
while (start < text.size()) {
auto end = text.find('\n', start);
if (end == std::string_view::npos) end = text.size();
list.AddEntry(ChatFilterWords::NormalizeEntry(text.substr(start, end - start)));
start = end + 1;
}
return list;
}
// A plain allow list (chatplus_en_us.txt): one word per line, lower case, compared whole (as the filter always has)
inline ChatFilterWords::WordList AllowListFromText(std::string_view text) {
ChatFilterWords::WordList list;
size_t start = 0;
while (start < text.size()) {
auto end = text.find('\n', start);
if (end == std::string_view::npos) end = text.size();
std::string line(text.substr(start, end - start));
std::erase(line, '\r');
line = ChatFilterWords::AsciiLower(std::move(line));
list.hashes.insert(ChatFilterWords::Hash(line));
list.maxWords = std::max(list.maxWords, 1u);
start = end + 1;
}
return list;
}
inline std::optional<std::string> ReadBytes(const std::filesystem::path& path) {
std::ifstream in(path, std::ios::binary);
if (!in) return std::nullopt;
return std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>());
}
inline ParseResult ReadFile(const std::filesystem::path& path) {
const auto bytes = ReadBytes(path);
if (!bytes) return { eStatus::MISSING };
return Parse(*bytes);
}
// Writes the file whole or not at all (a temporary file renamed over it), so servers starting together don't read half a file
inline bool WriteFile(const std::filesystem::path& path, const ChatFilterWords::WordList& list) {
auto temp = path;
temp += "." + std::to_string(std::random_device{}()) + ".tmp";
{
std::ofstream out(temp, std::ios::binary | std::ios::trunc);
if (!out) return false;
const auto bytes = Serialize(list);
out.write(bytes.data(), static_cast<std::streamsize>(bytes.size()));
if (!out) return false;
}
std::error_code error;
std::filesystem::rename(temp, path, error);
if (!error) return true;
std::filesystem::remove(temp, error);
return false;
}
}

View File

@@ -1,15 +1,8 @@
#include "dChatFilter.h" #include "dChatFilter.h"
#include "BinaryIO.h"
#include <fstream>
#include <string>
#include <functional>
#include <algorithm>
#include <sstream>
#include <regex>
#include "dCommonVars.h" #include <system_error>
#include "Logger.h" #include "Logger.h"
#include "dConfig.h"
#include "Database.h" #include "Database.h"
#include "Game.h" #include "Game.h"
#include "eGameMasterLevel.h" #include "eGameMasterLevel.h"
@@ -19,154 +12,96 @@ using namespace dChatFilterDCF;
dChatFilter::dChatFilter(const std::string& filepath, bool dontGenerateDCF) { dChatFilter::dChatFilter(const std::string& filepath, bool dontGenerateDCF) {
m_DontGenerateDCF = dontGenerateDCF; m_DontGenerateDCF = dontGenerateDCF;
if (!BinaryIO::DoesFileExist(filepath + ".dcf") || m_DontGenerateDCF) { LoadAllowList(filepath);
ReadWordlistPlaintext(filepath + ".txt", true); LoadBlockList();
if (!m_DontGenerateDCF) ExportWordlistToDCF(filepath + ".dcf", true);
} else if (!ReadWordlistDCF(filepath + ".dcf", true)) {
ReadWordlistPlaintext(filepath + ".txt", true);
ExportWordlistToDCF(filepath + ".dcf", true);
}
if (BinaryIO::DoesFileExist("blocklist.dcf")) { // Approved character names count as allowed words
ReadWordlistDCF("blocklist.dcf", false); for (const auto& name : Database::Get()->GetApprovedCharacterNames()) {
} m_Lists.approved.hashes.insert(ChatFilterWords::Hash(ChatFilterWords::AsciiLower(name)));
//Read player names that are ok as well:
auto approvedNames = Database::Get()->GetApprovedCharacterNames();
for (auto& name : approvedNames) {
std::transform(name.begin(), name.end(), name.begin(), ::tolower); //Transform to lowercase
m_ApprovedWords.push_back(CalculateHash(name));
} }
ReloadCustomWords(); ReloadCustomWords();
} }
void dChatFilter::ReloadCustomWords() { void dChatFilter::LoadAllowList(const std::string& filepath) {
m_CustomAllowedWords.clear(); const std::string dcf = filepath + ".dcf";
m_CustomBlockedWords.clear(); const std::string txt = filepath + ".txt";
// Words remembered as not allowed may be allowed now if (!m_DontGenerateDCF) {
m_UserUnapprovedWordCache.clear(); auto cached = ReadFile(dcf);
for (const auto& word : Database::Get()->GetChatFilterWords()) { if (cached.status == eStatus::OK) {
(word.allowed ? m_CustomAllowedWords : m_CustomBlockedWords).insert(CalculateHash(NormalizeWord(word.word))); m_Lists.approved = std::move(cached.list);
return;
} }
if (cached.status != eStatus::MISSING) LOG("%s is %s; building it again from %s", dcf.c_str(), StatusText(cached.status), txt.c_str());
} }
dChatFilter::~dChatFilter() { const auto text = ReadBytes(txt);
m_ApprovedWords.clear(); if (!text) {
m_DeniedWords.clear(); LOG("Could not read the chat filter's allowed words (%s)", txt.c_str());
return;
}
m_Lists.approved = AllowListFromText(*text);
if (!m_DontGenerateDCF && !WriteFile(dcf, m_Lists.approved)) LOG("Could not write %s", dcf.c_str());
} }
void dChatFilter::ReadWordlistPlaintext(const std::string& filepath, bool allowList) { void dChatFilter::LoadBlockList() {
std::ifstream file(filepath); std::error_code error;
if (file) { const bool hasText = std::filesystem::exists(BLOCK_LIST_TEXT, error);
std::string line; if (hasText) {
while (std::getline(file, line)) { const auto text = ReadBytes(BLOCK_LIST_TEXT);
line.erase(std::remove(line.begin(), line.end(), '\r'), line.end()); const auto existing = ReadFile(BLOCK_LIST_FILE);
std::transform(line.begin(), line.end(), line.begin(), ::tolower); //Transform to lowercase // Rebuilt when the .dcf is missing, unreadable or older than the text
if (allowList) m_ApprovedWords.push_back(CalculateHash(line)); bool stale = existing.status != eStatus::OK;
else m_DeniedWords.push_back(CalculateHash(line)); if (!stale) {
std::error_code textError, fileError;
const auto textTime = std::filesystem::last_write_time(BLOCK_LIST_TEXT, textError);
const auto fileTime = std::filesystem::last_write_time(BLOCK_LIST_FILE, fileError);
stale = textError || fileError || fileTime < textTime;
} }
if (text && (m_DontGenerateDCF || stale)) {
m_Lists.denied = BlockListFromText(*text);
if (m_DontGenerateDCF) {
LOG("Loaded %zu blocked words and phrases from %s", m_Lists.denied.Size(), BLOCK_LIST_TEXT);
return;
} }
} if (WriteFile(BLOCK_LIST_FILE, m_Lists.denied)) {
LOG("Built %s from %s (%zu words and phrases)", BLOCK_LIST_FILE, BLOCK_LIST_TEXT, m_Lists.denied.Size());
bool dChatFilter::ReadWordlistDCF(const std::string& filepath, bool allowList) {
std::ifstream file(filepath, std::ios::binary);
if (file) {
fileHeader hdr;
BinaryIO::BinaryRead(file, hdr);
if (hdr.header != header) {
file.close();
return false;
}
if (hdr.formatVersion == formatVersion) {
size_t wordsToRead = 0;
BinaryIO::BinaryRead(file, wordsToRead);
if (allowList) m_ApprovedWords.reserve(wordsToRead);
else m_DeniedWords.reserve(wordsToRead);
size_t word = 0;
for (size_t i = 0; i < wordsToRead; ++i) {
BinaryIO::BinaryRead(file, word);
if (allowList) m_ApprovedWords.push_back(word);
else m_DeniedWords.push_back(word);
}
return true;
} else { } else {
file.close(); LOG("Could not write %s", BLOCK_LIST_FILE);
return false; }
return;
} }
} }
return false; auto blocked = ReadFile(BLOCK_LIST_FILE);
switch (blocked.status) {
case eStatus::OK:
m_Lists.denied = std::move(blocked.list);
break;
case eStatus::MISSING:
LOG("No %s: best friends' free chat stops every message. Put the blocked words in %s next to the servers (one word or phrase per line) and start the servers again.",
BLOCK_LIST_FILE, BLOCK_LIST_TEXT);
break;
case eStatus::OLD_FORMAT:
LOG("%s is in the old format (version 2), whose hashes depend on the compiler and platform, so it can't be read; best friends' free chat stops every message. "
"Put the plain word list in %s next to the servers (one word or phrase per line) and start the servers again to rebuild it.",
BLOCK_LIST_FILE, BLOCK_LIST_TEXT);
break;
default:
LOG("%s is %s and can't be read; best friends' free chat stops every message. Rebuild it from %s.", BLOCK_LIST_FILE, StatusText(blocked.status), BLOCK_LIST_TEXT);
break;
}
} }
void dChatFilter::ExportWordlistToDCF(const std::string& filepath, bool allowList) { void dChatFilter::ReloadCustomWords() {
std::ofstream file(filepath, std::ios::binary | std::ios_base::out); m_Lists.customAllowed = {};
if (file) { m_Lists.customBlocked = {};
BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::header)); for (const auto& word : Database::Get()->GetChatFilterWords()) {
BinaryIO::BinaryWrite(file, uint32_t(dChatFilterDCF::formatVersion)); (word.allowed ? m_Lists.customAllowed : m_Lists.customBlocked).AddEntry(ChatFilterWords::NormalizeEntry(word.word));
BinaryIO::BinaryWrite(file, size_t(allowList ? m_ApprovedWords.size() : m_DeniedWords.size()));
for (size_t word : allowList ? m_ApprovedWords : m_DeniedWords) {
BinaryIO::BinaryWrite(file, word);
}
file.close();
} }
} }
std::set<std::pair<uint8_t, uint8_t>> dChatFilter::IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList) { std::set<std::pair<uint8_t, uint8_t>> dChatFilter::IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList) {
if (gmLevel > eGameMasterLevel::FORUM_MODERATOR) return { }; //If anything but a forum mod, return true. if (gmLevel > eGameMasterLevel::FORUM_MODERATOR) return { }; //If anything but a forum mod, return true.
if (message.empty()) return { }; return ChatFilterWords::CheckMessage(message, allowList, m_Lists);
if (!allowList && m_DeniedWords.empty()) return { { 0, message.length() } };
std::stringstream sMessage(message);
std::string segment;
std::set<std::pair<uint8_t, uint8_t>> listOfBadSegments;
uint32_t position = 0;
while (std::getline(sMessage, segment, ' ')) {
std::string originalSegment = segment;
segment = NormalizeWord(segment);
size_t hash = CalculateHash(segment);
// Blocked on the dashboard: stopped in every kind of chat
if (m_CustomBlockedWords.contains(hash)) {
listOfBadSegments.emplace(position, originalSegment.length());
position += originalSegment.length() + 1;
continue;
}
if (std::find(m_UserUnapprovedWordCache.begin(), m_UserUnapprovedWordCache.end(), hash) != m_UserUnapprovedWordCache.end() && allowList) {
listOfBadSegments.emplace(position, originalSegment.length());
}
if (std::find(m_ApprovedWords.begin(), m_ApprovedWords.end(), hash) == m_ApprovedWords.end() && !m_CustomAllowedWords.contains(hash) && allowList) {
m_UserUnapprovedWordCache.push_back(hash);
listOfBadSegments.emplace(position, originalSegment.length());
}
if (std::find(m_DeniedWords.begin(), m_DeniedWords.end(), hash) != m_DeniedWords.end() && !allowList) {
m_UserUnapprovedWordCache.push_back(hash);
listOfBadSegments.emplace(position, originalSegment.length());
}
position += originalSegment.length() + 1;
}
return listOfBadSegments;
}
size_t dChatFilter::CalculateHash(const std::string& word) {
std::hash<std::string> hash{};
size_t value = hash(word);
return value;
} }

View File

@@ -1,70 +1,52 @@
#pragma once #pragma once
#include <algorithm> #include <algorithm>
#include <cctype> #include <cctype>
#include <cstdint>
#include <filesystem>
#include <set> #include <set>
#include <unordered_set>
#include <vector>
#include <string> #include <string>
#include <vector>
#include "ChatFilterCore.h"
#include "dCommonVars.h" #include "dCommonVars.h"
enum class eGameMasterLevel : uint8_t; enum class eGameMasterLevel : uint8_t;
namespace dChatFilterDCF {
static const uint32_t header = ('D' + ('C' << 8) + ('F' << 16) + ('B' << 24));
static const uint32_t formatVersion = 2;
struct fileHeader {
uint32_t header;
uint32_t formatVersion;
};
};
class dChatFilter class dChatFilter
{ {
public: public:
/**
* Loads the allow list (filepath + ".txt", cached as filepath + ".dcf") and the block list (blocklist.dcf next to the
* servers, rebuilt from blocklist.txt there when that file is newer). dontGenerateDCF: read the plain lists only and
* write no .dcf files.
*/
dChatFilter(const std::string& filepath, bool dontGenerateDCF); dChatFilter(const std::string& filepath, bool dontGenerateDCF);
~dChatFilter(); ~dChatFilter() = default;
void ReadWordlistPlaintext(const std::string& filepath, bool allowList);
bool ReadWordlistDCF(const std::string& filepath, bool allowList);
void ExportWordlistToDCF(const std::string& filepath, bool allowList);
std::set<std::pair<uint8_t, uint8_t>> IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList = true); std::set<std::pair<uint8_t, uint8_t>> IsSentenceOkay(const std::string& message, eGameMasterLevel gmLevel, bool allowList = true);
// Whether a deny list is loaded (without one, IsSentenceOkay(..., false) refuses every message) // Whether a deny list is loaded (without one, IsSentenceOkay(..., false) refuses every message)
bool HasDenyList() const { return !m_DeniedWords.empty(); } bool HasDenyList() const { return !m_Lists.denied.Empty(); }
/** /**
* Load the words staff added on the dashboard (chat_filter_words) again, replacing the ones loaded before. * Load the words staff added on the dashboard (chat_filter_words) again, replacing the ones loaded before.
* Allowed words are accepted in whitelisted chat; blocked words are always stopped, even when a file allows them. * Allowed words are accepted in whitelisted chat; blocked words and phrases are always stopped, even when a file allows them.
*/ */
void ReloadCustomWords(); void ReloadCustomWords();
// A word as the filter compares it: lower case, without ! ? ; . , // A word as the filter compares it: lower case, without ! ? ; . ,
static std::string NormalizeWord(std::string word) { static std::string NormalizeWord(std::string word) { return ChatFilterWords::NormalizeWord(std::move(word)); }
std::erase_if(word, [](char c) { return c == '!' || c == '?' || c == ';' || c == '.' || c == ','; });
std::transform(word.begin(), word.end(), word.begin(), ::tolower); //Transform to lowercase
return word;
}
// A message split into words (at spaces) the way the filter checks it // A message split into words (at spaces) the way the filter checks it
static std::vector<std::string> Words(const std::string& message) { static std::vector<std::string> Words(const std::string& message) {
std::vector<std::string> words; std::vector<std::string> words;
size_t start = 0; for (auto& token : ChatFilterWords::Tokenize(message)) words.push_back(std::move(token.word));
while (start <= message.size()) {
const auto end = std::min(message.find(' ', start), message.size());
words.push_back(NormalizeWord(message.substr(start, end - start)));
start = end + 1;
}
return words; return words;
} }
private: private:
bool m_DontGenerateDCF; void LoadAllowList(const std::string& filepath);
std::vector<size_t> m_DeniedWords; void LoadBlockList();
std::vector<size_t> m_ApprovedWords;
std::vector<size_t> m_UserUnapprovedWordCache;
std::unordered_set<size_t> m_CustomAllowedWords;
std::unordered_set<size_t> m_CustomBlockedWords;
//Private functions: bool m_DontGenerateDCF;
size_t CalculateHash(const std::string& word); ChatFilterWords::Lists m_Lists;
}; };

View File

@@ -8,23 +8,26 @@
#include <string> #include <string>
#include <vector> #include <vector>
#include "dChatFilter.h" #include "ChatFilterCore.h"
// Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h) // Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h)
namespace ModerationTools { namespace ModerationTools {
constexpr size_t MAX_FILTER_WORD = 64; constexpr size_t MAX_FILTER_WORD = 64;
// A word staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,); nullopt if it isn't // A word or phrase staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,, words joined by
// one word of 1-64 characters. Pure; unit tested. // one space); nullopt if it isn't 1-64 characters or has no word. Pure; unit tested.
inline std::optional<std::string> FilterWord(std::string text) { inline std::optional<std::string> FilterWord(std::string text) {
text.erase(0, text.find_first_not_of(" \t\r\n")); text.erase(0, text.find_first_not_of(" \t\r\n"));
text.erase(text.find_last_not_of(" \t\r\n") + 1); text.erase(text.find_last_not_of(" \t\r\n") + 1);
if (text.empty() || text.size() > MAX_FILTER_WORD || text.find_first_of(" \t\r\n") != std::string::npos) return std::nullopt; if (text.empty() || text.size() > MAX_FILTER_WORD) return std::nullopt;
auto word = dChatFilter::NormalizeWord(text); auto entry = ChatFilterWords::NormalizeEntry(text);
if (word.empty()) return std::nullopt; if (entry.empty()) return std::nullopt;
return word; return entry;
} }
// Whether an entry is a phrase (more than one word). Phrases can only be blocked: whitelist chat checks one word at a time.
inline bool IsPhrase(const std::string& entry) { return ChatFilterWords::WordCount(entry) > 1; }
// The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once // The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once
inline std::vector<std::string> FileWords(const std::string& text) { inline std::vector<std::string> FileWords(const std::string& text) {
std::vector<std::string> words; std::vector<std::string> words;
@@ -33,7 +36,7 @@ namespace ModerationTools {
const auto end = std::min(text.find('\n', start), text.size()); const auto end = std::min(text.find('\n', start), text.size());
auto line = text.substr(start, end - start); auto line = text.substr(start, end - start);
std::erase(line, '\r'); std::erase(line, '\r');
std::transform(line.begin(), line.end(), line.begin(), ::tolower); line = ChatFilterWords::AsciiLower(std::move(line));
if (!line.empty()) words.push_back(std::move(line)); if (!line.empty()) words.push_back(std::move(line));
start = end + 1; start = end + 1;
} }
@@ -42,28 +45,12 @@ namespace ModerationTools {
return words; return words;
} }
// The hashes of a .dcf word list (blocklist.dcf), as dChatFilter::ReadWordlistDCF reads them; nullopt if it isn't one // Whether a message contains a word or phrase (FilterWord) as the chat filter reads it: whole words, in a row, skipping
inline std::optional<std::vector<size_t>> DcfHashes(const std::string& bytes) { // pieces that are only punctuation
dChatFilterDCF::fileHeader header{}; inline bool HasFilterWord(const std::string& message, const std::string& entry) {
size_t count = 0; const auto tokens = ChatFilterWords::Tokenize(message);
if (bytes.size() < sizeof(header) + sizeof(count)) return std::nullopt; const auto words = ChatFilterWords::WordCount(entry);
std::memcpy(&header, bytes.data(), sizeof(header)); return !ChatFilterWords::FindBlocked(tokens, words, [&entry](const std::string& run) { return run == entry; }).empty();
if (header.header != dChatFilterDCF::header || header.formatVersion != dChatFilterDCF::formatVersion) return std::nullopt;
std::memcpy(&count, bytes.data() + sizeof(header), sizeof(count));
const size_t offset = sizeof(header) + sizeof(count);
if (count > (bytes.size() - offset) / sizeof(size_t)) return std::nullopt;
std::vector<size_t> hashes(count);
if (count) std::memcpy(hashes.data(), bytes.data() + offset, count * sizeof(size_t));
return hashes;
}
// A word's hash as the filter stores it (dChatFilter::CalculateHash)
inline size_t WordHash(const std::string& word) { return std::hash<std::string>{}(word); }
// Whether a message contains `word` as one of the words the chat filter checks
inline bool HasFilterWord(const std::string& message, const std::string& word) {
const auto words = dChatFilter::Words(message);
return std::find(words.begin(), words.end(), word) != words.end();
} }
// What the filter decides about one word of a message, and why // What the filter decides about one word of a message, and why
@@ -72,6 +59,7 @@ namespace ModerationTools {
std::string word; // as the filter compares it std::string word; // as the filter compares it
bool stopped{}; bool stopped{};
std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file
std::string phrase; // the blocked phrase this word is part of (blocked_here or block_file), when it was a phrase
}; };
// Where the filter finds its words (callbacks keep this pure; the route reads the files and the database) // Where the filter finds its words (callbacks keep this pure; the route reads the files and the database)
@@ -81,33 +69,49 @@ namespace ModerationTools {
std::function<bool(const std::string&)> characterName; // approved character names count as allowed words std::function<bool(const std::string&)> characterName; // approved character names count as allowed words
std::function<bool(const std::string&)> blockFile; // blocklist.dcf (by hash) std::function<bool(const std::string&)> blockFile; // blocklist.dcf (by hash)
bool blockFileLoaded{}; bool blockFileLoaded{};
uint32_t maxWords{ 1 }; // the longest blocked phrase, in words (here or in the file)
}; };
/** /**
* Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher * Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher
* levels skip the filter). Normal chat (allowList) needs every word allowed; best friends' free chat (!allowList) stops * levels skip the filter). Blocked words and phrases (here always, the block file's in free chat) are stopped; a phrase
* only blocked words, or every word when there is no blocked words file. Words are split at spaces as the filter * stops each of its words. Normal chat (allowList) needs every other word allowed, one at a time; best friends' free
* splits them. Pure; unit tested. * chat stops only blocked ones, or every word when there is no blocked words file. Words are split at spaces as the
* filter splits them (ChatFilterWords::CheckMessage). Pure; unit tested.
*/ */
inline std::vector<WordVerdict> ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) { inline std::vector<WordVerdict> ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) {
const auto tokens = ChatFilterWords::Tokenize(message);
std::vector<WordVerdict> verdicts; std::vector<WordVerdict> verdicts;
std::stringstream stream(message); for (const auto& token : tokens) verdicts.push_back({ message.substr(token.position, token.length), token.word });
std::string segment;
while (std::getline(stream, segment, ' ')) {
WordVerdict verdict{ segment, dChatFilter::NormalizeWord(segment) };
const auto here = sources.dashboard(verdict.word);
if (!allowList && !sources.blockFileLoaded) { if (!allowList && !sources.blockFileLoaded) {
for (auto& verdict : verdicts) {
verdict.stopped = true; verdict.stopped = true;
verdict.reason = "no_block_file"; verdict.reason = "no_block_file";
} else if (here && !*here) { }
verdict.stopped = true; return verdicts;
verdict.reason = "blocked_here"; }
} else if (!allowList) {
verdict.stopped = sources.blockFile(verdict.word); const auto blockedHere = [&sources](const std::string& entry) { const auto here = sources.dashboard(entry); return here && !*here; };
verdict.reason = verdict.stopped ? "block_file" : "not_in_block_file"; const auto matches = ChatFilterWords::FindBlocked(tokens, std::max(sources.maxWords, 1u), [&](const std::string& entry) {
return blockedHere(entry) || (!allowList && sources.blockFile(entry));
});
for (const auto& match : matches) {
const auto reason = blockedHere(match.entry) ? "blocked_here" : "block_file";
for (size_t i = match.first; i <= match.last; i++) {
verdicts[i].stopped = true;
verdicts[i].reason = reason;
if (ChatFilterWords::WordCount(match.entry) > 1) verdicts[i].phrase = match.entry;
}
}
for (auto& verdict : verdicts) {
if (verdict.stopped) continue;
const auto here = sources.dashboard(verdict.word);
if (!allowList) {
verdict.reason = "not_in_block_file";
} else if (sources.allowFile(verdict.word)) { } else if (sources.allowFile(verdict.word)) {
verdict.reason = "allow_file"; verdict.reason = "allow_file";
} else if (here) { } else if (here && *here) {
verdict.reason = "allowed_here"; verdict.reason = "allowed_here";
} else if (sources.characterName(verdict.word)) { } else if (sources.characterName(verdict.word)) {
verdict.reason = "character_name"; verdict.reason = "character_name";
@@ -115,7 +119,6 @@ namespace ModerationTools {
verdict.stopped = true; verdict.stopped = true;
verdict.reason = "not_allowed"; verdict.reason = "not_allowed";
} }
verdicts.push_back(std::move(verdict));
} }
return verdicts; return verdicts;
} }

View File

@@ -31,7 +31,8 @@ namespace {
// The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked // The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked
// words (only their hashes) next to the servers // words (only their hashes) next to the servers
constexpr const char* ALLOW_FILE = "chatplus_en_us.txt"; constexpr const char* ALLOW_FILE = "chatplus_en_us.txt";
constexpr const char* BLOCK_FILE = "blocklist.dcf"; constexpr const char* BLOCK_FILE = dChatFilterDCF::BLOCK_LIST_FILE;
constexpr const char* BLOCK_TEXT = dChatFilterDCF::BLOCK_LIST_TEXT;
constexpr uint32_t FILE_WORDS_PAGE = 200; constexpr uint32_t FILE_WORDS_PAGE = 200;
// Recent chat searched when checking what a word would change // Recent chat searched when checking what a word would change
constexpr uint32_t CHECK_MESSAGES = 1000; constexpr uint32_t CHECK_MESSAGES = 1000;
@@ -216,11 +217,8 @@ namespace {
return text ? ModerationTools::FileWords(*text) : std::vector<std::string>{}; return text ? ModerationTools::FileWords(*text) : std::vector<std::string>{};
} }
std::optional<std::vector<size_t>> BlockFileHashes() { // blocklist.dcf as the servers read it (an old-format file is refused, as they refuse it)
std::ifstream in(BLOCK_FILE, std::ios::binary); dChatFilterDCF::ParseResult BlockFile() { return dChatFilterDCF::ReadFile(BLOCK_FILE); }
if (!in) return std::nullopt;
return ModerationTools::DcfHashes(std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>()));
}
// The dashboard's own lists by word: true allowed, false blocked // The dashboard's own lists by word: true allowed, false blocked
std::map<std::string, bool> DashboardWords() { std::map<std::string, bool> DashboardWords() {
@@ -258,25 +256,27 @@ namespace {
const auto it = dashboard.find(word); const auto it = dashboard.find(word);
words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} }); words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
} }
const auto blocked = BlockFileHashes(); const auto blocked = BlockFile();
uint32_t imported = 0; uint32_t imported = 0;
for (const auto& word : all) if (dashboard.contains(word)) imported++; for (const auto& word : all) if (dashboard.contains(word)) imported++;
JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched}, JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched},
{"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported}, {"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported},
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.has_value()}, {"blockTotal", blocked ? blocked->size() : 0} }); {"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK}, {"blockTotal", blocked.list.Size()},
{"blockMaxWords", blocked.list.maxWords}, {"blockFileStatus", dChatFilterDCF::StatusText(blocked.status)},
{"blockFileOld", blocked.status == dChatFilterDCF::eStatus::OLD_FORMAT}, {"blockText", BLOCK_TEXT} });
}); });
Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"), Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"),
"Where a word stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word", "Where a word or phrase stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
[](HTTPReply& reply, const HTTPContext& context) { [](HTTPReply& reply, const HTTPContext& context) {
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word")); const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const auto all = AllowFileWords(); const auto all = AllowFileWords();
const auto blocked = BlockFileHashes(); const auto blocked = BlockFile();
const auto dashboard = DashboardWords(); const auto dashboard = DashboardWords();
const auto it = dashboard.find(*word); const auto it = dashboard.find(*word);
JsonSuccess(reply, { {"word", *word}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)}, JsonSuccess(reply, { {"word", *word}, {"phrase", ModerationTools::IsPhrase(*word)}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
{"inBlockFile", blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(*word)) != blocked->end()}, {"inBlockFile", blocked.list.Contains(*word)},
{"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} }); {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
}); });
@@ -293,7 +293,7 @@ namespace {
for (const auto& word : all) { for (const auto& word : all) {
// Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only) // Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only)
const auto filterWord = ModerationTools::FilterWord(word); const auto filterWord = ModerationTools::FilterWord(word);
if (!filterWord || *filterWord != word || dashboard.contains(word)) continue; if (!filterWord || *filterWord != word || ModerationTools::IsPhrase(word) || dashboard.contains(word)) continue;
Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now }); Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now });
added++; added++;
} }
@@ -305,13 +305,17 @@ namespace {
}); });
Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"), Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"),
"Allow or block a word (or move it to the other list); running worlds pick it up at once. Body: {word, allowed: bool}", "Allow or block a word, or block a phrase (or move it to the other list); running worlds pick it up at once. Phrases can't be allowed: "
"whitelist chat checks each word on its own, as the client does. Body: {word, allowed: bool}",
[](HTTPReply& reply, const HTTPContext& context) { [](HTTPReply& reply, const HTTPContext& context) {
const auto body = ParseBody(context); const auto body = ParseBody(context);
if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON"); if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON");
const auto word = ModerationTools::FilterWord(body->value("word", "")); const auto word = ModerationTools::FilterWord(body->value("word", ""));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = body->value("allowed", false); const bool allowed = body->value("allowed", false);
if (allowed && ModerationTools::IsPhrase(*word)) {
return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead");
}
Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast<int64_t>(std::time(nullptr)) }); Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast<int64_t>(std::time(nullptr)) });
Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat"); Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat");
BroadcastTableChanged("chat_filter"); BroadcastTableChanged("chat_filter");
@@ -345,12 +349,11 @@ namespace {
if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters"); if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters");
const bool allowList = QueryValue(context.queryString, "chat") != "free"; const bool allowList = QueryValue(context.queryString, "chat") != "free";
const auto all = AllowFileWords(); const auto all = AllowFileWords();
const auto blocked = BlockFileHashes(); const auto blocked = BlockFile();
const auto dashboard = DashboardWords(); const auto dashboard = DashboardWords();
std::set<std::string> names; std::set<std::string> names;
for (auto name : Database::Get()->GetApprovedCharacterNames()) { for (auto name : Database::Get()->GetApprovedCharacterNames()) {
std::transform(name.begin(), name.end(), name.begin(), ::tolower); names.insert(ChatFilterWords::AsciiLower(std::move(name)));
names.insert(std::move(name));
} }
ModerationTools::WordSources sources; ModerationTools::WordSources sources;
sources.dashboard = [&dashboard](const std::string& word) -> std::optional<bool> { sources.dashboard = [&dashboard](const std::string& word) -> std::optional<bool> {
@@ -359,18 +362,18 @@ namespace {
}; };
sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); }; sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); };
sources.characterName = [&names](const std::string& word) { return names.contains(word); }; sources.characterName = [&names](const std::string& word) { return names.contains(word); };
sources.blockFile = [&blocked](const std::string& word) { sources.blockFile = [&blocked](const std::string& entry) { return blocked.list.Contains(entry); };
return blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(word)) != blocked->end(); sources.blockFileLoaded = !blocked.list.Empty();
}; sources.maxWords = blocked.list.maxWords;
sources.blockFileLoaded = blocked && !blocked->empty(); for (const auto& [entry, allowed] : dashboard) if (!allowed) sources.maxWords = std::max(sources.maxWords, ChatFilterWords::WordCount(entry));
nlohmann::json words = nlohmann::json::array(); nlohmann::json words = nlohmann::json::array();
bool stopped = false; bool stopped = false;
for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) { for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) {
stopped |= verdict.stopped; stopped |= verdict.stopped;
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason} }); words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason}, {"phrase", verdict.phrase} });
} }
JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words}, JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words},
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.has_value()} }); {"allowFileFound", !all.empty()}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK} });
}); });
Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"), Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"),
@@ -379,7 +382,7 @@ namespace {
[](HTTPReply& reply, const HTTPContext& context) { [](HTTPReply& reply, const HTTPContext& context) {
if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission"); if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission");
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word")); const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters"); if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = QueryValue(context.queryString, "allowed") == "1"; const bool allowed = QueryValue(context.queryString, "allowed") == "1";
IChatLog::ChatQuery query; IChatLog::ChatQuery query;
query.search = *word; query.search = *word;

View File

@@ -8,6 +8,9 @@
var page = document.getElementById('chatFilterPage'); var page = document.getElementById('chatFilterPage');
var canChat = !!page.dataset.canChat; var canChat = !!page.dataset.canChat;
var PAGE = 50; var PAGE = 50;
function isPhrase(word) { return String(word).indexOf(' ') !== -1; }
function phraseBadge(word) { return isPhrase(word) ? ' ' + fmt.badge('Phrase', 'secondary') : ''; }
var REASONS = { var REASONS = {
blocked_here: 'Blocked on the staff list', blocked_here: 'Blocked on the staff list',
allowed_here: 'Allowed on the staff list', allowed_here: 'Allowed on the staff list',
@@ -26,7 +29,8 @@
function wordAction(w) { function wordAction(w) {
if (!w.word) return ''; if (!w.word) return '';
if (w.reason === 'blocked_here' || w.reason === 'allowed_here') return '<button type="button" class="btn btn-sm btn-outline-secondary" data-remove="' + esc(w.word) + '">Remove…</button>'; if (w.reason === 'blocked_here' || w.reason === 'allowed_here') return '<button type="button" class="btn btn-sm btn-outline-secondary" data-remove="' + esc(w.phrase || w.word) + '">Remove…</button>';
if (w.phrase) return '';
if (w.reason === 'not_allowed') return '<button type="button" class="btn btn-sm btn-outline-success" data-add="allowed" data-word="' + esc(w.word) + '">Allow…</button>'; if (w.reason === 'not_allowed') return '<button type="button" class="btn btn-sm btn-outline-success" data-add="allowed" data-word="' + esc(w.word) + '">Allow…</button>';
return '<button type="button" class="btn btn-sm btn-outline-danger" data-add="blocked" data-word="' + esc(w.word) + '">Block…</button>'; return '<button type="button" class="btn btn-sm btn-outline-danger" data-add="blocked" data-word="' + esc(w.word) + '">Block…</button>';
} }
@@ -48,7 +52,7 @@
}).join(''); }).join('');
document.getElementById('testRows').innerHTML = d.words.map(function (w) { document.getElementById('testRows').innerHTML = d.words.map(function (w) {
return '<tr><td>' + esc(w.word || '(empty: two spaces in a row)') + '</td><td>' + (w.stopped ? fmt.badge('Stopped', 'danger') : fmt.badge('OK', 'success')) + '</td>' + return '<tr><td>' + esc(w.word || '(empty: two spaces in a row)') + '</td><td>' + (w.stopped ? fmt.badge('Stopped', 'danger') : fmt.badge('OK', 'success')) + '</td>' +
'<td class="small">' + esc(REASONS[w.reason] || w.reason) + '</td><td class="text-end">' + wordAction(w) + '</td></tr>'; '<td class="small">' + esc(REASONS[w.reason] || w.reason) + (w.phrase ? ' (the phrase <strong>' + esc(w.phrase) + '</strong>)' : '') + '</td><td class="text-end">' + wordAction(w) + '</td></tr>';
}).join(''); }).join('');
document.getElementById('testResult').classList.remove('d-none'); document.getElementById('testResult').classList.remove('d-none');
}).catch(function () {}); }).catch(function () {});
@@ -83,10 +87,10 @@
var shown = rows.slice(listStart, listStart + PAGE); var shown = rows.slice(listStart, listStart + PAGE);
document.getElementById('listRows').innerHTML = shown.map(function (w) { document.getElementById('listRows').innerHTML = shown.map(function (w) {
var other = w.allowed ? 'blocked' : 'allowed'; var other = w.allowed ? 'blocked' : 'allowed';
return '<tr><td>' + esc(w.word) + '</td><td>' + (w.allowed ? fmt.badge('Allowed', 'success') : fmt.badge('Blocked', 'danger')) + '</td>' + return '<tr><td>' + esc(w.word) + phraseBadge(w.word) + '</td><td>' + (w.allowed ? fmt.badge('Allowed', 'success') : fmt.badge('Blocked', 'danger')) + '</td>' +
'<td class="small">' + esc(w.added_by) + '</td><td class="small text-nowrap">' + esc(fmt.unix(w.added_at)) + '</td>' + '<td class="small">' + esc(w.added_by) + '</td><td class="small text-nowrap">' + esc(fmt.unix(w.added_at)) + '</td>' +
'<td class="text-end text-nowrap"><button type="button" class="btn btn-sm btn-outline-secondary" data-test="' + esc(w.word) + '">Test</button> ' + '<td class="text-end text-nowrap"><button type="button" class="btn btn-sm btn-outline-secondary" data-test="' + esc(w.word) + '">Test</button> ' +
'<button type="button" class="btn btn-sm btn-outline-' + (w.allowed ? 'danger' : 'success') + '" data-add="' + other + '" data-word="' + esc(w.word) + '">' + (w.allowed ? 'Block…' : 'Allow…') + '</button> ' + (isPhrase(w.word) && !w.allowed ? '' : '<button type="button" class="btn btn-sm btn-outline-' + (w.allowed ? 'danger' : 'success') + '" data-add="' + other + '" data-word="' + esc(w.word) + '">' + (w.allowed ? 'Block…' : 'Allow…') + '</button> ') +
'<button type="button" class="btn btn-sm btn-outline-secondary" data-remove="' + esc(w.word) + '">Remove…</button></td></tr>'; '<button type="button" class="btn btn-sm btn-outline-secondary" data-remove="' + esc(w.word) + '">Remove…</button></td></tr>';
}).join('') || '<tr><td colspan="5" class="text-body-secondary">' + (words.length ? 'No words match.' : 'No words added yet.') + '</td></tr>'; }).join('') || '<tr><td colspan="5" class="text-body-secondary">' + (words.length ? 'No words match.' : 'No words added yet.') + '</td></tr>';
document.getElementById('listRange').textContent = rows.length ? (listStart + 1) + '–' + (listStart + shown.length) + ' of ' + rows.length : ''; document.getElementById('listRange').textContent = rows.length ? (listStart + 1) + '–' + (listStart + shown.length) + ' of ' + rows.length : '';
@@ -114,8 +118,9 @@
addForm.addEventListener('click', function (e) { var b = e.target.closest('button[data-list]'); if (b) addList = b.dataset.list; }); addForm.addEventListener('click', function (e) { var b = e.target.closest('button[data-list]'); if (b) addList = b.dataset.list; });
addForm.addEventListener('submit', function (e) { addForm.addEventListener('submit', function (e) {
e.preventDefault(); e.preventDefault();
var word = addWord.value.trim(); var word = addWord.value.trim().replace(/\s+/g, ' ');
if (!word) return; if (!word) return;
if (addList === 'allowed' && isPhrase(word)) { toast('Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead', 'warning'); return; }
confirmAdd(word, addList === 'allowed').then(function (done) { if (done) addWord.value = ''; }); confirmAdd(word, addList === 'allowed').then(function (done) { if (done) addWord.value = ''; });
}); });
@@ -129,7 +134,7 @@
var parts = [d.inAllowFile ? 'in chatplus_en_us.txt' : 'not in chatplus_en_us.txt']; var parts = [d.inAllowFile ? 'in chatplus_en_us.txt' : 'not in chatplus_en_us.txt'];
if (d.inBlockFile) parts.push('in blocklist.dcf'); if (d.inBlockFile) parts.push('in blocklist.dcf');
parts.push(d.dashboard ? (d.dashboard === 'blocked' ? 'on the Blocked list' : 'on the Allowed list') : 'on neither staff list'); parts.push(d.dashboard ? (d.dashboard === 'blocked' ? 'on the Blocked list' : 'on the Allowed list') : 'on neither staff list');
return '<strong>' + esc(d.word) + '</strong> now: ' + esc(parts.join(', ')) + '.'; return '<strong>' + esc(d.word) + '</strong>' + phraseBadge(d.word) + ' now: ' + esc(parts.join(', ')) + '.';
} }
function openConfirm(options) { function openConfirm(options) {
@@ -186,7 +191,8 @@
function confirmAdd(word, allowed) { function confirmAdd(word, allowed) {
var p = openConfirm({ var p = openConfirm({
title: (allowed ? 'Allow "' : 'Block "') + word + '"?', title: (allowed ? 'Allow "' : 'Block "') + word + '"?',
text: allowed ? 'Players may use it in normal chat. Running worlds apply it at once.' : 'It is stopped in all chat, even where a word file allows it. Running worlds apply it at once.', text: allowed ? 'Players may use it in normal chat. Running worlds apply it at once.'
: (isPhrase(word) ? 'The phrase is stopped in all chat when its words come in a row. Running worlds apply it at once.' : 'It is stopped in all chat, even where a word file allows it. Running worlds apply it at once.'),
tone: allowed ? 'success' : 'danger', tone: allowed ? 'success' : 'danger',
button: allowed ? 'Allow' : 'Block', button: allowed ? 'Allow' : 'Block',
run: function () { return api.action('/api/chat_filter/words', { word: word, allowed: allowed }).then(function (d) { toast(d.message, 'success'); }); } run: function () { return api.action('/api/chat_filter/words', { word: word, allowed: allowed }).then(function (d) { toast(d.message, 'success'); }); }
@@ -228,8 +234,11 @@
? esc(d.allowTotal) + ' words normal chat may use (from the client); ' + esc(d.onDashboard) + ' are on a staff list too, which wins.' ? esc(d.allowTotal) + ' words normal chat may use (from the client); ' + esc(d.onDashboard) + ' are on a staff list too, which wins.'
: 'could not be read. Set <code>client_location</code>.') + '</li>' + : 'could not be read. Set <code>client_location</code>.') + '</li>' +
'<li><code>' + esc(d.blockFile) + '</code>: ' + (d.blockFileFound '<li><code>' + esc(d.blockFile) + '</code>: ' + (d.blockFileFound
? esc(d.blockTotal) + ' words best friends\' free chat may not use. Stored as hashes, so they can\'t be listed; test a word to see if it is one.' ? esc(d.blockTotal) + ' words' + (d.blockMaxWords > 1 ? ' and phrases (up to ' + esc(d.blockMaxWords) + ' words)' : '') + ' best friends\' free chat may not use. Stored as hashes, so they can\'t be listed; test a word or phrase to see if it is one.'
: 'not found next to the servers, so best friends\' free chat stops everything.') + '</li>'; : d.blockFileOld
? 'in the old format (its hashes depended on the platform), so the servers can\'t read it and best friends\' free chat stops everything.'
: 'not found next to the servers (' + esc(d.blockFileStatus) + '), so best friends\' free chat stops everything.') +
' To change it, put the words in <code>' + esc(d.blockText) + '</code> next to the servers (one word or phrase per line) and start the servers again; they build <code>' + esc(d.blockFile) + '</code> from it.</li>';
document.getElementById('fileWords').innerHTML = d.words.map(function (w) { document.getElementById('fileWords').innerHTML = d.words.map(function (w) {
var cls = w.dashboard === 'blocked' ? 'text-bg-danger' : w.dashboard === 'allowed' ? 'text-bg-success' : 'text-bg-secondary'; var cls = w.dashboard === 'blocked' ? 'text-bg-danger' : w.dashboard === 'allowed' ? 'text-bg-success' : 'text-bg-secondary';
return '<button type="button" class="badge border-0 ' + cls + '" data-test="' + esc(w.word) + '">' + esc(w.word) + '</button>'; return '<button type="button" class="badge border-0 ' + cls + '" data-test="' + esc(w.word) + '">' + esc(w.word) + '</button>';

View File

@@ -54,12 +54,12 @@
</div> </div>
<div class="card-body"> <div class="card-body">
<dl class="small cf-explain mb-3 row"> <dl class="small cf-explain mb-3 row">
<dt class="col-sm-2 text-danger">Blocked</dt><dd class="col-sm-10">Stopped in all chat, even if a word file allows it.</dd> <dt class="col-sm-2 text-danger">Blocked</dt><dd class="col-sm-10">Stopped in all chat, even if a word file allows it. A phrase (several words) is stopped when its words come in a row, whatever the spaces and punctuation between them.</dd>
<dt class="col-sm-2 text-success">Allowed</dt><dd class="col-sm-10">Usable in normal chat, like the words in <code>chatplus_en_us.txt</code>.</dd> <dt class="col-sm-2 text-success">Allowed</dt><dd class="col-sm-10">Usable in normal chat, like the words in <code>chatplus_en_us.txt</code>. Single words only: normal chat checks each word on its own, as the client does.</dd>
</dl> </dl>
<form class="row g-2 align-items-end mb-3" id="addForm"> <form class="row g-2 align-items-end mb-3" id="addForm">
<div class="col-md-6"><label class="form-label small mb-1" for="addWord">Add a word</label> <div class="col-md-6"><label class="form-label small mb-1" for="addWord">Add a word or phrase</label>
<input class="form-control form-control-sm" id="addWord" maxlength="64" required autocomplete="off" placeholder="One word, no spaces"></div> <input class="form-control form-control-sm" id="addWord" maxlength="64" required autocomplete="off" placeholder="A word, or a phrase to block"></div>
<div class="col-md-6 d-flex gap-2"> <div class="col-md-6 d-flex gap-2">
<button type="submit" class="btn btn-sm btn-danger" data-list="blocked">Block…</button> <button type="submit" class="btn btn-sm btn-danger" data-list="blocked">Block…</button>
<button type="submit" class="btn btn-sm btn-success" data-list="allowed">Allow…</button> <button type="submit" class="btn btn-sm btn-success" data-list="allowed">Allow…</button>
@@ -74,7 +74,7 @@
<input type="search" class="form-control form-control-sm w-auto ms-auto" id="listSearch" placeholder="Search the lists" aria-label="Search the lists"> <input type="search" class="form-control form-control-sm w-auto ms-auto" id="listSearch" placeholder="Search the lists" aria-label="Search the lists">
</div> </div>
<table class="table table-sm table-hover align-middle w-100 table-stack mb-2" id="cfWordTable"> <table class="table table-sm table-hover align-middle w-100 table-stack mb-2" id="cfWordTable">
<thead><tr><th>Word</th><th>List</th><th>Added by</th><th>Added</th><th></th></tr></thead> <thead><tr><th>Word or phrase</th><th>List</th><th>Added by</th><th>Added</th><th></th></tr></thead>
<tbody id="listRows"></tbody> <tbody id="listRows"></tbody>
</table> </table>
<div class="d-flex justify-content-between align-items-center small"> <div class="d-flex justify-content-between align-items-center small">

View File

@@ -1213,18 +1213,42 @@ character's owner sees only what is still in their mailbox. Deleting a character
The **Chat Filter** page (GM 5+, `chat_filter_manage`, under Moderation) decides which words players below GM 2 may use The **Chat Filter** page (GM 5+, `chat_filter_manage`, under Moderation) decides which words players below GM 2 may use
in chat. The filter's files: `chatplus_en_us.txt` (client `res` folder) lists the words normal chat may use, in chat. The filter's files: `chatplus_en_us.txt` (client `res` folder) lists the words normal chat may use,
`blocklist.dcf` (next to the servers, hashes only) the words best friends' free chat may not. Approved character names `blocklist.dcf` (next to the servers, hashes only) the words and phrases best friends' free chat may not. Approved
also count as allowed. Words are compared lower case, without `! ? ; . ,`. Changes apply at once in running worlds and in character names also count as allowed. Words are compared lower case, without `! ? ; . ,`. Changes apply at once in
the chat server's web chat; servers that start later read them. Changes are audited and go to the `moderation` webhook running worlds and in the chat server's web chat; servers that start later read them. Changes are audited and go to the
event. `moderation` webhook event.
Blocked entries can be phrases: a phrase is stopped when its words come in a row in a message, whatever the spaces and
punctuation between them, and the whole phrase is marked. Allowed entries are single words only, because normal
(whitelist) chat checks each word on its own, as the client does.
#### Block list file
`blocklist.dcf` is DLU's own file (the client reads no `.dcf` and doesn't hash chat words). To make or change it, put the
blocked words in `blocklist.txt` next to the servers (the build folder, beside `blocklist.dcf`): one word or phrase per
line, any case, punctuation `! ? ; . ,` ignored, blank lines skipped. The world and chat servers rebuild `blocklist.dcf`
from it when they start and it is newer than the `.dcf` (or the `.dcf` is missing or unreadable), then log how many
entries it has. With `dont_generate_dcf=1` they read `blocklist.txt` directly and write no file. `blocklist.txt` can be
removed afterwards; only the `.dcf` is needed.
Format (little-endian): `uint32` magic `DCFB`, `uint32` version `3`, `uint32` most words in one entry, `uint64` count,
then that many `uint64` hashes, sorted. Each hash is 64-bit FNV-1a (offset basis `0xcbf29ce484222325`, prime
`0x100000001b3`) over the entry's bytes: the words lower case (ASCII), without `! ? ; . ,`, joined by one space. The
same words give the same file on every platform. The allowed words cache, `chatplus_en_us.dcf` in the client's `res`
folder, uses the same format and is built from `chatplus_en_us.txt`.
Version 2 files (older DLU) stored `std::hash` values, which differ between compilers and platforms, so a list made on
one system never matched on another (issue 215). Servers refuse them: an old `chatplus_en_us.dcf` is rebuilt from
the `.txt`, and an old `blocklist.dcf` is logged as unreadable (free chat then stops every message) until it is rebuilt
from `blocklist.txt`. The **Word files** section shows which it is.
The page has three sections: The page has three sections:
- **Test a message**: type a message, pick normal or best friends' free chat, and see whether it would be sent and why, - **Test a message**: type a message, pick normal or best friends' free chat, and see whether it would be sent and why,
word by word (in the file, allowed or blocked here, a character name, not allowed, in the blocked words file). Each word by word (in the file, allowed or blocked here, a character name, not allowed, in the blocked words file). Each
word has a Block, Allow or Remove button. word has a Block, Allow or Remove button.
- **Staff lists**: **Blocked** words are stopped in all chat, even where a file allows them; **Allowed** words are usable - **Staff lists**: **Blocked** words and phrases are stopped in all chat, even where a file allows them (phrases are
in normal chat. Search, filter by list, 50 per page. Block, Allow (or move to the other list) and Remove each open a marked **Phrase**); **Allowed** words are usable in normal chat. Search, filter by list, 50 per page. Block, Allow (or move to the other list) and Remove each open a
confirmation that shows where the word stands now and, for Block and Allow, the recent chat it changes (players' chat confirmation that shows where the word stands now and, for Block and Allow, the recent chat it changes (players' chat
containing it that would have been stopped, or stopped messages containing it; the newest 1000 messages with the containing it that would have been stopped, or stopped messages containing it; the newest 1000 messages with the
text; needs `chat_view`). text; needs `chat_view`).

View File

@@ -10,6 +10,7 @@ State: **done** = fixed on this branch, needs an in-game check; **partial** = pa
|---|---|---|---| |---|---|---|---|
| 159 | BUG: Brick-by-brick models are deleted instead of put away | done | `6ce261c5` fix: brick by brick and model placement work the way the client expects | | 159 | BUG: Brick-by-brick models are deleted instead of put away | done | `6ce261c5` fix: brick by brick and model placement work the way the client expects |
| 185 | BUG: Assembly Engineer Fortress Knockback | partial | `2d76c81b` feat: server side knockback for AI moved objects | | 185 | BUG: Assembly Engineer Fortress Knockback | partial | `2d76c81b` feat: server side knockback for AI moved objects |
| 215 | ENH: Bring chat filter closer to Live | partial | `c1bcda8d` fix(chat-filter): portable .dcf hashing, block list phrases; `b279d187` shipped blocklist.dcf in the portable format; `cef170ae` dashboard phrases. The block list works on every platform and takes phrases; the rest of the issue is open |
| 225 | ENH: "bind_ip" config option | done | `e36f894f` feat: bind_ip setting for the server sockets | | 225 | ENH: "bind_ip" config option | done | `e36f894f` feat: bind_ip setting for the server sockets |
| 257 | EH: Crux Prime shields stun instead of knockback | done | `2d76c81b` feat: server side knockback for AI moved objects | | 257 | EH: Crux Prime shields stun instead of knockback | done | `2d76c81b` feat: server side knockback for AI moved objects |
| 307 | Spider Queen scream on spiderling death | done | `4900f11c` fix(scripts): the Spider Queen screams from the mountain when a spiderling dies (issue 307) | | 307 | Spider Queen scream on spiderling death | done | `4900f11c` fix(scripts): the Spider Queen screams from the mountain when a spiderling dies (issue 307) |

Binary file not shown.

View File

@@ -39,6 +39,7 @@ set(DCOMMONTEST_SOURCES
"FdbReaderTests.cpp" "FdbReaderTests.cpp"
"FdbSnapshotTests.cpp" "FdbSnapshotTests.cpp"
"WorldFileWatchTests.cpp" "WorldFileWatchTests.cpp"
"ChatFilterCoreTests.cpp"
) )
add_subdirectory(dEnumsTests) add_subdirectory(dEnumsTests)
@@ -59,6 +60,8 @@ endif()
target_link_libraries(dCommonTests ${COMMON_LIBRARIES} MD5 GTest::gtest_main) target_link_libraries(dCommonTests ${COMMON_LIBRARIES} MD5 GTest::gtest_main)
# SpareBackoff.h (header only) # SpareBackoff.h (header only)
target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dMasterServer") target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dMasterServer")
# ChatFilterCore.h (header only)
target_include_directories(dCommonTests PRIVATE "${PROJECT_SOURCE_DIR}/dChatFilter")
# Copy test files to testing directory # Copy test files to testing directory
add_subdirectory(TestBitStreams) add_subdirectory(TestBitStreams)

View File

@@ -0,0 +1,158 @@
#include <gtest/gtest.h>
#include <string>
#include <vector>
#include "ChatFilterCore.h"
using namespace ChatFilterWords;
using Spans = std::set<std::pair<uint8_t, uint8_t>>;
// The hash is a compile-time constant: the same value on every compiler, standard library and platform
static_assert(Hash("") == 0xcbf29ce484222325ULL);
static_assert(Hash("a") == 0xaf63dc4c8601ec8cULL);
static_assert(Hash("foobar") == 0x85944171f73967e8ULL);
TEST(ChatFilterCoreTest, HashIsFnv1a64) {
// The standard FNV-1a 64 test vectors
EXPECT_EQ(Hash(""), 0xcbf29ce484222325ULL);
EXPECT_EQ(Hash("a"), 0xaf63dc4c8601ec8cULL);
EXPECT_EQ(Hash("foobar"), 0x85944171f73967e8ULL);
// Words of the client's chatplus_en_us.txt, as the filter hashes them (lower case)
EXPECT_EQ(Hash("hello"), 0xa430d84680aabd0bULL);
EXPECT_EQ(Hash("brick"), 0xf9236d1e24832c9aULL);
EXPECT_EQ(Hash(AsciiLower("Brick")), Hash("brick"));
// A phrase is its words joined by one space
EXPECT_EQ(Hash("bad phrase"), 0x72e9fce49c1b0875ULL);
// Bytes, not chars: a high byte hashes as 0x80-0xFF whether char is signed or not
EXPECT_EQ(Hash("\xC3\xA9"), 0x0ac21707b7181e01ULL);
}
TEST(ChatFilterCoreTest, NormalizeEntry) {
EXPECT_EQ(NormalizeWord("Hello!?"), "hello");
EXPECT_EQ(NormalizeEntry(" Bad \t PHRASE! "), "bad phrase");
EXPECT_EQ(NormalizeEntry("word , here"), "word here");
EXPECT_EQ(NormalizeEntry("..."), "");
EXPECT_EQ(WordCount("bad phrase here"), 3u);
EXPECT_EQ(WordCount("word"), 1u);
EXPECT_EQ(WordCount(""), 0u);
// ASCII only: other bytes are left alone whatever the locale
EXPECT_EQ(AsciiLower("\xC3\x89Z"), "\xC3\x89z");
}
TEST(ChatFilterCoreTest, DcfBytesAreFixed) {
WordList list;
list.AddEntry("a");
list.AddEntry("bad phrase");
const auto bytes = dChatFilterDCF::Serialize(list);
// magic DCFB, version 3, 2 words at most, 2 hashes, then the hashes sorted, all little-endian
const std::string expected(
"DCFB" "\x03\x00\x00\x00" "\x02\x00\x00\x00" "\x02\x00\x00\x00\x00\x00\x00\x00"
"\x75\x08\x1b\x9c\xe4\xfc\xe9\x72" "\x8c\xec\x01\x86\x4c\xdc\x63\xaf", 4 + 4 + 4 + 8 + 16);
EXPECT_EQ(bytes, expected);
const auto parsed = dChatFilterDCF::Parse(bytes);
ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK);
EXPECT_EQ(parsed.list.maxWords, 2u);
EXPECT_EQ(parsed.list.hashes, list.hashes);
}
TEST(ChatFilterCoreTest, DcfRejectsOldAndBadFiles) {
// Version 2 (std::hash, platform dependent) is refused, not guessed at
const std::string old("DCFB" "\x02\x00\x00\x00" "\x01\x00\x00\x00\x00\x00\x00\x00" "\x11\x22\x33\x44\x55\x66\x77\x88", 24);
EXPECT_EQ(dChatFilterDCF::Parse(old).status, dChatFilterDCF::eStatus::OLD_FORMAT);
EXPECT_EQ(dChatFilterDCF::Parse("XCFB\x03\x00\x00\x00").status, dChatFilterDCF::eStatus::NOT_DCF);
EXPECT_EQ(dChatFilterDCF::Parse(std::string("DCFB\x09\x00\x00\x00", 8)).status, dChatFilterDCF::eStatus::UNKNOWN);
WordList list;
list.AddEntry("word");
auto bytes = dChatFilterDCF::Serialize(list);
bytes.pop_back();
EXPECT_EQ(dChatFilterDCF::Parse(bytes).status, dChatFilterDCF::eStatus::TRUNCATED);
EXPECT_EQ(dChatFilterDCF::ReadFile("no such blocklist.dcf").status, dChatFilterDCF::eStatus::MISSING);
}
TEST(ChatFilterCoreTest, BlockListFromText) {
const auto list = dChatFilterDCF::BlockListFromText("Badword\r\n\r\n Very BAD phrase!\nbadword\n...\n");
EXPECT_EQ(list.Size(), 2u);
EXPECT_TRUE(list.Contains("badword"));
EXPECT_TRUE(list.Contains("very bad phrase"));
EXPECT_EQ(list.maxWords, 3u);
// Written and read back, the same list
const auto parsed = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(list));
ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK);
EXPECT_EQ(parsed.list.hashes, list.hashes);
EXPECT_EQ(parsed.list.maxWords, 3u);
}
namespace {
Lists FreeChatLists(std::string_view blockText) {
Lists lists;
// Through the file format, as the servers load it
lists.denied = dChatFilterDCF::Parse(dChatFilterDCF::Serialize(dChatFilterDCF::BlockListFromText(blockText))).list;
lists.approved = dChatFilterDCF::AllowListFromText("hello\nthere\nfriend\nbad\nphrase\n");
return lists;
}
}
TEST(ChatFilterCoreTest, BlockedWordFromFileIsStopped) {
const auto lists = FreeChatLists("badword\n");
EXPECT_EQ(CheckMessage("hello badword there", false, lists), (Spans{ { 6, 7 } }));
EXPECT_EQ(CheckMessage("hello BadWord!", false, lists), (Spans{ { 6, 8 } }));
EXPECT_TRUE(CheckMessage("hello there", false, lists).empty());
// Whitelist chat doesn't use the block list: the word is simply not allowed there
EXPECT_EQ(CheckMessage("hello badword", true, lists), (Spans{ { 6, 7 } }));
// No block list: free chat stops everything
EXPECT_EQ(CheckMessage("hello", false, Lists{}), (Spans{ { 0, 5 } }));
}
TEST(ChatFilterCoreTest, PhrasesAtStartMiddleEnd) {
const auto lists = FreeChatLists("bad phrase\n");
EXPECT_EQ(CheckMessage("bad phrase hello", false, lists), (Spans{ { 0, 10 } }));
EXPECT_EQ(CheckMessage("hello bad phrase there", false, lists), (Spans{ { 6, 10 } }));
EXPECT_EQ(CheckMessage("hello bad phrase", false, lists), (Spans{ { 6, 10 } }));
// The words alone are fine
EXPECT_TRUE(CheckMessage("bad hello phrase", false, lists).empty());
EXPECT_TRUE(CheckMessage("phrase bad", false, lists).empty());
}
TEST(ChatFilterCoreTest, PhrasesWithExtraSpacesAndPunctuation) {
const auto lists = FreeChatLists("bad phrase\n");
// Two spaces: the span covers both words and the gap
EXPECT_EQ(CheckMessage("hi Bad Phrase!", false, lists), (Spans{ { 4, 12 } }));
// A piece that is only punctuation is skipped
EXPECT_EQ(CheckMessage("bad ... phrase", false, lists), (Spans{ { 0, 14 } }));
EXPECT_EQ(CheckMessage("bad, phrase.", false, lists), (Spans{ { 0, 12 } }));
}
TEST(ChatFilterCoreTest, PhrasesOverlapWithWords) {
const auto lists = FreeChatLists("bad phrase\nphrase here\nbadword\nfriend\n");
// Two phrases sharing a word become one span
EXPECT_EQ(CheckMessage("a bad phrase here b", false, lists), (Spans{ { 2, 15 } }));
// A blocked word inside a blocked phrase: one span for the phrase
EXPECT_EQ(CheckMessage("bad phrase", false, FreeChatLists("bad phrase\nphrase\n")), (Spans{ { 0, 10 } }));
// A blocked word right after a phrase: its own span
EXPECT_EQ(CheckMessage("bad phrase badword", false, lists), (Spans{ { 0, 10 }, { 11, 7 } }));
// Longest match wins where it starts
EXPECT_EQ(CheckMessage("friend bad phrase", false, FreeChatLists("friend\nfriend bad\nbad phrase\n")), (Spans{ { 0, 17 } }));
}
TEST(ChatFilterCoreTest, DashboardPhrasesInWhitelistChat) {
auto lists = FreeChatLists("");
lists.customBlocked.AddEntry(NormalizeEntry("Bad Phrase"));
// Blocked on the dashboard: stopped in whitelist chat too, even though each word is allowed
EXPECT_EQ(CheckMessage("hello bad phrase", true, lists), (Spans{ { 6, 10 } }));
EXPECT_TRUE(CheckMessage("hello bad there phrase", true, lists).empty());
// Whitelist chat checks one word at a time, as the client does
EXPECT_EQ(CheckMessage("hello stranger", true, lists), (Spans{ { 6, 8 } }));
lists.customAllowed.AddEntry("stranger");
EXPECT_TRUE(CheckMessage("hello stranger", true, lists).empty());
}
TEST(ChatFilterCoreTest, ShippedBlockListIsPortable) {
const auto parsed = dChatFilterDCF::ReadFile(std::string(DLU_SOURCE_DIR) + "/resources/blocklist.dcf");
ASSERT_EQ(parsed.status, dChatFilterDCF::eStatus::OK);
Lists lists;
lists.denied = parsed.list;
EXPECT_EQ(CheckMessage("what crap", false, lists), (Spans{ { 5, 4 } }));
EXPECT_TRUE(CheckMessage("hello there", false, lists).empty());
}

View File

@@ -43,7 +43,10 @@ TEST(ChatFilterWordsTest, FilterWord) {
EXPECT_EQ(ModerationTools::FilterWord(" Hello! "), "hello"); EXPECT_EQ(ModerationTools::FilterWord(" Hello! "), "hello");
EXPECT_EQ(ModerationTools::FilterWord("W.o,r;d?"), "word"); EXPECT_EQ(ModerationTools::FilterWord("W.o,r;d?"), "word");
EXPECT_FALSE(ModerationTools::FilterWord("")); EXPECT_FALSE(ModerationTools::FilterWord(""));
EXPECT_FALSE(ModerationTools::FilterWord("two words")); // Phrases: words normalized and joined by one space
EXPECT_EQ(ModerationTools::FilterWord(" Two Words! "), "two words");
EXPECT_TRUE(ModerationTools::IsPhrase("two words"));
EXPECT_FALSE(ModerationTools::IsPhrase("word"));
EXPECT_FALSE(ModerationTools::FilterWord("!!!")); EXPECT_FALSE(ModerationTools::FilterWord("!!!"));
EXPECT_FALSE(ModerationTools::FilterWord(std::string(65, 'a'))); EXPECT_FALSE(ModerationTools::FilterWord(std::string(65, 'a')));
} }
@@ -53,6 +56,11 @@ TEST(ChatFilterWordsTest, HasFilterWord) {
EXPECT_TRUE(ModerationTools::HasFilterWord("bad", "bad")); EXPECT_TRUE(ModerationTools::HasFilterWord("bad", "bad"));
EXPECT_FALSE(ModerationTools::HasFilterWord("badger badminton", "bad")); EXPECT_FALSE(ModerationTools::HasFilterWord("badger badminton", "bad"));
EXPECT_FALSE(ModerationTools::HasFilterWord("", "bad")); EXPECT_FALSE(ModerationTools::HasFilterWord("", "bad"));
// Phrases: the words in a row, whatever the spaces and punctuation between them
EXPECT_TRUE(ModerationTools::HasFilterWord("well, Bad Phrase!", "bad phrase"));
EXPECT_TRUE(ModerationTools::HasFilterWord("bad ... phrase", "bad phrase"));
EXPECT_FALSE(ModerationTools::HasFilterWord("bad other phrase", "bad phrase"));
EXPECT_FALSE(ModerationTools::HasFilterWord("phrase bad", "bad phrase"));
} }
TEST(ChatFilterWordsTest, FileWords) { TEST(ChatFilterWordsTest, FileWords) {
@@ -61,27 +69,6 @@ TEST(ChatFilterWordsTest, FileWords) {
ASSERT_TRUE(ModerationTools::FileWords("").empty()); ASSERT_TRUE(ModerationTools::FileWords("").empty());
} }
TEST(ChatFilterWordsTest, DcfHashes) {
const std::vector<size_t> hashes{ ModerationTools::WordHash("badword"), 42 };
std::string bytes(sizeof(dChatFilterDCF::fileHeader) + sizeof(size_t) * (hashes.size() + 1), '\0');
const dChatFilterDCF::fileHeader header{ dChatFilterDCF::header, dChatFilterDCF::formatVersion };
const size_t count = hashes.size();
std::memcpy(bytes.data(), &header, sizeof(header));
std::memcpy(bytes.data() + sizeof(header), &count, sizeof(count));
std::memcpy(bytes.data() + sizeof(header) + sizeof(count), hashes.data(), sizeof(size_t) * count);
ASSERT_EQ(ModerationTools::DcfHashes(bytes), hashes);
// Wrong header, other version, or fewer hashes than it says
auto wrong = bytes;
wrong[0] = 'X';
ASSERT_FALSE(ModerationTools::DcfHashes(wrong).has_value());
auto version = bytes;
version[sizeof(uint32_t)] = 9;
ASSERT_FALSE(ModerationTools::DcfHashes(version).has_value());
ASSERT_FALSE(ModerationTools::DcfHashes(bytes.substr(0, bytes.size() - sizeof(size_t) * 2)).has_value());
ASSERT_FALSE(ModerationTools::DcfHashes("DCFB").has_value());
}
namespace { namespace {
ModerationTools::WordSources Sources(bool blockFileLoaded = true) { ModerationTools::WordSources Sources(bool blockFileLoaded = true) {
ModerationTools::WordSources sources; ModerationTools::WordSources sources;
@@ -121,3 +108,22 @@ TEST(ChatFilterWordsTest, ExplainFreeChat) {
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("zzz darn", false, Sources(false))), EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("zzz darn", false, Sources(false))),
(std::vector<std::string>{ "x no_block_file", "x no_block_file" })); (std::vector<std::string>{ "x no_block_file", "x no_block_file" }));
} }
TEST(ChatFilterWordsTest, ExplainPhrases) {
auto sources = Sources();
sources.dashboard = [](const std::string& w) -> std::optional<bool> {
if (w == "no way") return false;
return std::nullopt;
};
sources.allowFile = [](const std::string& w) { return w == "hello" || w == "no" || w == "way" || w == "rude"; };
sources.blockFile = [](const std::string& w) { return w == "very rude"; };
sources.maxWords = 2;
// A phrase blocked here stops each of its words (and the empty piece between two spaces inside it), in normal chat too
auto verdicts = ModerationTools::ExplainMessage("hello No way!", true, sources);
EXPECT_EQ(Reasons(verdicts), (std::vector<std::string>{ "ok allow_file", "x blocked_here", "x blocked_here", "x blocked_here" }));
EXPECT_EQ(verdicts[1].phrase, "no way");
EXPECT_EQ(verdicts[3].text, "way!");
// The block file's phrases only in free chat
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("very rude", false, sources)), (std::vector<std::string>{ "x block_file", "x block_file" }));
EXPECT_EQ(Reasons(ModerationTools::ExplainMessage("rude very", false, sources)), (std::vector<std::string>{ "ok not_in_block_file", "ok not_in_block_file" }));
}