fix(chat-filter): portable .dcf hashing, block list phrases

The filter stored and compared words by std::hash<std::string> in size_t,
which differs between standard libraries and platforms, so a .dcf made on
one system never matched on another and the block list never worked
there (issue 215).

- ChatFilterCore.h: 64-bit FNV-1a over the entry's bytes, ASCII lower
  case, fixed-width uint64_t everywhere stored or compared.
- .dcf version 3: little-endian magic, version, longest entry in words,
  uint64 count and sorted uint64 hashes. Version 2 files are refused:
  the allowed words cache is rebuilt from its .txt, an old
  blocklist.dcf is logged as unreadable.
- The servers build blocklist.dcf from a plain blocklist.txt next to
  them (one word or phrase per line) when it is newer.
- Blocked entries can be phrases: runs of consecutive words up to the
  longest entry, the whole run marked. Whitelist chat still checks one
  word at a time, as the client does.
- Dashboard: the chat filter API reads blocklist.dcf the same way
  (status, phrase length), accepts blocked phrases, refuses allowed
  ones, and explains phrase matches in its message test.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Aaron Kimbrell
2026-09-30 08:05:00 -05:00
parent 3018312f29
commit c1bcda8dd1
8 changed files with 691 additions and 268 deletions

View File

@@ -8,23 +8,26 @@
#include <string>
#include <vector>
#include "dChatFilter.h"
#include "ChatFilterCore.h"
// Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h)
namespace ModerationTools {
constexpr size_t MAX_FILTER_WORD = 64;
// A word staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,); nullopt if it isn't
// one word of 1-64 characters. Pure; unit tested.
// A word or phrase staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,, words joined by
// one space); nullopt if it isn't 1-64 characters or has no word. Pure; unit tested.
inline std::optional<std::string> FilterWord(std::string text) {
text.erase(0, text.find_first_not_of(" \t\r\n"));
text.erase(text.find_last_not_of(" \t\r\n") + 1);
if (text.empty() || text.size() > MAX_FILTER_WORD || text.find_first_of(" \t\r\n") != std::string::npos) return std::nullopt;
auto word = dChatFilter::NormalizeWord(text);
if (word.empty()) return std::nullopt;
return word;
if (text.empty() || text.size() > MAX_FILTER_WORD) return std::nullopt;
auto entry = ChatFilterWords::NormalizeEntry(text);
if (entry.empty()) return std::nullopt;
return entry;
}
// Whether an entry is a phrase (more than one word). Phrases can only be blocked: whitelist chat checks one word at a time.
inline bool IsPhrase(const std::string& entry) { return ChatFilterWords::WordCount(entry) > 1; }
// The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once
inline std::vector<std::string> FileWords(const std::string& text) {
std::vector<std::string> words;
@@ -33,7 +36,7 @@ namespace ModerationTools {
const auto end = std::min(text.find('\n', start), text.size());
auto line = text.substr(start, end - start);
std::erase(line, '\r');
std::transform(line.begin(), line.end(), line.begin(), ::tolower);
line = ChatFilterWords::AsciiLower(std::move(line));
if (!line.empty()) words.push_back(std::move(line));
start = end + 1;
}
@@ -42,28 +45,12 @@ namespace ModerationTools {
return words;
}
// The hashes of a .dcf word list (blocklist.dcf), as dChatFilter::ReadWordlistDCF reads them; nullopt if it isn't one
inline std::optional<std::vector<size_t>> DcfHashes(const std::string& bytes) {
dChatFilterDCF::fileHeader header{};
size_t count = 0;
if (bytes.size() < sizeof(header) + sizeof(count)) return std::nullopt;
std::memcpy(&header, bytes.data(), sizeof(header));
if (header.header != dChatFilterDCF::header || header.formatVersion != dChatFilterDCF::formatVersion) return std::nullopt;
std::memcpy(&count, bytes.data() + sizeof(header), sizeof(count));
const size_t offset = sizeof(header) + sizeof(count);
if (count > (bytes.size() - offset) / sizeof(size_t)) return std::nullopt;
std::vector<size_t> hashes(count);
if (count) std::memcpy(hashes.data(), bytes.data() + offset, count * sizeof(size_t));
return hashes;
}
// A word's hash as the filter stores it (dChatFilter::CalculateHash)
inline size_t WordHash(const std::string& word) { return std::hash<std::string>{}(word); }
// Whether a message contains `word` as one of the words the chat filter checks
inline bool HasFilterWord(const std::string& message, const std::string& word) {
const auto words = dChatFilter::Words(message);
return std::find(words.begin(), words.end(), word) != words.end();
// Whether a message contains a word or phrase (FilterWord) as the chat filter reads it: whole words, in a row, skipping
// pieces that are only punctuation
inline bool HasFilterWord(const std::string& message, const std::string& entry) {
const auto tokens = ChatFilterWords::Tokenize(message);
const auto words = ChatFilterWords::WordCount(entry);
return !ChatFilterWords::FindBlocked(tokens, words, [&entry](const std::string& run) { return run == entry; }).empty();
}
// What the filter decides about one word of a message, and why
@@ -72,6 +59,7 @@ namespace ModerationTools {
std::string word; // as the filter compares it
bool stopped{};
std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file
std::string phrase; // the blocked phrase this word is part of (blocked_here or block_file), when it was a phrase
};
// Where the filter finds its words (callbacks keep this pure; the route reads the files and the database)
@@ -81,33 +69,49 @@ namespace ModerationTools {
std::function<bool(const std::string&)> characterName; // approved character names count as allowed words
std::function<bool(const std::string&)> blockFile; // blocklist.dcf (by hash)
bool blockFileLoaded{};
uint32_t maxWords{ 1 }; // the longest blocked phrase, in words (here or in the file)
};
/**
* Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher
* levels skip the filter). Normal chat (allowList) needs every word allowed; best friends' free chat (!allowList) stops
* only blocked words, or every word when there is no blocked words file. Words are split at spaces as the filter
* splits them. Pure; unit tested.
* levels skip the filter). Blocked words and phrases (here always, the block file's in free chat) are stopped; a phrase
* stops each of its words. Normal chat (allowList) needs every other word allowed, one at a time; best friends' free
* chat stops only blocked ones, or every word when there is no blocked words file. Words are split at spaces as the
* filter splits them (ChatFilterWords::CheckMessage). Pure; unit tested.
*/
inline std::vector<WordVerdict> ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) {
const auto tokens = ChatFilterWords::Tokenize(message);
std::vector<WordVerdict> verdicts;
std::stringstream stream(message);
std::string segment;
while (std::getline(stream, segment, ' ')) {
WordVerdict verdict{ segment, dChatFilter::NormalizeWord(segment) };
const auto here = sources.dashboard(verdict.word);
if (!allowList && !sources.blockFileLoaded) {
for (const auto& token : tokens) verdicts.push_back({ message.substr(token.position, token.length), token.word });
if (!allowList && !sources.blockFileLoaded) {
for (auto& verdict : verdicts) {
verdict.stopped = true;
verdict.reason = "no_block_file";
} else if (here && !*here) {
verdict.stopped = true;
verdict.reason = "blocked_here";
} else if (!allowList) {
verdict.stopped = sources.blockFile(verdict.word);
verdict.reason = verdict.stopped ? "block_file" : "not_in_block_file";
}
return verdicts;
}
const auto blockedHere = [&sources](const std::string& entry) { const auto here = sources.dashboard(entry); return here && !*here; };
const auto matches = ChatFilterWords::FindBlocked(tokens, std::max(sources.maxWords, 1u), [&](const std::string& entry) {
return blockedHere(entry) || (!allowList && sources.blockFile(entry));
});
for (const auto& match : matches) {
const auto reason = blockedHere(match.entry) ? "blocked_here" : "block_file";
for (size_t i = match.first; i <= match.last; i++) {
verdicts[i].stopped = true;
verdicts[i].reason = reason;
if (ChatFilterWords::WordCount(match.entry) > 1) verdicts[i].phrase = match.entry;
}
}
for (auto& verdict : verdicts) {
if (verdict.stopped) continue;
const auto here = sources.dashboard(verdict.word);
if (!allowList) {
verdict.reason = "not_in_block_file";
} else if (sources.allowFile(verdict.word)) {
verdict.reason = "allow_file";
} else if (here) {
} else if (here && *here) {
verdict.reason = "allowed_here";
} else if (sources.characterName(verdict.word)) {
verdict.reason = "character_name";
@@ -115,7 +119,6 @@ namespace ModerationTools {
verdict.stopped = true;
verdict.reason = "not_allowed";
}
verdicts.push_back(std::move(verdict));
}
return verdicts;
}

View File

@@ -31,7 +31,8 @@ namespace {
// The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked
// words (only their hashes) next to the servers
constexpr const char* ALLOW_FILE = "chatplus_en_us.txt";
constexpr const char* BLOCK_FILE = "blocklist.dcf";
constexpr const char* BLOCK_FILE = dChatFilterDCF::BLOCK_LIST_FILE;
constexpr const char* BLOCK_TEXT = dChatFilterDCF::BLOCK_LIST_TEXT;
constexpr uint32_t FILE_WORDS_PAGE = 200;
// Recent chat searched when checking what a word would change
constexpr uint32_t CHECK_MESSAGES = 1000;
@@ -216,11 +217,8 @@ namespace {
return text ? ModerationTools::FileWords(*text) : std::vector<std::string>{};
}
std::optional<std::vector<size_t>> BlockFileHashes() {
std::ifstream in(BLOCK_FILE, std::ios::binary);
if (!in) return std::nullopt;
return ModerationTools::DcfHashes(std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>()));
}
// blocklist.dcf as the servers read it (an old-format file is refused, as they refuse it)
dChatFilterDCF::ParseResult BlockFile() { return dChatFilterDCF::ReadFile(BLOCK_FILE); }
// The dashboard's own lists by word: true allowed, false blocked
std::map<std::string, bool> DashboardWords() {
@@ -258,25 +256,27 @@ namespace {
const auto it = dashboard.find(word);
words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
}
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
uint32_t imported = 0;
for (const auto& word : all) if (dashboard.contains(word)) imported++;
JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched},
{"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported},
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.has_value()}, {"blockTotal", blocked ? blocked->size() : 0} });
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK}, {"blockTotal", blocked.list.Size()},
{"blockMaxWords", blocked.list.maxWords}, {"blockFileStatus", dChatFilterDCF::StatusText(blocked.status)},
{"blockFileOld", blocked.status == dChatFilterDCF::eStatus::OLD_FORMAT}, {"blockText", BLOCK_TEXT} });
});
Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"),
"Where a word stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
"Where a word or phrase stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
[](HTTPReply& reply, const HTTPContext& context) {
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const auto all = AllowFileWords();
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
const auto dashboard = DashboardWords();
const auto it = dashboard.find(*word);
JsonSuccess(reply, { {"word", *word}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
{"inBlockFile", blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(*word)) != blocked->end()},
JsonSuccess(reply, { {"word", *word}, {"phrase", ModerationTools::IsPhrase(*word)}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
{"inBlockFile", blocked.list.Contains(*word)},
{"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
});
@@ -293,7 +293,7 @@ namespace {
for (const auto& word : all) {
// Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only)
const auto filterWord = ModerationTools::FilterWord(word);
if (!filterWord || *filterWord != word || dashboard.contains(word)) continue;
if (!filterWord || *filterWord != word || ModerationTools::IsPhrase(word) || dashboard.contains(word)) continue;
Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now });
added++;
}
@@ -305,13 +305,17 @@ namespace {
});
Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"),
"Allow or block a word (or move it to the other list); running worlds pick it up at once. Body: {word, allowed: bool}",
"Allow or block a word, or block a phrase (or move it to the other list); running worlds pick it up at once. Phrases can't be allowed: "
"whitelist chat checks each word on its own, as the client does. Body: {word, allowed: bool}",
[](HTTPReply& reply, const HTTPContext& context) {
const auto body = ParseBody(context);
if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON");
const auto word = ModerationTools::FilterWord(body->value("word", ""));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = body->value("allowed", false);
if (allowed && ModerationTools::IsPhrase(*word)) {
return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead");
}
Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast<int64_t>(std::time(nullptr)) });
Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat");
BroadcastTableChanged("chat_filter");
@@ -345,12 +349,11 @@ namespace {
if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters");
const bool allowList = QueryValue(context.queryString, "chat") != "free";
const auto all = AllowFileWords();
const auto blocked = BlockFileHashes();
const auto blocked = BlockFile();
const auto dashboard = DashboardWords();
std::set<std::string> names;
for (auto name : Database::Get()->GetApprovedCharacterNames()) {
std::transform(name.begin(), name.end(), name.begin(), ::tolower);
names.insert(std::move(name));
names.insert(ChatFilterWords::AsciiLower(std::move(name)));
}
ModerationTools::WordSources sources;
sources.dashboard = [&dashboard](const std::string& word) -> std::optional<bool> {
@@ -359,18 +362,18 @@ namespace {
};
sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); };
sources.characterName = [&names](const std::string& word) { return names.contains(word); };
sources.blockFile = [&blocked](const std::string& word) {
return blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(word)) != blocked->end();
};
sources.blockFileLoaded = blocked && !blocked->empty();
sources.blockFile = [&blocked](const std::string& entry) { return blocked.list.Contains(entry); };
sources.blockFileLoaded = !blocked.list.Empty();
sources.maxWords = blocked.list.maxWords;
for (const auto& [entry, allowed] : dashboard) if (!allowed) sources.maxWords = std::max(sources.maxWords, ChatFilterWords::WordCount(entry));
nlohmann::json words = nlohmann::json::array();
bool stopped = false;
for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) {
stopped |= verdict.stopped;
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason} });
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason}, {"phrase", verdict.phrase} });
}
JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words},
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.has_value()} });
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK} });
});
Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"),
@@ -379,7 +382,7 @@ namespace {
[](HTTPReply& reply, const HTTPContext& context) {
if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission");
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
const bool allowed = QueryValue(context.queryString, "allowed") == "1";
IChatLog::ChatQuery query;
query.search = *word;