mirror of
https://github.com/DarkflameUniverse/DarkflameServer.git
synced 2026-10-02 10:53:44 +00:00
fix(chat-filter): portable .dcf hashing, block list phrases
The filter stored and compared words by std::hash<std::string> in size_t, which differs between standard libraries and platforms, so a .dcf made on one system never matched on another and the block list never worked there (issue 215). - ChatFilterCore.h: 64-bit FNV-1a over the entry's bytes, ASCII lower case, fixed-width uint64_t everywhere stored or compared. - .dcf version 3: little-endian magic, version, longest entry in words, uint64 count and sorted uint64 hashes. Version 2 files are refused: the allowed words cache is rebuilt from its .txt, an old blocklist.dcf is logged as unreadable. - The servers build blocklist.dcf from a plain blocklist.txt next to them (one word or phrase per line) when it is newer. - Blocked entries can be phrases: runs of consecutive words up to the longest entry, the whole run marked. Whitelist chat still checks one word at a time, as the client does. - Dashboard: the chat filter API reads blocklist.dcf the same way (status, phrase length), accepts blocked phrases, refuses allowed ones, and explains phrase matches in its message test. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -8,23 +8,26 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "dChatFilter.h"
|
||||
#include "ChatFilterCore.h"
|
||||
|
||||
// Words for the chat filter page, compared the way the chat filter compares them (see ModerationTools.h)
|
||||
namespace ModerationTools {
|
||||
constexpr size_t MAX_FILTER_WORD = 64;
|
||||
|
||||
// A word staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,); nullopt if it isn't
|
||||
// one word of 1-64 characters. Pure; unit tested.
|
||||
// A word or phrase staff typed for the chat filter, as the filter compares it (lower case, no ! ? ; . ,, words joined by
|
||||
// one space); nullopt if it isn't 1-64 characters or has no word. Pure; unit tested.
|
||||
inline std::optional<std::string> FilterWord(std::string text) {
|
||||
text.erase(0, text.find_first_not_of(" \t\r\n"));
|
||||
text.erase(text.find_last_not_of(" \t\r\n") + 1);
|
||||
if (text.empty() || text.size() > MAX_FILTER_WORD || text.find_first_of(" \t\r\n") != std::string::npos) return std::nullopt;
|
||||
auto word = dChatFilter::NormalizeWord(text);
|
||||
if (word.empty()) return std::nullopt;
|
||||
return word;
|
||||
if (text.empty() || text.size() > MAX_FILTER_WORD) return std::nullopt;
|
||||
auto entry = ChatFilterWords::NormalizeEntry(text);
|
||||
if (entry.empty()) return std::nullopt;
|
||||
return entry;
|
||||
}
|
||||
|
||||
// Whether an entry is a phrase (more than one word). Phrases can only be blocked: whitelist chat checks one word at a time.
|
||||
inline bool IsPhrase(const std::string& entry) { return ChatFilterWords::WordCount(entry) > 1; }
|
||||
|
||||
// The words of a plain word list (chatplus_en_us.txt) as the filter reads them: one per line, lower case; sorted, each once
|
||||
inline std::vector<std::string> FileWords(const std::string& text) {
|
||||
std::vector<std::string> words;
|
||||
@@ -33,7 +36,7 @@ namespace ModerationTools {
|
||||
const auto end = std::min(text.find('\n', start), text.size());
|
||||
auto line = text.substr(start, end - start);
|
||||
std::erase(line, '\r');
|
||||
std::transform(line.begin(), line.end(), line.begin(), ::tolower);
|
||||
line = ChatFilterWords::AsciiLower(std::move(line));
|
||||
if (!line.empty()) words.push_back(std::move(line));
|
||||
start = end + 1;
|
||||
}
|
||||
@@ -42,28 +45,12 @@ namespace ModerationTools {
|
||||
return words;
|
||||
}
|
||||
|
||||
// The hashes of a .dcf word list (blocklist.dcf), as dChatFilter::ReadWordlistDCF reads them; nullopt if it isn't one
|
||||
inline std::optional<std::vector<size_t>> DcfHashes(const std::string& bytes) {
|
||||
dChatFilterDCF::fileHeader header{};
|
||||
size_t count = 0;
|
||||
if (bytes.size() < sizeof(header) + sizeof(count)) return std::nullopt;
|
||||
std::memcpy(&header, bytes.data(), sizeof(header));
|
||||
if (header.header != dChatFilterDCF::header || header.formatVersion != dChatFilterDCF::formatVersion) return std::nullopt;
|
||||
std::memcpy(&count, bytes.data() + sizeof(header), sizeof(count));
|
||||
const size_t offset = sizeof(header) + sizeof(count);
|
||||
if (count > (bytes.size() - offset) / sizeof(size_t)) return std::nullopt;
|
||||
std::vector<size_t> hashes(count);
|
||||
if (count) std::memcpy(hashes.data(), bytes.data() + offset, count * sizeof(size_t));
|
||||
return hashes;
|
||||
}
|
||||
|
||||
// A word's hash as the filter stores it (dChatFilter::CalculateHash)
|
||||
inline size_t WordHash(const std::string& word) { return std::hash<std::string>{}(word); }
|
||||
|
||||
// Whether a message contains `word` as one of the words the chat filter checks
|
||||
inline bool HasFilterWord(const std::string& message, const std::string& word) {
|
||||
const auto words = dChatFilter::Words(message);
|
||||
return std::find(words.begin(), words.end(), word) != words.end();
|
||||
// Whether a message contains a word or phrase (FilterWord) as the chat filter reads it: whole words, in a row, skipping
|
||||
// pieces that are only punctuation
|
||||
inline bool HasFilterWord(const std::string& message, const std::string& entry) {
|
||||
const auto tokens = ChatFilterWords::Tokenize(message);
|
||||
const auto words = ChatFilterWords::WordCount(entry);
|
||||
return !ChatFilterWords::FindBlocked(tokens, words, [&entry](const std::string& run) { return run == entry; }).empty();
|
||||
}
|
||||
|
||||
// What the filter decides about one word of a message, and why
|
||||
@@ -72,6 +59,7 @@ namespace ModerationTools {
|
||||
std::string word; // as the filter compares it
|
||||
bool stopped{};
|
||||
std::string reason; // blocked_here, allowed_here, allow_file, character_name, not_allowed, block_file, not_in_block_file, no_block_file
|
||||
std::string phrase; // the blocked phrase this word is part of (blocked_here or block_file), when it was a phrase
|
||||
};
|
||||
|
||||
// Where the filter finds its words (callbacks keep this pure; the route reads the files and the database)
|
||||
@@ -81,33 +69,49 @@ namespace ModerationTools {
|
||||
std::function<bool(const std::string&)> characterName; // approved character names count as allowed words
|
||||
std::function<bool(const std::string&)> blockFile; // blocklist.dcf (by hash)
|
||||
bool blockFileLoaded{};
|
||||
uint32_t maxWords{ 1 }; // the longest blocked phrase, in words (here or in the file)
|
||||
};
|
||||
|
||||
/**
|
||||
* Each word of a message with what dChatFilter::IsSentenceOkay decides about it for a player below GM level 2 (higher
|
||||
* levels skip the filter). Normal chat (allowList) needs every word allowed; best friends' free chat (!allowList) stops
|
||||
* only blocked words, or every word when there is no blocked words file. Words are split at spaces as the filter
|
||||
* splits them. Pure; unit tested.
|
||||
* levels skip the filter). Blocked words and phrases (here always, the block file's in free chat) are stopped; a phrase
|
||||
* stops each of its words. Normal chat (allowList) needs every other word allowed, one at a time; best friends' free
|
||||
* chat stops only blocked ones, or every word when there is no blocked words file. Words are split at spaces as the
|
||||
* filter splits them (ChatFilterWords::CheckMessage). Pure; unit tested.
|
||||
*/
|
||||
inline std::vector<WordVerdict> ExplainMessage(const std::string& message, bool allowList, const WordSources& sources) {
|
||||
const auto tokens = ChatFilterWords::Tokenize(message);
|
||||
std::vector<WordVerdict> verdicts;
|
||||
std::stringstream stream(message);
|
||||
std::string segment;
|
||||
while (std::getline(stream, segment, ' ')) {
|
||||
WordVerdict verdict{ segment, dChatFilter::NormalizeWord(segment) };
|
||||
const auto here = sources.dashboard(verdict.word);
|
||||
if (!allowList && !sources.blockFileLoaded) {
|
||||
for (const auto& token : tokens) verdicts.push_back({ message.substr(token.position, token.length), token.word });
|
||||
if (!allowList && !sources.blockFileLoaded) {
|
||||
for (auto& verdict : verdicts) {
|
||||
verdict.stopped = true;
|
||||
verdict.reason = "no_block_file";
|
||||
} else if (here && !*here) {
|
||||
verdict.stopped = true;
|
||||
verdict.reason = "blocked_here";
|
||||
} else if (!allowList) {
|
||||
verdict.stopped = sources.blockFile(verdict.word);
|
||||
verdict.reason = verdict.stopped ? "block_file" : "not_in_block_file";
|
||||
}
|
||||
return verdicts;
|
||||
}
|
||||
|
||||
const auto blockedHere = [&sources](const std::string& entry) { const auto here = sources.dashboard(entry); return here && !*here; };
|
||||
const auto matches = ChatFilterWords::FindBlocked(tokens, std::max(sources.maxWords, 1u), [&](const std::string& entry) {
|
||||
return blockedHere(entry) || (!allowList && sources.blockFile(entry));
|
||||
});
|
||||
for (const auto& match : matches) {
|
||||
const auto reason = blockedHere(match.entry) ? "blocked_here" : "block_file";
|
||||
for (size_t i = match.first; i <= match.last; i++) {
|
||||
verdicts[i].stopped = true;
|
||||
verdicts[i].reason = reason;
|
||||
if (ChatFilterWords::WordCount(match.entry) > 1) verdicts[i].phrase = match.entry;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto& verdict : verdicts) {
|
||||
if (verdict.stopped) continue;
|
||||
const auto here = sources.dashboard(verdict.word);
|
||||
if (!allowList) {
|
||||
verdict.reason = "not_in_block_file";
|
||||
} else if (sources.allowFile(verdict.word)) {
|
||||
verdict.reason = "allow_file";
|
||||
} else if (here) {
|
||||
} else if (here && *here) {
|
||||
verdict.reason = "allowed_here";
|
||||
} else if (sources.characterName(verdict.word)) {
|
||||
verdict.reason = "character_name";
|
||||
@@ -115,7 +119,6 @@ namespace ModerationTools {
|
||||
verdict.stopped = true;
|
||||
verdict.reason = "not_allowed";
|
||||
}
|
||||
verdicts.push_back(std::move(verdict));
|
||||
}
|
||||
return verdicts;
|
||||
}
|
||||
|
||||
@@ -31,7 +31,8 @@ namespace {
|
||||
// The chat filter's files, as the servers load them: the allowed words from the client's res folder, the blocked
|
||||
// words (only their hashes) next to the servers
|
||||
constexpr const char* ALLOW_FILE = "chatplus_en_us.txt";
|
||||
constexpr const char* BLOCK_FILE = "blocklist.dcf";
|
||||
constexpr const char* BLOCK_FILE = dChatFilterDCF::BLOCK_LIST_FILE;
|
||||
constexpr const char* BLOCK_TEXT = dChatFilterDCF::BLOCK_LIST_TEXT;
|
||||
constexpr uint32_t FILE_WORDS_PAGE = 200;
|
||||
// Recent chat searched when checking what a word would change
|
||||
constexpr uint32_t CHECK_MESSAGES = 1000;
|
||||
@@ -216,11 +217,8 @@ namespace {
|
||||
return text ? ModerationTools::FileWords(*text) : std::vector<std::string>{};
|
||||
}
|
||||
|
||||
std::optional<std::vector<size_t>> BlockFileHashes() {
|
||||
std::ifstream in(BLOCK_FILE, std::ios::binary);
|
||||
if (!in) return std::nullopt;
|
||||
return ModerationTools::DcfHashes(std::string((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>()));
|
||||
}
|
||||
// blocklist.dcf as the servers read it (an old-format file is refused, as they refuse it)
|
||||
dChatFilterDCF::ParseResult BlockFile() { return dChatFilterDCF::ReadFile(BLOCK_FILE); }
|
||||
|
||||
// The dashboard's own lists by word: true allowed, false blocked
|
||||
std::map<std::string, bool> DashboardWords() {
|
||||
@@ -258,25 +256,27 @@ namespace {
|
||||
const auto it = dashboard.find(word);
|
||||
words.push_back({ {"word", word}, {"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
|
||||
}
|
||||
const auto blocked = BlockFileHashes();
|
||||
const auto blocked = BlockFile();
|
||||
uint32_t imported = 0;
|
||||
for (const auto& word : all) if (dashboard.contains(word)) imported++;
|
||||
JsonSuccess(reply, { {"allowFile", ALLOW_FILE}, {"allowFileFound", !all.empty()}, {"allowTotal", all.size()}, {"matched", matched},
|
||||
{"start", start}, {"pageSize", FILE_WORDS_PAGE}, {"words", words}, {"onDashboard", imported},
|
||||
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.has_value()}, {"blockTotal", blocked ? blocked->size() : 0} });
|
||||
{"blockFile", BLOCK_FILE}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK}, {"blockTotal", blocked.list.Size()},
|
||||
{"blockMaxWords", blocked.list.maxWords}, {"blockFileStatus", dChatFilterDCF::StatusText(blocked.status)},
|
||||
{"blockFileOld", blocked.status == dChatFilterDCF::eStatus::OLD_FORMAT}, {"blockText", BLOCK_TEXT} });
|
||||
});
|
||||
|
||||
Route(eHTTPMethod::GET, "/api/chat_filter/lookup", Perm("chat_filter_manage"),
|
||||
"Where a word stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
|
||||
"Where a word or phrase stands: in the allowed words file, in the blocked words file (by its hash), and on the dashboard's lists. Query: word",
|
||||
[](HTTPReply& reply, const HTTPContext& context) {
|
||||
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
|
||||
const auto all = AllowFileWords();
|
||||
const auto blocked = BlockFileHashes();
|
||||
const auto blocked = BlockFile();
|
||||
const auto dashboard = DashboardWords();
|
||||
const auto it = dashboard.find(*word);
|
||||
JsonSuccess(reply, { {"word", *word}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
|
||||
{"inBlockFile", blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(*word)) != blocked->end()},
|
||||
JsonSuccess(reply, { {"word", *word}, {"phrase", ModerationTools::IsPhrase(*word)}, {"inAllowFile", std::binary_search(all.begin(), all.end(), *word)},
|
||||
{"inBlockFile", blocked.list.Contains(*word)},
|
||||
{"dashboard", it == dashboard.end() ? nlohmann::json(nullptr) : nlohmann::json(it->second ? "allowed" : "blocked")} });
|
||||
});
|
||||
|
||||
@@ -293,7 +293,7 @@ namespace {
|
||||
for (const auto& word : all) {
|
||||
// Only words the filter could compare (the file's odd lines with spaces or punctuation stay in the file only)
|
||||
const auto filterWord = ModerationTools::FilterWord(word);
|
||||
if (!filterWord || *filterWord != word || dashboard.contains(word)) continue;
|
||||
if (!filterWord || *filterWord != word || ModerationTools::IsPhrase(word) || dashboard.contains(word)) continue;
|
||||
Database::Get()->SetChatFilterWord({ word, true, context.authenticatedUser, now });
|
||||
added++;
|
||||
}
|
||||
@@ -305,13 +305,17 @@ namespace {
|
||||
});
|
||||
|
||||
Route(eHTTPMethod::POST, "/api/chat_filter/words", Perm("chat_filter_manage"),
|
||||
"Allow or block a word (or move it to the other list); running worlds pick it up at once. Body: {word, allowed: bool}",
|
||||
"Allow or block a word, or block a phrase (or move it to the other list); running worlds pick it up at once. Phrases can't be allowed: "
|
||||
"whitelist chat checks each word on its own, as the client does. Body: {word, allowed: bool}",
|
||||
[](HTTPReply& reply, const HTTPContext& context) {
|
||||
const auto body = ParseBody(context);
|
||||
if (!body) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Invalid JSON");
|
||||
const auto word = ModerationTools::FilterWord(body->value("word", ""));
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
|
||||
const bool allowed = body->value("allowed", false);
|
||||
if (allowed && ModerationTools::IsPhrase(*word)) {
|
||||
return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Phrases can only be blocked: normal chat checks each word on its own, so allow the words instead");
|
||||
}
|
||||
Database::Get()->SetChatFilterWord({ *word, allowed, context.authenticatedUser, static_cast<int64_t>(std::time(nullptr)) });
|
||||
Audit(context, allowed ? "chat_filter_allow" : "chat_filter_block", (allowed ? "Allowed \"" : "Blocked \"") + *word + "\" in chat");
|
||||
BroadcastTableChanged("chat_filter");
|
||||
@@ -345,12 +349,11 @@ namespace {
|
||||
if (message.empty() || message.size() > 300) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a message of 1 to 300 characters");
|
||||
const bool allowList = QueryValue(context.queryString, "chat") != "free";
|
||||
const auto all = AllowFileWords();
|
||||
const auto blocked = BlockFileHashes();
|
||||
const auto blocked = BlockFile();
|
||||
const auto dashboard = DashboardWords();
|
||||
std::set<std::string> names;
|
||||
for (auto name : Database::Get()->GetApprovedCharacterNames()) {
|
||||
std::transform(name.begin(), name.end(), name.begin(), ::tolower);
|
||||
names.insert(std::move(name));
|
||||
names.insert(ChatFilterWords::AsciiLower(std::move(name)));
|
||||
}
|
||||
ModerationTools::WordSources sources;
|
||||
sources.dashboard = [&dashboard](const std::string& word) -> std::optional<bool> {
|
||||
@@ -359,18 +362,18 @@ namespace {
|
||||
};
|
||||
sources.allowFile = [&all](const std::string& word) { return std::binary_search(all.begin(), all.end(), word); };
|
||||
sources.characterName = [&names](const std::string& word) { return names.contains(word); };
|
||||
sources.blockFile = [&blocked](const std::string& word) {
|
||||
return blocked && std::find(blocked->begin(), blocked->end(), ModerationTools::WordHash(word)) != blocked->end();
|
||||
};
|
||||
sources.blockFileLoaded = blocked && !blocked->empty();
|
||||
sources.blockFile = [&blocked](const std::string& entry) { return blocked.list.Contains(entry); };
|
||||
sources.blockFileLoaded = !blocked.list.Empty();
|
||||
sources.maxWords = blocked.list.maxWords;
|
||||
for (const auto& [entry, allowed] : dashboard) if (!allowed) sources.maxWords = std::max(sources.maxWords, ChatFilterWords::WordCount(entry));
|
||||
nlohmann::json words = nlohmann::json::array();
|
||||
bool stopped = false;
|
||||
for (const auto& verdict : ModerationTools::ExplainMessage(message, allowList, sources)) {
|
||||
stopped |= verdict.stopped;
|
||||
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason} });
|
||||
words.push_back({ {"text", verdict.text}, {"word", verdict.word}, {"stopped", verdict.stopped}, {"reason", verdict.reason}, {"phrase", verdict.phrase} });
|
||||
}
|
||||
JsonSuccess(reply, { {"message", message}, {"chat", allowList ? "normal" : "free"}, {"stopped", stopped}, {"words", words},
|
||||
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.has_value()} });
|
||||
{"allowFileFound", !all.empty()}, {"blockFileFound", blocked.status == dChatFilterDCF::eStatus::OK} });
|
||||
});
|
||||
|
||||
Route(eHTTPMethod::GET, "/api/chat_filter/check", Perm("chat_filter_manage"),
|
||||
@@ -379,7 +382,7 @@ namespace {
|
||||
[](HTTPReply& reply, const HTTPContext& context) {
|
||||
if (!Can(context, "chat_view")) return JsonError(reply, eHTTPStatusCode::FORBIDDEN, "Reading chat needs the chat_view permission");
|
||||
const auto word = ModerationTools::FilterWord(QueryValue(context.queryString, "word"));
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type one word (no spaces), up to 64 characters");
|
||||
if (!word) return JsonError(reply, eHTTPStatusCode::BAD_REQUEST, "Type a word or phrase, up to 64 characters");
|
||||
const bool allowed = QueryValue(context.queryString, "allowed") == "1";
|
||||
IChatLog::ChatQuery query;
|
||||
query.search = *word;
|
||||
|
||||
Reference in New Issue
Block a user