mirror of
https://github.com/DarkflameUniverse/DarkflameServer.git
synced 2026-10-02 02:43:44 +00:00
feat: live updates move every server onto a new build without a restart
With new binaries in place, master moves everything onto them while the server keeps running (the dashboard's Live update, /liveupdate or SIGUSR2 to master): - database migrations of the new build first; a failure stops there - UGC finishes the jobs it is running (queued rows stay pending), auth restarts, chat hands its teams to master for the next chat server; master starts the new processes and retries ones that don't come back - once the new chat server is up (CHAT_SERVER_READY) every world connects at once and sends its players again (LoginSessionNotify resync, no login logged) - every world instance is replaced with an instance migration: public worlds and private ones (same password) at once, properties after the old instance saved and froze the property (MIGRATE_PREPARE: no building, claiming or saving there any more), activity zones and character select once their players left or after a wait; empty instances just stop, zones in prestart_worlds get a new one first - the dashboard restarts last and picks the status up again Players land where they stood (position carried in CarriedPlayerState, also on properties and Moon Base). Draining instances get no new players (InstanceMigration::AcceptsNewPlayers) and show as "Moving players" in the world list. The order lives in LiveUpdateMachine.h without master state and is unit tested; master's glue is LiveUpdateCoordinator. Master itself is not replaced. Message IDs are appended only. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -150,6 +150,35 @@ void UgcProcessor::Stop() {
|
||||
m_Threads.clear();
|
||||
}
|
||||
|
||||
void UgcProcessor::Drain() {
|
||||
if (m_Draining) return;
|
||||
m_Draining = true;
|
||||
std::deque<Job> dropped;
|
||||
{
|
||||
std::lock_guard lock(m_Mutex);
|
||||
dropped.swap(m_Jobs);
|
||||
}
|
||||
// What was waiting for them is forgotten too; the database rows stay pending
|
||||
for (const auto& job : dropped) {
|
||||
if (job.kind == Kind::MODEL) {
|
||||
m_InFlight.erase({ job.kind, job.id });
|
||||
continue;
|
||||
}
|
||||
m_ComboJobs.erase(job.id);
|
||||
if (const auto rows = m_ComboRows.find(job.id); rows != m_ComboRows.end()) {
|
||||
for (const auto& [row, attempts] : rows->second) m_InFlight.erase({ Kind::MODULAR, row });
|
||||
m_ComboRows.erase(rows);
|
||||
}
|
||||
}
|
||||
LOG("Draining: %zu queued job(s) left for the next UGC server, waiting for the running ones", dropped.size());
|
||||
}
|
||||
|
||||
bool UgcProcessor::Drained() const {
|
||||
if (!m_Draining) return false;
|
||||
std::lock_guard lock(m_Mutex);
|
||||
return m_Active == 0 && m_Jobs.empty() && m_Done.empty();
|
||||
}
|
||||
|
||||
void UgcProcessor::Worker() {
|
||||
int appliedNice = 0;
|
||||
while (true) {
|
||||
@@ -546,6 +575,8 @@ void UgcProcessor::Backfill() {
|
||||
|
||||
void UgcProcessor::Update() {
|
||||
Collect();
|
||||
// Live update: only the running jobs' outcomes are recorded
|
||||
if (m_Draining) return;
|
||||
Backfill();
|
||||
const auto now = std::chrono::steady_clock::now();
|
||||
if (now - m_CpuSampled >= std::chrono::seconds(2)) SampleUsage();
|
||||
|
||||
@@ -61,6 +61,15 @@ public:
|
||||
// Waits for the jobs that are running; queued ones are dropped (they stay pending in the database)
|
||||
void Stop();
|
||||
|
||||
/**
|
||||
* Main thread, live updates: take no new work. Queued jobs are dropped (they stay pending in the database, so the
|
||||
* next UGC server makes them); the running ones finish and their outcome is recorded by Update as usual.
|
||||
*/
|
||||
void Drain();
|
||||
// Main thread: draining and nothing running or waiting to be recorded any more
|
||||
bool Drained() const;
|
||||
bool IsDraining() const { return m_Draining; }
|
||||
|
||||
// Main thread: poll, dispatch, record results, keep the storage under its cap
|
||||
void Update();
|
||||
|
||||
@@ -219,6 +228,7 @@ private:
|
||||
double m_CpuSeconds{};
|
||||
double m_CpuPercent{};
|
||||
bool m_Stopping{};
|
||||
bool m_Draining{}; // main thread: live update, no new work (Drain)
|
||||
std::vector<std::thread> m_Threads;
|
||||
|
||||
// Main thread only
|
||||
|
||||
@@ -697,10 +697,22 @@ int main(int argc, char** argv) {
|
||||
|
||||
Packet* packet = g_Server->ReceiveFromMaster();
|
||||
while (packet) {
|
||||
// Live update (docs/LiveUpdate.md): finish the running jobs, then stop; master starts the new build's server
|
||||
RakNet::BitStream inStream(packet->data, packet->length, false);
|
||||
LUBitStream header;
|
||||
if (header.ReadHeader(inStream) && header.connectionType == ServiceType::MASTER &&
|
||||
static_cast<MessageType::Master>(header.internalPacketID) == MessageType::Master::LIVE_UPDATE_RETIRE && !processor.IsDraining()) {
|
||||
LOG("Live update: finishing the running jobs, then stopping");
|
||||
processor.Drain();
|
||||
}
|
||||
g_Server->DeallocateMasterPacket(packet);
|
||||
packet = g_Server->ReceiveFromMaster();
|
||||
}
|
||||
processor.Update();
|
||||
if (processor.Drained()) {
|
||||
LOG("Live update: the running jobs are done; stopping");
|
||||
Game::lastSignal = -1;
|
||||
}
|
||||
// Worlds showing a model whose mesh changed tell their clients (docs/UgcServer.md, "Models without 3D services")
|
||||
if (auto changed = processor.TakeChangedMeshes(); !changed.empty()) {
|
||||
for (size_t start = 0; start < changed.size(); start += UgcModelsMade::MAX_MODELS) {
|
||||
|
||||
Reference in New Issue
Block a user