fix(update): stop blocking event loop during rebuild + fix infinite spinner

The update endpoint ran the entire git pull + docker build (up to 10 min)
synchronously inside the request handler, blocking the whole server for
everyone while it ran. Separately, the frontend spinner was only hidden on
error, never on success, so it spun forever even when the update worked.

- Run the update in a background thread; the request returns immediately
- Add GET /settings/update/status for progress polling (backup/pull/build/restart)
- Frontend polls status, then waits for the app to come back after the
  container restart, and shows a clear done/timeout message instead of an
  endless spinner

Co-Authored-By: Claude Sonnet 5 <[email protected]>
This commit is contained in:
2026-07-23 15:09:42 +02:00
co-authored by Claude Sonnet 5
parent c5189d88fe
commit a5988af6a3
6 changed files with 163 additions and 13 deletions
+1 -1
View File
@@ -33,7 +33,7 @@ logger = logging.getLogger(__name__)
app = FastAPI(
title="NetBird MSP Appliance",
description="Multi-tenant NetBird management platform for MSPs",
version="1.1.1",
version="1.1.2",
docs_url="/api/docs",
redoc_url="/api/redoc",
openapi_url="/api/openapi.json",
+39 -10
View File
@@ -4,6 +4,7 @@ There is no .env file. Every setting lives in the ``system_config`` table
(singleton row with id=1) and is editable via the Web UI settings page.
"""
import asyncio
import logging
import os
import shutil
@@ -358,13 +359,15 @@ async def trigger_update(
current_user: User = Depends(get_current_user),
db: Session = Depends(get_db),
):
"""Backup the database, git pull the latest code, and rebuild the container.
"""Kick off backup + git pull + container rebuild in the background.
Returns immediately — the actual work (which can take several minutes,
especially the ``--no-cache`` image build) runs in a background thread so
it doesn't block this request or any other user's requests while it
runs. Progress can be polled via GET /settings/update/status until the
container restarts with the new version.
The rebuild is fire-and-forget — the app will restart in ~60 seconds.
Only admin users may trigger an update.
Returns:
Dict with ok, message, and backup path.
"""
if getattr(current_user, "role", "admin") != "admin":
raise HTTPException(
@@ -383,11 +386,37 @@ async def trigger_update(
detail="git_repo_url is not configured in settings.",
)
result = update_service.trigger_update(config, DATABASE_PATH)
if not result.get("ok"):
current_status = update_service.get_update_status()
if current_status.get("state") == "running":
raise HTTPException(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail=result.get("message", "Update failed."),
status_code=status.HTTP_409_CONFLICT,
detail="An update is already in progress.",
)
# Snapshot the only fields trigger_update() needs — avoids passing a
# SQLAlchemy instance into a background thread after this request's
# session may already be closed.
class _ConfigSnapshot:
git_repo_url = config.git_repo_url
git_branch = config.git_branch
git_token = config.git_token
asyncio.create_task(asyncio.to_thread(update_service.trigger_update, _ConfigSnapshot(), DATABASE_PATH))
logger.info("Update triggered by %s.", current_user.username)
return result
return {
"ok": True,
"message": "Update gestartet. Dies kann mehrere Minuten dauern — Fortschritt via Status sichtbar.",
}
@router.get("/update/status")
async def update_status(
current_user: User = Depends(get_current_user),
):
"""Return progress of the currently running (or last) update.
Note: once the container restarts mid-update, this endpoint stops
responding for a few seconds — that itself is a signal the swap is
happening. The frontend falls back to polling for the app coming back up.
"""
return update_service.get_update_status()
+42
View File
@@ -20,6 +20,32 @@ SERVICE_NAME = "netbird-msp-appliance"
logger = logging.getLogger(__name__)
# In-memory progress tracker for the currently running (or last) update.
# The container gets replaced mid-update, so this deliberately does NOT need
# to survive a restart — the frontend detects completion by polling until the
# app comes back up and reports a new version, not by reading a final status
# here. It exists so the UI can show *something* other than a frozen spinner
# while the backup/pull/build steps are in progress.
_update_status: dict[str, Any] = {
"state": "idle", # idle | running | failed
"step": "",
"message": "",
"started_at": None,
}
def get_update_status() -> dict[str, Any]:
"""Return a snapshot of the current update progress."""
return dict(_update_status)
def _set_status(state: str, step: str, message: str = "") -> None:
_update_status["state"] = state
_update_status["step"] = step
_update_status["message"] = message
if step == "backup":
_update_status["started_at"] = datetime.utcnow().isoformat()
def _get_compose_project_name() -> str:
"""Detect the compose project name from the running container's labels.
@@ -233,10 +259,12 @@ def trigger_update(config: Any, db_path: str) -> dict:
Dict with ok (bool), message, backup path, and pulled_branch.
"""
# 1. Backup database before any changes
_set_status("running", "backup", "Datenbank wird gesichert …")
try:
backup_path = backup_database(db_path)
except Exception as exc:
logger.error("Database backup failed: %s", exc)
_set_status("failed", "backup", f"Database backup failed: {exc}")
return {"ok": False, "message": f"Database backup failed: {exc}", "backup": None}
# 2. Build git pull command (embed token in URL if provided)
@@ -252,6 +280,7 @@ def trigger_update(config: Any, db_path: str) -> dict:
pull_cmd = ["git", "-C", SOURCE_DIR, "pull", "origin", branch]
# 3. Git pull (synchronous — must complete before rebuild)
_set_status("running", "pull", f"Code wird von Branch '{branch}' geholt …")
# Ensure .git directory is owned by the process user (root inside container).
# The .git dir may be owned by the host user after manual operations.
try:
@@ -270,13 +299,16 @@ def trigger_update(config: Any, db_path: str) -> dict:
timeout=120,
)
except subprocess.TimeoutExpired:
_set_status("failed", "pull", "git pull timed out after 120s.")
return {"ok": False, "message": "git pull timed out after 120s.", "backup": backup_path}
except Exception as exc:
_set_status("failed", "pull", f"git pull error: {exc}")
return {"ok": False, "message": f"git pull error: {exc}", "backup": backup_path}
if result.returncode != 0:
stderr = result.stderr.strip()[:500]
logger.error("git pull failed (exit %d): %s", result.returncode, stderr)
_set_status("failed", "pull", f"git pull failed: {stderr}")
return {
"ok": False,
"message": f"git pull failed: {stderr}",
@@ -348,6 +380,7 @@ def trigger_update(config: Any, db_path: str) -> dict:
SERVICE_NAME,
]
logger.info("Phase A: building new image …")
_set_status("running", "build", "Docker-Image wird gebaut (kann mehrere Minuten dauern) …")
try:
build_result = subprocess.run(
build_cmd,
@@ -360,15 +393,18 @@ def trigger_update(config: Any, db_path: str) -> dict:
f.write(build_result.stderr)
if build_result.returncode != 0:
logger.error("Image build failed: %s", build_result.stderr[:500])
_set_status("failed", "build", f"Image build failed: {build_result.stderr[:300]}")
return {
"ok": False,
"message": f"Image build failed: {build_result.stderr[:300]}",
"backup": backup_path,
}
except subprocess.TimeoutExpired:
_set_status("failed", "build", "Image build timed out after 600s.")
return {"ok": False, "message": "Image build timed out after 600s.", "backup": backup_path}
logger.info("Phase A complete — image built successfully.")
_set_status("running", "restart", "Container wird neu gestartet …")
# Phase B — swap the container using a helper container.
# When compose recreates our container, ALL processes inside die (PID namespace
@@ -388,6 +424,7 @@ def trigger_update(config: Any, db_path: str) -> dict:
raise ValueError("Could not find /app-source mount")
except Exception as exc:
logger.error("Failed to discover host source path: %s", exc)
_set_status("failed", "restart", f"Could not find host source path: {exc}")
return {"ok": False, "message": f"Could not find host source path: {exc}", "backup": backup_path}
logger.info("Host source directory: %s", host_source_dir)
@@ -426,6 +463,10 @@ def trigger_update(config: Any, db_path: str) -> dict:
)
if result.returncode != 0:
logger.error("Failed to start updater container: %s", result.stderr.strip())
_set_status(
"failed", "restart",
f"Update-Container konnte nicht gestartet werden: {result.stderr.strip()[:200]}",
)
return {
"ok": False,
"message": f"Update-Container konnte nicht gestartet werden: {result.stderr.strip()[:200]}",
@@ -434,6 +475,7 @@ def trigger_update(config: Any, db_path: str) -> dict:
logger.info("Phase B: updater container started — this container will restart in ~5s.")
except Exception as exc:
logger.error("Failed to launch updater: %s", exc)
_set_status("failed", "restart", f"Updater launch failed: {exc}")
return {"ok": False, "message": f"Updater launch failed: {exc}", "backup": backup_path}
return {
+65 -2
View File
@@ -1374,11 +1374,12 @@ async function loadVersionInfo() {
if (needsUpdate) {
html += `<div class="mt-3">
<button class="btn btn-warning" onclick="triggerUpdate()">
<button class="btn btn-warning" id="update-trigger-btn" onclick="triggerUpdate()">
<span class="spinner-border spinner-border-sm d-none me-1" id="update-spinner"></span>
<i class="bi bi-arrow-repeat me-1"></i>${t('settings.triggerUpdate')}
</button>
<div class="text-muted small mt-1">${t('settings.updateWarning')}</div>
<div class="small mt-2 d-none" id="update-progress-text"></div>
</div>`;
}
el.innerHTML = html;
@@ -1390,14 +1391,76 @@ async function loadVersionInfo() {
async function triggerUpdate() {
if (!confirm(t('settings.confirmUpdate'))) return;
const spinner = document.getElementById('update-spinner');
const btn = document.getElementById('update-trigger-btn');
const progressText = document.getElementById('update-progress-text');
const setProgress = (msg) => {
if (!progressText) return;
progressText.classList.remove('d-none');
progressText.textContent = msg;
};
const stopUpdateUi = () => {
if (spinner) spinner.classList.add('d-none');
if (btn) btn.disabled = false;
};
if (spinner) spinner.classList.remove('d-none');
if (btn) btn.disabled = true;
setProgress(t('settings.updateStepStarting'));
try {
const data = await api('POST', '/settings/update');
showSettingsAlert('success', data.message || t('messages.updateStarted'));
} catch (err) {
showSettingsAlert('danger', t('errors.failed', { error: err.message }));
if (spinner) spinner.classList.add('d-none');
stopUpdateUi();
return;
}
// Phase 1: poll build/pull progress until the container restarts
// (connection drops, which is expected and is our cue to move to phase 2).
const stepLabelKey = {
backup: 'settings.updateStepBackup',
pull: 'settings.updateStepPull',
build: 'settings.updateStepBuild',
restart: 'settings.updateStepRestart',
};
for (let i = 0; i < 200; i++) {
await new Promise(r => setTimeout(r, 2000));
try {
const st = await api('GET', '/settings/update/status');
if (st.state === 'failed') {
showSettingsAlert('danger', st.message || t('errors.requestFailed'));
stopUpdateUi();
return;
}
setProgress(t(stepLabelKey[st.step] || 'settings.updateStepStarting') + (st.message ? `${st.message}` : ''));
} catch (err) {
// Connection dropped — the container is very likely mid-restart. Move on.
break;
}
}
// Phase 2: wait for the app to come back up, then reload version info.
setProgress(t('settings.updateStepReconnecting'));
for (let i = 0; i < 90; i++) {
await new Promise(r => setTimeout(r, 2000));
try {
await api('GET', '/settings/version');
setProgress(t('settings.updateStepDone'));
stopUpdateUi();
showSettingsAlert('success', t('settings.updateStepDone'));
await loadVersionInfo();
return;
} catch (err) {
// still restarting — keep polling
}
}
// Gave up waiting — surface this instead of spinning forever.
stopUpdateUi();
setProgress('');
if (progressText) progressText.classList.add('d-none');
showSettingsAlert('warning', t('settings.updateStepTimeout'));
}
// ---------------------------------------------------------------------------
+8
View File
@@ -230,6 +230,14 @@
"triggerUpdate": "Update starten",
"updateWarning": "Die App ist während des Rebuilds ca. 60 Sekunden nicht verfügbar.",
"confirmUpdate": "Update jetzt starten? Die Datenbank wird zuerst gesichert. Die App startet neu (~60 Sekunden Ausfallzeit).",
"updateStepStarting": "Update wird gestartet …",
"updateStepBackup": "Datenbank wird gesichert …",
"updateStepPull": "Code wird geholt …",
"updateStepBuild": "Docker-Image wird gebaut (kann mehrere Minuten dauern) …",
"updateStepRestart": "Container wird neu gestartet …",
"updateStepReconnecting": "Container startet neu — warte auf Verbindung …",
"updateStepDone": "Update abgeschlossen.",
"updateStepTimeout": "Update läuft länger als erwartet. Bitte Server-Logs prüfen oder die Seite in ein paar Minuten neu laden.",
"gitTitle": "Git-Repository Einstellungen",
"gitRepoUrl": "Repository URL",
"gitRepoUrlHint": "Wird für Versionsprüfungen und One-Click-Updates via Gitea API verwendet.",
+8
View File
@@ -262,6 +262,14 @@
"triggerUpdate": "Start Update",
"updateWarning": "The app will be unavailable for ~60 seconds during rebuild.",
"confirmUpdate": "Start the update now? The database will be backed up first. The app will restart (~60 seconds downtime).",
"updateStepStarting": "Starting update …",
"updateStepBackup": "Backing up database …",
"updateStepPull": "Fetching code …",
"updateStepBuild": "Building Docker image (can take several minutes) …",
"updateStepRestart": "Restarting container …",
"updateStepReconnecting": "Container is restarting — waiting for connection …",
"updateStepDone": "Update complete.",
"updateStepTimeout": "Update is taking longer than expected. Check the server logs or reload this page in a few minutes.",
"gitTitle": "Git Repository Settings",
"gitRepoUrl": "Repository URL",
"gitRepoUrlHint": "Used for version checks and one-click updates via Gitea API.",