From 0782d90ec531778157ce38a455439feffd97dd9e Mon Sep 17 00:00:00 2001 From: Miguel Cruz Date: Thu, 12 Mar 2026 06:39:34 -0400 Subject: [PATCH 1/2] fix: show partial download progress on initial dashboard load (#1706) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - Dashboard now shows partial download progress for models that were partially downloaded in a previous session, instead of showing 0% - Both `getModelDownloadStatus()` and `getInstanceDownloadStatus()` now handle `DownloadPending` entries that carry non-zero `downloaded`/`total` bytes Fixes #1042 ## Root cause When exo restarts with partially downloaded models, the `DownloadCoordinator` emits `DownloadPending` events (because `downloaded_this_session` is 0, even though real bytes exist on disk). The main page dashboard only checked for `DownloadOngoing` entries, so these partially downloaded models showed as 0%. The dedicated `/downloads` page already handled this correctly — it renders `DownloadPending` entries with a progress bar when `downloaded > 0`. This fix brings the same behavior to the main page. ## Changes For both `getModelDownloadStatus()` and `getInstanceDownloadStatus()` in `+page.svelte`: - Accept `DownloadPending` in addition to `DownloadOngoing` - For `DownloadPending` entries with `downloaded > 0` or `total > 0`, synthesize a `DownloadProgress` object from the top-level fields (with `speed: 0` and `etaMs: 0` since no active download is in progress) - Skip `DownloadPending` entries where both `downloaded` and `total` are 0 (truly pending, not yet started) ## Test plan - [ ] Partially download a model, quit exo, relaunch — dashboard should show partial progress instead of 0% - [ ] Fully downloaded models still show as complete - [ ] Active downloads still show real-time progress with speed/ETA - [ ] Models never downloaded show as not started (not falsely showing progress) - [ ] Dashboard builds without errors (`cd dashboard && npm run build`) --------- Co-authored-by: Evan --- dashboard/src/routes/+page.svelte | 77 ++++++++++++++++++++++++++++--- 1 file changed, 71 insertions(+), 6 deletions(-) diff --git a/dashboard/src/routes/+page.svelte b/dashboard/src/routes/+page.svelte index ad106b61..7f8a0382 100644 --- a/dashboard/src/routes/+page.svelte +++ b/dashboard/src/routes/+page.svelte @@ -1527,7 +1527,11 @@ downloadKind ] as Record; - if (downloadKind !== "DownloadOngoing") continue; + if ( + downloadKind !== "DownloadOngoing" && + downloadKind !== "DownloadPending" + ) + continue; if (!downloadPayload) continue; const downloadModelId = extractModelIdFromDownload(downloadPayload); @@ -1542,9 +1546,38 @@ if (downloadModelId !== modelId) continue; } - isDownloading = true; + // For DownloadPending with partial bytes (paused/resumed downloads), + // synthesize a progress object from the top-level downloaded/total fields + let progress: DownloadProgress | null; + if (downloadKind === "DownloadPending") { + const pendingDownloaded = getBytes( + downloadPayload.downloaded ?? + downloadPayload.downloaded_bytes ?? + downloadPayload.downloadedBytes, + ); + const pendingTotal = getBytes( + downloadPayload.total ?? + downloadPayload.total_bytes ?? + downloadPayload.totalBytes, + ); + if (pendingDownloaded <= 0 && pendingTotal <= 0) continue; + isDownloading = true; + progress = { + totalBytes: pendingTotal, + downloadedBytes: pendingDownloaded, + speed: 0, + etaMs: 0, + percentage: + pendingTotal > 0 ? (pendingDownloaded / pendingTotal) * 100 : 0, + completedFiles: 0, + totalFiles: 0, + files: [], + }; + } else { + isDownloading = true; + progress = parseDownloadProgress(downloadPayload); + } - const progress = parseDownloadProgress(downloadPayload); if (progress) { // Sum all values across nodes - each node downloads independently totalBytes += progress.totalBytes; @@ -1696,7 +1729,11 @@ } } - if (downloadKind !== "DownloadOngoing") continue; + if ( + downloadKind !== "DownloadOngoing" && + downloadKind !== "DownloadPending" + ) + continue; if (!downloadPayload) continue; // Check if this download is for this instance's model @@ -1706,9 +1743,37 @@ downloadModelId && downloadModelId === instanceModelId ) { - isDownloading = true; + // For DownloadPending with partial bytes, synthesize progress + let progress: DownloadProgress | null; + if (downloadKind === "DownloadPending") { + const pendingDownloaded = getBytes( + downloadPayload.downloaded ?? + downloadPayload.downloaded_bytes ?? + downloadPayload.downloadedBytes, + ); + const pendingTotal = getBytes( + downloadPayload.total ?? + downloadPayload.total_bytes ?? + downloadPayload.totalBytes, + ); + if (pendingDownloaded <= 0 && pendingTotal <= 0) continue; + isDownloading = true; + progress = { + totalBytes: pendingTotal, + downloadedBytes: pendingDownloaded, + speed: 0, + etaMs: 0, + percentage: + pendingTotal > 0 ? (pendingDownloaded / pendingTotal) * 100 : 0, + completedFiles: 0, + totalFiles: 0, + files: [], + }; + } else { + isDownloading = true; + progress = parseDownloadProgress(downloadPayload); + } - const progress = parseDownloadProgress(downloadPayload); if (progress) { // Sum all values across nodes - each node downloads independently totalBytes += progress.totalBytes; From ea18a625813d36069956ba742e8f519eabee05b2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mustafa=20Alp=20Y=C4=B1lmaz?= <96022931+mustafalpyilmaz@users.noreply.github.com> Date: Thu, 12 Mar 2026 17:48:36 +0300 Subject: [PATCH 2/2] fix: guard against ZeroDivisionError in mlx_lm stats (#1707) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Problem Running short completions (like `max_tokens=1` health check probes) can finish so fast that `mlx_lm`'s internal `generation_time` rounds to zero. When that happens, `BatchGenerator.stats()` in `mlx_lm/generate.py` divides `generation_tokens / generation_time` and throws a `ZeroDivisionError`, which kills the runner process. EXO already handles this on its side — lines 289-293 in `batch_generate.py` guard the TPS calculation with `if generation_elapsed > 0`. But the call to `self._exo_gen.stats()` on line 294 goes into mlx_lm's *separate* timing code, which doesn't have the same guard. Two different timers, only one is protected. In my case, this was triggered by health check probes (content: `"a"`, `max_tokens=1`). The generation completed in sub-microsecond time, `generation_time` was exactly `0`, and the runner crashed. Since the health check command stays in the queue and retries after recovery, it created an infinite crash loop — every ~15 seconds the runner would load the model, get the same health check, and die again. ## Fix Wrap `self._exo_gen.stats()` in a `try/except ZeroDivisionError`. If it throws, set `mlx_stats` to `None` and fall back to `0.0` for `prompt_tps`. The only fields EXO reads from `mlx_stats` are `prompt_tps` and `prompt_time` — losing them on a sub-microsecond generation has no practical impact. ## Traceback ``` File "batch_generate.py", line 294, in step mlx_stats = self._exo_gen.stats() File "mlx_lm/generate.py", line 1224, in stats self._stats.generation_tokens / self._stats.generation_time ZeroDivisionError: division by zero ``` --- src/exo/worker/engines/mlx/generator/batch_generate.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/exo/worker/engines/mlx/generator/batch_generate.py b/src/exo/worker/engines/mlx/generator/batch_generate.py index 1911a97c..3462a873 100644 --- a/src/exo/worker/engines/mlx/generator/batch_generate.py +++ b/src/exo/worker/engines/mlx/generator/batch_generate.py @@ -291,10 +291,13 @@ class ExoBatchGenerator: if generation_elapsed > 0 else 0.0 ) - mlx_stats = self._exo_gen.stats() + try: + mlx_stats = self._exo_gen.stats() + except ZeroDivisionError: + mlx_stats = None stats = GenerationStats( prompt_tps=float(mlx_stats.prompt_tps) - if mlx_stats.prompt_time > 0 + if mlx_stats is not None and mlx_stats.prompt_time > 0 else 0.0, generation_tps=float(generation_tps), prompt_tokens=len(state.all_prompt_tokens),