From 2af8f5ae6e98cdd1d902a9bf2bf0c0ad66609dbd Mon Sep 17 00:00:00 2001 From: Lumpiasty Date: Sat, 25 Jul 2026 14:36:16 +0200 Subject: [PATCH] server: coordinate model load with the VRAM arbiter to avoid load-time OOM Before uploading a model's weights, acquire the shared VRAM token (ring the doorbell so any resident model releases first). Previously load uploaded weights to VRAM before the arbiter was active, so loading a large model while another (e.g. the warm 4B task model) held VRAM could exceed the budget and OOM. - add vram_arena_open() (idempotent flock/doorbell setup) and vram_acquire_for_load(), called from load_model() before common_init_from_params(). - a coordinated load now stays warm holding the token and serves its first request without a re-warm (drop the init-time go_cold cold-start). Assisted-by: Claude --- tools/server/server-context.cpp | 60 +++++++++++++++++++++++++++------ 1 file changed, 50 insertions(+), 10 deletions(-) diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index a6a3389e5..770e5636e 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -927,14 +927,17 @@ private: std::thread vram_warden; std::atomic vram_warden_run{false}; - void vram_share_init() { + // open the shared arena (flock token + doorbell dir). Idempotent; sets vram_only/vram_flock. + void vram_arena_open() { if (getenv("LLAMA_SLEEP_VRAM_ONLY") == nullptr) { return; } vram_only = true; vram_evict_kv = getenv("LLAMA_SLEEP_EVICT_KV") != nullptr; - vram_cold = false; // weights are resident right after load #if !defined(_WIN32) + if (vram_lock_fd >= 0) { + return; // already open + } const char * arena_env = getenv("LLAMA_VRAM_ARENA"); const std::string arena = arena_env ? arena_env : "/dev/shm/llama-vram"; mkdir(arena.c_str(), 0777); @@ -945,16 +948,49 @@ private: vram_pid_str = std::to_string(getpid()); vram_doorbell_dir = arena + "/doorbell"; mkdir(vram_doorbell_dir.c_str(), 0777); + SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n", + lock_path.c_str(), vram_doorbell_dir.c_str()); + } else { + SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str()); + } +#endif + } + + // Acquire the VRAM token BEFORE uploading this model's weights, so any model currently resident + // on the GPU releases first and the load uploads into free VRAM instead of racing it (which + // could OOM, e.g. the 4B task model holding VRAM while a large model loads). Blocks until free. + // Called from load_model() right before common_init_from_params(). + void vram_acquire_for_load() { + vram_arena_open(); + if (!vram_only) { + return; + } +#if !defined(_WIN32) + if (vram_flock) { + vram_ring_doorbell(); // nudge whoever is resident to release + flock(vram_lock_fd, LOCK_EX); // block until the GPU is free + } +#endif + vram_cold = false; // we hold the token; weights will be resident after the upload + } + + // Start the doorbell warden. Called from init() after the (coordinated) load. The model stays + // warm (holding the token acquired in vram_acquire_for_load) and serves its first request + // without a re-warm; the warden releases it when another model rings the doorbell. + void vram_share_init() { + vram_arena_open(); + if (!vram_only) { + return; + } + vram_cold = false; // resident and holding the token after the coordinated load +#if !defined(_WIN32) + if (vram_flock && vram_inotify_fd < 0) { vram_inotify_fd = inotify_init1(IN_NONBLOCK); if (vram_inotify_fd >= 0) { inotify_add_watch(vram_inotify_fd, vram_doorbell_dir.c_str(), IN_CLOSE_WRITE); vram_warden_run = true; vram_warden = std::thread([this]{ vram_warden_loop(); }); } - SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n", - lock_path.c_str(), vram_doorbell_dir.c_str()); - } else { - SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str()); } #endif } @@ -1321,6 +1357,11 @@ private: params_base.load_progress_callback_user_data = &load_progress_text; } + // VRAM arbiter: acquire the GPU token before uploading weights, so any resident model (e.g. + // the warm 4B task model) releases first and this load uploads into free VRAM instead of + // racing it and OOM-ing. No-op unless LLAMA_SLEEP_VRAM_ONLY is set. + vram_acquire_for_load(); + llama_init = common_init_from_params(params_base); model_tgt = llama_init->model(); @@ -1588,11 +1629,10 @@ private: handle_sleeping_state(sleeping); }); - // VRAM arbiter: enable cross-process GPU time-sharing (if LLAMA_SLEEP_VRAM_ONLY is set) and - // start cold - release the just-loaded device weights and drop the shared token, so we only - // occupy VRAM while actually serving. The first decode re-acquires the token and restores. + // VRAM arbiter: start the doorbell warden. The load was coordinated (vram_acquire_for_load + // grabbed the token before uploading), so we stay warm holding the token and serve the first + // request without a re-warm; the warden releases us when another model rings the doorbell. vram_share_init(); - vram_go_cold(); metrics.init();