From 811eb55e12509382e6564c1469c7ebacf1c54d20 Mon Sep 17 00:00:00 2001 From: Lumpiasty Date: Fri, 24 Jul 2026 22:32:29 +0200 Subject: [PATCH] server: on-demand VRAM sharing to time-share one GPU between models Add release/restore of a model's GPU weight buffers (keeping a host shadow and the KV cache) so several always-loaded llama-server processes can time-share a single GPU without reloading or losing the prompt cache. - llama-model: release_device_weights()/restore_device_weights() capture a compact host shadow (stable iteration order, view-skipping) and free then realloc the device weight buffers; weights_resident() query. - llama-context: release_device()/restore_device() wrappers; decode() auto- restores; public C API llama_context_release_device/restore_device. - server: LLAMA_SLEEP_VRAM_ONLY makes idle-sleep release only the VRAM weights (not a full unload/reload). A cross-process flock token in LLAMA_VRAM_ARENA enforces "resident iff holds token"; an inotify doorbell forces the holder to release on contention. The warden thread only touches the task queue, so releases run on the loop thread and never race a decode. Validated on RX 580 (Vulkan): two models share 8GB, never both resident, correct output under contention, KV cache preserved (no re-prefill). Assisted-by: Claude --- include/llama.h | 8 ++ src/llama-context.cpp | 38 +++++++ src/llama-context.h | 6 + src/llama-model.cpp | 121 +++++++++++++++++++- src/llama-model.h | 8 ++ tools/server/server-context.cpp | 191 ++++++++++++++++++++++++++++++++ tools/server/server-queue.cpp | 21 +++- tools/server/server-queue.h | 6 + 8 files changed, 394 insertions(+), 5 deletions(-) diff --git a/include/llama.h b/include/llama.h index ef7a012c4..c0f5cf412 100644 --- a/include/llama.h +++ b/include/llama.h @@ -580,6 +580,14 @@ extern "C" { LLAMA_API const struct llama_model * llama_get_model (const struct llama_context * ctx); LLAMA_API llama_memory_t llama_get_memory (const struct llama_context * ctx); + + // On-demand device (VRAM) residency: free the model's GPU weight buffers (keeping a host + // shadow so no reload is needed) to hand VRAM to another model, and rebuild them on demand. + // The KV cache is left resident. Any subsequent llama_decode auto-restores the weights. + // Intended for time-sharing a single GPU between several always-loaded models. + LLAMA_API void llama_context_release_device(struct llama_context * ctx); + LLAMA_API void llama_context_restore_device(struct llama_context * ctx); + LLAMA_API bool llama_context_weights_resident(const struct llama_context * ctx); LLAMA_API enum llama_pooling_type llama_pooling_type(const struct llama_context * ctx); // TODO: rename to llama_get_pooling_type LLAMA_API const struct llama_vocab * llama_model_get_vocab(const struct llama_model * model); diff --git a/src/llama-context.cpp b/src/llama-context.cpp index 21501574a..b0887a603 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -745,6 +745,26 @@ void llama_context::synchronize() { t_compute_start_us = 0; } +void llama_context::release_device() { + if (!model.weights_resident()) { + return; + } + // ensure no compute is in flight before freeing the device buffers + synchronize(); + model.release_device_weights(); +} + +void llama_context::restore_device() { + if (model.weights_resident()) { + return; + } + model.restore_device_weights(); + // make sure all weight uploads have completed before any compute reads them + if (sched) { + ggml_backend_sched_synchronize(sched.get()); + } +} + const llama_model & llama_context::get_model() const { return model; } @@ -1656,6 +1676,12 @@ int llama_context::decode(const llama_batch & batch_inp) { return -1; } + // on-demand VRAM: if the weights were released while this model was idle, bring them back + // to the device before building the compute graph + if (!model.weights_resident()) { + restore_device(); + } + const auto & vocab = model.vocab; const auto & hparams = model.hparams; @@ -3850,6 +3876,18 @@ void llama_synchronize(llama_context * ctx) { ctx->synchronize(); } +void llama_context_release_device(llama_context * ctx) { + ctx->release_device(); +} + +void llama_context_restore_device(llama_context * ctx) { + ctx->restore_device(); +} + +bool llama_context_weights_resident(const llama_context * ctx) { + return ctx->get_model().weights_resident(); +} + float * llama_get_logits(llama_context * ctx) { ctx->synchronize(); diff --git a/src/llama-context.h b/src/llama-context.h index bf91daa8b..e50fbd515 100644 --- a/src/llama-context.h +++ b/src/llama-context.h @@ -57,6 +57,12 @@ struct llama_context { void synchronize(); + // on-demand device (VRAM) residency: free / rebuild the model's GPU weight buffers so that + // VRAM can be time-shared with another model. release_device() synchronizes first; decode() + // auto-restores when it finds the weights have been released. + void release_device(); + void restore_device(); + const llama_model & get_model() const; const llama_cparams & get_cparams() const; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 0adc07449..c62ef18cc 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1160,6 +1160,16 @@ struct llama_model::impl { // contexts where the model tensors metadata is stored as well as the corresponding buffers: std::vector>> ctxs_bufs; + // on-demand device (VRAM) weight residency: per-ctxs_bufs slot describing whether the + // device weight buffer can be freed and rebuilt from a host shadow (see release/restore_device_weights) + struct weight_release_slot { + ggml_backend_buffer_type_t buft = nullptr; // buffer type to reallocate on restore + bool releasable = false; // single-buffer alloc-path device (VRAM) buffer + bool resident = true; // currently allocated on device + std::vector shadow; // host copy, captured lazily on first release + }; + std::vector weight_release; // parallel to ctxs_bufs + buft_list_t cpu_buft_list; std::map gpu_buft_list; @@ -1742,8 +1752,9 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { // a lazy context is mapped whatever the load mode, but the memory-fit pass maps nothing const bool is_lazy_mapped = ctx_key.lazy && !ml.no_alloc; + const bool host_ptr_path = (ml.use_mmap || is_lazy_mapped) && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft; - if ((ml.use_mmap || is_lazy_mapped) && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft) { + if (host_ptr_path) { GGML_ASSERT(!ml.no_alloc); for (uint32_t idx = 0; idx < ml.files.size(); idx++) { // only the mmap region containing the tensors in the model is mapped to the backend buffer @@ -1797,6 +1808,23 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { pimpl->ctxs_bufs.emplace_back(std::move(ctx_ptr), std::move(bufs)); + // record on-demand release metadata (parallel to ctxs_bufs). Only the single-buffer + // alloc path on a device (non-host) buffer can be freed and rebuilt from a host shadow; + // the host_ptr path (mmap / lazy), the no_alloc dummy buffers and host (CPU) buffers are + // left resident. + { + llama_model::impl::weight_release_slot slot; + auto & last_bufs = pimpl->ctxs_bufs.back().second; + if (!host_ptr_path && !ml.no_alloc && last_bufs.size() == 1) { + ggml_backend_buffer_t b = last_bufs[0].get(); + if (b != nullptr && !ggml_backend_buffer_is_host(b)) { + slot.releasable = true; + slot.buft = buft; + } + } + pimpl->weight_release.push_back(std::move(slot)); + } + ctx_buf_maps.emplace_back(ctx, buf_map); } @@ -1859,6 +1887,97 @@ ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM tn, ne, flags); } +void llama_model::release_device_weights() const { + // NOTE: the caller is responsible for synchronizing the backend scheduler first, so that no + // compute is in flight referencing these buffers when they are freed. + size_t freed = 0; + for (size_t i = 0; i < pimpl->ctxs_bufs.size(); ++i) { + auto & slot = pimpl->weight_release[i]; + if (!slot.releasable || !slot.resident) { + continue; + } + ggml_context * ctx = pimpl->ctxs_bufs[i].first.get(); + ggml_backend_buffer_t buf = pimpl->ctxs_bufs[i].second[0].get(); + const size_t sz = ggml_backend_buffer_get_size(buf); + + // lazily capture a host shadow of the weights (read-only, so captured only once). + // Store tensor bytes compactly in stable iteration order (independent of buffer layout / + // alignment padding) so restore does not depend on the re-allocated offsets matching. + // View tensors alias their base, so they are skipped (restored implicitly via their base). + if (slot.shadow.empty()) { + size_t total = 0; + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) { + if (t->view_src != nullptr) continue; + total += ggml_nbytes(t); + } + slot.shadow.resize(total); + size_t off = 0; + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) { + if (t->view_src != nullptr) continue; + const size_t n = ggml_nbytes(t); + ggml_backend_tensor_get(t, slot.shadow.data() + off, 0, n); + off += n; + } + } + + // free the device (VRAM) buffer and clear the now-dangling tensor pointers so that + // ggml_backend_alloc_ctx_tensors_from_buft reallocates them cleanly on restore + pimpl->ctxs_bufs[i].second[0].reset(); + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) { + t->buffer = nullptr; + t->data = nullptr; + } + slot.resident = false; + freed += sz; + } + if (freed > 0) { + LLAMA_LOG_INFO("%s: released %.2f MiB of device weights\n", __func__, freed / 1024.0 / 1024.0); + } +} + +bool llama_model::restore_device_weights() const { + size_t restored = 0; + for (size_t i = 0; i < pimpl->ctxs_bufs.size(); ++i) { + auto & slot = pimpl->weight_release[i]; + if (!slot.releasable || slot.resident) { + continue; + } + ggml_context * ctx = pimpl->ctxs_bufs[i].first.get(); + ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, slot.buft); + if (buf == nullptr) { + LLAMA_LOG_ERROR("%s: failed to reallocate device weight buffer (out of VRAM?)\n", __func__); + return false; + } + ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); + + // re-upload weights from the compact host shadow, using the same stable iteration order + // and view-skipping as the capture above + size_t off = 0; + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) { + if (t->view_src != nullptr) continue; + const size_t n = ggml_nbytes(t); + ggml_backend_tensor_set(t, slot.shadow.data() + off, 0, n); + off += n; + } + pimpl->ctxs_bufs[i].second[0].reset(buf); + slot.resident = true; + restored += ggml_backend_buffer_get_size(buf); + } + if (restored > 0) { + LLAMA_LOG_INFO("%s: restored %.2f MiB of device weights\n", __func__, restored / 1024.0 / 1024.0); + } + return true; +} + +bool llama_model::weights_resident() const { + for (const auto & slot : pimpl->weight_release) { + if (slot.releasable && !slot.resident) { + return false; + } + } + return true; +} + std::string llama_model::arch_name() const { return llm_arch_name(arch); } diff --git a/src/llama-model.h b/src/llama-model.h index 4c4a30e01..7ff91a39c 100644 --- a/src/llama-model.h +++ b/src/llama-model.h @@ -746,6 +746,14 @@ struct llama_model { const struct ggml_tensor * get_tensor(const char * name) const; + // on-demand device (VRAM) weight residency: free the GPU weight buffers (keeping a host + // shadow) and rebuild them on demand. Used to time-share VRAM between models without + // reloading or losing the KV cache. release_device_weights() requires the caller to have + // synchronized the backend scheduler first. + void release_device_weights() const; + bool restore_device_weights() const; + bool weights_resident() const; + float get_rope_freq_base (const llama_cparams & cparams, int il) const; float get_rope_freq_scale(const llama_cparams & cparams, int il) const; diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index fe068d3e9..e1729841d 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -26,6 +26,8 @@ #include #include #include +#include +#include // fix problem with std::min and std::max #if defined(_WIN32) @@ -36,6 +38,16 @@ #include #endif +// POSIX file locking + inotify doorbell for the cross-process VRAM arbiter (see vram_share_* below) +#if !defined(_WIN32) +#include +#include +#include +#include +#include +#include +#endif + constexpr int HTTP_POLLING_SECONDS = 1; static common_speculative_output_limits server_output_limits(const common_params & params) { @@ -857,6 +869,8 @@ public: } ~server_context_impl() { + // stop the VRAM warden thread (and release the token) before tearing anything down + vram_share_shutdown(); if (!sleeping) { // destroy() is already called when entering sleeping state // we don't call it again here to avoid double free @@ -888,6 +902,153 @@ private: llama_model * model_dft = nullptr; llama_context * ctx_dft = nullptr; + // Cross-process VRAM arbiter: when LLAMA_SLEEP_VRAM_ONLY is set, several always-loaded + // llama-server processes time-share one GPU. A single flock() on /token.lock is the + // baton: "resident (weights in VRAM) iff I hold the token". A model warms up (locks + restores + // weights) right before it decodes, and goes cold (releases weights + unlocks) when it idles, + // so only one model holds VRAM at a time and the KV cache is never evicted (no re-prefill). + bool vram_only = false; // release-mode sleep active (LLAMA_SLEEP_VRAM_ONLY) + bool vram_flock = false; // cross-process flock coordination available + std::atomic vram_cold{true}; // weights currently released (read by warden thread) + int vram_lock_fd = -1; // fd for /token.lock + + // inotify "doorbell": a waiter that wants the VRAM token touches /doorbell/, which + // wakes the current holder's warden thread so it releases immediately instead of only on idle. + std::string vram_doorbell_dir; + std::string vram_pid_str; + int vram_inotify_fd = -1; + std::thread vram_warden; + std::atomic vram_warden_run{false}; + + void vram_share_init() { + if (getenv("LLAMA_SLEEP_VRAM_ONLY") == nullptr) { + return; + } + vram_only = true; + vram_cold = false; // weights are resident right after load +#if !defined(_WIN32) + const char * arena_env = getenv("LLAMA_VRAM_ARENA"); + const std::string arena = arena_env ? arena_env : "/dev/shm/llama-vram"; + mkdir(arena.c_str(), 0777); + const std::string lock_path = arena + "/token.lock"; + vram_lock_fd = open(lock_path.c_str(), O_RDWR | O_CREAT, 0666); + if (vram_lock_fd >= 0) { + vram_flock = true; + vram_pid_str = std::to_string(getpid()); + vram_doorbell_dir = arena + "/doorbell"; + mkdir(vram_doorbell_dir.c_str(), 0777); + vram_inotify_fd = inotify_init1(IN_NONBLOCK); + if (vram_inotify_fd >= 0) { + inotify_add_watch(vram_inotify_fd, vram_doorbell_dir.c_str(), IN_CLOSE_WRITE); + vram_warden_run = true; + vram_warden = std::thread([this]{ vram_warden_loop(); }); + } + SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n", + lock_path.c_str(), vram_doorbell_dir.c_str()); + } else { + SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str()); + } +#endif + } + + // warden thread: block on the doorbell; when another process rings (wants the token) and we + // currently hold it, ask the loop to yield (release) at its next idle point. Only sets a flag on + // the queue - it never posts a task and never touches the GPU. The flag is consumed by + // should_sleep() on the start_loop() thread, which is only reached after callback_update_slots() + // has returned, so a release can never happen underneath an in-flight decode - not even via the + // worker thread of yield_to_queue(), which declines everything except read-only tasks. + void vram_warden_loop() { +#if !defined(_WIN32) + char buf[4096]; + while (vram_warden_run.load()) { + struct pollfd pfd { vram_inotify_fd, POLLIN, 0 }; + int pr = poll(&pfd, 1, 500); // 500ms so we periodically re-check the run flag + if (pr <= 0) { + continue; + } + ssize_t n = read(vram_inotify_fd, buf, sizeof(buf)); + if (n <= 0) { + continue; + } + bool foreign_ring = false; + for (char * p = buf; p < buf + n; ) { + struct inotify_event * ev = (struct inotify_event *) p; + if (ev->len > 0 && vram_pid_str != ev->name) { + foreign_ring = true; // someone else wants the GPU + } + p += sizeof(struct inotify_event) + ev->len; + } + if (foreign_ring && !vram_cold) { + queue_tasks.request_yield(); + } + } +#endif + } + + // ring the doorbell so the current token holder releases promptly + void vram_ring_doorbell() { +#if !defined(_WIN32) + if (!vram_flock || vram_doorbell_dir.empty()) { + return; + } + const std::string f = vram_doorbell_dir + "/" + vram_pid_str; + int fd = open(f.c_str(), O_WRONLY | O_CREAT | O_TRUNC, 0666); + if (fd >= 0) { + ssize_t w = write(fd, "1", 1); + (void) w; + close(fd); + } +#endif + } + + // acquire the VRAM token (blocking) and bring weights back to the device. Called on the loop + // thread right before a decode, so it never races compute. + void vram_ensure_warm() { + if (!vram_only || !vram_cold) { + return; + } +#if !defined(_WIN32) + if (vram_flock) { + vram_ring_doorbell(); // nudge the current holder to release + flock(vram_lock_fd, LOCK_EX); // blocks until the current holder goes cold + } +#endif + llama_context_restore_device(ctx_tgt); + if (ctx_dft != nullptr) { + llama_context_restore_device(ctx_dft); + } + vram_cold = false; + } + + void vram_share_shutdown() { +#if !defined(_WIN32) + vram_warden_run = false; + if (vram_warden.joinable()) { + vram_warden.join(); + } + if (vram_inotify_fd >= 0) { close(vram_inotify_fd); vram_inotify_fd = -1; } + if (vram_lock_fd >= 0) { flock(vram_lock_fd, LOCK_UN); close(vram_lock_fd); vram_lock_fd = -1; } +#endif + } + + // release the device weights (keeping KV + host shadow) and drop the VRAM token so another + // model can warm up. Called on the loop thread when the server goes idle. + void vram_go_cold() { + if (!vram_only || vram_cold) { + return; + } + llama_context_release_device(ctx_tgt); + if (ctx_dft != nullptr) { + llama_context_release_device(ctx_dft); + } +#if !defined(_WIN32) + if (vram_flock) { + flock(vram_lock_fd, LOCK_UN); + } +#endif + vram_cold = true; + } + common_speculative_init_result_ptr spec_init; common_context_seq_rm_type ctx_tgt_seq_rm_type = COMMON_CONTEXT_SEQ_RM_TYPE_NO; @@ -953,6 +1114,25 @@ private: void handle_sleeping_state(bool new_state) { GGML_ASSERT(sleeping != new_state); + + // Lightweight VRAM-only sleep: instead of a full unload/reload (which frees RAM and the + // KV cache and pays a full reload on wake), just release the model's device (VRAM) weight + // buffers to a host shadow (keeping the context, KV cache and host weights) and drop the + // shared VRAM token. Waking is handled lazily by vram_ensure_warm() right before the next + // decode (which re-acquires the token first), so a single GPU is time-shared between models + // with no re-prefill. + if (vram_only && ctx_tgt != nullptr) { + if (new_state) { + SRV_INF("%s", "server entering sleeping state (VRAM-only: releasing device weights, keeping KV cache)\n"); + vram_go_cold(); + } else { + SRV_INF("%s", "server exiting sleeping state (VRAM-only: weights restored on next decode)\n"); + // token is acquired and weights restored by vram_ensure_warm() before decode + } + sleeping = new_state; + return; + } + if (new_state) { if (callback_state) { callback_state(SERVER_STATE_SLEEPING, {}); @@ -1415,6 +1595,12 @@ private: handle_sleeping_state(sleeping); }); + // VRAM arbiter: enable cross-process GPU time-sharing (if LLAMA_SLEEP_VRAM_ONLY is set) and + // start cold - release the just-loaded device weights and drop the shared token, so we only + // occupy VRAM while actually serving. The first decode re-acquires the token and restores. + vram_share_init(); + vram_go_cold(); + metrics.init(); if (params_base.cache_idle_slots) { @@ -3675,6 +3861,11 @@ private: has_output |= batch.tokens[i].output; } + // VRAM arbiter: make sure our weights are resident (acquiring the shared GPU token first) + // before we decode. Runs on the loop thread before entering yield_to_queue(), so it never + // races an in-flight decode. + vram_ensure_warm(); + // yield to the queue, so we can still handle metrics tasks while decoding // note: the sync is done here too, so that the wait is also covered by the yield int ret = 0; diff --git a/tools/server/server-queue.cpp b/tools/server/server-queue.cpp index 78169e9a5..f173043f0 100644 --- a/tools/server/server-queue.cpp +++ b/tools/server/server-queue.cpp @@ -129,6 +129,12 @@ void server_queue::wait_until_no_sleep() { } } +void server_queue::request_yield() { + std::unique_lock lock(mutex_tasks); + yield_requested = true; + condition_tasks.notify_all(); +} + void server_queue::terminate() { std::unique_lock lock(mutex_tasks); running = false; @@ -289,6 +295,9 @@ void server_queue::start_loop(int64_t idle_sleep_ms) { constexpr auto max_wait_time = std::chrono::seconds(1); auto should_sleep = [&]() -> bool { // caller must hold mutex_tasks + if (yield_requested) { + return true; // another process rang the VRAM doorbell - release now + } if (idle_sleep_ms < 0) { return false; } @@ -327,6 +336,7 @@ void server_queue::start_loop(int64_t idle_sleep_ms) { if (should_sleep()) { QUE_INF("%s", "entering sleeping state\n"); sleeping = true; + yield_requested = false; // consumed // Call order cb0 -> cb1 -> cb{N} for (auto & cb : callback_sleeping_state) { cb(true); @@ -350,14 +360,17 @@ void server_queue::start_loop(int64_t idle_sleep_ms) { condition_tasks.notify_all(); // notify wait_until_no_sleep() break; // process new tasks } else { - // wait for new tasks or timeout for checking sleeping condition + // wait for new tasks, a VRAM yield request, or timeout for checking sleeping condition bool res = condition_tasks.wait_for(lock, max_wait_time, [&]{ - return (!queue_tasks.empty() || !running); + return (!queue_tasks.empty() || !running || yield_requested); }); - if (res) { + if (res && !queue_tasks.empty()) { break; // new task arrived or terminate } - // otherwise, loop again to check sleeping condition + if (!running) { + break; + } + // otherwise (timeout or yield request), loop again to re-check should_sleep } } } diff --git a/tools/server/server-queue.h b/tools/server/server-queue.h index e17733a74..4bc390c93 100644 --- a/tools/server/server-queue.h +++ b/tools/server/server-queue.h @@ -18,6 +18,7 @@ private: bool running = false; bool sleeping = false; bool req_stop_sleeping = false; + bool yield_requested = false; // set by request_yield() when another process wants the VRAM token int64_t time_last_task = 0; // queues @@ -69,6 +70,11 @@ public: // returns immediately if not sleeping void wait_until_no_sleep(); + // request that the loop go to sleep (release VRAM) as soon as it is idle - called from the + // VRAM-arbiter warden thread when another process rings the doorbell for the GPU token. + // Thread-safe; wakes the loop so it releases promptly instead of at the next idle poll. + void request_yield(); + bool is_sleeping() { std::unique_lock lock(mutex_tasks); return sleeping;