server: on-demand VRAM sharing to time-share one GPU between models
Add release/restore of a model's GPU weight buffers (keeping a host shadow and the KV cache) so several always-loaded llama-server processes can time-share a single GPU without reloading or losing the prompt cache. - llama-model: release_device_weights()/restore_device_weights() capture a compact host shadow (stable iteration order, view-skipping) and free then realloc the device weight buffers; weights_resident() query. - llama-context: release_device()/restore_device() wrappers; decode() auto- restores; public C API llama_context_release_device/restore_device. - server: LLAMA_SLEEP_VRAM_ONLY makes idle-sleep release only the VRAM weights (not a full unload/reload). A cross-process flock token in LLAMA_VRAM_ARENA enforces "resident iff holds token"; an inotify doorbell forces the holder to release on contention. The warden thread only touches the task queue, so releases run on the loop thread and never race a decode. Validated on RX 580 (Vulkan): two models share 8GB, never both resident, correct output under contention, KV cache preserved (no re-prefill). Assisted-by: Claude
This commit is contained in:
@@ -565,6 +565,14 @@ extern "C" {
|
||||
|
||||
LLAMA_API const struct llama_model * llama_get_model (const struct llama_context * ctx);
|
||||
LLAMA_API llama_memory_t llama_get_memory (const struct llama_context * ctx);
|
||||
|
||||
// On-demand device (VRAM) residency: free the model's GPU weight buffers (keeping a host
|
||||
// shadow so no reload is needed) to hand VRAM to another model, and rebuild them on demand.
|
||||
// The KV cache is left resident. Any subsequent llama_decode auto-restores the weights.
|
||||
// Intended for time-sharing a single GPU between several always-loaded models.
|
||||
LLAMA_API void llama_context_release_device(struct llama_context * ctx);
|
||||
LLAMA_API void llama_context_restore_device(struct llama_context * ctx);
|
||||
LLAMA_API bool llama_context_weights_resident(const struct llama_context * ctx);
|
||||
LLAMA_API enum llama_pooling_type llama_pooling_type(const struct llama_context * ctx); // TODO: rename to llama_get_pooling_type
|
||||
|
||||
LLAMA_API const struct llama_vocab * llama_model_get_vocab(const struct llama_model * model);
|
||||
|
||||
@@ -728,6 +728,26 @@ void llama_context::synchronize() {
|
||||
t_compute_start_us = 0;
|
||||
}
|
||||
|
||||
void llama_context::release_device() {
|
||||
if (!model.weights_resident()) {
|
||||
return;
|
||||
}
|
||||
// ensure no compute is in flight before freeing the device buffers
|
||||
synchronize();
|
||||
model.release_device_weights();
|
||||
}
|
||||
|
||||
void llama_context::restore_device() {
|
||||
if (model.weights_resident()) {
|
||||
return;
|
||||
}
|
||||
model.restore_device_weights();
|
||||
// make sure all weight uploads have completed before any compute reads them
|
||||
if (sched) {
|
||||
ggml_backend_sched_synchronize(sched.get());
|
||||
}
|
||||
}
|
||||
|
||||
const llama_model & llama_context::get_model() const {
|
||||
return model;
|
||||
}
|
||||
@@ -1705,6 +1725,12 @@ int llama_context::decode(const llama_batch & batch_inp) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
// on-demand VRAM: if the weights were released while this model was idle, bring them back
|
||||
// to the device before building the compute graph
|
||||
if (!model.weights_resident()) {
|
||||
restore_device();
|
||||
}
|
||||
|
||||
const auto & vocab = model.vocab;
|
||||
const auto & hparams = model.hparams;
|
||||
|
||||
@@ -3675,6 +3701,18 @@ void llama_synchronize(llama_context * ctx) {
|
||||
ctx->synchronize();
|
||||
}
|
||||
|
||||
void llama_context_release_device(llama_context * ctx) {
|
||||
ctx->release_device();
|
||||
}
|
||||
|
||||
void llama_context_restore_device(llama_context * ctx) {
|
||||
ctx->restore_device();
|
||||
}
|
||||
|
||||
bool llama_context_weights_resident(const llama_context * ctx) {
|
||||
return ctx->get_model().weights_resident();
|
||||
}
|
||||
|
||||
float * llama_get_logits(llama_context * ctx) {
|
||||
ctx->synchronize();
|
||||
|
||||
|
||||
@@ -57,6 +57,12 @@ struct llama_context {
|
||||
|
||||
void synchronize();
|
||||
|
||||
// on-demand device (VRAM) residency: free / rebuild the model's GPU weight buffers so that
|
||||
// VRAM can be time-shared with another model. release_device() synchronizes first; decode()
|
||||
// auto-restores when it finds the weights have been released.
|
||||
void release_device();
|
||||
void restore_device();
|
||||
|
||||
const llama_model & get_model() const;
|
||||
const llama_cparams & get_cparams() const;
|
||||
|
||||
|
||||
@@ -1019,6 +1019,16 @@ struct llama_model::impl {
|
||||
// contexts where the model tensors metadata is stored as well as the corresponding buffers:
|
||||
std::vector<std::pair<ggml_context_ptr, std::vector<ggml_backend_buffer_ptr>>> ctxs_bufs;
|
||||
|
||||
// on-demand device (VRAM) weight residency: per-ctxs_bufs slot describing whether the
|
||||
// device weight buffer can be freed and rebuilt from a host shadow (see release/restore_device_weights)
|
||||
struct weight_release_slot {
|
||||
ggml_backend_buffer_type_t buft = nullptr; // buffer type to reallocate on restore
|
||||
bool releasable = false; // single-buffer alloc-path device (VRAM) buffer
|
||||
bool resident = true; // currently allocated on device
|
||||
std::vector<uint8_t> shadow; // host copy, captured lazily on first release
|
||||
};
|
||||
std::vector<weight_release_slot> weight_release; // parallel to ctxs_bufs
|
||||
|
||||
buft_list_t cpu_buft_list;
|
||||
std::map<ggml_backend_dev_t, buft_list_t> gpu_buft_list;
|
||||
|
||||
@@ -1614,6 +1624,23 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
||||
|
||||
pimpl->ctxs_bufs.emplace_back(std::move(ctx_ptr), std::move(bufs));
|
||||
|
||||
// record on-demand release metadata (parallel to ctxs_bufs). Only the single-buffer
|
||||
// alloc path on a device (non-host) buffer can be freed and rebuilt from a host shadow;
|
||||
// the mmap buffer_from_host_ptr path and host (CPU) buffers are left resident.
|
||||
{
|
||||
llama_model::impl::weight_release_slot slot;
|
||||
const bool mmap_host_ptr_path = ml.use_mmap && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft;
|
||||
auto & last_bufs = pimpl->ctxs_bufs.back().second;
|
||||
if (!mmap_host_ptr_path && last_bufs.size() == 1) {
|
||||
ggml_backend_buffer_t b = last_bufs[0].get();
|
||||
if (b != nullptr && !ggml_backend_buffer_is_host(b)) {
|
||||
slot.releasable = true;
|
||||
slot.buft = buft;
|
||||
}
|
||||
}
|
||||
pimpl->weight_release.push_back(std::move(slot));
|
||||
}
|
||||
|
||||
ctx_buf_maps.emplace_back(ctx, buf_map);
|
||||
}
|
||||
|
||||
@@ -1668,6 +1695,97 @@ ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM
|
||||
tn, ne, flags);
|
||||
}
|
||||
|
||||
void llama_model::release_device_weights() const {
|
||||
// NOTE: the caller is responsible for synchronizing the backend scheduler first, so that no
|
||||
// compute is in flight referencing these buffers when they are freed.
|
||||
size_t freed = 0;
|
||||
for (size_t i = 0; i < pimpl->ctxs_bufs.size(); ++i) {
|
||||
auto & slot = pimpl->weight_release[i];
|
||||
if (!slot.releasable || !slot.resident) {
|
||||
continue;
|
||||
}
|
||||
ggml_context * ctx = pimpl->ctxs_bufs[i].first.get();
|
||||
ggml_backend_buffer_t buf = pimpl->ctxs_bufs[i].second[0].get();
|
||||
const size_t sz = ggml_backend_buffer_get_size(buf);
|
||||
|
||||
// lazily capture a host shadow of the weights (read-only, so captured only once).
|
||||
// Store tensor bytes compactly in stable iteration order (independent of buffer layout /
|
||||
// alignment padding) so restore does not depend on the re-allocated offsets matching.
|
||||
// View tensors alias their base, so they are skipped (restored implicitly via their base).
|
||||
if (slot.shadow.empty()) {
|
||||
size_t total = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src != nullptr) continue;
|
||||
total += ggml_nbytes(t);
|
||||
}
|
||||
slot.shadow.resize(total);
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src != nullptr) continue;
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_get(t, slot.shadow.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
}
|
||||
|
||||
// free the device (VRAM) buffer and clear the now-dangling tensor pointers so that
|
||||
// ggml_backend_alloc_ctx_tensors_from_buft reallocates them cleanly on restore
|
||||
pimpl->ctxs_bufs[i].second[0].reset();
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
t->buffer = nullptr;
|
||||
t->data = nullptr;
|
||||
}
|
||||
slot.resident = false;
|
||||
freed += sz;
|
||||
}
|
||||
if (freed > 0) {
|
||||
LLAMA_LOG_INFO("%s: released %.2f MiB of device weights\n", __func__, freed / 1024.0 / 1024.0);
|
||||
}
|
||||
}
|
||||
|
||||
bool llama_model::restore_device_weights() const {
|
||||
size_t restored = 0;
|
||||
for (size_t i = 0; i < pimpl->ctxs_bufs.size(); ++i) {
|
||||
auto & slot = pimpl->weight_release[i];
|
||||
if (!slot.releasable || slot.resident) {
|
||||
continue;
|
||||
}
|
||||
ggml_context * ctx = pimpl->ctxs_bufs[i].first.get();
|
||||
ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, slot.buft);
|
||||
if (buf == nullptr) {
|
||||
LLAMA_LOG_ERROR("%s: failed to reallocate device weight buffer (out of VRAM?)\n", __func__);
|
||||
return false;
|
||||
}
|
||||
ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
|
||||
|
||||
// re-upload weights from the compact host shadow, using the same stable iteration order
|
||||
// and view-skipping as the capture above
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src != nullptr) continue;
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_set(t, slot.shadow.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
pimpl->ctxs_bufs[i].second[0].reset(buf);
|
||||
slot.resident = true;
|
||||
restored += ggml_backend_buffer_get_size(buf);
|
||||
}
|
||||
if (restored > 0) {
|
||||
LLAMA_LOG_INFO("%s: restored %.2f MiB of device weights\n", __func__, restored / 1024.0 / 1024.0);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool llama_model::weights_resident() const {
|
||||
for (const auto & slot : pimpl->weight_release) {
|
||||
if (slot.releasable && !slot.resident) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
std::string llama_model::arch_name() const {
|
||||
return llm_arch_name(arch);
|
||||
}
|
||||
|
||||
@@ -670,6 +670,14 @@ struct llama_model {
|
||||
|
||||
const struct ggml_tensor * get_tensor(const char * name) const;
|
||||
|
||||
// on-demand device (VRAM) weight residency: free the GPU weight buffers (keeping a host
|
||||
// shadow) and rebuild them on demand. Used to time-share VRAM between models without
|
||||
// reloading or losing the KV cache. release_device_weights() requires the caller to have
|
||||
// synchronized the backend scheduler first.
|
||||
void release_device_weights() const;
|
||||
bool restore_device_weights() const;
|
||||
bool weights_resident() const;
|
||||
|
||||
float get_rope_freq_base (const llama_cparams & cparams, int il) const;
|
||||
float get_rope_freq_scale(const llama_cparams & cparams, int il) const;
|
||||
|
||||
|
||||
@@ -25,6 +25,8 @@
|
||||
#include <filesystem>
|
||||
#include <utility>
|
||||
#include <fstream>
|
||||
#include <thread>
|
||||
#include <atomic>
|
||||
|
||||
// fix problem with std::min and std::max
|
||||
#if defined(_WIN32)
|
||||
@@ -35,6 +37,16 @@
|
||||
#include <windows.h>
|
||||
#endif
|
||||
|
||||
// POSIX file locking + inotify doorbell for the cross-process VRAM arbiter (see vram_share_* below)
|
||||
#if !defined(_WIN32)
|
||||
#include <sys/file.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/inotify.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <poll.h>
|
||||
#endif
|
||||
|
||||
using json = nlohmann::ordered_json;
|
||||
|
||||
constexpr int HTTP_POLLING_SECONDS = 1;
|
||||
@@ -871,6 +883,8 @@ public:
|
||||
}
|
||||
|
||||
~server_context_impl() {
|
||||
// stop the VRAM warden thread (and release the token) before tearing anything down
|
||||
vram_share_shutdown();
|
||||
if (!sleeping) {
|
||||
// destroy() is already called when entering sleeping state
|
||||
// we don't call it again here to avoid double free
|
||||
@@ -894,6 +908,150 @@ private:
|
||||
llama_model * model_dft = nullptr;
|
||||
llama_context * ctx_dft = nullptr;
|
||||
|
||||
// Cross-process VRAM arbiter: when LLAMA_SLEEP_VRAM_ONLY is set, several always-loaded
|
||||
// llama-server processes time-share one GPU. A single flock() on <arena>/token.lock is the
|
||||
// baton: "resident (weights in VRAM) iff I hold the token". A model warms up (locks + restores
|
||||
// weights) right before it decodes, and goes cold (releases weights + unlocks) when it idles,
|
||||
// so only one model holds VRAM at a time and the KV cache is never evicted (no re-prefill).
|
||||
bool vram_only = false; // release-mode sleep active (LLAMA_SLEEP_VRAM_ONLY)
|
||||
bool vram_flock = false; // cross-process flock coordination available
|
||||
std::atomic<bool> vram_cold{true}; // weights currently released (read by warden thread)
|
||||
int vram_lock_fd = -1; // fd for <arena>/token.lock
|
||||
|
||||
// inotify "doorbell": a waiter that wants the VRAM token touches <arena>/doorbell/<pid>, which
|
||||
// wakes the current holder's warden thread so it releases immediately instead of only on idle.
|
||||
std::string vram_doorbell_dir;
|
||||
std::string vram_pid_str;
|
||||
int vram_inotify_fd = -1;
|
||||
std::thread vram_warden;
|
||||
std::atomic<bool> vram_warden_run{false};
|
||||
|
||||
void vram_share_init() {
|
||||
if (getenv("LLAMA_SLEEP_VRAM_ONLY") == nullptr) {
|
||||
return;
|
||||
}
|
||||
vram_only = true;
|
||||
vram_cold = false; // weights are resident right after load
|
||||
#if !defined(_WIN32)
|
||||
const char * arena_env = getenv("LLAMA_VRAM_ARENA");
|
||||
const std::string arena = arena_env ? arena_env : "/dev/shm/llama-vram";
|
||||
mkdir(arena.c_str(), 0777);
|
||||
const std::string lock_path = arena + "/token.lock";
|
||||
vram_lock_fd = open(lock_path.c_str(), O_RDWR | O_CREAT, 0666);
|
||||
if (vram_lock_fd >= 0) {
|
||||
vram_flock = true;
|
||||
vram_pid_str = std::to_string(getpid());
|
||||
vram_doorbell_dir = arena + "/doorbell";
|
||||
mkdir(vram_doorbell_dir.c_str(), 0777);
|
||||
vram_inotify_fd = inotify_init1(IN_NONBLOCK);
|
||||
if (vram_inotify_fd >= 0) {
|
||||
inotify_add_watch(vram_inotify_fd, vram_doorbell_dir.c_str(), IN_CLOSE_WRITE);
|
||||
vram_warden_run = true;
|
||||
vram_warden = std::thread([this]{ vram_warden_loop(); });
|
||||
}
|
||||
SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n",
|
||||
lock_path.c_str(), vram_doorbell_dir.c_str());
|
||||
} else {
|
||||
SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// warden thread: block on the doorbell; when another process rings (wants the token) and we
|
||||
// currently hold it, ask the loop to yield (release) at its next idle point. Only touches the
|
||||
// thread-safe queue - never the GPU - so it cannot race a decode.
|
||||
void vram_warden_loop() {
|
||||
#if !defined(_WIN32)
|
||||
char buf[4096];
|
||||
while (vram_warden_run.load()) {
|
||||
struct pollfd pfd { vram_inotify_fd, POLLIN, 0 };
|
||||
int pr = poll(&pfd, 1, 500); // 500ms so we periodically re-check the run flag
|
||||
if (pr <= 0) {
|
||||
continue;
|
||||
}
|
||||
ssize_t n = read(vram_inotify_fd, buf, sizeof(buf));
|
||||
if (n <= 0) {
|
||||
continue;
|
||||
}
|
||||
bool foreign_ring = false;
|
||||
for (char * p = buf; p < buf + n; ) {
|
||||
struct inotify_event * ev = (struct inotify_event *) p;
|
||||
if (ev->len > 0 && vram_pid_str != ev->name) {
|
||||
foreign_ring = true; // someone else wants the GPU
|
||||
}
|
||||
p += sizeof(struct inotify_event) + ev->len;
|
||||
}
|
||||
if (foreign_ring && !vram_cold) {
|
||||
queue_tasks.request_yield();
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// ring the doorbell so the current token holder releases promptly
|
||||
void vram_ring_doorbell() {
|
||||
#if !defined(_WIN32)
|
||||
if (!vram_flock || vram_doorbell_dir.empty()) {
|
||||
return;
|
||||
}
|
||||
const std::string f = vram_doorbell_dir + "/" + vram_pid_str;
|
||||
int fd = open(f.c_str(), O_WRONLY | O_CREAT | O_TRUNC, 0666);
|
||||
if (fd >= 0) {
|
||||
ssize_t w = write(fd, "1", 1);
|
||||
(void) w;
|
||||
close(fd);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// acquire the VRAM token (blocking) and bring weights back to the device. Called on the loop
|
||||
// thread right before a decode, so it never races compute.
|
||||
void vram_ensure_warm() {
|
||||
if (!vram_only || !vram_cold) {
|
||||
return;
|
||||
}
|
||||
#if !defined(_WIN32)
|
||||
if (vram_flock) {
|
||||
vram_ring_doorbell(); // nudge the current holder to release
|
||||
flock(vram_lock_fd, LOCK_EX); // blocks until the current holder goes cold
|
||||
}
|
||||
#endif
|
||||
llama_context_restore_device(ctx_tgt);
|
||||
if (ctx_dft != nullptr) {
|
||||
llama_context_restore_device(ctx_dft);
|
||||
}
|
||||
vram_cold = false;
|
||||
}
|
||||
|
||||
void vram_share_shutdown() {
|
||||
#if !defined(_WIN32)
|
||||
vram_warden_run = false;
|
||||
if (vram_warden.joinable()) {
|
||||
vram_warden.join();
|
||||
}
|
||||
if (vram_inotify_fd >= 0) { close(vram_inotify_fd); vram_inotify_fd = -1; }
|
||||
if (vram_lock_fd >= 0) { flock(vram_lock_fd, LOCK_UN); close(vram_lock_fd); vram_lock_fd = -1; }
|
||||
#endif
|
||||
}
|
||||
|
||||
// release the device weights (keeping KV + host shadow) and drop the VRAM token so another
|
||||
// model can warm up. Called on the loop thread when the server goes idle.
|
||||
void vram_go_cold() {
|
||||
if (!vram_only || vram_cold) {
|
||||
return;
|
||||
}
|
||||
llama_context_release_device(ctx_tgt);
|
||||
if (ctx_dft != nullptr) {
|
||||
llama_context_release_device(ctx_dft);
|
||||
}
|
||||
#if !defined(_WIN32)
|
||||
if (vram_flock) {
|
||||
flock(vram_lock_fd, LOCK_UN);
|
||||
}
|
||||
#endif
|
||||
vram_cold = true;
|
||||
}
|
||||
|
||||
common_speculative_init_result_ptr spec_init;
|
||||
|
||||
common_context_seq_rm_type ctx_tgt_seq_rm_type = COMMON_CONTEXT_SEQ_RM_TYPE_NO;
|
||||
@@ -951,6 +1109,25 @@ private:
|
||||
|
||||
void handle_sleeping_state(bool new_state) {
|
||||
GGML_ASSERT(sleeping != new_state);
|
||||
|
||||
// Lightweight VRAM-only sleep: instead of a full unload/reload (which frees RAM and the
|
||||
// KV cache and pays a full reload on wake), just release the model's device (VRAM) weight
|
||||
// buffers to a host shadow (keeping the context, KV cache and host weights) and drop the
|
||||
// shared VRAM token. Waking is handled lazily by vram_ensure_warm() right before the next
|
||||
// decode (which re-acquires the token first), so a single GPU is time-shared between models
|
||||
// with no re-prefill.
|
||||
if (vram_only && ctx_tgt != nullptr) {
|
||||
if (new_state) {
|
||||
SRV_INF("%s", "server entering sleeping state (VRAM-only: releasing device weights, keeping KV cache)\n");
|
||||
vram_go_cold();
|
||||
} else {
|
||||
SRV_INF("%s", "server exiting sleeping state (VRAM-only: weights restored on next decode)\n");
|
||||
// token is acquired and weights restored by vram_ensure_warm() before decode
|
||||
}
|
||||
sleeping = new_state;
|
||||
return;
|
||||
}
|
||||
|
||||
if (new_state) {
|
||||
SRV_INF("%s", "server is entering sleeping state\n");
|
||||
destroy();
|
||||
@@ -1409,6 +1586,12 @@ private:
|
||||
handle_sleeping_state(sleeping);
|
||||
});
|
||||
|
||||
// VRAM arbiter: enable cross-process GPU time-sharing (if LLAMA_SLEEP_VRAM_ONLY is set) and
|
||||
// start cold - release the just-loaded device weights and drop the shared token, so we only
|
||||
// occupy VRAM while actually serving. The first decode re-acquires the token and restores.
|
||||
vram_share_init();
|
||||
vram_go_cold();
|
||||
|
||||
metrics.init();
|
||||
|
||||
if (params_base.cache_idle_slots) {
|
||||
@@ -3589,6 +3772,10 @@ private:
|
||||
n_empty_consecutive = 0;
|
||||
}
|
||||
|
||||
// VRAM arbiter: make sure our weights are resident (acquiring the shared GPU token first)
|
||||
// before we decode. Runs on the loop thread, so it never races an in-flight decode.
|
||||
vram_ensure_warm();
|
||||
|
||||
const int ret = llama_decode(ctx_tgt, batch_view);
|
||||
|
||||
metrics.on_decoded(slots);
|
||||
|
||||
@@ -116,6 +116,12 @@ void server_queue::wait_until_no_sleep() {
|
||||
}
|
||||
}
|
||||
|
||||
void server_queue::request_yield() {
|
||||
std::unique_lock<std::mutex> lock(mutex_tasks);
|
||||
yield_requested = true;
|
||||
condition_tasks.notify_all();
|
||||
}
|
||||
|
||||
void server_queue::terminate() {
|
||||
std::unique_lock<std::mutex> lock(mutex_tasks);
|
||||
running = false;
|
||||
@@ -129,6 +135,9 @@ void server_queue::start_loop(int64_t idle_sleep_ms) {
|
||||
constexpr auto max_wait_time = std::chrono::seconds(1);
|
||||
auto should_sleep = [&]() -> bool {
|
||||
// caller must hold mutex_tasks
|
||||
if (yield_requested) {
|
||||
return true; // another process rang the VRAM doorbell - release now
|
||||
}
|
||||
if (idle_sleep_ms < 0) {
|
||||
return false;
|
||||
}
|
||||
@@ -178,6 +187,7 @@ void server_queue::start_loop(int64_t idle_sleep_ms) {
|
||||
if (should_sleep()) {
|
||||
QUE_INF("%s", "entering sleeping state\n");
|
||||
sleeping = true;
|
||||
yield_requested = false; // consumed
|
||||
callback_sleeping_state(true);
|
||||
req_stop_sleeping = false;
|
||||
// wait until we are requested to exit sleeping state
|
||||
@@ -195,13 +205,17 @@ void server_queue::start_loop(int64_t idle_sleep_ms) {
|
||||
condition_tasks.notify_all(); // notify wait_until_no_sleep()
|
||||
break; // process new tasks
|
||||
} else {
|
||||
// wait for new tasks or timeout for checking sleeping condition
|
||||
// wait for new tasks, a VRAM yield request, or timeout for checking sleeping condition
|
||||
bool res = condition_tasks.wait_for(lock, max_wait_time, [&]{
|
||||
return (!queue_tasks.empty() || !running);
|
||||
return (!queue_tasks.empty() || !running || yield_requested);
|
||||
});
|
||||
if (res) {
|
||||
if (res && !queue_tasks.empty()) {
|
||||
break; // new task arrived or terminate
|
||||
}
|
||||
if (!running) {
|
||||
break;
|
||||
}
|
||||
// otherwise (timeout or yield request), loop again to re-check should_sleep
|
||||
// otherwise, loop again to check sleeping condition
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,6 +16,7 @@ private:
|
||||
bool running = false;
|
||||
bool sleeping = false;
|
||||
bool req_stop_sleeping = false;
|
||||
bool yield_requested = false; // set by request_yield() when another process wants the VRAM token
|
||||
int64_t time_last_task = 0;
|
||||
|
||||
// queues
|
||||
@@ -51,6 +52,11 @@ public:
|
||||
// returns immediately if not sleeping
|
||||
void wait_until_no_sleep();
|
||||
|
||||
// request that the loop go to sleep (release VRAM) as soon as it is idle - called from the
|
||||
// VRAM-arbiter warden thread when another process rings the doorbell for the GPU token.
|
||||
// Thread-safe; wakes the loop so it releases promptly instead of at the next idle poll.
|
||||
void request_yield();
|
||||
|
||||
bool is_sleeping() {
|
||||
std::unique_lock<std::mutex> lock(mutex_tasks);
|
||||
return sleeping;
|
||||
|
||||
Reference in New Issue
Block a user