Compare commits
10
Commits
62873a34bb
...
80b91cbf76
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
80b91cbf76 | ||
|
|
da76fcb326 | ||
|
|
6d476d6ccc | ||
|
|
a8dc3ebddb | ||
|
|
5ddcf40fc1 | ||
|
|
fd7cbd5f4f | ||
|
|
4d576de9dd | ||
|
|
f707a2430d | ||
|
|
2af8f5ae6e | ||
|
|
189856ff9b |
@@ -104,6 +104,11 @@ extern "C" {
|
||||
GGML_API enum ggml_status ggml_backend_graph_compute (ggml_backend_t backend, struct ggml_cgraph * cgraph);
|
||||
GGML_API enum ggml_status ggml_backend_graph_compute_async(ggml_backend_t backend, struct ggml_cgraph * cgraph);
|
||||
|
||||
// Free transient/scratch device memory the backend holds outside of any allocated buffer
|
||||
// (compute preallocations, staging buffers). No-op if the backend does not implement it.
|
||||
// The backend remains usable; scratch is reallocated lazily on the next compute.
|
||||
GGML_API void ggml_backend_free_scratch(ggml_backend_t backend);
|
||||
|
||||
// NOTE: will be removed, use device version instead
|
||||
GGML_API bool ggml_backend_supports_op(ggml_backend_t backend, const struct ggml_tensor * op);
|
||||
GGML_API bool ggml_backend_supports_buft(ggml_backend_t backend, ggml_backend_buffer_type_t buft);
|
||||
|
||||
@@ -137,6 +137,11 @@ extern "C" {
|
||||
|
||||
// (optional) sort/optimize the nodes in the graph
|
||||
void (*graph_optimize) (ggml_backend_t backend, struct ggml_cgraph * cgraph);
|
||||
|
||||
// (optional) free transient/scratch device memory the backend holds outside of any buffer
|
||||
// (e.g. compute preallocations and staging buffers). The backend stays usable; the scratch
|
||||
// is reallocated lazily on the next compute. Used to shrink an idle model's device footprint.
|
||||
void (*free_scratch) (ggml_backend_t backend);
|
||||
};
|
||||
|
||||
struct ggml_backend {
|
||||
|
||||
@@ -420,6 +420,15 @@ void ggml_backend_synchronize(ggml_backend_t backend) {
|
||||
backend->iface.synchronize(backend);
|
||||
}
|
||||
|
||||
void ggml_backend_free_scratch(ggml_backend_t backend) {
|
||||
GGML_ASSERT(backend);
|
||||
if (backend->iface.free_scratch == NULL) {
|
||||
return;
|
||||
}
|
||||
|
||||
backend->iface.free_scratch(backend);
|
||||
}
|
||||
|
||||
ggml_backend_graph_plan_t ggml_backend_graph_plan_create(ggml_backend_t backend, struct ggml_cgraph * cgraph) {
|
||||
GGML_ASSERT(backend);
|
||||
GGML_ASSERT(backend->iface.graph_plan_create != NULL);
|
||||
|
||||
@@ -15843,6 +15843,35 @@ static const char * ggml_backend_vk_name(ggml_backend_t backend) {
|
||||
return ctx->name.c_str();
|
||||
}
|
||||
|
||||
// Free the compute-scratch device buffers (prealloc_* and the transfer staging buffer) without
|
||||
// tearing down the backend (pipelines, command pools, fences and device stay alive). These buffers
|
||||
// are reallocated lazily by ggml_vk_preallocate_buffers() on the next compute, so this just shrinks
|
||||
// an idle model's device footprint. Mirrors the buffer-freeing subset of ggml_vk_cleanup().
|
||||
static void ggml_backend_vk_free_scratch(ggml_backend_t backend) {
|
||||
ggml_backend_vk_context * ctx = (ggml_backend_vk_context *)backend->context;
|
||||
VK_LOG_DEBUG("ggml_backend_vk_free_scratch(" << ctx->name << ")");
|
||||
|
||||
// discard any unsubmitted command buffer and wait for in-flight work before freeing
|
||||
ctx->compute_ctx.reset();
|
||||
ggml_vk_synchronize(ctx);
|
||||
|
||||
ggml_vk_destroy_buffer(ctx->prealloc_x);
|
||||
ggml_vk_destroy_buffer(ctx->prealloc_y);
|
||||
ggml_vk_destroy_buffer(ctx->prealloc_split_k);
|
||||
ggml_vk_destroy_buffer(ctx->prealloc_add_rms_partials);
|
||||
ggml_vk_destroy_buffer(ctx->sync_staging);
|
||||
|
||||
ctx->prealloc_y_last_pipeline_used = nullptr;
|
||||
ctx->prealloc_y_last_tensor_used = nullptr;
|
||||
ctx->prealloc_y_last_decode_vector_staging = false;
|
||||
|
||||
ctx->prealloc_size_x = 0;
|
||||
ctx->prealloc_size_y = 0;
|
||||
ctx->prealloc_size_split_k = 0;
|
||||
ctx->prealloc_size_add_rms_partials = 0;
|
||||
ctx->prealloc_size_add_rms_partials_offset = 0;
|
||||
}
|
||||
|
||||
static void ggml_backend_vk_free(ggml_backend_t backend) {
|
||||
ggml_backend_vk_context * ctx = (ggml_backend_vk_context *)backend->context;
|
||||
VK_LOG_DEBUG("ggml_backend_vk_free(" << ctx->name << ")");
|
||||
@@ -17378,6 +17407,7 @@ static ggml_backend_i ggml_backend_vk_interface = {
|
||||
/* .event_record = */ ggml_backend_vk_event_record,
|
||||
/* .event_wait = */ ggml_backend_vk_event_wait,
|
||||
/* .graph_optimize = */ ggml_vk_graph_optimize,
|
||||
/* .free_scratch = */ ggml_backend_vk_free_scratch,
|
||||
};
|
||||
|
||||
static ggml_guid_t ggml_backend_vk_guid() {
|
||||
|
||||
+29
-12
@@ -738,11 +738,20 @@ void llama_context::release_device(bool evict_kv) {
|
||||
if (evict_kv && memory && !kv_device_evicted) {
|
||||
memory->release_device_buffers();
|
||||
kv_device_evicted = true;
|
||||
// also free the scheduler and its worst-case compute buffer (hundreds of MiB) so a cold
|
||||
// model holds essentially no VRAM. Rebuilt lazily by sched_reserve() on restore.
|
||||
sched.reset();
|
||||
sched_need_reserve = true;
|
||||
// finally, free each backend's own compute-scratch (Vulkan prealloc/staging buffers, which
|
||||
// are owned by the backend and survive sched.reset()). Reallocated lazily on next compute.
|
||||
for (auto & backend : backends) {
|
||||
ggml_backend_free_scratch(backend.get());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void llama_context::restore_device() {
|
||||
if (model.weights_resident() && !kv_device_evicted) {
|
||||
if (model.weights_resident() && !kv_device_evicted && sched) {
|
||||
return;
|
||||
}
|
||||
model.restore_device_weights();
|
||||
@@ -750,6 +759,10 @@ void llama_context::restore_device() {
|
||||
memory->restore_device_buffers();
|
||||
kv_device_evicted = false;
|
||||
}
|
||||
// rebuild the scheduler + compute buffer if they were freed on release (evict_kv mode)
|
||||
if (!sched) {
|
||||
sched_reserve();
|
||||
}
|
||||
// make sure all weight/KV uploads have completed before any compute reads them
|
||||
if (sched) {
|
||||
ggml_backend_sched_synchronize(sched.get());
|
||||
@@ -3258,17 +3271,21 @@ llama_memory_breakdown llama_context::memory_breakdown() const {
|
||||
ret[buft].context += size;
|
||||
}
|
||||
}
|
||||
if (model.hparams.no_alloc) {
|
||||
for (size_t i = 0; i < backends.size(); ++i) {
|
||||
ggml_backend_t backend = backends[i].get();
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_sched_get_buffer_type(sched.get(), backend);
|
||||
ret[buft].compute += backend_buf_exp_size[i];
|
||||
}
|
||||
} else {
|
||||
for (const auto & backend_ptr : backends) {
|
||||
ggml_backend_t backend = backend_ptr.get();
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_sched_get_buffer_type(sched.get(), backend);
|
||||
ret[buft].compute += ggml_backend_sched_get_buffer_size(sched.get(), backend);
|
||||
// the scheduler (and its compute buffers) may have been freed while the model is cold
|
||||
// (on-demand VRAM eviction, see release_device); it contributes no compute VRAM then.
|
||||
if (sched) {
|
||||
if (model.hparams.no_alloc) {
|
||||
for (size_t i = 0; i < backends.size(); ++i) {
|
||||
ggml_backend_t backend = backends[i].get();
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_sched_get_buffer_type(sched.get(), backend);
|
||||
ret[buft].compute += backend_buf_exp_size[i];
|
||||
}
|
||||
} else {
|
||||
for (const auto & backend_ptr : backends) {
|
||||
ggml_backend_t backend = backend_ptr.get();
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_sched_get_buffer_type(sched.get(), backend);
|
||||
ret[buft].compute += ggml_backend_sched_get_buffer_size(sched.get(), backend);
|
||||
}
|
||||
}
|
||||
}
|
||||
return ret;
|
||||
|
||||
@@ -398,7 +398,9 @@ void llama_kv_cache::clear(bool data) {
|
||||
|
||||
if (data) {
|
||||
for (auto & [_, buf] : ctxs_bufs) {
|
||||
ggml_backend_buffer_clear(buf.get(), 0);
|
||||
if (buf) { // may be null if evicted for on-demand VRAM sharing
|
||||
ggml_backend_buffer_clear(buf.get(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -741,6 +743,9 @@ llama_pos llama_kv_cache::seq_pos_max(llama_seq_id seq_id) const {
|
||||
std::map<ggml_backend_buffer_type_t, size_t> llama_kv_cache::memory_breakdown() const {
|
||||
std::map<ggml_backend_buffer_type_t, size_t> ret;
|
||||
for (const auto & [ctx, buf] : ctxs_bufs) {
|
||||
if (!buf) { // may be null if evicted for on-demand VRAM sharing
|
||||
continue;
|
||||
}
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_buffer_get_type(buf.get());
|
||||
|
||||
if (hparams.no_alloc) {
|
||||
@@ -1953,7 +1958,9 @@ size_t llama_kv_cache::total_size() const {
|
||||
size_t size = 0;
|
||||
|
||||
for (const auto & [_, buf] : ctxs_bufs) {
|
||||
size += ggml_backend_buffer_get_size(buf.get());
|
||||
if (buf) { // may be null if evicted for on-demand VRAM sharing
|
||||
size += ggml_backend_buffer_get_size(buf.get());
|
||||
}
|
||||
}
|
||||
|
||||
return size;
|
||||
|
||||
@@ -202,7 +202,7 @@ void llama_memory_hybrid::state_read(llama_io_read_i & io, llama_seq_id seq_id,
|
||||
}
|
||||
|
||||
void llama_memory_hybrid::release_device_buffers() {
|
||||
// evict the attention KV (grows with context); the recurrent state uses the no-op default
|
||||
// evict both the attention KV (grows with context) and the recurrent/SSM state
|
||||
mem_attn->release_device_buffers();
|
||||
mem_recr->release_device_buffers();
|
||||
}
|
||||
|
||||
@@ -140,7 +140,9 @@ void llama_memory_recurrent::clear(bool data) {
|
||||
|
||||
if (data) {
|
||||
for (auto & [_, buf] : ctxs_bufs) {
|
||||
ggml_backend_buffer_clear(buf.get(), 0);
|
||||
if (buf) { // may be null if evicted for on-demand VRAM sharing
|
||||
ggml_backend_buffer_clear(buf.get(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -399,6 +401,7 @@ void llama_memory_recurrent::set_rs_idx(llama_seq_id seq_id, uint32_t idx) {
|
||||
std::map<ggml_backend_buffer_type_t, size_t> llama_memory_recurrent::memory_breakdown() const {
|
||||
std::map<ggml_backend_buffer_type_t, size_t> ret;
|
||||
for (const auto & [_, buf] : ctxs_bufs) {
|
||||
if (!buf) { continue; } // may be null if evicted for on-demand VRAM sharing
|
||||
ret[ggml_backend_buffer_get_type(buf.get())] += ggml_backend_buffer_get_size(buf.get());
|
||||
}
|
||||
return ret;
|
||||
@@ -700,12 +703,87 @@ bool llama_memory_recurrent::get_can_shift() const {
|
||||
size_t llama_memory_recurrent::total_size() const {
|
||||
size_t size = 0;
|
||||
for (const auto & [_, buf] : ctxs_bufs) {
|
||||
size += ggml_backend_buffer_get_size(buf.get());
|
||||
if (buf) { // may be null if evicted for on-demand VRAM sharing
|
||||
size += ggml_backend_buffer_get_size(buf.get());
|
||||
}
|
||||
}
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
void llama_memory_recurrent::release_device_buffers() {
|
||||
// Same mechanism as llama_kv_cache: the recurrent (SSM/conv) state is read-write, so its host
|
||||
// shadow is (re)captured on every release. The caller must have synchronized the backend.
|
||||
if (dev_released) {
|
||||
return;
|
||||
}
|
||||
dev_shadows.assign(ctxs_bufs.size(), device_buffer_shadow{});
|
||||
size_t freed = 0;
|
||||
for (size_t i = 0; i < ctxs_bufs.size(); ++i) {
|
||||
ggml_context * ctx = ctxs_bufs[i].first.get();
|
||||
ggml_backend_buffer_t buf = ctxs_bufs[i].second.get();
|
||||
if (buf == nullptr || ggml_backend_buffer_is_host(buf) || ggml_backend_buffer_get_size(buf) == 0) {
|
||||
continue;
|
||||
}
|
||||
auto & sh = dev_shadows[i];
|
||||
sh.releasable = true;
|
||||
sh.buft = ggml_backend_buffer_get_type(buf);
|
||||
|
||||
size_t total = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src == nullptr) { total += ggml_nbytes(t); }
|
||||
}
|
||||
sh.data.resize(total);
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src != nullptr) { continue; }
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_get(t, sh.data.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
|
||||
freed += ggml_backend_buffer_get_size(buf);
|
||||
ctxs_bufs[i].second.reset();
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
t->buffer = nullptr;
|
||||
t->data = nullptr;
|
||||
}
|
||||
}
|
||||
dev_released = true;
|
||||
if (freed > 0) {
|
||||
LLAMA_LOG_INFO("%s: released %.2f MiB of recurrent state from device\n", __func__, freed / 1024.0 / 1024.0);
|
||||
}
|
||||
}
|
||||
|
||||
bool llama_memory_recurrent::restore_device_buffers() {
|
||||
if (!dev_released) {
|
||||
return true;
|
||||
}
|
||||
for (size_t i = 0; i < ctxs_bufs.size(); ++i) {
|
||||
auto & sh = dev_shadows[i];
|
||||
if (!sh.releasable) {
|
||||
continue;
|
||||
}
|
||||
ggml_context * ctx = ctxs_bufs[i].first.get();
|
||||
ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, sh.buft);
|
||||
if (buf == nullptr) {
|
||||
LLAMA_LOG_ERROR("%s: failed to reallocate recurrent device buffer (out of VRAM?)\n", __func__);
|
||||
return false;
|
||||
}
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
|
||||
if (t->view_src != nullptr) { continue; }
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_set(t, sh.data.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
ctxs_bufs[i].second.reset(buf);
|
||||
}
|
||||
dev_released = false;
|
||||
dev_shadows.clear();
|
||||
return true;
|
||||
}
|
||||
|
||||
size_t llama_memory_recurrent::size_r_bytes() const {
|
||||
size_t size_r_bytes = 0;
|
||||
|
||||
|
||||
@@ -66,6 +66,10 @@ public:
|
||||
void state_write(llama_io_write_i & io, llama_seq_id seq_id = -1, llama_state_seq_flags flags = 0) const override;
|
||||
void state_read (llama_io_read_i & io, llama_seq_id seq_id = -1, llama_state_seq_flags flags = 0) override;
|
||||
|
||||
// on-demand device (VRAM) residency (see llama_memory_i)
|
||||
void release_device_buffers() override;
|
||||
bool restore_device_buffers() override;
|
||||
|
||||
uint32_t head = 0; // the location where the batch will be placed in the cache (see find_slot())
|
||||
uint32_t size = 0; // total number of cells, shared across all sequences
|
||||
uint32_t used = 0; // used cells (i.e. at least one seq_id)
|
||||
@@ -121,6 +125,16 @@ private:
|
||||
// ggml contexts for the KV cache along with the allocated backend buffers:
|
||||
std::vector<std::pair<ggml_context_ptr, ggml_backend_buffer_ptr>> ctxs_bufs;
|
||||
|
||||
// on-demand device eviction (see release_device_buffers): host shadow of each device buffer's
|
||||
// live contents (recaptured on every release since the recurrent state is read-write)
|
||||
struct device_buffer_shadow {
|
||||
ggml_backend_buffer_type_t buft = nullptr;
|
||||
bool releasable = false;
|
||||
std::vector<uint8_t> data;
|
||||
};
|
||||
std::vector<device_buffer_shadow> dev_shadows; // parallel to ctxs_bufs
|
||||
bool dev_released = false;
|
||||
|
||||
size_t total_size() const;
|
||||
|
||||
size_t size_r_bytes() const;
|
||||
|
||||
@@ -1838,6 +1838,11 @@ std::map<ggml_backend_buffer_type_t, size_t> llama_model::memory_breakdown() con
|
||||
ret[buft] += ggml_backend_alloc_ctx_tensors_from_buft_size(ctx.get(), buft);
|
||||
} else {
|
||||
for (const auto & buf : bufs) {
|
||||
// buf may be null if the device weights were released for on-demand VRAM sharing
|
||||
// (see release_device_weights); skip it in the breakdown
|
||||
if (!buf) {
|
||||
continue;
|
||||
}
|
||||
// GGML_ASSERT(ggml_backend_buffer_get_base(buf.get()) != nullptr); // multi_buffer does not have a defined base
|
||||
ret[ggml_backend_buffer_get_type(buf.get())] += ggml_backend_buffer_get_size(buf.get());
|
||||
}
|
||||
|
||||
@@ -158,6 +158,13 @@ struct clip_ctx {
|
||||
ggml_backend_t backend_cpu = nullptr;
|
||||
ggml_backend_buffer_ptr buf;
|
||||
|
||||
// on-demand device (VRAM) residency: the vision/audio encoder weights are read-only, so a host
|
||||
// shadow is captured once and the device buffer can be freed while the model is cold, then
|
||||
// rebuilt on wake (mirrors llama_model::release_device_weights). See clip_release_device().
|
||||
ggml_backend_buffer_type_t dev_buft = nullptr;
|
||||
std::vector<uint8_t> dev_shadow;
|
||||
bool dev_released = false;
|
||||
|
||||
|
||||
int max_nodes = 8192;
|
||||
ggml_backend_sched_ptr sched;
|
||||
@@ -3241,6 +3248,74 @@ void clip_free(clip_ctx * ctx) {
|
||||
delete ctx;
|
||||
}
|
||||
|
||||
void clip_release_device(struct clip_ctx * ctx) {
|
||||
if (ctx == nullptr || ctx->dev_released) {
|
||||
return;
|
||||
}
|
||||
ggml_backend_buffer_t buf = ctx->buf.get();
|
||||
if (buf == nullptr || ggml_backend_buffer_is_host(buf) || ggml_backend_buffer_get_size(buf) == 0) {
|
||||
return; // CPU-backed encoder: nothing in VRAM to free
|
||||
}
|
||||
// ensure no encode is in flight before freeing the weights
|
||||
if (ctx->sched) {
|
||||
ggml_backend_sched_synchronize(ctx->sched.get());
|
||||
}
|
||||
ctx->dev_buft = ggml_backend_buffer_get_type(buf);
|
||||
|
||||
ggml_context * cd = ctx->ctx_data.get();
|
||||
// capture the host shadow once (weights are read-only): compact, stable iteration order,
|
||||
// skipping view tensors (which alias a base and are restored implicitly)
|
||||
if (ctx->dev_shadow.empty()) {
|
||||
size_t total = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(cd); t != nullptr; t = ggml_get_next_tensor(cd, t)) {
|
||||
if (t->view_src == nullptr) { total += ggml_nbytes(t); }
|
||||
}
|
||||
ctx->dev_shadow.resize(total);
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(cd); t != nullptr; t = ggml_get_next_tensor(cd, t)) {
|
||||
if (t->view_src != nullptr) { continue; }
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_get(t, ctx->dev_shadow.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
}
|
||||
|
||||
// free the device buffer and clear the now-dangling tensor pointers so restore reallocates cleanly
|
||||
ctx->buf.reset();
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(cd); t != nullptr; t = ggml_get_next_tensor(cd, t)) {
|
||||
t->buffer = nullptr;
|
||||
t->data = nullptr;
|
||||
}
|
||||
ctx->dev_released = true;
|
||||
}
|
||||
|
||||
bool clip_restore_device(struct clip_ctx * ctx) {
|
||||
if (ctx == nullptr || !ctx->dev_released) {
|
||||
return true;
|
||||
}
|
||||
ggml_context * cd = ctx->ctx_data.get();
|
||||
ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(cd, ctx->dev_buft);
|
||||
if (buf == nullptr) {
|
||||
LOG_ERR("%s: failed to reallocate encoder device buffer (out of VRAM?)\n", __func__);
|
||||
return false;
|
||||
}
|
||||
ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
|
||||
size_t off = 0;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(cd); t != nullptr; t = ggml_get_next_tensor(cd, t)) {
|
||||
if (t->view_src != nullptr) { continue; }
|
||||
const size_t n = ggml_nbytes(t);
|
||||
ggml_backend_tensor_set(t, ctx->dev_shadow.data() + off, 0, n);
|
||||
off += n;
|
||||
}
|
||||
ctx->buf.reset(buf);
|
||||
ctx->dev_released = false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool clip_weights_resident(const struct clip_ctx * ctx) {
|
||||
return ctx == nullptr || !ctx->dev_released;
|
||||
}
|
||||
|
||||
const char * clip_patch_merge_type(const struct clip_ctx * ctx) {
|
||||
return ctx->model.hparams.mm_patch_merge_type == PATCH_MERGE_SPATIAL_UNPAD ? "spatial_unpad" : "flat";
|
||||
}
|
||||
|
||||
@@ -67,6 +67,12 @@ struct clip_init_result clip_init(const char * fname, struct clip_context_params
|
||||
|
||||
void clip_free(struct clip_ctx * ctx);
|
||||
|
||||
// on-demand device (VRAM) residency: free/rebuild the encoder weight buffer to shrink an idle
|
||||
// (cold) multimodal model's VRAM footprint. No-op for a CPU-backed encoder. See clip.cpp.
|
||||
void clip_release_device(struct clip_ctx * ctx);
|
||||
bool clip_restore_device(struct clip_ctx * ctx);
|
||||
bool clip_weights_resident(const struct clip_ctx * ctx);
|
||||
|
||||
// TODO: should be enum, not string
|
||||
const char * clip_patch_merge_type(const struct clip_ctx * ctx);
|
||||
|
||||
|
||||
@@ -814,6 +814,24 @@ void mtmd_free(mtmd_context * ctx) {
|
||||
delete ctx;
|
||||
}
|
||||
|
||||
void mtmd_release_device(mtmd_context * ctx) {
|
||||
if (ctx == nullptr) {
|
||||
return;
|
||||
}
|
||||
if (ctx->ctx_v) { clip_release_device(ctx->ctx_v); }
|
||||
if (ctx->ctx_a) { clip_release_device(ctx->ctx_a); }
|
||||
}
|
||||
|
||||
bool mtmd_restore_device(mtmd_context * ctx) {
|
||||
if (ctx == nullptr) {
|
||||
return true;
|
||||
}
|
||||
bool ok = true;
|
||||
if (ctx->ctx_v) { ok = clip_restore_device(ctx->ctx_v) && ok; }
|
||||
if (ctx->ctx_a) { ok = clip_restore_device(ctx->ctx_a) && ok; }
|
||||
return ok;
|
||||
}
|
||||
|
||||
struct mtmd_tokenizer {
|
||||
mtmd_context * ctx;
|
||||
|
||||
|
||||
@@ -127,6 +127,12 @@ MTMD_API mtmd_context * mtmd_init_from_file(const char * mmproj_fname,
|
||||
|
||||
MTMD_API void mtmd_free(mtmd_context * ctx);
|
||||
|
||||
// on-demand device (VRAM) residency: free / rebuild the vision+audio encoder weight buffers so an
|
||||
// idle (cold) multimodal model releases its encoder VRAM and reclaims it on wake. No-op for a
|
||||
// CPU-backed encoder. Restore returns false if reallocation failed (out of VRAM).
|
||||
MTMD_API void mtmd_release_device(mtmd_context * ctx);
|
||||
MTMD_API bool mtmd_restore_device(mtmd_context * ctx);
|
||||
|
||||
// whether we need to set non-causal mask before llama_decode
|
||||
// if chunk is nullptr, we assume the default case where chunk is an image chunk
|
||||
MTMD_API bool mtmd_decode_use_non_causal(const mtmd_context * ctx, const mtmd_input_chunk * chunk);
|
||||
|
||||
+108
-14
@@ -780,7 +780,36 @@ struct server_slot {
|
||||
// TODO @ngxson : move this log line to debug when it become more stable
|
||||
SLT_TRC(*this, "encoding mtmd batch from idx = %zu, n_chunks = %d\n", idx, n_added);
|
||||
|
||||
// Bring the vision/audio encoder into VRAM just for this encode. In on-demand mode the
|
||||
// encoder is normally kept in RAM (its VRAM freed for KV / expert cache). If it does not
|
||||
// fit, evict the LLM backbone weights first: they are NOT needed while the encoder runs (the
|
||||
// decode that uses them happens afterwards) and their host shadow is read-only, so this is a
|
||||
// cheap free (no D2H) and leaves the KV cache / prompt cache untouched.
|
||||
static const bool mmproj_ondemand = getenv("LLAMA_MMPROJ_ONDEMAND") != nullptr;
|
||||
bool weights_evicted = false;
|
||||
if (mctx && !mtmd_restore_device(mctx)) {
|
||||
if (mmproj_ondemand) {
|
||||
llama_context_release_device(ctx_tgt, /* evict_kv = */ false); // free backbone, keep KV
|
||||
weights_evicted = true;
|
||||
}
|
||||
if (!mtmd_restore_device(mctx)) {
|
||||
if (weights_evicted) { llama_context_restore_device(ctx_tgt); }
|
||||
SLT_ERR(*this, "%s", "failed to bring the multimodal encoder into VRAM for encoding\n");
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
res = mtmd_batch_encode(mbatch.get());
|
||||
|
||||
// release the encoder VRAM again (on-demand), then restore the backbone weights for decode.
|
||||
// Order matters: free the encoder BEFORE re-uploading the weights so the peak stays within VRAM.
|
||||
if (mctx && mmproj_ondemand) {
|
||||
mtmd_release_device(mctx);
|
||||
}
|
||||
if (weights_evicted) {
|
||||
llama_context_restore_device(ctx_tgt);
|
||||
}
|
||||
|
||||
if (res != 0) {
|
||||
SLT_ERR(*this, "failed to encode mtmd batch for chunk idx = %zu, res = %d\n", idx, res);
|
||||
return -1;
|
||||
@@ -927,14 +956,17 @@ private:
|
||||
std::thread vram_warden;
|
||||
std::atomic<bool> vram_warden_run{false};
|
||||
|
||||
void vram_share_init() {
|
||||
// open the shared arena (flock token + doorbell dir). Idempotent; sets vram_only/vram_flock.
|
||||
void vram_arena_open() {
|
||||
if (getenv("LLAMA_SLEEP_VRAM_ONLY") == nullptr) {
|
||||
return;
|
||||
}
|
||||
vram_only = true;
|
||||
vram_evict_kv = getenv("LLAMA_SLEEP_EVICT_KV") != nullptr;
|
||||
vram_cold = false; // weights are resident right after load
|
||||
#if !defined(_WIN32)
|
||||
if (vram_lock_fd >= 0) {
|
||||
return; // already open
|
||||
}
|
||||
const char * arena_env = getenv("LLAMA_VRAM_ARENA");
|
||||
const std::string arena = arena_env ? arena_env : "/dev/shm/llama-vram";
|
||||
mkdir(arena.c_str(), 0777);
|
||||
@@ -945,16 +977,49 @@ private:
|
||||
vram_pid_str = std::to_string(getpid());
|
||||
vram_doorbell_dir = arena + "/doorbell";
|
||||
mkdir(vram_doorbell_dir.c_str(), 0777);
|
||||
SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n",
|
||||
lock_path.c_str(), vram_doorbell_dir.c_str());
|
||||
} else {
|
||||
SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// Acquire the VRAM token BEFORE uploading this model's weights, so any model currently resident
|
||||
// on the GPU releases first and the load uploads into free VRAM instead of racing it (which
|
||||
// could OOM, e.g. the 4B task model holding VRAM while a large model loads). Blocks until free.
|
||||
// Called from load_model() right before common_init_from_params().
|
||||
void vram_acquire_for_load() {
|
||||
vram_arena_open();
|
||||
if (!vram_only) {
|
||||
return;
|
||||
}
|
||||
#if !defined(_WIN32)
|
||||
if (vram_flock) {
|
||||
vram_ring_doorbell(); // nudge whoever is resident to release
|
||||
flock(vram_lock_fd, LOCK_EX); // block until the GPU is free
|
||||
}
|
||||
#endif
|
||||
vram_cold = false; // we hold the token; weights will be resident after the upload
|
||||
}
|
||||
|
||||
// Start the doorbell warden. Called from init() after the (coordinated) load. The model stays
|
||||
// warm (holding the token acquired in vram_acquire_for_load) and serves its first request
|
||||
// without a re-warm; the warden releases it when another model rings the doorbell.
|
||||
void vram_share_init() {
|
||||
vram_arena_open();
|
||||
if (!vram_only) {
|
||||
return;
|
||||
}
|
||||
vram_cold = false; // resident and holding the token after the coordinated load
|
||||
#if !defined(_WIN32)
|
||||
if (vram_flock && vram_inotify_fd < 0) {
|
||||
vram_inotify_fd = inotify_init1(IN_NONBLOCK);
|
||||
if (vram_inotify_fd >= 0) {
|
||||
inotify_add_watch(vram_inotify_fd, vram_doorbell_dir.c_str(), IN_CLOSE_WRITE);
|
||||
vram_warden_run = true;
|
||||
vram_warden = std::thread([this]{ vram_warden_loop(); });
|
||||
}
|
||||
SRV_INF("VRAM arbiter: cross-process GPU sharing via %s (doorbell %s)\n",
|
||||
lock_path.c_str(), vram_doorbell_dir.c_str());
|
||||
} else {
|
||||
SRV_WRN("VRAM arbiter: cannot open %s, running VRAM-only sleep without cross-process lock\n", lock_path.c_str());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -1022,6 +1087,11 @@ private:
|
||||
if (ctx_dft != nullptr) {
|
||||
llama_context_restore_device(ctx_dft);
|
||||
}
|
||||
// NOTE: the multimodal (vision/audio) encoder is intentionally NOT restored here. It is
|
||||
// brought into VRAM just-in-time before an image/audio encode (process_mtmd_chunk) and, in
|
||||
// on-demand mode, released again right after - so a warm text-only model holds no encoder
|
||||
// VRAM (that space is free for KV / expert cache). A cold->warm wake for a *text* request
|
||||
// therefore leaves the encoder in RAM; an image request restores it at encode time.
|
||||
vram_cold = false;
|
||||
}
|
||||
|
||||
@@ -1046,6 +1116,11 @@ private:
|
||||
if (ctx_dft != nullptr) {
|
||||
llama_context_release_device(ctx_dft, vram_evict_kv);
|
||||
}
|
||||
// also release the multimodal (vision/audio) encoder weights (dead weight while cold);
|
||||
// its read-only host shadow is captured once and rebuilt on wake by vram_ensure_warm()
|
||||
if (mctx != nullptr) {
|
||||
mtmd_release_device(mctx);
|
||||
}
|
||||
#if !defined(_WIN32)
|
||||
if (vram_flock) {
|
||||
flock(vram_lock_fd, LOCK_UN);
|
||||
@@ -1123,8 +1198,12 @@ private:
|
||||
SRV_INF("%s", "server entering sleeping state (VRAM-only: releasing device weights, keeping KV cache)\n");
|
||||
vram_go_cold();
|
||||
} else {
|
||||
SRV_INF("%s", "server exiting sleeping state (VRAM-only: weights restored on next decode)\n");
|
||||
// token is acquired and weights restored by vram_ensure_warm() before decode
|
||||
SRV_INF("%s", "server exiting sleeping state (VRAM-only: restoring device weights/KV)\n");
|
||||
// Restore NOW, on wake, before update_slots runs. update_slots touches the KV cache
|
||||
// (e.g. SWA checkpoint creation reads it via ggml_backend_tensor_get) before the
|
||||
// decode-time vram_ensure_warm(), so with KV eviction the KV must already be resident
|
||||
// here or those reads hit a freed (null) buffer.
|
||||
vram_ensure_warm();
|
||||
}
|
||||
sleeping = new_state;
|
||||
return;
|
||||
@@ -1321,6 +1400,11 @@ private:
|
||||
params_base.load_progress_callback_user_data = &load_progress_text;
|
||||
}
|
||||
|
||||
// VRAM arbiter: acquire the GPU token before uploading weights, so any resident model (e.g.
|
||||
// the warm 4B task model) releases first and this load uploads into free VRAM instead of
|
||||
// racing it and OOM-ing. No-op unless LLAMA_SLEEP_VRAM_ONLY is set.
|
||||
vram_acquire_for_load();
|
||||
|
||||
llama_init = common_init_from_params(params_base);
|
||||
|
||||
model_tgt = llama_init->model();
|
||||
@@ -1391,6 +1475,13 @@ private:
|
||||
}
|
||||
SRV_INF("loaded multimodal model, '%s'\n", mmproj_path.c_str());
|
||||
|
||||
// on-demand encoder: keep the vision/audio encoder weights in RAM (out of VRAM) until an
|
||||
// image/audio actually needs encoding, freeing that VRAM for KV / expert cache on the
|
||||
// common text-only path. process_mtmd_chunk() brings the encoder in just-in-time.
|
||||
if (getenv("LLAMA_MMPROJ_ONDEMAND") != nullptr) {
|
||||
mtmd_release_device(mctx);
|
||||
}
|
||||
|
||||
if (params_base.ctx_shift) {
|
||||
params_base.ctx_shift = false;
|
||||
SRV_WRN("%s\n", "ctx_shift is not supported by multimodal, it will be disabled");
|
||||
@@ -1588,11 +1679,10 @@ private:
|
||||
handle_sleeping_state(sleeping);
|
||||
});
|
||||
|
||||
// VRAM arbiter: enable cross-process GPU time-sharing (if LLAMA_SLEEP_VRAM_ONLY is set) and
|
||||
// start cold - release the just-loaded device weights and drop the shared token, so we only
|
||||
// occupy VRAM while actually serving. The first decode re-acquires the token and restores.
|
||||
// VRAM arbiter: start the doorbell warden. The load was coordinated (vram_acquire_for_load
|
||||
// grabbed the token before uploading), so we stay warm holding the token and serve the first
|
||||
// request without a re-warm; the warden releases us when another model rings the doorbell.
|
||||
vram_share_init();
|
||||
vram_go_cold();
|
||||
|
||||
metrics.init();
|
||||
|
||||
@@ -4199,8 +4289,12 @@ struct server_res_generator : server_res_spipe {
|
||||
server_response_reader rd;
|
||||
server_res_generator(server_queue & queue_tasks, server_response & queue_results, int sleep_idle_seconds, bool bypass_sleep = false)
|
||||
: rd(queue_tasks, queue_results, HTTP_POLLING_SECONDS) {
|
||||
// fast path in case sleeping is disabled
|
||||
bypass_sleep |= sleep_idle_seconds < 0;
|
||||
// fast path in case sleeping is disabled. Note: the VRAM arbiter (LLAMA_SLEEP_VRAM_ONLY)
|
||||
// can put the server to sleep via the cross-process doorbell even when idle-sleep is
|
||||
// disabled (sleep_idle_seconds < 0), so in that case requests must still wake it - otherwise
|
||||
// a doorbell-slept server would hang, never returning from a request.
|
||||
static const bool vram_arbiter = getenv("LLAMA_SLEEP_VRAM_ONLY") != nullptr;
|
||||
bypass_sleep |= (sleep_idle_seconds < 0 && !vram_arbiter);
|
||||
if (!bypass_sleep) {
|
||||
queue_tasks.wait_until_no_sleep();
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user