llama: improve TENSOR_READ_LAZY handling (#27837)

* force lazy tensor on cpu if lazy is on

* llama: improve TENSOR_READ_LAZY handling
This commit is contained in:
Xuan-Son Nguyen
2026-08-30 16:59:48 +02:00
committed by GitHub
parent f1793c1c4e
commit 2578138397
4 changed files with 118 additions and 28 deletions
+61 -19
View File
@@ -1070,11 +1070,52 @@ static ggml_backend_buffer_type_t select_weight_buft(const llama_hparams & hpara
return nullptr;
}
ggml_backend_buffer_type_t llama_model_loader::lazy_read::buft() {
auto * cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);
if (!cpu_dev) {
throw std::runtime_error("no CPU backend found");
}
return ggml_backend_dev_buffer_type(cpu_dev);
}
bool llama_model_loader::lazy_read::add(const std::string & name, const ggml_tensor * t, const llama_tensor_weight * w) {
if (mode == LLAMA_LAZY_MODE_OFF) {
return false;
}
// do not lazy-read small tensors, it has significant overhead and is not worth it
constexpr size_t auto_min_size = 4ull * 1024 * 1024 * 1024;
if (mode != LLAMA_LAZY_MODE_ON && ggml_nbytes(t) <= auto_min_size) {
return false;
}
if (!llama_mmap::SUPPORTED) {
LLAMA_LOG_WARN("%s: mmap is not available, so tensor %s (size = %zu MiB) is loaded into RAM in full\n",
__func__, name.c_str(), ggml_nbytes(t)/1024/1024);
return false;
}
if (w) {
ranges[w->idx].emplace_back(w->offs, w->offs + ggml_nbytes(t));
tensors.insert(name);
LLAMA_LOG_INFO("%s: tensor %s (size = %zu MiB) lazy read enabled\n",
__func__, name.c_str(), ggml_nbytes(t)/1024/1024);
}
return true;
}
struct ggml_tensor * llama_model_loader::create_tensor(
const llama_hparams & hparams, const buft_list_t * buft_list_cpu, const buft_list_t * buft_list_input, const buft_list_t * buft_list_output,
const buft_list_t * buft_list_layer, const LLM_TN_IMPL & tn, const std::initializer_list<int64_t> & ne, int flags) {
// set below, before buft_for_tensor() runs
bool is_lazy = false;
auto ctx_for_buft = [&](ggml_backend_buffer_type_t buft) -> ggml_context * {
auto it = ctx_map.find(buft);
const ctx_key key { buft, is_lazy };
auto it = ctx_map.find(key);
if (it == ctx_map.end()) {
// one ggml context per buffer type
int max_n_tensors = n_tensors;
@@ -1096,7 +1137,7 @@ struct ggml_tensor * llama_model_loader::create_tensor(
throw std::runtime_error(format("failed to create ggml context"));
}
ctx_map.emplace(buft, ctx);
ctx_map.emplace(key, ctx);
return ctx;
}
@@ -1160,6 +1201,10 @@ struct ggml_tensor * llama_model_loader::create_tensor(
}
}
if (is_lazy) {
return lazy_read::buft();
}
// select the buffer type for this tensor
const buft_list_t * buft_list;
switch (info.layer) {
@@ -1287,16 +1332,9 @@ struct ggml_tensor * llama_model_loader::create_tensor(
return NULL;
}
if ((flags & TENSOR_READ_LAZY) && use_mmap && lazy_mode != LLAMA_LAZY_MODE_OFF) {
// in auto mode, small tensors are cheap enough to keep resident
constexpr size_t auto_lazy_min_size = 4ull * 1024 * 1024 * 1024;
if (lazy_mode == LLAMA_LAZY_MODE_ON || ggml_nbytes(cur) > auto_lazy_min_size) {
const auto & w = require_weight(tn.str().c_str());
lazy_tensor_ranges[w.idx].emplace_back(w.offs, w.offs + ggml_nbytes(cur));
LLAMA_LOG_INFO("%s: tensor %s (size = %zu MiB) lazy read enabled\n",
__func__, tn.str().c_str(), ggml_nbytes(cur)/1024/1024);
}
if (flags & TENSOR_READ_LAZY) {
// the decision must not depend on the load mode, or the memory-fit pass (no_alloc, no mmap)
is_lazy = lazy.add(tn.str(), cur, no_alloc ? nullptr : &require_weight(tn.str().c_str()));
}
ggml_tensor t_meta = *cur;
@@ -1363,7 +1401,8 @@ void llama_model_loader::done_getting_tensors(bool partial) const {
}
void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps) {
if (use_mmap) {
// note: read_lazy also requires mmap; this condition make sure it's usable even when --load-mode is not set to mmap
if (use_mmap || lazy.any()) {
mappings.reserve(files.size());
mmaps_used.reserve(files.size());
for (uint32_t idx = 0; idx < files.size(); idx++) {
@@ -1380,11 +1419,10 @@ void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps
}
}
const auto it_lazy = lazy_tensor_ranges.find(idx);
static const llama_mmap::ranges no_lazy_ranges;
const size_t prefetch_size = prefetch && use_mmap ? -1 : 0;
std::unique_ptr<llama_mmap> mapping = std::make_unique<llama_mmap>(file.get(), prefetch ? -1 : 0, is_numa,
it_lazy != lazy_tensor_ranges.end() ? it_lazy->second : no_lazy_ranges);
std::unique_ptr<llama_mmap> mapping = std::make_unique<llama_mmap>(file.get(), prefetch_size, is_numa,
lazy.for_file(idx));
mmaps_used.emplace_back(mapping->size(), 0);
if (mlock_mmaps) {
std::unique_ptr<llama_mlock> mlock_mmap(new llama_mlock());
@@ -1575,7 +1613,9 @@ bool llama_model_loader::load_all_data(
size_t n_size = ggml_nbytes(cur);
if (use_mmap) {
const bool from_mapping = use_mmap || lazy.has(cur);
if (from_mapping) {
const auto & mapping = mappings.at(weight->idx);
ggml_backend_buffer_t buf_mmap = nullptr;
if (bufs.count(weight->idx)) {
@@ -1592,7 +1632,9 @@ bool llama_model_loader::load_all_data(
GGML_ASSERT(buf_mmap || cur->data); // either we have a buffer to allocate the tensor in, or it is already allocated
if (buf_mmap && cur->data == nullptr) {
ggml_backend_tensor_alloc(buf_mmap, cur, data);
if (lmlocks) {
// locking a lazy tensor would fault all of it in, which is what lazy avoids
if (lmlocks && !lazy.has(cur)) {
const auto & lmlock = lmlocks->at(weight->idx);
lmlock->grow_to(weight->offs + n_size);
}
+49 -6
View File
@@ -12,6 +12,7 @@
#include <cstddef>
#include <cstring>
#include <map>
#include <set>
#include <stdexcept>
#include <unordered_map>
@@ -83,8 +84,38 @@ struct llama_model_loader {
bool no_alloc;
bool load_mtp;
// set by the caller before the create_tensor() calls
enum llama_lazy_mode lazy_mode = LLAMA_LAZY_MODE_OFF;
// handle TENSOR_READ_LAZY
// use case: keep PLE / engrams embd tensors on disk, read them on demand
struct lazy_read {
// set by the caller before the create_tensor() calls
enum llama_lazy_mode mode = LLAMA_LAZY_MODE_OFF;
// decide whether this tensor is read lazily
// pass w to also record it, or nullptr to only ask
bool add(const std::string & name, const ggml_tensor * t, const llama_tensor_weight * w);
bool any() const {
return !ranges.empty();
}
bool has(const ggml_tensor * t) const {
return tensors.count(ggml_get_name(t)) > 0;
}
const llama_mmap::ranges & for_file(uint32_t idx) const {
static const llama_mmap::ranges none;
const auto it = ranges.find(idx);
return it == ranges.end() ? none : it->second;
}
// lazy tensors are gathered on the host, so no offload setting applies to them
static ggml_backend_buffer_type_t buft();
private:
std::map<uint32_t, llama_mmap::ranges> ranges;
std::set<std::string> tensors;
} lazy;
llama_files files;
llama_ftype ftype;
@@ -92,9 +123,6 @@ struct llama_model_loader {
llama_mmaps mappings;
// byte ranges of TENSOR_READ_LAZY tensors, per file index
std::map<uint32_t, llama_mmap::ranges> lazy_tensor_ranges;
std::map<std::string, llama_tensor_weight, weight_name_comparer> weights_map;
std::unordered_map<std::string, llama_model_kv_override> kv_overrides;
const llama_model_tensor_buft_override * tensor_buft_overrides;
@@ -119,7 +147,22 @@ struct llama_model_loader {
}
};
std::map<ggml_backend_buffer_type_t, ggml_context_ptr, ggml_backend_buft_comparator> ctx_map;
// lazy tensors need dedicated context
struct ctx_key {
ggml_backend_buffer_type_t buft;
bool lazy;
};
struct ctx_key_comparator {
bool operator()(const ctx_key & lhs, const ctx_key & rhs) const {
if (lhs.lazy != rhs.lazy) {
return lhs.lazy < rhs.lazy;
}
return strcmp(ggml_backend_buft_name(lhs.buft), ggml_backend_buft_name(rhs.buft)) < 0;
}
};
std::map<ctx_key, ggml_context_ptr, ctx_key_comparator> ctx_map;
// track tensors that had to be moved for debugging:
size_t n_tensors_moved = 0;
+7 -2
View File
@@ -1689,7 +1689,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
const size_t n_max_backend_buffer = ml.ctx_map.size() * ml.files.size();
pimpl->ctxs_bufs.reserve(n_max_backend_buffer);
for (auto & [buft, ctx_ptr] : ml.ctx_map) {
for (auto & [ctx_key, ctx_ptr] : ml.ctx_map) {
ggml_backend_buffer_type_t buft = ctx_key.buft;
ggml_context * ctx = ctx_ptr.get();
// skip contexts without tensors
@@ -1715,7 +1716,11 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
bool is_default_buft = buft == ggml_backend_dev_buffer_type(dev);
std::vector<ggml_backend_buffer_ptr> bufs;
if (ml.use_mmap && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft) {
// a lazy context is mapped whatever the load mode, but the memory-fit pass maps nothing
const bool is_lazy_mapped = ctx_key.lazy && !ml.no_alloc;
if ((ml.use_mmap || is_lazy_mapped) && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft) {
GGML_ASSERT(!ml.no_alloc);
for (uint32_t idx = 0; idx < ml.files.size(); idx++) {
// only the mmap region containing the tensors in the model is mapped to the backend buffer
+1 -1
View File
@@ -318,7 +318,7 @@ static std::pair<int, llama_model *> llama_model_load(struct gguf_context * meta
llama_model_loader ml(metadata, set_tensor_data, set_tensor_data_ud, fname, splits, file, params.load_mode,
params.check_tensors, params.no_alloc, params.load_mtp, params.kv_overrides, params.tensor_buft_overrides);
ml.lazy_mode = params.lazy_mode;
ml.lazy.mode = params.lazy_mode;
ml.print_info();
std::unique_ptr<llama_model> model_ptr(llama_model_create(ml, params));