ggml-org/llama.cpp · #27837

llama: improve TENSOR_READ_LAZY handling

ngxson · merged Aug 30, 20264 files · 118 + / 28
src/llama-model-loader.cpp61 + / 19
@@ -1070,11 +1070,52 @@ static ggml_backend_buffer_type_t select_weight_buft(const llama_hparams & hpara     return nullptr; } +ggml_backend_buffer_type_t llama_model_loader::lazy_read::buft() {+    auto * cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);+    if (!cpu_dev) {+        throw std::runtime_error("no CPU backend found");+    }+    return ggml_backend_dev_buffer_type(cpu_dev);+}++bool llama_model_loader::lazy_read::add(const std::string & name, const ggml_tensor * t, const llama_tensor_weight * w) {+    if (mode == LLAMA_LAZY_MODE_OFF) {+        return false;+    }++    // do not lazy-read small tensors, it has significant overhead and is not worth it+    constexpr size_t auto_min_size = 4ull * 1024 * 1024 * 1024;+    if (mode != LLAMA_LAZY_MODE_ON && ggml_nbytes(t) <= auto_min_size) {+        return false;+    }++    if (!llama_mmap::SUPPORTED) {+        LLAMA_LOG_WARN("%s: mmap is not available, so tensor %s (size = %zu MiB) is loaded into RAM in full\n",+                __func__, name.c_str(), ggml_nbytes(t)/1024/1024);+        return false;+    }++    if (w) {+        ranges[w->idx].emplace_back(w->offs, w->offs + ggml_nbytes(t));+        tensors.insert(name);++        LLAMA_LOG_INFO("%s: tensor %s (size = %zu MiB) lazy read enabled\n",+                __func__, name.c_str(), ggml_nbytes(t)/1024/1024);+    }++    return true;+}+ struct ggml_tensor * llama_model_loader::create_tensor(         const llama_hparams & hparams, const buft_list_t * buft_list_cpu, const buft_list_t * buft_list_input, const buft_list_t * buft_list_output,         const buft_list_t * buft_list_layer, const LLM_TN_IMPL & tn, const std::initializer_list<int64_t> & ne, int flags) {+    // set below, before buft_for_tensor() runs+    bool is_lazy = false;+     auto ctx_for_buft = [&](ggml_backend_buffer_type_t buft) -> ggml_context * {-        auto it = ctx_map.find(buft);+        const ctx_key key { buft, is_lazy };++        auto it = ctx_map.find(key);         if (it == ctx_map.end()) {             // one ggml context per buffer type             int max_n_tensors = n_tensors;@@ -1096,7 +1137,7 @@ struct ggml_tensor * llama_model_loader::create_tensor(                 throw std::runtime_error(format("failed to create ggml context"));             } -            ctx_map.emplace(buft, ctx);+            ctx_map.emplace(key, ctx);              return ctx;         }@@ -1160,6 +1201,10 @@ struct ggml_tensor * llama_model_loader::create_tensor(             }         } +        if (is_lazy) {+            return lazy_read::buft();+        }+         // select the buffer type for this tensor         const buft_list_t * buft_list;         switch (info.layer) {@@ -1287,16 +1332,9 @@ struct ggml_tensor * llama_model_loader::create_tensor(         return NULL;     } -    if ((flags & TENSOR_READ_LAZY) && use_mmap && lazy_mode != LLAMA_LAZY_MODE_OFF) {-        // in auto mode, small tensors are cheap enough to keep resident-        constexpr size_t auto_lazy_min_size = 4ull * 1024 * 1024 * 1024;-        if (lazy_mode == LLAMA_LAZY_MODE_ON || ggml_nbytes(cur) > auto_lazy_min_size) {-            const auto & w = require_weight(tn.str().c_str());-            lazy_tensor_ranges[w.idx].emplace_back(w.offs, w.offs + ggml_nbytes(cur));--            LLAMA_LOG_INFO("%s: tensor %s (size = %zu MiB) lazy read enabled\n",-                    __func__, tn.str().c_str(), ggml_nbytes(cur)/1024/1024);-        }+    if (flags & TENSOR_READ_LAZY) {+        // the decision must not depend on the load mode, or the memory-fit pass (no_alloc, no mmap)+        is_lazy = lazy.add(tn.str(), cur, no_alloc ? nullptr : &require_weight(tn.str().c_str()));     }      ggml_tensor t_meta = *cur;@@ -1363,7 +1401,8 @@ void llama_model_loader::done_getting_tensors(bool partial) const { }  void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps) {-    if (use_mmap) {+    // note: read_lazy also requires mmap; this condition make sure it's usable even when --load-mode is not set to mmap+    if (use_mmap || lazy.any()) {         mappings.reserve(files.size());         mmaps_used.reserve(files.size());         for (uint32_t idx = 0; idx < files.size(); idx++) {@@ -1380,11 +1419,10 @@ void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps                 }             } -            const auto it_lazy = lazy_tensor_ranges.find(idx);-            static const llama_mmap::ranges no_lazy_ranges;+            const size_t prefetch_size = prefetch && use_mmap ? -1 : 0; -            std::unique_ptr<llama_mmap> mapping = std::make_unique<llama_mmap>(file.get(), prefetch ? -1 : 0, is_numa,-                    it_lazy != lazy_tensor_ranges.end() ? it_lazy->second : no_lazy_ranges);+            std::unique_ptr<llama_mmap> mapping = std::make_unique<llama_mmap>(file.get(), prefetch_size, is_numa,+                    lazy.for_file(idx));             mmaps_used.emplace_back(mapping->size(), 0);             if (mlock_mmaps) {                 std::unique_ptr<llama_mlock> mlock_mmap(new llama_mlock());@@ -1575,7 +1613,9 @@ bool llama_model_loader::load_all_data(          size_t n_size = ggml_nbytes(cur); -        if (use_mmap) {+        const bool from_mapping = use_mmap || lazy.has(cur);++        if (from_mapping) {             const auto & mapping = mappings.at(weight->idx);             ggml_backend_buffer_t buf_mmap = nullptr;             if (bufs.count(weight->idx)) {@@ -1592,7 +1632,9 @@ bool llama_model_loader::load_all_data(             GGML_ASSERT(buf_mmap || cur->data); // either we have a buffer to allocate the tensor in, or it is already allocated             if (buf_mmap && cur->data == nullptr) {                 ggml_backend_tensor_alloc(buf_mmap, cur, data);-                if (lmlocks) {++                // locking a lazy tensor would fault all of it in, which is what lazy avoids+                if (lmlocks && !lazy.has(cur)) {                     const auto & lmlock = lmlocks->at(weight->idx);                     lmlock->grow_to(weight->offs + n_size);                 }
src/llama-model-loader.h49 + / 6
@@ -12,6 +12,7 @@ #include <cstddef> #include <cstring> #include <map>+#include <set> #include <stdexcept> #include <unordered_map> @@ -83,18 +84,45 @@ struct llama_model_loader {     bool no_alloc;     bool load_mtp; -    // set by the caller before the create_tensor() calls-    enum llama_lazy_mode lazy_mode = LLAMA_LAZY_MODE_OFF;+    // handle TENSOR_READ_LAZY+    // use case: keep PLE / engrams embd tensors on disk, read them on demand+    struct lazy_read {+        // set by the caller before the create_tensor() calls+        enum llama_lazy_mode mode = LLAMA_LAZY_MODE_OFF;++        // decide whether this tensor is read lazily+        // pass w to also record it, or nullptr to only ask+        bool add(const std::string & name, const ggml_tensor * t, const llama_tensor_weight * w);++        bool any() const {+            return !ranges.empty();+        }++        bool has(const ggml_tensor * t) const {+            return tensors.count(ggml_get_name(t)) > 0;+        }++        const llama_mmap::ranges & for_file(uint32_t idx) const {+            static const llama_mmap::ranges none;++            const auto it = ranges.find(idx);+            return it == ranges.end() ? none : it->second;+        }++        // lazy tensors are gathered on the host, so no offload setting applies to them+        static ggml_backend_buffer_type_t buft();++    private:+        std::map<uint32_t, llama_mmap::ranges> ranges;+        std::set<std::string>                  tensors;+    } lazy;      llama_files files;     llama_ftype ftype;     llama_fver  fver;      llama_mmaps mappings; -    // byte ranges of TENSOR_READ_LAZY tensors, per file index-    std::map<uint32_t, llama_mmap::ranges> lazy_tensor_ranges;-     std::map<std::string, llama_tensor_weight, weight_name_comparer> weights_map;     std::unordered_map<std::string, llama_model_kv_override> kv_overrides;     const llama_model_tensor_buft_override * tensor_buft_overrides;@@ -119,7 +147,22 @@ struct llama_model_loader {         }     }; -    std::map<ggml_backend_buffer_type_t, ggml_context_ptr, ggml_backend_buft_comparator> ctx_map;+    // lazy tensors need dedicated context+    struct ctx_key {+        ggml_backend_buffer_type_t buft;+        bool lazy;+    };++    struct ctx_key_comparator {+        bool operator()(const ctx_key & lhs, const ctx_key & rhs) const {+            if (lhs.lazy != rhs.lazy) {+                return lhs.lazy < rhs.lazy;+            }+            return strcmp(ggml_backend_buft_name(lhs.buft), ggml_backend_buft_name(rhs.buft)) < 0;+        }+    };++    std::map<ctx_key, ggml_context_ptr, ctx_key_comparator> ctx_map;      // track tensors that had to be moved for debugging:     size_t n_tensors_moved = 0;
src/llama-model.cpp7 + / 2
@@ -1689,7 +1689,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {     const size_t n_max_backend_buffer = ml.ctx_map.size() * ml.files.size();     pimpl->ctxs_bufs.reserve(n_max_backend_buffer); -    for (auto & [buft, ctx_ptr] : ml.ctx_map) {+    for (auto & [ctx_key, ctx_ptr] : ml.ctx_map) {+        ggml_backend_buffer_type_t buft = ctx_key.buft;         ggml_context * ctx = ctx_ptr.get();          // skip contexts without tensors@@ -1715,7 +1716,11 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {         bool is_default_buft = buft == ggml_backend_dev_buffer_type(dev);          std::vector<ggml_backend_buffer_ptr> bufs;-        if (ml.use_mmap && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft) {++        // a lazy context is mapped whatever the load mode, but the memory-fit pass maps nothing+        const bool is_lazy_mapped = ctx_key.lazy && !ml.no_alloc;++        if ((ml.use_mmap || is_lazy_mapped) && use_mmap_buffer && buffer_from_host_ptr_supported && is_default_buft) {             GGML_ASSERT(!ml.no_alloc);             for (uint32_t idx = 0; idx < ml.files.size(); idx++) {                 // only the mmap region containing the tensors in the model is mapped to the backend buffer
src/llama.cpp1 + / 1
@@ -318,7 +318,7 @@ static std::pair<int, llama_model *> llama_model_load(struct gguf_context * meta         llama_model_loader ml(metadata, set_tensor_data, set_tensor_data_ud, fname, splits, file, params.load_mode,             params.check_tensors, params.no_alloc, params.load_mtp, params.kv_overrides, params.tensor_buft_overrides); -        ml.lazy_mode = params.lazy_mode;+        ml.lazy.mode = params.lazy_mode;          ml.print_info();         std::unique_ptr<llama_model> model_ptr(llama_model_create(ml, params));