Take into account layer sizes for setting GPU layers (cont) (#1476)

* Also take into account KV cache

* Take into account attn_wkv_b and mla = 3 compute buffers
This commit is contained in:
Kawrakow
2026-03-23 17:46:53 +01:00
committed by GitHub
parent 4eb08208f2
commit b7a2bde4cc
5 changed files with 207 additions and 56 deletions
+71 -48
View File
@@ -3235,46 +3235,6 @@ void llama_lora_adapters_apply(struct llama_context * ctx, std::vector<llama_lor
}
}
struct llama_model_params common_model_params_to_llama(const gpt_params & params) {
auto mparams = llama_model_default_params();
mparams.devices = params.devices.c_str();
if (params.n_gpu_layers != -1) {
mparams.n_gpu_layers = params.n_gpu_layers;
}
mparams.mla = params.mla_attn;
mparams.dry_run = params.dry_run;
mparams.rpc_servers = params.rpc_servers.c_str();
mparams.main_gpu = params.main_gpu;
mparams.max_gpu = params.max_gpu;
mparams.ncmoe = params.ncmoe;
mparams.split_mode = params.split_mode;
mparams.tensor_split = params.tensor_split;
mparams.use_mmap = params.use_mmap;
mparams.use_mlock = params.use_mlock;
mparams.check_tensors = params.check_tensors;
mparams.repack_tensors = params.repack_tensors;
mparams.use_thp = params.use_thp;
mparams.validate_quants = params.validate_quants;
mparams.merge_qkv = params.merge_qkv;
mparams.merge_up_gate_exps = params.merge_up_gate_exps;
mparams.mtp = params.has_mtp;
if (params.kv_overrides.empty()) {
mparams.kv_overrides = NULL;
} else {
GGML_ASSERT(params.kv_overrides.back().key[0] == 0 && "KV overrides not terminated with empty key");
mparams.kv_overrides = params.kv_overrides.data();
}
if (params.tensor_buft_overrides.empty()) {
mparams.tensor_buft_overrides = NULL;
} else {
GGML_ASSERT(params.tensor_buft_overrides.back().pattern == nullptr && "Tensor buffer overrides not terminated with empty pattern");
mparams.tensor_buft_overrides = params.tensor_buft_overrides.data();
}
return mparams;
}
static ggml_type kv_cache_type_from_str(const std::string & s) {
if (s == "f32") {
return GGML_TYPE_F32;
@@ -3313,6 +3273,73 @@ static ggml_type kv_cache_type_from_str(const std::string & s) {
throw std::runtime_error("Invalid cache type: " + s);
}
static std::pair<int, int> get_batch_ubatch(const gpt_params & params) {
int n_batch = params.n_batch;
int n_ubatch = params.n_ubatch;
if (params.n_ctx > 0) {
n_batch = std::min(n_batch, params.n_ctx);
}
if (!params.mmproj.path.empty()) {
// temporary fix for qwen mtmd
n_batch = std::max(n_batch, n_ubatch);
n_ubatch = n_batch;
fprintf(stdout, "Adjust batch size for mtmd: u_batch = %d, batch = %d\n", n_ubatch, n_batch);
} else {
n_ubatch = std::min(n_batch, n_ubatch);
}
return {n_batch, n_ubatch};
}
struct llama_model_params common_model_params_to_llama(const gpt_params & params) {
auto mparams = llama_model_default_params();
mparams.devices = params.devices.c_str();
if (params.n_gpu_layers != -1) {
mparams.n_gpu_layers = params.n_gpu_layers;
}
mparams.mla = params.mla_attn;
mparams.dry_run = params.dry_run;
mparams.rpc_servers = params.rpc_servers.c_str();
mparams.main_gpu = params.main_gpu;
mparams.max_gpu = params.max_gpu;
mparams.ncmoe = params.ncmoe;
mparams.type_k = kv_cache_type_from_str(params.cache_type_k);
mparams.type_v = kv_cache_type_from_str(params.cache_type_v);
mparams.max_ctx_size = params.n_ctx;
mparams.n_seq_max = params.n_parallel;
mparams.n_ubatch = get_batch_ubatch(params).second;
mparams.amb = params.attn_max_batch;
mparams.split_mode = params.split_mode;
mparams.tensor_split = params.tensor_split;
mparams.use_mmap = params.use_mmap;
mparams.use_mlock = params.use_mlock;
mparams.check_tensors = params.check_tensors;
mparams.repack_tensors = params.repack_tensors;
mparams.use_thp = params.use_thp;
mparams.validate_quants = params.validate_quants;
mparams.merge_qkv = params.merge_qkv;
mparams.merge_up_gate_exps = params.merge_up_gate_exps;
mparams.mtp = params.has_mtp;
mparams.flash_attn = params.flash_attn;
if (params.kv_overrides.empty()) {
mparams.kv_overrides = NULL;
} else {
GGML_ASSERT(params.kv_overrides.back().key[0] == 0 && "KV overrides not terminated with empty key");
mparams.kv_overrides = params.kv_overrides.data();
}
if (params.tensor_buft_overrides.empty()) {
mparams.tensor_buft_overrides = NULL;
} else {
GGML_ASSERT(params.tensor_buft_overrides.back().pattern == nullptr && "Tensor buffer overrides not terminated with empty pattern");
mparams.tensor_buft_overrides = params.tensor_buft_overrides.data();
}
if (!mparams.flash_attn && ggml_is_quantized(mparams.type_v)) {
throw std::runtime_error("Quantized V cache cannot be used without flash attention");
}
return mparams;
}
static ggml_type ggml_type_from_str(const std::string & s) {
if (s == "f32") {
return GGML_TYPE_F32;
@@ -3331,15 +3358,8 @@ static ggml_type ggml_type_from_str(const std::string & s) {
struct llama_context_params common_context_params_to_llama(const gpt_params & params) {
auto cparams = llama_context_default_params();
int n_batch = params.n_batch;
int n_ubatch = params.n_ubatch;
// temporary fix for qwen mtmd
if (!params.mmproj.path.empty()) {
n_batch = std::max(params.n_batch, params.n_ubatch);
n_ubatch = params.n_batch;
fprintf(stdout, "Adjust batch size for mtmd: u_batch = %d, batch = %d\n", n_ubatch, n_batch);
}
auto [n_batch, n_ubatch] = get_batch_ubatch(params);
cparams.n_ctx = params.n_ctx;
cparams.n_seq_max = params.n_parallel;
@@ -3387,6 +3407,9 @@ struct llama_context_params common_context_params_to_llama(const gpt_params & pa
cparams.type_k = kv_cache_type_from_str(params.cache_type_k);
cparams.type_v = kv_cache_type_from_str(params.cache_type_v);
cparams.type_reduce = ggml_type_from_str(params.reduce_type);
if (!cparams.flash_attn && ggml_is_quantized(cparams.type_v)) {
throw std::runtime_error("Quantized V cache cannot be used without flash attention");
}
if (!params.offload_policy.empty()) cparams.offload_policy = (void *)&params.offload_policy;
if (!params.cuda_params.empty()) cparams.cuda_params = (void *)params.cuda_params.data();
+8
View File
@@ -372,6 +372,13 @@ extern "C" {
int32_t max_gpu;
int32_t ncmoe;
enum ggml_type type_k;
enum ggml_type type_v;
uint32_t max_ctx_size;
int32_t n_seq_max;
int32_t n_ubatch;
int32_t amb;
// proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()
const float * tensor_split;
@@ -403,6 +410,7 @@ extern "C" {
bool merge_up_gate_exps; // if true, merge ffn_up_exps and ffn_gate_exps tensors into a single, contiguous tensor
bool mtp; // if true, load MTP layers if present
bool dry_run; // skip loading tensors
bool flash_attn;
};
// NOTE: changing the default values of parameters marked as [EXPERIMENTAL] may cause crashes or incorrect results in certain configurations
+27
View File
@@ -1,4 +1,5 @@
#include "llama-model.h"
#include "llama-cparams.h"
#include <map>
@@ -1835,3 +1836,29 @@ bool llama_model_is_hybrid(const llama_model * model) {
bool llama_model_has_recurrent(const llama_model * model) {
return llm_arch_is_hybrid(model->arch) || llm_arch_is_recurrent(model->arch);
}
size_t llama_model::cache_size(int il, ggml_type type_k, ggml_type type_v, uint32_t kv_size, int mla_attn, int n_seq_max, bool flash_attn) const {
if (il < 0 || il >= hparams.n_layer) return 0;
if (hparams.recurrent_layer_arr[il]) {
auto state_sots = std::min<uint32_t>(std::max<uint32_t>(1, n_seq_max), kv_size);
return hparams.n_embd_v_s() * state_sots * sizeof(float);
}
bool is_mla_attn = arch == LLM_ARCH_DEEPSEEK2 || arch == LLM_ARCH_GLM_DSA || arch == LLM_ARCH_MISTRAL4;
if (is_mla_attn && mla_attn) {
auto n_embd_head_qk_rope = hparams.n_rot;
auto kv_lora_rank = hparams.n_lora_kv;
if (flash_attn) {
return ggml_row_size(type_k, kv_lora_rank + n_embd_head_qk_rope) * kv_size;
}
auto kv_type = mla_attn == 1 ? type_k : type_v;
auto size = ggml_row_size(kv_type, kv_lora_rank + n_embd_head_qk_rope) * kv_size;
if (mla_attn == 1) {
size += ggml_row_size(type_v, kv_lora_rank*kv_size);
}
return size;
}
auto n_head_kv = hparams.n_head_kv(il);
auto k_size = ggml_row_size(type_k, hparams.n_embd_head_k) * n_head_kv*kv_size;
auto v_size = ggml_row_size(type_v, hparams.n_embd_v_gqa(il)) * kv_size;
return k_size + v_size;
}
+4
View File
@@ -362,6 +362,8 @@ struct rpc_device {
uint32_t device;
};
struct llama_cparams;
struct llama_model {
e_model type = MODEL_UNKNOWN;
llm_arch arch = LLM_ARCH_UNKNOWN;
@@ -457,6 +459,8 @@ struct llama_model {
return tensor_overrides;
}
size_t cache_size(int il, ggml_type type_k, ggml_type type_v, uint32_t kv_size, int mla_attn, int n_seq_max, bool flash_attn) const;
void set_tensor_overrides(const llama_model_params& params);
int device_count() const;
+97 -8
View File
@@ -1997,9 +1997,22 @@ static bool is_model_split_supported(const llama_model & model) {
return it != k_supported.end();
}
static std::vector<double> get_layer_sizes(const llama_model_loader & ml, const llama_model & model) {
static std::pair<std::vector<double>, double> get_layer_sizes(const llama_model_loader & ml, const llama_model & model,
ggml_type cache_type_k, ggml_type cache_type_v, uint32_t max_ctx_size, int mla_attn, int n_seq_max, int n_ubatch, int amb, bool flash_attn) {
int n_layer = model.hparams.n_layer;
std::vector<double> result(n_layer+1, 0);
std::vector<double> compute(n_layer+1, 0);
struct mla_tensors {
ggml_tensor * wk_b = nullptr;
ggml_tensor * wv_b = nullptr;
ggml_tensor * wkv_b = nullptr;
};
std::vector<mla_tensors> mla_tensors;
bool has_mla = model.arch == LLM_ARCH_DEEPSEEK2 || model.arch == LLM_ARCH_GLM_DSA || model.arch == LLM_ARCH_MISTRAL4;
if (has_mla) {
mla_tensors.resize(n_layer);
}
size_t ow_size = 0;
size_t embd_size = 0;
for (int i = 0; i < ml.n_tensors; ++i) {
@@ -2036,13 +2049,70 @@ static std::vector<double> get_layer_sizes(const llama_model_loader & ml, const
continue;
}
result[il] += size;
if (has_mla) {
if (name.find("attn_k_b.weight") != std::string::npos) {
mla_tensors[il].wk_b = t;
}
else if (name.find("attn_v_b.weight") != std::string::npos) {
mla_tensors[il].wv_b = t;
}
else if (name.find("attn_kv_b.weight") != std::string::npos) {
mla_tensors[il].wkv_b = t;
}
}
}
if (has_mla) {
for (int il = 0; il < n_layer; ++il) {
auto & mla = mla_tensors[il];
if (mla.wk_b && mla.wv_b && !mla.wkv_b) {
auto type = ggml_is_quantized(mla.wk_b->type) ? GGML_TYPE_Q8_0 : mla.wk_b->type;
auto wkv_b_size = ggml_row_size(type, mla.wv_b->ne[0]) * (mla.wv_b->ne[1] + mla.wk_b->ne[1]) * mla.wv_b->ne[2] * 2;
result[il] += wkv_b_size;
}
else if (mla.wkv_b) {
if (!mla.wk_b) result[il] += ggml_nbytes(mla.wkv_b)/2;
if (!mla.wv_b) result[il] += ggml_nbytes(mla.wkv_b)/2;
}
if (mla_attn == 3 && mla.wv_b) {
auto kv_f32_size = (2 * mla.wv_b->ne[1] * mla.wv_b->ne[2] * max_ctx_size * sizeof(float))/(1024.*1024.);
//printf("wv_b: %ld x %ld x %ld -> %g\n", mla.wv_b->ne[0], mla.wv_b->ne[1], mla.wv_b->ne[2], kv_f32_size);
int n_head = mla.wv_b->ne[2];
int n_max_head = n_head;
if (amb > 0 && kv_f32_size > amb) {
n_max_head = 1;
for (int niter = 2; niter < n_head; ++niter) {
if (n_head % niter == 0 && kv_f32_size/niter <= amb) {
n_max_head = n_head/niter;
break;
}
}
}
kv_f32_size = 2. * mla.wv_b->ne[1] * n_max_head * max_ctx_size * sizeof(float);
compute[il] = std::max(compute[il], kv_f32_size);
}
}
}
if (!ow_size) ow_size = embd_size;
result[n_layer] = ow_size;
LLAMA_LOG_INFO("------------------- Layer sizes:\n");
for (int il = 0; il < n_layer; ++il) LLAMA_LOG_INFO("Layer %2d: %g MiB\n", il, result[il]/1024./1024.);
LLAMA_LOG_INFO("Layer %2d: %g MiB (output layer)\n", n_layer, result[n_layer]/1024./1024.);
return result;
double tot_model = 0, tot_cache = 0, max_compute = 0;
for (int il = 0; il < n_layer; ++il) {
auto kv_size = model.cache_size(il, cache_type_k, cache_type_v, max_ctx_size, mla_attn, n_seq_max, flash_attn);
LLAMA_LOG_INFO("Layer %2d: %9.2f, %9.2f, %9.2f %9.2f MiB\n", il, result[il]/1024./1024., kv_size/1024./1024., (result[il] + kv_size)/1024./1024., compute[il]/1024./1024.);
max_compute = std::max(max_compute, compute[il]);
tot_model += result[il];
tot_cache += kv_size;
result[il] += kv_size;
}
size_t output_size = model.hparams.n_vocab * n_ubatch * sizeof(float);
if (output_size < max_compute) output_size = max_compute;
output_size -= max_compute;
LLAMA_LOG_INFO("Layer %2d: %9.2f, %9.2f, %9.2f MiB (output layer)\n", n_layer, result[n_layer]/1024./1024., output_size/1024./1024., (result[n_layer] + output_size)/1024./1024.);
result[n_layer] += output_size;
tot_cache += output_size;
LLAMA_LOG_INFO("--------------------------------------------------------------------------\n");
LLAMA_LOG_INFO("Total : %9.2f, %9.2f, %9.2f MiB\n", tot_model/1024./1024., tot_cache/1024./1024., (tot_model + tot_cache)/1024./1024.);
return std::make_pair(std::move(result), max_compute);
}
// Returns false if cancelled by progress_callback
@@ -2055,6 +2125,13 @@ static bool llm_load_tensors(
int main_gpu,
int max_gpu,
const float * tensor_split,
ggml_type cache_type_k,
ggml_type cache_type_v,
uint32_t max_ctx_size,
int n_seq_max,
int n_ubatch,
int amb,
bool flash_attn,
bool use_mlock,
bool validate_quants,
bool mtp,
@@ -2088,6 +2165,9 @@ static bool llm_load_tensors(
} else {
LLAMA_LOG_INFO("======================================= HAVE_FANCY_SIMD is NOT defined\n");
}
if (max_ctx_size == 0) {
max_ctx_size = model.hparams.n_ctx_train;
}
model.split_mode = split_mode;
model.main_gpu = main_gpu;
@@ -2139,14 +2219,15 @@ static bool llm_load_tensors(
model.default_layer_device = std::vector<int32_t>(hparams.n_layer+1, device_count-1);
int act_gpu_layers = std::min(n_gpu_layers, (int)n_layer + 1);
if (device_count > 1) {
auto layer_sizes = get_layer_sizes(ml, model);
auto [layer_sizes, max_compute] = get_layer_sizes(ml, model, cache_type_k, cache_type_v, max_ctx_size, mla_attn, n_seq_max, n_ubatch, amb, flash_attn);
int n_last = n_layer;
if (n_gpu_layers > n_layer) ++n_last;
double sum = 0;
double sum = max_compute * device_count;
for (int i = i_gpu_start; i < n_last; ++i) sum += layer_sizes[i];
int last = i_gpu_start;
float loaded_sum = 0;
for (int id = 0; id < int(model.splits.size()); ++id) {
for (int id = 0; id < device_count; ++id) {
loaded_sum += max_compute;
float split_size = model.splits[id]*sum;
int il = last;
for (; il < n_last; ++il) {
@@ -2513,7 +2594,8 @@ static int llama_model_load(const std::string & fname, llama_model & model, llam
#endif
if (!llm_load_tensors(
ml, model, params.n_gpu_layers, params.mla, params.split_mode, params.main_gpu, params.max_gpu, params.tensor_split,
ml, model, params.n_gpu_layers, params.mla, params.split_mode, params.main_gpu, params.max_gpu, params.tensor_split,
params.type_k, params.type_v, params.max_ctx_size, params.n_seq_max, params.n_ubatch, params.amb, params.flash_attn,
params.use_mlock, params.validate_quants, params.mtp, params.dry_run,
params.progress_callback, params.progress_callback_user_data
)) {
@@ -4477,6 +4559,12 @@ struct llama_model_params llama_model_default_params() {
/*.main_gpu =*/ 0,
/*.max_gpu =*/ 0,
/*.ncmoe =*/ 0,
/*.type_k =*/ GGML_TYPE_F16,
/*.type_v =*/ GGML_TYPE_F16,
/*.max_ctx_size =*/ 0,
/*.n_seq_max =*/ 1,
/*.n_ubatch =*/ 512,
/*.amb =*/ 0,
/*.tensor_split =*/ nullptr,
/*.rpc_servers =*/ nullptr,
/*.progress_callback =*/ nullptr,
@@ -4494,6 +4582,7 @@ struct llama_model_params llama_model_default_params() {
/*.merge_up_gate_exps =*/ false,
/*.mtp =*/ false,
/*.dry_run =*/ false,
/*.flash_attn =*/ true,
};
#ifdef GGML_USE_METAL