From: Meshnet Subject: [PATCH] llama: add dense owned-range tensor loading Concern: range loading. Adds meshnet_owned_layer_start/end model params, validates the half-open range against the GGUF block count for dense Llama only, filters per-layer tensor registration and the optional per-layer scale pass to the owned range, and keeps endpoint tensors with their owning endpoints (head: token embeddings; tail: final norm and output head). Stock zero/zero params preserve whole-model loading. --- diff --git a/include/llama.h b/include/llama.h index a311ac202..229946ede 100644 --- a/include/llama.h +++ b/include/llama.h @@ -319,6 +319,12 @@ extern "C" { // override key-value pairs of the model meta data const struct llama_model_kv_override * kv_overrides; + // Project-owned dense-Llama owned range [start, end). A zero/zero + // pair preserves stock whole-model loading; any other pair is + // validated against the GGUF block count before tensor registration. + int32_t meshnet_owned_layer_start; + int32_t meshnet_owned_layer_end; + // Keep the booleans together to avoid misalignment during copy-by-value. bool vocab_only; // only load the vocabulary, no weights bool use_mmap; // use mmap if possible diff --git a/src/llama-model.cpp b/src/llama-model.cpp index d87481381..05b8b9c91 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1236,6 +1236,19 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { const bool use_mmap_buffer = true; + const bool meshnet_range_requested = params.meshnet_owned_layer_start != 0 || params.meshnet_owned_layer_end != 0; + const int meshnet_start = params.meshnet_owned_layer_start; + const int meshnet_end = params.meshnet_owned_layer_end; + if (meshnet_range_requested) { + if (arch != LLM_ARCH_LLAMA) { + throw std::runtime_error("Meshnet owned range currently supports dense Llama only"); + } + if (meshnet_start < 0 || meshnet_end <= meshnet_start || meshnet_end > static_cast(hparams.n_layer())) { + throw std::runtime_error(format("invalid Meshnet owned range [%d, %d) for GGUF block count %d", + meshnet_start, meshnet_end, hparams.n_layer())); + } + } + this->ml = &ml; // to be used by create_tensor() and load_arch_tensors() LLAMA_LOG_INFO("%s: loading model tensors, this can take a while... (mmap = %s, direct_io = %s)\n", @@ -1336,7 +1349,9 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { // generic pass: load optional per-tensor/per-expert ".scale" tensors (e.g. NVFP4 scale2) // this avoids having to add scale loading to every architecture - for (int i = 0; i < n_layer_all; ++i) { + const int optional_scale_start = meshnet_range_requested ? meshnet_start : 0; + const int optional_scale_end = meshnet_range_requested ? meshnet_end : n_layer_all; + for (int i = optional_scale_start; i < optional_scale_end; ++i) { auto & layer = layers[i]; // attention weight scales (per-tensor, shape {1}) @@ -1487,7 +1502,7 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { } } } - ml.done_getting_tensors(); + ml.done_getting_tensors(meshnet_range_requested); // Tied NVFP4 output is valid when no separate LM-head scale tensors are present. // If sidecar scales exist, the output weight must be an actual output tensor. @@ -2308,6 +2323,8 @@ llama_model_params llama_model_default_params() { /*.progress_callback =*/ nullptr, /*.progress_callback_user_data =*/ nullptr, /*.kv_overrides =*/ nullptr, + /*.meshnet_owned_layer_start =*/ 0, + /*.meshnet_owned_layer_end =*/ 0, /*.vocab_only =*/ false, /*.use_mmap =*/ true, /*.use_direct_io =*/ false, diff --git a/src/models/llama.cpp b/src/models/llama.cpp index 4bfebc884..c3092763b 100644 --- a/src/models/llama.cpp +++ b/src/models/llama.cpp @@ -34,18 +34,29 @@ void llama_model_llama::load_arch_hparams(llama_model_loader & ml) { void llama_model_llama::load_arch_tensors(llama_model_loader &) { LLAMA_LOAD_LOCALS; - tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); + const bool meshnet_range_requested = params.meshnet_owned_layer_start != 0 || params.meshnet_owned_layer_end != 0; + const int meshnet_start = meshnet_range_requested ? params.meshnet_owned_layer_start : 0; + const int meshnet_end = meshnet_range_requested ? params.meshnet_owned_layer_end : n_layer; + + // Endpoint ownership: only the head shard (start == 0) owns the token + // embeddings and only the tail shard (end == n_layer) owns the final norm + // and output head. Middle ranges register per-layer tensors only. + if (!meshnet_range_requested || meshnet_start == 0) { + tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); + } - // output - output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0); - output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED); + if (!meshnet_range_requested || meshnet_end == n_layer) { + // output + output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0); + output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED); - // if output is NULL, init from the input tok embed - if (output == NULL) { - output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED); + // if output is NULL, init from the input tok embed + if (output == NULL) { + output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED); + } } - for (int i = 0; i < n_layer; ++i) { + for (int i = meshnet_start; i < meshnet_end; ++i) { auto & layer = layers[i]; layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);