From: Meshnet Subject: [PATCH] llama: guard dense graph behind boundary endpoint ownership Concern: boundary I/O. Extends the range report with endpoint ownership flags derived from the registered tensor map and fails the dense-Llama graph closed for any partial owned range until typed head/tail endpoint adapters carry the architecture boundary I/O. --- diff --git a/include/llama.h b/include/llama.h index 6fd7ad509..8a7521349 100644 --- a/include/llama.h +++ b/include/llama.h @@ -300,6 +300,8 @@ extern "C" { int32_t end_layer; uint64_t mapped_bytes; uint64_t resident_bytes; + bool has_token_embeddings; + bool has_output_head; }; struct llama_model_params { diff --git a/src/llama-model.cpp b/src/llama-model.cpp index efb290c1f..2ea8598ad 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1649,11 +1649,17 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { if (ml.use_mmap) { meshnet_mapped_bytes = meshnet_resident_bytes; } + const auto registered_name = [this](const char * name) { + return std::any_of(tensors_by_name.begin(), tensors_by_name.end(), + [name](const auto & entry) { return entry.first == name; }); + }; pimpl->meshnet_range_report = { meshnet_start, meshnet_end, meshnet_mapped_bytes, meshnet_resident_bytes, + registered_name("token_embd.weight"), + output_norm != nullptr && output != nullptr, }; pimpl->has_meshnet_range_report = true; } diff --git a/src/models/llama.cpp b/src/models/llama.cpp index c3092763b..3b6854d0c 100644 --- a/src/models/llama.cpp +++ b/src/models/llama.cpp @@ -108,6 +108,25 @@ std::unique_ptr llama_model_llama::build_arch_graph(const llm template llama_model_llama::graph::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) { + llama_meshnet_range_report meshnet_report = {}; + if (model.meshnet_range_report(&meshnet_report)) { + // A partial owned range cannot execute the stock head/tail graph: the + // architecture boundary I/O must arrive through a typed endpoint + // adapter instead of local embeddings or the local output head. + if (meshnet_report.start_layer != 0) { + throw std::runtime_error("Meshnet dense-Llama graph requires a head endpoint adapter"); + } + if (meshnet_report.end_layer != n_layer) { + throw std::runtime_error("Meshnet dense-Llama graph requires a tail endpoint adapter"); + } + if (!meshnet_report.has_token_embeddings) { + throw std::runtime_error("Meshnet dense-Llama head range is missing token embeddings"); + } + if (!meshnet_report.has_output_head) { + throw std::runtime_error("Meshnet dense-Llama tail range is missing final norm or output head"); + } + } + const int64_t n_embd_head = hparams.n_embd_head_v(); GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());