llama.cpp version update (#16548)

This commit is contained in:
Daniel Hiltgen
2026-06-24 14:03:12 -07:00
committed by GitHub
parent 0a408b2225
commit e11eeb3ba0
2 changed files with 21 additions and 21 deletions

View File

@@ -1 +1 @@
b9672
b9781

View File

@@ -1,5 +1,5 @@
diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp
index 4e65a45..a6e4fe2 100644
index 474cabdfc..2bfe43a28 100644
--- a/src/llama-model-loader.cpp
+++ b/src/llama-model-loader.cpp
@@ -4,6 +4,7 @@
@@ -10,7 +10,7 @@ index 4e65a45..a6e4fe2 100644
#include <algorithm>
#include <array>
@@ -549,6 +550,7 @@ llama_model_loader::llama_model_loader(
@@ -551,6 +552,7 @@ llama_model_loader::llama_model_loader(
}
get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE), arch_name, false);
@@ -18,7 +18,7 @@ index 4e65a45..a6e4fe2 100644
llm_kv = LLM_KV(llm_arch_from_string(arch_name));
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
@@ -573,6 +575,9 @@ llama_model_loader::llama_model_loader(
@@ -575,6 +577,9 @@ llama_model_loader::llama_model_loader(
// so we build a unified tensors index for weights.
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
std::string tensor_name = std::string(cur->name);
@@ -28,7 +28,7 @@ index 4e65a45..a6e4fe2 100644
// make sure there is no duplicated tensor names
if (weights_map.find(tensor_name) != weights_map.end()) {
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
@@ -683,6 +688,9 @@ llama_model_loader::llama_model_loader(
@@ -685,6 +690,9 @@ llama_model_loader::llama_model_loader(
// Save tensors data offset info of the main file.
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
std::string tensor_name = std::string(cur->name);
@@ -38,7 +38,7 @@ index 4e65a45..a6e4fe2 100644
// make sure there is no duplicated tensor names
if (weights_map.find(tensor_name) != weights_map.end()) {
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
@@ -1375,6 +1383,7 @@ void llama_model_loader::get_mapping_range(size_t * first, size_t * last, void *
@@ -1384,6 +1392,7 @@ void llama_model_loader::get_mapping_range(size_t * first, size_t * last, void *
void llama_model_loader::load_data_for(struct ggml_tensor * cur) const {
const auto & w = require_weight(ggml_get_name(cur));
@@ -46,7 +46,7 @@ index 4e65a45..a6e4fe2 100644
if (use_mmap) {
const auto & mapping = mappings.at(w.idx);
@@ -1525,6 +1534,7 @@ bool llama_model_loader::load_all_data(
@@ -1534,6 +1543,7 @@ bool llama_model_loader::load_all_data(
}
size_t n_size = ggml_nbytes(cur);
@@ -55,7 +55,7 @@ index 4e65a45..a6e4fe2 100644
if (use_mmap) {
const auto & mapping = mappings.at(weight->idx);
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index 2e0cfa6..a0f2955 100644
index 7bd486030..6c3b23da0 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -10,6 +10,8 @@
@@ -67,7 +67,7 @@ index 2e0cfa6..a0f2955 100644
#include <algorithm>
#include <cassert>
#include <cmath>
@@ -1009,6 +1011,11 @@ struct clip_model_loader {
@@ -1070,6 +1072,11 @@ struct clip_model_loader {
ctx_meta.reset(meta);
@@ -79,15 +79,18 @@ index 2e0cfa6..a0f2955 100644
const int n_tensors = gguf_get_n_tensors(ctx_gguf.get());
// print gguf info
@@ -2611,6 +2618,7 @@ struct clip_model_loader {
auto it_off = tensor_offset.find(t->name);
GGML_ASSERT(it_off != tensor_offset.end() && "no offset for tensor");
const size_t offset = it_off->second;
+ if (llama_ollama_compat::maybe_load_tensor(cur, fname.c_str(), offset, buft)) continue;
fin.seekg(offset, std::ios::beg);
if (!fin) {
throw std::runtime_error(string_format("%s: failed to seek for tensor %s\n", __func__, t->name));
@@ -4312,6 +4320,15 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
@@ -2822,6 +2829,7 @@ struct clip_model_loader {
auto it_off = tensor_offset.find(t->name);
GGML_ASSERT(it_off != tensor_offset.end() && "no offset for tensor");
const size_t offset = it_off->second;
+ if (llama_ollama_compat::maybe_load_tensor(cur, fname.c_str(), offset, buft)) continue;
fin.seekg(offset, std::ios::beg);
if (!fin) {
throw std::runtime_error(string_format("%s: failed to seek for tensor %s\n", __func__, t->name));
@@ -4489,6 +4497,15 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
}
int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
+ const auto projector_type = PROJECTOR_TYPE_NAMES.find(ctx->model.proj_type);
+ if (projector_type != PROJECTOR_TYPE_NAMES.end()) {
+ if (int n = llama_ollama_compat::maybe_clip_mmproj_embd(
@@ -100,6 +103,3 @@ index 2e0cfa6..a0f2955 100644
switch (ctx->model.proj_type) {
case PROJECTOR_TYPE_LDP:
return ctx->model.mm_model_block_1_block_2_1_b->ne[0];
case PROJECTOR_TYPE_LDPV2:
return ctx->model.mm_model_peg_0_b->ne[0];
case PROJECTOR_TYPE_MLP: