llama.cpp version update (#16548)
This commit is contained in:
@@ -1 +1 @@
|
|||||||
b9672
|
b9781
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp
|
diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp
|
||||||
index 4e65a45..a6e4fe2 100644
|
index 474cabdfc..2bfe43a28 100644
|
||||||
--- a/src/llama-model-loader.cpp
|
--- a/src/llama-model-loader.cpp
|
||||||
+++ b/src/llama-model-loader.cpp
|
+++ b/src/llama-model-loader.cpp
|
||||||
@@ -4,6 +4,7 @@
|
@@ -4,6 +4,7 @@
|
||||||
@@ -10,7 +10,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
|
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <array>
|
#include <array>
|
||||||
@@ -549,6 +550,7 @@ llama_model_loader::llama_model_loader(
|
@@ -551,6 +552,7 @@ llama_model_loader::llama_model_loader(
|
||||||
}
|
}
|
||||||
|
|
||||||
get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE), arch_name, false);
|
get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE), arch_name, false);
|
||||||
@@ -18,7 +18,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
llm_kv = LLM_KV(llm_arch_from_string(arch_name));
|
llm_kv = LLM_KV(llm_arch_from_string(arch_name));
|
||||||
|
|
||||||
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
|
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
|
||||||
@@ -573,6 +575,9 @@ llama_model_loader::llama_model_loader(
|
@@ -575,6 +577,9 @@ llama_model_loader::llama_model_loader(
|
||||||
// so we build a unified tensors index for weights.
|
// so we build a unified tensors index for weights.
|
||||||
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
||||||
std::string tensor_name = std::string(cur->name);
|
std::string tensor_name = std::string(cur->name);
|
||||||
@@ -28,7 +28,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
// make sure there is no duplicated tensor names
|
// make sure there is no duplicated tensor names
|
||||||
if (weights_map.find(tensor_name) != weights_map.end()) {
|
if (weights_map.find(tensor_name) != weights_map.end()) {
|
||||||
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
|
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
|
||||||
@@ -683,6 +688,9 @@ llama_model_loader::llama_model_loader(
|
@@ -685,6 +690,9 @@ llama_model_loader::llama_model_loader(
|
||||||
// Save tensors data offset info of the main file.
|
// Save tensors data offset info of the main file.
|
||||||
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
||||||
std::string tensor_name = std::string(cur->name);
|
std::string tensor_name = std::string(cur->name);
|
||||||
@@ -38,7 +38,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
// make sure there is no duplicated tensor names
|
// make sure there is no duplicated tensor names
|
||||||
if (weights_map.find(tensor_name) != weights_map.end()) {
|
if (weights_map.find(tensor_name) != weights_map.end()) {
|
||||||
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
|
throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
|
||||||
@@ -1375,6 +1383,7 @@ void llama_model_loader::get_mapping_range(size_t * first, size_t * last, void *
|
@@ -1384,6 +1392,7 @@ void llama_model_loader::get_mapping_range(size_t * first, size_t * last, void *
|
||||||
|
|
||||||
void llama_model_loader::load_data_for(struct ggml_tensor * cur) const {
|
void llama_model_loader::load_data_for(struct ggml_tensor * cur) const {
|
||||||
const auto & w = require_weight(ggml_get_name(cur));
|
const auto & w = require_weight(ggml_get_name(cur));
|
||||||
@@ -46,7 +46,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
|
|
||||||
if (use_mmap) {
|
if (use_mmap) {
|
||||||
const auto & mapping = mappings.at(w.idx);
|
const auto & mapping = mappings.at(w.idx);
|
||||||
@@ -1525,6 +1534,7 @@ bool llama_model_loader::load_all_data(
|
@@ -1534,6 +1543,7 @@ bool llama_model_loader::load_all_data(
|
||||||
}
|
}
|
||||||
|
|
||||||
size_t n_size = ggml_nbytes(cur);
|
size_t n_size = ggml_nbytes(cur);
|
||||||
@@ -55,7 +55,7 @@ index 4e65a45..a6e4fe2 100644
|
|||||||
if (use_mmap) {
|
if (use_mmap) {
|
||||||
const auto & mapping = mappings.at(weight->idx);
|
const auto & mapping = mappings.at(weight->idx);
|
||||||
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
|
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
|
||||||
index 2e0cfa6..a0f2955 100644
|
index 7bd486030..6c3b23da0 100644
|
||||||
--- a/tools/mtmd/clip.cpp
|
--- a/tools/mtmd/clip.cpp
|
||||||
+++ b/tools/mtmd/clip.cpp
|
+++ b/tools/mtmd/clip.cpp
|
||||||
@@ -10,6 +10,8 @@
|
@@ -10,6 +10,8 @@
|
||||||
@@ -67,7 +67,7 @@ index 2e0cfa6..a0f2955 100644
|
|||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <cmath>
|
#include <cmath>
|
||||||
@@ -1009,6 +1011,11 @@ struct clip_model_loader {
|
@@ -1070,6 +1072,11 @@ struct clip_model_loader {
|
||||||
|
|
||||||
ctx_meta.reset(meta);
|
ctx_meta.reset(meta);
|
||||||
|
|
||||||
@@ -79,15 +79,18 @@ index 2e0cfa6..a0f2955 100644
|
|||||||
const int n_tensors = gguf_get_n_tensors(ctx_gguf.get());
|
const int n_tensors = gguf_get_n_tensors(ctx_gguf.get());
|
||||||
|
|
||||||
// print gguf info
|
// print gguf info
|
||||||
@@ -2611,6 +2618,7 @@ struct clip_model_loader {
|
@@ -2822,6 +2829,7 @@ struct clip_model_loader {
|
||||||
auto it_off = tensor_offset.find(t->name);
|
auto it_off = tensor_offset.find(t->name);
|
||||||
GGML_ASSERT(it_off != tensor_offset.end() && "no offset for tensor");
|
GGML_ASSERT(it_off != tensor_offset.end() && "no offset for tensor");
|
||||||
const size_t offset = it_off->second;
|
const size_t offset = it_off->second;
|
||||||
+ if (llama_ollama_compat::maybe_load_tensor(cur, fname.c_str(), offset, buft)) continue;
|
+ if (llama_ollama_compat::maybe_load_tensor(cur, fname.c_str(), offset, buft)) continue;
|
||||||
fin.seekg(offset, std::ios::beg);
|
fin.seekg(offset, std::ios::beg);
|
||||||
if (!fin) {
|
if (!fin) {
|
||||||
throw std::runtime_error(string_format("%s: failed to seek for tensor %s\n", __func__, t->name));
|
throw std::runtime_error(string_format("%s: failed to seek for tensor %s\n", __func__, t->name));
|
||||||
@@ -4312,6 +4320,15 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
|
@@ -4489,6 +4497,15 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
|
||||||
|
}
|
||||||
|
|
||||||
|
int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
|
||||||
+ const auto projector_type = PROJECTOR_TYPE_NAMES.find(ctx->model.proj_type);
|
+ const auto projector_type = PROJECTOR_TYPE_NAMES.find(ctx->model.proj_type);
|
||||||
+ if (projector_type != PROJECTOR_TYPE_NAMES.end()) {
|
+ if (projector_type != PROJECTOR_TYPE_NAMES.end()) {
|
||||||
+ if (int n = llama_ollama_compat::maybe_clip_mmproj_embd(
|
+ if (int n = llama_ollama_compat::maybe_clip_mmproj_embd(
|
||||||
@@ -100,6 +103,3 @@ index 2e0cfa6..a0f2955 100644
|
|||||||
switch (ctx->model.proj_type) {
|
switch (ctx->model.proj_type) {
|
||||||
case PROJECTOR_TYPE_LDP:
|
case PROJECTOR_TYPE_LDP:
|
||||||
return ctx->model.mm_model_block_1_block_2_1_b->ne[0];
|
return ctx->model.mm_model_block_1_block_2_1_b->ne[0];
|
||||||
case PROJECTOR_TYPE_LDPV2:
|
|
||||||
return ctx->model.mm_model_peg_0_b->ne[0];
|
|
||||||
case PROJECTOR_TYPE_MLP:
|
|
||||||
|
|||||||
Reference in New Issue
Block a user