Merge branch 'upstream' into concedo_experimental

# Conflicts: # .github/workflows/build.yml # Makefile # README.md # docs/backend/SYCL.md # examples/llava/README.md # examples/llava/clip.cpp # examples/run/run.cpp # examples/server/README.md # examples/sycl/run-llama2.sh # ggml/CMakeLists.txt # ggml/src/ggml-cpu/CMakeLists.txt # ggml/src/ggml-cuda/CMakeLists.txt # ggml/src/ggml-hip/CMakeLists.txt # ggml/src/ggml-musa/CMakeLists.txt # ggml/src/ggml-sycl/CMakeLists.txt # tests/test-backend-ops.cpp
2026-05-08 18:30:50 +00:00 · 2025-02-26 18:44:45 +08:00 · 2025-02-26 18:44:45 +08:00 · 722fc2dbf1
commit 722fc2dbf1
parent 50eae1ffeb 53e4db1012
79 changed files with 9330 additions and 5299 deletions
--- a/examples/llava/llava.cpp
+++ b/examples/llava/llava.cpp
@ -353,9 +353,10 @@ static bool encode_image_with_clip(clip_ctx * ctx_clip, int n_threads, const cli
        LOG_INF("%s: %d segments encoded in %8.2f ms\n", __func__, (int)img_res_v.size, (t_img_enc_batch_us - t_img_enc_start_us) / 1000.0);

        const int32_t * image_grid = clip_image_grid(ctx_clip);
+        const size_t num_gridpoints = get_clip_image_grid_size(ctx_clip);

        std::vector<std::pair<int, int>> grid_pinpoints;
-        for (int i = 0; i < 32 && image_grid[i] != 0; i += 2) {
+        for (size_t i = 0; i < num_gridpoints; i += 2) {
            grid_pinpoints.push_back({image_grid[i], image_grid[i+1]});
        }

@ -405,7 +406,8 @@ bool llava_validate_embed_size(const llama_context * ctx_llama, const clip_ctx *
 }

 bool llava_image_embed_make_with_clip_img(clip_ctx * ctx_clip, int n_threads, const clip_image_u8 * img, float ** image_embd_out, int * n_img_pos_out) {
-    int num_max_patches = 6;
+    // Granite vision uses up to 10 patches + base patch
+    int num_max_patches = 11;
    if (clip_is_minicpmv(ctx_clip)) {
        num_max_patches = 10;
    }