]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
mtmd, common: various fixes (#27071)
authorXuan-Son Nguyen <redacted>
Fri, 14 Aug 2026 21:34:56 +0000 (23:34 +0200)
committerGitHub <redacted>
Fri, 14 Aug 2026 21:34:56 +0000 (23:34 +0200)
* apply fixes

* cont

* revert gguf fix

common/arg.cpp
src/models/dflash.cpp
src/models/minimax-m3.cpp
tools/mtmd/clip-impl.h
tools/mtmd/clip.cpp
tools/mtmd/mtmd-helper-common.h
tools/mtmd/mtmd-image.cpp

index 3e63ce0a6dfd48685798e22f18c107f2cad8d74b..49b33806a389b31706d7d896e59f7d9543d3edbd 100644 (file)
@@ -4077,6 +4077,9 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
         {"--spec-draft-n-max"}, "N",
         string_format("number of tokens to draft for speculative decoding (default: %d)", params.speculative.draft.n_max),
         [](common_params & params, int value) {
+            if (value < 0) {
+                throw std::invalid_argument("invalid value");
+            }
             params.speculative.draft.n_max = value;
         }
     ).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_LOOKUP, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_N_MAX"));
index daaa20826adf5f5283726b3aa004ae04001662f4..39e4f5dd8c51d5ad1cfed11765d3435f3666b54a 100644 (file)
@@ -43,6 +43,8 @@ void llama_model_dflash::load_arch_hparams(llama_model_loader & ml) {
         ml.get_key(LLM_KV_HYPER_CONNECTION_EPSILON,             hparams.dsv4_hc_eps);
         ml.get_arr(LLM_KV_ATTENTION_COMPRESS_RATIOS,            hparams.dsv4_compress_ratios, false);
 
+        GGML_ASSERT(hparams.dsv4_o_group_count > 0); // avoid div by zero
+
         if (hparams.expert_gating_func != LLAMA_EXPERT_GATING_FUNC_TYPE_SQRT_SOFTPLUS) {
             throw std::runtime_error("DSpark DSV4 draft expects sqrtsoftplus MoE scoring");
         }
index 854d5aed0f822df145801d13790872275dfa5cb0..1ba699d01665345a078e875d2d74671289befe35 100644 (file)
@@ -25,6 +25,8 @@ void llama_model_minimax_m3::load_arch_hparams(llama_model_loader & ml) {
     ml.get_key(LLM_KV_ATTENTION_INDEXER_LOCAL_BLOCKS,  hparams.indexer_local_blocks);
     msa_p = { (int) hparams.indexer_block_size, (int) hparams.indexer_top_k, (int) hparams.indexer_local_blocks };
 
+    GGML_ASSERT(hparams.indexer_block_size > 0); // avoid div by zero
+
     switch (hparams.n_layer()) {
         case 60: type = LLM_TYPE_428B_A23B; break;
         default: type = LLM_TYPE_UNKNOWN;
index b2c8b402132162ab026ea97efbc3a4cd3b9f6ac0..464f74ca05817d2d090aa5455bb72227cd12a442 100644 (file)
@@ -603,7 +603,7 @@ struct clip_image_u8 {
             // return a dummy value, so that legacy code can still process image without errors
             return { 0, 0, 0 };
         }
-        int idx = (y * nx + x) * 3;
+        size_t idx = ((size_t) y * (size_t) nx + (size_t) x) * 3;
         return { buf[idx], buf[idx + 1], buf[idx + 2] };
     }
 
@@ -611,8 +611,8 @@ struct clip_image_u8 {
         if (is_placeholder()) {
             return; // no-op
         }
-        int idx = (y * nx + x) * 3;
-        buf[idx] = rgb[0];
+        size_t idx = ((size_t) y * (size_t) nx + (size_t) x) * 3;
+        buf[idx]     = rgb[0];
         buf[idx + 1] = rgb[1];
         buf[idx + 2] = rgb[2];
     }
index 2fb2b5041dcd410cc5b2d51b5e6b4ea9deebaf27..c08b41d5907dcac67c638877f46d2db1d84f79a9 100644 (file)
@@ -1595,6 +1595,9 @@ struct clip_model_loader {
                         hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
                         hparams.image_resize_pad  = PAD_NONE;
                         get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
+                        // n_merge is used as a divisor in clip_image_batch_encode
+                        // (gh / n_merge); reject 0 to avoid int div-by-zero (DoS).
+                        GGML_ASSERT(hparams.n_merge > 0);
                         hparams.rope_theta = 10000.0f; // vision_config.rope_theta
                         // MiniMax-M3: max_pixels 451584 (=672^2) -> 576 merged tokens (image_seq_length)
                         hparams.set_limit_image_tokens(8, 576);
@@ -1823,7 +1826,9 @@ struct clip_model_loader {
                         // unlimited-ocr shares the v1 projector but tiles up to 32
                         get_u32(KEY_PREPROC_MIN_TILES, hparams.preproc_min_tiles, false);
                         get_u32(KEY_PREPROC_MAX_TILES, hparams.preproc_max_tiles, false);
-                        GGML_ASSERT(hparams.preproc_min_tiles <= hparams.preproc_max_tiles);
+                        GGML_ASSERT(hparams.preproc_min_tiles >= 0
+                                    && hparams.preproc_min_tiles <= hparams.preproc_max_tiles
+                                    && hparams.preproc_max_tiles <= 256);
                      } break;
                 case PROJECTOR_TYPE_HUNYUANVL:
                     {
@@ -1888,6 +1893,9 @@ struct clip_model_loader {
                         hparams.audio_window_len       = 400;
                         hparams.audio_hop_len          = 160;
                         get_u32(KEY_A_CHUNK_SIZE,           hparams.audio_chunk_size);
+                        // context_size is squared for the attn_dists/mask buffers; cap to prevent int32 overflow
+                        // (legitimate values are small, e.g. 12-200; 8192^2 = 67M still fits int32)
+                        GGML_ASSERT(hparams.audio_chunk_size > 0 && hparams.audio_chunk_size <= 8192);
                         get_u32(KEY_A_CONV_KERNEL_SIZE,     hparams.audio_conv_kernel_size);
                         get_u32(KEY_A_MAX_POS_EMB,          hparams.audio_max_pos_emb);
                         get_u32(KEY_A_PROJ_WINDOW_SIZE,     hparams.audio_proj_window_size);
@@ -1927,8 +1935,9 @@ struct clip_model_loader {
                     // note: some models having hparams.image_size == 0, which means the image size is dynamic
                     throw std::runtime_error(string_format("%s: image_size (%d) cannot be negative\n", __func__, hparams.image_size));
                 }
-                if (hparams.image_size > 65536) {
-                    throw std::runtime_error(string_format("%s: image_size (%d) is too large (max 65536)\n", __func__, hparams.image_size));
+                if (hparams.image_size > 8192) {
+                    // cap prevents int32 overflow in n_patches = (image_size/patch_size)^2
+                    throw std::runtime_error(string_format("%s: image_size (%d) is too large (max 8192)\n", __func__, hparams.image_size));
                 }
                 if (hparams.patch_size <= 0 || hparams.patch_size >= 65536) {
                     throw std::runtime_error(string_format("%s: patch_size (%d) must be positive and less than 65536\n", __func__, hparams.patch_size));
@@ -1939,9 +1948,12 @@ struct clip_model_loader {
                 if (hparams.image_max_pixels < hparams.image_min_pixels) {
                     throw std::runtime_error(string_format("%s: image_max_pixels (%d) is less than image_min_pixels (%d)\n", __func__, hparams.image_max_pixels, hparams.image_min_pixels));
                 }
-                if (hparams.n_merge < 0 || hparams.n_merge >= 65536) {
+                if (hparams.n_merge <= 0 || hparams.n_merge >= 65536) {
                     throw std::runtime_error(string_format("%s: n_merge (%d) must be greater than 0 and less than 65536\n", __func__, hparams.n_merge));
                 }
+                if (hparams.attn_window_size > 4096) {
+                    throw std::runtime_error(string_format("%s: attn_window_size (%d) is too large (max 4096)\n", __func__, hparams.attn_window_size));
+                }
             }
 
             LOG_INF("%s: projector:          %s\n", __func__, proj_type.c_str());
@@ -5408,13 +5420,13 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
                 const int context_size = ctx->model.hparams.audio_chunk_size;
                 const int max_pos_emb  = ctx->model.hparams.audio_max_pos_emb;
 
-                std::vector<int32_t> dists(context_size * context_size);
+                std::vector<int32_t> dists((size_t) context_size * (size_t) context_size);
                 for (int i = 0; i < context_size; i++) {
                     for (int j = 0; j < context_size; j++) {
                         int d = i - j;
                         if (d < -context_size) d = -context_size;
                         if (d >  context_size) d =  context_size;
-                        dists[i * context_size + j] = d + max_pos_emb;
+                        dists[(size_t) i * (size_t) context_size + (size_t) j] = d + max_pos_emb;
                     }
                 }
                 set_input_i32("attn_dists", dists);
@@ -5423,13 +5435,13 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
                 const int remainder  = n_frames % context_size;
                 if (remainder > 0) {
                     const int num_blocks = (n_frames + context_size - 1) / context_size;
-                    std::vector<float> mask(context_size * context_size * num_blocks, 0.0f);
+                    std::vector<float> mask((size_t) context_size * (size_t) context_size * (size_t) num_blocks, 0.0f);
                     const float neg_inf = -INFINITY;
-                    const int last_block_offset = (num_blocks - 1) * context_size * context_size;
+                    const size_t last_block_offset = (size_t) (num_blocks - 1) * (size_t) context_size * (size_t) context_size;
                     for (int q = 0; q < context_size; q++) {
                         for (int k = 0; k < context_size; k++) {
                             if (q >= remainder || k >= remainder) {
-                                mask[last_block_offset + q * context_size + k] = neg_inf;
+                                mask[last_block_offset + (size_t) q * (size_t) context_size + (size_t) k] = neg_inf;
                             }
                         }
                     }
index 968b4df9c8f7cec4f8054681eac3745f2ad2bfcc..f907346c7b58c02e3274d581f8e375cbab58a272 100644 (file)
@@ -82,7 +82,7 @@ struct decode_embd_batch {
     llama_batch batch;
     decode_embd_batch(float * embd, int32_t n_tokens, int n_pos_per_embd, int n_mmproj_embd) : n_pos_per_embd(n_pos_per_embd), n_mmproj_embd(n_mmproj_embd) {
         GGML_ASSERT(n_tokens > 0 && n_pos_per_embd > 0 && n_mmproj_embd > 0);
-        pos     .resize(n_tokens * n_pos_per_embd);
+        pos     .resize((size_t) n_tokens * (size_t) n_pos_per_embd);
         n_seq_id.resize(n_tokens);
         seq_ids .resize(n_tokens + 1);
         logits  .resize(n_tokens);
@@ -115,10 +115,12 @@ struct decode_embd_batch {
         GGML_ASSERT(!rel_pos.empty() && (int32_t)rel_pos.size() == batch.n_tokens);
         seq_id_0[0] = seq_id;
         for (int32_t i = 0; i < batch.n_tokens; i++) {
-            pos[i                     ] = rel_pos[i].t;
-            pos[i + batch.n_tokens    ] = rel_pos[i].y;
-            pos[i + batch.n_tokens * 2] = rel_pos[i].x;
-            pos[i + batch.n_tokens * 3] = rel_pos[i].z;
+            const size_t idx = (size_t) i;
+            const size_t n_tokens = (size_t) batch.n_tokens;
+            pos[idx                    ] = rel_pos[i].t;
+            pos[idx + n_tokens         ] = rel_pos[i].y;
+            pos[idx + n_tokens * 2     ] = rel_pos[i].x;
+            pos[idx + n_tokens * 3     ] = rel_pos[i].z;
         }
         for (int i = 0; i < batch.n_tokens; i++) {
             batch.n_seq_id[i] = 1;
@@ -132,10 +134,12 @@ struct decode_embd_batch {
         GGML_ASSERT(n_pos_per_embd == 4);
         seq_id_0[0] = seq_id;
         for (int i = 0; i < batch.n_tokens; i++) {
-            pos[i                     ] = pos_0 + i;
-            pos[i + batch.n_tokens    ] = pos_0 + i;
-            pos[i + batch.n_tokens * 2] = pos_0 + i;
-            pos[i + batch.n_tokens * 3] = pos_0 + i;
+            const size_t idx = (size_t) i;
+            const size_t n_tokens = (size_t) batch.n_tokens;
+            pos[idx                    ] = pos_0 + i;
+            pos[idx + n_tokens         ] = pos_0 + i;
+            pos[idx + n_tokens * 2     ] = pos_0 + i;
+            pos[idx + n_tokens * 3     ] = pos_0 + i;
         }
         for (int i = 0; i < batch.n_tokens; i++) {
             batch.n_seq_id[i] = 1;
@@ -148,7 +152,7 @@ struct decode_embd_batch {
         GGML_ASSERT(offset >= 0 && n_tokens > 0 && offset + n_tokens <= batch.n_tokens);
         llama_pos * pos_ptr;
         pos_view.clear();
-        pos_view.reserve(n_tokens * n_pos_per_embd);
+        pos_view.reserve((size_t) n_tokens * (size_t) n_pos_per_embd);
         if (n_pos_per_embd > 1) {
             // mrope
             // for example, with layout of src: 1234...1234...1234...1234...
@@ -157,7 +161,7 @@ struct decode_embd_batch {
                 // assume n_tokens is less than or equal to batch.n_tokens
                 // batch.n_tokens is number of **total** tokens
                 // n_tokens is number of viewed token
-                size_t src_idx = i * batch.n_tokens + offset;
+                size_t src_idx = (size_t) i * (size_t) batch.n_tokens + (size_t) offset;
                 pos_view.insert(pos_view.end(),
                     pos.data() + src_idx,
                     pos.data() + src_idx + n_tokens);
index 813fe493fa0ad11d88c21f3471630f2744704d2b..02a0a29cd81db5dfd09d2d528e79d8833b4db906 100644 (file)
@@ -1317,7 +1317,7 @@ void mtmd_image_preprocessor_step3vl::img_u8_resize_bilinear_to_f32(
     const float scale_x = static_cast<float>(src_size.width)  / target_width;
     const float scale_y = static_cast<float>(src_size.height) / target_height;
 
-    std::vector<float> local_buf(3 * target_width * target_height);
+    std::vector<float> local_buf((size_t) 3 * (size_t) target_width * (size_t) target_height);
 
     for (int y = 0; y < target_height; ++y) {
         const float src_y = (static_cast<float>(y) + 0.5f) * scale_y - 0.5f;
@@ -1338,7 +1338,7 @@ void mtmd_image_preprocessor_step3vl::img_u8_resize_bilinear_to_f32(
             const auto p10 = src.get_pixel(x0, y1);
             const auto p11 = src.get_pixel(x1, y1);
 
-            const size_t idx_dst = 3 * (y * target_width + x);
+            const size_t idx_dst = (size_t) 3 * ((size_t) y * (size_t) target_width + (size_t) x);
             for (int c = 0; c < 3; ++c) {
                 const float v00 = (static_cast<float>(p00[c]) / 255.0f - mean[c]) / std[c];
                 const float v01 = (static_cast<float>(p01[c]) / 255.0f - mean[c]) / std[c];