diff --git a/common/arg.cpp b/common/arg.cpp index 1f70d0ad45..a5a72970d6 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -4065,6 +4065,9 @@ common_params_context common_params_parser_init(common_params & params, llama_ex {"--spec-draft-n-max"}, "N", string_format("number of tokens to draft for speculative decoding (default: %d)", params.speculative.draft.n_max), [](common_params & params, int value) { + if (value < 0) { + throw std::invalid_argument("invalid value"); + } params.speculative.draft.n_max = value; } ).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_LOOKUP, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_N_MAX")); diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index c2d5059e02..c08b41d590 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1948,9 +1948,12 @@ struct clip_model_loader { if (hparams.image_max_pixels < hparams.image_min_pixels) { throw std::runtime_error(string_format("%s: image_max_pixels (%d) is less than image_min_pixels (%d)\n", __func__, hparams.image_max_pixels, hparams.image_min_pixels)); } - if (hparams.n_merge < 0 || hparams.n_merge >= 65536) { + if (hparams.n_merge <= 0 || hparams.n_merge >= 65536) { throw std::runtime_error(string_format("%s: n_merge (%d) must be greater than 0 and less than 65536\n", __func__, hparams.n_merge)); } + if (hparams.attn_window_size > 4096) { + throw std::runtime_error(string_format("%s: attn_window_size (%d) is too large (max 4096)\n", __func__, hparams.attn_window_size)); + } } LOG_INF("%s: projector: %s\n", __func__, proj_type.c_str()); @@ -1985,8 +1988,6 @@ struct clip_model_loader { LOG_INF("%s: preproc_tiles: %d - %d\n", __func__, hparams.preproc_min_tiles, hparams.preproc_max_tiles); } } else if (is_audio) { - GGML_ASSERT(hparams.attn_window_size <= 4096); // avoid int32_t overflow in attn_dists/mask buffers - LOG_INF("\n--- audio hparams ---\n"); LOG_INF("%s: n_mel_bins: %d\n", __func__, hparams.n_mel_bins); LOG_INF("%s: proj_stack_factor: %d\n", __func__, hparams.proj_stack_factor); diff --git a/tools/mtmd/mtmd-helper-common.h b/tools/mtmd/mtmd-helper-common.h index d027e4c9cb..f907346c7b 100644 --- a/tools/mtmd/mtmd-helper-common.h +++ b/tools/mtmd/mtmd-helper-common.h @@ -117,7 +117,7 @@ struct decode_embd_batch { for (int32_t i = 0; i < batch.n_tokens; i++) { const size_t idx = (size_t) i; const size_t n_tokens = (size_t) batch.n_tokens; - pos[idx ] = rel_pos[i].t; + pos[idx ] = rel_pos[i].t; pos[idx + n_tokens ] = rel_pos[i].y; pos[idx + n_tokens * 2 ] = rel_pos[i].x; pos[idx + n_tokens * 3 ] = rel_pos[i].z; @@ -136,6 +136,7 @@ struct decode_embd_batch { for (int i = 0; i < batch.n_tokens; i++) { const size_t idx = (size_t) i; const size_t n_tokens = (size_t) batch.n_tokens; + pos[idx ] = pos_0 + i; pos[idx + n_tokens ] = pos_0 + i; pos[idx + n_tokens * 2 ] = pos_0 + i; pos[idx + n_tokens * 3 ] = pos_0 + i;