mtmd : support Kimi VL model (#15458)

* convert : fix tensor naming conflict for llama 4 vision * convert ok * support kimi vision model * clean up * fix style * fix calc number of output tokens * refactor resize_position_embeddings * add test case * rename build fn * correct a small bug
2025-10-27 08:21:30 +00:00 · 2025-08-26 12:54:19 +02:00
parent 85cc1ae998
commit 79a546220c
6 changed files with 211 additions and 61 deletions
--- a/tools/mtmd/clip-impl.h
+++ b/tools/mtmd/clip-impl.h
@@ -135,6 +135,7 @@ enum projector_type {
    PROJECTOR_TYPE_QWEN25O, // will be replaced by QWEN2A or QWEN25VL depending on clip_ctx
    PROJECTOR_TYPE_VOXTRAL,
    PROJECTOR_TYPE_LFM2,
+    PROJECTOR_TYPE_KIMIVL,
    PROJECTOR_TYPE_UNKNOWN,
 };

@@ -156,6 +157,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
    { PROJECTOR_TYPE_QWEN25O,   "qwen2.5o"},
    { PROJECTOR_TYPE_VOXTRAL,   "voxtral"},
    { PROJECTOR_TYPE_LFM2,      "lfm2"},
+    { PROJECTOR_TYPE_KIMIVL,    "kimivl"},
 };

 static projector_type clip_projector_type_from_string(const std::string & str) {