Multimodal Model¶
3 Models
Instruction Model¶
Alibaba-NLP/GVE-3B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 3.8B | 2048 | 32.8K | 7.0 GB | 2025-10-31 | eng-Latn |
Citation
@misc{guo2025gve,
title={Towards Universal Video Retrieval: Generalizing Video Embedding via Synthesized Multimodal Pyramid Curriculum},
author={Zhuoning Guo and Mingxin Li and Yanzhao Zhang and Dingkun Long and Pengjun Xie and Xiaowen Chu},
year={2025},
eprint={2510.27571},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2510.27571}
}
Alibaba-NLP/GVE-7B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.3B | 3584 | 32.8K | 15.4 GB | 2025-10-31 | eng-Latn |
Citation
@misc{guo2025gve,
title={Towards Universal Video Retrieval: Generalizing Video Embedding via Synthesized Multimodal Pyramid Curriculum},
author={Zhuoning Guo and Mingxin Li and Yanzhao Zhang and Dingkun Long and Pengjun Xie and Xiaowen Chu},
year={2025},
eprint={2510.27571},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2510.27571}
}
Alibaba-NLP/UEmbed-2B¶
License: cc-by-4.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.6B | 2048 | 8.2K | 4.8 GB | 2026-08-04 | eng-Latn |
Citation
@misc{uembed2026,
title={UEmbed: Unified Sparse and Dense Multimodal Embeddings},
author={Tingyu Song and Mingxin Li and Yanzhao Zhang and Dingkun Long and Pengjun Xie and Zhijie Nie and Yilun Zhao and Shu Wu},
year={2026},
eprint={2608.02583},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.02583},
}
Alibaba-NLP/UEmbed-4B¶
License: cc-by-4.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 5.0B | 2560 | 8.2K | 9.3 GB | 2026-08-04 | eng-Latn |
Citation
@misc{uembed2026,
title={UEmbed: Unified Sparse and Dense Multimodal Embeddings},
author={Tingyu Song and Mingxin Li and Yanzhao Zhang and Dingkun Long and Pengjun Xie and Zhijie Nie and Yilun Zhao and Shu Wu},
year={2026},
eprint={2608.02583},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.02583},
}
Alibaba-NLP/UEmbed-9B¶
License: cc-by-4.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 9.1B | 4096 | 8.2K | 17.0 GB | 2026-08-04 | eng-Latn |
Citation
@misc{uembed2026,
title={UEmbed: Unified Sparse and Dense Multimodal Embeddings},
author={Tingyu Song and Mingxin Li and Yanzhao Zhang and Dingkun Long and Pengjun Xie and Zhijie Nie and Yilun Zhao and Shu Wu},
year={2026},
eprint={2608.02583},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.02583},
}
OpenSearch-AI/Ops-MM-embedding-v1-2B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.2B | 1536 | 32.8K | 4.3 GB | 2025-07-03 | afr-Latn, ara-Arab, aze-Latn, bel-Cyrl, ben-Beng, ... (73) |
Citation
@misc{ops_mm_embedding_v1,
author = {OpenSearch-AI},
title = {Ops-MM-embedding-v1: State-of-the-Art Multimodal Embedding Models},
year = {2025},
url = {https://huggingface.co/OpenSearch-AI/Ops-MM-embedding-v1-2B},
}
OpenSearch-AI/Ops-MM-embedding-v1-7B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.3B | 3584 | 32.8K | 16.2 GB | 2025-07-03 | afr-Latn, ara-Arab, aze-Latn, bel-Cyrl, ben-Beng, ... (73) |
Citation
@misc{ops_mm_embedding_v1,
author = {OpenSearch-AI},
title = {Ops-MM-embedding-v1: State-of-the-Art Multimodal Embedding Models},
year = {2025},
url = {https://huggingface.co/OpenSearch-AI/Ops-MM-embedding-v1-2B},
}
Qwen/Qwen3-VL-Embedding-2B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.1B | 2048 | 32.8K | 7.5 GB | 2026-01-08 | eng-Latn |
Citation
@article{qwen3vlembedding,
title={Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking},
author={Li, Mingxin and Zhang, Yanzhao and Long, Dingkun and Chen Keqin and Song, Sibo and Bai, Shuai and Yang, Zhibo and Xie, Pengjun and Yang, An and Liu, Dayiheng and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2601.04720},
year={2026}
}
Qwen/Qwen3-VL-Embedding-8B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.1B | 4096 | 32.8K | 29.8 GB | 2026-01-08 | eng-Latn |
Citation
@article{qwen3vlembedding,
title={Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking},
author={Li, Mingxin and Zhang, Yanzhao and Long, Dingkun and Chen Keqin and Song, Sibo and Bai, Shuai and Yang, Zhibo and Xie, Pengjun and Yang, An and Liu, Dayiheng and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2601.04720},
year={2026}
}
Qwen/Qwen3-VL-Reranker-2B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.1B | not specified | 32.8K | 4.0 GB | 2026-01-07 | ara-Arab, ces-Latn, cmn-Hans, dan-Latn, deu-Latn, ... (33) |
Citation
@article{qwen3vlembedding,
title={Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking},
author={Li, Mingxin and Zhang, Yanzhao and Long, Dingkun and Chen Keqin and Song, Sibo and Bai, Shuai and Yang, Zhibo and Xie, Pengjun and Yang, An and Liu, Dayiheng and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2601.04720},
year={2026}
}
Qwen/Qwen3-VL-Reranker-8B¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.8B | not specified | 32.8K | 16.3 GB | 2026-01-07 | ara-Arab, ces-Latn, cmn-Hans, dan-Latn, deu-Latn, ... (33) |
Citation
@article{qwen3vlembedding,
title={Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking},
author={Li, Mingxin and Zhang, Yanzhao and Long, Dingkun and Chen Keqin and Song, Sibo and Bai, Shuai and Yang, Zhibo and Xie, Pengjun and Yang, An and Liu, Dayiheng and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2601.04720},
year={2026}
}
TianchengGu/UniME-V2-LLaVA-OneVision-8B¶
License: apache-2.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.0B | 3584 | 32.8K | 15.0 GB | 2025-10-15 | eng-Latn |
Citation
@misc{gu2025unimev2mllmasajudgeuniversalmultimodal,
title={UniME-V2: MLLM-as-a-Judge for Universal Multimodal Embedding Learning},
author={Tiancheng Gu and Kaicheng Yang and Kaichen Zhang and Xiang An and Ziyong Feng and Yueyi Zhang and Weidong Cai and Jiankang Deng and Lidong Bing},
year={2025},
eprint={2510.13515},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2510.13515},
}
dnotitia/DNA-VL-STEER-2B¶
License: apache-2.0 • Openness: 3/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.1B | 2048 | 32.8K | 7.5 GB | 2026-08-27 | ara-Arab, ben-Beng, ces-Latn, dan-Latn, deu-Latn, ... (36) |
qihoo360/RzenEmbed¶
License: mit • Openness: 5/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.3B | 3584 | 32.8K | 16.2 GB | 2025-11-06 | eng-Latn, zho-Hans |
Citation
@article{jian2025rzenembed,
title={RzenEmbed: Towards Comprehensive Multimodal Retrieval},
author={Jian, Weijian and Zhang, Yajun and Liang, Dawei and Xie, Chunyu and He, Yixiao and Leng, Dawei and Yin, Yuhui},
journal={arXiv preprint arXiv:2510.27350},
year={2025}
}
tencent/WeMM-Embedding-2B¶
License: apache-2.0 • Openness: 5/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.2B | [64, 128, 256, 512, 1024, 2048] | 262.1K | 4.1 GB | 2026-08-25 | eng-Latn, zho-Hans |
Citation
@article{wemm-embedding,
title={WeMM-Embedding: WeChat Multi-Modal Embedding Technical Report},
author={Junjie Zhou and Ke Mei and Lei Li and Tianyi Wang and Fengyun Rao and Jing Lyu},
year={2026},
eprint={2608.24053},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.24053},
}
tencent/WeMM-Embedding-4B¶
License: apache-2.0 • Openness: 5/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 4.2B | [64, 128, 256, 512, 1024, 2560] | 262.1K | 7.9 GB | 2026-08-25 | eng-Latn, zho-Hans |
Citation
@article{wemm-embedding,
title={WeMM-Embedding: WeChat Multi-Modal Embedding Technical Report},
author={Junjie Zhou and Ke Mei and Lei Li and Tianyi Wang and Fengyun Rao and Jing Lyu},
year={2026},
eprint={2608.24053},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.24053},
}
tencent/WeMM-Embedding-9B¶
License: apache-2.0 • Openness: 5/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 9.0B | [64, 128, 256, 512, 1024, 2048, 4096] | 262.1K | 16.8 GB | 2026-08-25 | eng-Latn, zho-Hans |
Citation
@article{wemm-embedding,
title={WeMM-Embedding: WeChat Multi-Modal Embedding Technical Report},
author={Junjie Zhou and Ke Mei and Lei Li and Tianyi Wang and Fengyun Rao and Jing Lyu},
year={2026},
eprint={2608.24053},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2608.24053},
}
zhibinlan/UME-R1-2B¶
License: apache-2.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.2B | 1536 | 32.8K | 8.2 GB | 2025-11-10 | eng-Latn |
Citation
@article{lan2025ume,
title={UME-R1: Exploring Reasoning-Driven Generative Multimodal Embeddings},
author={Lan, Zhibin and Niu, Liqiang and Meng, Fandong and Zhou, Jie and Su, Jinsong},
journal={arXiv preprint arXiv:2511.00405},
year={2025}
}
zhibinlan/UME-R1-7B¶
License: apache-2.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 8.3B | 3584 | 32.8K | 30.9 GB | 2025-11-10 | eng-Latn |
Citation
@article{lan2025ume,
title={UME-R1: Exploring Reasoning-Driven Generative Multimodal Embeddings},
author={Lan, Zhibin and Niu, Liqiang and Meng, Fandong and Zhou, Jie and Su, Jinsong},
journal={arXiv preprint arXiv:2511.00405},
year={2025}
}
Non-instruction Model¶
VLM2Vec/VLM2Vec-V2.0¶
License: apache-2.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.2B | 1536 | 32.8K | 4.1 GB | 2025-04-30 | eng-Latn |
Citation
@misc{meng2025vlm2vecv2advancingmultimodalembedding,
title={VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents},
author={Rui Meng and Ziyan Jiang and Ye Liu and Mingyi Su and Xinyi Yang and Yuepeng Fu and Can Qin and Zeyuan Chen and Ran Xu and Caiming Xiong and Yingbo Zhou and Wenhu Chen and Semih Yavuz},
year={2025},
eprint={2507.04590},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2507.04590},
}
encord-team/ebind-points-vision¶
License: cc-by-nc-sa-4.0 • Openness: 4/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 1.7B | 1024 | 512 | 6.3 GB | 2025-11-19 | eng-Latn |
Citation
@misc{broadbent2025ebindpracticalapproachspace,
title={{EBind}: a practical approach to space binding},
author={Jim Broadbent and Felix Cohen and Frederik Hvilshøj and Eric Landau and Eren Sasoglu},
year={2025},
eprint={2511.14229},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2511.14229},
}
friedrichor/Unite-Base-Qwen2-VL-2B¶
License: apache-2.0 • Openness: 6/6 • Learn more →
| Parameters | Emb. Dim | Max Tokens | Memory | Released | Languages |
|---|---|---|---|---|---|
| 2.2B | 1536 | 32.8K | 8.2 GB | 2025-05-28 | eng-Latn |
Citation
@article{kong2025modality,
title={Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval},
author={Kong, Fanheng and Zhang, Jingyuan and Liu, Yahui and Zhang, Hongzhi and Feng, Shi and Yang, Xiaocui and Wang, Daling and Tian, Yu and W., Victoria and Zhang, Fuzheng and Zhou, Guorui},
journal={arXiv preprint arXiv:2505.19650},
year={2025}
}