@inproceedings{4f8ddb021ce04565959b4725148ab494,
title = "Multi-Modal Multi-Task Unified Embedding Model (M3T-UEM): A Task-Adaptive Representation Learning Framework",
abstract = "We present Multi-Modal Multi-Task Unified Embedding Model (M3T-UEM), a framework that advances visionlanguage matching and retrieval by leveraging a large language model (LLM) backbone. While concurrent LLMbased approaches have demonstrated impressive capabilities in multimodal and multitask scenarios; our work introduces novel mechanisms for task-adaptive learning and embedding extraction that further enhance the potential of LLM-based retrieval systems. Our key technical contribution lies in the development of a task-aware contrastive learning framework with an automated Bayesian weighing mechanism. This approach provides a principled way to balance multiple tasks during training, departing from conventional contrastive learning strategies. We further enhance the framework through a multiple token summarization strategy and an auxiliary language modeling objective, which together significantly improve retrieval performance. Comprehensive experiments on M-BEIR and ICinW benchmarks demonstrate the effectiveness of M3T-UEM, showing competitive or superior performance compared to both traditional encoder-based methods and recent LLMbased approaches. Furthermore, we demonstrate particular strengths in handling compositional conceptual changes and multilingual scenarios owing to the incorporation of an LLM backbone where the method drastically outperforms CLIP in zero-shot settings, often by orders of magnitude.∗",
keywords = "contrastive learning, embedding models, llm, multi-modal retrieval",
author = "Rohan Sharma and Changyou Chen and Chang, \{Feng Ju\} and Seongjun Yun and Xiaohu Xie and Rui Meng and Dehong Xu and Alejandro Mottini and Qingjun Cui",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 IEEE/CVF International Conference on Computer Vision, ICCV 2025 ; Conference date: 19-10-2025 Through 23-10-2025",
year = "2025",
doi = "10.1109/ICCV51701.2025.02115",
language = "English",
series = "Proceedings of the IEEE International Conference on Computer Vision",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "22783--22793",
booktitle = "Proceedings - 2025 IEEE/CVF International Conference on Computer Vision, ICCV 2025",
address = "United States",
}