@inproceedings{87cae5a59b0040c2953b75c7d536f9eb,
title = "Expanding the Image Embedding Space for Language-Free Text-to-Face Image Generation",
abstract = "Recent advancements in text-to-image (T2I) generation have revolutionized image synthesis, but conventional text-image paired training poses challenges when confronted with limited dataset size and narrow descriptive breadth. In particular, limited descriptive breadth can significantly impair a model{\textquoteright}s ability to generate unmentioned image features. This issue is also evident in text-to-face image generation, where a method will underperform when rendering an image based on the text description of a specific facial feature not present in the training dataset. Language-free training emerges as a promising solution to this problem, leveraging latent spaces like CLIP to facilitate generalization from image embeddings to text embeddings. However, the modality gap remains a hurdle for language-free trained models. To address this, we propose a Gaussian perturbation-based technique that enhances perturbation coverage and robustness across varying modality gap sizes without the need to train a prior model. Our method achieves new state-of-the-art results on MM-CelebA-HQ in the language-free setting, presenting a novel solution to challenges in text-to-face image generation on limited datasets.",
keywords = "diffusion, language-free, text-to-face, text-to-image",
author = "Joakim Jensen and Sukalpa Chanda and Krishnan, \{Narayanan C.\} and David Doermann",
note = "Publisher Copyright: {\textcopyright} The Author(s), under exclusive license to Springer Nature Switzerland AG 2025.; 46th Annual Conference of the German Association for Pattern Recognition, DAGM GCPR 2024, co-hosted with VMV 2024 ; Conference date: 10-09-2024 Through 13-09-2024",
year = "2025",
doi = "10.1007/978-3-031-85187-2\_5",
language = "English",
isbn = "9781424469116",
series = "Lecture Notes in Computer Science",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "72--86",
editor = "Daniel Cremers and Zorah L{\"a}hner and Michael Moeller and Matthias Nie{\ss}ner and Bj{\"o}rn Ommer and Rudolph Triebel",
booktitle = "Pattern Recognition - 46th DAGM German Conference, DAGM GCPR 2024, Proceedings",
address = "Germany",
}