@online{Farina2606.28551,
TITLE = {{DataComp}-{VLM}: Improved Open Datasets for Vision-Language Models},
AUTHOR = {Farina, Matteo and Udandarao, Vishaal and Nguyen, Thao and Kuzucu, Selim and B{\"o}ther, Maximilian and Hochlehnert, Andreas and Ghosh, Adhiraj and Nezhurina, Marianna and Roth, Karsten and Struber, Joschka and Zhang, Yuhui and Dziadzio, Sebastian and Sui, Elaine and Jahagirdar, Soumya and Ghosh, Dhruba and Hammoud, Hasan and De Min, Thomas and Caldarella, Simone and Mirza, Jehanzeb and Keh, Sedrick and Cherti, Mehdi and Kuehne, Hilde and Schiele, Bernt and Yeung-Levy, Serena and Naeem, Muhammad Ferjad and Tombari, Federico and Klimovic, Ana and Ricci, Elisa and Bethge, Matthias and Oh, Sewoong and Prabhu, Ameya and Tonioni, Alessio and Jitsev, Jenia and Mancini, Massimiliano and Schmidt, Ludwig and Parthasarathy, Nikhil},
LANGUAGE = {eng},
URL = {https://arxiv.org/abs/2606.28551},
EPRINT = {2606.28551},
EPRINTTYPE = {arXiv},
YEAR = {2026},
ABSTRACT = {Building performant Vision-Language Models (VLMs) requires carefully curating large-scale training datasets, yet the community lacks systematic benchmarks for evaluating such curation strategies. We introduce DataComp for VLMs (DCVLM), a benchmark for controlled data-centric experiments to improve VLM training. As part of DCVLM, we collect 160 datasets spanning four data types -- image-caption pairs, multimodal interleaved documents, text-only, and instruction-tuning data -- into a corpus of 6T multimodal tokens. DCVLM allows participants to test curation strategies (filtering, mixing, formatting, sampling) across 1B-8B models and 6.25B-200B token budgets. Models are then evaluated on a carefully selected suite of up to 52 downstream benchmarks across 9 domains. We conduct extensive experiments on DCVLM and find that data mixing, not filtering, is key to a high-quality training dataset: instruction-heavy mixtures scale better than caption-heavy ones, with gains widening at larger scales. The resulting dataset, DCVLM-Baseline, enables training an 8B VLM to 63.6% accuracy on our 33-task core suite with 200B training tokens. Compared to FineVision, the state-of-the-art open VLM training dataset, this represents an improvement of +5.4pp. DCVLM and all accompanying artifacts will be made publicly available at https://www.datacomp.ai/dcvlm/.},
}
