@inproceedings{11568_1361188,
 abstract = {Emotion recognition on social media is often approached in unimodal or single-label settings, despite the multimodal nature of online communication. This paper presents a study of multilabel emotion recognition from paired text-image data. We evaluate vision--language encoders and compare them with strong unimodal baselines and a zero-shot multimodal LLM. A simple multimodal classifier built on CLIP achieves the most reliable performance. Data-centric additions such as emoji transcription, caption augmentation, and pseudo-labelling offer limited gains, whereas calibrated decision thresholds have a consistent effect. The results highlight the value of visual cues and show limitations of recent VLMs.},
 author = {Passaro, Lucia and Amadei, Davide and Bacciu, Davide},
 booktitle = {{{ESANN}} 2026 Proceedings, European Symposium on Artificial Neural Networks, Computational Intelligence and Machine Learning},
 doi = {10.14428/esann/2026.es2026-287},
 isbn = {978-2-87587-096-4},
 keywords = {emotion detection,multimodal emotion recognition,vision and language},
 pages = {235--240},
 title = {Emotion Recognition in Multimodal Social Data},
 year = {2026}
}

