@inproceedings{d439e9303a634a2fa80bd2466dc20046,
title = "Semi-supervised Visual Feature Integration for Language Models through Sentence Visualization",
abstract = "Integrating visual features has been proved useful for natural language understanding tasks. Nevertheless, most existing multimodal language models highly rely on training on aligned image and text data. In this paper, we propose a novel semi-supervised visual integration framework for pre-trained language models. In the framework, the visual features are obtained through a sentence visualization and vision-language fusion mechanism. The uniqueness includes: 1) the integration is conducted via a semi-supervised framework and does not require aligned images for the processed sentences. 2) the framework works as an auxiliary component, and will not affect the language processing ability of the integrated language model. Experimental results on both natural language inference and reading comprehension tasks demonstrate that our framework improves the strong baseline language models. Considering that our framework only requires an image database, and does not require aligned images for the processed texts, it provides a feasible way for multimodal language learning.",
keywords = "language model, multimodal fusion, vision and language",
author = "Lisai Zhang and Qingcai Chen and Joanna Siebert and Buzhou Tang",
note = "Publisher Copyright: {\textcopyright} 2021 ACM.; 23rd ACM International Conference on Multimodal Interaction, ICMI 2021 ; Conference date: 18-10-2021 Through 22-10-2021",
year = "2021",
month = oct,
day = "18",
doi = "10.1145/3462244.3479965",
language = "英语",
series = "ICMI 2021 - Proceedings of the 2021 International Conference on Multimodal Interaction",
publisher = "Association for Computing Machinery, Inc",
pages = "682--686",
booktitle = "ICMI 2021 - Proceedings of the 2021 International Conference on Multimodal Interaction",
}