@inproceedings{6f5ae32baab243ed8ce5fd904f44b415,
title = "PeaCap: Patch-Level Retrieval for Lightweight Retrieval-Augmented Image Captioning",
abstract = "Retrieval-augmented image captioning aims to improve caption quality by grounding generation in external evidence, but most prior systems retrieve evidence using coarse whole-image similarity, which can miss small or rare objects in cluttered scenes. We propose PeaCap, a patch-based retrieval-augmented captioning framework that explicitly studies how retrieval granularity affects the quality of retrieved object evidence and downstream caption generation. PeaCap decomposes a query image into patches, performs patch-level image-to-image retrieval to obtain object tags, and fuses the retrieved tags with the whole-image embedding via a lightweight cross-attention module and an alignment loss to robustly prompt a frozen LLM. Analyses on retrieval (encoder choice, patch-vs.-whole retrieval, and patch-grid ablations) show that patch-level retrieval can improve object coverage, and experiments on COCO and out-of-domain benchmarks demonstrate competitive captioning performance under a lightweight training setup.",
keywords = "image captioning, retrieval-augmented generation",
author = "Robin Viltoriano and Zhang, \{Wei Emma\} and Hu Wang and Sim, \{Mong Yuan\} and Yanjun Shu",
note = "Publisher Copyright: {\textcopyright} 2026 Owner/Author.; 49th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2026 ; Conference date: 20-07-2026 Through 24-07-2026",
year = "2026",
month = jul,
day = "19",
doi = "10.1145/3805712.3809956",
language = "英语",
series = "SIGIR 2026 - Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval",
publisher = "Association for Computing Machinery, Inc",
pages = "4216--4220",
booktitle = "SIGIR 2026 - Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval",
}