|
200 | 200 | "_thumbnail": "/assets/images/papers/2509.24709.png" |
201 | 201 | }, |
202 | 202 | { |
| 203 | + "arXiv主页": { |
| 204 | + "link": "https://arxiv.org/abs/2508.05502", |
| 205 | + "text": "https://arxiv.org/abs/2508.05502" |
| 206 | + }, |
203 | 207 | "作者信息(每人一行,分号换行,数字表示单位信息,*表示Equal Contribution, ^表示通讯作者)": "Yufei Gao, Jiaying Fei, Nuo Chen, Ruirui Chen, Guohang Yan, Yunshi Lan, Botian Shi", |
204 | 208 | "单位信息(每个单位一行,分号换行)": "1 Shanghai AI Laboratory\n2 East China Normal University\n3 The Chinese University of Hong Kong, Shenzhen\n4\nInstitute of High Performance Computing, A*STAR", |
205 | 209 | "摘要": "Multimodal Large Language Models (MLLMs) perform strongly in high-resource languages, yet their effectiveness drops sharply in low-resource settings, largely due to the scarcity of aligned and culturally informative multimodal data. Existing multilingual enhancement approaches predominantly rely on text-only resources or translation-based pipelines, which improve surface-level fluency but often fail to capture culturally specific visual knowledge.\nIn this work, we present MELLA, a large-scale multimodal multilingual dataset designed to support both linguistic fluency and culturally grounded visual understanding in low-resource languages. MELLA is constructed using a dual-source data curation strategy that combines (i) native web image-alt-text pairs, which provide in-context, culture-specific visual-textual alignments, and (ii) high-quality image descriptions generated in a high-resource language and translated into target languages to ensure linguistic richness and structural completeness. Rather than expanding multilingual coverage alone, this design explicitly disentangles two complementary learning signals that are conflated in existing multilingual multimodal datasets.\nMELLA covers eight low-resource languages and contains 6.8M image-text pairs spanning diverse domains and visual categories. Through controlled diagnostic fine-tuning experiments on multiple MLLM backbones, we show that training on MELLA mitigates the cultural hallucination gap, often manifested as culturally “thin“ descriptions, by enabling models to recognize and articulate culturally specific entities that are systematically overlooked by translation-centric pipelines. Our findings underscore the central role of data alignment, rather than model modification, in achieving culturally grounded multimodal understanding for low-resource languages.", |
206 | 210 | "是否为团队主导工作": true, |
207 | 211 | "期刊/会议": "Under Submission", |
208 | 212 | "记录创建日期": 1776268800000, |
| 213 | + "论文发表日期": 1754496000000, |
209 | 214 | "论文标题": "MELLA: Bridging Linguistic Capability and Cultural Groundedness for Low-Resource Language MLLMs", |
210 | 215 | "责任人": [ |
211 | 216 | { |
|
214 | 219 | "id": "ou_33cfcf2048b911c367917bf320f12862", |
215 | 220 | "name": "闫国行" |
216 | 221 | } |
217 | | - ] |
| 222 | + ], |
| 223 | + "_thumbnail": "/assets/images/papers/2508.05502.png" |
218 | 224 | }, |
219 | 225 | { |
220 | 226 | "作者信息(每人一行,分号换行,数字表示单位信息,*表示Equal Contribution, ^表示通讯作者)": "Hongwei Zhang, Zehui Ling, Ruicheng Zhu, Yue Zhang, ShaoxiongGuo, Jinrong Wen, Pinlong Cai, Botian Shi, Guohang Yan", |
|
223 | 229 | "是否为团队主导工作": true, |
224 | 230 | "期刊/会议": "Under Submission", |
225 | 231 | "记录创建日期": 1776268800000, |
| 232 | + "论文发表日期": 1776614400000, |
226 | 233 | "论文标题": "PropRAG: Seeded Relevance Diffusion on Chunk–Page Graphs for Long Multimodal Document Retrieval", |
227 | 234 | "责任人": [ |
228 | 235 | { |
|
371 | 378 | ], |
372 | 379 | "期刊/会议": "Under Submission", |
373 | 380 | "记录创建日期": 1776355200000, |
| 381 | + "论文发表日期": 1761580800000, |
374 | 382 | "论文标题": "MGA: Memory-Driven GUI Agent for Observation-Centric Interaction", |
375 | 383 | "论文状态": "已投稿并挂arXiv", |
376 | 384 | "_thumbnail": "/assets/images/papers/2510.24168.png" |
|
0 commit comments