
%Aigaion2 BibTeX export from HES SO Valais Publications
%Monday 31 August 2026 06:24:03 AM

@ARTICLE{,
    author = {Damm, Hendrik and Pakull, Tabea M. G. and Becker, Helmut and Bracke, Benjamin and Eryilmaz, Bahadir and Bloch, Louise and Br{\"{u}}ngel, Raphael and Schmidt, Cynthia S. and R{\"{u}}ckert, Johannes and Pelka, Obioma and Sch{\"{a}}fer, Henning and Idrissi-Yaghir, Ahmad and Abacha, Asma Ben and de Herrera, Alba G. Seco and M{\"{u}}ller, Henning and Friedrich, Christoph M.},
  keywords = {computer vision, Explainable AI, Image Captioning, Image Understanding, ImageCLEF, Multi-label classification, Radiology},
     month = sep,
     title = {Overview of ImageCLEFmedical 2025 – Medical Concept Detection and Interpretable Caption Generation},
   journal = {CEUR-WS.org/Vol-4038 - CLEF 2025 Working Notes},
    volume = {4038},
    number = {Paper 170},
      year = {2025},
       url = {https://ceur-ws.org/Vol-4038/paper_170.pdf},
  abstract = {The ImageCLEFmedical 2025 Caption task follows challenges held from 2017–2024 and comprises three subtasks:
concept detection, caption prediction, and a newly introduced explainability task. The goal is to extract Unified
Medical Language System (UMLS) concepts, generate fluent captions from medical images, and provide humaninterpretable justifications for the outputs. This year’s edition used an enlarged version of the Radiology Objects
in COntext version 2 (ROCOv2) dataset, which was expanded with new articles and the inclusion of the optical
coherence tomography (OCT) imaging modality. For concept detection, the F1-score was used to evaluate
predictions against UMLS terms. For caption prediction, evaluation was updated to a composite score averaging
six metrics to assess both relevance and factuality. The new explainability submissions were manually judged by a
radiologist. The 2025 task attracted 80 registered research groups, with 11 teams submitting a total of 149 graded
runs across the three subtasks. Top-performing systems for concept detection were predominantly based on
ensembles of Convolutional Neural Networks (CNNs). For caption prediction, a general shift towards fine-tuning
Vision-Language Models (VLMs) was observed, with adapted architectures like BLIP leading to strong results
across the new composite metrics. Finally, the inaugural explainability task saw initial submissions of post-hoc
visualizations, establishing a baseline and clarifying the need for model-intrinsic explanations in future editions.}
}

