
%Aigaion2 BibTeX export van HES SO Valais Publications
%Monday 31 August 2026 06:00:28 AM

@ARTICLE{MOLCHANOVA2026104007,
    author = {Molchanova, Nataliia and Cagol, Alessandro and Ocampo-Pineda, Mario and Lu, Po-Jui and Weigel, Matthias and Chen, Xinjie and Beck, Erin S. and Tsagkas, Charidimos and Reich, Daniel S. and Bulcke, Colin Vanden and Stolting, Anna and Borrelli, Serena and Maggi, Pietro and Lugo, Sebastian Baez and Lemay, Delphine Ribes and Depeursinge, Adrien and Granziera, Cristina and M{\"{u}}ller, Henning and Gordaliza, Pedro M. and Bach Cuadra, Meritxell},
  keywords = {Brain, Cortical lesions, Deep Learning, detection, Magnetic resonance imaging, Multiple sclerosis, segmentation, Trustworthy AI},
     title = {A comparative study of deep learning for cortical lesion MRI segmentation with explainability analysis in multiple sclerosis},
   journal = {NeuroImage: Clinical},
    volume = {50},
      year = {2026},
     pages = {104007},
      issn = {2213-1582},
       url = {https://www.sciencedirect.com/science/article/pii/S2213158226000665},
       doi = {https://doi.org/10.1016/j.nicl.2026.104007},
  abstract = {Cortical lesions (CLs) have emerged as valuable biomarkers in multiple sclerosis (MS), offering high diagnostic specificity and prognostic relevance. However, their routine clinical integration remains limited due to subtle magnetic resonance imaging (MRI) appearance, challenges in expert annotation, and a lack of standardized automated methods. We present a multi-centric comparative study of CL detection and segmentation in MRI. A total of 656 MRI scans, including clinical trial and research data from four institutions, were acquired at 3T and 7T using MP2RAGE and MPRAGE sequences with expert-consensus annotations. We rely on the self-configuring nnU-Net framework, designed for medical imaging segmentation, and propose adaptations tailored to the improved CL detection. We evaluated model generalization through out-of-distribution testing, demonstrating promising lesion detection capabilities with an F1-score of 0.64 and 0.5 in and out of the domain, respectively. We also analyze internal model features and model errors for a better understanding of AI decision-making. Our study examines how data variability, lesion ambiguity, and protocol differences impact model performance, offering future recommendations to address these barriers to clinical adoption. Furthermore, we designed and implemented a medical expert questionnaire for better assessment of clinical value of the model predictions. To reinforce the reproducibility, the implementation and models will be publicly accessible and ready to use at GitHub and Zenodo.}
}

