@InProceedings{10.1007/978-3-032-38069-2_54,
author="Kim, Taehyo
and Eads, Amanda
and Mcfee, Brian
and McAllister, Tara
and Shu, Hai",
editor="Yang, Guang
and Adeli, Ehsan
and de Bruijne, Marleen
and Papie{\.{z}}, Bart{\l}omiej W.
and Speidel, Stefanie
and Tiwari, Pallavi
and Zheng, Guoyan
and Yaqub, Mohammad
and Dou, Qi
and Rekik, Islem",
title="Self-supervised Learning on Lingual Ultrasound Video Encodes Clinically Meaningful Articulatory Structure of Rhotic Production in Residual Speech Sound Disorder",
booktitle="Medical Image Computing and Computer Assisted Intervention -- MICCAI 2026",
year="2027",
publisher="Springer Nature Switzerland",
address="Cham",
pages="570--580",
abstract="Speech sound disorder (SSD) affects many school-aged children and can lead to long-term academic, social, and mental health difficulties. Speech-language pathologists use ultrasound, a non-invasive imaging modality, to visualize tongue motion during speech and provide biofeedback treatment. The American English rhotic (r sound) is particularly challenging, as distinct articulatory strategies can produce the same perceived sound. Label-efficient self-supervised learning systems are increasingly being developed to automate scalable and objective SSD assessment. However, existing work has primarily focused on phoneme-level classification and provides limited insight into within-phoneme articulatory variations. In this paper, we present the first self-supervised learning framework that focuses on within-phoneme articulatory structure of American English rhotic production from lingual ultrasound video of children with residual SSD. We further introduce speaker adversarial regularization to reduce inter-speaker variability while preserving articulatory structure in the learned representations. Our results on the recently curated PERCEPT-US multimodal corpus demonstrate that self-supervised learning without task specific fine-tuning yields embeddings that accurately discriminate between perceptually accurate and inaccurate rhotic productions (mean ACC = 0.974; mean MCC = 0.941) and between articulatory subtypes of perceptually accurate rhotics, namely retroflex and bunched configurations (mean ACC = 0.961; mean MCC = 0.897). Code is available at https://github.com/kimtae55/BYOL-VS.",
isbn="978-3-032-38069-2"
}

