Real‐Time Classification of Functional Laryngeal Behaviors From Vocal Fold Kinematics
{"OBJECTIVES:":[0],"Develop":[1],"a":[2,41,132,230],"real-time":[3],"prediction":[4],"model":[5,48,101,130,168],"to":[6,27,65,187],"detect":[7],"laryngeal":[8,56,69,160,205,240],"behaviors":[9],"from":[10,59,95,183,207],"videolaryngoscopy-based":[11,208],"tracking":[12,52,210],"data":[13,20,53],"during":[14,211],"clinical":[15,212],"examinations.":[16],"This":[17],"will":[18],"improve":[19],"collection":[21],"quality":[22],"by":[23],"providing":[24],"immediate":[25],"feedback":[26],"clinicians":[28],"and":[29,81,113,122,138,155,225,237],"enable":[30],"future":[31],"automation":[32],"pipelines":[33],"for":[34,185,189,197,232],"efficient":[35],"large-scale":[36],"analysis.":[37],"METHODS:":[38],"We":[39],"trained":[40],"stateful":[42],"residual":[43],"Gated":[44],"Recurrent":[45],"Unit":[46],"(GRU)":[47],"that":[49],"analyzed":[50],"the":[51,67,146,164,167,179],"of":[54,83,136,141,159,204,220,239],"39":[55],"keypoints,":[57],"derived":[58],"our":[60],"published":[61],"keypoint":[62],"detection":[63,219],"model,":[64],"predict":[66],"patient's":[68],"task":[70,161],"state.":[71],"These":[72],"included":[73],"\\"phonation,\\"":[74],"\\"sustained":[75],"phonation,\\"":[76],"\\"swallowing,\\"":[77],"\\"idle,\\"":[78],"\\"coughing,\\"":[79],"\\"sniffing,\\"":[80],"\\"out":[82],"view.\\"":[84],"Model":[85],"development":[86],"used":[87],"916":[88],"state":[89,173],"segments":[90],"comprising":[91,108],"222,065":[92],"video":[93],"frames":[94],"72":[96],"laryngoscopy":[97],"videos.":[98],"The":[99,129,191],"resulting":[100],"was":[102,117,195],"evaluated":[103,144],"on":[104,145,178],"an":[105,139],"independent":[106,147],"dataset":[107],"8":[109],"videos,":[110],"123":[111],"segments,":[112],"49,770":[114],"frames.":[115],"Performance":[116],"assessed":[118],"using":[119],"classification":[120,203],"metrics":[121],"temporal":[123,157],"intersection":[124],"over":[125],"union":[126],"(mIoU).":[127],"RESULTS:":[128],"achieved":[131],"mean":[133],"accuracy":[134],"score":[135],"92%":[137],"mIoU":[140],"0.82":[142],"when":[143],"test":[148,180],"dataset,":[149],"indicating":[150],"agreement":[151],"with":[152,175],"manual":[153],"annotations":[154],"accurate":[156],"identification":[158],"states.":[162],"At":[163],"class":[165],"level,":[166],"performed":[169],"consistently":[170],"across":[171],"most":[172],"categories,":[174],"per-class":[176],"F1-scores":[177],"set":[181],"ranging":[182],"83%":[184],"phonation":[186],"97%":[188],"sniffing.":[190],"lowest":[192],"validation":[193],"F1":[194],"observed":[196],"cough":[198],"(82%":[199],"[70%-92%]).":[200],"CONCLUSION:":[201],"Real-time":[202],"states":[206,221],"pose":[209],"examinations":[213],"is":[214],"feasible.":[215],"With":[216],"reliable":[217],"automated":[218,234],"such":[222],"as":[223],"swallowing":[224],"phonation,":[226],"this":[227],"work":[228],"establishes":[229],"foundation":[231],"further":[233],"analysis,":[235],"interpretation,":[236],"documentation":[238],"pathology.":[241]}
Authors
- Kristina Simonyan (ORCID: https://orcid.org/0000-0001-7444-0437)
- Matthew R. Naunheim (ORCID: https://orcid.org/0000-0002-3927-3984)
- Aki Koivu (ORCID: https://orcid.org/0000-0003-4116-344X)
- Pin‐Yu Lin
Institutions
- Massachusetts Eye and Ear Infirmary (US)
- Harvard University (US)
Publication Details
- Journal
- The Laryngoscope
- Published
- 2026-09-18
- DOI
- https://doi.org/10.1002/lary.70927
- Primary Topic
- Voice and Speech Disorders
- Type
- article
- Field-Weighted Citation Impact
- 0.00