<?xml version="1.0" encoding="iso-8859-1" standalone="no"?>
<!DOCTYPE GmsArticle SYSTEM "http://www.egms.de/dtd/2.0.34/GmsArticle.dtd">
<GmsArticle xmlns:xlink="http://www.w3.org/1999/xlink">
  <MetaData>
    <Identifier>26dgpp19</Identifier>
    <IdentifierDoi>10.3205/26dgpp19</IdentifierDoi>
    <IdentifierUrn>urn:nbn:de:0183-26dgpp195</IdentifierUrn>
    <ArticleType>Vortrag</ArticleType>
    <TitleGroup>
      <Title language="en">Automated hoarseness severity estimation using high-speed videoendoscopy-synchronized acoustic signals under phonation constraints</Title>
    </TitleGroup>
    <CreatorList>
      <Creator>
        <PersonNames>
          <Lastname>Patel</Lastname>
          <LastnameHeading>Patel</LastnameHeading>
          <Firstname>P.</Firstname>
          <Initials>P</Initials>
        </PersonNames>
        <Address>University Hospital Erlangen, Division of Phoniatrics and Pediatric Audiology, Department of Otorhinolaryngology Erlangen, Head and Neck Surgery, Waldstra&#223;e 1, 91054 Erlangen, Deutschland<Affiliation>University Hospital Erlangen, Division of Phoniatrics and Pediatric Audiology, Department of Otorhinolaryngology Erlangen, Head and Neck Surgery, Erlangen, Germany</Affiliation></Address>
        <Email>param.patel&#64;uk-erlangen.de</Email>
        <Creatorrole corresponding="yes" presenting="yes">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Donhauser</Lastname>
          <LastnameHeading>Donhauser</LastnameHeading>
          <Firstname>J.</Firstname>
          <Initials>J</Initials>
        </PersonNames>
        <Address>
          <Affiliation>University Hospital Erlangen, Division of Phoniatrics and Pediatric Audiology, Department of Otorhinolaryngology Erlangen, Head and Neck Surgery, Erlangen, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Sch&#252;tzenberger</Lastname>
          <LastnameHeading>Sch&#252;tzenberger</LastnameHeading>
          <Firstname>A.</Firstname>
          <Initials>A</Initials>
        </PersonNames>
        <Address>
          <Affiliation>University Hospital Erlangen, Division of Phoniatrics and Pediatric Audiology, Department of Otorhinolaryngology Erlangen, Head and Neck Surgery, Erlangen, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Kunduk</Lastname>
          <LastnameHeading>Kunduk</LastnameHeading>
          <Firstname>M.</Firstname>
          <Initials>M</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Louisiana State University, Department of Communication Sciences and Disorders, Louisiana, United States</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>D&#246;llinger</Lastname>
          <LastnameHeading>D&#246;llinger</LastnameHeading>
          <Firstname>M.</Firstname>
          <Initials>M</Initials>
        </PersonNames>
        <Address>
          <Affiliation>University Hospital Erlangen, Division of Phoniatrics and Pediatric Audiology, Department of Otorhinolaryngology Erlangen, Head and Neck Surgery, Erlangen, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
    </CreatorList>
    <PublisherList>
      <Publisher>
        <Corporation>
          <Corporatename>German Medical Science GMS Publishing House</Corporatename>
        </Corporation>
        <Address>D&#252;sseldorf</Address>
      </Publisher>
    </PublisherList>
    <SubjectGroup>
      <SubjectheadingDDB>610</SubjectheadingDDB>
    </SubjectGroup>
    <DatePublishedList>
      <DatePublished>20260826</DatePublished>
    </DatePublishedList>
    <Language>engl</Language>
    <License license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
      <AltText language="en">This is an Open Access article distributed under the terms of the Creative Commons Attribution 4.0 License.</AltText>
      <AltText language="de">Dieser Artikel ist ein Open-Access-Artikel und steht unter den Lizenzbedingungen der Creative Commons Attribution 4.0 License (Namensnennung).</AltText>
    </License>
    <SourceGroup>
      <Meeting>
        <MeetingId>M0654</MeetingId>
        <MeetingSequence>19</MeetingSequence>
        <MeetingCorporation>Deutsche Gesellschaft f&#252;r Phoniatrie und P&#228;daudiologie</MeetingCorporation>
        <MeetingName></MeetingName>
        <MeetingTitle>42. Wissenschaftliche Jahrestagung der Deutschen Gesellschaft f&#252;r Phoniatrie und P&#228;daudiologie (DGPP)</MeetingTitle>
        <MeetingSession>Stimme: Stimmanalyse II</MeetingSession>
        <MeetingCity>Starnberg</MeetingCity>
        <MeetingDate>
          <DateFrom>20260916</DateFrom>
          <DateTo>20260919</DateTo>
        </MeetingDate>
      </Meeting>
    </SourceGroup>
    <ArticleNo>V16</ArticleNo>
  </MetaData>
  <OrigData>
    <Abstract language="en" linked="yes"><Pgraph><Mark1>Background:</Mark1> Hoarseness caused by different voice disorders may exhibit different acoustic signatures, complicating standardized quantitative assessment. Acoustic voice signals recorded during sustained phonation are commonly used for objective assessment of voice quality. Laryngeal high-speed videoendoscopy (HSV) enables simultaneous recording of acoustic signals during phonation. However, the presence of rigid endoscope in the oral cavity results in acoustic signals that may differ from natural voice production. Hence, the goals of this study are to determine an optimal, as short as possible, analysis time interval and analyze the potential of HSV-synchronized acoustic signals for machine learning (ML)-based hoarseness severity estimation.</Pgraph><Pgraph><Mark1>Materials and methods:</Mark1> Two databases containing normal voices, functional and organic voice disorders were constructed. Database D<Subscript>1</Subscript> comprises 824 HSV-synchronized acoustic recordings of sustained vowel &#47;i&#47;, while Database D<Subscript>2</Subscript> includes 804 sustained vowel &#47;a&#47; recordings from speech therapy sessions. Recording segments of 250 ms, 500 ms, and 1000 ms were analyzed. RBH ratings derived from continuous speech served as ground truth. Subjects were categorized into two hoarseness levels (H &#60; 2 vs. H &#8805; 2), while ML-based estimated probability was interpreted as a continuous interval-scaled severity score between 0 and 1. Comprehensive acoustic features were extracted and reduced using ensemble feature selection. ML models of varying complexity (logistic regression, SVM, XGBoost, TabNet) were evaluated across durations.</Pgraph><Pgraph><Mark1>Results:</Mark1> At 1000 ms, Logistic regression for D<Subscript>1</Subscript> and XGBoost for D<Subscript>2</Subscript> achieved the best performance, yielding Spearman rank correlations of 0.62 and 0.75 respectively between predicted severity scores and perceptual ratings. Across models, D<Subscript>2</Subscript> demonstrated consistently higher performance than D<Subscript>1</Subscript>, with mean differences of approximately 10&#37; in accuracy and 8&#37; in ROC-AUC. </Pgraph><Pgraph><Mark1>Conclusion:</Mark1> The lower performance of HSV-synchronized acoustic recordings is likely related to the background noise from the equipment, which may degrade signal quality. Perceptual uncertainty in the ground truth (H &#61; 1 vs. H &#61; 2) may further influence model predictions. HSV-synchronized recordings show potential for reliable hoarseness severity estimation even under phonation constraints, with a 500 ms analysis interval sufficient for clinical assessment of voice quality. </Pgraph></Abstract>
    <TextBlock name="Text" linked="yes">
      <MainHeadline>Text</MainHeadline><SubHeadline>Background</SubHeadline><Pgraph>Hoarseness caused by different voice disorders may exhibit different acoustic signatures, complicating standardized quantitative assessment. Acoustic voice signals recorded during sustained phonation are commonly used for objective assessment of voice quality <TextLink reference="1"></TextLink>. Laryngeal high-speed videoendoscopy (HSV) enables simultaneous recording of acoustic signals during phonation. However, the presence of a rigid endoscope in the oral cavity results in acoustic signals that may differ from natural voice production <TextLink reference="2"></TextLink>. Hence, the goals of this study are to determine an optimal, as short as possible, analysis time interval and analyze the potential of HSV-synchronized acoustic signals for machine learning (ML)-based hoarseness severity estimation.</Pgraph><SubHeadline>Materials and methods</SubHeadline><Pgraph>Two databases containing normal voices, functional and organic voice disorders were constructed. Database <Mark1>D</Mark1><Mark1><Subscript>1</Subscript></Mark1> comprises 824 HSV-synchronized acoustic recordings of sustained vowel &#47;i&#47;, while Database <Mark1>D</Mark1><Mark1><Subscript>2</Subscript></Mark1> includes 804 sustained vowel &#47;a&#47; recordings from speech therapy sessions. Recording segments of 250 ms, 500 ms, and 1000 ms were analyzed. RBH ratings derived from continuous speech served as ground truth <TextLink reference="1"></TextLink>. Subjects were categorized into two hoarseness levels (H &#60; 2 vs. H &#8805; 2), while ML-based estimated probability was interpreted as a continuous interval-scaled severity score between 0 and 1. Comprehensive acoustic features were extracted and reduced using ensemble feature selection. ML models of varying complexity (logistic regression, SVM, XGBoost, TabNet) were evaluated across durations. </Pgraph><SubHeadline>Results</SubHeadline><Pgraph>At 1000 ms, Logistic regression for <Mark1>D</Mark1><Mark1><Subscript>1</Subscript></Mark1> and XGBoost for <Mark1>D</Mark1><Mark1><Subscript>2</Subscript></Mark1> achieved the best performance, yielding Spearman rank correlations of <Mark1>0.62</Mark1> and <Mark1>0.75</Mark1> respectively between predicted severity scores and perceptual ratings. Across models, D2 demonstrated consistently higher performance than D1, with mean differences of approximately 10&#37; in accuracy and 8&#37; in ROC-AUC. </Pgraph><Pgraph>Figure 1 <ImgLink imgNo="1" imgType="figure" />, Figure 2 <ImgLink imgNo="2" imgType="figure" /></Pgraph><SubHeadline>Conclusion</SubHeadline><Pgraph>The lower performance of HSV-synchronized acoustic recordings is likely related to the background noise from the equipment, which may degrade signal quality. Perceptual uncertainty in the ground truth (H &#61; 1 vs. H &#61; 2) may further influence model predictions. Despite these limitations, HSV-synchronized recordings show potential for reliable hoarseness severity estimation even under phonation constraints, with a <Mark1>500 ms</Mark1> analysis interval sufficient for clinical assessment of voice quality.</Pgraph></TextBlock>
    <References linked="yes">
      <Reference refNo="1">
        <RefAuthor>Dejonckere PH</RefAuthor>
        <RefAuthor>Bradley P</RefAuthor>
        <RefAuthor>Clemente P</RefAuthor>
        <RefAuthor>Cornut G</RefAuthor>
        <RefAuthor>Crevier-Buchman L</RefAuthor>
        <RefAuthor>Friedrich G</RefAuthor>
        <RefAuthor>Van De Heyning P</RefAuthor>
        <RefAuthor>Remacle M</RefAuthor>
        <RefAuthor>Woisard V</RefAuthor>
        <RefAuthor> Committee on Phoniatrics of the European Laryngological Society (ELS)</RefAuthor>
        <RefTitle>A basic protocol for functional assessment of voice pathology, especially for investigating the efficacy of (phonosurgical) treatments and evaluating new assessment techniques. Guideline elaborated by the Committee on Phoniatrics of the European Laryngological Society (ELS)</RefTitle>
        <RefYear>2001</RefYear>
        <RefJournal>Eur Arch Otorhinolaryngol</RefJournal>
        <RefPage>77-82</RefPage>
        <RefTotal>Dejonckere PH, Bradley P, Clemente P, Cornut G, Crevier-Buchman L, Friedrich G, Van De Heyning P, Remacle M, Woisard V; Committee on Phoniatrics of the European Laryngological Society (ELS). A basic protocol for functional assessment of voice pathology, especially for investigating the efficacy of (phonosurgical) treatments and evaluating new assessment techniques. Guideline elaborated by the Committee on Phoniatrics of the European Laryngological Society (ELS). Eur Arch Otorhinolaryngol. 2001 Feb;258(2):77-82. DOI: 10.1007&#47;s004050000299</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1007&#47;s004050000299</RefLink>
      </Reference>
      <Reference refNo="2">
        <RefAuthor>Ng ML</RefAuthor>
        <RefAuthor>Bailey RL</RefAuthor>
        <RefTitle>Acoustic changes related to laryngeal examination with a rigid telescope</RefTitle>
        <RefYear>2006</RefYear>
        <RefJournal>Folia Phoniatr Logop</RefJournal>
        <RefPage>353-62</RefPage>
        <RefTotal>Ng ML, Bailey RL. Acoustic changes related to laryngeal examination with a rigid telescope. Folia Phoniatr Logop. 2006;58(5):353-62. DOI: 10.1159&#47;000094569</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1159&#47;000094569</RefLink>
      </Reference>
    </References>
    <Media>
      <Tables>
        <NoOfTables>0</NoOfTables>
      </Tables>
      <Figures>
        <Figure width="832" height="374" format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Figure 1: Model predicted probability scores stratified by ground truth hoarseness ratings </Mark1><Mark1><Mark2>H</Mark2></Mark1><Mark1> for databases </Mark1><Mark1><Mark2>D</Mark2></Mark1><Mark1><Mark2><Subscript>1</Subscript></Mark2></Mark1><Mark1> (Logistic regression) and </Mark1><Mark1><Mark2>D</Mark2></Mark1><Mark1><Mark2><Subscript>2</Subscript></Mark2></Mark1> <Mark1>(XGBoost), with Spearman rank correlation (&#961;) indicated. A regression line is fitted over predictions.</Mark1></Pgraph></Caption>
        </Figure>
        <Figure width="557" height="380" format="png">
          <MediaNo>2</MediaNo>
          <MediaID>2</MediaID>
          <Caption><Pgraph><Mark1>Figure 2: ROC-AUC score on test sets for </Mark1><Mark1><Mark2>D</Mark2></Mark1><Mark1><Mark2><Subscript>1</Subscript></Mark2></Mark1><Mark1> and </Mark1><Mark1><Mark2>D</Mark2></Mark1><Mark1><Mark2><Subscript>2</Subscript></Mark2></Mark1><Mark1> across analysis intervals. Shaded bands represent the standard error. </Mark1></Pgraph></Caption>
        </Figure>
        <NoOfPictures>2</NoOfPictures>
      </Figures>
      <InlineFigures>
        <NoOfPictures>0</NoOfPictures>
      </InlineFigures>
      <Attachments>
        <NoOfAttachments>0</NoOfAttachments>
      </Attachments>
    </Media>
  </OrigData>
</GmsArticle>