<?xml version="1.0" encoding="utf-8"?>
<export-example>
  <doc>
    <id>1852</id>
    <completedYear>2021</completedYear>
    <publishedYear>2021</publishedYear>
    <thesisYearAccepted/>
    <language>eng</language>
    <pageFirst>1079</pageFirst>
    <pageLast>1086</pageLast>
    <pageNumber>1079 - 1086</pageNumber>
    <edition/>
    <issue/>
    <volume>2021</volume>
    <type>conferenceobject</type>
    <publisherName>IEEE</publisherName>
    <publisherPlace/>
    <creatingCorporation/>
    <contributingCorporation/>
    <belongsToBibliography>1</belongsToBibliography>
    <completedDate>--</completedDate>
    <publishedDate>--</publishedDate>
    <thesisDateAccepted>--</thesisDateAccepted>
    <title language="eng">Applying X-Vectors on Pathological Speech After Larynx Removal</title>
    <abstract language="eng">Speaker embeddings extracted from time delayed neural networks (TDNNs) contributed to major recent advancements in speaker recognition and verification. We use an X-Vector system trained on augmented VoxCeleb1 and VoxCeleb2 data to obtain embeddings for pathological speech after total or partial larynx removal. We show that our model is able to effectively distinguish and visualize patient groups when generating embeddings. We further compare various regression models on the task of automatically predicting different perceptual ratings by speech therapists (intelligibility, vocal effort, and overall quality) based on the extracted speaker embeddings. For both patient groups we show Pearson correlations in the range of +0.8; we find that Random Forest and Support Vector Regression produce scores that best resemble the experts' assessments.</abstract>
    <parentTitle language="eng">2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)</parentTitle>
    <identifier type="isbn">978-1-6654-3739-4</identifier>
    <identifier type="doi">10.1109/asru51503.2021.9688278</identifier>
    <enrichment key="opus_doi_flag">true</enrichment>
    <enrichment key="opus_import_data">{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T05:55:18Z","timestamp":1715234118302},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,12,13]]},"DOI":"10.1109\/asru51503.2021.9688278","type":"proceedings-article","created":{"date-parts":[[2022,2,3]],"date-time":"2022-02-03T20:31:00Z","timestamp":1643920260000},"source":"Crossref","is-referenced-by-count":1,"title":["Applying X-Vectors on Pathological Speech After Larynx Removal"],"prefix":"10.1109","author":[{"given":"Ralph","family":"Scheuerer","sequence":"first","affiliation":[{"name":"Technische Hochschule N&amp;#x00FC;rnberg,Germany"}]},{"given":"Tino","family":"Haderlein","sequence":"additional","affiliation":[{"name":"Universit&amp;#x00E4;t Erlangen-N&amp;#x00FC;rnberg,Germany"}]},{"given":"Elmar","family":"Noth","sequence":"additional","affiliation":[{"name":"Universit&amp;#x00E4;t Erlangen-N&amp;#x00FC;rnberg,Germany"}]},{"given":"Tobias","family":"Bocklet","sequence":"additional","affiliation":[{"name":"Technische Hochschule N&amp;#x00FC;rnberg,Germany"}]}],"member":"263","reference":[{"key":"ref33","author":"kumar","year":"2020","journal-title":"Designing neural speaker embeddings with meta learning"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref31","article-title":"MUSAN: A Music, Speech, and Noise Corpus","author":"snyder","year":"2015","journal-title":"ArXiv"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2587"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1159\/000492219"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1044\/1092-4388(2007\/102)"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1044\/jshr.3402.285"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijporl.2020.110191"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1121\/1.1398053"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s00268-003-7107-4"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1159\/000048592"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s00405-003-0681-0"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvoice.2011.04.010"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-620"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.3389\/fninf.2021.578369"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s11136-018-2033-y"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053770"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1002\/lary.29027"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1159\/000266164"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101198"},{"key":"ref5","article-title":"Perceptual evaluation of tra-cheoesophageal speech by naive and experienced judges through the use of semantic differential scales","author":"van as","year":"2003","journal-title":"Journal of Speech Language and Hearing Research"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvoice.2005.09.005"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1044\/jslhr.4303.697"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1308\/147870811X13137608455253"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/S0021-9924(98)00008-2"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.3109\/14015439609099197"},{"key":"ref20","article-title":"Voxceleb: Large-scale speaker verification in the wild","author":"nagrani","year":"2019","journal-title":"Computer Science and Language"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2013-313"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163978"},{"key":"ref24","first-page":"6544","article-title":"Comparison of user models based on gmm-ubm and i-vectors for speech, hand-writing, and gait assessment of parkinson's disease patients","author":"vasquez-correa","year":"0","journal-title":"ICASSP 2020 &amp;#x2014; 2020 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/j.jcomdis.2018.08.002"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1431"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1266"}],"event":{"name":"2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"Cartagena, Colombia","start":{"date-parts":[[2021,12,13]]},"end":{"date-parts":[[2021,12,17]]}},"container-title":["2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9687821\/9687855\/09688278.pdf?arnumber=9688278","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,16]],"date-time":"2022-05-16T20:42:07Z","timestamp":1652733727000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9688278\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,12,13]]},"references-count":33,"URL":"http:\/\/dx.doi.org\/10.1109\/asru51503.2021.9688278","relation":{},"subject":[],"published":{"date-parts":[[2021,12,13]]}}}</enrichment>
    <enrichment key="local_crossrefDocumentType">proceedings-article</enrichment>
    <enrichment key="local_crossrefLicence">https://doi.org/10.15223/policy-029</enrichment>
    <enrichment key="local_import_origin">crossref</enrichment>
    <enrichment key="local_doiImportPopulated">PersonAuthorFirstName_1,PersonAuthorLastName_1,PersonAuthorFirstName_2,PersonAuthorLastName_2,PersonAuthorFirstName_3,PersonAuthorLastName_3,PersonAuthorFirstName_4,PersonAuthorLastName_4,Enrichmentconference_title,Enrichmentconference_place,PublisherName,TitleMain_1,TitleParent_1,CompletedYear,Enrichmentlocal_crossrefLicence</enrichment>
    <enrichment key="conference_title">2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)</enrichment>
    <enrichment key="conference_place">Cartagena, Colombia</enrichment>
    <enrichment key="opus.source">doi-import</enrichment>
    <enrichment key="Reviewstatus">Begutachtet/Reviewed</enrichment>
    <author>Ralph Scheuerer</author>
    <author>Tino Haderlein</author>
    <author>Elmar Nöth</author>
    <author>Tobias Bocklet</author>
    <subject>
      <language>eng</language>
      <type>uncontrolled</type>
      <value>laryngectomy</value>
    </subject>
    <subject>
      <language>eng</language>
      <type>uncontrolled</type>
      <value>intelligibility</value>
    </subject>
    <subject>
      <language>eng</language>
      <type>uncontrolled</type>
      <value>pathological speech</value>
    </subject>
    <subject>
      <language>eng</language>
      <type>uncontrolled</type>
      <value>x-vectors</value>
    </subject>
    <collection role="institutes" number="">Fakultät Informatik</collection>
    <collection role="Forschungsschwerpunkt" number="5">Digitalisierung &amp; Künstliche Intelligenz</collection>
    <collection role="institutes" number="">Zentrum für Künstliche Intelligenz (KIZ)</collection>
  </doc>
</export-example>
