<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xml:lang="en" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
	<front>
		<journal-meta>
			<journal-id journal-id-type="publisher-id">LOQ</journal-id>
			<journal-title-group>
				<journal-title>Loquens</journal-title>
				<abbrev-journal-title abbrev-type="publisher">Loquens</abbrev-journal-title>
			</journal-title-group>
			<issn publication-format="electronic">2386-2637</issn>
			<publisher>
				<publisher-name>Consejo Superior de Investigaciones Cient&#xed;ficas</publisher-name>
			</publisher>
		</journal-meta>
		<article-meta>
			<article-id pub-id-type="publisher-id">loquens.2025.e116</article-id>
			<article-id pub-id-type="doi">10.3989/loquens.2025.e116</article-id>
			<article-categories>
				<subj-group subj-group-type="heading">
					<subject>Articles</subject>
				</subj-group>
			</article-categories>
			<title-group>
				<article-title>Evaluation of German Automatic Speech Recognition solutions in the context of speech and language therapy support of people with aphasia</article-title>
				<trans-title-group xml:lang="es">
					<trans-title>Evaluaci&#xf3;n de soluciones de reconocimiento autom&#xe1;tico del habla en alem&#xe1;n en el contexto del apoyo a la terapia del lenguaje para las personas con afasia</trans-title>
				</trans-title-group>
			</title-group>
			<contrib-group>
				<contrib contrib-type="author">
					<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-3819-0949</contrib-id>
					<name>
						<surname>Rykova</surname>
						<given-names>Eugenia</given-names>
					</name>
					<email xlink:href="eugenryk@uef.fi">eugenryk@uef.fi</email>
					<aff id="aff-1-e116">
						<institution content-type="university">University of Eastern Finland</institution>
						<country country="FI">Finland</country>
					</aff>
					<aff id="aff-2-e116">
						<institution content-type="university">Technical University of Applied Sciences TH Wildau</institution>
						<country country="DE">Germany</country>
					</aff>
					<aff id="aff-3-e116">
						<institution content-type="university">Catholic University Eichst&#xe4;tt-Ingolstadt</institution>
						<country country="DE">Germany</country>
					</aff>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/" vocab-term="Conceptualization">Conceptualization</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/" vocab-term="Data curation">Data curation</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/" vocab-term="Formal analysis">Formal analysis</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/" vocab-term="Investigation">Investigation</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/" vocab-term="Methodology">Methodology</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/" vocab-term="Visualization">Visualization</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/" vocab-term="Writing &#x2013; original draft">Writing &#x2013; original draft</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/" vocab-term="Writing &#x2013; review &amp; editing">Writing &#x2013; review &amp; editing</role>
				</contrib>
				<contrib contrib-type="author">
					<contrib-id contrib-id-type="orcid">https://orcid.org/0009-0001-8451-8290</contrib-id>
					<name>
						<surname>Walther</surname>
						<given-names>Mathias</given-names>
					</name>
					<email xlink:href="mathias.walther@th-wildau.de">mathias.walther@th-wildau.de</email>
					<aff id="aff-4-e116">
						<institution content-type="university">Technical University of Applied Sciences TH Wildau</institution>
						<country country="DE">Germany</country>
					</aff>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/" vocab-term="Conceptualization">Conceptualization</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/" vocab-term="Funding acquisition">Funding acquisition</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/" vocab-term="Project administration">Project administration</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/" vocab-term="Resources">Resources</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/" vocab-term="Supervision">Supervision</role>
					<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/" vocab-term="Writing &#x2013; review &amp; editing">Writing &#x2013; review &amp; editing</role>
				</contrib>
			</contrib-group>
			<pub-date pub-type="epub">
				<day>30</day>
				<month>12</month>
				<year>2025</year>
			</pub-date>
			<pub-date pub-type="collection">
				<day>30</day>
				<month>12</month>
				<year>2025</year>
			</pub-date>
			<volume>12</volume>
			<elocation-id>e116</elocation-id>
			<pub-history>
				<event>
					<event-desc>Submitted</event-desc>
					<date date-type="received">
						<day>11</day>
						<month>06</month>
						<year>2024</year>
					</date>
				</event>
				<event>
					<event-desc>Accepted</event-desc>
					<date date-type="accepted">
						<day>31</day>
						<month>07</month>
						<year>2024</year>
					</date>
				</event>
				<event>
					<event-desc>Published online</event-desc>
					<date date-type="pub">
						<day>30</day>
						<month>06</month>
						<year>2025</year>
					</date>
				</event>
			</pub-history>
			<self-uri xlink:href="https://loquens.revistas.csic.es/index.php/loquens/article/view/XXXX/XXXX"/>
			<abstract>
				<title>Abstract</title>
				<p>Those who suffer from aphasia benefit from digital speech and language therapy solutions, and automatic speech recognition (ASR) has been already used for giving feedback on the correctness of the answers in naming exercises. AphaDIGITAL application is to provide German-speaking users with detailed feedback on phonemic/phonetic and semantic errors, based on automatic speech and language processing. For this purpose, open-source ASR solutions for German were evaluated on different corpora of atypical speech, including two small datasets with aphasic speech samples. Character error rate, the number of precisely recognized items and empty outputs served as evaluation metrics. The four selected models are generally robust to the deteriorated condition of speech and audio quality and consistently outperform commercial models in atypical speech recognition. Applying error acceptance threshold, additional use of phonemic error rate, and other valuable insights for ASR implementation in aphaDIGITAL are discussed.</p>
			</abstract>
			<trans-abstract xml:lang="es">
				<title>Resumen</title>
				<p>Aquellos que sufren de afasia se benefician de soluciones digitales de terapia del lenguaje, y el reconocimiento autom&#xe1;tico del habla (RAH) ya se ha utilizado para proporcionar retroalimentaci&#xf3;n sobre la correcci&#xf3;n de las respuestas en ejercicios de denominaci&#xf3;n. La aplicaci&#xf3;n aphaDIGITAL debe ofrecer a los usuarios germanohablantes una realimentaci&#xf3;n detallada sobre errores fon&#xe9;micos/fon&#xe9;ticos y sem&#xe1;nticos, basada en el procesamiento autom&#xe1;tico del habla y lenguaje. A tal fin, se evaluaron soluciones del RAH de c&#xf3;digo abierto para el alem&#xe1;n con diferentes corpus de habla at&#xed;pica, incluidos dos peque&#xf1;os conjuntos de datos con muestras de habla af&#xe1;sica. Se utilizaron como m&#xe9;tricas de evaluaci&#xf3;n la tasa de errores en los caracteres, el n&#xfa;mero de elementos precisamente reconocidos y las salidas vac&#xed;as. Los cuatro modelos seleccionados son generalmente robustos frente al deteriorado estado del habla y la calidad del audio, y consistentemente superan a los modelos comerciales en el reconocimiento del habla at&#xed;pica. Se discuten la aplicaci&#xf3;n del umbral de aceptaci&#xf3;n de errores, el uso adicional de la tasa de errores en fonemas y otros conocimientos valiosos para la implementaci&#xf3;n del RAH en aphaDIGITAL.</p>
			</trans-abstract>
			<kwd-group>
				<kwd>aphasia</kwd>
				<kwd>automatic speech recognition</kwd>
				<kwd>speech and language therapy</kwd>
				<kwd>digital health</kwd>
			</kwd-group>
			<kwd-group xml:lang="es">
				<kwd>afasia</kwd>
				<kwd>reconocimiento autom&#xe1;tico del habla</kwd>
				<kwd>terapia del lenguaje</kwd>
				<kwd>software m&#xe9;dico</kwd>
			</kwd-group>
			<funding-group id="fug-1-e116">
				<award-group id="awg-1-e116">
					<funding-source id="fus-1-e116">German Federal Ministry of Education and Research</funding-source>
					<award-id id="awi-1-e116">03WIR3108A</award-id>
				</award-group>
				<funding-statement>AphaDIGITAL project is sponsored by German Federal Ministry of Education and Research under funding code 03WIR3108A via the TDG innovation ecosystem (Translationsregion f&#xfc;r digitale Gesundheitsversorgung [Translational region for digital healthcare]) and &#x201e;WIR! &#x2013; Wandel durch Innovation in der Region&#x201d; [Change through innovation in the region] program.</funding-statement>
			</funding-group>
			<counts>
				<fig-count count="0"/>
				<table-count count="6"/>
				<equation-count count="1"/>
				<ref-count count="87"/>
				<page-count count="0"/>
			</counts>
		</article-meta>
	</front>
	<body>
		<sec id="sec-1-e116" sec-type="intro">
			<label>1.</label>
			<title>Introduction</title>
			<p>Using automatic speech processing tools, including Automatic Speech Recognition (ASR), in speech and language pathology has become increasingly popular in the last two decades. Such tools provide valuable help in diagnostics and therapy when used by a speech and language therapy (SLT) practitioner (<xref ref-type="bibr" rid="ref-38-e116">Keshet, 2018</xref>), on the one hand, and on the other hand, contribute to more autonomous healthcare (<xref ref-type="bibr" rid="ref-31-e116">H&#xf6;nig &amp; N&#xf6;th, 2016</xref>). In particular, mobile applications to support SLT are becoming popular (<xref ref-type="bibr" rid="ref-22-e116">Griffel <italic>et al.</italic>, 2019</xref>; <xref ref-type="bibr" rid="ref-80-e116">Vaezipour <italic>et al.</italic>, 2020</xref>).</p>
			<p>Aphasia is a language disorder that occurs after completed language development due to brain damage, which in 80% of the cases is caused by a stroke. Every year, aphasia affects 25,000 new patients in Germany (<xref ref-type="bibr" rid="ref-84-e116">Wiehage &amp; Heide, 2016</xref>). SLT improves functional communication of those who suffer from aphasia, with certain benefits brought by high intensity and duration of the therapy (<xref ref-type="bibr" rid="ref-10-e116">Bhogal and Speechley, 2003</xref>; <xref ref-type="bibr" rid="ref-12-e116">Brady <italic>et al.</italic>, 2016</xref>). In reality, not everyone has enough access to extensive or even sufficient SLT because of geographical remoteness, lack of specialists, or other reasons. Nevertheless, in-person therapy can be efficiently supplemented with digital therapy solutions used independently (<xref ref-type="bibr" rid="ref-81-e116">van de Sandt-Koenderman, 2011</xref>; <xref ref-type="bibr" rid="ref-18-e116">Des Roches and Kiran, 2017</xref>; <xref ref-type="bibr" rid="ref-13-e116">Braley <italic>et al.</italic>, 2021</xref>), and oral speech production exercises with adequate feedback are highly desired by users (<xref ref-type="bibr" rid="ref-40-e116">Kitzing <italic>et al.</italic>, 2009</xref>). <xref ref-type="bibr" rid="ref-80-e116">Vaezipour <italic>et al.</italic> (2020)</xref> have analyzed SLT apps for English-speaking people with aphasia (PWA), and from those 70 meeting the eligibility criteria only 24% offer exercises on perceiving and producing oral speech, and while some of them provide automatic feedback, it does not necessarily have high quality. </p>
			<p>The aphaDIGITAL project (<xref ref-type="bibr" rid="ref-76-e116">TDG, 2021</xref>) focuses on developing a mobile application for German-speaking PWA that is to provide detailed feedback with the help of speech and text processing in a variety of exercises (cf. <xref ref-type="bibr" rid="ref-22-e116">Griffel <italic>et al.</italic>, 2019</xref>). There are different requirements for the speech recognition solution(s) in the framework of the aphaDIGITAL app. First, it must provide certain phonetic precision (reflecting acoustic modeling), in other words, be able to produce output independently (at least partially) from the existing vocabulary and spelling of the language, or pronunciation and language models in terms of ASR (<xref ref-type="bibr" rid="ref-38-e116">Keshet, 2018</xref>). This is needed for the feedback on pronunciation, which incorporates the committed error(s), for example, phoneme deletion or substitution. On the other hand, a pronounced word must be recognized as an existing one (or at least close to the language reality) in order to be passed further in the pipeline for semantic and grammatical analysis (<xref ref-type="bibr" rid="ref-69-e116">Rykova &amp; Walther, 2024a</xref>). The current paper presents the process of evaluating and selecting ASR solutions for the aphaDIGITAL app, answering the following research questions (RQs):</p>
			<list list-type="order" id="lst-1-e116">
				<list-item>
					<p>Which existing ASR solutions are suitable for the task-specific speech of German-speaking PWA?</p>
				</list-item>
				<list-item>
					<p>How do open-source ASR models perform in comparison to commercial solutions?</p>
				</list-item>
				<list-item>
					<p>Which aspects should be considered when implementing an ASR solution for SLT support of PWA?</p>
				</list-item>
			</list>
		</sec>
		<sec id="sec-2-e116">
			<label>2.</label>
			<title>Background</title>
			<sec id="sec-2.1-e116">
				<label>2.1.</label>
				<title>Aphasia speech features in the light of ASR</title>
				<p>Aphasia could be translated as &#x201c;speechlessness&#x201d; from (Ancient) Greek (<xref ref-type="bibr" rid="ref-68-e116">Ryalls, 1984</xref>). It affects all language modalities: reading and listening (comprehension), and speaking and writing (production). There are several typical clinical pictures of the disorder, but some linguistic symptoms can be considered the most noticeable and universal across PWA. Anomia, or word-finding problems, is one of them (<xref ref-type="bibr" rid="ref-9-e116">Benson, 1988</xref>). This deficit is treated with naming-oriented semantic exercises, which can be automated with the help of ASR (see Section 2.2.2). </p>
				<p>Aachen Aphasia Test (AAT) (<xref ref-type="bibr" rid="ref-32-e116">Huber, 1983</xref>) is considered the gold standard in Germany for aphasia diagnosis. Assessment at phonetic and phonemic levels includes a mostly qualitative description of fluidity, vocalization, preciseness, speed, and rhythm (articulation and prosody level), and a quantitative evaluation of the phonemic structure correctness: added, dropped, repeated, or shuffled phonemes in speech output. </p>
				<p>Contrary to motor speech disorders, phonetic and phonemic errors in aphasia (a language disorder) are mostly inconsistent and unpredictable. Aphasia can be, however, comorbid with motor speech disorders: apraxia of speech (AOS), and, much less frequently, dysarthria. In AOS, the neurologic mechanisms for motor planning and programming are affected, while the motor function itself remains intact (<xref ref-type="bibr" rid="ref-62-e116">Qualls, 2011</xref>). That results in phonemic structure distortions, speech disfluency, prolonged sound duration, and other prosodic/temporal abnormalities (<xref ref-type="bibr" rid="ref-45-e116">Le <italic>et al.</italic>, 2016</xref>, see also <xref ref-type="bibr" rid="ref-83-e116">Wambaugh <italic>et al.</italic>, 1996</xref>). Dysarthria manifests itself in weakness, slowness, poor coordination, and restricted and imprecise movements of muscles that take part in oral speech production. That causes low intelligibility of speech in general, and such particular deviations as, for example, slower speech rate, strained phonation, irregular articulation, and reduction or deletion of word-initial consonants (<xref ref-type="bibr" rid="ref-14-e116">Caballero Morales and Cox, 2009</xref>; <xref ref-type="bibr" rid="ref-62-e116">Qualls, 2011</xref>). The research on ASR for dysarthric speakers is actively ongoing, for example in adapting acoustic models or modeling the errors (<xref ref-type="bibr" rid="ref-27-e116">Gutz, 2022</xref>).</p>
				<p>Aphasia generally affects more men than women, and age is another risk factor for stroke and aphasia (<xref ref-type="bibr" rid="ref-75-e116">Schulz and Werner, 2019</xref>; see also <xref ref-type="bibr" rid="ref-36-e116">Johnson <italic>et al.</italic>, 2022</xref>). Furthermore, older individuals tend to recover from aphasia slower and to a lesser extent. Age per se can influence speech production on various linguistic levels, including acoustics. For example, older individuals speak at a slower speech rate (<xref ref-type="bibr" rid="ref-36-e116">Johnson <italic>et al.</italic>, 2022</xref>). Changes in acoustic features are reflected in poorer ASR performance for older speakers, which might be more drastic for female voices (<xref ref-type="bibr" rid="ref-82-e116">Vipperla <italic>et al.</italic>, 2008</xref>).</p>
			</sec>
			<sec id="sec-2.2-e116">
				<label>2.2.</label>
				<title>ASR in aphasia diagnostics and therapy</title>
				<sec id="sec-2.2.1-e116">
					<label>2.2.1.</label>
					<title>PWA&#x2019;s speech assessment</title>
					<p>Plenty of studies have explored the potential of ASR systems to automatically assess the intelligibility of pathological speech (for reviews see <xref ref-type="bibr" rid="ref-35-e116">Jamal <italic>et al.</italic>, 2017</xref>; <xref ref-type="bibr" rid="ref-38-e116">Keshet, 2018</xref>; <xref ref-type="bibr" rid="ref-2-e116">Adikari <italic>et al.</italic>, 2024</xref>). In general, deteriorated condition of speech, high variability among speakers, and insufficiency of data make it difficult to use ASR for aphasic speech. The corresponding solutions should be dynamic and flexible, in the best-case scenario allowing personal tailoring (<xref ref-type="bibr" rid="ref-81-e116">van de Sandt-Koenderman, 2011</xref>). It must be noted that commercial systems with excellent results in many applications for typical speakers demonstrate poor performance on the material of impaired speech. In its turn, personalized models can reach very high recognition rates for the latter (<xref ref-type="bibr" rid="ref-21-e116">Green <italic>et al.</italic>, 2021</xref>).</p>
					<p>As applied to aphasia, Le and colleagues explore the possibilities of automatic assessment of the continuous speech produced by English speakers with aphasia (<xref ref-type="bibr" rid="ref-43-e116">Le <italic>et al.</italic>, 2016</xref>), the ways of improving the automatic recognition of aphasic speech (<xref ref-type="bibr" rid="ref-45-e116">Le &amp; Provost, 2016</xref>) and consequent detection of phonemic and neologistic paraphasias (<xref ref-type="bibr" rid="ref-44-e116">Le <italic>et al.</italic>, 2017</xref>). <xref ref-type="bibr" rid="ref-78-e116">Torre <italic>et al.</italic> (2021)</xref> set a new benchmark in ASR for PWA in English and provide the first adapted system for Spanish. </p>
					<p>
						<xref ref-type="bibr" rid="ref-46-e116">Lee <italic>et al.</italic> (2016)</xref> evaluate the feasibility and challenges of assessing continuous speech of Cantonese speakers with aphasia with the help of ASR. <xref ref-type="bibr" rid="ref-15-e116">Chatzoudis <italic>et al</italic>. (2022)</xref> propose fine-tuning of ASR models that share cross-lingual speech representation to low-resource languages and present models for detecting aphasia and transcribing PWA&#x2019;s speech for French and Greek. </p>
					<p>Another area of ASR in aphasia assessment includes automatic transcription of the PWA&#x2019;s speech and further analysis of text features, possibly in combination with acoustic features. Such work has been done, for example, on the material of English (<xref ref-type="bibr" rid="ref-20-e116">Fraser <italic>et al.</italic>, 2013</xref>) and Cantonese (<xref ref-type="bibr" rid="ref-60-e116">Qin <italic>et al.</italic>, 2018</xref>; <xref ref-type="bibr" rid="ref-61-e116">Qin <italic>et al.</italic>, 2020</xref>). <xref ref-type="bibr" rid="ref-41-e116">Kohlschein <italic>et al</italic>. (2018)</xref> aim at an automatic version of German AAT (<xref ref-type="bibr" rid="ref-32-e116">Huber, 1983</xref>), which uses acoustic features, phonemic structure, and higher-level linguistic features for diagnosing and classifying aphasia.</p>
				</sec>
				<sec id="sec-2.2.2-e116">
					<label>2.2.2.</label>
					<title>Automatic feedback in naming exercises</title>
					<p>Virtual Therapist for Aphasia Treatment (VIRTHEA) in European Portuguese, introduced in 2011 (<xref ref-type="bibr" rid="ref-58-e116">Pompili <italic>et al.</italic>, 2011</xref>), seems to be the first system that uses ASR (an in-house ASR engine) to process what is said by the user and evaluate whether the answer was correct or incorrect &#x2013; a verification task, in other words. The system focuses on naming exercises to improve the word-retrieval ability. <xref ref-type="bibr" rid="ref-1-e116">Abad <italic>et al.</italic> (2013)</xref> state that since VIRTHEA is assumed to be used by PWA with very low (or none) motor speech deficits, a general acoustic model can be used without retraining. However, PWA&#x2019;s speech may contain a considerable number of hesitations, repetitions, and other disruptive factors that weaken ASR performance. Therefore, a keyword spotting method is proposed in order to verify that a correct word has been pronounced during the analyzed speech segment. The system demonstrates promising results on a corpus of nomination tests from native Portuguese speakers with different types of aphasia, in particular high correlation between human and automatic naming scores, and high word verification rates &#x2013; 82% accuracy on average. VIRTHEA is positively perceived by SLT practitioners and is being updated according to the wishes of the latter (<xref ref-type="bibr" rid="ref-57-e116">Pompili <italic>et al.</italic>, 2020</xref>).</p>
					<p>
						<xref ref-type="bibr" rid="ref-7-e116">Ballard and colleagues (2019)</xref> evaluate an open-source ASR engine to provide binary feedback (correct/incorrect) in a picture-naming task for Australian English and reach a mean accuracy performance of 75%. Naming Utterance Verifier for Aphasia Treatment (NUVA), developed by <xref ref-type="bibr" rid="ref-8-e116">Barbera <italic>et al.</italic> (2021)</xref> for British English, reaches a mean accuracy of 89.5% with a smaller range than VIRTHEA and the system for Australian English (see <xref ref-type="bibr" rid="ref-8-e116">Barbera <italic>et al.</italic>, 2021</xref> for a detailed comparison). In NUVA, a word pronounced by a user with aphasia is compared to two recordings of healthy speakers and classified as correct or incorrect using a verification threshold. Different threshold calibration methods are applied to a proposed ASR model with a phone error rate of 15.85%. This model consequently outperforms Google Cloud Platform speech-to-text service used as an ASR baseline.</p>
					<p>Several research teams work on SLT solutions for German-speaking PWA with ASR-based feedback. Nevertheless, to the best of the authors&#x2019; knowledge, there are currently no such apps in active use. <xref ref-type="bibr" rid="ref-47-e116">Lin <italic>et al.</italic> (2022)</xref> report 83% recognition accuracy of target words pronounced by PWA in the research for <xref ref-type="bibr" rid="ref-51-e116">neolexon Aphasie-App (2023)</xref> and propose that an SLT specialist should posteriorly analyze the problem cases. Dietmar Bothe, project manager of <xref ref-type="bibr" rid="ref-4-e116">aphavox (2020)</xref>, presents the app with automatic recognition of PWA&#x2019;s speech and corresponding feedback in an interview (<xref ref-type="bibr" rid="ref-28-e116">Halling, 2023</xref>). <xref ref-type="bibr" rid="ref-65-e116">RehaLingo (2023)</xref> seeks to combine several speech recognizers and model possible erroneous inputs (<xref ref-type="bibr" rid="ref-30-e116">Hirsch <italic>et al.</italic>, 2023</xref>). LingoTalk (<xref ref-type="bibr" rid="ref-48-e116">LingoLab, 2020</xref>) exploits built-in iOS or Android ASR software in naming exercises and reaches 98% accuracy with typical speech, but there is no data on PWA&#x2019;s speech (<xref ref-type="bibr" rid="ref-52-e116">Netzebandt <italic>et al.</italic>, 2022</xref>).</p>
				</sec>
			</sec>
		</sec>
		<sec id="sec-3-e116" sec-type="materials|methods">
			<label>3.</label>
			<title>Materials and methods</title>
			<sec id="sec-3.1-e116">
				<label>3.1.</label>
				<title>Process and models overview</title>
				<p>In the present research, the ASR selection process consisted of several steps. First, more than 50 open-source ASR solutions, including models available from <xref ref-type="bibr" rid="ref-3-e116">Alpha Cephei (2022)</xref>, Mozilla Deepspeech (<xref ref-type="bibr" rid="ref-86-e116">Xu <italic>et al.</italic>, 2020</xref>), and via <xref ref-type="bibr" rid="ref-33-e116">Hugging Face (2022)</xref>, were screened for further suitability (<xref ref-type="bibr" rid="ref-71-e116">Rykova <italic>et al.</italic>, 2022</xref>). The screening procedure was also applied to the commercial models. Next, 13 selected open-source models were evaluated with a considerable amount of atypical speech data. They are presented in <xref ref-type="table" rid="taw-1-e116">Table 1</xref>. Eleven models were accessed via Hugging Face framework, and ims_0 and ims_35 are modified versions of the original model with language model (lm) weights set to 0 and 0.35, respectively.</p>
				<table-wrap id="taw-1-e116">
					<label>Table 1</label>
					<caption>
						<title>Thirteen open-source ASR models evaluated after the initial screening.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col/>
							<col/>
						</colgroup>
						<thead>
							<tr>
								<th align="center">Model name in the current paper</th>
								<th align="center">Author(s)</th>
								<th align="center">Description given by the author(s) of the model</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="center">andrew</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-50-e116">McDowell, 2022</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLS-R-1B model (<xref ref-type="bibr" rid="ref-6-e116">Babu <italic>et al.</italic>, 2022</xref>) on German Common Voice (CV) 8.0 dataset.</td>
							</tr>
							<tr>
								<td align="center">ims_0 (lm weight = 0)</td>
								<td align="center" rowspan="2">
									<xref ref-type="bibr" rid="ref-17-e116">Denisov and Vu, 2019</xref>
								</td>
								<td align="center" rowspan="2">The original IMS model (lm weight = 0.7) was trained using kaldi German ASR recipe and implemented with ESPnet end-to-end speech recognition toolkit. Datasets: Tuda-De, SWC, M-AILABS, Verbmobil 1 and 2, VoxForge, RVG 1, PhonDat1.</td>
							</tr>
							<tr>
								<td align="center">ims_35 (lm weight = 0.35)</td>
							</tr>
							<tr>
								<td align="center">jonatas53</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-23-e116">Grosman, 2022a</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLSR-53 model (<xref ref-type="bibr" rid="ref-16-e116">Conneau <italic>et al.</italic>, 2021</xref>) on German CV 6.1 dataset.</td>
							</tr>
							<tr>
								<td align="center">jonatas1b</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-24-e116">Grosman, 2022b</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLS-R-1B model on German using CV 8.0, Multilingual TEDx, Multilingual LibriSpeech, and Voxpopuli datasets.</td>
							</tr>
							<tr>
								<td align="center">jsnfly</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-37-e116">Jsnfly, 2022</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLS-R-1B model on German CV 8.0 dataset.</td>
							</tr>
							<tr>
								<td align="center">marcel</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-11-e116">Bischoff, 2022</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLSR-53 model on German using the CV dataset.</td>
							</tr>
							<tr>
								<td align="center">maxidl</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-34-e116">Idahl, 2022</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLSR-53 model on German using the CV dataset.</td>
							</tr>
							<tr>
								<td align="center">mfleck</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-19-e116">Fleck, 2022</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLS-R-300M model (<xref ref-type="bibr" rid="ref-16-e116">Conneau <italic>et al.</italic>, 2021</xref>) on German CV dataset.</td>
							</tr>
							<tr>
								<td align="center">nvidia1</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-54-e116">NVIDIA, 2022a</xref>
								</td>
								<td align="center">A "large" version of Conformer model, trained on several thousand hours of German speech data, NeMo toolkit (<xref ref-type="bibr" rid="ref-42-e116">Kuchaiev <italic>et al.</italic>, 2019</xref>).</td>
							</tr>
							<tr>
								<td align="center">nvidia2</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-55-e116">NVIDIA, 2022b</xref>
								</td>
								<td align="center">A "large" version of Conformer-Transducer model, trained on several thousand hours of German speech data, NeMo toolkit.</td>
							</tr>
							<tr>
								<td align="center">oliver8</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-25-e116">Guhr, 2022a</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLSR-53 on German CV 8.0 dataset.</td>
							</tr>
							<tr>
								<td align="center">oliver9</td>
								<td align="center">
									<xref ref-type="bibr" rid="ref-26-e116">Guhr, 2022b</xref>
								</td>
								<td align="center">Fine-tuned Facebook&#x2019;s Wav2Vec2-XLSR-53 on German CV 9.0 dataset.</td>
							</tr>
						</tbody>
					</table>
				</table-wrap>
				<p>Lastly, the thirteen open-source models and the commercial ones were tested with the PWA&#x2019;s speech. Four commercial models were subject to comparison, namely Fraunhofer German ASR (fr-hofer), European Media Lab transcription service (eml), Google Speech Cloud ASR (google), and IBM Watson ASR (watson). The outputs were obtained via BAS web services, available for academic purposes (<xref ref-type="bibr" rid="ref-39-e116">Kisler <italic>et al.</italic>, 2017</xref>). One may use these services for a limited amount of data only, therefore the commercial models were not evaluated together with the open-source ones at the previous step.</p>
				<p>During the evaluation, a new highly performing ASR model, Whisper (<xref ref-type="bibr" rid="ref-64-e116">Radford <italic>et al.</italic>, 2023</xref>) was released. A screening with PWA&#x2019;s samples showed, however, that this model would not be suitable for aphaDIGITAL purposes because it failed to recognize speech in the given samples. </p>
			</sec>
			<sec id="sec-3.2-e116">
				<label>3.2.</label>
				<title>Datasets</title>
				<p>Due to the requirements of some ASR models, all audio recordings described below were (if necessary) converted to one channel and resampled to 16 kHz. For the screening step, individual recordings were selected from the following German corpora: </p>
				<list list-type="bullet" id="lst-2-e116">
					<list-item>
						<p>speech of cochlear implants (CI) users and normal-hearing speakers from CI Articulation Corpus (<xref ref-type="bibr" rid="ref-53-e116">Neumeyer, 2009</xref>) &#x2013; hereinafter CI corpus, </p>
					</list-item>
					<list-item>
						<p>speech of intoxicated and sober speakers from Alcohol Language Corpus (<xref ref-type="bibr" rid="ref-73-e116">Schiel <italic>et al.</italic>, 2008</xref>) &#x2013; hereinafter ALC corpus, </p>
					</list-item>
					<list-item>
						<p>speech of a person with aphasia from AphasiaBank (<xref ref-type="bibr" rid="ref-49-e116">MacWhinney <italic>et al.</italic>, 2011</xref>), </p>
					</list-item>
					<list-item>
						<p>speech of eight PWA extracted from a YouTube video (<xref ref-type="bibr" rid="ref-66-e116">Rhein-Zeitung, 2018</xref>),</p>
					</list-item>
					<list-item>
						<p>typical speech from PHONDAT2 (<xref ref-type="bibr" rid="ref-29-e116">Hess <italic>et al.</italic>, 1995</xref>) for comparison (cf. <xref ref-type="bibr" rid="ref-85-e116">Wirth and Peinl, 2022</xref>).</p>
					</list-item>
				</list>
				<p>The transcriptions provided together with the audio were used for the recordings from CI, ALC and PHONDAT2 corpora, and AphasiaBank. The YouTube video was transcribed by the speech science students. The annotators followed the principle of phonemic orthography: they transcribed actual pronunciation rather than a standard orthographic form but used the German graphemes as the output form.</p>
				<p>In the absence of necessary data from PWA, test material from other corpora with atypical speech was considered for the main evaluation. Thus, the speech of adult CI users can be characterized by decreased vowel exactness and precision of articulatory movements. It is considered deteriorated, especially with a longer period of deafness or pre-lingual onset, which is also reflected in automatic recognition rates (<xref ref-type="bibr" rid="ref-67-e116">Ruff <italic>et al.</italic>, 2017</xref>; <xref ref-type="bibr" rid="ref-5-e116">Arias-Vergara <italic>et al.</italic>, 2022</xref>). The changes in speech production under intoxicated condition include decreased speech rate and weakened speech motor control, which can be captured by both human perception and digital acoustical analysis (<xref ref-type="bibr" rid="ref-56-e116">Pisoni &amp; Martin, 1989</xref>; <xref ref-type="bibr" rid="ref-77-e116">Tislj&#xe1;r-Szab&#xf3; <italic>et al.</italic>, 2014</xref>). Hence, the selected 13 models were evaluated with the help of material from ALC and CI corpora, which covers female and male speakers of different ages, presented in <xref ref-type="table" rid="taw-2-e116">Table 2</xref>.</p>
				<table-wrap id="taw-2-e116">
					<label>Table 2</label>
					<caption>
						<title>Datasets of atypical speech used for the evaluation of ASR models.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col/>
							<col/>
						</colgroup>
						<thead>
							<tr>
								<th align="center">Dataset name</th>
								<th align="center">Number of elements</th>
								<th align="center">Description</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="center">NA_phrases</td>
								<td align="center">1274</td>
								<td align="center">phrases uttered by sober speakers from ALC corpus</td>
							</tr>
							<tr>
								<td align="center">A_phrases</td>
								<td align="center">1404</td>
								<td align="center">phrases uttered by intoxicated speakers from ALC corpus</td>
							</tr>
							<tr>
								<td align="center">NA_words</td>
								<td align="center">1976</td>
								<td align="center">words, automatically segmented out of the tongue-twisting lists uttered by sober speakers from ALC corpus</td>
							</tr>
							<tr>
								<td align="center">A_words</td>
								<td align="center">2249</td>
								<td align="center">words, automatically segmented out of the tongue-twisting lists uttered by intoxicated speakers from ALC corpus</td>
							</tr>
							<tr>
								<td align="center">NORM_words</td>
								<td align="center">1032</td>
								<td align="center">words, automatically segmented out of the sentences uttered by normal-hearing speakers from CI corpus</td>
							</tr>
							<tr>
								<td align="center">CI_words</td>
								<td align="center">1021</td>
								<td align="center">words, automatically segmented out of the sentences uttered by CI users from CI corpus</td>
							</tr>
						</tbody>
					</table>
				</table-wrap>
				<p>For the last evaluation and comparison step, two datasets with aphasic speech, internally named AvEv and UniSt, were used. AvEv is a small dataset obtained from four PWA who took part in the avatar evaluation experiment (<xref ref-type="bibr" rid="ref-87-e116">Zeuner <italic>et al.</italic>, 2022</xref>). While selecting the correct option in a PC-based picture-naming task, the participants incidentally pronounced the corresponding words. The experiment was videotaped. The audio was extracted from the videos and the words were segmented out, which made a set of 39 single words. It must be, however, kept in mind that the quality of these recordings is low. Besides that, a lot of words were pronounced in a manner deviating from the standard pronunciation (e.g., due to dialectal differences or the presence of aphasia). Two speech science students provided separate annotations (based on the principle of phonemic orthography described above) as alternative ground truth in addition to a standard orthographic form of the target words. UniSt is a dataset comprising 61 words uttered by SLT specialists, and 79 recordings of PWA&#x2019;s responses. The recordings had been made during AAT screening sessions (repetition and picture-naming tasks) with six PWA and were obtained on request from University of Stuttgart Institute for Natural Language Processing, where they are used as learning material in neurolinguistics online tutorial (<xref ref-type="bibr" rid="ref-79-e116">Universit&#xe4;t Stuttgart, 2023</xref>). These recordings were also transcribed by the speech science students, following the principle described above. Phonemic transcriptions were generated automatically with a slightly modified version of Deep Phonemizer (<xref ref-type="bibr" rid="ref-72-e116">Sch&#xe4;fer <italic>et al.</italic>, 2023</xref>). Additionally, an SLT specialist classified the PWA&#x2019;s responses in UniSt dataset as containing no error, a phonemic/phonetic error, or a semantic error. A phonemic/phonetic error was understood as such a deviation in a segmental structure of the word that would result in a transcription distinct from the standard orthographic form. The answers with no error or phonemic/phonetic error were considered semantically acceptable. The SLT specialist also provided finer classification of errors according to the ICF (International Classification of Functioning, Disability and Health) guidelines (<xref ref-type="bibr" rid="ref-74-e116">Schneider <italic>et al.</italic>, 2021</xref>).</p>
			</sec>
			<sec id="sec-3.3-e116">
				<label>3.3.</label>
				<title>Measurements</title>
				<p>Character Error Rate (CER) was the main accuracy metric to evaluate the ASR systems:</p>
				<disp-formula id="dif-1-e116">
					<mml:math id="mml-1-e116">
						<mml:mi>C</mml:mi>
						<mml:mi>E</mml:mi>
						<mml:mi>R</mml:mi>
						<mml:mo>=</mml:mo>
						<mml:mi> </mml:mi>
						<mml:mfrac>
							<mml:mrow>
								<mml:mi>S</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>D</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>I</mml:mi>
							</mml:mrow>
							<mml:mrow>
								<mml:mi>N</mml:mi>
							</mml:mrow>
						</mml:mfrac>
						<mml:mo>=</mml:mo>
						<mml:mi> </mml:mi>
						<mml:mfrac>
							<mml:mrow>
								<mml:mi>S</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>D</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>I</mml:mi>
							</mml:mrow>
							<mml:mrow>
								<mml:mi>S</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>D</mml:mi>
								<mml:mo>+</mml:mo>
								<mml:mi>C</mml:mi>
							</mml:mrow>
						</mml:mfrac>
					</mml:math>
				</disp-formula>
				<def-list id="del-1-e116">
					<def-item>
						<term id="trm-1-e116">S &#x2013;</term>
						<def>
							<p>the number of substitutions,</p>
						</def>
					</def-item>
					<def-item>
						<term id="trm-2-e116">D &#x2013;</term>
						<def>
							<p>the number of deletions,</p>
						</def>
					</def-item>
					<def-item>
						<term id="trm-3-e116">I &#x2013;</term>
						<def>
							<p>the number of insertions,</p>
						</def>
					</def-item>
					<def-item>
						<term id="trm-4-e116">N &#x2013;</term>
						<def>
							<p>the number of characters in the reference (target),</p>
						</def>
					</def-item>
					<def-item>
						<term id="trm-5-e116">C &#x2013;</term>
						<def>
							<p>the number of correct characters.</p>
						</def>
					</def-item>
				</def-list>
				<p>If there are too many substitutions and/or insertions in the ASR transcription, the CER value can be higher than 1 (or 100%). For some comparisons, a normalized CER was calculated. In this case, the total number of substitutions, deletions, and insertions is divided by the maximum length of the sequences in question. CER does not only reflect the performance of an ASR system, but is relevant for granular analysis of impaired speech input. It was calculated separately for each of the evaluation material sets (according to the target phrase/word) and then ranked. The HITS measurement was used to assess the number of precisely recognized words. In word sets, the percentage of empty outputs was also taken into consideration in the evaluation process. Thus, each of the metrics was ranked, and the mean rank was calculated for each model/dataset. CER and HITS (correctly recognized words) were computed with the help of the JiWER Python library (<xref ref-type="bibr" rid="ref-59-e116">Python Software Foundation, 2022</xref>).</p>
				<p>CER values were subject to Student&#x2019;s t-test (datasets from ALC and CI corpora) and Wilcoxon Rank Sum Test (AvEv and UniSt datasets). It was decided not to use any correction for multiple comparisons to decrease the risk of Type II error. In other words, it was more important to detect the difference between the models&#x2019; performance when it was insignificant than miss a significant difference. All the analyses were performed in R (<xref ref-type="bibr" rid="ref-63-e116">R Core Team, 2023</xref>) at 95% confidence.</p>
			</sec>
		</sec>
		<sec id="sec-4-e116" sec-type="results">
			<label>4.</label>
			<title>Results</title>
			<sec id="sec-4.1-e116">
				<label>4.1.</label>
				<title>Model selection</title>
				<p>
					<xref ref-type="bibr" rid="ref-71-e116">Rykova <italic>et al.</italic> (2022)</xref> show some preliminary results of the ASR model screening. <xref ref-type="table" rid="taw-3-e116">Table 3</xref> contains mean CER values (in percent), mean ranks, HITS (H), and empty outputs (E) percentage for each of the 13 models obtained with atypical speech from ALC and CI corpora, CER values for typical and atypical speech are in most cases significantly different. Recognition results on the other datasets can be found in the <xref ref-type="app" rid="app-1-e116">Annex A</xref>. The last two columns present the mean (M) rank for all the datasets used for evaluation and its absolute value (ab), respectively. For words datasets, CER values are given ignoring the missing values. The lowest CER values, the highest ranks and HITS are in bold. The CER values that are significantly lower than the others in a pairwise t-test comparison (p-values &lt; 0.05) are marked with an asterisk.</p>
				<table-wrap id="taw-3-e116">
					<label>Table 3</label>
					<caption>
						<title>ASR results for 13 models obtained with atypical speech from ALC and CI corpora.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col span="3"/>
							<col span="4"/>
							<col span="4"/>
							<col span="2"/>
						</colgroup>
						<thead>
							<tr>
								<th align="center" rowspan="3">Model</th>
								<th align="center" colspan="3">A_phrases </th>
								<th align="center" colspan="4">A_words </th>
								<th align="center" colspan="4">CI_words </th>
								<th align="center" colspan="2">all datasets </th>
							</tr>
							<tr>
								<th align="center" rowspan="2">CER</th>
								<th align="center" rowspan="2">H</th>
								<th align="center" rowspan="2">M rank</th>
								<th align="center" rowspan="2">CER</th>
								<th align="center" rowspan="2">H</th>
								<th align="center" rowspan="2">E</th>
								<th align="center" rowspan="2">M rank</th>
								<th align="center" rowspan="2">CER</th>
								<th align="center" rowspan="2">H</th>
								<th align="center" rowspan="2">E</th>
								<th align="center" rowspan="2">M rank</th>
								<th align="center" colspan="2">rank </th>
							</tr>
							<tr>
								<th align="center">M</th>
								<th align="center">ab</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="center">andrew</td>
								<td align="center">7.4</td>
								<td align="center">64.3</td>
								<td align="center">11.6</td>
								<td align="center">12.5</td>
								<td align="center">49.4</td>
								<td align="center">0</td>
								<td align="center">4.4</td>
								<td align="center">33.7</td>
								<td align="center">37.6</td>
								<td align="center">2.4</td>
								<td align="center">7.3</td>
								<td align="center">7.0</td>
								<td align="center">9</td>
							</tr>
							<tr>
								<td align="center">ims_0</td>
								<td align="center">9.2</td>
								<td align="center">69.8</td>
								<td align="center">12.1</td>
								<td align="center">23.1</td>
								<td align="center">37.3</td>
								<td align="center">1.2</td>
								<td align="center">10.6</td>
								<td align="center">55.9</td>
								<td align="center">24.0</td>
								<td align="center">26.8</td>
								<td align="center">10.4</td>
								<td align="center">10.4</td>
								<td align="center">12</td>
							</tr>
							<tr>
								<td align="center">ims_35</td>
								<td align="center">8.6</td>
								<td align="center">75.1</td>
								<td align="center">9.1</td>
								<td align="center">25.1</td>
								<td align="center">39.7</td>
								<td align="center">2.8</td>
								<td align="center">11.0</td>
								<td align="center">64.0</td>
								<td align="center">21.4</td>
								<td align="center">38.9</td>
								<td align="center">11.9</td>
								<td align="center">10.4</td>
								<td align="center">13</td>
							</tr>
							<tr>
								<td align="center" style="background: lightgrey;">jonatas53</td>
								<td align="center" style="background: lightgrey;">5.3</td>
								<td align="center" style="background: lightgrey;">76.4</td>
								<td align="center" style="background: lightgrey;">5.6</td>
								<td align="center" style="background: lightgrey;">14.0</td>
								<td align="center" style="background: lightgrey;">44.7</td>
								<td align="center" style="background: lightgrey;">0</td>
								<td align="center" style="background: lightgrey;">5.3</td>
								<td align="center" style="background: lightgrey;">
									<bold>20.9*</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>59.5</bold>
								</td>
								<td align="center" style="background: lightgrey;">0</td>
								<td align="center" style="background: lightgrey;">
									<bold>1.3</bold>
								</td>
								<td align="center" style="background: lightgrey;">4.2</td>
								<td align="center" style="background: lightgrey;">
									<bold>2</bold>
								</td>
							</tr>
							<tr>
								<td align="center">jonatas1b</td>
								<td align="center">5.4</td>
								<td align="center">79.5</td>
								<td align="center">4.4</td>
								<td align="center">16.4</td>
								<td align="center">
									<bold>57.5</bold>
								</td>
								<td align="center">0.3</td>
								<td align="center">6.2</td>
								<td align="center">37.0</td>
								<td align="center">36.9</td>
								<td align="center">4.9</td>
								<td align="center">8.7</td>
								<td align="center">5.6</td>
								<td align="center">6</td>
							</tr>
							<tr>
								<td align="center">jsnfly</td>
								<td align="center">5.8</td>
								<td align="center">72.4</td>
								<td align="center">8.1</td>
								<td align="center">12.5</td>
								<td align="center">
									<bold>58</bold>
								</td>
								<td align="center">0</td>
								<td align="center">
									<bold>2.3</bold>
								</td>
								<td align="center">31.5</td>
								<td align="center">47.0</td>
								<td align="center">0</td>
								<td align="center">4.5</td>
								<td align="center">5.2</td>
								<td align="center">4</td>
							</tr>
							<tr>
								<td align="center">marcel</td>
								<td align="center">6.8</td>
								<td align="center">70.2</td>
								<td align="center">10.0</td>
								<td align="center">13.4</td>
								<td align="center">46.8</td>
								<td align="center">0</td>
								<td align="center">5.0</td>
								<td align="center">23.8</td>
								<td align="center">45.4</td>
								<td align="center">0</td>
								<td align="center">4.0</td>
								<td align="center">7.0</td>
								<td align="center">10</td>
							</tr>
							<tr>
								<td align="center">maxidl</td>
								<td align="center">6.6</td>
								<td align="center">73.2</td>
								<td align="center">8.4</td>
								<td align="center">13.8</td>
								<td align="center">49.6</td>
								<td align="center">0</td>
								<td align="center">4.6</td>
								<td align="center">27.5</td>
								<td align="center">49.9</td>
								<td align="center">0</td>
								<td align="center">3.8</td>
								<td align="center">6.6</td>
								<td align="center">7</td>
							</tr>
							<tr>
								<td align="center" style="background: lightgrey;">mfleck</td>
								<td align="center" style="background: lightgrey;">5.2</td>
								<td align="center" style="background: lightgrey;">77.4</td>
								<td align="center" style="background: lightgrey;">4.1</td>
								<td align="center" style="background: lightgrey;">
									<bold>10.9*</bold>
								</td>
								<td align="center" style="background: lightgrey;">57.5</td>
								<td align="center" style="background: lightgrey;">0</td>
								<td align="center" style="background: lightgrey;">2.4</td>
								<td align="center" style="background: lightgrey;">24.6</td>
								<td align="center" style="background: lightgrey;">54.3</td>
								<td align="center" style="background: lightgrey;">0</td>
								<td align="center" style="background: lightgrey;">3.2</td>
								<td align="center" style="background: lightgrey;">
									<bold>2.7</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>1</bold>
								</td>
							</tr>
							<tr>
								<td align="center">nvidia1</td>
								<td align="center">
									<bold>3.8*</bold>
								</td>
								<td align="center">
									<bold>86.9</bold>
								</td>
								<td align="center">2.9</td>
								<td align="center">44.1</td>
								<td align="center">28.2</td>
								<td align="center">19.6</td>
								<td align="center">12.9</td>
								<td align="center">78.7</td>
								<td align="center">17.0</td>
								<td align="center">55.5</td>
								<td align="center">12.8</td>
								<td align="center">8.7</td>
								<td align="center">11</td>
							</tr>
							<tr>
								<td align="center" style="background: lightgrey;">nvidia2</td>
								<td align="center" style="background: lightgrey;">
									<bold>3.7*</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>87.0</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>1.3</bold>
								</td>
								<td align="center" style="background: lightgrey;">21.2</td>
								<td align="center" style="background: lightgrey;">55.6</td>
								<td align="center" style="background: lightgrey;">7.4</td>
								<td align="center" style="background: lightgrey;">8.7</td>
								<td align="center" style="background: lightgrey;">65.0</td>
								<td align="center" style="background: lightgrey;">31.4</td>
								<td align="center" style="background: lightgrey;">27.2</td>
								<td align="center" style="background: lightgrey;">10.7</td>
								<td align="center" style="background: lightgrey;">6.7</td>
								<td align="center" style="background: lightgrey;">8</td>
							</tr>
							<tr>
								<td align="center">oliver8</td>
								<td align="center">5.7</td>
								<td align="center">71.7</td>
								<td align="center">8.1</td>
								<td align="center">12.7</td>
								<td align="center">52.3</td>
								<td align="center">0</td>
								<td align="center">3.8</td>
								<td align="center">23.6</td>
								<td align="center">54.6</td>
								<td align="center">0</td>
								<td align="center">2.7</td>
								<td align="center">5.5</td>
								<td align="center">5</td>
							</tr>
							<tr>
								<td align="center" style="background: lightgrey;">oliver9</td>
								<td align="center" style="background: lightgrey;">5.2</td>
								<td align="center" style="background: lightgrey;">77.3</td>
								<td align="center" style="background: lightgrey;">5.3</td>
								<td align="center" style="background: lightgrey;">12.1</td>
								<td align="center" style="background: lightgrey;">52.2</td>
								<td align="center" style="background: lightgrey;">0</td>
								<td align="center" style="background: lightgrey;">4.1</td>
								<td align="center" style="background: lightgrey;">24.7</td>
								<td align="center" style="background: lightgrey;">58.1</td>
								<td align="center" style="background: lightgrey;">0.1</td>
								<td align="center" style="background: lightgrey;">4.4</td>
								<td align="center" style="background: lightgrey;">4.3</td>
								<td align="center" style="background: lightgrey;">
									<bold>3</bold>
								</td>
							</tr>
						</tbody>
					</table>
					<table-wrap-foot>
						<fn id="twf-1-e116">
							<p>CER &#x2013; character error rate (in percent); H &#x2013; HITS percentage; E &#x2013; empty outputs percentage; M &#x2013; mean value; ab &#x2013; absolute value.</p>
						</fn>
						<fn id="twf-2-e116">
							<p>The lowest CER values, the highest ranks and HITS are in bold. The CER values that are significantly lower are marked with an asterisk.</p>
						</fn>
						<fn id="twf-3-e116">
							<p>The models selected for the current app after the evaluation are marked with grey.</p>
						</fn>
					</table-wrap-foot>
				</table-wrap>
				<p>
					<xref ref-type="table" rid="taw-4-e116">Table 4</xref> contains mean CER values (in percent), HITS (H), and empty outputs (E) percentage for the 13 open-source models and 4 commercial ones obtained with AvEv and UniSt datasets. For the open-source models, the mean (M) rank per dataset is given for the comparison among them only, and the absolute (ab) rank value is given for the comparisons among all 17. The last column presents the absolute rank among the open-source and commercial models for PWA&#x2019;s speech from AvEv and UniSt datasets. For the AvEv dataset, manual transcriptions differ equally from the orthographic target (normalized CER = 26% and 25%) and achieve a 17% normalized CER in comparison to each other. However, there are no statistical differences between CER values, obtained in the three comparisons: ASR output vs target and ASR output vs two manual transcriptions. To avoid any bias, the CER values obtained from comparisons with target orthographic transcriptions are considered. For the UniSt dataset, manual transcriptions achieve a 4% normalized CER in comparison to each other, and there are no significant differences in CER values obtained in the comparisons of ASR output vs manual transcriptions, so the quantitative results are presented for one of the manual transcriptions only. The results for speech therapists&#x2019; and PWA&#x2019;s speech are treated separately, as the CER values differ significantly. The best results among open-source models are marked in bold.</p>
				<p>Three models with the highest ranks, namely jonatas53, mfleck, and oliver9, are selected as those providing phonetic level granularity, on the one hand, and robust to degraded speech and audio quality, on the other. Additionally, nvidia2 is selected as the model that is able to recognize words close to language reality (i.e., in accordance with pronunciation and language models). They are marked with grey in <xref ref-type="table" rid="taw-3-e116">Table 3</xref> and <xref ref-type="table" rid="taw-4-e116">Table 4</xref>. Some further details on the performance of these models can be found in <xref ref-type="bibr" rid="ref-70-e116">Rykova and Walther (2024b)</xref>.</p>
				<table-wrap id="taw-4-e116">
					<label>Table 4</label>
					<caption>
						<title>ASR results for 13 open-source and 4 commercial models obtained with AvEv and UniSt datasets.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col span="5"/>
							<col span="5"/>
							<col span="5"/>
							<col/>
						</colgroup>
						<thead>
							<tr>
								<th align="center" rowspan="3">Model </th>
								<th align="center" colspan="5">AvEv </th>
								<th align="center" colspan="5">UniSt therapist </th>
								<th align="center" colspan="5">UniSt PWA </th>
								<th align="center">all PWA</th>
							</tr>
							<tr>
								<th align="center" rowspan="2">CER</th>
								<th align="center" rowspan="2">H </th>
								<th align="center" rowspan="2">E</th>
								<th align="center" colspan="2">rank </th>
								<th align="center" rowspan="2">CER</th>
								<th align="center" rowspan="2">H </th>
								<th align="center" rowspan="2">E</th>
								<th align="center" colspan="2">rank </th>
								<th align="center" rowspan="2">CER </th>
								<th align="center" rowspan="2">H</th>
								<th align="center" rowspan="2">E</th>
								<th align="center" colspan="2">rank</th>
								<th align="center" rowspan="2">ab rank</th>
							</tr>
							<tr>
								<th align="center">M</th>
								<th align="center">ab</th>
								<th align="center">M</th>
								<th align="center">ab</th>
								<th align="center">M</th>
								<th align="center">ab</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="justify">andrew</td>
								<td align="right">65.6</td>
								<td align="right">5.1</td>
								<td align="right">2.6</td>
								<td align="right">6.0</td>
								<td align="right">8</td>
								<td align="right">38.2</td>
								<td align="right">16.4</td>
								<td align="right">0</td>
								<td align="right">6.0</td>
								<td align="right">10</td>
								<td align="right">49.1</td>
								<td align="right">6.3</td>
								<td align="right">0</td>
								<td align="right">3.3</td>
								<td align="right">4</td>
								<td align="right">5</td>
							</tr>
							<tr>
								<td align="justify">ims_0</td>
								<td align="right">58.4</td>
								<td align="right">2.6</td>
								<td align="right">79.5</td>
								<td align="right">8.0</td>
								<td align="right">12</td>
								<td align="right">29.8</td>
								<td align="right">19.7</td>
								<td align="right">29.5</td>
								<td align="right">8.0</td>
								<td align="right">14</td>
								<td align="right">62.8</td>
								<td align="right">3.2</td>
								<td align="right">30.4</td>
								<td align="right">11</td>
								<td align="right">15</td>
								<td align="right">14</td>
							</tr>
							<tr>
								<td align="justify">ims_35</td>
								<td align="right">86.5</td>
								<td align="right">0</td>
								<td align="right">89.7</td>
								<td align="right">12.3</td>
								<td align="right">16</td>
								<td align="right">29.1</td>
								<td align="right">21.3</td>
								<td align="right">42.6</td>
								<td align="right">7.3</td>
								<td align="right">13</td>
								<td align="right">63.3</td>
								<td align="right">5.3</td>
								<td align="right">38.0</td>
								<td align="right">11</td>
								<td align="right">15</td>
								<td align="right">16</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">jonatas 53</td>
								<td align="right" style="background: lightgrey;">67.8</td>
								<td align="right" style="background: lightgrey;">7.7</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">4.3</td>
								<td align="right" style="background: lightgrey;">
									<bold>3</bold>
								</td>
								<td align="right" style="background: lightgrey;">41.3</td>
								<td align="right" style="background: lightgrey;">24.6</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">5.0</td>
								<td align="right" style="background: lightgrey;">4</td>
								<td align="right" style="background: lightgrey;">58.5</td>
								<td align="right" style="background: lightgrey;">6.3</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">4</td>
								<td align="right" style="background: lightgrey;">6</td>
								<td align="right" style="background: lightgrey;">4</td>
							</tr>
							<tr>
								<td align="justify">jonatas 1b</td>
								<td align="right">62.2</td>
								<td align="right">7.7</td>
								<td align="right">2.6</td>
								<td align="right">4.7</td>
								<td align="right">6</td>
								<td align="right">43.5</td>
								<td align="right">31.1</td>
								<td align="right">0</td>
								<td align="right">4.7</td>
								<td align="right">4</td>
								<td align="right">58.5</td>
								<td align="right">
									<bold>12.6</bold>
								</td>
								<td align="right">0</td>
								<td align="right">
									<bold>2</bold>
								</td>
								<td align="right">
									<bold>2</bold>
								</td>
								<td align="right">
									<bold>2</bold>
								</td>
							</tr>
							<tr>
								<td align="justify">jsnfly</td>
								<td align="right">98.2</td>
								<td align="right">0</td>
								<td align="right">0</td>
								<td align="right">8.7</td>
								<td align="right">13</td>
								<td align="right">42.7</td>
								<td align="right">23.0</td>
								<td align="right">0</td>
								<td align="right">5.7</td>
								<td align="right">7</td>
								<td align="right">54.4</td>
								<td align="right">8.4</td>
								<td align="right">0</td>
								<td align="right">3.3</td>
								<td align="right">5</td>
								<td align="right">8</td>
							</tr>
							<tr>
								<td align="justify">marcel</td>
								<td align="right">70.9</td>
								<td align="right">2.6</td>
								<td align="right">2.6</td>
								<td align="right">8.3</td>
								<td align="right">11</td>
								<td align="right">52.2</td>
								<td align="right">4.9</td>
								<td align="right">0</td>
								<td align="right">9.0</td>
								<td align="right">15</td>
								<td align="right">68.7</td>
								<td align="right">1.1</td>
								<td align="right">0</td>
								<td align="right">9</td>
								<td align="right">12</td>
								<td align="right">13</td>
							</tr>
							<tr>
								<td align="justify">maxidl</td>
								<td align="right">64.9</td>
								<td align="right">2.6</td>
								<td align="right">10.3</td>
								<td align="right">7.7</td>
								<td align="right">9</td>
								<td align="right">39.7</td>
								<td align="right">19.7</td>
								<td align="right">0</td>
								<td align="right">5.7</td>
								<td align="right">7</td>
								<td align="right">62.3</td>
								<td align="right">4.2</td>
								<td align="right">1.3</td>
								<td align="right">9.3</td>
								<td align="right">12</td>
								<td align="right">12</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">mfleck</td>
								<td align="right" style="background: lightgrey;">54.0</td>
								<td align="right" style="background: lightgrey;">17.9</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">
									<bold>1.7</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>28.0</bold>
								</td>
								<td align="right" style="background: lightgrey;">31.1</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">
									<bold>1.3</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>45.1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>12.6</bold>
								</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">
									<bold>1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>1</bold>
								</td>
							</tr>
							<tr>
								<td align="justify">nvidia1</td>
								<td align="right">74.1</td>
								<td align="right">5.1</td>
								<td align="right">41.0</td>
								<td align="right">9.3</td>
								<td align="right">14</td>
								<td align="right">34.4</td>
								<td align="right">29.5</td>
								<td align="right">4.9</td>
								<td align="right">6.0</td>
								<td align="right">7</td>
								<td align="right">60.6</td>
								<td align="right">10.5</td>
								<td align="right">8.9</td>
								<td align="right">6.7</td>
								<td align="right">7</td>
								<td align="right">11</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">nvidia2</td>
								<td align="right" style="background: lightgrey;">
									<bold>53.1</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>20.5</bold>
								</td>
								<td align="right" style="background: lightgrey;">20.5</td>
								<td align="right" style="background: lightgrey;">4.0</td>
								<td align="right" style="background: lightgrey;">
									<bold>2</bold>
								</td>
								<td align="right" style="background: lightgrey;">39.9</td>
								<td align="right" style="background: lightgrey;">
									<bold>32.8</bold>
								</td>
								<td align="right" style="background: lightgrey;">6.6</td>
								<td align="right" style="background: lightgrey;">6.7</td>
								<td align="right" style="background: lightgrey;">11</td>
								<td align="right" style="background: lightgrey;">63.3</td>
								<td align="right" style="background: lightgrey;">10.5</td>
								<td align="right" style="background: lightgrey;">12.7</td>
								<td align="right" style="background: lightgrey;">8</td>
								<td align="right" style="background: lightgrey;">10</td>
								<td align="right" style="background: lightgrey;">6</td>
							</tr>
							<tr>
								<td align="justify">oliver8</td>
								<td align="right">66.4</td>
								<td align="right">5.1</td>
								<td align="right">0</td>
								<td align="right">4.7</td>
								<td align="right">6</td>
								<td align="right">43.9</td>
								<td align="right">19.7</td>
								<td align="right">0</td>
								<td align="right">7.3</td>
								<td align="right">12</td>
								<td align="right">63.6</td>
								<td align="right">6.3</td>
								<td align="right">0</td>
								<td align="right">6.7</td>
								<td align="right">9</td>
								<td align="right">7</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">oliver9</td>
								<td align="right" style="background: lightgrey;">69.2</td>
								<td align="right" style="background: lightgrey;">10.3</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">4.3</td>
								<td align="right" style="background: lightgrey;">
									<bold>3</bold>
								</td>
								<td align="right" style="background: lightgrey;">38.9</td>
								<td align="right" style="background: lightgrey;">21.3</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">4.7</td>
								<td align="right" style="background: lightgrey;">
									<bold>3</bold>
								</td>
								<td align="right" style="background: lightgrey;">58.9</td>
								<td align="right" style="background: lightgrey;">10.5</td>
								<td align="right" style="background: lightgrey;">0</td>
								<td align="right" style="background: lightgrey;">3.3</td>
								<td align="right" style="background: lightgrey;">
									<bold>3</bold>
								</td>
								<td align="right" style="background: lightgrey;">
									<bold>3</bold>
								</td>
							</tr>
							<tr>
								<td align="justify">eml</td>
								<td align="right">n/a</td>
								<td align="right">0</td>
								<td align="right">100</td>
								<td align="right"> </td>
								<td align="right">17</td>
								<td align="right">65.8</td>
								<td align="right">4.9</td>
								<td align="right">67.2</td>
								<td align="right"> </td>
								<td align="right">17</td>
								<td align="right">73.8</td>
								<td align="right">1.1</td>
								<td align="right">77.2</td>
								<td align="right"> </td>
								<td align="right">17</td>
								<td align="right">17</td>
							</tr>
							<tr>
								<td align="justify">fr-hofer</td>
								<td align="right">56.6</td>
								<td align="right">
									<bold>23.1</bold>
								</td>
								<td align="right">23.1</td>
								<td align="right"> </td>
								<td align="right">
									<bold>3</bold>
								</td>
								<td align="right">38.6</td>
								<td align="right">
									<bold>41.0</bold>
								</td>
								<td align="right">6.6</td>
								<td align="right"> </td>
								<td align="right">4</td>
								<td align="right">67.1</td>
								<td align="right">9.5</td>
								<td align="right">12.7</td>
								<td align="right"> </td>
								<td align="right">11</td>
								<td align="right">9</td>
							</tr>
							<tr>
								<td align="justify">google</td>
								<td align="right">
									<bold>53.7</bold>
								</td>
								<td align="right">2.6</td>
								<td align="right">84.6</td>
								<td align="right"> </td>
								<td align="right">10</td>
								<td align="right">
									<bold>26.3</bold>
								</td>
								<td align="right">39.3</td>
								<td align="right">21.3</td>
								<td align="right"> </td>
								<td align="right">
									<bold>2</bold>
								</td>
								<td align="right">50.1</td>
								<td align="right">9.5</td>
								<td align="right">38.0</td>
								<td align="right"> </td>
								<td align="right">8</td>
								<td align="right">10</td>
							</tr>
							<tr>
								<td align="justify">watson</td>
								<td align="right">78.1</td>
								<td align="right">0</td>
								<td align="right">66.7</td>
								<td align="right"> </td>
								<td align="right">15</td>
								<td align="right">43.2</td>
								<td align="right">18.0</td>
								<td align="right">36.1</td>
								<td align="right"> </td>
								<td align="right">16</td>
								<td align="right">60.0</td>
								<td align="right">6.3</td>
								<td align="right">45.6</td>
								<td align="right"> </td>
								<td align="right">12</td>
								<td align="right">15</td>
							</tr>
						</tbody>
					</table>
					<table-wrap-foot>
						<fn id="twf-6-e116">
							<p>CER &#x2013; character error rate (in percent); H &#x2013; HITS percentage; E &#x2013; empty outputs percentage; M &#x2013; mean value; ab &#x2013; absolute value.</p>
						</fn>
						<fn id="twf-4-e116">
							<p>The lowest CER values, the highest ranks and HITS are in bold.</p>
						</fn>
						<fn id="twf-5-e116">
							<p>The models selected for the current app after the evaluation are marked with grey.</p>
						</fn>
					</table-wrap-foot>
				</table-wrap>
			</sec>
			<sec id="sec-4.2-e116">
				<label>4.2.</label>
				<title>Open-source vs commercial models</title>
				<p>The results of the screening phase (<xref ref-type="bibr" rid="ref-71-e116">Rykova <italic>et al.</italic>, 2022</xref>) demonstrate that although google, fr-hofer, and, to some extent, watson models show top results in precise word recognition, this performance drops on atypical speech, which includes both speech in deteriorated condition and unusual phrases uttered by typical speakers. Besides that, the mean CER values of commercial models are noticeably higher than those of open-source ones. </p>
				<p>From the evaluation of the AvEv dataset, the performance of the fr-hofer model seems to be the best among the commercial models and comparable to the top open-source models. It also has the highest percentage of HITS among all models and a relatively low CER value, yet this holds true for about three-quarters of the initial data only (non-empty output). Although the mean CER value for the google model also is relatively low, it only accounts for less than 16% of the initial data, and the number of HITS is in the lowest range. </p>
				<p>The performance of the fr-hofer and google models on therapists&#x2019; speech from the UniSt dataset is at the top: among all the models, they have the highest number of HITS and the mean CER for google is the lowest (but on about 80% of the data only). The results change extremely when the models deal with PWA&#x2019;s speech. Thus, both the mean CER value and the percentage of empty outputs for google almost double, and the number of HITS decreases more than four times. The drop in the fr-hofer model performance is similar (its mean CER value increases 1.7 times). The mean CER values of the four selected open-source models increase approximately 1.5 times; the highest numbers of HITS (obtained with mfleck and nvidia2) decrease 2.5-3 times; and the empty output of nvidia2 doubles, while the other three selected models produce results on all the data. Furthermore, none of the commercial models recognizes distorted pronunciations as the human transcribers, producing HITS only on canonical transcriptions of existing words, in distinction to jonatas53, mfleck, and oliver9.</p>
			</sec>
			<sec id="sec-4.3-e116">
				<label>4.3.</label>
				<title>Evaluating the models on PWA&#x2019;s speech</title>
				<p>The lowest mean CER value for the whole AvEv dataset &#x2013; 54% (ranging from 0 to 125%, standard deviation SD = 34.6%) &#x2013; is achieved with the mfleck model. To compare, a mean CER value of 53.1% (ranging from 0 to 150%, SD = 42%) is achieved with nvidia2, but on 80% of the data (20% is empty output). If this value is taken as a threshold for accepting ASR output text as correct, the total number of the words accepted by any of the four selected models reaches 28, which is 72% of the given dataset.</p>
				<p>The lowest mean CER values for both parts of the UniSt dataset are also reached with the mfleck model: 28% on therapists&#x2019; speech (ranging from 0 to 117%, SD = 29.5%), and 45.1% on PWA&#x2019;s speech (ranging from 0 to 150%, SD = 32.1%). Most of the HITS are obtained on the orthographic form of existing words, even when these represent erroneous speech production. For example, a person says &#x201c;twist&#x201d; instead of the target <italic>Zwist</italic> &#x2018;dispute&#x2019;, and &#x201c;twist&#x201d; is recognized correctly by an ASR model (i.e., it&#x2019;s a HIT), but meanwhile <italic>Twist</italic> &#x2018;twist&#x2019; is an actual word, too. Few non-existing forms, representing different degrees of deterioration, are recognized: &#x201c;schwern&#x201d; (target <italic>Stern</italic> &#x2018;star&#x2019;) &#x2013; with oliver9; &#x201c;schweibmaschine&#x201d; (target <italic>Schreibmaschine</italic> &#x2018;typewriter&#x2019;) &#x2013; with jonatas53; &#x201c;schwo&#x201d; (target <italic>Strumpf</italic> &#x2018;stocking&#x2019;), &#x201c;losig&#x201d; (separately pronounced part of target <italic>Verantwortunglosigkeit</italic> &#x2018;irresponsibility&#x2019;), and &#x201c;poloret&#x201d; (target <italic>Lotterie</italic> &#x2018;lottery&#x2019;) &#x2013; with mfleck. Considering transcriptions of both transcribers as a possible target, the four selected models together can recognize 54% HITS on the speech of speech therapists, and 24% on the PWA&#x2019;s speech. </p>
				<p>It must be noted that German orthography principles include several ambiguities, and the same sounds or sound combinations can be transcribed in different ways, which is especially relevant for non-existing words. For example, spellings &#x201c;schweibmaschine&#x201d; (jonatas53) and &#x201c;schwaibmaschine&#x201d; (mfleck) correspond to the same pronunciation; or the initial phoneme /&#x283;/ is transcribed as &#x201c;sch&#x201d; or &#x201c;s&#x201d; in &#x201c;schwern&#x201d; and &#x201c;stern&#x201d;, respectively. Thus, an additional comparison of (automatically generated) phonemic transcriptions (i.e., CER for phonemic transcriptions &#x2013; PER) might be relevant. In the case of the UniSt dataset, such comparison brings one more HIT among non-existing words. In the AvEv dataset, there are two additional words, whose PER is lower than the 54% threshold, which increases the joint acceptance rate to 77%. </p>
				<p>The proposed approach was tested using 54% as a CER/PER threshold to accept PWA&#x2019;s answers as semantically correct (the error rate value is below the threshold) or not. First, the human transcriptions were compared to the corresponding orthographic target as if that were an ideal ASR model. For six recordings, there was a mismatch between human and threshold-based answer acceptance. One error, classified as semantic paraphasia by the SLT specialist (&#x201c;kraftfahr brief&#x201d; vs target <italic>Kraftfahrzeugschein</italic>, which refer to two different documents), would be automatically attributed to a phonemic/phonetic error because more than half of the word is pronounced correctly. Five of them were classified as a phonemic/phonetic error by the SLT specialist (in particular, phonemic conduite d&#x2019;approche &#x2013; &#x201c;approaching&#x201d; the target with self-corrections, and phonemic neologisms &#x2013; phonemic changes in the target that make the latter hard to recognize), but the error rates exceeded 54%. Non-normalized error rates are especially sensitive to insertions, which can be ignored by a human listener to recognize the target word surrounded by extra phonemes (e.g., &#x201c;likurk&#x201d; vs target <italic>Kur</italic> &#x2018;cure&#x2019;). In this case, using normalized error rates could be a solution, which reduces the total number of error classification mismatches to five. It must be noted that for two more semantically accepted phonemic neologisms, the CER/PER values were only slightly below the threshold (50%).</p>
				<p>Furthermore, PWA might utter extra words together with the target word (e.g., articles or false starts). If the target is pronounced correctly, there will be a HIT, but the CER value will be higher than 0, including higher than the acceptance threshold. If there is a deviation in pronunciation or a flaw in automatic recognition, apart from high error rates, there will be no HIT, although an SLT specialist would accept such an answer. Thus, it seems reasonable not only to look for a target word in the uttered phrase but perform a CER/PER analysis for each recognized word of the output. On the other hand, some PWA speak so slowly and carefully/laboriously that the syllables of one word are recognized as separate words. That causes a rise in CER/PER values, which might lead to the rejection of an answer that would be accepted by an SLT specialist. In this case, deleting the spaces between the output chunks and treating the whole output as one word might be useful.</p>
				<p>Taking into consideration the above-mentioned points and using the 54% normalized CER/PER acceptances threshold, ASR outputs of the four selected models were compared to the target words with a subsequent automatic error classification. The comparison of the manual and automatic error classification is displayed in <xref ref-type="table" rid="taw-5-e116">Table 5</xref>. Five cases of error mismatches described above are excluded from the table. The ASR outputs in these cases yield the same results (i.e. mismatches) as the human transcriptions.</p>
				<table-wrap id="taw-5-e116">
					<label>Table 5</label>
					<caption>
						<title>Manual and automatic error classification on the UniSt PWA dataset.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col span="3"/>
						</colgroup>
						<thead>
							<tr>
								<th align="center" rowspan="2">Manual classification</th>
								<th align="center" colspan="3">Automatic classification </th>
							</tr>
							<tr>
								<th align="center">no error</th>
								<th align="center">phonemic/phonetic error</th>
								<th align="center">semantic error</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="justify">no error</td>
								<td align="center">12</td>
								<td align="center">
										20<break/>
										+ 3 gained through error rate normalization<break/>
										+ 1 gained through separate analysis of each word<break/>
										+ 2 gained through deleting the spaces
									</td>
								<td align="center">3</td>
							</tr>
							<tr>
								<td align="justify">phonemic/ phonetic error</td>
								<td align="center">0</td>
								<td align="center">
										16<break/>
										+ 3 gained through separate analysis of each word<break/>
										+ 1 gained through deleting the spaces</td>
								<td align="center">4</td>
							</tr>
							<tr>
								<td align="justify">semantic error</td>
								<td align="center">0</td>
								<td align="center">0</td>
								<td align="center">9</td>
							</tr>
						</tbody>
					</table>
				</table-wrap>
				<p>As one can see, there are no false positives among the automatically classified errors. From the samples accepted by the SLT practitioner, 10.8% are erroneously classified as semantic errors. In 63% of the fully correct answers, ASR models are only able to reach the level of a phonemic/phonetic error, although the answer would be accepted as semantically correct.</p>
			</sec>
		</sec>
		<sec id="sec-5-e116" sec-type="discussion">
			<label>5.</label>
			<title>Discussion</title>
			<sec id="sec-5.1-e116">
				<label>5.1.</label>
				<title>Selection of the open-source models for aphasic speech recognition (RQ 1)</title>
				<p>Based on the experiments with various speech material in German, including speech samples from PWA and other atypical speech, four open-source ASR models are selected for the backend of the aphaDIGITAL app. Three of these models (jonatas53, mfleck, oliver9) are to a certain extent independent of pronunciation and language models and are suitable for phoneme-level pronunciation analysis, while the fourth model (nvidia2) gives only existing orthographic forms as output, which is more suitable for subsequent semantic and grammatical error analysis. The error-analysis component will use every distinct ASR output for comparison to the target. The selected four models present a possibility to be fine-tuned: to PWA&#x2019;s speech or speech of a particular user in a customized version, and to word recognition task rather than continuous speech recognition. </p>
			</sec>
			<sec id="sec-5.2-e116">
				<label>5.2.</label>
				<title>Comparison of the selected open-source models to the commercial ones (RQ2)</title>
				<p>The selected models have consistently outperformed commercial systems in recognition of atypical speech, which in the current paper is primarily reflected in a high amount of empty outputs in the experiments with PWA&#x2019;s speech, and a great contrast between the results on speech therapists&#x2019; and PWA&#x2019;s speech samples from UniSt dataset (cf. <xref ref-type="bibr" rid="ref-21-e116">Green <italic>et al.</italic>, 2021</xref> and <xref ref-type="bibr" rid="ref-85-e116">Wirth and Peinl, 2022</xref>). Results of the screening phase (<xref ref-type="bibr" rid="ref-71-e116">Rykova <italic>et al.</italic>, 2022</xref>) suggest that when Google Speech Cloud ASR and Fraunhofer German ASR recognize the words, they recognize them precisely, but imprecisely recognized words are far from the target (cf. <xref ref-type="bibr" rid="ref-8-e116">Barbera <italic>et al.</italic>, 2021</xref>). Such precise recognition of Google and Fraunhofer models holds true for the data from the AvEv dataset (orthographical form was used as the target in the experiments) and for canonically pronounced words from the UniSt dataset when the target (manual transcription) coincided with the orthographic form of the word. The commercial models cannot recognize distorted pronunciations as such, in contrast to the &#x201c;rule-independent&#x201d; open-source models.</p>
			</sec>
			<sec id="sec-5.3-e116">
				<label>5.3.</label>
				<title>Model performance on PWA&#x2019;s speech (RQ 3)</title>
				<p>The experiments with the AvEv corpus suggest that the selected models together can reach a 72% recognition rate if the CER threshold of acceptance is set to 54%. For the UniSt dataset, manual transcription was used as the target, and the four selected models together transcribed 24% of the words uttered by PWA exactly like human experts (cf. 54% of words uttered by speech therapists in the same dataset), including some non-existing forms. Considering the ambiguities of German orthography principles, incorporating phonemic comparisons and an additional PER threshold seems to be reasonable for the proposed application. Introducing PER with the same 54% threshold allows increasing the joint acceptance rate on the AvEv dataset to 77%. Further research will be dedicated to exploring the threshold further, evaluating false positives. On the other hand, a number of dialect variations seen in the data suggest the relevance of incorporating the knowledge on systematic dialect changes into the pipeline (cf. <xref ref-type="bibr" rid="ref-58-e116">Pompili <italic>et al.</italic>, 2011</xref>).</p>
				<p>The variability of answers produced by PWA suggests the following ASR implementation concerns. First, to overcome the fact that there might be other words in the answer besides the target, the ASR output should be segmented into separate words for further analysis. To reduce the adverse effect of insertions further, the CER/PER threshold should be used for a normalized comparison, in other words, the distance between orthographic/phonemic transcriptions should be divided not by the length of the target, but by the length of the longest word in the comparison. Finally, due to PWA&#x2019;s laborious speech production, ASR output could be considered as one word, removing the spaces between the segments (if applicable).</p>
				<p>The proposed threshold-based acceptance approach has proven to be valid on most of the manual transcriptions of PWA&#x2019;s speech samples from the UniSt dataset. Thus, it has not worked for conduite d&#x2019;approche type of error, for 40% of the semantically acceptable phonemic neologisms, and for semantic paraphasia as part of a compound word. The transcriptions of ASR models yield the same results on these samples. Detecting and further analysis of these types of errors is a subject for further research. </p>
				<p>On the rest of the UniSt PWA data, the four selected models reach 90.3% acceptance accuracy, with 100% specificity and 89.2% sensitivity, which is one of the top results among the existing applications (see Section 2.2.2). However, more than half of PWA&#x2019;s fully correct answers were recognized as containing a phonemic/phonetic error due to the audio quality and flaws of ASR models. Working on the models&#x2019; improvement and testing them with a designated device are seen as the next steps.</p>
			</sec>
		</sec>
		<sec id="sec-6-e116" sec-type="conclusions">
			<label>6.</label>
			<title>Conclusions</title>
			<p>The paper describes the evaluation of open-source German ASR solutions for further use in a mobile SLT application. As a result, four open-source models have been selected. These models fulfill the suitability requirements for both phoneme-level pronunciation analysis and subsequent semantic and grammatical error analysis. They also outperform commercial models in atypical speech recognition, including audio recordings of low quality. </p>
			<p>Using 54% as a normalized error acceptance threshold for orthographic/phonemic transcriptions, analyzing ASR output segment per segment, on the one hand, and as one word with no spaces, on the other, allows reaching promising results in the experiments with aphasic speech data. Improving ASR models&#x2019; performance (e.g. combining several models in a model ensemble or speaker adaptation techniques), making the approach more sensitive to error types, and implementing and evaluating the whole speech analysis pipeline are foreseen for the following stages of the project.</p>
		</sec>
	</body>
	<back>
		<app-group id="appg-1-e116">
			<app id="app-1-e116">
				<title>ANNEX A</title>
				<table-wrap id="taw-6-e116">
					<caption>
						<title>Recognition results for 13 open-source models on NA_phrases, NA_words, and NORM_words datasets.</title>
					</caption>
					<table>
						<colgroup>
							<col/>
							<col span="3"/>
							<col span="4"/>
							<col span="4"/>
						</colgroup>
						<thead>
							<tr>
								<th align="justify" rowspan="2">Model</th>
								<th align="center" colspan="3">NA_phrases </th>
								<th align="center" colspan="4">NA_words </th>
								<th align="center" colspan="4">NORM_words </th>
							</tr>
							<tr>
								<th align="center">CER</th>
								<th align="center">HITS</th>
								<th align="center">M rank</th>
								<th align="center">CER</th>
								<th align="center">HITS</th>
								<th align="center">empty</th>
								<th align="center">M rank</th>
								<th align="center">CER</th>
								<th align="center">HITS</th>
								<th align="center">empty</th>
								<th align="center">M rank</th>
							</tr>
						</thead>
						<tbody>
							<tr>
								<td align="justify">andrew</td>
								<td align="center">6.0</td>
								<td align="center">66.5</td>
								<td align="center">11.3</td>
								<td align="center">12.2</td>
								<td align="center">49.7</td>
								<td align="center"> </td>
								<td align="center">5.1</td>
								<td align="center">24.6</td>
								<td align="center">26.6</td>
								<td align="center">1.5</td>
								<td align="center">7.7</td>
							</tr>
							<tr>
								<td align="justify">ims_0</td>
								<td align="center">7.3</td>
								<td align="center">71.1</td>
								<td align="center">12.3</td>
								<td align="center">21.8</td>
								<td align="center">39.0</td>
								<td align="center">1.3</td>
								<td align="center">10.3</td>
								<td align="center">52.4</td>
								<td align="center">21.8</td>
								<td align="center">20.3</td>
								<td align="center">10.5</td>
							</tr>
							<tr>
								<td align="justify">ims_35</td>
								<td align="center">6.8</td>
								<td align="center">77.0</td>
								<td align="center">9.1</td>
								<td align="center">23.1</td>
								<td align="center">41.9</td>
								<td align="center">2.5</td>
								<td align="center">10.7</td>
								<td align="center">61.4</td>
								<td align="center">19.9</td>
								<td align="center">31.6</td>
								<td align="center">11.7</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">jonatas53</td>
								<td align="center" style="background: lightgrey;">3.9</td>
								<td align="center" style="background: lightgrey;">79.2</td>
								<td align="center" style="background: lightgrey;">5.4</td>
								<td align="center" style="background: lightgrey;">12.9</td>
								<td align="center" style="background: lightgrey;">48.1</td>
								<td align="center" style="background: lightgrey;"> </td>
								<td align="center" style="background: lightgrey;">5.2</td>
								<td align="center" style="background: lightgrey;">
									<bold>12.9</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>47.1</bold>
								</td>
								<td align="center" style="background: lightgrey;"> </td>
								<td align="center" style="background: lightgrey;">
									<bold>1.3</bold>
								</td>
							</tr>
							<tr>
								<td align="justify">jonatas1b</td>
								<td align="center">4.0</td>
								<td align="center">82.5</td>
								<td align="center">4.3</td>
								<td align="center">17.6</td>
								<td align="center">56.1</td>
								<td align="center">0.1</td>
								<td align="center">7.3</td>
								<td align="center">31.8</td>
								<td align="center">29.7</td>
								<td align="center">3.5</td>
								<td align="center">8.5</td>
							</tr>
							<tr>
								<td align="justify">jsnfly</td>
								<td align="center">4.3</td>
								<td align="center">76.4</td>
								<td align="center">7.1</td>
								<td align="center">12.2</td>
								<td align="center">
									<bold>58.6</bold>
								</td>
								<td align="center"> </td>
								<td align="center">2.8</td>
								<td align="center">19.9</td>
								<td align="center">31.2</td>
								<td align="center"> </td>
								<td align="center">4.6</td>
							</tr>
							<tr>
								<td align="justify">marcel</td>
								<td align="center">5.1</td>
								<td align="center">72.8</td>
								<td align="center">9.5</td>
								<td align="center">12.0</td>
								<td align="center">50.1</td>
								<td align="center"> </td>
								<td align="center">4.6</td>
								<td align="center">18.6</td>
								<td align="center">35.9</td>
								<td align="center"> </td>
								<td align="center">3.4</td>
							</tr>
							<tr>
								<td align="justify">maxidl</td>
								<td align="center">5.2</td>
								<td align="center">76.1</td>
								<td align="center">9.1</td>
								<td align="center">13.0</td>
								<td align="center">51.6</td>
								<td align="center">0.1</td>
								<td align="center">6.8</td>
								<td align="center">19.3</td>
								<td align="center">35.9</td>
								<td align="center"> </td>
								<td align="center">3.8</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">mfleck</td>
								<td align="center" style="background: lightgrey;">4.0</td>
								<td align="center" style="background: lightgrey;">80.3</td>
								<td align="center" style="background: lightgrey;">5.0</td>
								<td align="center" style="background: lightgrey;">
									<bold>10.7</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>58.7</bold>
								</td>
								<td align="center" style="background: lightgrey;"> </td>
								<td align="center" style="background: lightgrey;">
									<bold>2.3</bold>
								</td>
								<td align="center" style="background: lightgrey;">15.6</td>
								<td align="center" style="background: lightgrey;">39.3</td>
								<td align="center" style="background: lightgrey;"> </td>
								<td align="center" style="background: lightgrey;">3.2</td>
							</tr>
							<tr>
								<td align="justify">nvidia1</td>
								<td align="center">
									<bold>2.3</bold>
								</td>
								<td align="center">
									<bold>90.2</bold>
								</td>
								<td align="center">
									<bold>2.0</bold>
								</td>
								<td align="center">43.0</td>
								<td align="center">27.8</td>
								<td align="center">17.9</td>
								<td align="center">12.9</td>
								<td align="center">70.2</td>
								<td align="center">8.7</td>
								<td align="center">60.7</td>
								<td align="center">12.7</td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">nvidia2</td>
								<td align="center" style="background: lightgrey;">
									<bold>2.5</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>89.7</bold>
								</td>
								<td align="center" style="background: lightgrey;">
									<bold>1.8</bold>
								</td>
								<td align="center" style="background: lightgrey;">21.3</td>
								<td align="center" style="background: lightgrey;">57.3</td>
								<td align="center" style="background: lightgrey;">8.1</td>
								<td align="center" style="background: lightgrey;">8.5</td>
								<td align="center" style="background: lightgrey;">56.4</td>
								<td align="center" style="background: lightgrey;">23.3</td>
								<td align="center" style="background: lightgrey;">35.7</td>
								<td align="center" style="background: lightgrey;">11.0</td>
							</tr>
							<tr>
								<td align="justify">oliver8</td>
								<td align="center">4.6</td>
								<td align="center">74.2</td>
								<td align="center">9.1</td>
								<td align="center">11.5</td>
								<td align="center">54.9</td>
								<td align="center"> </td>
								<td align="center">4.0</td>
								<td align="center">17.0</td>
								<td align="center">41.3</td>
								<td align="center"> </td>
								<td align="center">2.9 </td>
							</tr>
							<tr>
								<td align="justify" style="background: lightgrey;">oliver9</td>
								<td align="center" style="background: lightgrey;">4.0</td>
								<td align="center" style="background: lightgrey;">80.5</td>
								<td align="center" style="background: lightgrey;">5.1</td>
								<td align="center" style="background: lightgrey;">11.0</td>
								<td align="center" style="background: lightgrey;">56.7</td>
								<td align="center" style="background: lightgrey;"> </td>
								<td align="center" style="background: lightgrey;"> 3.0</td>
								<td align="center" style="background: lightgrey;">15.5</td>
								<td align="center" style="background: lightgrey;">43.2</td>
								<td align="center" style="background: lightgrey;">0.1</td>
								<td align="center" style="background: lightgrey;">4.3 </td>
							</tr>
						</tbody>
					</table>
					<table-wrap-foot>
						<fn id="twf-8-e116">
							<p>CER &#x2013; character error rate (in percent); empty &#x2013; empty outputs percentage; M &#x2013; mean value.</p>
						</fn>
						<fn id="twf-9-e116">
							<p>The lowest CER values, the highest ranks and HITS are in bold.</p>
						</fn>
						<fn id="twf-10-e116">
							<p>The models selected for the current app after the evaluation are marked with grey.</p>
						</fn>
					</table-wrap-foot>
				</table-wrap>
			</app>
		</app-group>
		<sec id="sec-7-e116" sec-type="data-availability">
			<title>Data availability</title>
			<p>ALC, CI, and PHONDAT2 corpora were downloaded from BAS CLARIN repository (<ext-link ext-link-type="uri" xlink:href="https://clarin.phonetik.uni-muenchen.de/BASRepository/" id="exl-1-e116">https://clarin.phonetik.uni-muenchen.de/BASRepository/</ext-link>) under free access for scientists. </p>
			<p>AphasiaBank (<ext-link ext-link-type="uri" xlink:href="https://aphasia.talkbank.org/" id="exl-2-e116">https://aphasia.talkbank.org/</ext-link>) data was accessed with the permission for research and education purposes.</p>
			<p>The samples of UniSt dataset were obtained with the permission for research and education purposes from the Institute for Natural Language Processing of the University of Stuttgart (<ext-link ext-link-type="uri" xlink:href="https://www2.ims.uni-stuttgart.de/sgtutorial/index.html" id="exl-3-e116">https://www2.ims.uni-stuttgart.de/sgtutorial/index.html</ext-link>). </p>
		</sec>
		<ack>
			<title>Acknowledgments</title>
			<p>Elisabeth Zeuner, Aischa Khader-Lindholz, and Ida Luise Kranzfelder from the Speech Science Department of the Martin-Luther-University Halle-Wittenberg annotated the PWA&#x2019;s data from the YouTube video, and AvEv and UniSt datasets. Elisabeth Zeuner classified the errors of PWA in UniSt dataset.</p>
		</ack>
		<sec id="sec-8-e116">
			<title>Declaration of competing interest</title>
			<p>The authors of this article declare that they have no financial, professional or personal conflicts of interest that could have inappropriately influenced this work. </p>
		</sec>
		<sec id="sec-9-e116" sec-type="apoyo">
			<title>Funding sources</title>
			<p>AphaDIGITAL project is sponsored by German Federal Ministry of Education and Research under funding code 03WIR3108A via the TDG innovation ecosystem (Translationsregion f&#xfc;r digitale Gesundheitsversorgung [Translational region for digital healthcare]) and &#x201e;WIR! &#x2013; Wandel durch Innovation in der Region&#x201d; [Change through innovation in the region] program.</p>
		</sec>
		<sec id="sec-10-e116" sec-type="author-contributions">
			<title>Authorship contribution statement</title>
			<p>Eugenia Rykova: Conceptualization, Data Curation, Formal analysis, Investigation, Methodology, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
			<p>Mathias Walther: Conceptualization, Funding Acquisition, Resources, Project Administration, Supervision, Writing &#x2013; review &amp; editing.</p>
		</sec>
		<ref-list id="refl-1-e116">
			<title>References</title>
			<ref id="ref-1-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Abad</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Pompili</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Costa</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Trancoso</surname>
							<given-names>I.</given-names>
						</name>
						<name>
							<surname>Fonseca</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Leal</surname>
							<given-names>G.</given-names>
						</name>
						<name>
							<surname>Martins</surname>
							<given-names>I. P.</given-names>
						</name>
					</person-group>
					<year>2013</year>
					<article-title>Automatic word naming recognition for an on&#x2010;line aphasia treatment system</article-title>
					<source>Computer Speech &amp; Language</source>
					<volume>27</volume>
					<fpage>1235</fpage>
					<lpage>1248</lpage>
				</element-citation>
			</ref>
			<ref id="ref-2-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Adikari</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Hernandez</surname>
							<given-names>N.</given-names>
						</name>
						<name>
							<surname>Alahakoon</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Rose</surname>
							<given-names>M. L.</given-names>
						</name>
						<name>
							<surname>Pierce</surname>
							<given-names>J. E.</given-names>
						</name>
					</person-group>
					<year>2024</year>
					<article-title>From concept to practice: A scoping review of the application of AI to aphasia diagnosis and management</article-title>
					<source>Disability and Rehabilitation</source>
					<volume>46</volume>
					<issue>7</issue>
					<fpage>1288</fpage>
					<lpage>1297</lpage>
				</element-citation>
			</ref>
			<ref id="ref-3-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Alpha Cephei, Inc.</collab>
					</person-group>
					<year>2022</year>
					<source>Vosk speech recognition toolkit models</source>
					<ext-link ext-link-type="uri" xlink:href="https://alphacephei.com/vosk/models" id="exl-5-e116">https://alphacephei.com/vosk/models</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-4-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Aphavox</collab>
					</person-group>
					<year>2020</year>
					<source>Therapieunterst&#xfc;tzung f&#xfc;r Aphasiker. Eigentraining mit dem Tablet und Feedback durch Spracherkennung</source>
					<comment>Therapy support for people with aphasia. Self&#x2010;training with a tablet and feedback through speech recognition</comment>
					<ext-link ext-link-type="uri" xlink:href="https://aphavox.de/media/201029_aphavox_faltblatt_web.pdf" id="exl-7-e116">https://aphavox.de/media/201029_aphavox_faltblatt_web.pdf</ext-link>
				</element-citation>
			</ref>
			<ref id="ref-5-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Arias-Vergara</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Batliner</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Rader</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Polterauer</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>H&#xf6;gerle</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>M&#xfc;ller</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Orozco-Arroyave</surname>
							<given-names>J.-R.</given-names>
						</name>
						<name>
							<surname>N&#xf6;th</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Schuster</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<article-title>Adult cochlear implant users versus typical hearing persons: An automatic analysis of acoustic&#x2010;prosodic parameters</article-title>
					<source>Journal of Speech, Language, and Hearing Research</source>
					<volume>65</volume>
					<issue>12</issue>
					<fpage>4623</fpage>
					<lpage>4636</lpage>
				</element-citation>
			</ref>
			<ref id="ref-6-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Babu</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Wang</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Tjandra</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Lakhotia</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>Xu</surname>
							<given-names>Q.</given-names>
						</name>
						<name>
							<surname>Goyal</surname>
							<given-names>N.</given-names>
						</name>
						<name>
							<surname>Auli</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>XLS-R: self-supervised cross-lingual speech representation learning at scale</source>
					<conf-name>Interspeech 2022</conf-name>
					<fpage>2278</fpage>
					<lpage>2282</lpage>
				</element-citation>
			</ref>
			<ref id="ref-7-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Ballard</surname>
							<given-names>K. J.</given-names>
						</name>
						<name>
							<surname>Etter</surname>
							<given-names>N. M.</given-names>
						</name>
						<name>
							<surname>Shen</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Monroe</surname>
							<given-names>P.</given-names>
						</name>
						<name>
							<surname>Tien Tan</surname>
							<given-names>C.</given-names>
						</name>
					</person-group>
					<year>2019</year>
					<article-title>Feasibility of automatic speech recognition for providing feedback during tablet-based treatment for apraxia of speech plus aphasia</article-title>
					<source>American Journal of Speech-Language Pathology</source>
					<volume>28</volume>
					<fpage>818</fpage>
					<lpage>834</lpage>
				</element-citation>
			</ref>
			<ref id="ref-8-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Barbera</surname>
							<given-names>D. S.</given-names>
						</name>
						<name>
							<surname>Huckvale</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Fleming</surname>
							<given-names>V.</given-names>
						</name>
						<name>
							<surname>Upton</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Coley-Fisher</surname>
							<given-names>H.</given-names>
						</name>
						<name>
							<surname>Doogan</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Crinion</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<article-title>NUVA: A Naming Utterance Verifier for Aphasia Treatment</article-title>
					<source>Computer Speech &amp; Language</source>
					<volume>69</volume>
					<elocation-id>101221</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-9-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Benson</surname>
							<given-names>D. F.</given-names>
						</name>
					</person-group>
					<year>1988</year>
					<article-title>Anomia in aphasia</article-title>
					<source>Aphasiology</source>
					<volume>2</volume>
					<issue>3-4</issue>
					<fpage>229</fpage>
					<lpage>235</lpage>
				</element-citation>
			</ref>
			<ref id="ref-10-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Bhogal</surname>
							<given-names>S. K.</given-names>
						</name>
						<name>
							<surname>Teasell</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Speechley</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2003</year>
					<article-title>Intensity of aphasia therapy, impact on recovery</article-title>
					<source>Stroke</source>
					<volume>34</volume>
					<fpage>987</fpage>
					<lpage>993</lpage>
				</element-citation>
			</ref>
			<ref id="ref-11-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Bischoff</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>Wav2Vec2-Large-XLSR-53-German</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/marcel/wav2vec2-large-xlsr-53-german" id="exl-9-e116">https://huggingface.co/marcel/wav2vec2-large-xlsr-53-german</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-12-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Brady</surname>
							<given-names>M. C.</given-names>
						</name>
						<name>
							<surname>Kelly</surname>
							<given-names>H.</given-names>
						</name>
						<name>
							<surname>Godwin</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Enderby</surname>
							<given-names>P.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<article-title>Speech and language therapy for aphasia following stroke (Review)</article-title>
					<source>Cochrane Database of Systematic Reviews 2016</source>
					<issue>6</issue>
					<elocation-id>CD000425</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-13-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Braley</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Pierce</surname>
							<given-names>J. S.</given-names>
						</name>
						<name>
							<surname>Saxena</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Oliveira</surname>
							<given-names>E. D.</given-names>
						</name>
						<name>
							<surname>Taraboanta</surname>
							<given-names>L.</given-names>
						</name>
						<name>
							<surname>Anantha</surname>
							<given-names>V.</given-names>
						</name>
						<name>
							<surname>Kiran</surname>
							<given-names>S.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<article-title>A virtual, randomized, control trial of a digital therapeutic for speech, language, and cognitive intervention in post&#x2010;stroke persons with aphasia</article-title>
					<source>Frontiers in Neurology</source>
					<volume>12</volume>
					<elocation-id>626780</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-14-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Caballero Morales</surname>
							<given-names>S. O.</given-names>
						</name>
						<name>
							<surname>Cox</surname>
							<given-names>S. J.</given-names>
						</name>
					</person-group>
					<year>2009</year>
					<article-title>Modelling errors in automatic speech recognition for dysarthric speakers</article-title>
					<source>EURASIP Journal on Advances in Signal Processing</source>
					<volume>2009</volume>
					<elocation-id>308340</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-15-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Chatzoudis</surname>
							<given-names>G.</given-names>
						</name>
						<name>
							<surname>Plitsis</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Stamouli</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Dimou</surname>
							<given-names>A.&#x2013;L.</given-names>
						</name>
						<name>
							<surname>Katsamanis</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Katsouros</surname>
							<given-names>V.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>Zero-shot cross-lingual aphasia detection using automatic speech recognition</source>
					<conf-name>Interspeech 2022</conf-name>
					<fpage>2178</fpage>
					<lpage>2182</lpage>
				</element-citation>
			</ref>
			<ref id="ref-16-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Conneau</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Baevski</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Collobert</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Mohamed</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Auli</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<source>Unsupervised cross-lingual representation learning for speech recognition</source>
					<conf-name>Interspeech 2021</conf-name>
					<fpage>2426</fpage>
					<lpage>2430</lpage>
				</element-citation>
			</ref>
			<ref id="ref-17-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Denisov</surname>
							<given-names>P.</given-names>
						</name>
						<name>
							<surname>Vu</surname>
							<given-names>N.T.</given-names>
						</name>
					</person-group>
					<year>2019</year>
					<article-title>IMS-speech: A speech to text tool</article-title>
					<source>Studientexte zur Sprachkommunikation: Elektronische Sprachsignalverarbeitung 2019</source>
					<fpage>170</fpage>
					<lpage>177</lpage>
				</element-citation>
			</ref>
			<ref id="ref-18-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Des Roches</surname>
							<given-names>C. A.</given-names>
						</name>
						<name>
							<surname>Kiran</surname>
							<given-names>S.</given-names>
						</name>
					</person-group>
					<year>2017</year>
					<article-title>Technology-based rehabilitation to improve communication after acquired brain injury</article-title>
					<source>Frontiers in Neuroscience</source>
					<volume>11</volume>
					<elocation-id>382</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-19-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Fleck</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>Wav2vec2-large-xls-r-300m-german-with-lm</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/mfleck/wav2vec2-large-xls-r-300m-german-with-lm" id="exl-11-e116">https://huggingface.co/mfleck/wav2vec2-large-xls-r-300m-german-with-lm</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-20-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Fraser</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>Rudzicz</surname>
							<given-names>F.</given-names>
						</name>
						<name>
							<surname>Graham</surname>
							<given-names>N.</given-names>
						</name>
						<name>
							<surname>Rochon</surname>
							<given-names>E.</given-names>
						</name>
					</person-group>
					<year>2013</year>
					<source>Automatic speech recognition in the diagnosis of primary progressive aphasia</source>
					<conf-name>SLPAT 2013, 4th Workshop on Speech and Language Processing for Assistive Technologies</conf-name>
					<fpage>47</fpage>
					<lpage>54</lpage>
				</element-citation>
			</ref>
			<ref id="ref-21-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Green</surname>
							<given-names>J. R.</given-names>
						</name>
						<name>
							<surname>MacDonald</surname>
							<given-names>R. L.</given-names>
						</name>
						<name>
							<surname>Jiang</surname>
							<given-names>P.-P.</given-names>
						</name>
						<name>
							<surname>Cattiau</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Heywood</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Cave</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Tomanek</surname>
							<given-names>K.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<source>Automatic speech recognition of disordered speech: Personalized models outperforming human listeners on short phrases</source>
					<conf-name>Interspeech 2021</conf-name>
					<fpage>4778</fpage>
					<lpage>4782</lpage>
				</element-citation>
			</ref>
			<ref id="ref-22-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Griffel</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Leinweber</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Spelter</surname>
							<given-names>B.</given-names>
						</name>
						<name>
							<surname>Roddam</surname>
							<given-names>H.</given-names>
						</name>
					</person-group>
					<year>2019</year>
					<article-title>Patient-centred design of aphasia therapy apps: a scoping review</article-title>
					<source>Aphasie und verwandte Gebiete | Aphasie et domaines associ&#xe9;s</source>
					<volume>46</volume>
					<issue>2</issue>
					<fpage>6</fpage>
					<lpage>21</lpage>
				</element-citation>
			</ref>
			<ref id="ref-23-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Grosman</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2022a</year>
					<source>Fine-tuned XLSR-53 large model for speech recognition in German</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/jonatasgrosman/wav2vec2-large-xlsr-53-german" id="exl-13-e116">https://huggingface.co/jonatasgrosman/wav2vec2-large-xlsr-53-german</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-24-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Grosman</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2022b</year>
					<source>Fine-tuned XLS-R 1B model for speech recognition in German</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/jonatasgrosman/wav2vec2-xls-r-1b-german" id="exl-15-e116">https://huggingface.co/jonatasgrosman/wav2vec2-xls-r-1b-german</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-25-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Guhr</surname>
							<given-names>O.</given-names>
						</name>
					</person-group>
					<year>2022a</year>
					<source>Wav2vec2-large-xlsr-53-german-cv8-dropout</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/oliverguhr/wav2vec2-large-xlsr-53-german-cv8" id="exl-17-e116">https://huggingface.co/oliverguhr/wav2vec2-large-xlsr-53-german-cv8</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-26-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Guhr</surname>
							<given-names>O.</given-names>
						</name>
					</person-group>
					<year>2022b</year>
					<source>wav2vec2-large-xlsr-53-german-cv9</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/oliverguhr/wav2vec2-large-xlsr-53-german-cv9" id="exl-19-e116">https://huggingface.co/oliverguhr/wav2vec2-large-xlsr-53-german-cv9</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-27-e116">
				<element-citation publication-type="thesis">
					<person-group person-group-type="author">
						<name>
							<surname>Gutz</surname>
							<given-names>S. E.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>Automatic speech recognition as a clinical tool: Implications for speech assessment and intervention</source>
					<comment content-type="degree">Doctoral dissertation</comment>
					<publisher-name>Harvard University</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-28-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Halling</surname>
							<given-names>S.</given-names>
						</name>
					</person-group>
					<year>2023</year>
					<article-title>Ein Trainingsprogramm f&#xfc;r AphasiepatientInnen: aphavox [Training software for aphasia patients: aphavox]</article-title>
					<source>Logos</source>
					<volume>31</volume>
					<issue>1</issue>
					<fpage>46</fpage>
					<lpage>48</lpage>
				</element-citation>
			</ref>
			<ref id="ref-29-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Hess</surname>
							<given-names>W.J.</given-names>
						</name>
						<name>
							<surname>Kohler</surname>
							<given-names>K.J.</given-names>
						</name>
						<name>
							<surname>Tillmann</surname>
							<given-names>H.-G.</given-names>
						</name>
					</person-group>
					<year>1995</year>
					<source>The Phondat-verbmobil speech corpus</source>
					<conf-name>4th European Conference on Speech Communication and Technology (Eurospeech 1995)</conf-name>
					<fpage>863</fpage>
					<lpage>866</lpage>
				</element-citation>
			</ref>
			<ref id="ref-30-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>Hirsch</surname>
							<given-names>H.-G.</given-names>
						</name>
						<name>
							<surname>Neumann</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Tiggelkamp</surname>
							<given-names>Y.</given-names>
						</name>
						<name>
							<surname>Fiorista</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Knecht</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Schnitzler</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Frieg</surname>
							<given-names>H.</given-names>
						</name>
					</person-group>
					<year>2023</year>
					<chapter-title>RehaLingo - towards a speech training system for aphasia</chapter-title>
					<person-group person-group-type="editor">
						<name>
							<surname>Draxler</surname>
							<given-names>C.</given-names>
						</name>
					</person-group>
					<source>Proceedings of the 34th conference Elektronische Sprachsignalverarbeitung</source>
					<fpage>134</fpage>
					<lpage>141</lpage>
					<publisher-name>TUDpress</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-31-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>H&#xf6;nig</surname>
							<given-names>F.</given-names>
						</name>
						<name>
							<surname>N&#xf6;th</surname>
							<given-names>E.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<chapter-title>Automatische Sprachverarbeitung in der Sprachtherapie [Automatic signal processing in speech and language therapy]</chapter-title>
					<person-group person-group-type="editor">
						<name>
							<surname>Bilda</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>M&#xfc;hlhaus</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Ritterfeld</surname>
							<given-names>U.</given-names>
						</name>
					</person-group>
					<source>Neue Technologien in der Sprachtherapie [New Technologies in Speech and Language Therapy]</source>
					<fpage>173</fpage>
					<lpage>184</lpage>
					<publisher-name>Thieme</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-32-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>Huber</surname>
							<given-names>W.</given-names>
						</name>
					</person-group>
					<year>1983</year>
					<source>Aachener aphasie test (AAT) [Aachen Aphasia Test]</source>
					<publisher-name>Verlag f&#xfc;r Psychologie Hogrefe</publisher-name>
					<publisher-loc>G&#xf6;ttingen, Z&#xfc;rich</publisher-loc>
				</element-citation>
			</ref>
			<ref id="ref-33-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Hugging Face, Inc.</collab>
					</person-group>
					<year>2022</year>
					<source>The AI community building the future</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co" id="exl-21-e116">https://huggingface.co</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-12-12">December 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-34-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Idahl</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>Wav2Vec2-Large-XLSR-53-German</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/maxidl/wav2vec2-large-xlsr-german" id="exl-23-e116">https://huggingface.co/maxidl/wav2vec2-large-xlsr-german</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-12-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-35-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Jamal</surname>
							<given-names>N.</given-names>
						</name>
						<name>
							<surname>Shanta</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Mahmud</surname>
							<given-names>F.</given-names>
						</name>
						<name>
							<surname>Sha'abani</surname>
							<given-names>M. N.</given-names>
						</name>
					</person-group>
					<year>2017</year>
					<source>Automatic speech recognition (ASR) based approach for speech therapy of aphasic patients: A review</source>
					<conf-name>AIP Conference Proceedings, 1883</conf-name>
					<issue>1</issue>
					<elocation-id>020028</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-36-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Johnson</surname>
							<given-names>L.</given-names>
						</name>
						<name>
							<surname>Nemati</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Bonilha</surname>
							<given-names>L.</given-names>
						</name>
						<name>
							<surname>Rorden</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Busby</surname>
							<given-names>N.</given-names>
						</name>
						<name>
							<surname>Basilakos</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Fridriksson</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<article-title>Predictors beyond the lesion: Health and demographic factors associated with aphasia severity</article-title>
					<source>Cortex</source>
					<volume>154</volume>
					<fpage>375</fpage>
					<lpage>389</lpage>
				</element-citation>
			</ref>
			<ref id="ref-37-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<collab>Jsnfly</collab>
					</person-group>
					<year>2022</year>
					<source>XLS-R-1b-DE</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/jsnfly/wav2vec2-xls-r-1b-de-cv8" id="exl-25-e116">https://huggingface.co/jsnfly/wav2vec2-xls-r-1b-de-cv8</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-38-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Keshet</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2018</year>
					<article-title>Automatic speech recognition: A primer for speech-language pathology researchers</article-title>
					<source>International Journal of Speech-Language Pathology</source>
					<volume>20</volume>
					<fpage>599</fpage>
					<lpage>609</lpage>
				</element-citation>
			</ref>
			<ref id="ref-39-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Kisler</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Reichel</surname>
							<given-names>U.</given-names>
						</name>
						<name>
							<surname>Schiel</surname>
							<given-names>F.</given-names>
						</name>
					</person-group>
					<year>2017</year>
					<article-title>Multilingual processing of speech via web services</article-title>
					<source>Computer Speech &amp; Language</source>
					<volume>45</volume>
					<issue>C</issue>
					<fpage>326</fpage>
					<lpage>347</lpage>
				</element-citation>
			</ref>
			<ref id="ref-40-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Kitzing</surname>
							<given-names>P.</given-names>
						</name>
						<name>
							<surname>Maier</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>&#xc5;hlander</surname>
							<given-names>V. L.</given-names>
						</name>
					</person-group>
					<year>2009</year>
					<article-title>Automatic speech recognition (ASR) and its use as a tool for assessment or therapy of voice, speech, and language disorders</article-title>
					<source>Logopedics Phoniatrics Vocology</source>
					<volume>34</volume>
					<issue>2</issue>
					<fpage>91</fpage>
					<lpage>96</lpage>
				</element-citation>
			</ref>
			<ref id="ref-41-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Kohlschein</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Klischies</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Meisen</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Schuller</surname>
							<given-names>B. W.</given-names>
						</name>
						<name>
							<surname>Werner</surname>
							<given-names>C. J.</given-names>
						</name>
					</person-group>
					<year>2018</year>
					<source>Automatic processing of clinical aphasia data collected during diagnosis sessions: Challenges and prospects</source>
					<person-group person-group-type="editor">
						<name>
							<surname>Kokkinakis</surname>
							<given-names>D.</given-names>
						</name>
					</person-group>
					<conf-name>Proceedings of the LREC 2018 Workshop &#x201c;Resources and ProcessIng of linguistic, para-linguistic andextra-linguistic Data from people with various forms of cognitive/psychiatric impairments (RaPID-2)&#x201d;</conf-name>
					<fpage>11</fpage>
					<lpage>18</lpage>
				</element-citation>
			</ref>
			<ref id="ref-42-e116">
				<element-citation publication-type="article">
					<person-group person-group-type="author">
						<name>
							<surname>Kuchaiev</surname>
							<given-names>O.</given-names>
						</name>
						<name>
							<surname>Li</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Nguyen</surname>
							<given-names>H.</given-names>
						</name>
						<name>
							<surname>Hrinchuk</surname>
							<given-names>O.</given-names>
						</name>
						<name>
							<surname>Leary</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Ginsburg</surname>
							<given-names>B.</given-names>
						</name>
						<name>
							<surname>Castonguay</surname>
							<given-names>P.</given-names>
						</name>
					</person-group>
					<year>2019</year>
					<article-title>NeMo: a toolkit for building AI applications using Neural Modules</article-title>
					<source>ArXiv</source>
					<elocation-id>1909.09577</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-43-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Le</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Licata</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>Persad</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Provost</surname>
							<given-names>E.M.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<article-title>Automatic assessment of speech intelligibility for individuals with aphasia</article-title>
					<source>IEEE/ACM Transactions on Audio, Speech, and Language Processing</source>
					<volume>24</volume>
					<fpage>2187</fpage>
					<lpage>2199</lpage>
				</element-citation>
			</ref>
			<ref id="ref-44-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Le</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Licata</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>Provost</surname>
							<given-names>E.M.</given-names>
						</name>
					</person-group>
					<year>2017</year>
					<source>Automatic paraphasia detection from aphasic speech: A preliminary study</source>
					<conf-name>Interspeech 2017</conf-name>
					<fpage>294</fpage>
					<lpage>298</lpage>
				</element-citation>
			</ref>
			<ref id="ref-45-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Le</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Provost</surname>
							<given-names>E. M.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<source>Improving automatic recognition of aphasic speech with AphasiaBank</source>
					<conf-name>Interspeech 2016</conf-name>
					<fpage>2681</fpage>
					<lpage>2685</lpage>
				</element-citation>
			</ref>
			<ref id="ref-46-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Lee</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Liu</surname>
							<given-names>Y.</given-names>
						</name>
						<name>
							<surname>Huang</surname>
							<given-names>P.-W.</given-names>
						</name>
						<name>
							<surname>Chien</surname>
							<given-names>J.-T.</given-names>
						</name>
						<name>
							<surname>Lam</surname>
							<given-names>W. K.</given-names>
						</name>
						<name>
							<surname>Yeung</surname>
							<given-names>Y. T.</given-names>
						</name>
						<name>
							<surname>Law</surname>
							<given-names>S.-P.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<source>Automatic speech recognition for acoustical analysis and assessment of aCantonese pathological voice and speech</source>
					<conf-name>ICASSP 2016</conf-name>
					<fpage>6475</fpage>
					<lpage>6479</lpage>
					<conf-sponsor>IEEE</conf-sponsor>
				</element-citation>
			</ref>
			<ref id="ref-47-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Lin</surname>
							<given-names>Y.</given-names>
						</name>
						<name>
							<surname>Klumpp</surname>
							<given-names>P.</given-names>
						</name>
						<name>
							<surname>Pfab</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Abdelioua</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Gebray</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Sp&#xe4;th</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<month>04</month>
					<year>2022</year>
					<source>Eine automatische Sprachbewertung f&#xfc;r die neolexon Aphasie-App mithilfe K&#xfc;nstlicher Intelligenz [Automatic language assessment with artificial intelligence. for the neolexon aphasia app]</source>
					<conf-name>Sprachtherapie aktuell: Forschung - Wissen &#x2013; Transfer</conf-name>
					<volume>9</volume>
					<issue>1</issue>
					<elocation-id>XXXIV</elocation-id>
					<conf-sponsor>Workshop Klinische Linguistik e2022-11</conf-sponsor>
				</element-citation>
			</ref>
			<ref id="ref-48-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>LingoLab</collab>
					</person-group>
					<year>2020</year>
					<source>Digitale L&#xf6;sungen f&#xfc;r die Sprachtherapie [Digital solutions for speech and language therapy]</source>
					<ext-link ext-link-type="uri" xlink:href="https://lingo-lab.de/" id="exl-27-e116">https://lingo-lab.de/</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-06-07">June 7, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-49-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>MacWhinney</surname>
							<given-names>B.</given-names>
						</name>
						<name>
							<surname>Fromm</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Forbes</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Holland</surname>
							<given-names>A.</given-names>
						</name>
					</person-group>
					<year>2011</year>
					<article-title>AphasiaBank: Methods for Studying Discourse</article-title>
					<source>Aphasiology</source>
					<volume>25</volume>
					<issue>11</issue>
					<fpage>1286</fpage>
					<lpage>1307</lpage>
				</element-citation>
			</ref>
			<ref id="ref-50-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>McDowell</surname>
							<given-names>A.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>A fine-tuned version of facebook/wav2vec2-xls-r-1b on the MOZILLA-FOUNDATION/COMMON_VOICE_8_0 - DE dataset</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/AndrewMcDowell/wav2vec2-xls-r-1B-german" id="exl-29-e116">https://huggingface.co/AndrewMcDowell/wav2vec2-xls-r-1B-german</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-51-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Neolexon</collab>
					</person-group>
					<year>2023</year>
					<source>Logop&#xe4;die-Apps [Speech and language pathology apps]</source>
					<ext-link ext-link-type="uri" xlink:href="https://neolexon.de/" id="exl-31-e116">https://neolexon.de/</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-06-07">June 7, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-52-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Netzebandt</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Schmitz-Antonischki</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Heide</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<article-title>Hochfrequente Wortabruftherapie mit LingoTalk [High-frequency word retrieval therapy with LingoTalk]</article-title>
					<source>forum:logop&#xe4;die</source>
					<volume>36</volume>
					<issue>3</issue>
					<fpage>18</fpage>
					<lpage>24</lpage>
				</element-citation>
			</ref>
			<ref id="ref-53-e116">
				<element-citation publication-type="thesis">
					<person-group person-group-type="author">
						<name>
							<surname>Neumeyer</surname>
							<given-names>V.</given-names>
						</name>
					</person-group>
					<year>2009</year>
					<source>Phonetische Untersuchungender Artikulation von CI-Tr&#xe4;gern [Phonetic studies of CI users' articulation]</source>
					<comment content-type="degree">Master's thesis</comment>
					<publisher-name>Ludwig-Maximilians-Universit&#xe4;t M&#xfc;nchen</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-54-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<collab>NVIDIA</collab>
					</person-group>
					<year>2022a</year>
					<source>NVIDIA Conformer-CTC Large (de)</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/nvidia/stt_de_conformer_ctc_large" id="exl-33-e116">https://huggingface.co/nvidia/stt_de_conformer_ctc_large</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-55-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<collab>NVIDIA</collab>
					</person-group>
					<year>2022b</year>
					<source>NVIDIA Conformer-Transducer Large (de)</source>
					<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/nvidia/stt_de_conformer_transducer_large" id="exl-35-e116">https://huggingface.co/nvidia/stt_de_conformer_transducer_large</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-09-12">September 12, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-56-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Pisoni</surname>
							<given-names>D. B.</given-names>
						</name>
						<name>
							<surname>Martin</surname>
							<given-names>C. S.</given-names>
						</name>
					</person-group>
					<year>1989</year>
					<article-title>Effects of alcohol on the acoustic-phonetic properties of speech: Perceptual and acoustic analyses</article-title>
					<source>Alcoholism-Clinical and Experimental Research</source>
					<volume>13</volume>
					<issue>4</issue>
					<fpage>577</fpage>
					<lpage>587</lpage>
				</element-citation>
			</ref>
			<ref id="ref-57-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>Pompili</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Abad</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Trancoso</surname>
							<given-names>I.</given-names>
						</name>
						<name>
							<surname>Fonseca</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Martins</surname>
							<given-names>I. P.</given-names>
						</name>
					</person-group>
					<year>2020</year>
					<chapter-title>Evaluation and extensions and of an automatic and speech therapy and platform</chapter-title>
					<person-group person-group-type="editor">
						<name>
							<surname>Quaresma</surname>
							<given-names>P.</given-names>
						</name>
						<name>
							<surname>Vieira</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Alu&#xed;sio</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Moniz</surname>
							<given-names>H.</given-names>
						</name>
						<name>
							<surname>Batista</surname>
							<given-names>F.</given-names>
						</name>
						<name>
							<surname>Gon&#xe7;alves</surname>
							<given-names>T.</given-names>
						</name>
					</person-group>
					<source>Computational Processing of the Portuguese Language: 14th International Conference, PROPOR 2020, Portugal</source>
					<fpage>43</fpage>
					<lpage>52</lpage>
					<publisher-name>Springer International Publishing</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-58-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Pompili</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Abad</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Trancoso</surname>
							<given-names>I.</given-names>
						</name>
						<name>
							<surname>Fonseca</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Martins</surname>
							<given-names>I. P.</given-names>
						</name>
						<name>
							<surname>Leal</surname>
							<given-names>G.</given-names>
						</name>
						<name>
							<surname>Farrajota</surname>
							<given-names>L.</given-names>
						</name>
					</person-group>
					<year>2011</year>
					<source>An on-line system for remote treatment of aphasia</source>
					<conf-name>Second Workshop on Speech and Language Processing for Assistive Technologies</conf-name>
					<fpage>1</fpage>
					<lpage>10</lpage>
				</element-citation>
			</ref>
			<ref id="ref-59-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<collab>Python Software Foundation</collab>
					</person-group>
					<year>2022</year>
					<source>JiWER: Similarity measures for automatic speech recognition evaluation</source>
					<ext-link ext-link-type="uri" xlink:href="https://pypi.org/project/jiwer/" id="exl-37-e116">https://pypi.org/project/jiwer/</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-12-15">December 15, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-60-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Qin</surname>
							<given-names>Y.</given-names>
						</name>
						<name>
							<surname>Lee</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Feng</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Hin Kong</surname>
							<given-names>A. P.</given-names>
						</name>
					</person-group>
					<year>2018</year>
					<source>Automatic speech assessment for people with aphasia using TDNN-BLSTM with multi-task learning</source>
					<conf-name>Interspeech 2018</conf-name>
					<fpage>3418</fpage>
					<lpage>3422</lpage>
				</element-citation>
			</ref>
			<ref id="ref-61-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Qin</surname>
							<given-names>Y.</given-names>
						</name>
						<name>
							<surname>Lee</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Hin Kong</surname>
							<given-names>A. P.</given-names>
						</name>
					</person-group>
					<year>2020</year>
					<article-title>Automatic Assessment of Speech Impairment in Cantonese-Speaking People with Aphasia</article-title>
					<source>IEEE Journal of Selected Topics in Signal Processing</source>
					<volume>14</volume>
					<issue>2</issue>
					<fpage>331</fpage>
					<lpage>345</lpage>
				</element-citation>
			</ref>
			<ref id="ref-62-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>Qualls</surname>
							<given-names>C. D.</given-names>
						</name>
					</person-group>
					<year>2011</year>
					<chapter-title>Neurogenic disorders of speech, language, cognition-communication, and swallowing</chapter-title>
					<person-group person-group-type="editor">
						<name>
							<surname>Battle</surname>
							<given-names>D. E.</given-names>
						</name>
					</person-group>
					<source>Communication Disorders in Multicultural and International Populations</source>
					<fpage>148</fpage>
					<lpage>163</lpage>
					<publisher-name>Mosby</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-63-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<collab>R Core Team</collab>
					</person-group>
					<year>2022</year>
					<source>R: A language and environment for statistical computing</source>
					<publisher-name>R Foundation for Statistical Computing</publisher-name>
					<publisher-loc>Vienna, Austria</publisher-loc>
					<ext-link ext-link-type="uri" xlink:href="https://www.R-project.org" id="exl-39-e116">https://www.R-project.org</ext-link>
				</element-citation>
			</ref>
			<ref id="ref-64-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Radford</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Kim</surname>
							<given-names>J. W.</given-names>
						</name>
						<name>
							<surname>Xu</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>Brockman</surname>
							<given-names>G.</given-names>
						</name>
						<name>
							<surname>McLeavey</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Sutskever</surname>
							<given-names>I.</given-names>
						</name>
					</person-group>
					<year>2023</year>
					<source>Robust speech recognition via large-scale weak supervision</source>
					<conf-name>40th International Conference on Machine Learning (ICML'23)</conf-name>
					<fpage>28492</fpage>
					<lpage>28518</lpage>
					<ext-link ext-link-type="uri" xlink:href="https://jmlr.org/" id="exl-41-e116">JMLR.org</ext-link>
				</element-citation>
			</ref>
			<ref id="ref-65-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>RehaLingo</collab>
					</person-group>
					<year>2023</year>
					<source>Language rehabilitation software for the 21st Century</source>
					<ext-link ext-link-type="uri" xlink:href="https://www.rehalingo.com" id="exl-43-e116">https://www.rehalingo.com</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-06-07">June 7, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-66-e116">
				<element-citation publication-type="newspaper">
					<person-group person-group-type="author">
						<collab>Rhein-Zeitung</collab>
					</person-group>
					<year>2018</year>
					<article-title>Am Anfang war das Wort: Zu Besuch bei einem Aphasiker [In the beginning was the word: Visiting a person with aphasia]</article-title>
					<source>Rhein-Zeitung</source>
					<ext-link ext-link-type="uri" xlink:href="https://www.youtube.com/watch?v=Z1ZglYMSx1Y" id="exl-45-e116">https://www.youtube.com/watch?v=Z1ZglYMSx1Y</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2022-05-16">May 16, 2022</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-67-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Ruff</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Bocklet</surname>
							<given-names>T.</given-names>
						</name>
						<name>
							<surname>N&#xf6;th</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>M&#xfc;ller</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Hoster</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Schuster</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2017</year>
					<article-title>Speech production quality of cochlear implant users with respect to duration and onset of hearing loss</article-title>
					<source>ORL, Journal of Oto-Rhino-Laryngology and Its Related Specialities</source>
					<volume>79</volume>
					<issue>5</issue>
					<fpage>282</fpage>
					<lpage>294</lpage>
				</element-citation>
			</ref>
			<ref id="ref-68-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Ryalls</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>1984</year>
					<article-title>Where does the term &#x201c;aphasia&#x201d; come from?</article-title>
					<source>Brain and Language</source>
					<volume>21</volume>
					<fpage>358</fpage>
					<lpage>363</lpage>
				</element-citation>
			</ref>
			<ref id="ref-69-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Rykova</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Walther</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2024a</year>
					<source>AphaDIGITAL &#x2013; digital speech therapy solution for aphasia patients with automatic feedback provided by a virtual assistant</source>
					<conf-name>57th Hawaii International Conference on System Sciences (HICSS)</conf-name>
					<fpage>3385</fpage>
					<lpage>3394</lpage>
				</element-citation>
			</ref>
			<ref id="ref-70-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Rykova</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Walther</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2024b</year>
					<source>Linguistic and extralinguistic factors in automatic speech recognition of German atypical speech</source>
					<conf-name>20th Conference on Natural Language Processing (KONVENS 2024)</conf-name>
					<fpage>358</fpage>
					<lpage>367</lpage>
					<publisher-name>Association for Computational Linguistics</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-71-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Rykova</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Walther</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Zeuner</surname>
							<given-names>E.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>AphaDIGITAL &#x2014; Avatar-based digital speech therapy solution for aphasia patients: first evaluation</source>
					<conf-name>35th Fonetiikan P&#xe4;iv&#xe4;t</conf-name>
					<conf-loc>Joensuu, Finland</conf-loc>
					<ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/364676432_aphaDIGITAL_-Avatar-based_digital_speech_therapy_solution_for_aphasia_patients_first_evaluation" id="exl-47-e116">https://www.researchgate.net/publication/364676432_aphaDIGITAL_-Avatar-based_digital_speech_therapy_solution_for_aphasia_patients_first_evaluation</ext-link>
				</element-citation>
			</ref>
			<ref id="ref-72-e116">
				<element-citation publication-type="software">
					<person-group person-group-type="author">
						<name>
							<surname>Sch&#xe4;fer</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Oliznyk</surname>
							<given-names>O.</given-names>
						</name>
						<name>
							<surname>Haller</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<year>2023</year>
					<source>Deep Phonemizer: a G2P library in PyTorch</source>
					<ext-link ext-link-type="uri" xlink:href="https://github.com/as-ideas/DeepPhonemizer" id="exl-49-e116">https://github.com/as-ideas/DeepPhonemizer</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-05-17">May 17, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-73-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Schiel</surname>
							<given-names>F.</given-names>
						</name>
						<name>
							<surname>Heinrich</surname>
							<given-names>C.</given-names>
						</name>
						<name>
							<surname>Barf&#xfc;sser</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Gilg</surname>
							<given-names>T.</given-names>
						</name>
					</person-group>
					<year>2008</year>
					<source>ALC &#x2014; Alcohol Language Corpus</source>
					<conf-name>Sixth International Conference on Language Resources and Evaluation (LREC'08)</conf-name>
					<fpage>1641</fpage>
					<lpage>1645</lpage>
				</element-citation>
			</ref>
			<ref id="ref-74-e116">
				<element-citation publication-type="book">
					<person-group person-group-type="author">
						<name>
							<surname>Schneider</surname>
							<given-names>B.</given-names>
						</name>
						<name>
							<surname>Wehmeyer</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Gr&#xf6;tzbach</surname>
							<given-names>H.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<chapter-title>Aphasische Symptome und Syndrome [Aphasia symptoms and syndroms]</chapter-title>
					<person-group person-group-type="editor">
						<name>
							<surname>Schneider</surname>
							<given-names>B.</given-names>
						</name>
						<name>
							<surname>Wehmeyer</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>Gr&#xf6;tzbach</surname>
							<given-names>H.</given-names>
						</name>
					</person-group>
					<source>Aphasie: ICF-orientierte Diagnostik und Therapie [Aphasia: ICF-based diagnostics and therapy]</source>
					<fpage>25</fpage>
					<lpage>56</lpage>
					<publisher-name>Praxiswissen Logop&#xe4;die</publisher-name>
					<publisher-name>Monika Maria Thiel</publisher-name>
					<publisher-name>Mascha Wanke</publisher-name>
					<publisher-name>Susanne Weber</publisher-name>
				</element-citation>
			</ref>
			<ref id="ref-75-e116">
				<element-citation publication-type="report">
					<person-group person-group-type="author">
						<name>
							<surname>Schulz</surname>
							<given-names>J. B.</given-names>
						</name>
						<name>
							<surname>Werner</surname>
							<given-names>C. J.</given-names>
						</name>
					</person-group>
					<year>2019</year>
					<source>Statistischer Jahresbericht 2018. [Year 2018 statistics report]</source>
					<publisher-name>Aphasiestation</publisher-name>
					<publisher-name>Klinik f&#xfc;r Neurologie</publisher-name>
					<publisher-name>Uniklinik RWTH Aachen</publisher-name>
					<publisher-loc>Germany</publisher-loc>
				</element-citation>
			</ref>
			<ref id="ref-76-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Translationsregion f&#xfc;r digitale Gesundheits-versorgung [Translational Region for Digital Healthcare]</collab>
					</person-group>
					<abbrev>TDG</abbrev>
					<year>2021</year>
					<source>AphaDIGITAL: Entwicklung einer digitalen, dezentralen sprachtherapeuti-schen Versorgung [Development of digital, de-centralized speech therapy solutions]</source>
					<ext-link ext-link-type="uri" xlink:href="https://inno-tdg.de/projekte/aphadigital/" id="exl-51-e116">https://inno-tdg.de/projekte/aphadigital/</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-01-25">January 25, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-77-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Tislj&#xe1;r-Szab&#xf3;</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Rossu</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Varga</surname>
							<given-names>V.</given-names>
						</name>
						<name>
							<surname>Pl&#xe9;h</surname>
							<given-names>C.</given-names>
						</name>
					</person-group>
					<year>2014</year>
					<article-title>The effect of alcohol on speech production</article-title>
					<source>Journal of Psycholinguistic Research</source>
					<volume>43</volume>
					<fpage>737</fpage>
					<lpage>748</lpage>
				</element-citation>
			</ref>
			<ref id="ref-78-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Torre</surname>
							<given-names>I. G.</given-names>
						</name>
						<name>
							<surname>Romero</surname>
							<given-names>M.</given-names>
						</name>
						<name>
							<surname>&#xc1;lvarez</surname>
							<given-names>A.</given-names>
						</name>
					</person-group>
					<year>2021</year>
					<article-title>Improving aphasic speech recognition by using novel semi-supervised learning methods on AphasiaBank for English and Spanish</article-title>
					<source>Applied Sciences</source>
					<volume>11</volume>
					<issue>19</issue>
					<elocation-id>8872</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-79-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<collab>Universit&#xe4;t Stuttgart</collab>
					</person-group>
					<year>2023</year>
					<source>Sprache und Gehirn: Ein neurolinguistisches Tutorial [Language and brain: a neurolinguistics tutorial]</source>
					<ext-link ext-link-type="uri" xlink:href="https://www2.ims.uni-stuttgart.de/sgtutorial/index.html" id="exl-53-e116">https://www2.ims.uni-stuttgart.de/sgtutorial/index.html</ext-link>
					<date-in-citation content-type="access-date" iso-8601-date="2023-06-17">June 17, 2023</date-in-citation>
				</element-citation>
			</ref>
			<ref id="ref-80-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Vaezipour</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Campbell</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Theodoros</surname>
							<given-names>D.</given-names>
						</name>
						<name>
							<surname>Russell</surname>
							<given-names>T.</given-names>
						</name>
					</person-group>
					<year>2020</year>
					<article-title>Mobile apps for speech-language therapy in adults with communication disorders: review of content and quality</article-title>
					<source>JMIR mHealth and uHealth</source>
					<volume>8</volume>
					<issue>10</issue>
					<elocation-id>e18858</elocation-id>
				</element-citation>
			</ref>
			<ref id="ref-81-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>van de Sandt-Koenderman</surname>
							<given-names>W. M.</given-names>
						</name>
					</person-group>
					<year>2011</year>
					<article-title>Aphasia rehabilitation and the role of computer technology: Can we keep up with modern times?</article-title>
					<source>International Journal of Speech-Language Pathology</source>
					<volume>13</volume>
					<fpage>21</fpage>
					<lpage>27</lpage>
				</element-citation>
			</ref>
			<ref id="ref-82-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Vipperla</surname>
							<given-names>R.</given-names>
						</name>
						<name>
							<surname>Renals</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Frankel</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2008</year>
					<source>Longitudinal study of ASR performance on ageing voices</source>
					<conf-name>Interspeech 2008</conf-name>
					<fpage>2550</fpage>
					<lpage>2553</lpage>
				</element-citation>
			</ref>
			<ref id="ref-83-e116">
				<element-citation publication-type="journal">
					<person-group person-group-type="author">
						<name>
							<surname>Wambaugh</surname>
							<given-names>J. L.</given-names>
						</name>
						<name>
							<surname>Doyle</surname>
							<given-names>P. J.</given-names>
						</name>
						<name>
							<surname>Kalinyak</surname>
							<given-names>M. M.</given-names>
						</name>
						<name>
							<surname>West</surname>
							<given-names>J. E.</given-names>
						</name>
					</person-group>
					<year>1996</year>
					<article-title>A critical review of acoustic analyses of aphasic and-or apraxic speech</article-title>
					<source>Clinical Aphasiology</source>
					<volume>24</volume>
					<fpage>35</fpage>
					<lpage>63</lpage>
				</element-citation>
			</ref>
			<ref id="ref-84-e116">
				<element-citation publication-type="webpage">
					<person-group person-group-type="author">
						<name>
							<surname>Wiehage</surname>
							<given-names>A.</given-names>
						</name>
						<name>
							<surname>Heide</surname>
							<given-names>J.</given-names>
						</name>
					</person-group>
					<year>2016</year>
					<source>Aphasie: Informationen f&#xfc;r Betroffene und Angeh&#xf6;rige [Aphasia: information for the affected and relatives]</source>
					<ext-link ext-link-type="uri" xlink:href="https://www.dbs-ev.de/fileadmin/dokumente/Publikationen/dbs-Broschuere_Aphasie_2016.pdf" id="exl-55-e116">https://www.dbs-ev.de/fileadmin/dokumente/Publikationen/dbs-Broschuere_Aphasie_2016.pdf</ext-link>
				</element-citation>
			</ref>
			<ref id="ref-85-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Wirth</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Peinl</surname>
							<given-names>R.</given-names>
						</name>
					</person-group>
					<year>2022</year>
					<source>ASR in German - a detailed error analysis</source>
					<conf-name>2022 IEEE International Conference on Omni-layer Intelligent Systems (COINS)</conf-name>
					<fpage>1</fpage>
					<lpage>8</lpage>
					<conf-sponsor>IEEE</conf-sponsor>
				</element-citation>
			</ref>
			<ref id="ref-86-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Xu</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Matta</surname>
							<given-names>K.</given-names>
						</name>
						<name>
							<surname>Islam</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>N&#xfc;rnberger</surname>
							<given-names>A.</given-names>
						</name>
					</person-group>
					<year>2020</year>
					<source>German speech recognition system using DeepSpeech</source>
					<conf-name>4th International Conference on Natural Language Processing and Information Retrieval</conf-name>
					<fpage>102</fpage>
					<lpage>106</lpage>
				</element-citation>
			</ref>
			<ref id="ref-87-e116">
				<element-citation publication-type="confproc">
					<person-group person-group-type="author">
						<name>
							<surname>Zeuner</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Pietschmann</surname>
							<given-names>J.</given-names>
						</name>
						<name>
							<surname>Voigt-Zimmermann</surname>
							<given-names>S.</given-names>
						</name>
						<name>
							<surname>Rykova</surname>
							<given-names>E.</given-names>
						</name>
						<name>
							<surname>Walther</surname>
							<given-names>M.</given-names>
						</name>
					</person-group>
					<month>09</month>
					<year>2022</year>
					<source>aphaDIGITAL - Avatar-gest&#xfc;tzte digitale Aphasietherapie: Evaluation [aphaDIGITAL: Avatar-supported digital aphasia therapy - evaluation study]</source>
					<conf-name>DGSS Annual Conference 2022&#x201a; Stimme und Geschlecht im Wandel&#x2018; &#x2013; Implikationen f&#xfc;r Theorie und Praxis in der Sprechwissenschaft und Phonetics</conf-name>
					<conf-loc>Jena, Germany</conf-loc>
					<ext-link ext-link-type="uri" xlink:href="https://www.researchgate.net/publication/371510421_aphaDIGITAL_-Avatar-gestutzte_digitale_Aphasietherapie_Evaluation" id="exl-57-e116">https://www.researchgate.net/publication/371510421_aphaDIGITAL_-Avatar-gestutzte_digitale_Aphasietherapie_Evaluation</ext-link>
				</element-citation>
			</ref>
		</ref-list>
	</back>
</article>