var bibbase_data = {"data":"\"Loading..\"\n\n
\n\n \n\n \n\n \n \n\n \n\n \n \n\n \n\n \n
\n generated by\n \n \"bibbase.org\"\n\n \n
\n \n\n
\n\n \n\n\n
\n\n Excellent! Next you can\n create a new website with this list, or\n embed it in an existing web page by copying & pasting\n any of the following snippets.\n\n
\n JavaScript\n (easiest)\n
\n \n <script src=\"https://bibbase.org/show?bib=https%3A%2F%2Fbibbase.org%2Fzotero-mypublications%2FVolkerDellwo&jsonp=1&jsonp=1\"></script>\n \n
\n\n PHP\n
\n \n <?php\n $contents = file_get_contents(\"https://bibbase.org/show?bib=https%3A%2F%2Fbibbase.org%2Fzotero-mypublications%2FVolkerDellwo&jsonp=1\");\n print_r($contents);\n ?>\n \n
\n\n iFrame\n (not recommended)\n
\n \n <iframe src=\"https://bibbase.org/show?bib=https%3A%2F%2Fbibbase.org%2Fzotero-mypublications%2FVolkerDellwo&jsonp=1\"></iframe>\n \n
\n\n

\n For more details see the documention.\n

\n
\n
\n\n
\n\n This is a preview! To use this list on your own web site\n or create a new web site from it,\n create a free account. The file will be added\n and you will be able to edit it in the File Manager.\n We will show you instructions once you've created your account.\n
\n\n
\n\n

To the site owner:

\n\n

Action required! Mendeley is changing its\n API. In order to keep using Mendeley with BibBase past April\n 14th, you need to:\n

    \n
  1. renew the authorization for BibBase on Mendeley, and
  2. \n
  3. update the BibBase URL\n in your page the same way you did when you initially set up\n this page.\n
  4. \n
\n

\n\n

\n \n \n Fix it now\n

\n
\n\n
\n\n\n
\n \n \n
\n
\n  \n 2026\n \n \n (9)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Systematic F1 increase in Korean aegyo speech may signal childlike qualities.\n \n \n \n \n\n\n \n Kim, J.; and Dellwo, V.\n\n\n \n\n\n\n JASA Express Letters, 6(8): 085204. August 2026.\n \n\n\n\n
\n\n\n\n \n \n \"SystematicPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kimSystematicF1Increase2026,\n\ttitle = {Systematic {F1} increase in {Korean} \\textit{aegyo} speech may signal childlike qualities},\n\tvolume = {6},\n\tissn = {2691-1191},\n\turl = {https://pubs.aip.org/jel/article/6/8/085204/3401298/Systematic-F1-increase-in-Korean-aegyo-speech-may},\n\tdoi = {10.1121/10.0044475},\n\tabstract = {Korean aegyo is a socially recognized childlike speaking style used predominantly in romantic interactions among adults. This study examined vowel space modification in aegyo by analyzing formant frequencies from twelve Seoul Korean speakers who produced identical scripts in aegyo and non-aegyo styles. Results show that aegyo speech features a significant increase in F1 values across vowels and selective fronting of front vowels, leading to some expansion of the vowel space but mainly shifting to higher F1. These findings motivate the hypothesis that adult speakers construct a childlike vocal quality by increasing F1, possibly through a combination of different factors like global vowel lowering and partial fronting.},\n\tlanguage = {en},\n\tnumber = {8},\n\turldate = {2026-08-20},\n\tjournal = {JASA Express Letters},\n\tauthor = {Kim, Ji-eun and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2026},\n\tpages = {085204},\n}\n\n\n\n\n\n\n\n
\n
\n\n\n
\n Korean aegyo is a socially recognized childlike speaking style used predominantly in romantic interactions among adults. This study examined vowel space modification in aegyo by analyzing formant frequencies from twelve Seoul Korean speakers who produced identical scripts in aegyo and non-aegyo styles. Results show that aegyo speech features a significant increase in F1 values across vowels and selective fronting of front vowels, leading to some expansion of the vowel space but mainly shifting to higher F1. These findings motivate the hypothesis that adult speakers construct a childlike vocal quality by increasing F1, possibly through a combination of different factors like global vowel lowering and partial fronting.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Impaired self-other voice discrimination in patients with auditory-verbal hallucinations and nonclinical hallucination proneness.\n \n \n \n\n\n \n Vukojević, J.; Jelić, L.; McCormack, K.; Sušac, J.; Muselimović, I.; Bagarić, M.; Brečić, P.; Dellwo, V.; Cifrek, M.; Savić, A.; and others\n\n\n \n\n\n\n Schizophrenia bulletin, 52(3): sbag073. 2026.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{vukojevic2026impaired,\n\ttitle = {Impaired self-other voice discrimination in patients with auditory-verbal hallucinations and nonclinical hallucination proneness},\n\tvolume = {52},\n\tnumber = {3},\n\tjournal = {Schizophrenia bulletin},\n\tpublisher = {Oxford University Press},\n\tauthor = {Vukojević, Jakša and Jelić, Luka and McCormack, Kieren and Sušac, Jelena and Muselimović, Ivan and Bagarić, Mihovil and Brečić, Petrana and Dellwo, Volker and Cifrek, Mario and Savić, Aleksandar and {others}},\n\tyear = {2026},\n\tpages = {sbag073},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Micro-expression-aware avatar fingerprinting via inter-frame feature differencing.\n \n \n \n\n\n \n Chapariniya, M.; Odobez, J.; Dellwo, V.; and Vuković, T.\n\n\n \n\n\n\n In 2026 IEEE 20th international conference on automatic face and gesture recognition (FG), pages 1–8, 2026. IEEE\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{chapariniya2026micro,\n\ttitle = {Micro-expression-aware avatar fingerprinting via inter-frame feature differencing},\n\tbooktitle = {2026 {IEEE} 20th international conference on automatic face and gesture recognition ({FG})},\n\tpublisher = {IEEE},\n\tauthor = {Chapariniya, Masoumeh and Odobez, Jean-Marc and Dellwo, Volker and Vuković, Teodora},\n\tyear = {2026},\n\tpages = {1--8},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Towards language-independent face-voice association with multimodal foundation models.\n \n \n \n\n\n \n Farhadipour, A.; Vukovic, T.; and Dellwo, V.\n\n\n \n\n\n\n In ICASSP 2026-2026 IEEE international conference on acoustics, speech and signal processing (ICASSP), pages 21787–21789, 2026. IEEE\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{farhadipour2026towards,\n\ttitle = {Towards language-independent face-voice association with multimodal foundation models},\n\tbooktitle = {{ICASSP} 2026-2026 {IEEE} international conference on acoustics, speech and signal processing ({ICASSP})},\n\tpublisher = {IEEE},\n\tauthor = {Farhadipour, Aref and Vukovic, Teodora and Dellwo, Volker},\n\tyear = {2026},\n\tpages = {21787--21789},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Vocal-tract length estimation from vowel formants benchmarked against acoustic pharyngometry.\n \n \n \n \n\n\n \n Friedrichs, D.; Guerrini, U.; Ekström, A.; Dellwo, V.; and Moran, S.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 159(6): 5650–5666. June 2026.\n \n\n\n\n
\n\n\n\n \n \n \"Vocal-tractPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{friedrichsVocaltractLengthEstimation2026a,\n\ttitle = {Vocal-tract length estimation from vowel formants benchmarked against acoustic pharyngometry},\n\tvolume = {159},\n\tissn = {1520-8524},\n\turl = {https://pubs.aip.org/jasa/article/159/6/5650/3396117/Vocal-tract-length-estimation-from-vowel-formants},\n\tdoi = {10.1121/10.0044192},\n\tabstract = {Estimating vocal-tract length (VTL) from vowel formants can aid speaker normalization, but few methods have been benchmarked against an anatomical reference in the same speakers. We combined acoustic pharyngometry (APh) and speech data from 42 adults to benchmark eight widely used formant-based VTL estimators against incisors-to-glottis length and to test an interpretable two-stage bias-corrected linear estimator. Across more than 400 000 central frames with valid F1–F4, traditional quarter-wave, odd-harmonic, and dispersion-type estimators correlated with VTLAPh but showed poor out-of-sample anatomical recovery and strong calibration compression. Re-estimated one-stage linear models reduced mean absolute error (MAE; median �1.0 cm) but still overestimated shorter tracts and underestimated longer tracts. A two-stage model markedly improved calibration and agreement, outperforming one-stage linear and nonlinear alternatives (median per-vowel MAE 0.39 cm, median out-of-sample R2 ¼ 0:83). Front and front-rounded vowels were the most informative. Speaker-level 95\\% limits of agreement were about 60.9 cm, indicating that the method is better suited to aggregated tract-scale estimation than to direct anatomical measurement. These results identify calibration bias as a central limitation of standard formant-based VTL estimators and provide a practical, interpretable route to tract-scale estimation from similarly processed labeled-vowel data under matched conditions.},\n\tlanguage = {en},\n\tnumber = {6},\n\turldate = {2026-07-06},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {Friedrichs, Daniel and Guerrini, Urs and Ekström, Axel and Dellwo, Volker and Moran, Steven},\n\tmonth = jun,\n\tyear = {2026},\n\tpages = {5650--5666},\n}\n\n\n\n
\n
\n\n\n
\n Estimating vocal-tract length (VTL) from vowel formants can aid speaker normalization, but few methods have been benchmarked against an anatomical reference in the same speakers. We combined acoustic pharyngometry (APh) and speech data from 42 adults to benchmark eight widely used formant-based VTL estimators against incisors-to-glottis length and to test an interpretable two-stage bias-corrected linear estimator. Across more than 400 000 central frames with valid F1–F4, traditional quarter-wave, odd-harmonic, and dispersion-type estimators correlated with VTLAPh but showed poor out-of-sample anatomical recovery and strong calibration compression. Re-estimated one-stage linear models reduced mean absolute error (MAE; median �1.0 cm) but still overestimated shorter tracts and underestimated longer tracts. A two-stage model markedly improved calibration and agreement, outperforming one-stage linear and nonlinear alternatives (median per-vowel MAE 0.39 cm, median out-of-sample R2 ¼ 0:83). Front and front-rounded vowels were the most informative. Speaker-level 95% limits of agreement were about 60.9 cm, indicating that the method is better suited to aggregated tract-scale estimation than to direct anatomical measurement. These results identify calibration bias as a central limitation of standard formant-based VTL estimators and provide a practical, interpretable route to tract-scale estimation from similarly processed labeled-vowel data under matched conditions.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n A multimodal speech-production dataset with time-aligned articulography, EEG, audio, and vocal-tract anatomy.\n \n \n \n \n\n\n \n Friedrichs, D.; Vyshnevetska, V.; Lancheros, M.; Bolt, E.; Dellwo, V.; and Moran, S.\n\n\n \n\n\n\n Scientific Data. July 2026.\n \n\n\n\n
\n\n\n\n \n \n \"APaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{friedrichsMultimodalSpeechproductionDataset2026a,\n\ttitle = {A multimodal speech-production dataset with time-aligned articulography, {EEG}, audio, and vocal-tract anatomy},\n\tissn = {2052-4463},\n\turl = {https://www.nature.com/articles/s41597-026-07704-3},\n\tdoi = {10.1038/s41597-026-07704-3},\n\tabstract = {Abstract\n            We present a multimodal speech-production dataset combining simultaneous electromagnetic articulography (EMA), electroencephalography (EEG), and audio from 29 adult native speakers of German. All participants have external craniofacial anthropometry; an anatomical subset (N = 18) additionally contributed acoustic pharyngometry, rhinometry, and 3D head surface meshes. Speech materials include high-trial diadochokinetic sequences at habitual and maximally fast rates, plus EMA+audio for passage reading and sustained vowels, as well as EMA for palate tracing and non-speech oromotor actions. The corpus contains 8,700 syllable-task trials and approximately 17 h of EMA with matched audio. A microcontroller mirrors each EMA sweep’s start and stop as 1 ms transistor—transistor logic (TTL) pulses on the EEG digital  channel, enabling sub-millisecond alignment. We distribute raw and minimally processed streams, stable event codes, and machine-readable metadata, plus example Python utilities (and a container) for loading, synchronising, and basic preprocessing. The resource supports studies that exploit articulatory landmarks for EEG alignment, examine pre-movement activity, assess overt-speech EEG artefact handling, and develop anatomy-informed models linking vocal-tract structure to articulatory dynamics and acoustics. Data and code are openly available under a CC-BY 4.0 licence with versioned DOIs.},\n\tlanguage = {en},\n\turldate = {2026-08-03},\n\tjournal = {Scientific Data},\n\tauthor = {Friedrichs, Daniel and Vyshnevetska, Valeriia and Lancheros, Monica and Bolt, Elena and Dellwo, Volker and Moran, Steven},\n\tmonth = jul,\n\tyear = {2026},\n}\n\n\n\n
\n
\n\n\n
\n Abstract We present a multimodal speech-production dataset combining simultaneous electromagnetic articulography (EMA), electroencephalography (EEG), and audio from 29 adult native speakers of German. All participants have external craniofacial anthropometry; an anatomical subset (N = 18) additionally contributed acoustic pharyngometry, rhinometry, and 3D head surface meshes. Speech materials include high-trial diadochokinetic sequences at habitual and maximally fast rates, plus EMA+audio for passage reading and sustained vowels, as well as EMA for palate tracing and non-speech oromotor actions. The corpus contains 8,700 syllable-task trials and approximately 17 h of EMA with matched audio. A microcontroller mirrors each EMA sweep’s start and stop as 1 ms transistor—transistor logic (TTL) pulses on the EEG digital channel, enabling sub-millisecond alignment. We distribute raw and minimally processed streams, stable event codes, and machine-readable metadata, plus example Python utilities (and a container) for loading, synchronising, and basic preprocessing. The resource supports studies that exploit articulatory landmarks for EEG alignment, examine pre-movement activity, assess overt-speech EEG artefact handling, and develop anatomy-informed models linking vocal-tract structure to articulatory dynamics and acoustics. Data and code are openly available under a CC-BY 4.0 licence with versioned DOIs.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Implicit voice learning through discrimination outperforms explicit listen-and-memorize tasks.\n \n \n \n \n\n\n \n Fröhlich, A.; Ramon, M.; French, P.; and Dellwo, V.\n\n\n \n\n\n\n Scientific Reports, 16(1): 13498. March 2026.\n \n\n\n\n
\n\n\n\n \n \n \"ImplicitPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{frohlichImplicitVoiceLearning2026a,\n\ttitle = {Implicit voice learning through discrimination outperforms explicit listen-and-memorize tasks},\n\tvolume = {16},\n\tissn = {2045-2322},\n\turl = {https://www.nature.com/articles/s41598-026-41541-z},\n\tdoi = {10.1038/s41598-026-41541-z},\n\tabstract = {Abstract\n            \n              Voice learning primarily occurs\n              implicitly\n              in everyday situations—as an incidental by-product of other activities such as participating in conversation or listening to voices in the media. Most research on voice learning is conducted in laboratory settings, where participants are\n              explicitly\n              instructed to attend to and memorize voices for later recognition. Yet, the impact of task awareness (awareness regarding the goal of voice learning) on voice recognition performance remains poorly understood. To address this gap, we conducted a study comparing two voice-learning modalities: explicit learning, instructing participants to listen to and memorize voices for later recognition, and implicit learning, based on exposure during a voice discrimination task, without awareness of a subsequent recognition test. After both exposure phases, participants completed a surreptitious old–new voice recognition task. To further examine whether task awareness is modulated by voice load (number of voice identities introduced in the experiment), we implemented both a simple and a challenging version of the experiment. We found that, irrespective of voice load, implicit learning through participation in a discrimination task, resulted in higher recognition performance than explicit listen-and-memorize training. These findings suggest that highly demanding explicit listen-and-memorize tasks may benefit from incorporating ecologically valid familiarization paradigms, such as voice discrimination. We discuss the implications of our findings in relation to previous empirical research and their relevance for forensic applications.},\n\tlanguage = {en},\n\tnumber = {1},\n\turldate = {2026-07-10},\n\tjournal = {Scientific Reports},\n\tauthor = {Fröhlich, Andrea and Ramon, Meike and French, Peter and Dellwo, Volker},\n\tmonth = mar,\n\tyear = {2026},\n\tpages = {13498},\n}\n\n\n\n
\n
\n\n\n
\n Abstract Voice learning primarily occurs implicitly in everyday situations—as an incidental by-product of other activities such as participating in conversation or listening to voices in the media. Most research on voice learning is conducted in laboratory settings, where participants are explicitly instructed to attend to and memorize voices for later recognition. Yet, the impact of task awareness (awareness regarding the goal of voice learning) on voice recognition performance remains poorly understood. To address this gap, we conducted a study comparing two voice-learning modalities: explicit learning, instructing participants to listen to and memorize voices for later recognition, and implicit learning, based on exposure during a voice discrimination task, without awareness of a subsequent recognition test. After both exposure phases, participants completed a surreptitious old–new voice recognition task. To further examine whether task awareness is modulated by voice load (number of voice identities introduced in the experiment), we implemented both a simple and a challenging version of the experiment. We found that, irrespective of voice load, implicit learning through participation in a discrimination task, resulted in higher recognition performance than explicit listen-and-memorize training. These findings suggest that highly demanding explicit listen-and-memorize tasks may benefit from incorporating ecologically valid familiarization paradigms, such as voice discrimination. We discuss the implications of our findings in relation to previous empirical research and their relevance for forensic applications.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n As group size increases, individuals modify their vocal features to signal cooperation while remaining recognizable.\n \n \n \n \n\n\n \n Pellegrino, E.; and Dellwo, V.\n\n\n \n\n\n\n Frontiers in Psychology, 17: 1765090. June 2026.\n \n\n\n\n
\n\n\n\n \n \n \"AsPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoGroupSizeIncreases2026a,\n\ttitle = {As group size increases, individuals modify their vocal features to signal cooperation while remaining recognizable},\n\tvolume = {17},\n\tissn = {1664-1078},\n\turl = {https://www.frontiersin.org/articles/10.3389/fpsyg.2026.1765090/full},\n\tdoi = {10.3389/fpsyg.2026.1765090},\n\tabstract = {Introduction\n              Cooperation is essential to humans and manifests in speech through acoustic convergence, in which interlocutors’ voices become more similar. Yet convergence can be limited when signaling individuality is more important than aligning with others. In such contexts, speakers may adopt non-accommodative strategies to preserve their vocal identity and recognizability. Drawing on animal communication research—which shows that species in larger social groups exhibit greater vocal individuality—we test how group size (3 and 5 interactants) shapes the balance between remaining identifiable while still cooperating to support communication.\n            \n            \n              Methods\n              In an interactive online game, players collaborated on a shared task while trying to recognize one another by voice, with the player recognized best receiving a reward. To assess how group size influences accommodation under these dual demands, we analyzed the speech of three players who participated in both game sessions. Acoustic features relevant to voice identity and amenable to accommodation—harmonicity, jitter, F0, formant dispersion, and duration—were extracted and reduced through Principal Component Analysis to two dimensions accounting for 52.9\\% of the variance. Following the identification of group size effects on these components, we assessed whether inter-speaker acoustic differences increased, decreased, or remained stable across conditions. Additionally, within-speaker variability was examined as a function of group size to determine whether observed changes in the five-player condition were driven by all speakers or only a subset.\n            \n            \n              Results\n              In larger groups, accommodation was selective: players modulated their acoustic features, maintaining some while converging on others. No significant differences as a function of group size were observed for PC1, interpreted as reflecting maintenance and primarily associated with voice quality measures (harmonicity and jitter). In contrast, significant differences emerged for PC2, largely driven by F0 standard deviation and duration. These changes indicated reduced inter-speaker differences in the larger group, consistent with convergence. In the larger group setting, convergence was observed when two speakers showed mutual alignment and also shifted toward the speech patterns of a third participant, who remained comparatively stable in their acoustic behavior.\n            \n            \n              Discussion\n              The pattern of selective and asymmetrical convergence indicates that speakers strategically balance the goals of cooperation and individuality, suggesting that recognizability demands also shape accommodation.},\n\tlanguage = {en},\n\turldate = {2026-07-06},\n\tjournal = {Frontiers in Psychology},\n\tauthor = {Pellegrino, Elisa and Dellwo, Volker},\n\tmonth = jun,\n\tyear = {2026},\n\tpages = {1765090},\n}\n\n\n\n
\n
\n\n\n
\n Introduction Cooperation is essential to humans and manifests in speech through acoustic convergence, in which interlocutors’ voices become more similar. Yet convergence can be limited when signaling individuality is more important than aligning with others. In such contexts, speakers may adopt non-accommodative strategies to preserve their vocal identity and recognizability. Drawing on animal communication research—which shows that species in larger social groups exhibit greater vocal individuality—we test how group size (3 and 5 interactants) shapes the balance between remaining identifiable while still cooperating to support communication. Methods In an interactive online game, players collaborated on a shared task while trying to recognize one another by voice, with the player recognized best receiving a reward. To assess how group size influences accommodation under these dual demands, we analyzed the speech of three players who participated in both game sessions. Acoustic features relevant to voice identity and amenable to accommodation—harmonicity, jitter, F0, formant dispersion, and duration—were extracted and reduced through Principal Component Analysis to two dimensions accounting for 52.9% of the variance. Following the identification of group size effects on these components, we assessed whether inter-speaker acoustic differences increased, decreased, or remained stable across conditions. Additionally, within-speaker variability was examined as a function of group size to determine whether observed changes in the five-player condition were driven by all speakers or only a subset. Results In larger groups, accommodation was selective: players modulated their acoustic features, maintaining some while converging on others. No significant differences as a function of group size were observed for PC1, interpreted as reflecting maintenance and primarily associated with voice quality measures (harmonicity and jitter). In contrast, significant differences emerged for PC2, largely driven by F0 standard deviation and duration. These changes indicated reduced inter-speaker differences in the larger group, consistent with convergence. In the larger group setting, convergence was observed when two speakers showed mutual alignment and also shifted toward the speech patterns of a third participant, who remained comparatively stable in their acoustic behavior. Discussion The pattern of selective and asymmetrical convergence indicates that speakers strategically balance the goals of cooperation and individuality, suggesting that recognizability demands also shape accommodation.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Dominant role of temporal–prosodic cues in sentence-level voice discrimination: Insights from interpretable machine learning.\n \n \n \n \n\n\n \n Xu, T.; Jiang, X.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2026, pages 609–613, May 2026. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"DominantPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{xuDominantRoleTemporal2026,\n\ttitle = {Dominant role of temporal–prosodic cues in sentence-level voice discrimination: {Insights} from interpretable machine learning},\n\tshorttitle = {Dominant role of temporal–prosodic cues in sentence-level voice discrimination},\n\turl = {https://www.isca-archive.org/speechprosody_2026/xu26_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2026-123},\n\tabstract = {Voice perception has often been modeled within a multidimensional acoustic space, yet the stability of its underlying dimensions across linguistic levels (e.g., syllables, words, sentences) remains unclear. This study investigated whether the acoustic basis of voice discrimination remains stable across the linguistic levels of words and sentences. Mandarin-speaking listeners judged speaker identity (same/different) for word and sentence pairs with controlled similarity in a classic voice space. Interpretable machine learning classifiers were trained on acoustic differences between speech pairs to predict listeners’ judgments. Models performed well within levels, but generalization across levels declined and was asymmetric: sentence-trained models tended to overproduce different-speaker responses when tested on word data, whereas word-trained models showed a bias toward same-speaker responses when tested on sentence data. Feature importance rankings converged on a shared acoustic scaffold (spectral balance, formant structure, higher harmonics, and pitch) but highlighted the dominant role of temporal–prosodic cues (speech rate) and the growing relevance of variability-based cues for sentences compared with words. These findings refine prototype-based models of voice perception by demonstrating adaptive cue weighting across linguistic structure and offer implications for voice perception in naturalistic contexts. Methodologically, this study addressed an issue in the literature concerning the computation of variability measures for harmonic spectral features.},\n\tlanguage = {en},\n\turldate = {2026-08-03},\n\tbooktitle = {Speech {Prosody} 2026},\n\tpublisher = {ISCA},\n\tauthor = {Xu, Tianze and Jiang, Xiaoming and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2026},\n\tpages = {609--613},\n}\n\n\n\n
\n
\n\n\n
\n Voice perception has often been modeled within a multidimensional acoustic space, yet the stability of its underlying dimensions across linguistic levels (e.g., syllables, words, sentences) remains unclear. This study investigated whether the acoustic basis of voice discrimination remains stable across the linguistic levels of words and sentences. Mandarin-speaking listeners judged speaker identity (same/different) for word and sentence pairs with controlled similarity in a classic voice space. Interpretable machine learning classifiers were trained on acoustic differences between speech pairs to predict listeners’ judgments. Models performed well within levels, but generalization across levels declined and was asymmetric: sentence-trained models tended to overproduce different-speaker responses when tested on word data, whereas word-trained models showed a bias toward same-speaker responses when tested on sentence data. Feature importance rankings converged on a shared acoustic scaffold (spectral balance, formant structure, higher harmonics, and pitch) but highlighted the dominant role of temporal–prosodic cues (speech rate) and the growing relevance of variability-based cues for sentences compared with words. These findings refine prototype-based models of voice perception by demonstrating adaptive cue weighting across linguistic structure and offer implications for voice perception in naturalistic contexts. Methodologically, this study addressed an issue in the literature concerning the computation of variability measures for harmonic spectral features.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2025\n \n \n (7)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Beyond appearance: Transformer-based person identification from conversational dynamics.\n \n \n \n\n\n \n Chapariniya, M.; Vuković, T.; Ebling, S.; and Dellwo, V.\n\n\n \n\n\n\n In 2025 15th international conference on computer and knowledge engineering (ICCKE), pages 1–6, 2025. IEEE\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{chapariniya2025beyond,\n\ttitle = {Beyond appearance: {Transformer}-based person identification from conversational dynamics},\n\tbooktitle = {2025 15th international conference on computer and knowledge engineering ({ICCKE})},\n\tpublisher = {IEEE},\n\tauthor = {Chapariniya, Masoumeh and Vuković, Teodora and Ebling, Sarah and Dellwo, Volker},\n\tyear = {2025},\n\tpages = {1--6},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Rhythmic types vs. continuum: An empirical test on 53 languages.\n \n \n \n\n\n \n Seifart, F.; and Dellwo, V.\n\n\n \n\n\n\n In Societas linguistica europaea 58th annual meeting, 2025. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{seifart2025rhythmic,\n\ttitle = {Rhythmic types vs. continuum: {An} empirical test on 53 languages},\n\tbooktitle = {Societas linguistica europaea 58th annual meeting},\n\tauthor = {Seifart, Frank and Dellwo, Volker},\n\tyear = {2025},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Voxplorer: Voice data exploration and projection in an interactive dashboard.\n \n \n \n\n\n \n De Luca, A.; Madikeri, S.; and Dellwo, V.\n\n\n \n\n\n\n In Proc. Interspeech 2025, pages 296–297, 2025. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{de2025voxplorer,\n\ttitle = {Voxplorer: {Voice} data exploration and projection in an interactive dashboard},\n\tbooktitle = {Proc. {Interspeech} 2025},\n\tauthor = {De Luca, Alessandro and Madikeri, Srikanth and Dellwo, Volker},\n\tyear = {2025},\n\tpages = {296--297},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Listeners are biased towards voices of young speakers and female speakers when discriminating voices.\n \n \n \n\n\n \n Vyshnevetska, V.; Giroud, N.; Ramon, M.; and Dellwo, V.\n\n\n \n\n\n\n Cognitive Research: Principles and Implications, 10(1): 28. 2025.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{vyshnevetska2025listeners,\n\ttitle = {Listeners are biased towards voices of young speakers and female speakers when discriminating voices},\n\tvolume = {10},\n\tnumber = {1},\n\tjournal = {Cognitive Research: Principles and Implications},\n\tpublisher = {Springer International Publishing Cham},\n\tauthor = {Vyshnevetska, Valeriia and Giroud, Nathalie and Ramon, Meike and Dellwo, Volker},\n\tyear = {2025},\n\tpages = {28},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n GOOSVC: Version control for content creation with generative AI.\n \n \n \n\n\n \n Grünert, D.; de Spindler, A.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the 10th edition of the swiss text analytics conference, pages 66–74, 2025. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{grunert2025goosvc,\n\ttitle = {{GOOSVC}: {Version} control for content creation with generative {AI}},\n\tbooktitle = {Proceedings of the 10th edition of the swiss text analytics conference},\n\tauthor = {Grünert, David and de Spindler, Alexandre and Dellwo, Volker},\n\tyear = {2025},\n\tpages = {66--74},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Between- and within-speaker variability of voiceless fricatives in Persian.\n \n \n \n \n\n\n \n Asadi, H.; Alinezhad, B.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 158(6): 4294–4307. December 2025.\n \n\n\n\n
\n\n\n\n \n \n \"Between-Paper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{asadiWithinspeakerVariabilityVoiceless2025a,\n\ttitle = {Between- and within-speaker variability of voiceless fricatives in {Persian}},\n\tvolume = {158},\n\tissn = {1520-8524},\n\turl = {https://pubs.aip.org/jasa/article/158/6/4294/3373627/Between-and-within-speaker-variability-of},\n\tdoi = {10.1121/10.0039894},\n\tabstract = {Fricatives vary acoustically across languages and individuals, with speaker variability shaped by both phonetic and non-phonetic factors. This study examined between- and within-speaker variability in Persian voiceless fricatives (/f/, /s/, /S/, /x/) and how linguistic environments, such as syllable position and lexical stress, affect this variability. A gender-balanced sample of 24 Persian speakers was recorded in two sessions, 1–2 two weeks apart. Acoustic analysis targeted the first four spectral moments and duration. Results showed that center of gravity captured the greatest between-speaker variability, followed by standard deviation, skewness, duration, and kurtosis. Across segments, the alveolar /s/ exhibited the highest speaker-specificity, followed by /S/, /f/, and /x/. Gender-based patterns emerged: for males, the center of gravity and skewness of /s/ were most discriminative, whereas for females, the center of gravity and standard deviation of /S/ were most effective. The labiodental /f/ showed some speaker-specific characteristics only in the male group. Voiceless fricatives in syllable-initial positions demonstrated more speaker-specificity, while lexical stress did not impact between-speaker variability. Results also highlight cross-linguistic differences in the acoustic cues most effective for speaker differentiation and demonstrate that optimal features can vary across speaker populations. Adaptive algorithms are therefore crucial for improving forensic speaker comparison.},\n\tlanguage = {en},\n\tnumber = {6},\n\turldate = {2025-12-21},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {Asadi, Homa and Alinezhad, Batool and Dellwo, Volker},\n\tmonth = dec,\n\tyear = {2025},\n\tpages = {4294--4307},\n}\n\n\n\n
\n
\n\n\n
\n Fricatives vary acoustically across languages and individuals, with speaker variability shaped by both phonetic and non-phonetic factors. This study examined between- and within-speaker variability in Persian voiceless fricatives (/f/, /s/, /S/, /x/) and how linguistic environments, such as syllable position and lexical stress, affect this variability. A gender-balanced sample of 24 Persian speakers was recorded in two sessions, 1–2 two weeks apart. Acoustic analysis targeted the first four spectral moments and duration. Results showed that center of gravity captured the greatest between-speaker variability, followed by standard deviation, skewness, duration, and kurtosis. Across segments, the alveolar /s/ exhibited the highest speaker-specificity, followed by /S/, /f/, and /x/. Gender-based patterns emerged: for males, the center of gravity and skewness of /s/ were most discriminative, whereas for females, the center of gravity and standard deviation of /S/ were most effective. The labiodental /f/ showed some speaker-specific characteristics only in the male group. Voiceless fricatives in syllable-initial positions demonstrated more speaker-specificity, while lexical stress did not impact between-speaker variability. Results also highlight cross-linguistic differences in the acoustic cues most effective for speaker differentiation and demonstrate that optimal features can vary across speaker populations. Adaptive algorithms are therefore crucial for improving forensic speaker comparison.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The role of phonetic overlap for speaker discrimination.\n \n \n \n \n\n\n \n Bradshaw, L.; Chodroff, E.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 157(5): 3572–3589. May 2025.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{bradshawRolePhoneticOverlap2025,\n\ttitle = {The role of phonetic overlap for speaker discrimination},\n\tvolume = {157},\n\tissn = {1520-8524},\n\turl = {https://pubs.aip.org/jasa/article/157/5/3572/3346794/The-role-of-phonetic-overlap-for-speaker},\n\tdoi = {10.1121/10.0036562},\n\tabstract = {Linguistic information influences processing of speaker information in a multitude of ways, whether this arises from the listener’s familiarity with the language or dialect spoken, or existing linguistic relationships between spoken words. Specifically, phonological and semantic relationships between spoken words have been observed to influence a listeners’ ability to discriminate voices. This study aims to develop our understanding of how different kinds of linguistic relationships, namely, phonetic relationships, influence the processing of speaker information. We conducted two experiments, a voice discrimination task and a voice similarity rating task, in which listeners were presented with pairs of speakers producing two words with various degrees of phonetic overlap. On the whole, higher quantities of phonetic overlap corresponded to higher speaker discrimination performance and higher similarity scores; however, the type of the phonetic overlap also mattered. Overlapping vowel segments showed substantial utility, while overlap of the phonological rhyme alone substantially lower performance. Results from this condition suggest that a phonological relationship within the word pair can interfere with otherwise increased quantities of phonetic overlap. These findings highlight the salience of the phonological rhyme in voice processing, as well as the overall impact of phonetic overlap. CV 2025 Acoustical Society of America.},\n\tlanguage = {en},\n\tnumber = {5},\n\turldate = {2026-08-03},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {Bradshaw, Leah and Chodroff, Eleanor and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2025},\n\tpages = {3572--3589},\n}\n\n\n\n
\n
\n\n\n
\n Linguistic information influences processing of speaker information in a multitude of ways, whether this arises from the listener’s familiarity with the language or dialect spoken, or existing linguistic relationships between spoken words. Specifically, phonological and semantic relationships between spoken words have been observed to influence a listeners’ ability to discriminate voices. This study aims to develop our understanding of how different kinds of linguistic relationships, namely, phonetic relationships, influence the processing of speaker information. We conducted two experiments, a voice discrimination task and a voice similarity rating task, in which listeners were presented with pairs of speakers producing two words with various degrees of phonetic overlap. On the whole, higher quantities of phonetic overlap corresponded to higher speaker discrimination performance and higher similarity scores; however, the type of the phonetic overlap also mattered. Overlapping vowel segments showed substantial utility, while overlap of the phonological rhyme alone substantially lower performance. Results from this condition suggest that a phonological relationship within the word pair can interfere with otherwise increased quantities of phonetic overlap. These findings highlight the salience of the phonological rhyme in voice processing, as well as the overall impact of phonetic overlap. CV 2025 Acoustical Society of America.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2024\n \n \n (8)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Leveraging self-supervised models for automatic whispered speech recognition.\n \n \n \n\n\n \n Farhadipour, A.; Asadi, H.; and Dellwo, V.\n\n\n \n\n\n\n In 2024 14th international conference on computer and knowledge engineering (ICCKE), pages 188–193, 2024. IEEE\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{farhadipour2024leveraging,\n\ttitle = {Leveraging self-supervised models for automatic whispered speech recognition},\n\tbooktitle = {2024 14th international conference on computer and knowledge engineering ({ICCKE})},\n\tpublisher = {IEEE},\n\tauthor = {Farhadipour, Aref and Asadi, Homa and Dellwo, Volker},\n\tyear = {2024},\n\tpages = {188--193},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n NumberLie: a game-based experiment to understand the acoustics of deception and truthfulness.\n \n \n \n\n\n \n De Luca, A.; Clark, A.; and Dellwo, V.\n\n\n \n\n\n\n In Proc. Interspeech 2024, pages 3659–3663, 2024. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{de2024numberlie,\n\ttitle = {{NumberLie}: a game-based experiment to understand the acoustics of deception and truthfulness},\n\tbooktitle = {Proc. {Interspeech} 2024},\n\tauthor = {De Luca, Alessandro and Clark, Andrew and Dellwo, Volker},\n\tyear = {2024},\n\tpages = {3659--3663},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Temporal co-registration of simultaneous electromagnetic articulography and electroencephalography for precise articulatory and neural data alignment.\n \n \n \n\n\n \n Friedrichs, D.; Lancheros, M.; Kirkham, S.; He, L.; Clark, A.; Lutz, C.; Dellwo, V.; and Moran, S.\n\n\n \n\n\n\n In Interspeech, 2024. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{friedrichs2024temporal,\n\ttitle = {Temporal co-registration of simultaneous electromagnetic articulography and electroencephalography for precise articulatory and neural data alignment.},\n\tbooktitle = {Interspeech},\n\tauthor = {Friedrichs, Daniel and Lancheros, Monica and Kirkham, Sam and He, Lei and Clark, Andrew and Lutz, Clemens and Dellwo, Volker and Moran, Steven},\n\tyear = {2024},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Comparative analysis of modality fusion approaches for audio-visual person identification and verification.\n \n \n \n\n\n \n Farhadipour, A.; Chapariniya, M.; Vuković, T.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the 7th international conference on natural language and speech processing (ICNLSP 2024), pages 168–177, 2024. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{farhadipour2024comparative,\n\ttitle = {Comparative analysis of modality fusion approaches for audio-visual person identification and verification},\n\tbooktitle = {Proceedings of the 7th international conference on natural language and speech processing ({ICNLSP} 2024)},\n\tauthor = {Farhadipour, Aref and Chapariniya, Masoumeh and Vuković, Teodora and Dellwo, Volker},\n\tyear = {2024},\n\tpages = {168--177},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Shouting affects temporal properties of the speech amplitude envelope.\n \n \n \n \n\n\n \n Dimos, K.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n JASA Express Letters, 4(1): 015202. January 2024.\n \n\n\n\n
\n\n\n\n \n \n \"ShoutingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dimosShoutingAffectsTemporal2024a,\n\ttitle = {Shouting affects temporal properties of the speech amplitude envelope},\n\tvolume = {4},\n\tcopyright = {All rights reserved},\n\tissn = {2691-1191},\n\turl = {https://pubs.aip.org/jel/article/4/1/015202/2932311/Shouting-affects-temporal-properties-of-the-speech},\n\tdoi = {10.1121/10.0023995},\n\tabstract = {Distinguishing shouted from non-shouted speech is crucial in communication. We examined how shouting affects temporal properties of the amplitude envelope (ENV) in a total of 720 sentences read by 18 Swiss German speakers in normal and shouted modes; shouting was characterised by maintaining sound pressure levels of 80 dB sound pressure level (dB-SPL) (C-weighted) at a 1-meter distance from the mouth. Generalized additive models revealed significant temporal alterations of ENV in shouted speech, marked by steeper ascent, delayed peak, and extended high levels. These findings offer potential cues for identifying shouting, particularly useful when fine-structure and dynamic range cues are absent, for example, in cochlear implant users. CV 2024 Author(s). All article content, except where otherwise noted, is licensed under a Creative Commons Attribution (CC BY) license (http://creativecommons.org/licenses/by/4.0/).},\n\tlanguage = {en},\n\tnumber = {1},\n\turldate = {2024-11-18},\n\tjournal = {JASA Express Letters},\n\tauthor = {Dimos, Kostis and He, Lei and Dellwo, Volker},\n\tmonth = jan,\n\tyear = {2024},\n\tpages = {015202},\n}\n\n\n\n
\n
\n\n\n
\n Distinguishing shouted from non-shouted speech is crucial in communication. We examined how shouting affects temporal properties of the amplitude envelope (ENV) in a total of 720 sentences read by 18 Swiss German speakers in normal and shouted modes; shouting was characterised by maintaining sound pressure levels of 80 dB sound pressure level (dB-SPL) (C-weighted) at a 1-meter distance from the mouth. Generalized additive models revealed significant temporal alterations of ENV in shouted speech, marked by steeper ascent, delayed peak, and extended high levels. These findings offer potential cues for identifying shouting, particularly useful when fine-structure and dynamic range cues are absent, for example, in cochlear implant users. CV 2024 Author(s). All article content, except where otherwise noted, is licensed under a Creative Commons Attribution (CC BY) license (http://creativecommons.org/licenses/by/4.0/).\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaking the language of voice recognition.\n \n \n \n \n\n\n \n EUresearch\n\n\n \n\n\n\n EU Research, Summer 2024(38): 33–34. July 2024.\n \n\n\n\n
\n\n\n\n \n \n \"SpeakingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{euresearchSpeakingLanguageVoice2024a,\n\ttitle = {Speaking the language of voice recognition},\n\tvolume = {Summer 2024},\n\tcopyright = {CC0 1.0 Universal Public Domain Dedication},\n\tissn = {27524728, 27524736},\n\turl = {https://issuu.com/euresearcher/docs/the_dynamics_of_indexical_information_in_speech_eu},\n\tdoi = {10.56181//EEQL7196},\n\tabstract = {Voice recognition processes are fundamental to human social interaction, enabling us to rapidly identify a speaker and participate in discourse with other people. While previously a common assumption was that we recognise speakers on the basis of static features in their voice, Professor Volker Dellwo and his colleagues in an SNSF-funded project are now exploring a different hypothesis.},\n\tlanguage = {en},\n\tnumber = {38},\n\turldate = {2024-11-18},\n\tjournal = {EU Research},\n\tauthor = {{EUresearch}},\n\tmonth = jul,\n\tyear = {2024},\n\tpages = {33--34},\n}\n\n\n\n
\n
\n\n\n
\n Voice recognition processes are fundamental to human social interaction, enabling us to rapidly identify a speaker and participate in discourse with other people. While previously a common assumption was that we recognise speakers on the basis of static features in their voice, Professor Volker Dellwo and his colleagues in an SNSF-funded project are now exploring a different hypothesis.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Cortical-striatal brain network distinguishes deepfake from real speaker identity.\n \n \n \n \n\n\n \n Roswandowitz, C.; Kathiresan, T.; Pellegrino, E.; Dellwo, V.; and Frühholz, S.\n\n\n \n\n\n\n Communications Biology, 7(1): 711. June 2024.\n \n\n\n\n
\n\n\n\n \n \n \"Cortical-striatalPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{roswandowitzCorticalstriatalBrainNetwork2024a,\n\ttitle = {Cortical-striatal brain network distinguishes deepfake from real speaker identity},\n\tvolume = {7},\n\tcopyright = {All rights reserved},\n\tissn = {2399-3642},\n\turl = {https://www.nature.com/articles/s42003-024-06372-6},\n\tdoi = {10.1038/s42003-024-06372-6},\n\tabstract = {Abstract\n            Deepfakes are viral ingredients of digital environments, and they can trick human cognition into misperceiving the fake as real. Here, we test the neurocognitive sensitivity of 25 participants to accept or reject person identities as recreated in audio deepfakes. We generate high-quality voice identity clones from natural speakers by using advanced deepfake technologies. During an identity matching task, participants show intermediate performance with deepfake voices, indicating levels of deception and resistance to deepfake identity spoofing. On the brain level, univariate and multivariate analyses consistently reveal a central cortico-striatal network that decoded the vocal acoustic pattern and deepfake-level (auditory cortex), as well as natural speaker identities (nucleus accumbens), which are valued for their social relevance. This network is embedded in a broader neural identity and object recognition network. Humans can thus be partly tricked by deepfakes, but the neurocognitive mechanisms identified during deepfake processing open windows for strengthening human resilience to fake information.},\n\tlanguage = {en},\n\tnumber = {1},\n\turldate = {2024-10-07},\n\tjournal = {Communications Biology},\n\tauthor = {Roswandowitz, Claudia and Kathiresan, Thayabaran and Pellegrino, Elisa and Dellwo, Volker and Frühholz, Sascha},\n\tmonth = jun,\n\tyear = {2024},\n\tpages = {711},\n}\n\n\n\n
\n
\n\n\n
\n Abstract Deepfakes are viral ingredients of digital environments, and they can trick human cognition into misperceiving the fake as real. Here, we test the neurocognitive sensitivity of 25 participants to accept or reject person identities as recreated in audio deepfakes. We generate high-quality voice identity clones from natural speakers by using advanced deepfake technologies. During an identity matching task, participants show intermediate performance with deepfake voices, indicating levels of deception and resistance to deepfake identity spoofing. On the brain level, univariate and multivariate analyses consistently reveal a central cortico-striatal network that decoded the vocal acoustic pattern and deepfake-level (auditory cortex), as well as natural speaker identities (nucleus accumbens), which are valued for their social relevance. This network is embedded in a broader neural identity and object recognition network. Humans can thus be partly tricked by deepfakes, but the neurocognitive mechanisms identified during deepfake processing open windows for strengthening human resilience to fake information.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Deep neural networks for automatic speaker recognition do not learn supra-segmental temporal features.\n \n \n \n \n\n\n \n Neururer, D.; Dellwo, V.; and Stadelmann, T.\n\n\n \n\n\n\n Pattern Recognition Letters, 181: 64–69. May 2024.\n \n\n\n\n
\n\n\n\n \n \n \"DeepPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{neururerDeepNeuralNetworks2024a,\n\ttitle = {Deep neural networks for automatic speaker recognition do not learn supra-segmental temporal features},\n\tvolume = {181},\n\tcopyright = {All rights reserved},\n\tissn = {01678655},\n\turl = {https://linkinghub.elsevier.com/retrieve/pii/S0167865524000849},\n\tdoi = {10.1016/j.patrec.2024.03.016},\n\tabstract = {While deep neural networks have shown impressive results in automatic speaker recognition and related tasks, it is dissatisfactory how little is understood about what exactly is responsible for these results. Part of the success has been attributed in prior work to their capability to model supra-segmental temporal information (SST), i.e., learn rhythmic-prosodic characteristics of speech in addition to spectral features. In this paper, we (i) present and apply a novel test to quantify to what extent the performance of state-of-the-art neural networks for speaker recognition can be explained by modeling SST; and (ii) present several means to force respective nets to focus more on SST and evaluate their merits. We find that a variety of CNN- and RNN-based neural network architectures for speaker recognition do not model SST to any sufficient degree, even when forced. The results provide a highly relevant basis for impactful future research into better exploitation of the full speech signal and give insights into the inner workings of such networks, enhancing explainability of deep learning for speech technologies.},\n\tlanguage = {en},\n\turldate = {2024-11-18},\n\tjournal = {Pattern Recognition Letters},\n\tauthor = {Neururer, Daniel and Dellwo, Volker and Stadelmann, Thilo},\n\tmonth = may,\n\tyear = {2024},\n\tpages = {64--69},\n}\n\n\n\n
\n
\n\n\n
\n While deep neural networks have shown impressive results in automatic speaker recognition and related tasks, it is dissatisfactory how little is understood about what exactly is responsible for these results. Part of the success has been attributed in prior work to their capability to model supra-segmental temporal information (SST), i.e., learn rhythmic-prosodic characteristics of speech in addition to spectral features. In this paper, we (i) present and apply a novel test to quantify to what extent the performance of state-of-the-art neural networks for speaker recognition can be explained by modeling SST; and (ii) present several means to force respective nets to focus more on SST and evaluate their merits. We find that a variety of CNN- and RNN-based neural network architectures for speaker recognition do not model SST to any sufficient degree, even when forced. The results provide a highly relevant basis for impactful future research into better exploitation of the full speech signal and give insights into the inner workings of such networks, enhancing explainability of deep learning for speech technologies.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2023\n \n \n (10)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Acoustic compression in Zoom audio does not compromise voice recognition performance.\n \n \n \n\n\n \n Perepelytsia, V.; and Dellwo, V.\n\n\n \n\n\n\n Scientific Reports, 13(1): 18742. 2023.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{perepelytsia2023acoustic,\n\ttitle = {Acoustic compression in {Zoom} audio does not compromise voice recognition performance},\n\tvolume = {13},\n\tnumber = {1},\n\tjournal = {Scientific Reports},\n\tpublisher = {Nature Publishing Group UK London},\n\tauthor = {Perepelytsia, Valeriia and Dellwo, Volker},\n\tyear = {2023},\n\tpages = {18742},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speaker idiosyncratic intensity and mouth opening-closing variations: The case of English.\n \n \n \n\n\n \n Zhang, Y.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the 20th international congress of phonetic sciences (ICPhS), pages 7–11, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{zhang2023speaker,\n\ttitle = {Speaker idiosyncratic intensity and mouth opening-closing variations: {The} case of {English}},\n\tbooktitle = {Proceedings of the 20th international congress of phonetic sciences ({ICPhS})},\n\tauthor = {Zhang, Yu and He, Lei and Dellwo, Volker},\n\tyear = {2023},\n\tpages = {7--11},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Voice discrimination across speaking styles in Persian.\n \n \n \n\n\n \n Pellegrino, E.; Asadi, H.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the international congress of phonetic sciences. International phonetic association. https://www. internationalphoneticassociation. org/icphs-proceedings/ICPhS2023/full_papers/952. pdf, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{pellegrino2023voice,\n\ttitle = {Voice discrimination across speaking styles in {Persian}},\n\tbooktitle = {Proceedings of the international congress of phonetic sciences. {International} phonetic association. https://www. internationalphoneticassociation. org/icphs-proceedings/{ICPhS2023}/full\\_papers/952. pdf},\n\tauthor = {Pellegrino, Elisa and Asadi, Homa and Dellwo, Volker},\n\tyear = {2023},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n L2 phonological contrasts are not necessarily lost: Evidence from french listeners detecting mandarin tones.\n \n \n \n\n\n \n Dellwo, V.; Schwab, S.; and Shi, R.\n\n\n \n\n\n\n In 20th international congress of phonetic sciences, pages 1896–1900, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwo2023l2,\n\ttitle = {L2 phonological contrasts are not necessarily lost: {Evidence} from french listeners detecting mandarin tones},\n\tbooktitle = {20th international congress of phonetic sciences},\n\tauthor = {Dellwo, Volker and Schwab, Sandra and Shi, Rushen},\n\tyear = {2023},\n\tpages = {1896--1900},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Idear: a speech database of identity-marked, clear, and read speech.\n \n \n \n\n\n \n Perepelytsia, V.; Bradshaw, L.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of ICPhS, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{perepelytsia2023idear,\n\ttitle = {Idear: a speech database of identity-marked, clear, and read speech},\n\tbooktitle = {Proceedings of {ICPhS}},\n\tauthor = {Perepelytsia, Valeriia and Bradshaw, Leah and Dellwo, Volker},\n\tyear = {2023},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Vocal effort in human interactions with voice-AI.\n \n \n \n\n\n \n Bradshaw, L.; Perepelytsia, V.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of ICPhS, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{bradshaw2023vocal,\n\ttitle = {Vocal effort in human interactions with voice-{AI}},\n\tbooktitle = {Proceedings of {ICPhS}},\n\tauthor = {Bradshaw, Leah and Perepelytsia, Valeriia and Dellwo, Volker},\n\tyear = {2023},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n A longitudinal study of individual difference in foreign language pronunciation development: The case of vowel production in Ecuadorian learners of English.\n \n \n \n\n\n \n Pesantez, A.; and Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the 7th international conference on english pronunciation: Issues and practices, pages 214–224, 2023. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{pesantez2023longitudinal,\n\ttitle = {A longitudinal study of individual difference in foreign language pronunciation development: {The} case of vowel production in {Ecuadorian} learners of {English}},\n\tbooktitle = {Proceedings of the 7th international conference on english pronunciation: {Issues} and practices},\n\tauthor = {Pesantez, Alejandra and Dellwo, Volker},\n\tyear = {2023},\n\tpages = {214--224},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speakers are more cooperative and less individual when interacting in larger group sizes.\n \n \n \n\n\n \n Pellegrino, E.; and Dellwo, V.\n\n\n \n\n\n\n Frontiers in Psychology, 14: 1145572. 2023.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrino2023speakers,\n\ttitle = {Speakers are more cooperative and less individual when interacting in larger group sizes},\n\tvolume = {14},\n\tjournal = {Frontiers in Psychology},\n\tpublisher = {Frontiers Media SA},\n\tauthor = {Pellegrino, Elisa and Dellwo, Volker},\n\tyear = {2023},\n\tpages = {1145572},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speakers are more cooperative and less individual when interacting in larger group sizes.\n \n \n \n \n\n\n \n Pellegrino, E.; and Dellwo, V.\n\n\n \n\n\n\n Frontiers in Psychology, 14: 1145572. June 2023.\n \n\n\n\n
\n\n\n\n \n \n \"SpeakersPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoSpeakersAreMore2023a,\n\ttitle = {Speakers are more cooperative and less individual when interacting in larger group sizes},\n\tvolume = {14},\n\tcopyright = {All rights reserved},\n\tissn = {1664-1078},\n\turl = {https://www.frontiersin.org/articles/10.3389/fpsyg.2023.1145572/full},\n\tdoi = {10.3389/fpsyg.2023.1145572},\n\tabstract = {Introduction\n              Cooperation, acoustically signaled through vocal convergence, is facilitated when group members are more similar. Excessive vocal convergence may, however, weaken individual recognizability. This study aimed to explore whether constraints to convergence can arise in circumstances where interlocutors need to enhance their vocal individuality. Therefore, we tested the effects of group size (3 and 5 interactants) on vocal convergence and individualization in a social communication scenario in which individual recognition by voice is at stake.\n            \n            \n              Methods\n              In an interactive game, players had to recognize each other through their voices while solving a cooperative task online. The vocal similarity was quantified through similarities in speaker i-vectors obtained through probabilistic linear discriminant analysis (PLDA). Speaker recognition performance was measured through the system Equal Error Rate (EER).\n            \n            \n              Results\n              Vocal similarity between-speakers increased with a larger group size which indicates a higher cooperative vocal behavior. At the same time, there was an increase in EER for the same speakers between the smaller and the larger group size, meaning a decrease in overall recognition performance.\n            \n            \n              Discussion\n              The decrease in vocal individualization in the larger group size suggests that ingroup cooperation and social cohesion conveyed through acoustic convergence have priority over individualization in larger groups of unacquainted speakers.},\n\tlanguage = {en},\n\turldate = {2024-11-18},\n\tjournal = {Frontiers in Psychology},\n\tauthor = {Pellegrino, Elisa and Dellwo, Volker},\n\tmonth = jun,\n\tyear = {2023},\n\tpages = {1145572},\n}\n\n\n\n
\n
\n\n\n
\n Introduction Cooperation, acoustically signaled through vocal convergence, is facilitated when group members are more similar. Excessive vocal convergence may, however, weaken individual recognizability. This study aimed to explore whether constraints to convergence can arise in circumstances where interlocutors need to enhance their vocal individuality. Therefore, we tested the effects of group size (3 and 5 interactants) on vocal convergence and individualization in a social communication scenario in which individual recognition by voice is at stake. Methods In an interactive game, players had to recognize each other through their voices while solving a cooperative task online. The vocal similarity was quantified through similarities in speaker i-vectors obtained through probabilistic linear discriminant analysis (PLDA). Speaker recognition performance was measured through the system Equal Error Rate (EER). Results Vocal similarity between-speakers increased with a larger group size which indicates a higher cooperative vocal behavior. At the same time, there was an increase in EER for the same speakers between the smaller and the larger group size, meaning a decrease in overall recognition performance. Discussion The decrease in vocal individualization in the larger group size suggests that ingroup cooperation and social cohesion conveyed through acoustic convergence have priority over individualization in larger groups of unacquainted speakers.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Acoustic compression in Zoom audio does not compromise voice recognition performance.\n \n \n \n \n\n\n \n Perepelytsia, V.; and Dellwo, V.\n\n\n \n\n\n\n Scientific Reports, 13(1): 18742. October 2023.\n \n\n\n\n
\n\n\n\n \n \n \"AcousticPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{perepelytsiaAcousticCompressionZoom2023a,\n\ttitle = {Acoustic compression in {Zoom} audio does not compromise voice recognition performance},\n\tvolume = {13},\n\tissn = {2045-2322},\n\turl = {https://www.nature.com/articles/s41598-023-45971-x},\n\tdoi = {10.1038/s41598-023-45971-x},\n\tabstract = {Abstract\n            Human voice recognition over telephone channels typically yields lower accuracy when compared to audio recorded in a studio environment with higher quality. Here, we investigated the extent to which audio in video conferencing, subject to various lossy compression mechanisms, affects human voice recognition performance. Voice recognition performance was tested in an old–new recognition task under three audio conditions (telephone, Zoom, studio) across all matched (familiarization and test with same audio condition) and mismatched combinations (familiarization and test with different audio conditions). Participants were familiarized with female voices presented in either studio-quality (N = 22), Zoom-quality (N = 21), or telephone-quality (N = 20) stimuli. Subsequently, all listeners performed an identical voice recognition test containing a balanced stimulus set from all three conditions. Results revealed that voice recognition performance (dʹ) in Zoom audio was not significantly different to studio audio but both in Zoom and studio audio listeners performed significantly better compared to telephone audio. This suggests that signal processing of the speech codec used by Zoom provides equally relevant information in terms of voice recognition compared to studio audio. Interestingly, listeners familiarized with voices via Zoom audio showed a trend towards a better recognition performance in the test (p = 0.056) compared to listeners familiarized with studio audio. We discuss future directions according to which a possible advantage of Zoom audio for voice recognition might be related to some of the speech coding mechanisms used by Zoom.},\n\tlanguage = {en},\n\tnumber = {1},\n\turldate = {2024-11-18},\n\tjournal = {Scientific Reports},\n\tauthor = {Perepelytsia, Valeriia and Dellwo, Volker},\n\tmonth = oct,\n\tyear = {2023},\n\tpages = {18742},\n}\n\n\n\n
\n
\n\n\n
\n Abstract Human voice recognition over telephone channels typically yields lower accuracy when compared to audio recorded in a studio environment with higher quality. Here, we investigated the extent to which audio in video conferencing, subject to various lossy compression mechanisms, affects human voice recognition performance. Voice recognition performance was tested in an old–new recognition task under three audio conditions (telephone, Zoom, studio) across all matched (familiarization and test with same audio condition) and mismatched combinations (familiarization and test with different audio conditions). Participants were familiarized with female voices presented in either studio-quality (N = 22), Zoom-quality (N = 21), or telephone-quality (N = 20) stimuli. Subsequently, all listeners performed an identical voice recognition test containing a balanced stimulus set from all three conditions. Results revealed that voice recognition performance (dʹ) in Zoom audio was not significantly different to studio audio but both in Zoom and studio audio listeners performed significantly better compared to telephone audio. This suggests that signal processing of the speech codec used by Zoom provides equally relevant information in terms of voice recognition compared to studio audio. Interestingly, listeners familiarized with voices via Zoom audio showed a trend towards a better recognition performance in the test (p = 0.056) compared to listeners familiarized with studio audio. We discuss future directions according to which a possible advantage of Zoom audio for voice recognition might be related to some of the speech coding mechanisms used by Zoom.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2022\n \n \n (13)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Vocal accommodation in speech communication.\n \n \n \n\n\n \n Pardo, J. S; Pellegrino, E.; Dellwo, V.; and Möbius, B.\n\n\n \n\n\n\n Journal of Phonetics, 95: 101196. 2022.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pardo2022vocal,\n\ttitle = {Vocal accommodation in speech communication},\n\tvolume = {95},\n\tjournal = {Journal of Phonetics},\n\tpublisher = {Academic Press},\n\tauthor = {Pardo, Jennifer S and Pellegrino, Elisa and Dellwo, Volker and Möbius, Bernd},\n\tyear = {2022},\n\tpages = {101196},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Vowel convergence does not affect auditory speaker discriminability in humans and machine in a case study on Swiss German dialects.\n \n \n \n\n\n \n Pellegrino, E.; Kathiresan, T.; and Dellwo, V.\n\n\n \n\n\n\n The International Journal of Speech, Language and the Law, 29(1): 60–84. 2022.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrino2022vowel,\n\ttitle = {Vowel convergence does not affect auditory speaker discriminability in humans and machine in a case study on {Swiss} {German} dialects},\n\tvolume = {29},\n\tnumber = {1},\n\tjournal = {The International Journal of Speech, Language and the Law},\n\tpublisher = {Equinox Publishing Ltd.},\n\tauthor = {Pellegrino, Elisa and Kathiresan, Thayabaran and Dellwo, Volker},\n\tyear = {2022},\n\tpages = {60--84},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Idiosyncratic lingual articulation of American English/æ/and/ɑ/using network analysis.\n \n \n \n\n\n \n Machado, C. L.; Dellwo, V.; and He, L.\n\n\n \n\n\n\n In Proc. Interspeech 2022, pages 754–758, 2022. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{machado2022idiosyncratic,\n\ttitle = {Idiosyncratic lingual articulation of {American} {English}/æ/and/ɑ/using network analysis},\n\tbooktitle = {Proc. {Interspeech} 2022},\n\tauthor = {Machado, Carolina Lins and Dellwo, Volker and He, Lei},\n\tyear = {2022},\n\tpages = {754--758},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Fundamental frequency variability over time in telephone interactions.\n \n \n \n\n\n \n Bradshaw, L.; Chodroff, E.; Jäger, L.; and Dellwo, V.\n\n\n \n\n\n\n In Proc. Interspeech 2022, pages 101–105, 2022. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{bradshaw2022fundamental,\n\ttitle = {Fundamental frequency variability over time in telephone interactions},\n\tbooktitle = {Proc. {Interspeech} 2022},\n\tauthor = {Bradshaw, Leah and Chodroff, Eleanor and Jäger, Lena and Dellwo, Volker},\n\tyear = {2022},\n\tpages = {101--105},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Stress crossover in intimate relationships: A new framework for studying dynamic co-regulation patterns in dyadic interactions.\n \n \n \n\n\n \n Hilpert, P.; Butner, J.; Atkins, D.; Baucom, B.; Dellwo, V.; Bodenmann, G.; and Bradbury, T.\n\n\n \n\n\n\n . 2022.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{hilpert2022stress,\n\ttitle = {Stress crossover in intimate relationships: {A} new framework for studying dynamic co-regulation patterns in dyadic interactions},\n\tauthor = {Hilpert, Peter and Butner, JE and Atkins, DC and Baucom, Brian and Dellwo, Volker and Bodenmann, Guy and Bradbury, TN},\n\tyear = {2022},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Explicit versus non-explicit prosodic training in the learning of Spanish L2 stress contrasts by French listeners.\n \n \n \n\n\n \n Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Second Language Studies, 5(2): 266–306. 2022.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schwab2022explicit,\n\ttitle = {Explicit versus non-explicit prosodic training in the learning of {Spanish} {L2} stress contrasts by {French} listeners},\n\tvolume = {5},\n\tnumber = {2},\n\tjournal = {Journal of Second Language Studies},\n\tpublisher = {John Benjamins Publishing Company Amsterdam/Philadelphia},\n\tauthor = {Schwab, Sandra and Dellwo, Volker},\n\tyear = {2022},\n\tpages = {266--306},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Fundamental Frequency Variability over Time in Telephone Interactions.\n \n \n \n \n\n\n \n Bradshaw, L.; Chodroff, E.; Jäger, L.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2022, pages 101–105, September 2022. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"FundamentalPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{bradshawFundamentalFrequencyVariability2022a,\n\ttitle = {Fundamental {Frequency} {Variability} over {Time} in {Telephone} {Interactions}},\n\turl = {https://www.isca-archive.org/interspeech_2022/bradshaw22_interspeech.html},\n\tdoi = {10.21437/Interspeech.2022-10669},\n\tabstract = {Speech signals contain substantial fundamental frequency (f0) variability. Even within a single utterance, speakers modify f0 to create different intonational patterns. Previous studies have identified markers of increased f0 variability, such as the introduction of a new topic or greetings, but these are limited in the scope of their analyses. In the present study, we investigate f0 variability over the course of a telephone conversation, with a focus on the initial and medial utterances within the exchange. We examined f0 standard deviation of each utterance in over 2000 telephone conversations from 509 American English speakers from the Switchboard corpus. Findings showed that on average, speakers exhibit more f0 variability in the opening compared to mid-conversation utterances. Further, findings suggest that the inclusion of a greeting word in an initial turn, e.g., “hello” or “hi”, corresponds to an increase in f0 standard deviation. These results suggest that speakers employed more variable f0 in the initial few turns of a telephone conversation. The interpretation of this finding is multifaceted and may be linked to several communicative goals, including the placement of identity markers in conversation or the attraction of attention, or the role of openings as boundary markers.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2022},\n\tpublisher = {ISCA},\n\tauthor = {Bradshaw, Leah and Chodroff, Eleanor and Jäger, Lena and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2022},\n\tpages = {101--105},\n}\n\n\n\n
\n
\n\n\n
\n Speech signals contain substantial fundamental frequency (f0) variability. Even within a single utterance, speakers modify f0 to create different intonational patterns. Previous studies have identified markers of increased f0 variability, such as the introduction of a new topic or greetings, but these are limited in the scope of their analyses. In the present study, we investigate f0 variability over the course of a telephone conversation, with a focus on the initial and medial utterances within the exchange. We examined f0 standard deviation of each utterance in over 2000 telephone conversations from 509 American English speakers from the Switchboard corpus. Findings showed that on average, speakers exhibit more f0 variability in the opening compared to mid-conversation utterances. Further, findings suggest that the inclusion of a greeting word in an initial turn, e.g., “hello” or “hi”, corresponds to an increase in f0 standard deviation. These results suggest that speakers employed more variable f0 in the initial few turns of a telephone conversation. The interpretation of this finding is multifaceted and may be linked to several communicative goals, including the placement of identity markers in conversation or the attraction of attention, or the role of openings as boundary markers.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Degraded and computer-generated speech processing in a bonobo.\n \n \n \n\n\n \n Lahiff, N. J.; Slocombe, K. E.; Taglialatela, J.; Dellwo, V.; and Townsend, S. W.\n\n\n \n\n\n\n Animal Cognition. 2022.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{lahiffDegradedComputergeneratedSpeech2022a,\n\ttitle = {Degraded and computer-generated speech processing in a bonobo},\n\tissn = {14359456},\n\tdoi = {10.1007/S10071-022-01621-9},\n\tabstract = {The human auditory system is capable of processing human speech even in situations when it has been heavily degraded, such as during noise-vocoding, when frequency domain-based cues to phonetic content are strongly reduced. This has contributed to arguments that speech processing is highly specialized and likely a de novo evolved trait in humans. Previous comparative research has demonstrated that a language competent chimpanzee was also capable of recognizing degraded speech, and therefore that the mechanisms underlying speech processing may not be uniquely human. However, to form a robust reconstruction of the evolutionary origins of speech processing, additional data from other closely related ape species is needed. Specifically, such data can help disentangle whether these capabilities evolved independently in humans and chimpanzees, or if they were inherited from our last common ancestor. Here we provide evidence of processing of highly varied (degraded and computer-generated) speech in a language competent bonobo, Kanzi. We took advantage of Kanzi’s existing proficiency with touchscreens and his ability to report his understanding of human speech through interacting with arbitrary symbols called lexigrams. Specifically, we asked Kanzi to recognise both human (natural) and computer-generated forms of 40 highly familiar words that had been degraded (noise-vocoded and sinusoidal forms) using a match-to-sample paradigm. Results suggest that—apart from noise-vocoded computer-generated speech—Kanzi recognised both natural and computer-generated voices that had been degraded, at rates significantly above chance. Kanzi performed better with all forms of natural voice speech compared to computer-generated speech. This work provides additional support for the hypothesis that the processing apparatus necessary to deal with highly variable speech, including for the first time in nonhuman animals, computer-generated speech, may be at least as old as the last common ancestor we share with bonobos and chimpanzees.},\n\turldate = {2022-07-18},\n\tjournal = {Animal Cognition},\n\tpublisher = {Springer Science and Business Media Deutschland GmbH},\n\tauthor = {Lahiff, Nicole J. and Slocombe, Katie E. and Taglialatela, Jared and Dellwo, Volker and Townsend, Simon W.},\n\tyear = {2022},\n}\n\n\n\n
\n
\n\n\n
\n The human auditory system is capable of processing human speech even in situations when it has been heavily degraded, such as during noise-vocoding, when frequency domain-based cues to phonetic content are strongly reduced. This has contributed to arguments that speech processing is highly specialized and likely a de novo evolved trait in humans. Previous comparative research has demonstrated that a language competent chimpanzee was also capable of recognizing degraded speech, and therefore that the mechanisms underlying speech processing may not be uniquely human. However, to form a robust reconstruction of the evolutionary origins of speech processing, additional data from other closely related ape species is needed. Specifically, such data can help disentangle whether these capabilities evolved independently in humans and chimpanzees, or if they were inherited from our last common ancestor. Here we provide evidence of processing of highly varied (degraded and computer-generated) speech in a language competent bonobo, Kanzi. We took advantage of Kanzi’s existing proficiency with touchscreens and his ability to report his understanding of human speech through interacting with arbitrary symbols called lexigrams. Specifically, we asked Kanzi to recognise both human (natural) and computer-generated forms of 40 highly familiar words that had been degraded (noise-vocoded and sinusoidal forms) using a match-to-sample paradigm. Results suggest that—apart from noise-vocoded computer-generated speech—Kanzi recognised both natural and computer-generated voices that had been degraded, at rates significantly above chance. Kanzi performed better with all forms of natural voice speech compared to computer-generated speech. This work provides additional support for the hypothesis that the processing apparatus necessary to deal with highly variable speech, including for the first time in nonhuman animals, computer-generated speech, may be at least as old as the last common ancestor we share with bonobos and chimpanzees.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Idiosyncratic lingual articulation of American English /æ/ and /ɑ/ using network analysis.\n \n \n \n \n\n\n \n Machado, C. L.; Dellwo, V.; and He, L.\n\n\n \n\n\n\n In Interspeech 2022, pages 754–758, September 2022. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"IdiosyncraticPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{machadoIdiosyncraticLingualArticulation2022a,\n\ttitle = {Idiosyncratic lingual articulation of {American} {English} /æ/ and /ɑ/ using network analysis},\n\turl = {https://www.isca-archive.org/interspeech_2022/machado22_interspeech.html},\n\tdoi = {10.21437/Interspeech.2022-10397},\n\tabstract = {Formant dynamics are believed to reflect the characteristic articulatory behavior of a speaker. The present study aims to explore individual articulatory behaviors when producing American English /æ/ and /ɑ/. The two vowels differ in the degree of inherent spectral change, a property believed to carry information about vowel-phoneme identity, which may be reflected in the articulatory movements. We measured first and second formants together with tongue blade and dorsum trajectories from 20 speakers producing 330 words in citation forms. Using the network analysis, the relationships between acoustic and kinematic variables were revealed. In particular, between-speaker articulatory behaviors were most dissimilar in /ɑ/ which requires less inherent spectral change. Moreover, when networks of speakers with similar formant patterns were compared, it was revealed that their articulatory behaviors also shared similarities, although they seemed to be organized in characteristic ways. These findings contribute to our understanding of the complex interaction between articulatory variables and the acoustic outcome.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2022},\n\tpublisher = {ISCA},\n\tauthor = {Machado, Carolina Lins and Dellwo, Volker and He, Lei},\n\tmonth = sep,\n\tyear = {2022},\n\tpages = {754--758},\n}\n\n\n\n
\n
\n\n\n
\n Formant dynamics are believed to reflect the characteristic articulatory behavior of a speaker. The present study aims to explore individual articulatory behaviors when producing American English /æ/ and /ɑ/. The two vowels differ in the degree of inherent spectral change, a property believed to carry information about vowel-phoneme identity, which may be reflected in the articulatory movements. We measured first and second formants together with tongue blade and dorsum trajectories from 20 speakers producing 330 words in citation forms. Using the network analysis, the relationships between acoustic and kinematic variables were revealed. In particular, between-speaker articulatory behaviors were most dissimilar in /ɑ/ which requires less inherent spectral change. Moreover, when networks of speakers with similar formant patterns were compared, it was revealed that their articulatory behaviors also shared similarities, although they seemed to be organized in characteristic ways. These findings contribute to our understanding of the complex interaction between articulatory variables and the acoustic outcome.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Vowel convergence does not affect auditory speaker discriminability in humans and machine in a case study on Swiss German dialects.\n \n \n \n\n\n \n Pellegrino, E.; Kathiresan, T.; and Dellwo, V.\n\n\n \n\n\n\n International Journal of Speech, Language and the Law. November 2022.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoVowelConvergenceDoes2022a,\n\ttitle = {Vowel convergence does not affect auditory speaker discriminability in humans and machine in a case study on {Swiss} {German} dialects},\n\tissn = {1748-8885},\n\tdoi = {10.1558/ijsll.19954},\n\tabstract = {In this study, we examined whether the convergence in interlocutors’ vowel acoustics leads to decreasing discriminability between interlocutors’ voices. Ten pairs of Grison and Zürich German speakers produced lexical items before and after dialogue interactions with evidence of vowel convergence in post-dialogue productions. In Experiment 1, native and non-native Swiss German listeners discriminated pairs of speakers whose speech was obtained pre- and post-dialogue. Results showed that listeners’ sensitivity (A’) was higher for native than non-native listeners, but comparable for pre- and post-dialogue recordings. The observed negative correlation between voice discrimination and the acoustic distance in formant space was mainly driven by a single speaker pair. In Experiment 2, the speaker recognition performance of an i-vector-based software was compared in pre- and post-dialogue speech. Results revealed no difference in the system performance between the two conditions. The findings suggest that vowel convergence does not compromise voice discriminability under the given experimental conditions.},\n\tjournal = {International Journal of Speech, Language and the Law},\n\tpublisher = {Equinox Publishing},\n\tauthor = {Pellegrino, Elisa and Kathiresan, Thayabaran and Dellwo, Volker},\n\tmonth = nov,\n\tyear = {2022},\n}\n\n\n\n
\n
\n\n\n
\n In this study, we examined whether the convergence in interlocutors’ vowel acoustics leads to decreasing discriminability between interlocutors’ voices. Ten pairs of Grison and Zürich German speakers produced lexical items before and after dialogue interactions with evidence of vowel convergence in post-dialogue productions. In Experiment 1, native and non-native Swiss German listeners discriminated pairs of speakers whose speech was obtained pre- and post-dialogue. Results showed that listeners’ sensitivity (A’) was higher for native than non-native listeners, but comparable for pre- and post-dialogue recordings. The observed negative correlation between voice discrimination and the acoustic distance in formant space was mainly driven by a single speaker pair. In Experiment 2, the speaker recognition performance of an i-vector-based software was compared in pre- and post-dialogue speech. Results revealed no difference in the system performance between the two conditions. The findings suggest that vowel convergence does not compromise voice discriminability under the given experimental conditions.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Special issue: Vocal accommodation in speech communication.\n \n \n \n \n\n\n \n Pardo, J. S.; Pellegrino, E.; Dellwo, V.; and Möbius, B.\n\n\n \n\n\n\n Journal of Phonetics, 95: 101196. 2022.\n \n\n\n\n
\n\n\n\n \n \n \"SpecialPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pardoSpecialIssueVocal2022a,\n\ttitle = {Special issue: {Vocal} accommodation in speech communication},\n\tvolume = {95},\n\tissn = {00954470},\n\turl = {https://doi.org/10.1016/j.wocn.2022.101196},\n\tdoi = {10.1016/j.wocn.2022.101196},\n\tabstract = {This introductory article for the Special Issue on Vocal Accommodation in Speech Communication provides an overview of prevailing theories of vocal accommodation and summarizes the ten papers in the collection. Communication Accommodation Theory focusses on social factors evoking accent convergence or divergence, while the Interactive Alignment Model proposes cognitive integration of perception and production as an automatic priming mechanism driving convergence language production. Recent research including most of the papers in this Special Issue indicates that a hybrid or interactive synergy model provides a more comprehensive account of observed patterns of phonetic convergence than purely automatic mechanisms. Some of the fundamental questions that this special collection aimed to cover concerned (1) the nature of vocal accommodation in terms of underlying mechanisms and social functions in human–human and human–computer interaction; (2) the effect of task-specific and talker-specific characteristics (gender, age, personality, linguistic and cultural background, role in interaction) on degree and direction of convergence towards human and computer interlocutors; (3) integration of articulatory, perceptual, neurocognitive, and/or multimodal data to the analysis of acoustic accommodation in interactive and non-interactive speech tasks; and (4) the contribution of short/long-term accommodation in human–human and human–computer interactions to the diffusion of linguistic innovation and ultimately language variation and change.},\n\tjournal = {Journal of Phonetics},\n\tpublisher = {Elsevier Ltd},\n\tauthor = {Pardo, Jennifer S. and Pellegrino, Elisa and Dellwo, Volker and Möbius, Bernd},\n\tyear = {2022},\n\tpages = {101196},\n}\n\n\n\n
\n
\n\n\n
\n This introductory article for the Special Issue on Vocal Accommodation in Speech Communication provides an overview of prevailing theories of vocal accommodation and summarizes the ten papers in the collection. Communication Accommodation Theory focusses on social factors evoking accent convergence or divergence, while the Interactive Alignment Model proposes cognitive integration of perception and production as an automatic priming mechanism driving convergence language production. Recent research including most of the papers in this Special Issue indicates that a hybrid or interactive synergy model provides a more comprehensive account of observed patterns of phonetic convergence than purely automatic mechanisms. Some of the fundamental questions that this special collection aimed to cover concerned (1) the nature of vocal accommodation in terms of underlying mechanisms and social functions in human–human and human–computer interaction; (2) the effect of task-specific and talker-specific characteristics (gender, age, personality, linguistic and cultural background, role in interaction) on degree and direction of convergence towards human and computer interlocutors; (3) integration of articulatory, perceptual, neurocognitive, and/or multimodal data to the analysis of acoustic accommodation in interactive and non-interactive speech tasks; and (4) the contribution of short/long-term accommodation in human–human and human–computer interactions to the diffusion of linguistic innovation and ultimately language variation and change.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n L2 phonetic accuracy development in a foreign language setting: A longitudinal study.\n \n \n \n \n\n\n \n Pesantez Pesantez, A. C.; and Dellwo, V.\n\n\n \n\n\n\n . April 2022.\n \n\n\n\n
\n\n\n\n \n \n \"L2Paper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pesantezpesantezL2PhoneticAccuracy2022,\n\ttitle = {L2 phonetic accuracy development in a foreign language setting: {A} longitudinal study},\n\tshorttitle = {L2 phonetic accuracy development in a foreign language setting},\n\turl = {https://www.zora.uzh.ch/id/eprint/237520},\n\tdoi = {10.5167/UZH-237520},\n\tabstract = {This study investigates second language phonetic development in the accuracy of the high front vowels producedby 21 Ecuadorian learners of English as a foreign language. The participants were enrolled in the First and ForeignLanguage Teaching program at the University of Cuenca-Ecuador. At the outset of the study, they were in the third semester (T1) and at the completion of it, they were in the fifth semester (T3) of the program.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {Universitat de Girona},\n\tauthor = {Pesantez Pesantez, Alejandra Carolina and Dellwo, Volker},\n\tmonth = apr,\n\tyear = {2022},\n}\n\n\n\n
\n
\n\n\n
\n This study investigates second language phonetic development in the accuracy of the high front vowels producedby 21 Ecuadorian learners of English as a foreign language. The participants were enrolled in the First and ForeignLanguage Teaching program at the University of Cuenca-Ecuador. At the outset of the study, they were in the third semester (T1) and at the completion of it, they were in the fifth semester (T3) of the program.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Mothers Reveal More of Their Vocal Identity When Talking to Infants.\n \n \n \n \n\n\n \n Kathiresan, T.; Hervais-Adelman, A.; Townsend, S. W.; Dilley, L.; Shi, R.; Daum, M.; and Dellwo, V.\n\n\n \n\n\n\n SSRN Electronic Journal. 2022.\n \n\n\n\n
\n\n\n\n \n \n \"MothersPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kathiresanMothersRevealMore2022,\n\ttitle = {Mothers {Reveal} {More} of {Their} {Vocal} {Identity} {When} {Talking} to {Infants}},\n\tissn = {1556-5068},\n\turl = {https://www.ssrn.com/abstract=4088888},\n\tdoi = {10.2139/ssrn.4088888},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tjournal = {SSRN Electronic Journal},\n\tauthor = {Kathiresan, Thayabaran and Hervais-Adelman, Alexis and Townsend, Simon William and Dilley, Laura and Shi, Rushen and Daum, Moritz and Dellwo, Volker},\n\tyear = {2022},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2021\n \n \n (5)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Do speakers converge rhythmically? A study on segmental timing properties of Grison and Zurich German before and after dialogical interactions.\n \n \n \n\n\n \n Pellegrino, E.; Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n Loquens, 8(1-2): e078–e078. 2021.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrino2021speakers,\n\ttitle = {Do speakers converge rhythmically? {A} study on segmental timing properties of {Grison} and {Zurich} {German} before and after dialogical interactions},\n\tvolume = {8},\n\tnumber = {1-2},\n\tjournal = {Loquens},\n\tauthor = {Pellegrino, Elisa and Schwab, Sandra and Dellwo, Volker},\n\tyear = {2021},\n\tpages = {e078--e078},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Native listeners rely on rhythmic cues when deciding on the nativeness of speech.\n \n \n \n\n\n \n Pellegrino, E.; Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 150(4): 2836–2853. 2021.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoNativeListenersRely2021a,\n\ttitle = {Native listeners rely on rhythmic cues when deciding on the nativeness of speech},\n\tvolume = {150},\n\tissn = {0001-4966},\n\tdoi = {10.1121/10.0006537},\n\tabstract = {Foreign-accented speech typically deviates segmentally and suprasegmentally from native-accented speech. Two experiments were conducted to investigate the role of amplitude envelope (ENV), segment duration (DUR), and speech rate (SR) on Italian listeners' ability to identify native-accented Italian in utterances produced by Zurich German speakers. In experiment 1, listeners judged in a two-alternative forced-choice perception task which of the two stimuli in a trial they perceived as more native-like. Stimuli in each trial only varied in ENV and DUR, which were retrieved either from a native Italian speaker [first language (L1) donor] or from a German speaker of Italian [second language (L2) donor]. Results revealed that listeners make use of both DUR and ENV to identify the more native-like stimuli, but the effect of ENV was more subtle. In experiment 2, SR differences (resulting from native and non-native segment duration differences in experiment 1) were normalized for. It was found that this drastically reduced the effect of segment durations in terms of perceived nativeness; however, the ENV effect still remained. This was not the case in a control group of listeners without competence in Italian. Though effects were subtle, the study shows that ENV cues contribute to the percept of nativeness in L2 speech.(c) 2021 Acoustical Society of America PU  - ACOUSTICAL SOC AMER AMER INST PHYSICS PI  - MELVILLE PA  - STE 1 NO 1, 2 HUNTINGTON QUADRANGLE, MELVILLE, NY 11747-4502 USA},\n\tnumber = {4},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {Pellegrino, Elisa and Schwab, Sandra and Dellwo, Volker},\n\tyear = {2021},\n\tpages = {2836--2853},\n}\n\n\n\n
\n
\n\n\n
\n Foreign-accented speech typically deviates segmentally and suprasegmentally from native-accented speech. Two experiments were conducted to investigate the role of amplitude envelope (ENV), segment duration (DUR), and speech rate (SR) on Italian listeners' ability to identify native-accented Italian in utterances produced by Zurich German speakers. In experiment 1, listeners judged in a two-alternative forced-choice perception task which of the two stimuli in a trial they perceived as more native-like. Stimuli in each trial only varied in ENV and DUR, which were retrieved either from a native Italian speaker [first language (L1) donor] or from a German speaker of Italian [second language (L2) donor]. Results revealed that listeners make use of both DUR and ENV to identify the more native-like stimuli, but the effect of ENV was more subtle. In experiment 2, SR differences (resulting from native and non-native segment duration differences in experiment 1) were normalized for. It was found that this drastically reduced the effect of segment durations in terms of perceived nativeness; however, the ENV effect still remained. This was not the case in a control group of listeners without competence in Italian. Though effects were subtle, the study shows that ENV cues contribute to the percept of nativeness in L2 speech.(c) 2021 Acoustical Society of America PU - ACOUSTICAL SOC AMER AMER INST PHYSICS PI - MELVILLE PA - STE 1 NO 1, 2 HUNTINGTON QUADRANGLE, MELVILLE, NY 11747-4502 USA\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Explicit versus non-explicit prosodic training in the learning of Spanish L2 stress contrasts by French listeners.\n \n \n \n\n\n \n Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Second Language Studies, (October): 1–41. 2021.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schwabExplicitNonexplicitProsodic2021a,\n\ttitle = {Explicit versus non-explicit prosodic training in the learning of {Spanish} {L2} stress contrasts by {French} listeners},\n\tissn = {2542-3835},\n\tdoi = {10.1075/jsls.21017.sch},\n\tabstract = {Different methods to acquire a language can contribute differently to learning success. In the present study we tested the success of L2 stress contrasts acquisition, when ab initio learners were taught or not about the theoretic nature of L2 stress contrasts. In two 4-hour perceptual training methods, French-speaking listeners received either (a) explicit instructions about Spanish stress patterns and perception activities commonly used in L2 pronunciation courses or (b) no explicit instructions and a unique perception activity, a shape/word matching task. Results showed that French-speaking listeners improved their ability to identify and discriminate stress contrasts in Spanish after training. However, there was no significant difference between explicit and non-explicit training nor was there an effect on stress processing under different phonetic variability conditions. This suggests that in L2 stress acquisition, non-explicit training may benefit ab initio learners as much as explicit instruction and activities used in L2 pronunciation courses.},\n\tnumber = {October},\n\tjournal = {Journal of Second Language Studies},\n\tauthor = {Schwab, Sandra and Dellwo, Volker},\n\tyear = {2021},\n\tpages = {1--41},\n}\n\n\n\n
\n
\n\n\n
\n Different methods to acquire a language can contribute differently to learning success. In the present study we tested the success of L2 stress contrasts acquisition, when ab initio learners were taught or not about the theoretic nature of L2 stress contrasts. In two 4-hour perceptual training methods, French-speaking listeners received either (a) explicit instructions about Spanish stress patterns and perception activities commonly used in L2 pronunciation courses or (b) no explicit instructions and a unique perception activity, a shape/word matching task. Results showed that French-speaking listeners improved their ability to identify and discriminate stress contrasts in Spanish after training. However, there was no significant difference between explicit and non-explicit training nor was there an effect on stress processing under different phonetic variability conditions. This suggests that in L2 stress acquisition, non-explicit training may benefit ab initio learners as much as explicit instruction and activities used in L2 pronunciation courses.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Age-related rhythmic variations : The role of syllable intensity variability.\n \n \n \n \n\n\n \n Pellegrino, E.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n Travaux neuchâtelois de linguistique, (74): 167–185. January 2021.\n \n\n\n\n
\n\n\n\n \n \n \"Age-relatedPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoAgerelatedRhythmicVariations2021,\n\ttitle = {Age-related rhythmic variations : {The} role of syllable intensity variability},\n\tissn = {2504-205X, 1010-1705},\n\tshorttitle = {Age-related rhythmic variations},\n\turl = {https://www.revue-tranel.ch/article/view/2924},\n\tdoi = {10.26034/tranel.2021.2924},\n\tabstract = {Speech rhythm varies with age. In this paper, we examined the role of mean and peak syllable intensity variability in age-related rhythmic changes. Si xteen younger adults and 10 older speakers read 60 sentences in Zurich German. Results revealed that peak syllable intensity variability is significantly smaller in older compared to younger adults; there was no such effect for syllable mean intensity. Reduced fluency, changes in the biomechanical properties of articulators controlling the mouth opening cycles, and compensation strategies for subglottal pressure generation were the most plausible reasons for the obtained age-related syllable intensity variability.},\n\tlanguage = {en},\n\tnumber = {74},\n\turldate = {2025-04-18},\n\tjournal = {Travaux neuchâtelois de linguistique},\n\tauthor = {Pellegrino, Elisa and He, Lei and Dellwo, Volker},\n\tmonth = jan,\n\tyear = {2021},\n\tpages = {167--185},\n}\n\n\n\n
\n
\n\n\n
\n Speech rhythm varies with age. In this paper, we examined the role of mean and peak syllable intensity variability in age-related rhythmic changes. Si xteen younger adults and 10 older speakers read 60 sentences in Zurich German. Results revealed that peak syllable intensity variability is significantly smaller in older compared to younger adults; there was no such effect for syllable mean intensity. Reduced fluency, changes in the biomechanical properties of articulators controlling the mouth opening cycles, and compensation strategies for subglottal pressure generation were the most plausible reasons for the obtained age-related syllable intensity variability.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Word stress processing integrates phonological abstraction with lexical access – An ERP study.\n \n \n \n \n\n\n \n Broś, K.; Meyer, M.; Kliesch, M.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Neurolinguistics, 57: 100959. February 2021.\n \n\n\n\n
\n\n\n\n \n \n \"WordPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{brosWordStressProcessing2021,\n\ttitle = {Word stress processing integrates phonological abstraction with lexical access – {An} {ERP} study},\n\tvolume = {57},\n\tissn = {09116044},\n\turl = {https://linkinghub.elsevier.com/retrieve/pii/S0911604420301196},\n\tdoi = {10.1016/j.jneuroling.2020.100959},\n\tabstract = {It is unclear whether word stress in a language is stored as part of the word or whether it is generated by a rule. We test the generativist hypothesis of lexical storage stating that only unpredictable stress is stored in long-term memory against the contrasting usage-based approach assuming that all phonetic information regardless of its (un)predictability is stored in the mental lexicon together with the word. In a correctness judgment task involving correctly and incorrectly stressed penults and antepenults, we found that incorrectly stressed penults do not evoke an N400 effect, whereas incorrectly stressed antepenults do: there is increased negativity with a peak latency around 350–600 ms from word onset. Only changes to words with exceptional stress cause lexical inhibition, hence exceptional but not default stress markers are stored in the lexicon. Additionally, differences in processing patterns between the N400 and the late positivity component window point to an integration of two stages of word processing: pre-lexical stress recognition and stress-to-meaning matching. The results of the study support the view that stress should be understood as abstract phonological information.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tjournal = {Journal of Neurolinguistics},\n\tauthor = {Broś, Karolina and Meyer, Martin and Kliesch, Maria and Dellwo, Volker},\n\tmonth = feb,\n\tyear = {2021},\n\tpages = {100959},\n}\n\n\n\n
\n
\n\n\n
\n It is unclear whether word stress in a language is stored as part of the word or whether it is generated by a rule. We test the generativist hypothesis of lexical storage stating that only unpredictable stress is stored in long-term memory against the contrasting usage-based approach assuming that all phonetic information regardless of its (un)predictability is stored in the mental lexicon together with the word. In a correctness judgment task involving correctly and incorrectly stressed penults and antepenults, we found that incorrectly stressed penults do not evoke an N400 effect, whereas incorrectly stressed antepenults do: there is increased negativity with a peak latency around 350–600 ms from word onset. Only changes to words with exceptional stress cause lexical inhibition, hence exceptional but not default stress markers are stored in the lexicon. Additionally, differences in processing patterns between the N400 and the late positivity component window point to an integration of two stages of word processing: pre-lexical stress recognition and stress-to-meaning matching. The results of the study support the view that stress should be understood as abstract phonological information.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2020\n \n \n (3)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Voice quality and f0 correlates of Burmese tone in the standard and southern varieties.\n \n \n \n\n\n \n Dellwo, V.; and Jenny, M.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 148(4_Supplement): 2656–2656. 2020.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwo2020voice,\n\ttitle = {Voice quality and f0 correlates of {Burmese} tone in the standard and southern varieties},\n\tvolume = {148},\n\tnumber = {4\\_Supplement},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {Dellwo, Volker and Jenny, Mathias},\n\tyear = {2020},\n\tpages = {2656--2656},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Working memory and not acoustic sensitivity is related to stress processing ability in a foreign language: An ERP study.\n \n \n \n\n\n \n Schwab, S.; Giroud, N.; Meyer, M.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Neurolinguistics, 55: 100897. 2020.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n  \n \n 1 download\n \n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schwab2020working,\n\ttitle = {Working memory and not acoustic sensitivity is related to stress processing ability in a foreign language: {An} {ERP} study},\n\tvolume = {55},\n\tjournal = {Journal of Neurolinguistics},\n\tpublisher = {Pergamon},\n\tauthor = {Schwab, Sandra and Giroud, Nathalie and Meyer, Martin and Dellwo, Volker},\n\tyear = {2020},\n\tpages = {100897},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Working memory and not acoustic sensitivity is related to stress processing ability in a foreign language: An ERP study.\n \n \n \n \n\n\n \n Schwab, S.; Giroud, N.; Meyer, M.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Neurolinguistics, 55: 100897. August 2020.\n \n\n\n\n
\n\n\n\n \n \n \"WorkingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n  \n \n 1 download\n \n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schwabWorkingMemoryNot2020,\n\ttitle = {Working memory and not acoustic sensitivity is related to stress processing ability in a foreign language: {An} {ERP} study},\n\tvolume = {55},\n\tissn = {09116044},\n\tshorttitle = {Working memory and not acoustic sensitivity is related to stress processing ability in a foreign language},\n\turl = {https://linkinghub.elsevier.com/retrieve/pii/S0911604419300818},\n\tdoi = {10.1016/j.jneuroling.2020.100897},\n\tabstract = {Listeners in fixed-stress languages are less sensitive in processing stress contrasts in a second language with contrastive stress (stress 'deafness'). We investigated whether native speakers of French (fixed-stress language) can acquire the ability to distinguish stress contrasts in Spanish (free-stress language). In behavioral experiments, we found that French listeners were able to improve their ability to discriminate stress contrasts in Spanish after a 4-h training. This indicates that French listeners' stress detection disadvantage can be reduced by a short exposure to L2 stress contrasts. An ERP experiment administered after the training evidenced that the larger the P3b amplitude, the better the listeners' training outcome. In contrast, listeners' performance was not reflected by the N2b amplitude. In other words, listeners with high performance after training showed similar auditory sensitivity to stress in comparison to listeners with poor performance, but they better maintained stress information in working memory, as indicated by the larger amplitude of P3b. The present research indicates that individual differences in working memory processing should be considered in the acquisition of second language prosody.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tjournal = {Journal of Neurolinguistics},\n\tauthor = {Schwab, Sandra and Giroud, Nathalie and Meyer, Martin and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2020},\n\tpages = {100897},\n}\n\n\n\n
\n
\n\n\n
\n Listeners in fixed-stress languages are less sensitive in processing stress contrasts in a second language with contrastive stress (stress 'deafness'). We investigated whether native speakers of French (fixed-stress language) can acquire the ability to distinguish stress contrasts in Spanish (free-stress language). In behavioral experiments, we found that French listeners were able to improve their ability to discriminate stress contrasts in Spanish after a 4-h training. This indicates that French listeners' stress detection disadvantage can be reduced by a short exposure to L2 stress contrasts. An ERP experiment administered after the training evidenced that the larger the P3b amplitude, the better the listeners' training outcome. In contrast, listeners' performance was not reflected by the N2b amplitude. In other words, listeners with high performance after training showed similar auditory sensitivity to stress in comparison to listeners with poor performance, but they better maintained stress information in working memory, as indicated by the larger amplitude of P3b. The present research indicates that individual differences in working memory processing should be considered in the acquisition of second language prosody.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2019\n \n \n (12)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n the Role of the First Five Formants in Three Vowels of.\n \n \n \n\n\n \n Cao, H.; and Dellwo, V.\n\n\n \n\n\n\n In International Congress of Phonetic Sciences (ICPhS), pages 617–621, 2019. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Cao,\n\ttitle = {the {Role} of the {First} {Five} {Formants} in {Three} {Vowels} of},\n\tabstract = {Formant characteristics are most commonly part of forensic speaker comparison (FSC). However, only formants F1 to F3 typically occur in evidence material because it is mostly recorded via telephone. Given recent technological advances in telephony (e.g. WeChat or WhatsApp) higher formants (F4-F5) are becoming increasingly part of evidence material. The present study investigated the speaker- distinguishing properties of F1 to F5 of three sustained vowels /i/, /y/ and /ɤ/ in Mandarin produced by 20 young male speakers. Based on discriminant analysis, for each single formant, the best predictors were F5 for /i/ and F4 for /y/ and /ɤ/. Classification performance varied between vowels. Inclusion of two and three formants yielded higher classification rates of 30−80\\%. The best value was provided by the combination of F2, F4 and F5 of /ɤ/. The value and limitations of F4 and F5 for FSC are discussed.},\n\tbooktitle = {International {Congress} of {Phonetic} {Sciences} ({ICPhS})},\n\tauthor = {Cao, Honglin and Dellwo, Volker},\n\tyear = {2019},\n\tpages = {617--621},\n}\n\n\n\n
\n
\n\n\n
\n Formant characteristics are most commonly part of forensic speaker comparison (FSC). However, only formants F1 to F3 typically occur in evidence material because it is mostly recorded via telephone. Given recent technological advances in telephony (e.g. WeChat or WhatsApp) higher formants (F4-F5) are becoming increasingly part of evidence material. The present study investigated the speaker- distinguishing properties of F1 to F5 of three sustained vowels /i/, /y/ and /ɤ/ in Mandarin produced by 20 young male speakers. Based on discriminant analysis, for each single formant, the best predictors were F5 for /i/ and F4 for /y/ and /ɤ/. Classification performance varied between vowels. Inclusion of two and three formants yielded higher classification rates of 30−80%. The best value was provided by the combination of F2, F4 and F5 of /ɤ/. The value and limitations of F4 and F5 for FSC are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Fundamental Frequency Accommodation in Multi-Party Human-Robot Game Interactions: The Effect of Winning or Losing.\n \n \n \n \n\n\n \n Ibrahim, O.; Skantze, G.; Stoll, S.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2019, pages 3980–3984, September 2019. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"FundamentalPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{ibrahimFundamentalFrequencyAccommodation2019,\n\ttitle = {Fundamental {Frequency} {Accommodation} in {Multi}-{Party} {Human}-{Robot} {Game} {Interactions}: {The} {Effect} of {Winning} or {Losing}},\n\tshorttitle = {Fundamental {Frequency} {Accommodation} in {Multi}-{Party} {Human}-{Robot} {Game} {Interactions}},\n\turl = {https://www.isca-archive.org/interspeech_2019/ibrahim19_interspeech.html},\n\tdoi = {10.21437/Interspeech.2019-2496},\n\tabstract = {In human-human interactions, the situational context plays a large role in the degree of speakers’ accommodation. In this paper, we investigate whether the degree of accommodation in a human-robot computer game is affected by (a) the duration of the interaction and (b) the success of the players in the game. 30 teams of two players played two card games with a conversational robot in which they had to find a correct order of five cards. After game 1, the players received the result of the game on a success scale from 1 (lowest success) to 5 (highest). Speakers’ fo accommodation was measured as the Euclidean distance between the human speakers and each human and the robot. Results revealed that (a) the duration of the game had no influence on the degree of fo accommodation and (b) the result of Game 1 correlated with the degree of fo accommodation in Game 2 (higher success equals lower Euclidean distance). We argue that game success is most likely considered as a sign of the success of players’ cooperation during the discussion, which leads to a higher accommodation behavior in speech.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2019},\n\tpublisher = {ISCA},\n\tauthor = {Ibrahim, Omnia and Skantze, Gabriel and Stoll, Sabine and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2019},\n\tpages = {3980--3984},\n}\n\n\n\n
\n
\n\n\n
\n In human-human interactions, the situational context plays a large role in the degree of speakers’ accommodation. In this paper, we investigate whether the degree of accommodation in a human-robot computer game is affected by (a) the duration of the interaction and (b) the success of the players in the game. 30 teams of two players played two card games with a conversational robot in which they had to find a correct order of five cards. After game 1, the players received the result of the game on a success scale from 1 (lowest success) to 5 (highest). Speakers’ fo accommodation was measured as the Euclidean distance between the human speakers and each human and the robot. Results revealed that (a) the duration of the game had no influence on the degree of fo accommodation and (b) the result of Game 1 correlated with the degree of fo accommodation in Game 2 (higher success equals lower Euclidean distance). We argue that game success is most likely considered as a sign of the success of players’ cooperation during the discussion, which leads to a higher accommodation behavior in speech.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Cepstral derivatives in MFCCS for emotion recognition.\n \n \n \n\n\n \n Kathiresan, T.; and Dellwo, V.\n\n\n \n\n\n\n In 2019 IEEE 4th International Conference on Signal and Image Processing, ICSIP 2019, 2019. \n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{kathiresanCepstralDerivativesMFCCS2019,\n\ttitle = {Cepstral derivatives in {MFCCS} for emotion recognition},\n\tisbn = {978-1-7281-3660-8},\n\tdoi = {10.1109/SIPROCESS.2019.8868573},\n\tabstract = {© 2019 IEEE. Recent advancements and growing interest in developing affective interaction systems demand more efficient and robust solutions for emotion recognition. Most of the current emotion recognizer systems use Mel frequency cepstral coefficients (MFCCs) along with its temporal derivatives, and other static spectral features. In this paper, we propose a new dynamic feature called cepstral deltas/derivatives to improve emotion recognition performance. We have used two emotion speech datasets of different languages (German, and English) to study the performance of cepstral deltas. The MFCCs are used as a base feature to compare the performance of the proposed feature and traditionally used temporal deltas. The result shows that the addition of cepstral deltas with MFCCs improves the performance of specific emotions in different languages. The possible performance of cepstral deltas in other classifiers like support vector machine and deep neural networks is discussed.},\n\tbooktitle = {2019 {IEEE} 4th {International} {Conference} on {Signal} and {Image} {Processing}, {ICSIP} 2019},\n\tauthor = {Kathiresan, T. and Dellwo, V.},\n\tyear = {2019},\n}\n\n\n\n
\n
\n\n\n
\n © 2019 IEEE. Recent advancements and growing interest in developing affective interaction systems demand more efficient and robust solutions for emotion recognition. Most of the current emotion recognizer systems use Mel frequency cepstral coefficients (MFCCs) along with its temporal derivatives, and other static spectral features. In this paper, we propose a new dynamic feature called cepstral deltas/derivatives to improve emotion recognition performance. We have used two emotion speech datasets of different languages (German, and English) to study the performance of cepstral deltas. The MFCCs are used as a base feature to compare the performance of the proposed feature and traditionally used temporal deltas. The result shows that the addition of cepstral deltas with MFCCs improves the performance of specific emotions in different languages. The possible performance of cepstral deltas in other classifiers like support vector machine and deep neural networks is discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Music and L2 Prosody : the Role of Music Aptitude on the Discrimination of Stress Contrasts in L2.\n \n \n \n \n\n\n \n Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n In International Congress of Phonetic Sciences (ICPhS), pages 1927–1931, 2019. \n \n\n\n\n
\n\n\n\n \n \n \"MusicPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Schwab1927,\n\ttitle = {Music and {L2} {Prosody} : the {Role} of {Music} {Aptitude} on the {Discrimination} of {Stress} {Contrasts} in {L2}},\n\turl = {https://doi.org/10.5167/uzh-177496},\n\tdoi = {https://doi.org/10.5167/uzh-177496},\n\tabstract = {In this study, we investigated the effect of music aptitude on French and German listeners' performance at discriminating stress contrasts in Spanish L2, before and after a 4-hour perceptual training in Spanish. For the French listeners, results showed that the better the music aptitude the better the stress discrimination performance (before and after training). Regarding German listeners, music aptitude did not show any effect on the discrimination of stress contrasts in Spanish L2. The link between music and L2 stress discrimination in French listeners (and its absence in German listeners) suggests that French and German listeners do not process stress information in the same way. It might be that, since French listeners, contrary to German listeners, are not used to stress encoding mechanism in their native language, they interpret information related to L2 prosody (such as lexical stress) in a more "musical way".},\n\tbooktitle = {International {Congress} of {Phonetic} {Sciences} ({ICPhS})},\n\tauthor = {Schwab, Sandra and Dellwo, Volker},\n\tyear = {2019},\n\tpages = {1927--1931},\n}\n\n\n\n
\n
\n\n\n
\n In this study, we investigated the effect of music aptitude on French and German listeners' performance at discriminating stress contrasts in Spanish L2, before and after a 4-hour perceptual training in Spanish. For the French listeners, results showed that the better the music aptitude the better the stress discrimination performance (before and after training). Regarding German listeners, music aptitude did not show any effect on the discrimination of stress contrasts in Spanish L2. The link between music and L2 stress discrimination in French listeners (and its absence in German listeners) suggests that French and German listeners do not process stress information in the same way. It might be that, since French listeners, contrary to German listeners, are not used to stress encoding mechanism in their native language, they interpret information related to L2 prosody (such as lexical stress) in a more \"musical way\".\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n The Dynamics of Indexical Information in Speech: Can Recognizability be Controlled by the Speaker?.\n \n \n \n\n\n \n Dellwo, V; Pellegrino, E; He, L; and Kathiresan, T\n\n\n \n\n\n\n Acta Universitatis Carolinae Philologica 2, 2019(2): 57–75. 2019.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoDynamicsIndexicalInformation2019,\n\ttitle = {The {Dynamics} of {Indexical} {Information} in {Speech}: {Can} {Recognizability} be {Controlled} by the {Speaker}?},\n\tvolume = {2019},\n\tdoi = {https://doi.org/10.14712/24646830.2019.18},\n\tnumber = {2},\n\tjournal = {Acta Universitatis Carolinae Philologica 2},\n\tauthor = {Dellwo, V and Pellegrino, E and He, L and Kathiresan, T},\n\tyear = {2019},\n\tpages = {57--75},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Bridging the brain structure—brain function gap in prosodic speech processing in older adults.\n \n \n \n\n\n \n Giroud, N.; Keller, M.; Hirsiger, S.; Dellwo, V.; and Meyer, M.\n\n\n \n\n\n\n Neurobiology of Aging, 80. 2019.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{giroudBridgingBrainStructure2019,\n\ttitle = {Bridging the brain structure—brain function gap in prosodic speech processing in older adults},\n\tvolume = {80},\n\tissn = {15581497},\n\tdoi = {10.1016/j.neurobiolaging.2019.04.017},\n\tabstract = {© 2019 The Author(s) Age-related decline in speech perception may result in difficulties partaking in spoken conversation and potentially lead to social isolation and cognitive decline in older adults. It is therefore important to better understand how age-related differences in neurostructural factors such as cortical thickness (CT) and cortical surface area (CSA) are related to neurophysiological sensitivity to speech cues in younger and older adults. Age-related differences in CT and CSA of bilateral auditory-related areas were extracted using FreeSurfer in younger and older adults with normal peripheral hearing. Behavioral and neurophysiological sensitivity to prosodic speech cues (word stress and fundamental frequency of oscillation) was evaluated using discrimination tasks and a passive oddball paradigm, while EEG was recorded, to quantify mismatch negativity responses. Results revealed (a) higher neural sensitivity (i.e., larger mismatch negativity responses) to word stress in older adults compared to younger adults, suggesting a higher importance of prosodic speech cues in the speech processing of older adults, and (b) lower CT in auditory-related regions in older compared to younger individuals, suggesting neuronal loss associated with aging. Within the older age group, less neuronal loss (i.e., higher CT) in a right auditory-related area (i.e., the supratemporal sulcus) was related to better performance in fundamental frequency discrimination, while higher CSA in left auditory-related areas was associated with higher neural sensitivity toward prosodic speech cues as evident in the mismatch negativity patterns. Overall, our results offer evidence for neurostructural changes in aging that are associated with differences in the extent to which left and right auditory-related areas are involved in speech processing in older adults. We argue that exploring age-related differences in brain structure and function associated with decline in speech perception in older adults may help develop much needed rehabilitation strategies for older adults with central age-related hearing loss.},\n\tjournal = {Neurobiology of Aging},\n\tauthor = {Giroud, N. and Keller, M. and Hirsiger, S. and Dellwo, V. and Meyer, M.},\n\tyear = {2019},\n}\n\n\n\n
\n
\n\n\n
\n © 2019 The Author(s) Age-related decline in speech perception may result in difficulties partaking in spoken conversation and potentially lead to social isolation and cognitive decline in older adults. It is therefore important to better understand how age-related differences in neurostructural factors such as cortical thickness (CT) and cortical surface area (CSA) are related to neurophysiological sensitivity to speech cues in younger and older adults. Age-related differences in CT and CSA of bilateral auditory-related areas were extracted using FreeSurfer in younger and older adults with normal peripheral hearing. Behavioral and neurophysiological sensitivity to prosodic speech cues (word stress and fundamental frequency of oscillation) was evaluated using discrimination tasks and a passive oddball paradigm, while EEG was recorded, to quantify mismatch negativity responses. Results revealed (a) higher neural sensitivity (i.e., larger mismatch negativity responses) to word stress in older adults compared to younger adults, suggesting a higher importance of prosodic speech cues in the speech processing of older adults, and (b) lower CT in auditory-related regions in older compared to younger individuals, suggesting neuronal loss associated with aging. Within the older age group, less neuronal loss (i.e., higher CT) in a right auditory-related area (i.e., the supratemporal sulcus) was related to better performance in fundamental frequency discrimination, while higher CSA in left auditory-related areas was associated with higher neural sensitivity toward prosodic speech cues as evident in the mismatch negativity patterns. Overall, our results offer evidence for neurostructural changes in aging that are associated with differences in the extent to which left and right auditory-related areas are involved in speech processing in older adults. We argue that exploring age-related differences in brain structure and function associated with decline in speech perception in older adults may help develop much needed rehabilitation strategies for older adults with central age-related hearing loss.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Between-speaker variability and temporal organization of the first formant.\n \n \n \n \n\n\n \n He, L.; Zhang, Y.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 145(3): EL209–EL214. March 2019.\n \n\n\n\n
\n\n\n\n \n \n \"Between-speakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{heBetweenspeakerVariabilityTemporal2019,\n\ttitle = {Between-speaker variability and temporal organization of the first formant},\n\tvolume = {145},\n\tissn = {0001-4966},\n\turl = {http://asa.scitation.org/doi/10.1121/1.5093450},\n\tdoi = {10.1121/1.5093450},\n\tabstract = {First formant (F1) trajectories of vocalic intervals were divided into positive and negative dynamics. Positive F1 dynamics were defined as the speeds of F1 increases to reach the maxima, and negative F1 dynamics as the speeds of F1 decreases away from the maxima. Mean, standard deviation, and sequential variability were measured for both dynamics. Results showed that measures of negative F1 dynamics explained more between-speaker variability, which was highly congruent with a previous study using intensity dynamics [He and Dellwo (2017). J. Acoust. Soc. Am. 141, EL488–EL494]. The results may be explained by speaker idiosyncratic articulation.},\n\tnumber = {3},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {He, Lei and Zhang, Yu and Dellwo, Volker},\n\tmonth = mar,\n\tyear = {2019},\n\tpages = {EL209--EL214},\n}\n\n\n\n
\n
\n\n\n
\n First formant (F1) trajectories of vocalic intervals were divided into positive and negative dynamics. Positive F1 dynamics were defined as the speeds of F1 increases to reach the maxima, and negative F1 dynamics as the speeds of F1 decreases away from the maxima. Mean, standard deviation, and sequential variability were measured for both dynamics. Results showed that measures of negative F1 dynamics explained more between-speaker variability, which was highly congruent with a previous study using intensity dynamics [He and Dellwo (2017). J. Acoust. Soc. Am. 141, EL488–EL494]. The results may be explained by speaker idiosyncratic articulation.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Highly spectrally undersampled vowels can be classified by machines without supervision.\n \n \n \n\n\n \n Kathiresan, T.; Maurer, D.; and Dellwo, V.\n\n\n \n\n\n\n , 146(1): EL1–EL7. 2019.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kathiresanHighlySpectrallyUndersampled2019,\n\ttitle = {Highly spectrally undersampled vowels can be classified by machines without supervision},\n\tvolume = {146},\n\tdoi = {10.1121/1.5111154},\n\tabstract = {© 2019 Acoustical Society of America. An unsupervised automatic clustering algorithm (k-means) classified 1282 Mel frequency cepstral coefficient (MFCC) representations of isolated steady-state vowel utterances from eight standard German vowel categories with fo between 196 and 698 Hz. Experiment I obtained the number of MFCCs (1-20) in connection with the spectral bandwidth (2-20 kHz) at which performance peaked (five MFCCs at 4 kHz). In experiment II, classification performance with different ranges of fo revealed that ranges with fo {\\textgreater} 500 Hz reduced classification performance but it remained well above chance. This shows that isolated steady state vowels with strongly undersampled spectra contain sufficient acoustic information to be classified automatically.},\n\tnumber = {1},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {Kathiresan, Thayabaran and Maurer, Dieter and Dellwo, Volker},\n\tyear = {2019},\n\tpages = {EL1--EL7},\n}\n\n\n\n
\n
\n\n\n
\n © 2019 Acoustical Society of America. An unsupervised automatic clustering algorithm (k-means) classified 1282 Mel frequency cepstral coefficient (MFCC) representations of isolated steady-state vowel utterances from eight standard German vowel categories with fo between 196 and 698 Hz. Experiment I obtained the number of MFCCs (1-20) in connection with the spectral bandwidth (2-20 kHz) at which performance peaked (five MFCCs at 4 kHz). In experiment II, classification performance with different ranges of fo revealed that ranges with fo \\textgreater 500 Hz reduced classification performance but it remained well above chance. This shows that isolated steady state vowels with strongly undersampled spectra contain sufficient acoustic information to be classified automatically.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Evaluation of VOCALISE under conditions reflecting those of a real forensic voice comparison case (forensic_eval_01).\n \n \n \n\n\n \n Kelly, F.; Fröhlich, A.; Dellwo, V.; Forth, O.; Kent, S.; and Alexander, A.\n\n\n \n\n\n\n Speech Communication, 112. 2019.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kellyEvaluationVOCALISEConditions2019,\n\ttitle = {Evaluation of {VOCALISE} under conditions reflecting those of a real forensic voice comparison case (forensic\\_eval\\_01)},\n\tvolume = {112},\n\tissn = {01676393},\n\tdoi = {10.1016/j.specom.2019.06.005},\n\tjournal = {Speech Communication},\n\tauthor = {Kelly, F. and Fröhlich, A. and Dellwo, V. and Forth, O. and Kent, S. and Alexander, A.},\n\tyear = {2019},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Phonetic Sources of Sound Change: The Influence of Thai on Nasality in Pwo Karen.\n \n \n \n\n\n \n Kerdpol, K.; Dellwo, V.; and Jenny, M.\n\n\n \n\n\n\n Manusya, 19(1): 45–66. 2019.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Kerdpol2019,\n\ttitle = {Phonetic {Sources} of {Sound} {Change}: {The} {Influence} of {Thai} on {Nasality} in {Pwo} {Karen}},\n\tvolume = {19},\n\tissn = {0859-9920},\n\tdoi = {10.1163/26659077-01901003},\n\tabstract = {The phonetic realization of nasal vowels produced by Pwo speakers of different ages can vary. The present study investigated mid and low nasal vowels of Pwo speakers from Mae Hong Son province, Thailand. Due to the higher tendency of language contact with Thai, the younger group’s nasal vowels were expected to lose more nasality than the older group. The emergence of final nasal consonants was also expected in the younger group. The nasalization duration and consonant duration of both groups were analyzed. The results showed that, regardless of age, mid nasal vowels of some speakers had final nasal consonants, while low nasal vowels of all speakers did not. Furthermore, the older group had both longer nasalization duration and consonant duration than the younger group, suggesting their higher tendency to preserve nasality. The younger group had shorter nasalization duration and consonant duration, indicating the loss of nasality in vowels without compensatory final nasal consonants. The change might be due to the vowel quality. High vowels were fully denasalized with no compensatory final nasal consonants. Mid vowels were nasalized with the emergence of final nasal consonants. Low vowels remained nasalized without final nasal consonants. We could not confirm that the emergence of final nasal consonants was induced by Thai because it occurred in both groups. The existence of final nasal consonants in the younger group could not be used as evidence of an effect of contact.},\n\tnumber = {1},\n\tjournal = {Manusya},\n\tauthor = {Kerdpol, Karnthida and Dellwo, Volker and Jenny, Mathias},\n\tyear = {2019},\n\tpages = {45--66},\n}\n\n\n\n
\n
\n\n\n
\n The phonetic realization of nasal vowels produced by Pwo speakers of different ages can vary. The present study investigated mid and low nasal vowels of Pwo speakers from Mae Hong Son province, Thailand. Due to the higher tendency of language contact with Thai, the younger group’s nasal vowels were expected to lose more nasality than the older group. The emergence of final nasal consonants was also expected in the younger group. The nasalization duration and consonant duration of both groups were analyzed. The results showed that, regardless of age, mid nasal vowels of some speakers had final nasal consonants, while low nasal vowels of all speakers did not. Furthermore, the older group had both longer nasalization duration and consonant duration than the younger group, suggesting their higher tendency to preserve nasality. The younger group had shorter nasalization duration and consonant duration, indicating the loss of nasality in vowels without compensatory final nasal consonants. The change might be due to the vowel quality. High vowels were fully denasalized with no compensatory final nasal consonants. Mid vowels were nasalized with the emergence of final nasal consonants. Low vowels remained nasalized without final nasal consonants. We could not confirm that the emergence of final nasal consonants was induced by Thai because it occurred in both groups. The existence of final nasal consonants in the younger group could not be used as evidence of an effect of contact.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speaker Individuality in the Durational Characteristics of Voiced Intervals : the Case of Chinese Bi-Dialectal Speakers.\n \n \n \n\n\n \n Zhang, Y.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n In International Congress of Phonetic Sciences (ICPhS), pages 3075–3079, Melbourne, 2019. \n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Zhang,\n\taddress = {Melbourne},\n\ttitle = {Speaker {Individuality} in the {Durational} {Characteristics} of {Voiced} {Intervals} : the {Case} of {Chinese} {Bi}-{Dialectal} {Speakers}},\n\tdoi = {doi.org/10.5167/uzh-177495},\n\tabstract = {Temporal organizations of the speech signal are highly individual among speakers of the same language. In the present study, we looked at speech production of bi-dialectal speakers using two varieties of the same language. We aimed at testing whether speaker-specific temporal features present in one dialect remain in another dialect of the same speaker. 20 sentences and one passage in both Mandarin and Danyang Dialect of 14 bi-dialectal speakers were recorded. We measured between- speaker variability of the percentage of voiced interval duration (percentVO) in both dialect conditions using linear mixed effect models. Results revealed that speakers exhibited distinct between- speaker variability when dialect variability and style variability were introduced. However, within-speaker variability was also present and the magnitude of the variability differences varied among different speakers and in different speaking styles. Findings of the current study are particularly relevant for forensic voice comparison tasks when a mismatch in speaking languages in trace and suspect materials is present.},\n\tbooktitle = {International {Congress} of {Phonetic} {Sciences} ({ICPhS})},\n\tauthor = {Zhang, Yu and He, Lei and Dellwo, Volker},\n\tyear = {2019},\n\tpages = {3075--3079},\n}\n\n\n\n
\n
\n\n\n
\n Temporal organizations of the speech signal are highly individual among speakers of the same language. In the present study, we looked at speech production of bi-dialectal speakers using two varieties of the same language. We aimed at testing whether speaker-specific temporal features present in one dialect remain in another dialect of the same speaker. 20 sentences and one passage in both Mandarin and Danyang Dialect of 14 bi-dialectal speakers were recorded. We measured between- speaker variability of the percentage of voiced interval duration (percentVO) in both dialect conditions using linear mixed effect models. Results revealed that speakers exhibited distinct between- speaker variability when dialect variability and style variability were introduced. However, within-speaker variability was also present and the magnitude of the variability differences varied among different speakers and in different speaking styles. Findings of the current study are particularly relevant for forensic voice comparison tasks when a mismatch in speaking languages in trace and suspect materials is present.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker individuality in the durational characteristics of voiced intervals: the case of chinese bi-dialectal speakers.\n \n \n \n \n\n\n \n Zhang, Y.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n . August 2019.\n \n\n\n\n
\n\n\n\n \n \n \"SpeakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{zhangSpeakerIndividualityDurational2019a,\n\ttitle = {Speaker individuality in the durational characteristics of voiced intervals: the case of chinese bi-dialectal speakers},\n\tshorttitle = {Speaker individuality in the durational characteristics of voiced intervals},\n\turl = {https://www.zora.uzh.ch/id/eprint/177495},\n\tdoi = {10.5167/UZH-177495},\n\tabstract = {Temporal organizations of the speech signal are highly individual among speakers of the same language. In the present study, we looked at speech production of bi-dialectal speakers using two varieties of the same language. We aimed at testing whether speaker-specific temporal features present in one dialect remain in another dialect of the same speaker. 20 sentences and one passage in both Mandarin and Danyang Dialect of 14 bi-dialectal speakers were recorded. We measured betweenspeaker variability of the percentage of voiced interval duration (percentVO) in both dialect conditions using linear mixed effect models. Results revealed that speakers exhibited distinct betweenspeaker variability when dialect variability and style variability were introduced. However, within-speaker variability was also present and the magnitude of the variability differences varied among different speakers and in different speaking styles. Findings of the current study are particularly relevant for forensic voice comparison tasks when a mismatch in speaking languages in trace and suspect materials is present.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Phonetic Association},\n\tauthor = {Zhang, Yu and He, Lei and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2019},\n}\n\n\n\n
\n
\n\n\n
\n Temporal organizations of the speech signal are highly individual among speakers of the same language. In the present study, we looked at speech production of bi-dialectal speakers using two varieties of the same language. We aimed at testing whether speaker-specific temporal features present in one dialect remain in another dialect of the same speaker. 20 sentences and one passage in both Mandarin and Danyang Dialect of 14 bi-dialectal speakers were recorded. We measured betweenspeaker variability of the percentage of voiced interval duration (percentVO) in both dialect conditions using linear mixed effect models. Results revealed that speakers exhibited distinct betweenspeaker variability when dialect variability and style variability were introduced. However, within-speaker variability was also present and the magnitude of the variability differences varied among different speakers and in different speaking styles. Findings of the current study are particularly relevant for forensic voice comparison tasks when a mismatch in speaking languages in trace and suspect materials is present.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2018\n \n \n (5)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Voice Biometrics for FORENSIC Speaker Recognition Applications.\n \n \n \n\n\n \n Dellwo, V.; French, P.; He, L.; Dellwo, V.; French, P.; and He, L.\n\n\n \n\n\n\n In Frühholz, S.; and Belin, P., editor(s), The Oxford Handbook of Voice Perception, pages 776–796. Oxford University Press, Oxford, January 2018.\n Issue: April\n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@incollection{Dellwo2018,\n\taddress = {Oxford},\n\ttitle = {Voice {Biometrics} for {FORENSIC} {Speaker} {Recognition} {Applications}},\n\tisbn = {978-0-19-874318-7},\n\tdoi = {10.1093/oxfordhb/9780198743187.013.36},\n\tabstract = {Voices are highly individual, and this information may be used to recognize people. This chapter provides an overview of how speaker-specific voice information can be used to assist in recognizing unknown speakers in evidential audio recordings in order to assist in progressing criminal investigations or for evidential purposes. While the chapter is predominantly concerned with forensic applications of speaker recognition, it also considers commercially-orientated applications such as the voice-access systems increasingly used in banking, for example. The chapter distinguishes between biometric features of voice (i.e. aspects of speech that facilitate recognition of an individual) and biometric recognition (i.e. the process of using those features to undertake the recognition). Biometric features include a range of frequency- and time-domain characteristics that vary between speakers and are useable in human- based and automatic speaker-recognition processes, the principles of which the chapter explains and discusses. Finally, the chapter considers possible future scenarios in which biometric features of voice might be of use, should certain methodological obstacles be overcome.},\n\tbooktitle = {The {Oxford} {Handbook} of {Voice} {Perception}},\n\tpublisher = {Oxford University Press},\n\tauthor = {Dellwo, Volker and French, Peter and He, Lei and Dellwo, Volker and French, Peter and He, Lei},\n\teditor = {Frühholz, Sascha and Belin, Pascal},\n\tmonth = jan,\n\tyear = {2018},\n\tnote = {Issue: April},\n\tpages = {776--796},\n}\n\n\n\n
\n
\n\n\n
\n Voices are highly individual, and this information may be used to recognize people. This chapter provides an overview of how speaker-specific voice information can be used to assist in recognizing unknown speakers in evidential audio recordings in order to assist in progressing criminal investigations or for evidential purposes. While the chapter is predominantly concerned with forensic applications of speaker recognition, it also considers commercially-orientated applications such as the voice-access systems increasingly used in banking, for example. The chapter distinguishes between biometric features of voice (i.e. aspects of speech that facilitate recognition of an individual) and biometric recognition (i.e. the process of using those features to undertake the recognition). Biometric features include a range of frequency- and time-domain characteristics that vary between speakers and are useable in human- based and automatic speaker-recognition processes, the principles of which the chapter explains and discusses. Finally, the chapter considers possible future scenarios in which biometric features of voice might be of use, should certain methodological obstacles be overcome.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Influences of fundamental oscillation on speaker identification in vocalic utterances by humans and computers.\n \n \n \n \n\n\n \n Dellwo, V.; Kathiresan, T.; Pellegrino, E.; He, L.; Schwab, S.; and Maurer, D.\n\n\n \n\n\n\n In Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH, volume 2018-Septe, pages 3795–3799, Hyderabad, India, September 2018. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"InfluencesPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwoInfluencesFundamentalOscillation2018,\n\taddress = {Hyderabad, India},\n\ttitle = {Influences of fundamental oscillation on speaker identification in vocalic utterances by humans and computers},\n\tvolume = {2018-Septe},\n\tissn = {19909772},\n\turl = {https://doi.org/10.5167/uzh-156785},\n\tdoi = {10.21437/Interspeech.2018-2331},\n\tabstract = {We tested the influence of fundamental oscillation (fo) on human and machine speaker recognition performance in vocalic test utterances. In experiment I, we trained a Gaussian-Mixture model on 15 speakers (80 multi-word utterances each) and tested it with sustained vowel utterances (/a:/, /i:/ and /u:/) under six fo conditions, three changing (fall, rise, fall-rise) and three steady-state (high, mid, low). Results revealed better performance for the steady-state compared to the changing conditions and within the steady-state condition, performance was poorest for high fo. In experiment II, we tested 9 human listeners on a subset of 4 speakers from experiment I. They went through two training tasks (training 1: multi-word utterances; training 2: words). In the test, they recognized speakers based on the same vocalic utterances as in experiment I (for these 4 speakers). Results showed that performance was about equally high for the changing and steady-state vowels, however, in the steady-state condition performance was best for high fo vowels. The experiments suggest that (a) fo has an influence on the strength of speaker specific characteristics in vowels and (b) humans - compared to machines - pay attention to different acoustic information in vocalic utterances for speaker recognition.},\n\tnumber = {September},\n\tbooktitle = {Proceedings of the {Annual} {Conference} of the {International} {Speech} {Communication} {Association}, {INTERSPEECH}},\n\tpublisher = {ISCA},\n\tauthor = {Dellwo, Volker and Kathiresan, Thayabaran and Pellegrino, Elisa and He, Lei and Schwab, Sandra and Maurer, Dieter},\n\tmonth = sep,\n\tyear = {2018},\n\tpages = {3795--3799},\n}\n\n\n\n
\n
\n\n\n
\n We tested the influence of fundamental oscillation (fo) on human and machine speaker recognition performance in vocalic test utterances. In experiment I, we trained a Gaussian-Mixture model on 15 speakers (80 multi-word utterances each) and tested it with sustained vowel utterances (/a:/, /i:/ and /u:/) under six fo conditions, three changing (fall, rise, fall-rise) and three steady-state (high, mid, low). Results revealed better performance for the steady-state compared to the changing conditions and within the steady-state condition, performance was poorest for high fo. In experiment II, we tested 9 human listeners on a subset of 4 speakers from experiment I. They went through two training tasks (training 1: multi-word utterances; training 2: words). In the test, they recognized speakers based on the same vocalic utterances as in experiment I (for these 4 speakers). Results showed that performance was about equally high for the changing and steady-state vowels, however, in the steady-state condition performance was best for high fo vowels. The experiments suggest that (a) fo has an influence on the strength of speaker specific characteristics in vowels and (b) humans - compared to machines - pay attention to different acoustic information in vocalic utterances for speaker recognition.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The Zurich Corpus of Vowel and Voice Quality , Version 1.0.\n \n \n \n \n\n\n \n Maurer, D.; D'Heureuse, C.; Suter, H.; Dellwo, V.; Friedrichs, D.; and Kathiresan, T.\n\n\n \n\n\n\n In Interspeech, pages 1417–1421, Hyderabad, India, September 2018. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{maurerZurichCorpusVowel2018,\n\taddress = {Hyderabad, India},\n\ttitle = {The {Zurich} {Corpus} of {Vowel} and {Voice} {Quality} , {Version} 1.0},\n\turl = {https://doi.org/10.5167/uzh-156786},\n\tabstract = {Existing databases of isolated vowel sounds or vowel sounds embedded in consonantal context generally document only limited variation of basic production parameters. Thus, concerning the possible variation range of vowel and voice quality-related sound characteristics, there is a lack of broad phenomenological and descriptive references that allow for a comprehensive understanding of vowel acoustics and for an evaluation of the extent to which corresponding existing approaches and models can be generalised. In order to contribute to the building up of such references, a novel database of vowel sounds that exceeds any existing collection by size and diversity of vocalic characteristicsis presented here, comprised of c. 34600 utterances of 70 speakers (46 nonprofessional speakers, children, women and men, and 24 professional actors/actresses and singers of straight theatre, contemporary singing, and European classical singing). The database focuses on sounds of the long Standard German vowels /i?y?e?ø???a?o?u/ produced with varying basic production parameters such as phonation type, vocal effort, fundamental frequency, vowel context and speaking or singing style. In addition, a read text and, for professionals, songs are also included. The database is accessible for scientific use, and further extensions are in progress.},\n\tnumber = {September},\n\tbooktitle = {Interspeech},\n\tpublisher = {ISCA},\n\tauthor = {Maurer, Dieter and D'Heureuse, Christian and Suter, Heidy and Dellwo, Volker and Friedrichs, Daniel and Kathiresan, Thayabaran},\n\tmonth = sep,\n\tyear = {2018},\n\tpages = {1417--1421},\n}\n\n\n\n
\n
\n\n\n
\n Existing databases of isolated vowel sounds or vowel sounds embedded in consonantal context generally document only limited variation of basic production parameters. Thus, concerning the possible variation range of vowel and voice quality-related sound characteristics, there is a lack of broad phenomenological and descriptive references that allow for a comprehensive understanding of vowel acoustics and for an evaluation of the extent to which corresponding existing approaches and models can be generalised. In order to contribute to the building up of such references, a novel database of vowel sounds that exceeds any existing collection by size and diversity of vocalic characteristicsis presented here, comprised of c. 34600 utterances of 70 speakers (46 nonprofessional speakers, children, women and men, and 24 professional actors/actresses and singers of straight theatre, contemporary singing, and European classical singing). The database focuses on sounds of the long Standard German vowels /i?y?e?ø???a?o?u/ produced with varying basic production parameters such as phonation type, vocal effort, fundamental frequency, vowel context and speaking or singing style. In addition, a read text and, for professionals, songs are also included. The database is accessible for scientific use, and further extensions are in progress.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The Effect of Ageing on Speech Rhythm: A Study on Zurich German.\n \n \n \n \n\n\n \n Pellegrino, E.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2018, pages 133–137, June 2018. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{pellegrinoEffectAgeingSpeech2018,\n\ttitle = {The {Effect} of {Ageing} on {Speech} {Rhythm}: {A} {Study} on {Zurich} {German}},\n\tshorttitle = {The {Effect} of {Ageing} on {Speech} {Rhythm}},\n\turl = {https://www.isca-archive.org/speechprosody_2018/pellegrino18_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2018-27},\n\tabstract = {Speech segmental and suprasegmental characteristics vary considerably across the life span, for example, due to degenerative changes in speech production mechanisms and neuro-muscolar control. A great deal of research on the acoustic correlates of adult speakers’ voice has focussed on changes in voice quality, vowel formant patterns, f0, amplitude and speech rate. Only little attention has been paid on speech rhythm variability due to advancing age.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2018},\n\tpublisher = {ISCA},\n\tauthor = {Pellegrino, Elisa and He, Lei and Dellwo, Volker},\n\tmonth = jun,\n\tyear = {2018},\n\tpages = {133--137},\n}\n\n\n\n
\n
\n\n\n
\n Speech segmental and suprasegmental characteristics vary considerably across the life span, for example, due to degenerative changes in speech production mechanisms and neuro-muscolar control. A great deal of research on the acoustic correlates of adult speakers’ voice has focussed on changes in voice quality, vowel formant patterns, f0, amplitude and speech rate. Only little attention has been paid on speech rhythm variability due to advancing age.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Between-speaker rhythmic variability is not dependent on language rhythm, as evidence from Persian reveals.\n \n \n \n \n\n\n \n Asadi, H.; Nourbakhsh, M.; He, L.; Pellegrino, E.; and Dellwo, V.\n\n\n \n\n\n\n International Journal of Speech Language and the Law, 25(2): 151–174. November 2018.\n \n\n\n\n
\n\n\n\n \n \n \"Between-speakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{asadiBetweenspeakerRhythmicVariability2018,\n\ttitle = {Between-speaker rhythmic variability is not dependent on language rhythm, as evidence from {Persian} reveals},\n\tvolume = {25},\n\tissn = {1748-8885},\n\turl = {https://doi.org/10.5167/uzh-159521},\n\tdoi = {10.1558/ijsll.37110},\n\tabstract = {Acoustic measures of speech rhythm based on the durational characteristics of consonantal and vocalic intervals (henceforth C- or V-intervals) as well as syllabic intensity reveal between-speaker variability. The evidence obtained so far is based on speakers of stressed-timed languages, which are assumed to have complex consonant clusters and a higher degree of vowel reduction. Speakers of stressed-timed languages might operate their articulatory organs in different ways due to the syllable complexity and vowel reduction. Complex consonant clusters are released differently, and vowel reduction tends to be produced more or less strongly depending on speakers. When a language lacks such features, it is possible that rhythmic variation between its speakers decreases. In the present study, we aimed at exploring between- and within-speaker rhythmic variability in Persian, an Indo-European language categorised as syllable-timed. Acoustic correlates of speech rhythm (\\%V, \\{{\\textbackslash}ensuremath\\{{\\textbackslash}Delta\\}\\}V[ln], \\{{\\textbackslash}ensuremath\\{{\\textbackslash}Delta\\}\\}C[ln], n-PVI-V) and articulation rate were obtained from two Persian corpora with different sources of within-speaker variability. In the first corpus, the source of within-speaker variability mainly comes from non-contemporaneous recording sessions, and in the second corpus, from different speech rates. Results revealed that there were significant differences between speakers in all investigated speech rhythm measures in Persian and \\%V best discriminated between speakers. This reveals that the lack of typical stress-time features does not affect between-speaker variability in speech rhythm.},\n\tnumber = {2},\n\tjournal = {International Journal of Speech Language and the Law},\n\tpublisher = {Equinox Publishing Ltd.},\n\tauthor = {Asadi, Homa and Nourbakhsh, Mandana and He, Lei and Pellegrino, Elisa and Dellwo, Volker},\n\tmonth = nov,\n\tyear = {2018},\n\tpages = {151--174},\n}\n\n\n\n
\n
\n\n\n
\n Acoustic measures of speech rhythm based on the durational characteristics of consonantal and vocalic intervals (henceforth C- or V-intervals) as well as syllabic intensity reveal between-speaker variability. The evidence obtained so far is based on speakers of stressed-timed languages, which are assumed to have complex consonant clusters and a higher degree of vowel reduction. Speakers of stressed-timed languages might operate their articulatory organs in different ways due to the syllable complexity and vowel reduction. Complex consonant clusters are released differently, and vowel reduction tends to be produced more or less strongly depending on speakers. When a language lacks such features, it is possible that rhythmic variation between its speakers decreases. In the present study, we aimed at exploring between- and within-speaker rhythmic variability in Persian, an Indo-European language categorised as syllable-timed. Acoustic correlates of speech rhythm (%V, \\\\ensuremath\\\\Delta\\\\V[ln], \\\\ensuremath\\\\Delta\\\\C[ln], n-PVI-V) and articulation rate were obtained from two Persian corpora with different sources of within-speaker variability. In the first corpus, the source of within-speaker variability mainly comes from non-contemporaneous recording sessions, and in the second corpus, from different speech rates. Results revealed that there were significant differences between speakers in all investigated speech rhythm measures in Persian and %V best discriminated between speakers. This reveals that the lack of typical stress-time features does not affect between-speaker variability in speech rhythm.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2017\n \n \n (9)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Vowel recognition at fundamental frequencies up to 1 kHz reveals point vowels as acoustic landmarks.\n \n \n \n \n\n\n \n Friedrichs, D.; Maurer, D.; Rosen, S.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 142(2): 1025–1033. 2017.\n \n\n\n\n
\n\n\n\n \n \n \"VowelPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{friedrichsVowelRecognitionFundamental2017,\n\ttitle = {Vowel recognition at fundamental frequencies up to 1 {kHz} reveals point vowels as acoustic landmarks},\n\tvolume = {142},\n\tissn = {0001-4966},\n\turl = {http://dx.doi.org/10.1121/1.4998706},\n\tdoi = {10.1121/1.4998706},\n\tabstract = {© 2017 Acoustical Society of America. The phonological function of vowels can be maintained at fundamental frequencies (f o ) up to 880 Hz [Friedrichs, Maurer, and Dellwo (2015). J. Acoust. Soc. Am. 138, EL36-EL42]. Here, the influence of talker variability and multiple response options on vowel recognition at high f o s is assessed. The stimuli (n = 264) consisted of eight isolated vowels (/i y e ø ϵ a o u/) produced by three female native German talkers at 11 f o s within a range of 220-1046 Hz. In a closed-set identification task, 21 listeners were presented excised 700-ms vowel nuclei with quasi-flat f o contours and resonance trajectories. The results show that listeners can identify the point vowels /i a u/ at f o s up to almost 1 kHz, with a significant decrease for the vowels /y ϵ/ and a drop to chance level for the vowels /e ø o/ toward the upper f o s. Auditory excitation patterns reveal highly differentiable representations for /i a u/ that can be used as landmarks for vowel category perception at high f o s. These results suggest that theories of vowel perception based on overall spectral shape will provide a fuller account of vowel perception than those based solely on formant frequency patterns.},\n\tnumber = {2},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {Friedrichs, Daniel and Maurer, Dieter and Rosen, Stuart and Dellwo, Volker},\n\tyear = {2017},\n\tpages = {1025--1033},\n}\n\n\n\n
\n
\n\n\n
\n © 2017 Acoustical Society of America. The phonological function of vowels can be maintained at fundamental frequencies (f o ) up to 880 Hz [Friedrichs, Maurer, and Dellwo (2015). J. Acoust. Soc. Am. 138, EL36-EL42]. Here, the influence of talker variability and multiple response options on vowel recognition at high f o s is assessed. The stimuli (n = 264) consisted of eight isolated vowels (/i y e ø ϵ a o u/) produced by three female native German talkers at 11 f o s within a range of 220-1046 Hz. In a closed-set identification task, 21 listeners were presented excised 700-ms vowel nuclei with quasi-flat f o contours and resonance trajectories. The results show that listeners can identify the point vowels /i a u/ at f o s up to almost 1 kHz, with a significant decrease for the vowels /y ϵ/ and a drop to chance level for the vowels /e ø o/ toward the upper f o s. Auditory excitation patterns reveal highly differentiable representations for /i a u/ that can be used as landmarks for vowel category perception at high f o s. These results suggest that theories of vowel perception based on overall spectral shape will provide a fuller account of vowel perception than those based solely on formant frequency patterns.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Between-speaker variability in temporal organizations of intensity contours.\n \n \n \n \n\n\n \n He, L.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 141(5): EL488–EL494. 2017.\n \n\n\n\n
\n\n\n\n \n \n \"Between-speakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{He2017,\n\ttitle = {Between-speaker variability in temporal organizations of intensity contours},\n\tvolume = {141},\n\tissn = {0001-4966},\n\turl = {http://asa.scitation.org/doi/10.1121/1.4983398},\n\tdoi = {10.1121/1.4983398},\n\tabstract = {Intensity contours of speech signals were sub-divided into positive and negative dynamics. Positive dynamics were defined as the speed of increases in intensity from amplitude troughs to subsequent peaks, and negative dynamics as the speed of decreases in intensity from peaks to troughs. Mean, standard deviation, and sequential variability were measured for both dynamics in each sentence. Analyses showed that measures of both dynamics were separately classified and between- speaker variability was largely explained by measures of negative dynamics. This suggests that parts of the signal where intensity decreases from syllable peaks are more speaker-specific. Idiosyncratic articulation may explain such results.},\n\tnumber = {5},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {He, Lei and Dellwo, Volker},\n\tyear = {2017},\n\tpages = {EL488--EL494},\n}\n\n\n\n
\n
\n\n\n
\n Intensity contours of speech signals were sub-divided into positive and negative dynamics. Positive dynamics were defined as the speed of increases in intensity from amplitude troughs to subsequent peaks, and negative dynamics as the speed of decreases in intensity from peaks to troughs. Mean, standard deviation, and sequential variability were measured for both dynamics in each sentence. Analyses showed that measures of both dynamics were separately classified and between- speaker variability was largely explained by measures of negative dynamics. This suggests that parts of the signal where intensity decreases from syllable peaks are more speaker-specific. Idiosyncratic articulation may explain such results.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Listeners use temporal information to identify French- and English-accented speech.\n \n \n \n\n\n \n Kolly, M. M. J.; Boula de Mareüil, P.; Leemann, A.; and Dellwo, V.\n\n\n \n\n\n\n Speech Communication, 86: 121–134. 2017.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Kolly2017,\n\ttitle = {Listeners use temporal information to identify {French}- and {English}-accented speech},\n\tvolume = {86},\n\tissn = {01676393},\n\tdoi = {10.1016/j.specom.2016.11.006},\n\tabstract = {Which acoustic cues can be used by listeners to identify speakers’ linguistic origins in foreign-accented speech? We investigated accent identification performance in signal-manipulated speech, where (a) Swiss German listeners heard native German speech to which we transplanted segment durations of French-accented German and English-accented German, and (b) Swiss German listeners heard 6-band noise-vocoded French-accented and English-accented German speech to which we transplanted native German segment durations. Therefore, the foreign accent cues in the stimuli consisted of only temporal information (in a) and only strongly degraded spectral information (in b). Findings suggest that listeners were able to identify the linguistic origin of French and English speakers in their foreign-accented German speech based on temporal features alone, as well as based on strongly degraded spectral features alone. When comparing these results to previous research, we found an additive trend of temporal and spectral cues: identification performance tended to be higher when both cues were present in the signal. Acoustic measures of temporal variability could not easily explain the perceptual results. However, listeners were drawn towards some of the native German segmental cues in condition (a), which biased responses towards ‘French’ when stimuli featured uvular /r/s and towards ‘English’ when they contained vocalized /r/s or lacked /r/.},\n\tjournal = {Speech Communication},\n\tauthor = {Kolly, M.-J. Marie José and Boula de Mareüil, Philippe and Leemann, Adrian and Dellwo, Volker},\n\tyear = {2017},\n\tpages = {121--134},\n}\n\n\n\n
\n
\n\n\n
\n Which acoustic cues can be used by listeners to identify speakers’ linguistic origins in foreign-accented speech? We investigated accent identification performance in signal-manipulated speech, where (a) Swiss German listeners heard native German speech to which we transplanted segment durations of French-accented German and English-accented German, and (b) Swiss German listeners heard 6-band noise-vocoded French-accented and English-accented German speech to which we transplanted native German segment durations. Therefore, the foreign accent cues in the stimuli consisted of only temporal information (in a) and only strongly degraded spectral information (in b). Findings suggest that listeners were able to identify the linguistic origin of French and English speakers in their foreign-accented German speech based on temporal features alone, as well as based on strongly degraded spectral features alone. When comparing these results to previous research, we found an additive trend of temporal and spectral cues: identification performance tended to be higher when both cues were present in the signal. Acoustic measures of temporal variability could not easily explain the perceptual results. However, listeners were drawn towards some of the native German segmental cues in condition (a), which biased responses towards ‘French’ when stimuli featured uvular /r/s and towards ‘English’ when they contained vocalized /r/s or lacked /r/.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Intonation and talker variability in the discrimination of Spanish lexical stress contrasts by Spanish, German and French listeners.\n \n \n \n\n\n \n Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 142(4): 2419–2429. 2017.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Schwab2017,\n\ttitle = {Intonation and talker variability in the discrimination of {Spanish} lexical stress contrasts by {Spanish}, {German} and {French} listeners},\n\tvolume = {142},\n\tissn = {0001-4966},\n\tdoi = {10.1121/1.5008849},\n\tabstract = {© 2017 Acoustical Society of America. The perception of stress is highly influenced by listeners' native language. In this research, the authors examined the effect of intonation and talker variability (here: phonetic variability) in the discrimination of Spanish lexical stress contrasts by native Spanish (N = 17), German (N = 21), and French (N = 27) listeners. Participants listened to 216 trials containing three Spanish disyllabic words, where one word carried a different lexical stress to the others. The listeners' task was to identify the deviant word in each trial (Odd-One-Out task). The words in the trials were produced by either the same talker or by two different talkers, and carried the same or varying intonation patterns. The German listeners' performance was lower compared to the Spanish listeners but higher than that of the French listeners. French listeners performed above chance level with and without talker variability, and performed at chance level when intonation variability was introduced. Results are discussed in the context of the stress "deafness" hypothesis.},\n\tnumber = {4},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tauthor = {Schwab, Sandra and Dellwo, Volker},\n\tyear = {2017},\n\tpages = {2419--2429},\n}\n\n\n\n
\n
\n\n\n
\n © 2017 Acoustical Society of America. The perception of stress is highly influenced by listeners' native language. In this research, the authors examined the effect of intonation and talker variability (here: phonetic variability) in the discrimination of Spanish lexical stress contrasts by native Spanish (N = 17), German (N = 21), and French (N = 27) listeners. Participants listened to 216 trials containing three Spanish disyllabic words, where one word carried a different lexical stress to the others. The listeners' task was to identify the deviant word in each trial (Odd-One-Out task). The words in the trials were produced by either the same talker or by two different talkers, and carried the same or varying intonation patterns. The German listeners' performance was lower compared to the Spanish listeners but higher than that of the French listeners. French listeners performed above chance level with and without talker variability, and performed at chance level when intonation variability was introduced. Results are discussed in the context of the stress \"deafness\" hypothesis.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Perception of vocal tract tension: exploring possible prosodic correlates.\n \n \n \n \n\n\n \n San Segundo, E.; Schwab, S.; Dellwo, V.; He, L.; and Mompeán, J.\n\n\n \n\n\n\n . November 2017.\n \n\n\n\n
\n\n\n\n \n \n \"PerceptionPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{sansegundoPerceptionVocalTract2017,\n\ttitle = {Perception of vocal tract tension: exploring possible prosodic correlates},\n\tshorttitle = {Perception of vocal tract tension},\n\turl = {https://www.zora.uzh.ch/id/eprint/145206},\n\tdoi = {10.5167/UZH-145206},\n\tabstract = {A recent study involving the perceptual analysis of 24 speakers by two raters (San Segundo Mompeán, 2017) revealed a slight inter-rater agreement in the assessment of vocal tract tension. In the current investigation several prosodic measures related to intensity and durational variability have been extracted per speaker with the aim of testing whether they correlate with the perceptual ratings for VT tension provided by the two trained raters. The correlation test showed a significant positive correlation between the ratings of Rater 1 and the variable varcoM (mean intensity variability across syllables). In contrast, the ratings of Rater 2 correlated positive and significantly with two rhythmic measures related to mean consonant duration. These results suggest that the acoustic cues playing a role in each rater’s auditory judgements are not the same. The different salience of intensity and durational characteristics should be taken into account in future studies on voice quality perception.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {UNED},\n\tauthor = {San Segundo, Eugenia and Schwab, Sandra and Dellwo, Volker and He, Lei and Mompeán, José},\n\tmonth = nov,\n\tyear = {2017},\n}\n\n\n\n
\n
\n\n\n
\n A recent study involving the perceptual analysis of 24 speakers by two raters (San Segundo Mompeán, 2017) revealed a slight inter-rater agreement in the assessment of vocal tract tension. In the current investigation several prosodic measures related to intensity and durational variability have been extracted per speaker with the aim of testing whether they correlate with the perceptual ratings for VT tension provided by the two trained raters. The correlation test showed a significant positive correlation between the ratings of Rater 1 and the variable varcoM (mean intensity variability across syllables). In contrast, the ratings of Rater 2 correlated positive and significantly with two rhythmic measures related to mean consonant duration. These results suggest that the acoustic cues playing a role in each rater’s auditory judgements are not the same. The different salience of intensity and durational characteristics should be taken into account in future studies on voice quality perception.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Computation of L2 speech rhythm based on duration and fundamental frequency.\n \n \n \n \n\n\n \n Pellegrino, E.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n . March 2017.\n \n\n\n\n
\n\n\n\n \n \n \"ComputationPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{pellegrinoComputationL2Speech2017,\n\ttitle = {Computation of {L2} speech rhythm based on duration and fundamental frequency},\n\turl = {https://www.zora.uzh.ch/id/eprint/136466},\n\tdoi = {10.5167/UZH-136466},\n\turldate = {2025-04-18},\n\tpublisher = {TUDpress},\n\tauthor = {Pellegrino, Elisa and He, Lei and Dellwo, Volker},\n\tmonth = mar,\n\tyear = {2017},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speaker Identification for Swiss German with Spectral and Rhythm Features.\n \n \n \n\n\n \n Lykartsis, A.; Weinzierl, S.; and Dellwo, V.\n\n\n \n\n\n\n . 2017.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{lykartsisSpeakerIdentificationSwiss2017,\n\ttitle = {Speaker {Identification} for {Swiss} {German} with {Spectral} and {Rhythm} {Features}},\n\tabstract = {We present results of speech rhythm analysis for automatic speaker identification. We expand previous experiments using similar methods for language identification. Features describing the rhythmic properties of salient changes in signal components are extracted and used in an speaker identification task to determine to which extent they are descriptive of speaker variability. We also test the performance of state-of-the-art but simple-to-extract frame-based features. The paper focus is the evaluation on one corpus (swiss german, TEVOID) using support vector machines. Results suggest that the general spectral features can provide very good performance on this dataset, whereas the rhythm features are not as successful in the task, indicating either the lack of suitability for this task or the dataset specificity.},\n\tlanguage = {en},\n\tauthor = {Lykartsis, Athanasios and Weinzierl, Stefan and Dellwo, Volker},\n\tyear = {2017},\n}\n\n\n\n
\n
\n\n\n
\n We present results of speech rhythm analysis for automatic speaker identification. We expand previous experiments using similar methods for language identification. Features describing the rhythmic properties of salient changes in signal components are extracted and used in an speaker identification task to determine to which extent they are descriptive of speaker variability. We also test the performance of state-of-the-art but simple-to-extract frame-based features. The paper focus is the evaluation on one corpus (swiss german, TEVOID) using support vector machines. Results suggest that the general spectral features can provide very good performance on this dataset, whereas the rhythm features are not as successful in the task, indicating either the lack of suitability for this task or the dataset specificity.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Enhancing the objectivity of interactive formant estimation: introducing euclidean distance measure and numerical conditions for numbers and frequency ranges of formants.\n \n \n \n \n\n\n \n Kathiresan, T.; Maurer, D.; and Dellwo, V.\n\n\n \n\n\n\n . 2017.\n \n\n\n\n
\n\n\n\n \n \n \"EnhancingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kathiresanEnhancingObjectivityInteractive2017,\n\ttitle = {Enhancing the objectivity of interactive formant estimation: introducing euclidean distance measure and numerical conditions for numbers and frequency ranges of formants},\n\tshorttitle = {Enhancing the objectivity of interactive formant estimation},\n\turl = {https://www.zora.uzh.ch/id/eprint/149558},\n\tdoi = {10.5167/UZH-149558},\n\tabstract = {Current formant measurement studies of vowel sounds generally use a Linear Predictive Coding (LPC) algorithm and rely on an interactive method of formant estimation which includes a comparison of measured formant tracks and characteristics of the spectrogram. Thereby, the selection of LPC parameters is based on the assumption that the number of poles for the analysis of a given frequency range is age- and gender-specific. However, when crosschecking measured formant tracks with the spectrogram, mismatches occur in a significant number of cases. In these cases, the investigators try to minimize these mismatches by modifying the number of poles of LPC. Such an interaction is based on phonetic knowledge, analytical experience and related expectations. Several authors have pointed towards the lack of objectivity and the inherent circularity as well as the fact that similar formant estimations performed by different researchers may yield different results. As of yet, the issue of an improvement and objectification of formant estimation procedure is still a matter of debate. The present paper describes such a corresponding approach: basing the LPC pole-number selection on objective criteria by introducing Euclidean distance measure and formant frequency conditions as references for interactive formant frequency estimation. The paper further presents and discusses the results of a pilot evaluation using the method proposed on 224 long Standard German vowel sounds /i-y-e-ø-ɛ-a-o-u/ produced by eight children, ten women and ten men on fundamental frequencies of 262 Hz (children), 220 Hz (women) and 131 Hz (men), respectively.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {TUDpress},\n\tauthor = {Kathiresan, Thayabaran and Maurer, Dieter and Dellwo, Volker},\n\tyear = {2017},\n}\n\n\n\n
\n
\n\n\n
\n Current formant measurement studies of vowel sounds generally use a Linear Predictive Coding (LPC) algorithm and rely on an interactive method of formant estimation which includes a comparison of measured formant tracks and characteristics of the spectrogram. Thereby, the selection of LPC parameters is based on the assumption that the number of poles for the analysis of a given frequency range is age- and gender-specific. However, when crosschecking measured formant tracks with the spectrogram, mismatches occur in a significant number of cases. In these cases, the investigators try to minimize these mismatches by modifying the number of poles of LPC. Such an interaction is based on phonetic knowledge, analytical experience and related expectations. Several authors have pointed towards the lack of objectivity and the inherent circularity as well as the fact that similar formant estimations performed by different researchers may yield different results. As of yet, the issue of an improvement and objectification of formant estimation procedure is still a matter of debate. The present paper describes such a corresponding approach: basing the LPC pole-number selection on objective criteria by introducing Euclidean distance measure and formant frequency conditions as references for interactive formant frequency estimation. The paper further presents and discusses the results of a pilot evaluation using the method proposed on 224 long Standard German vowel sounds /i-y-e-ø-ɛ-a-o-u/ produced by eight children, ten women and ten men on fundamental frequencies of 262 Hz (children), 220 Hz (women) and 131 Hz (men), respectively.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Amplitude envelope kinematics of speech signal: parameter extraction and applications.\n \n \n \n \n\n\n \n He, L.; and Dellwo, V.\n\n\n \n\n\n\n . 2017.\n \n\n\n\n
\n\n\n\n \n \n \"AmplitudePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{heAmplitudeEnvelopeKinematics2017,\n\ttitle = {Amplitude envelope kinematics of speech signal: parameter extraction and applications},\n\tshorttitle = {Amplitude envelope kinematics of speech signal},\n\turl = {https://www.zora.uzh.ch/id/eprint/136290},\n\tdoi = {10.5167/UZH-136290},\n\tabstract = {In this paper, we model the amplitude envelope of the broadband speech signal as a kinematic system and calculate its basic parameters, including displacement, velocity and acceleration. Such system captures the smoothed amplitude fluctuation pattern over time, illustrating how energy is distributed across the signal. Although the pulmonic air pressure is the primary energy source of speech, the amplitude modulation pattern is largely determined by articulatory behaviours, especially mandible and lip movements. Therefore, there should be a correspondence between signal envelope kinematics and articulator kinematics. Previous research showed that a tremendous amount of speaker idiosyncrasies in articulation existed. Such idiosyncrasies should therefore be reflected in the envelope kinematics as well. From the signal envelope kinematics, it may be possible to infer individual articulatory behaviours. This is particularly useful for forensic phoneticians who usually have no access to articulatory data, and clinical speech pathologists who usually find it difficult to make articulatory measurement in clinical consultations. Also in this paper, we illustrate a correspondence between the amplitude envelope kinematics and the lower lip kinematics (X-ray pellet history data) of one speaker reading one sentence. For future research, more speakers are needed to record both speech and articulatory signals to build a statistical model between the kinematics data of both domains.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {TUDpress},\n\tauthor = {He, Lei and Dellwo, Volker},\n\tyear = {2017},\n}\n\n\n\n
\n
\n\n\n
\n In this paper, we model the amplitude envelope of the broadband speech signal as a kinematic system and calculate its basic parameters, including displacement, velocity and acceleration. Such system captures the smoothed amplitude fluctuation pattern over time, illustrating how energy is distributed across the signal. Although the pulmonic air pressure is the primary energy source of speech, the amplitude modulation pattern is largely determined by articulatory behaviours, especially mandible and lip movements. Therefore, there should be a correspondence between signal envelope kinematics and articulator kinematics. Previous research showed that a tremendous amount of speaker idiosyncrasies in articulation existed. Such idiosyncrasies should therefore be reflected in the envelope kinematics as well. From the signal envelope kinematics, it may be possible to infer individual articulatory behaviours. This is particularly useful for forensic phoneticians who usually have no access to articulatory data, and clinical speech pathologists who usually find it difficult to make articulatory measurement in clinical consultations. Also in this paper, we illustrate a correspondence between the amplitude envelope kinematics and the lower lip kinematics (X-ray pellet history data) of one speaker reading one sentence. For future research, more speakers are needed to record both speech and articulatory signals to build a statistical model between the kinematics data of both domains.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2016\n \n \n (7)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n A Praat-Based Algorithm to Extract the Amplitude Envelope and Temporal Fine Structure Using the Hilbert Transform.\n \n \n \n \n\n\n \n He, L.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2016, pages 530–534, September 2016. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"APaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{hePraatBasedAlgorithmExtract2016,\n\ttitle = {A {Praat}-{Based} {Algorithm} to {Extract} the {Amplitude} {Envelope} and {Temporal} {Fine} {Structure} {Using} the {Hilbert} {Transform}},\n\turl = {https://www.isca-archive.org/interspeech_2016/he16_interspeech.html},\n\tdoi = {10.21437/Interspeech.2016-1447},\n\tabstract = {A speech signal can be viewed as a high frequency carrier signal containing the temporal fine structure (TFS) that is modulated by a low frequency envelope (ENV). A widely used method to decompose a speech signal into the TFS and ENV is the Hilbert transform. Although this method has been available for about one century and is widely applied in various kinds of speech processing tasks (e.g. speech chimeras), there are only very few speech processing packages that contain readily available functions for the Hilbert transform, and there is very little textbook type literature tailored for speech scientists to explain the processes behind the transform. With this paper we provide the code for carrying out the Hilbert operation to obtain the TFS and ENV in the widely used speech processing software Praat, and explain the basics of the procedure. To verify our code, we compare the Hilbert transform in Praat with a widely applied function for the same purpose in MATLAB (“hilbert(...)”). We can confirm that both methods arrive at identical outputs.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2016},\n\tpublisher = {ISCA},\n\tauthor = {He, Lei and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2016},\n\tpages = {530--534},\n}\n\n\n\n
\n
\n\n\n
\n A speech signal can be viewed as a high frequency carrier signal containing the temporal fine structure (TFS) that is modulated by a low frequency envelope (ENV). A widely used method to decompose a speech signal into the TFS and ENV is the Hilbert transform. Although this method has been available for about one century and is widely applied in various kinds of speech processing tasks (e.g. speech chimeras), there are only very few speech processing packages that contain readily available functions for the Hilbert transform, and there is very little textbook type literature tailored for speech scientists to explain the processes behind the transform. With this paper we provide the code for carrying out the Hilbert operation to obtain the TFS and ENV in the widely used speech processing software Praat, and explain the basics of the procedure. To verify our code, we compare the Hilbert transform in Praat with a widely applied function for the same purpose in MATLAB (“hilbert(...)”). We can confirm that both methods arrive at identical outputs.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Auditory voice identification based on visual information: Listeners seem to rely on fine-grained temporal information.\n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n In Proceedings of the International Association of Forensic Fonetics and Acoustics, 2016. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Dellwo2016,\n\ttitle = {Auditory voice identification based on visual information: {Listeners} seem to rely on fine-grained temporal information},\n\tbooktitle = {Proceedings of the {International} {Association} of {Forensic} {Fonetics} and {Acoustics}},\n\tauthor = {Dellwo, Volker},\n\tyear = {2016},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The use of the Odd-One-Out task in the study of the perception of lexical stress in Spanish by German-speaking listeners.\n \n \n \n \n\n\n \n Schwab, S.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2016, pages 252–256, May 2016. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{schwabUseOddOneOutTask2016,\n\ttitle = {The use of the {Odd}-{One}-{Out} task in the study of the perception of lexical stress in {Spanish} by {German}-speaking listeners},\n\turl = {https://www.isca-archive.org/speechprosody_2016/schwab16b_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2016-52},\n\tabstract = {In the present research we investigated the perception of Spanish stress in German-speaking listeners in comparison with native Spanish listeners. We used a cognitively demanding Odd-One-Out task and stimuli with variability in voice and/or in intonation. The main findings showed that the German-speaking listeners were able to perceive the Spanish lexical stress to a very high degree (76\\% of correct responses), but that their performance was lower than the Spanish listeners' performance (90\\%). The difference between German and Spanish speakers was mainly due to the German speakers' poorer detection of the odd in two specific accentual contrasts. The implications on the stress deafness hypothesis are discussed.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2016},\n\tpublisher = {ISCA},\n\tauthor = {Schwab, Sandra and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2016},\n\tpages = {252--256},\n}\n\n\n\n
\n
\n\n\n
\n In the present research we investigated the perception of Spanish stress in German-speaking listeners in comparison with native Spanish listeners. We used a cognitively demanding Odd-One-Out task and stimuli with variability in voice and/or in intonation. The main findings showed that the German-speaking listeners were able to perceive the Spanish lexical stress to a very high degree (76% of correct responses), but that their performance was lower than the Spanish listeners' performance (90%). The difference between German and Spanish speakers was mainly due to the German speakers' poorer detection of the odd in two specific accentual contrasts. The implications on the stress deafness hypothesis are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker-individual rhythmic characteristics in read speech of German-Italian bilinguals.\n \n \n \n \n\n\n \n Schmid, S.; and Dellwo, V.\n\n\n \n\n\n\n Trends in Phonetics and Phonology,349–362. September 2016.\n Place: Bern ISBN: 978-3-0343-1653-8\n\n\n\n
\n\n\n\n \n \n \"Speaker-individualPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schmidSpeakerindividualRhythmicCharacteristics2016,\n\ttitle = {Speaker-individual rhythmic characteristics in read speech of {German}-{Italian} bilinguals},\n\turl = {https://doi.org/10.5167/uzh-114508},\n\tdoi = {10.3726/978-3-0351-0869-9/38},\n\tjournal = {Trends in Phonetics and Phonology},\n\tpublisher = {Peter Lang},\n\tauthor = {Schmid, Stephan and Dellwo, Volker},\n\teditor = {Leemann, Adrian and Kolly, Marie-José and Schmid, Stephan and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2016},\n\tnote = {Place: Bern\nISBN: 978-3-0343-1653-8},\n\tpages = {349--362},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n The role of syllable intensity in between-speaker rhythmic variability.\n \n \n \n\n\n \n He, L.; and Dellwo, V.\n\n\n \n\n\n\n International Journal of Speech, Language and the Law, 23(2): 243–273. 2016.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{He2016,\n\ttitle = {The role of syllable intensity in between-speaker rhythmic variability},\n\tvolume = {23},\n\tissn = {17488893},\n\tdoi = {10.1558/ijsll.v23i2.30345},\n\tabstract = {Speech rhythm in terms of durational variability of different levels of phonetic intervals can vary between speakers. The present article examines the role of syllabic intensity characteristics in rhythmic variability. Mean and peak intensity variability across syllables (stdevM, varcoM, stdevP, varcoP, rPVIm, nPVIm, rPVIp, nPVIp; henceforth: intensity measures) were investigated as a function of speaker in a database where within-speaker variability was strong (BonnTempo) and another database designed to examine between-speaker rhythmic variability (TEVOID). It was found that the intensity measures varied significantly between speakers in both databases. Semiautomatic speaker recognition based on duration measures (\\%V, ΔV(ln), ΔC(ln), ΔPeak(ln), ΔSyll(ln) and nPVISyll) and intensity measures using multinomial logistic regression and feedforward neural networks was carried out for the two databases. Results showed that intensity measures contained stronger speaker specific information compared to measures based on durational variability of phonetic intervals. In addition, effects of the recognition algorithms (speaker recognition using multinomial logistic regression was significantly better than neural networks for BonnTempo) and data normalisation procedures (z-score normalised data was significantly better than non-normalised data in TEVOID) were discovered. This means that syllable intensity characteristics play an important role in between-speaker rhythmic differences and possibly in speech rhythm variability in general.},\n\tnumber = {2},\n\tjournal = {International Journal of Speech, Language and the Law},\n\tauthor = {He, Lei and Dellwo, Volker},\n\tyear = {2016},\n\tpages = {243--273},\n}\n\n\n\n
\n
\n\n\n
\n Speech rhythm in terms of durational variability of different levels of phonetic intervals can vary between speakers. The present article examines the role of syllabic intensity characteristics in rhythmic variability. Mean and peak intensity variability across syllables (stdevM, varcoM, stdevP, varcoP, rPVIm, nPVIm, rPVIp, nPVIp; henceforth: intensity measures) were investigated as a function of speaker in a database where within-speaker variability was strong (BonnTempo) and another database designed to examine between-speaker rhythmic variability (TEVOID). It was found that the intensity measures varied significantly between speakers in both databases. Semiautomatic speaker recognition based on duration measures (%V, ΔV(ln), ΔC(ln), ΔPeak(ln), ΔSyll(ln) and nPVISyll) and intensity measures using multinomial logistic regression and feedforward neural networks was carried out for the two databases. Results showed that intensity measures contained stronger speaker specific information compared to measures based on durational variability of phonetic intervals. In addition, effects of the recognition algorithms (speaker recognition using multinomial logistic regression was significantly better than neural networks for BonnTempo) and data normalisation procedures (z-score normalised data was significantly better than non-normalised data in TEVOID) were discovered. This means that syllable intensity characteristics play an important role in between-speaker rhythmic differences and possibly in speech rhythm variability in general.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Trends in Phonetics and Phonology.\n \n \n \n \n\n\n \n Leemann, A.; Kolly, M.; Schmid, S.; and Dellwo, V.\n\n\n \n\n\n\n Peter Lang, January 2016.\n \n\n\n\n
\n\n\n\n \n \n \"TrendsPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@book{leemannTrendsPhoneticsPhonology2016,\n\ttitle = {Trends in {Phonetics} and {Phonology}},\n\tisbn = {978-3-0351-0869-9 978-3-0351-9369-5 978-3-0351-9368-8 978-3-0343-1653-8},\n\turl = {https://www.peterlang.com/view/product/46578},\n\tdoi = {10.3726/978-3-0351-0869-9},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {Peter Lang},\n\tauthor = {Leemann, Adrian and Kolly, Marie-José and Schmid, Stephan and Dellwo, Volker},\n\tmonth = jan,\n\tyear = {2016},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n PresenterPro: A tool for recording, indexing and processing prompted speech with Praat.\n \n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n . October 2016.\n \n\n\n\n
\n\n\n\n \n \n \"PresenterPro:Paper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoPresenterProToolRecording2016,\n\ttitle = {{PresenterPro}: {A} tool for recording, indexing and processing prompted speech with {Praat}},\n\tshorttitle = {{PresenterPro}},\n\turl = {https://www.zora.uzh.ch/id/eprint/127817},\n\tdoi = {10.5167/UZH-127817},\n\tabstract = {Praat (www.praat.org) is a powerful tool for a wide variety of speech analysis and processing tasks. When it comes to recording speech, however, it lacks some fundamental functions that allow a user to prompt a reader with a written list of words or sentences (henceforth: speech prompts) on a screen and index the prompted recordings for further processing. PresenterPro a Praat plug-in - fills this gap. It (a) prompts a reader to read utterances from a screen, (b) automatically indexes the recorded speech prompts in a Praat TextGrid and (c) extracts all recorded speech prompts into individual files. It thus offers an efficient solution for recording large lists of speech prompts. The present paper describes the plug-in and discusses in which situations it is particularly useful.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Dellwo, Volker},\n\tmonth = oct,\n\tyear = {2016},\n}\n\n\n\n
\n
\n\n\n
\n Praat (www.praat.org) is a powerful tool for a wide variety of speech analysis and processing tasks. When it comes to recording speech, however, it lacks some fundamental functions that allow a user to prompt a reader with a written list of words or sentences (henceforth: speech prompts) on a screen and index the prompted recordings for further processing. PresenterPro a Praat plug-in - fills this gap. It (a) prompts a reader to read utterances from a screen, (b) automatically indexes the recorded speech prompts in a Praat TextGrid and (c) extracts all recorded speech prompts into individual files. It thus offers an efficient solution for recording large lists of speech prompts. The present paper describes the plug-in and discusses in which situations it is particularly useful.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2015\n \n \n (13)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Stable and unstable intervals as a basic segmentation procedure of the speech signal.\n \n \n \n \n\n\n \n Glavitsch, U.; He, L.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2015, pages 31–35, September 2015. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"StablePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{glavitschStableUnstableIntervals2015,\n\ttitle = {Stable and unstable intervals as a basic segmentation procedure of the speech signal},\n\turl = {https://www.isca-archive.org/interspeech_2015/glavitsch15_interspeech.html},\n\tdoi = {10.21437/Interspeech.2015-7},\n\tabstract = {The concept of acoustically stable and unstable intervals to structure continuous speech is introduced. We present a method to compute stable intervals efficiently and reliably as a bottom-up approach at an early processing stage. We argue that such intervals stand in close relation to the rhythm of speech as they contribute to the overall temporal organization of the speech production process and the acoustic signal (stable intervals = intervals of reduced movement of certain articulators; unstable intervals = intervals of enhanced movement of certain articulators). To test the relationship of stability intervals with speech rhythm we investigated the between-speaker variability of stable and unstable intervals in the TEVOID corpus. Results revealed that significant between-speaker variability exists. We hypothesize from our findings that the basic segmentation of speech into stable and unstable intervals is a process that might play a role in human perception and processing of speech.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2015},\n\tpublisher = {ISCA},\n\tauthor = {Glavitsch, Ulrike and He, Lei and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2015},\n\tpages = {31--35},\n}\n\n\n\n
\n
\n\n\n
\n The concept of acoustically stable and unstable intervals to structure continuous speech is introduced. We present a method to compute stable intervals efficiently and reliably as a bottom-up approach at an early processing stage. We argue that such intervals stand in close relation to the rhythm of speech as they contribute to the overall temporal organization of the speech production process and the acoustic signal (stable intervals = intervals of reduced movement of certain articulators; unstable intervals = intervals of enhanced movement of certain articulators). To test the relationship of stability intervals with speech rhythm we investigated the between-speaker variability of stable and unstable intervals in the TEVOID corpus. Results revealed that significant between-speaker variability exists. We hypothesize from our findings that the basic segmentation of speech into stable and unstable intervals is a process that might play a role in human perception and processing of speech.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speaker-idiosyncrasy in pausing behavior: Evidence from a cross-linguistic study.\n \n \n \n\n\n \n Kolly, M.; Leemann, A.; Boula de Mareüil, P.; and Dellwo, V.\n\n\n \n\n\n\n In 18th International Congress of Phonetic Sciences ICPHS 2015, 2015. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Kolly2015,\n\ttitle = {Speaker-idiosyncrasy in pausing behavior: {Evidence} from a cross-linguistic study},\n\tabstract = {Phoneticians study acoustic speech signals. But what about the aspects of speech where the signal is silent? The present study investigated speakers’ pausing behavior in their native and non-native speech. Pausing measures were applied in order to study between-speaker and within-speaker variability, where within-speaker variability was introduced by recording speakers in their native Zurich German, and in their second languages English and French. Results showed that pausing measures in the form of pause numbers and pause durations are speaker-specific. Furthermore, this speaker-specificity became evident across different languages. Results are discussed in the context of forensic voice comparison. Keywords:},\n\tbooktitle = {18th {International} {Congress} of {Phonetic} {Sciences} {ICPHS} 2015},\n\tauthor = {Kolly, Marie-José and Leemann, Adrian and Boula de Mareüil, Philippe and Dellwo, Volker},\n\tyear = {2015},\n}\n\n\n\n
\n
\n\n\n
\n Phoneticians study acoustic speech signals. But what about the aspects of speech where the signal is silent? The present study investigated speakers’ pausing behavior in their native and non-native speech. Pausing measures were applied in order to study between-speaker and within-speaker variability, where within-speaker variability was introduced by recording speakers in their native Zurich German, and in their second languages English and French. Results showed that pausing measures in the form of pause numbers and pause durations are speaker-specific. Furthermore, this speaker-specificity became evident across different languages. Results are discussed in the context of forensic voice comparison. Keywords:\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Voice Äpp: a mobile app for crowdsourcing Swiss German dialect data.\n \n \n \n \n\n\n \n Leemann, A.; Kolly, M.; Goldman, J.; Dellwo, V.; Hove, I.; Almajai, I.; Grimm, S.; Robert, S.; and Wanitsch, D.\n\n\n \n\n\n\n In Interspeech 2015, pages 2804–2808, September 2015. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"VoicePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{leemannVoiceAppMobile2015,\n\ttitle = {Voice Äpp: a mobile app for crowdsourcing {Swiss} {German} dialect data},\n\tshorttitle = {Voice Äpp},\n\turl = {https://www.isca-archive.org/interspeech_2015/leemann15b_interspeech.html},\n\tdoi = {10.21437/Interspeech.2015-590},\n\tabstract = {Crowdsourcing speech data through mobile applications is relatively new. In the present contribution we add to the existing body of research an innovative Android and iOS application called ‘Voice Äpp’. The free app is pioneering in the sense that it leverages its function as a medium for science communication – thus attracting an extensive user base – to crowdsource audio and dialect data. The app was launched in early 2015 and has already been downloaded 19k times. Nearly half a million audio tokens have been crowdsourced. In this system levels contribution we describe the basic functionalities of the app – voice and dialect analysis –, we present the scientific potential of the corpus created, and discuss methodological issues related to crowdsourcing audio data through mobile applications.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2015},\n\tpublisher = {ISCA},\n\tauthor = {Leemann, Adrian and Kolly, Marie-José and Goldman, Jean-Philippe and Dellwo, Volker and Hove, Ingrid and Almajai, Ibrahim and Grimm, Sarah and Robert, Sylvain and Wanitsch, Daniel},\n\tmonth = sep,\n\tyear = {2015},\n\tpages = {2804--2808},\n}\n\n\n\n
\n
\n\n\n
\n Crowdsourcing speech data through mobile applications is relatively new. In the present contribution we add to the existing body of research an innovative Android and iOS application called ‘Voice Äpp’. The free app is pioneering in the sense that it leverages its function as a medium for science communication – thus attracting an extensive user base – to crowdsource audio and dialect data. The app was launched in early 2015 and has already been downloaded 19k times. Nearly half a million audio tokens have been crowdsourced. In this system levels contribution we describe the basic functionalities of the app – voice and dialect analysis –, we present the scientific potential of the corpus created, and discuss methodological issues related to crowdsourcing audio data through mobile applications.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Age-Related Neural Oscillation Patterns During the Processing of Temporally Manipulated Speech.\n \n \n \n \n\n\n \n Rufener, K. S.; Oechslin, M. S.; Wöstmann, M.; Dellwo, V.; and Meyer, M.\n\n\n \n\n\n\n Brain Topography, 29(3): 1–19. 2015.\n \n\n\n\n
\n\n\n\n \n \n \"Age-RelatedPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{rufenerAgeRelatedNeuralOscillation2015,\n\ttitle = {Age-{Related} {Neural} {Oscillation} {Patterns} {During} the {Processing} of {Temporally} {Manipulated} {Speech}},\n\tvolume = {29},\n\tissn = {15736792},\n\turl = {http://www.zora.uzh.ch/id/eprint/115560/},\n\tdoi = {10.1007/s10548-015-0464-0},\n\tabstract = {This EEG-study aims to investigate age-related differences in the neural oscillation patterns during the processing of temporally modulated speech. Viewing from a lifespan perspective, we recorded the electroencephalogram (EEG) data of three age samples: young adults, middle-aged adults and older adults. Stimuli consisted of temporally degraded sentences in Swedish-a language unfamiliar to all participants. We found age-related differences in phonetic pattern matching when participants were presented with envelope-degraded sentences, whereas no such age-effect was observed in the processing of fine-structure-degraded sentences. Irrespective of age, during speech processing the EEG data revealed a relationship between envelope information and the theta band (4-8 Hz) activity. Additionally, an association between fine-structure information and the gamma band (30-48 Hz) activity was found. No interaction, however, was found between acoustic manipulation of stimuli and age. Importantly, our main finding was paralleled by an overall enhanced power in older adults in high frequencies (gamma: 30-48 Hz). This occurred irrespective of condition. For the most part, this result is in line with the Asymmetric Sampling in Time framework (Poeppel in Speech Commun 41:245-255, 2003), which assumes an isomorphic correspondence between frequency modulations in neurophysiological patterns and acoustic oscillations in spoken language. We conclude that speech-specific neural networks show strong stability over adulthood, despite initial processes of cortical degeneration indicated by enhanced gamma power. The results of our study therefore confirm the concept that sensory and cognitive processes undergo multidirectional trajectories within the context of healthy aging.},\n\tnumber = {3},\n\tjournal = {Brain Topography},\n\tpublisher = {Springer},\n\tauthor = {Rufener, Katharina Simone and Oechslin, Mathias S. and Wöstmann, Malte and Dellwo, Volker and Meyer, Martin},\n\tyear = {2015},\n\tpages = {1--19},\n}\n\n\n\n
\n
\n\n\n
\n This EEG-study aims to investigate age-related differences in the neural oscillation patterns during the processing of temporally modulated speech. Viewing from a lifespan perspective, we recorded the electroencephalogram (EEG) data of three age samples: young adults, middle-aged adults and older adults. Stimuli consisted of temporally degraded sentences in Swedish-a language unfamiliar to all participants. We found age-related differences in phonetic pattern matching when participants were presented with envelope-degraded sentences, whereas no such age-effect was observed in the processing of fine-structure-degraded sentences. Irrespective of age, during speech processing the EEG data revealed a relationship between envelope information and the theta band (4-8 Hz) activity. Additionally, an association between fine-structure information and the gamma band (30-48 Hz) activity was found. No interaction, however, was found between acoustic manipulation of stimuli and age. Importantly, our main finding was paralleled by an overall enhanced power in older adults in high frequencies (gamma: 30-48 Hz). This occurred irrespective of condition. For the most part, this result is in line with the Asymmetric Sampling in Time framework (Poeppel in Speech Commun 41:245-255, 2003), which assumes an isomorphic correspondence between frequency modulations in neurophysiological patterns and acoustic oscillations in spoken language. We conclude that speech-specific neural networks show strong stability over adulthood, despite initial processes of cortical degeneration indicated by enhanced gamma power. The results of our study therefore confirm the concept that sensory and cognitive processes undergo multidirectional trajectories within the context of healthy aging.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The phonological function of vowels is maintained at fundamental frequencies up to 880 Hz.\n \n \n \n \n\n\n \n Friedrichs, D.; Maurer, D.; and Dellwo, V.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 138(1): EL36–EL42. 2015.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{friedrichsPhonologicalFunctionVowels2015,\n\ttitle = {The phonological function of vowels is maintained at fundamental frequencies up to 880 {Hz}},\n\tvolume = {138},\n\tissn = {0001-4966},\n\turl = {https://doi.org/10.5167/uzh-111558},\n\tdoi = {10.1121/1.4922534},\n\tabstract = {In a between-subject perception task, listeners either identified full words or vowels isolated from these words at F0s between 220 and 880Hz. They received two written words as response options (minimal pair with the stimulus vowel in contrastive position). Listeners? sensitivity (A0) was extremely high in both conditions at all F0s, showing that the phonological function of vowels can also be maintained at high F0s. This indicates that vowel sounds may carry strong acoustic cues departing from common formant frequencies at high F0s and that listeners do not rely on consonantal context phenomena for their identification performance.},\n\tnumber = {1},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {Friedrichs, Daniel and Maurer, Dieter and Dellwo, Volker},\n\tyear = {2015},\n\tpages = {EL36--EL42},\n}\n\n\n\n
\n
\n\n\n
\n In a between-subject perception task, listeners either identified full words or vowels isolated from these words at F0s between 220 and 880Hz. They received two written words as response options (minimal pair with the stimulus vowel in contrastive position). Listeners? sensitivity (A0) was extremely high in both conditions at all F0s, showing that the phonological function of vowels can also be maintained at high F0s. This indicates that vowel sounds may carry strong acoustic cues departing from common formant frequencies at high F0s and that listeners do not rely on consonantal context phenomena for their identification performance.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n The recognition of read and spontaneous speech in local vernacular: The case of Zurich German.\n \n \n \n\n\n \n Dellwo, V.; Leemann, A.; and Kolly, M. J.\n\n\n \n\n\n\n Journal of Phonetics, 48: 13–28. 2015.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Dellwo2015b,\n\ttitle = {The recognition of read and spontaneous speech in local vernacular: {The} case of {Zurich} {German}},\n\tvolume = {48},\n\tissn = {00954470},\n\tdoi = {10.1016/j.wocn.2014.10.011},\n\tabstract = {Listeners are typically able to identify speech as either being produced spontaneously or read from a transcript. In the present research we investigated whether this is true in vernacular speech when typical cues to read and spontaneous speech are either missing and/or ambiguous. In addition it was investigated what the acoustic cues for listeners' identification ability were. 26 listeners of Zurich German participated in two perception experiments. In Experiment I, listeners judged 128 stimuli (64 spontaneous, 64 read) in a two-alternative identification task as either spontaneous or read. Results revealed that overall listener performance was well above chance (mean A'=0.82) while there was no bias for either read or spontaneous speech (mean BD′=0). There were significant effects of speaker and listener in A' and BD′. In Experiment II, the same 26 listeners rated the same 64 read speech stimuli from Experiment I as to whether they sounded more or less read. Results revealed that there was considerable within-category variability as a function of speaker. From eight acoustic prosodic parameters only articulation rate explained listener behavior to some degree in Experiments I and II. Overall the study suggested that read and spontaneous speech can be recognized based on very subtle cues to these speaking styles.},\n\tjournal = {Journal of Phonetics},\n\tauthor = {Dellwo, Volker and Leemann, Adrian and Kolly, Marie José},\n\tyear = {2015},\n\tpages = {13--28},\n}\n\n\n\n
\n
\n\n\n
\n Listeners are typically able to identify speech as either being produced spontaneously or read from a transcript. In the present research we investigated whether this is true in vernacular speech when typical cues to read and spontaneous speech are either missing and/or ambiguous. In addition it was investigated what the acoustic cues for listeners' identification ability were. 26 listeners of Zurich German participated in two perception experiments. In Experiment I, listeners judged 128 stimuli (64 spontaneous, 64 read) in a two-alternative identification task as either spontaneous or read. Results revealed that overall listener performance was well above chance (mean A'=0.82) while there was no bias for either read or spontaneous speech (mean BD′=0). There were significant effects of speaker and listener in A' and BD′. In Experiment II, the same 26 listeners rated the same 64 read speech stimuli from Experiment I as to whether they sounded more or less read. Results revealed that there was considerable within-category variability as a function of speaker. From eight acoustic prosodic parameters only articulation rate explained listener behavior to some degree in Experiments I and II. Overall the study suggested that read and spontaneous speech can be recognized based on very subtle cues to these speaking styles.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Rhythmic variability between speakers: Articulatory, prosodic, and linguistic factors.\n \n \n \n \n\n\n \n Dellwo, V.; Leemann, A.; and Kolly, M.\n\n\n \n\n\n\n The Journal of the Acoustical Society of America, 137(3): 1513–1528. 2015.\n \n\n\n\n
\n\n\n\n \n \n \"RhythmicPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoRhythmicVariabilitySpeakers2015,\n\ttitle = {Rhythmic variability between speakers: {Articulatory}, prosodic, and linguistic factors},\n\tvolume = {137},\n\tissn = {0001-4966},\n\turl = {http://dx.doi.org/10.1121/1.4906837},\n\tdoi = {10.1121/1.4906837},\n\tabstract = {Between-speaker variability of acoustically measurable speech rhythm [\\%V, ΔV(ln), ΔC(ln), and Δpeak(ln)] was investigated when within-speaker variability of (a) articulation rate and (b) linguistic structural characteristics was introduced. To study (a), 12 speakers of Standard German read seven lexically identical sentences under five different intended tempo conditions (very slow, slow, normal, fast, very fast). To study (b), 16 speakers of Zurich Swiss German produced 16 spontaneous utterances each (256 in total) for which transcripts were made and then read by all speakers (4096 sentences; 16 speaker × 256 sentences). Between-speaker variability was tested using analysis of variance with repeated measures on within-speaker factors. Results revealed strong and consistent between-speaker variability while within-speaker variability as a function of articulation rate and linguistic characteristics was typically not significant. It was concluded that between-speaker variability of acoustically measurable speech rhythm is strong and robust against various sources of within-speaker variability. Idiosyncratic articulatory movements were found to be the most plausible factor explaining between-speaker differences.},\n\tnumber = {3},\n\tjournal = {The Journal of the Acoustical Society of America},\n\tpublisher = {Acoustical Society of America},\n\tauthor = {Dellwo, Volker and Leemann, Adrian and Kolly, Marie-José},\n\tyear = {2015},\n\tpages = {1513--1528},\n}\n\n\n\n
\n
\n\n\n
\n Between-speaker variability of acoustically measurable speech rhythm [%V, ΔV(ln), ΔC(ln), and Δpeak(ln)] was investigated when within-speaker variability of (a) articulation rate and (b) linguistic structural characteristics was introduced. To study (a), 12 speakers of Standard German read seven lexically identical sentences under five different intended tempo conditions (very slow, slow, normal, fast, very fast). To study (b), 16 speakers of Zurich Swiss German produced 16 spontaneous utterances each (256 in total) for which transcripts were made and then read by all speakers (4096 sentences; 16 speaker × 256 sentences). Between-speaker variability was tested using analysis of variance with repeated measures on within-speaker factors. Results revealed strong and consistent between-speaker variability while within-speaker variability as a function of articulation rate and linguistic characteristics was typically not significant. It was concluded that between-speaker variability of acoustically measurable speech rhythm is strong and robust against various sources of within-speaker variability. Idiosyncratic articulatory movements were found to be the most plausible factor explaining between-speaker differences.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Swiss graphogame: concept and design presentation of a computerised reading intervention for children with high risk for poor reading outcomes.\n \n \n \n \n\n\n \n Röthlisberger, M.; Karipidis, I. I.; Pleisch, G.; Dellwo, V.; Richardson, U.; and Brem, S.\n\n\n \n\n\n\n . September 2015.\n \n\n\n\n
\n\n\n\n \n \n \"SwissPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{rothlisbergerSwissGraphogameConcept2015,\n\ttitle = {Swiss graphogame: concept and design presentation of a computerised reading intervention for children with high risk for poor reading outcomes},\n\tshorttitle = {Swiss graphogame},\n\turl = {https://www.zora.uzh.ch/id/eprint/120197},\n\tdoi = {10.5167/UZH-120197},\n\tabstract = {Developmental dyslexia is found in 30 to 65\\% of the children from high risk families (parent or sibling affected). The computerised GraphoGame training program aims to improve reading acquisition for children with a high risk for developing reading and spelling problems. GraphoGame® is a learning platform developed at the University of Jyväskylä in Finland to support poor reading children during reading acquisition. Here, we present the concept and design of the new Swiss GraphoGame, which at a linguistic level has been developed to especially suit the needs of children speaking an orthographically semi-transparent language (German).},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {Interspeech},\n\tauthor = {Röthlisberger, Martina and Karipidis, Iliana I. and Pleisch, Georgette and Dellwo, Volker and Richardson, Ulla and Brem, Silvia},\n\tmonth = sep,\n\tyear = {2015},\n}\n\n\n\n
\n
\n\n\n
\n Developmental dyslexia is found in 30 to 65% of the children from high risk families (parent or sibling affected). The computerised GraphoGame training program aims to improve reading acquisition for children with a high risk for developing reading and spelling problems. GraphoGame® is a learning platform developed at the University of Jyväskylä in Finland to support poor reading children during reading acquisition. Here, we present the concept and design of the new Swiss GraphoGame, which at a linguistic level has been developed to especially suit the needs of children speaking an orthographically semi-transparent language (German).\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker-individuality in Fujisaki model f0 features: implications for forensic voice comparison.\n \n \n \n \n\n\n \n Leeman, A.; Mixdorff, H.; O'Reilly, M.; Kolly, M.; and Dellwo, V.\n\n\n \n\n\n\n The International Journal of Speech, Language and the Law, 21(2): 343–370. February 2015.\n \n\n\n\n
\n\n\n\n \n \n \"Speaker-individualityPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{leemanSpeakerindividualityFujisakiModel2015,\n\ttitle = {Speaker-individuality in {Fujisaki} model f0 features: implications for forensic voice comparison},\n\tvolume = {21},\n\tissn = {1748-8885, 1748-8893},\n\tshorttitle = {Speaker-individuality in {Fujisaki} model f0 features},\n\turl = {https://utppublishing.com/doi/10.1558/ijsll.v21i2.343},\n\tdoi = {10.1558/ijsll.v21i2.343},\n\tabstract = {Fundamental frequency (f0) is a highly speaker-specific feature. Consequently, practitioners often use f0 information in forensic casework. Current research principally examines the use of long-term f0 statistics such as f0 means and standard deviations for forensic voice comparison. The present study investigates how short-term f0 features such as measured by the Fujisaki intonation model capture speaker-individuality. Based on data of a homogeneous group of Zurich German speakers, we conducted an experiment on a large corpus of read speech and on a subset of sentences that included speaking style variability (spontaneous vs read). The latter is characteristic of forensic casework. Speakers demonstrated high between-speaker variability and low within-speaker variability across the two speaking styles for a number of f0 features. Given this evidence of speakerindividuality, we discuss Fujisaki f0 features’ potential for forensic voice comparison.},\n\tlanguage = {en},\n\tnumber = {2},\n\turldate = {2025-04-18},\n\tjournal = {The International Journal of Speech, Language and the Law},\n\tauthor = {Leeman, Adrian and Mixdorff, Hansjörg and O'Reilly, Maria and Kolly, Marie-José and Dellwo, Volker},\n\tmonth = feb,\n\tyear = {2015},\n\tpages = {343--370},\n}\n\n\n\n
\n
\n\n\n
\n Fundamental frequency (f0) is a highly speaker-specific feature. Consequently, practitioners often use f0 information in forensic casework. Current research principally examines the use of long-term f0 statistics such as f0 means and standard deviations for forensic voice comparison. The present study investigates how short-term f0 features such as measured by the Fujisaki intonation model capture speaker-individuality. Based on data of a homogeneous group of Zurich German speakers, we conducted an experiment on a large corpus of read speech and on a subset of sentences that included speaking style variability (spontaneous vs read). The latter is characteristic of forensic casework. Speakers demonstrated high between-speaker variability and low within-speaker variability across the two speaking styles for a number of f0 features. Given this evidence of speakerindividuality, we discuss Fujisaki f0 features’ potential for forensic voice comparison.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Comparisons of speaker recognition strengths using suprasegmental duration and intensity variability: an artificial neural networks approach.\n \n \n \n \n\n\n \n He, L.; Glavitsch, U.; and Dellwo, V.\n\n\n \n\n\n\n . August 2015.\n \n\n\n\n
\n\n\n\n \n \n \"ComparisonsPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{heComparisonsSpeakerRecognition2015,\n\ttitle = {Comparisons of speaker recognition strengths using suprasegmental duration and intensity variability: an artificial neural networks approach},\n\tshorttitle = {Comparisons of speaker recognition strengths using suprasegmental duration and intensity variability},\n\turl = {https://www.zora.uzh.ch/id/eprint/127262},\n\tdoi = {10.5167/UZH-127262},\n\tabstract = {This study compares the speaker recognition strengths based on suprasegmental duration and intensity variability in the speech signal using artificial neural networks. Such algorithm can well capture the nonlinear effects in the data, and is more robust against noise in the data. Three rounds of classification tasks were performed with 1) duration metrics, 2) intensity metrics, and 3) the combination of duration and intensity metrics as the independent variables. The results indicated that both intensity and combined metrics significantly outperformed the duration metrics. Moreover, the combination of intensity and duration metrics showed higher probability of improved speaker classifications than intensity metrics over duration metrics.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Phonetic Association},\n\tauthor = {He, Lei and Glavitsch, Ulrike and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2015},\n}\n\n\n\n
\n
\n\n\n
\n This study compares the speaker recognition strengths based on suprasegmental duration and intensity variability in the speech signal using artificial neural networks. Such algorithm can well capture the nonlinear effects in the data, and is more robust against noise in the data. Three rounds of classification tasks were performed with 1) duration metrics, 2) intensity metrics, and 3) the combination of duration and intensity metrics as the independent variables. The results indicated that both intensity and combined metrics significantly outperformed the duration metrics. Moreover, the combination of intensity and duration metrics showed higher probability of improved speaker classifications than intensity metrics over duration metrics.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Vowel identification at high fundamental frequencies in minimal pairs.\n \n \n \n \n\n\n \n Friedrichs, D.; Maurer, D.; Suter, H.; and Dellwo, V.\n\n\n \n\n\n\n . August 2015.\n \n\n\n\n
\n\n\n\n \n \n \"VowelPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{friedrichsVowelIdentificationHigh2015,\n\ttitle = {Vowel identification at high fundamental frequencies in minimal pairs},\n\turl = {https://www.zora.uzh.ch/id/eprint/112398},\n\tdoi = {10.5167/UZH-112398},\n\tabstract = {The question of vowel intelligibility as a function of F0 is still a matter of debate. Above all concerning vowel sounds produced at F0s exceeding vowelrelated statistical F1 in citation-form words (‘oversinging’ F1), it is unclear whether vowel category perception inevitably shifts towards the neighboring category with a higher F1 or can be maintained in such cases. In this study, we tested listeners’ perception of the long German vowels /i-y-e-ø-ε-a-o/ produced by a trained female speaker in the context of minimal pair words (/l-V-gen/) at nine F0-levels between 220 and 880 Hz. Results showed that vowel identification was maintained {\\textgreater} 80\\% up to F0=740 Hz for /e-ø-ε/ and up to F0=880 Hz for /i-y-a-o/. Thus, vowel identification could be maintained in cases of F0 significantly exceeding F1. The role of neighboring vowels, vowel duration, and other productional and acoustical aspects relevant for vowel perception at different F0s is discussed.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Friedrichs, Daniel and Maurer, Dieter and Suter, Heidy and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2015},\n}\n\n\n\n
\n
\n\n\n
\n The question of vowel intelligibility as a function of F0 is still a matter of debate. Above all concerning vowel sounds produced at F0s exceeding vowelrelated statistical F1 in citation-form words (‘oversinging’ F1), it is unclear whether vowel category perception inevitably shifts towards the neighboring category with a higher F1 or can be maintained in such cases. In this study, we tested listeners’ perception of the long German vowels /i-y-e-ø-ε-a-o/ produced by a trained female speaker in the context of minimal pair words (/l-V-gen/) at nine F0-levels between 220 and 880 Hz. Results showed that vowel identification was maintained \\textgreater 80% up to F0=740 Hz for /e-ø-ε/ and up to F0=880 Hz for /i-y-a-o/. Thus, vowel identification could be maintained in cases of F0 significantly exceeding F1. The role of neighboring vowels, vowel duration, and other productional and acoustical aspects relevant for vowel perception at different F0s is discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Use of speech and prosody in Composed Theatre.\n \n \n \n \n\n\n \n Dimos, K.; Dick, L.; and Dellwo, V.\n\n\n \n\n\n\n . 2015.\n \n\n\n\n
\n\n\n\n \n \n \"UsePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dimosUseSpeechProsody2015,\n\ttitle = {Use of speech and prosody in {Composed} {Theatre}},\n\turl = {https://www.zora.uzh.ch/id/eprint/117113},\n\tdoi = {10.5167/UZH-117113},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {Peter Lang},\n\tauthor = {Dimos, Kostis and Dick, Leopold and Dellwo, Volker},\n\tyear = {2015},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Auditory speaker discrimination by forensic phoneticians and naive listeners in voiced and whispered speech.\n \n \n \n \n\n\n \n Bartle, A.; and Dellwo, V.\n\n\n \n\n\n\n International Journal of Speech, Language and the Law, 22(2): 229–248. 2015.\n \n\n\n\n
\n\n\n\n \n \n \"AuditoryPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Bartle2015,\n\ttitle = {Auditory speaker discrimination by forensic phoneticians and naive listeners in voiced and whispered speech},\n\tvolume = {22},\n\tissn = {17488893},\n\turl = {http://www.zora.uzh.ch/id/eprint/114532/},\n\tdoi = {10.1558/ijsll.v22i2.23101},\n\tabstract = {In whispered speech some important cues to a speaker’s identity (e.g. fundamental frequency, intonation) are inevitably absent. In the present study we investigated listeners’ ability to discriminate between speakers in short utterances in voiced and whispered speech. The performances of a group of 11 forensic phoneticians and a group of 22 naive listeners were compared in a binary forced-choice speaker discrimination task, with 48 same-speaker and 60 different-speaker pairs of short speech samples (≤ 3 s) in each test. Listeners were asked to say whether the two voice samples in each pair were produced by the same or different speakers, and to give a certainty rating. The results showed that speaker discrimination is more difficult in whispered than in voiced speech, and that while the phoneticians were only slightly better than the naive listeners in voiced speech, the gap widened in whispered speech. Phoneticians were more cautious in their responses, but also more accurate than naive listeners. When unsure, the phoneticians tended to say two utterances came from different speakers, whereas naive listeners tended to say two utterances came from the same speaker. Results support the view that trained phoneticians may have an advantage over naive listeners in auditory speaker discrimination when the signal is degraded.},\n\tnumber = {2},\n\tjournal = {International Journal of Speech, Language and the Law},\n\tpublisher = {Equinox Publishing Ltd.},\n\tauthor = {Bartle, Anna and Dellwo, Volker},\n\tyear = {2015},\n\tpages = {229--248},\n}\n
\n
\n\n\n
\n In whispered speech some important cues to a speaker’s identity (e.g. fundamental frequency, intonation) are inevitably absent. In the present study we investigated listeners’ ability to discriminate between speakers in short utterances in voiced and whispered speech. The performances of a group of 11 forensic phoneticians and a group of 22 naive listeners were compared in a binary forced-choice speaker discrimination task, with 48 same-speaker and 60 different-speaker pairs of short speech samples (≤ 3 s) in each test. Listeners were asked to say whether the two voice samples in each pair were produced by the same or different speakers, and to give a certainty rating. The results showed that speaker discrimination is more difficult in whispered than in voiced speech, and that while the phoneticians were only slightly better than the naive listeners in voiced speech, the gap widened in whispered speech. Phoneticians were more cautious in their responses, but also more accurate than naive listeners. When unsure, the phoneticians tended to say two utterances came from different speakers, whereas naive listeners tended to say two utterances came from the same speaker. Results support the view that trained phoneticians may have an advantage over naive listeners in auditory speaker discrimination when the signal is degraded.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2014\n \n \n (15)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Rhythmic variability between some asian languages: results from an automatic analysis of temporal characteristics.\n \n \n \n \n\n\n \n Dellwo, V.; Mok, P.; and Jenny, M.\n\n\n \n\n\n\n In Interspeech 2014, pages 1708–1711, September 2014. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"RhythmicPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwoRhythmicVariabilityAsian2014,\n\ttitle = {Rhythmic variability between some asian languages: results from an automatic analysis of temporal characteristics},\n\tshorttitle = {Rhythmic variability between some asian languages},\n\turl = {https://www.isca-archive.org/interspeech_2014/dellwo14_interspeech.html},\n\tdoi = {10.21437/Interspeech.2014-33},\n\tabstract = {The rhythmic organization of speech can vary between languages. In the present research we studied rhythmic variability between Mandarin, Cantonese and Thai using automatically retrieved prosodic temporal characteristics from read speech. We measured the variability of intervals between amplitude peaks in the amplitude envelope ({\\textless}10 Hz) and the durational characteristics of intervals with and without glottal activity (voiced and unvoiced intervals) in speech. Results for between language comparisons revealed significant differences between languages in both amplitude peak interval variability and voiced-voiceless interval durational characteristics. Results are discussed in connection with language specific phonotactic/phonological properties and hypotheses about the perceptual significance of the acoustic measurements in terms of speech rhythm.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2014},\n\tpublisher = {ISCA},\n\tauthor = {Dellwo, Volker and Mok, Peggy and Jenny, Mathias},\n\tmonth = sep,\n\tyear = {2014},\n\tpages = {1708--1711},\n}\n\n\n\n
\n
\n\n\n
\n The rhythmic organization of speech can vary between languages. In the present research we studied rhythmic variability between Mandarin, Cantonese and Thai using automatically retrieved prosodic temporal characteristics from read speech. We measured the variability of intervals between amplitude peaks in the amplitude envelope (\\textless10 Hz) and the durational characteristics of intervals with and without glottal activity (voiced and unvoiced intervals) in speech. Results for between language comparisons revealed significant differences between languages in both amplitude peak interval variability and voiced-voiceless interval durational characteristics. Results are discussed in connection with language specific phonotactic/phonological properties and hypotheses about the perceptual significance of the acoustic measurements in terms of speech rhythm.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker idiosyncratic variability of intensity across syllables.\n \n \n \n \n\n\n \n He, L.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2014, pages 233–237, September 2014. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"SpeakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{heSpeakerIdiosyncraticVariability2014,\n\ttitle = {Speaker idiosyncratic variability of intensity across syllables},\n\turl = {https://www.isca-archive.org/interspeech_2014/he14_interspeech.html},\n\tdoi = {10.21437/Interspeech.2014-59},\n\tabstract = {This study explored speaker idiosyncrasy by measuring the syllabic intensity variability in the speech signal. Sixteen speakers of the TEVOID corpus, each producing 256 read sentences, were analyzed. Characteristics of intensity variability (average or peak) between syllables were measured either holistically (standard deviation of intensity changes between syllables) or locally (pairwise variability indices of intensity changes between syllables). The results indicated significant effects of the speakers in all the metrics, suggesting a potential application of the methods for speaker recognition, and in particular for forensic speaker comparison.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2014},\n\tpublisher = {ISCA},\n\tauthor = {He, Lei and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2014},\n\tpages = {233--237},\n}\n\n\n\n
\n
\n\n\n
\n This study explored speaker idiosyncrasy by measuring the syllabic intensity variability in the speech signal. Sixteen speakers of the TEVOID corpus, each producing 256 read sentences, were analyzed. Characteristics of intensity variability (average or peak) between syllables were measured either holistically (standard deviation of intensity changes between syllables) or locally (pairwise variability indices of intensity changes between syllables). The results indicated significant effects of the speakers in all the metrics, suggesting a potential application of the methods for speaker recognition, and in particular for forensic speaker comparison.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Foreign accent recognition based on temporal information contained in lowpass-filtered speech.\n \n \n \n \n\n\n \n Kolly, M.; Leemann, A.; and Dellwo, V.\n\n\n \n\n\n\n In Interspeech 2014, pages 2175–2179, September 2014. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ForeignPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{kollyForeignAccentRecognition2014,\n\ttitle = {Foreign accent recognition based on temporal information contained in lowpass-filtered speech},\n\turl = {https://www.isca-archive.org/interspeech_2014/kolly14_interspeech.html},\n\tdoi = {10.21437/Interspeech.2014-487},\n\tabstract = {Can the foreign accent of a speaker be recognized based on suprasegmental temporal information? For a perception experiment we created stimuli based on German sentences read by six French and six English speakers. These foreignaccented sentences were manipulated by (1) applying a lowpass filter with a cutoff frequency of 300 Hz and (2) applying the same lowpass filter and monotonizing F0. In a between-subject 2AFC perception experiment we tested the accent recognition ability of 15 Swiss German listeners per signal manipulation condition. The results showed that speakers’ native language could be recognized above chance in both conditions. However, listeners obtained significantly lower recognition scores in the monotonized condition. Furthermore, higher recognition scores were obtained for French-accented speech in the monotonized condition, a result that is discussed in light of research on speech rhythm. We further report an effect for speaker within each accent group. The results suggest that suprasegmental temporal information allows for foreign accent recognition to some degree.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2014},\n\tpublisher = {ISCA},\n\tauthor = {Kolly, Marie-José and Leemann, Adrian and Dellwo, Volker},\n\tmonth = sep,\n\tyear = {2014},\n\tpages = {2175--2179},\n}\n\n\n\n
\n
\n\n\n
\n Can the foreign accent of a speaker be recognized based on suprasegmental temporal information? For a perception experiment we created stimuli based on German sentences read by six French and six English speakers. These foreignaccented sentences were manipulated by (1) applying a lowpass filter with a cutoff frequency of 300 Hz and (2) applying the same lowpass filter and monotonizing F0. In a between-subject 2AFC perception experiment we tested the accent recognition ability of 15 Swiss German listeners per signal manipulation condition. The results showed that speakers’ native language could be recognized above chance in both conditions. However, listeners obtained significantly lower recognition scores in the monotonized condition. Furthermore, higher recognition scores were obtained for French-accented speech in the monotonized condition, a result that is discussed in light of research on speech rhythm. We further report an effect for speaker within each accent group. The results suggest that suprasegmental temporal information allows for foreign accent recognition to some degree.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Listeners may rely on intonation to distinguish languages of different rhythm classes.\n \n \n \n \n\n\n \n Hagmann, L.; and Dellwo, V.\n\n\n \n\n\n\n Loquens, 1(1): e008. 2014.\n \n\n\n\n
\n\n\n\n \n \n \"ListenersPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{hagmannListenersMayRely2014,\n\ttitle = {Listeners may rely on intonation to distinguish languages of different rhythm classes},\n\tvolume = {1},\n\tissn = {2386-2637},\n\turl = {https://doi.org/10.5167/uzh-103020},\n\tdoi = {10.3989/loquens.2014.008},\n\tabstract = {Previous research argued that listeners can distinguish between languages of different rhythm class but not of the same class (class discrimination hypothesis). In the present research we tested the role of duration and pitch cues (intonation) in this process. In Experiment I we tested whether we could replicate previous findings on listeners’ language discrimination ability with native Swiss German listeners. Results showed that the discrimination of English and Japanese based on durational cues led to the same results as in previous experiments. In Experiment II we tested listeners’ ability to distinguish between languages belonging to different rhythm classes (English-French, French-Japanese, Spanish-Japanese) and the same rhythm class (Spanish-French). Results revealed that listeners’ distinction was not above chance level for all language contrasts. In Experiment III we added intonation to a French- English and a Spanish-French language contrast. Results revealed a significant effect of intonation for the French- English but not the Spanish-French contrast. The experiments showed that the primary cue for listeners to distinguish between languages of different rhythm class is not generally duration, as previously hypothesized, but it can also be intonation. Implications of the findings on the theory that languages can be classified according to their speech rhythm (rhythm class hypothesis) are discussed.},\n\tnumber = {1},\n\tjournal = {Loquens},\n\tpublisher = {Consejo Superior de Investigaciones Científicas (CSIC)},\n\tauthor = {Hagmann, Lea and Dellwo, Volker},\n\tyear = {2014},\n\tpages = {e008},\n}\n\n\n\n
\n
\n\n\n
\n Previous research argued that listeners can distinguish between languages of different rhythm class but not of the same class (class discrimination hypothesis). In the present research we tested the role of duration and pitch cues (intonation) in this process. In Experiment I we tested whether we could replicate previous findings on listeners’ language discrimination ability with native Swiss German listeners. Results showed that the discrimination of English and Japanese based on durational cues led to the same results as in previous experiments. In Experiment II we tested listeners’ ability to distinguish between languages belonging to different rhythm classes (English-French, French-Japanese, Spanish-Japanese) and the same rhythm class (Spanish-French). Results revealed that listeners’ distinction was not above chance level for all language contrasts. In Experiment III we added intonation to a French- English and a Spanish-French language contrast. Results revealed a significant effect of intonation for the French- English but not the Spanish-French contrast. The experiments showed that the primary cue for listeners to distinguish between languages of different rhythm class is not generally duration, as previously hypothesized, but it can also be intonation. Implications of the findings on the theory that languages can be classified according to their speech rhythm (rhythm class hypothesis) are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Inter-speaker variability in intensity dynamics.\n \n \n \n\n\n \n He, L.; Glavitsch, U.; and Dellwo, V.\n\n\n \n\n\n\n In International Association of Forensic Phonetics and Acoustics, pages 1–2, Leiden/Netherlands, 2014. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{He2014a,\n\taddress = {Leiden/Netherlands},\n\ttitle = {Inter-speaker variability in intensity dynamics},\n\tbooktitle = {International {Association} of {Forensic} {Phonetics} and {Acoustics}},\n\tauthor = {He, Lei and Glavitsch, Ulrike and Dellwo, Volker},\n\tyear = {2014},\n\tpages = {1--2},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Integration of Spoken and Written Words in Beginning Readers: A Topographic ERP Study.\n \n \n \n \n\n\n \n Jost, L. B.; Eberhard-Moscicka, A. K.; Frisch, C.; Dellwo, V.; and Maurer, U.\n\n\n \n\n\n\n Brain Topography, 27(6): 786–800. 2014.\n \n\n\n\n
\n\n\n\n \n \n \"IntegrationPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Jost2014,\n\ttitle = {Integration of {Spoken} and {Written} {Words} in {Beginning} {Readers}: {A} {Topographic} {ERP} {Study}},\n\tvolume = {27},\n\tissn = {15736792},\n\turl = {http://www.scopus.com/inward/record.url?eid=2-s2.0-84912005268&partnerID=MN8TOARS},\n\tdoi = {10.1007/s10548-013-0336-4},\n\tabstract = {Integrating visual and auditory language information is critical for reading. Suppression and congruency effects in audiovisual paradigms with letters and speech sounds have provided information about low-level mechanisms of grapheme-phoneme integration during reading. However, the central question about how such processes relate to reading entire words remains unexplored. Using ERPs, we investigated whether audiovisual integration occurs for words already in beginning readers, and if so, whether this integration is reflected by differences in map strength or topography (aim 1); and moreover, whether such integration is associated with reading fluency (aim 2). A 128-channel EEG was recorded while 69 monolingual (Swiss)-German speaking first-graders performed a detection task with rare targets. Stimuli were presented in blocks either auditorily (A), visually (V) or audiovisually (matching: AVM; nonmatching: AVN). Corresponding ERPs were computed, and unimodal ERPs summated (A + V = sumAV). We applied TANOVAs to identify time windows with significant integration effects: suppression (sumAV–AVM) and congruency (AVN–AVM). They were further characterized using GFP and 3D-centroid analyses, and significant effects were correlated with reading fluency. The results suggest that audiovisual suppression effects occur for familiar German and unfamiliar English words, whereas audiovisual congruency effects can be found only for familiar German words, probably due to lexical-semantic processes involved. Moreover, congruency effects were characterized by topographic differences, indicating that different sources are active during processing of congruent compared to incongruent audiovisual words. Furthermore, no clear associations between audiovisual integration and reading fluency were found. The degree to which such associations develop in beginning readers remains open to further investigation.},\n\tnumber = {6},\n\tjournal = {Brain Topography},\n\tauthor = {Jost, Lea B. and Eberhard-Moscicka, Aleksandra K. and Frisch, Christine and Dellwo, Volker and Maurer, Urs},\n\tyear = {2014},\n\tpages = {786--800},\n}\n\n\n\n
\n
\n\n\n
\n Integrating visual and auditory language information is critical for reading. Suppression and congruency effects in audiovisual paradigms with letters and speech sounds have provided information about low-level mechanisms of grapheme-phoneme integration during reading. However, the central question about how such processes relate to reading entire words remains unexplored. Using ERPs, we investigated whether audiovisual integration occurs for words already in beginning readers, and if so, whether this integration is reflected by differences in map strength or topography (aim 1); and moreover, whether such integration is associated with reading fluency (aim 2). A 128-channel EEG was recorded while 69 monolingual (Swiss)-German speaking first-graders performed a detection task with rare targets. Stimuli were presented in blocks either auditorily (A), visually (V) or audiovisually (matching: AVM; nonmatching: AVN). Corresponding ERPs were computed, and unimodal ERPs summated (A + V = sumAV). We applied TANOVAs to identify time windows with significant integration effects: suppression (sumAV–AVM) and congruency (AVN–AVM). They were further characterized using GFP and 3D-centroid analyses, and significant effects were correlated with reading fluency. The results suggest that audiovisual suppression effects occur for familiar German and unfamiliar English words, whereas audiovisual congruency effects can be found only for familiar German words, probably due to lexical-semantic processes involved. Moreover, congruency effects were characterized by topographic differences, indicating that different sources are active during processing of congruent compared to incongruent audiovisual words. Furthermore, no clear associations between audiovisual integration and reading fluency were found. The degree to which such associations develop in beginning readers remains open to further investigation.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Cues to linguistic origin: The contribution of speech temporal information to foreign accent recognition.\n \n \n \n\n\n \n Kolly, M. J. J.; and Dellwo, V.\n\n\n \n\n\n\n Journal of Phonetics, 42(1): 12–23. 2014.\n ISBN: 00954470\n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kollyCuesLinguisticOrigin2014,\n\ttitle = {Cues to linguistic origin: {The} contribution of speech temporal information to foreign accent recognition},\n\tvolume = {42},\n\tissn = {00954470},\n\tdoi = {10.1016/j.wocn.2013.11.004},\n\tabstract = {Foreign-accented speech typically contains information about speakers' linguistic origin, i.e., their native language. The present study explored the importance of different temporal and rhythmic prosodic characteristics for the recognition of French- and English-accented German. In perception experiments with Swiss German listeners, stimuli for accent recognition contained speech that was reduced artificially to convey temporal and rhythmic prosodic characteristics: (a) amplitude envelope durational information (by noise vocoding), (b) segment durations (by 1-bit requantisation) and (c) durations of voiced and voiceless intervals (by sasasa-delexicalisation). This preserved mainly time domain characteristics and different degrees of rudimentary information from the frequency domain. Results showed that listeners could recognise French- and English-accented German above chance even when their access to segmental and spectral cues was strongly reduced. Different types of temporal cues led to different recognition scores - segment durations were found to be the temporal cue most salient for accent recognition. Signal conditions that contained fewer segmental and spectral cues led to lower accent recognition scores. ?? 2013 Elsevier Ltd.},\n\tnumber = {1},\n\tjournal = {Journal of Phonetics},\n\tauthor = {Kolly, Marie José Jos?? and Dellwo, Volker},\n\tyear = {2014},\n\tnote = {ISBN: 00954470},\n\tpages = {12--23},\n}\n\n\n\n
\n
\n\n\n
\n Foreign-accented speech typically contains information about speakers' linguistic origin, i.e., their native language. The present study explored the importance of different temporal and rhythmic prosodic characteristics for the recognition of French- and English-accented German. In perception experiments with Swiss German listeners, stimuli for accent recognition contained speech that was reduced artificially to convey temporal and rhythmic prosodic characteristics: (a) amplitude envelope durational information (by noise vocoding), (b) segment durations (by 1-bit requantisation) and (c) durations of voiced and voiceless intervals (by sasasa-delexicalisation). This preserved mainly time domain characteristics and different degrees of rudimentary information from the frequency domain. Results showed that listeners could recognise French- and English-accented German above chance even when their access to segmental and spectral cues was strongly reduced. Different types of temporal cues led to different recognition scores - segment durations were found to be the temporal cue most salient for accent recognition. Signal conditions that contained fewer segmental and spectral cues led to lower accent recognition scores. ?? 2013 Elsevier Ltd.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Speaker-individuality in suprasegmental temporal features: Implications for forensic voice comparison.\n \n \n \n\n\n \n Leemann, A.; Kolly, M. J.; and Dellwo, V.\n\n\n \n\n\n\n Forensic Science International, 238: 59–67. 2014.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{leemannSpeakerindividualitySuprasegmentalTemporal2014,\n\ttitle = {Speaker-individuality in suprasegmental temporal features: {Implications} for forensic voice comparison},\n\tvolume = {238},\n\tissn = {03790738},\n\tdoi = {10.1016/j.forsciint.2014.02.019},\n\tabstract = {Everyday experience tells us that it is often possible to identify a familiar speaker solely by his/her voice. Such observations reveal that speakers carry individual features in their voices. The present study examines how suprasegmental temporal features contribute to speaker-individuality. Based on data of a homogeneous group of Zurich German speakers, we conducted an experiment that included speaking style variability (spontaneous vs. read speech) and channel variability (high-quality vs. mobile phone-transmitted speech), both of which are characteristic of forensic casework. Speakers demonstrated high between-speaker variability in both read and spontaneous speech, and low within-speaker variability across the two speaking styles. Results further revealed that distortions of the type introduced by mobile telephony had little effect on suprasegmental temporal characteristics. Given this evidence of speaker-individuality, we discuss suprasegmental temporal features' potential for forensic voice comparison. © 2014 Elsevier Ireland Ltd.},\n\tjournal = {Forensic Science International},\n\tauthor = {Leemann, Adrian and Kolly, Marie José and Dellwo, Volker},\n\tyear = {2014},\n\tpages = {59--67},\n}\n\n\n\n
\n
\n\n\n
\n Everyday experience tells us that it is often possible to identify a familiar speaker solely by his/her voice. Such observations reveal that speakers carry individual features in their voices. The present study examines how suprasegmental temporal features contribute to speaker-individuality. Based on data of a homogeneous group of Zurich German speakers, we conducted an experiment that included speaking style variability (spontaneous vs. read speech) and channel variability (high-quality vs. mobile phone-transmitted speech), both of which are characteristic of forensic casework. Speakers demonstrated high between-speaker variability in both read and spontaneous speech, and low within-speaker variability across the two speaking styles. Results further revealed that distortions of the type introduced by mobile telephony had little effect on suprasegmental temporal characteristics. Given this evidence of speaker-individuality, we discuss suprasegmental temporal features' potential for forensic voice comparison. © 2014 Elsevier Ireland Ltd.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The influence of speech rate on Fujisaki model parameters.\n \n \n \n \n\n\n \n Mixdorff, H.; Leemann, A.; and Dellwo, V.\n\n\n \n\n\n\n Eurasip Journal on Audio, Speech, and Music Processing, 2014(1): 1–11. 2014.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Mixdorff2014,\n\ttitle = {The influence of speech rate on {Fujisaki} model parameters},\n\tvolume = {2014},\n\tissn = {16874722},\n\turl = {http://asmp.eurasipjournals.com/content/2014/1/33},\n\tdoi = {10.1186/s13636-014-0033-6},\n\tabstract = {The current paper examines influences of speech rate on Fujisaki model parameters based on read speech from the BonnTempo-Corpus containing productions by 12 native speakers of German at five different intended tempo levels (very slow, slow, normal, fast, fastest possible). The normal condition was produced at an average rate of 6.34 syllables/s or 100\\%, the very slow version at 67\\%, and the fastest version at 161\\% of the normal rate. We extracted F0 contours and subjected them to decomposition using the Fujisaki model. We ordered all the data with respect to their actual speech rates. First, we assessed how prosodic realizations vary with speech rate and examined phrase command magnitudes, the number of phrase commands as well as the base frequency, accent command amplitudes, and the timing of accent command with respects to the underlying syllables and their nuclear vowels. Second, we analyzed between-sentence variability within and between speakers and investigated whether and how the prosodic structure is preserved at different speech rates. For very slow speech, we found for some of the speakers that the original phrase structure had disintegrated into something like a list of isolated words separated by pauses. Very fast speech became chains of uniform syllables at very high pitch and with almost flat intonation. With respect to the F0 range reflected by the amplitude of accent commands, we found strong interspeaker differences. While four of the subjects exhibited a significant reduction at higher speech rates, the others did not. As speed increases, it appears that F0 gestures commence earlier in the syllable, that is, the onset time of accent commands is located closer to the syllable/vowel onset than at lower speed.},\n\tnumber = {1},\n\tjournal = {Eurasip Journal on Audio, Speech, and Music Processing},\n\tauthor = {Mixdorff, Hansjörg and Leemann, Adrian and Dellwo, Volker},\n\tyear = {2014},\n\tpages = {1--11},\n}\n\n\n\n
\n
\n\n\n
\n The current paper examines influences of speech rate on Fujisaki model parameters based on read speech from the BonnTempo-Corpus containing productions by 12 native speakers of German at five different intended tempo levels (very slow, slow, normal, fast, fastest possible). The normal condition was produced at an average rate of 6.34 syllables/s or 100%, the very slow version at 67%, and the fastest version at 161% of the normal rate. We extracted F0 contours and subjected them to decomposition using the Fujisaki model. We ordered all the data with respect to their actual speech rates. First, we assessed how prosodic realizations vary with speech rate and examined phrase command magnitudes, the number of phrase commands as well as the base frequency, accent command amplitudes, and the timing of accent command with respects to the underlying syllables and their nuclear vowels. Second, we analyzed between-sentence variability within and between speakers and investigated whether and how the prosodic structure is preserved at different speech rates. For very slow speech, we found for some of the speakers that the original phrase structure had disintegrated into something like a list of isolated words separated by pauses. Very fast speech became chains of uniform syllables at very high pitch and with almost flat intonation. With respect to the F0 range reflected by the amplitude of accent commands, we found strong interspeaker differences. While four of the subjects exhibited a significant reduction at higher speech rates, the others did not. As speed increases, it appears that F0 gestures commence earlier in the syllable, that is, the onset time of accent commands is located closer to the syllable/vowel onset than at lower speed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Die Leistung im Memorieren und Nachsprechen von Pseudowörtern: Eine Untersuchung zum Wortakzent im Deutschen.\n \n \n \n \n\n\n \n Riedl, L.; Wiese, R.; Dellwo, V.; and Wittig, A.\n\n\n \n\n\n\n Linguistische Berichte, 2014(240): 447–470. November 2014.\n \n\n\n\n
\n\n\n\n \n \n \"DiePaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{riedlLeistungImMemorieren2014,\n\ttitle = {Die {Leistung} im {Memorieren} und {Nachsprechen} von {Pseudowörtern}: {Eine} {Untersuchung} zum {Wortakzent} im {Deutschen}},\n\tvolume = {2014},\n\tissn = {0024-3930},\n\turl = {http://www.zora.uzh.ch/id/eprint/103009/},\n\tabstract = {The adequate description of word stress is still a matter of discussion in phonological research. There are two types of approaches to explain German word stress: quantity-sensitive approaches (e.g., Giegerich 1985), on the one hand, claim that stress depends on syllable weight (the inherent structure of a syllable), quantity-insensitive approaches (e.g., Wiese 2000), on the other hand, claim that German word stress falls on a specific position in a word. There are some studies on the assignment of word stress by (language impaired) native speakers of German. Janßen (2003) found proof for the quantity-sensitive approach to German when the participants were urged to read out pseudowords. The present experiment is on perception: we presented spoken three syllable pseudowords to healthy participants and instructed them to: (a) remember as many items as they could (memory task), and (b) repeat the words (repetition task). Since regular word stress is assumed to make use of fewer cognitive resources than irregular word stress we expected participants to prefer one specific type of word stress in the memory task as well as in the repetition task. We found a preference for pseudowords stressed on the antepenultima (and penultima), supporting neither quantity-sensitive nor quantity-insensitive approaches, but an alternative approach connecting both approaches.},\n\tnumber = {240},\n\tjournal = {Linguistische Berichte},\n\tpublisher = {Helmut Buske Verlag},\n\tauthor = {Riedl, Lydia and Wiese, Richard and Dellwo, Volker and Wittig, Annika},\n\tmonth = nov,\n\tyear = {2014},\n\tpages = {447--470},\n}\n\n\n\n
\n
\n\n\n
\n The adequate description of word stress is still a matter of discussion in phonological research. There are two types of approaches to explain German word stress: quantity-sensitive approaches (e.g., Giegerich 1985), on the one hand, claim that stress depends on syllable weight (the inherent structure of a syllable), quantity-insensitive approaches (e.g., Wiese 2000), on the other hand, claim that German word stress falls on a specific position in a word. There are some studies on the assignment of word stress by (language impaired) native speakers of German. Janßen (2003) found proof for the quantity-sensitive approach to German when the participants were urged to read out pseudowords. The present experiment is on perception: we presented spoken three syllable pseudowords to healthy participants and instructed them to: (a) remember as many items as they could (memory task), and (b) repeat the words (repetition task). Since regular word stress is assumed to make use of fewer cognitive resources than irregular word stress we expected participants to prefer one specific type of word stress in the memory task as well as in the repetition task. We found a preference for pseudowords stressed on the antepenultima (and penultima), supporting neither quantity-sensitive nor quantity-insensitive approaches, but an alternative approach connecting both approaches.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Intelligibility of High-Pitched Vowel Sounds in the Singing and Speaking of a Female Cantonese Opera Singer.\n \n \n \n\n\n \n Maurer, D.; Mok, P.; Friedrichs, D.; and Dellwo, V.\n\n\n \n\n\n\n . 2014.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{maurerIntelligibilityHighPitchedVowel2014,\n\ttitle = {Intelligibility of {High}-{Pitched} {Vowel} {Sounds} in the {Singing} and {Speaking} of a {Female} {Cantonese} {Opera} {Singer}},\n\tabstract = {The question whether or not vowel quality can be maintained at F0 of the sounds exceeding statistical F1 of „normal“ speech is still a matter of debate. The present study investigates the perception of long Cantonese vowels /i, y, œ, a, ɔ, u/, spoken and sung in (C)V and (C)V:S context by a well-known female Cantonese Opera singer in the range of F0 of c. 560–860Hz. 172 high-pitched syllables or isolated vowel sounds were selected and extracted from a live recording. Corresponding vowel perception was tested in a listening test performed by 26 students of linguistics. All six vowels proved to be identifiable {\\textgreater} 80\\% up to F0 of c. 700Hz, and sounds of /i, a, ɔ, u/ proved to be identifiable {\\textgreater} 80\\% up to a range of F0 of c. 820–860Hz. Confusion matrices are provided in the present paper and spectral illustrations of all sounds are presented at http://is2014.phones-and-phonemes.org.},\n\tlanguage = {en},\n\tauthor = {Maurer, Dieter and Mok, Peggy and Friedrichs, Daniel and Dellwo, Volker},\n\tyear = {2014},\n}\n\n\n\n
\n
\n\n\n
\n The question whether or not vowel quality can be maintained at F0 of the sounds exceeding statistical F1 of „normal“ speech is still a matter of debate. The present study investigates the perception of long Cantonese vowels /i, y, œ, a, ɔ, u/, spoken and sung in (C)V and (C)V:S context by a well-known female Cantonese Opera singer in the range of F0 of c. 560–860Hz. 172 high-pitched syllables or isolated vowel sounds were selected and extracted from a live recording. Corresponding vowel perception was tested in a listening test performed by 26 students of linguistics. All six vowels proved to be identifiable \\textgreater 80% up to F0 of c. 700Hz, and sounds of /i, a, ɔ, u/ proved to be identifiable \\textgreater 80% up to a range of F0 of c. 820–860Hz. Confusion matrices are provided in the present paper and spectral illustrations of all sounds are presented at http://is2014.phones-and-phonemes.org.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n A Crowdsourcing Smartphone Application for Swiss German: Putting Language Documentation in the Hands of the Users.\n \n \n \n \n\n\n \n Goldman, J.; Leemann, A.; Kolly, M.; Hove, I.; Almajai, I.; Dellwo, V.; and Moran, S.\n\n\n \n\n\n\n . May 2014.\n \n\n\n\n
\n\n\n\n \n \n \"APaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{goldmanCrowdsourcingSmartphoneApplication2014,\n\ttitle = {A {Crowdsourcing} {Smartphone} {Application} for {Swiss} {German}: {Putting} {Language} {Documentation} in the {Hands} of the {Users}},\n\tshorttitle = {A {Crowdsourcing} {Smartphone} {Application} for {Swiss} {German}},\n\turl = {https://www.zora.uzh.ch/id/eprint/103791},\n\tdoi = {10.5167/UZH-103791},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {European Language Resources Association (ELRA)},\n\tauthor = {Goldman, Jean-Philippe and Leemann, Adrian and Kolly, Marie-José and Hove, Ingrid and Almajai, Ibrahim and Dellwo, Volker and Moran, Steven},\n\tmonth = may,\n\tyear = {2014},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Crowdsourcing regional variation in speaking rate through the iOS app ‘Dialäkt Äpp’.\n \n \n \n \n\n\n \n Leemann, A.; Kolly, M. J.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2014, pages 217–221, May 2014. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"CrowdsourcingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{leemannCrowdsourcingRegionalVariation2014,\n\ttitle = {Crowdsourcing regional variation in speaking rate through the {iOS} app ‘{Dialäkt} Äpp’},\n\turl = {https://www.isca-archive.org/speechprosody_2014/leemann14_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2014-31},\n\tabstract = {It is a common stereotype in Switzerland that speakers from Bern speak slowly and speakers from Zurich speak quickly. Are these differences in perception at all mirrored in production? We present a new method of crowdsourcing speaking rate through a free of charge iOS application. Astonishingly, results indicate that the temporal structure of a few words alone – as spoken by a few hundred speakers – are sufficient to tell apart the two dialects in speaking rate. In line with previous literature, females articulate more slowly than males. Further potential fields of application of the introduced method are discussed.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2014},\n\tpublisher = {ISCA},\n\tauthor = {Leemann, Adrian and Kolly, Marie José and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2014},\n\tpages = {217--221},\n}\n\n\n\n
\n
\n\n\n
\n It is a common stereotype in Switzerland that speakers from Bern speak slowly and speakers from Zurich speak quickly. Are these differences in perception at all mirrored in production? We present a new method of crowdsourcing speaking rate through a free of charge iOS application. Astonishingly, results indicate that the temporal structure of a few words alone – as spoken by a few hundred speakers – are sufficient to tell apart the two dialects in speaking rate. In line with previous literature, females articulate more slowly than males. Further potential fields of application of the introduced method are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Disentangling sources of rhythmic variability between dialects.\n \n \n \n \n\n\n \n Leemann, A.; Dellwo, V.; Kolly, M. J.; and Schmid, S.\n\n\n \n\n\n\n In Speech Prosody 2014, pages 693–697, May 2014. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"DisentanglingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{leemannDisentanglingSourcesRhythmic2014,\n\ttitle = {Disentangling sources of rhythmic variability between dialects},\n\turl = {https://www.isca-archive.org/speechprosody_2014/leemann14b_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2014-127},\n\tabstract = {Speech rhythm is highly variable. Previous studies reported variability between languages, dialects, speakers, and labelers. Research further revealed an effect of sentence in the rhythmic characteristics of speakers of the same language. In the present study we tested whether the effect of sentence material is constant across varieties of the same language. We addressed this question by an example of analyzing rhythmic variability between eight dialects of Swiss German in three different sentences. Results showed a significant interaction for dialect*sentence for most of the tested rhythm metrics. We take this as evidence that differences between dialects are contingent upon the sentences used in the experiment. We further investigated which sources in the sentence material caused between-dialect differences in rhythm scores to vary. We found exemplary evidence that dialect-specific phonological and morphological phenomena contained in the individual sentences are the prime suspects. Implications for future speech rhythm research are discussed.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2014},\n\tpublisher = {ISCA},\n\tauthor = {Leemann, Adrian and Dellwo, Volker and Kolly, Marie José and Schmid, Stephan},\n\tmonth = may,\n\tyear = {2014},\n\tpages = {693--697},\n}\n\n\n\n
\n
\n\n\n
\n Speech rhythm is highly variable. Previous studies reported variability between languages, dialects, speakers, and labelers. Research further revealed an effect of sentence in the rhythmic characteristics of speakers of the same language. In the present study we tested whether the effect of sentence material is constant across varieties of the same language. We addressed this question by an example of analyzing rhythmic variability between eight dialects of Swiss German in three different sentences. Results showed a significant interaction for dialect*sentence for most of the tested rhythm metrics. We take this as evidence that differences between dialects are contingent upon the sentences used in the experiment. We further investigated which sources in the sentence material caused between-dialect differences in rhythm scores to vary. We found exemplary evidence that dialect-specific phonological and morphological phenomena contained in the individual sentences are the prime suspects. Implications for future speech rhythm research are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n ‘Swiss Voice App’: A smartphone application for crowdsourcing Swiss German dialect data.\n \n \n \n \n\n\n \n Kolly, M.; Leemann, A.; Dellwo, V.; Goldman, J.; Hove, I.; and Ibrahim, A.\n\n\n \n\n\n\n . July 2014.\n \n\n\n\n
\n\n\n\n \n \n \"‘SwissPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{kollySwissVoiceApp2014,\n\ttitle = {‘{Swiss} {Voice} {App}’: {A} smartphone application for crowdsourcing {Swiss} {German} dialect data},\n\tshorttitle = {‘{Swiss} {Voice} {App}’},\n\turl = {https://www.zora.uzh.ch/id/eprint/105411},\n\tdoi = {10.5167/UZH-105411},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Kolly, Marie-José and Leemann, Adrian and Dellwo, Volker and Goldman, Jean-Philippe and Hove, Ingrid and Ibrahim, Almajai},\n\tmonth = jul,\n\tyear = {2014},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2013\n \n \n (5)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Rhythmic characteristics of voice between and within languages.\n \n \n \n \n\n\n \n Dellwo, V.; and Fourcin, A.\n\n\n \n\n\n\n Travaux neuchatelois de linguistique, 59: 87–107. 2013.\n \n\n\n\n
\n\n\n\n \n \n \"RhythmicPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoRhythmicCharacteristicsVoice2013,\n\ttitle = {Rhythmic characteristics of voice between and within languages},\n\tvolume = {59},\n\tissn = {1010-1705},\n\turl = {https://doi.org/10.5167/uzh-91230},\n\tabstract = {Die vorliegende Studie untersuchte die Rolle von stimmhaften Intervallen (d.h. Intervalle laryngaler Aktivität) rhythmische Charakteristika im Sprachsignal zu kodieren. Die Dauercharakteristika stimmhafter und stimmloser intervalle (\\%VO, deltaUV, VarcoUV, VarcoVO, n-PVI\\_VO, r-PVI\\_UV) wurden analysiert. Aufgrund der untersuchten Sprachen konnten wir zeigen, dass stimmhafte Dauercharakteristika effektiv zu einer Klassifizierung von Sprachen führen, die einer auditorischen Klassifizierung der Sprachen in Rhythmusklassen (akzentzählend, silbenzählend) entspricht. Weiterhin fanden wir Variation zwischen den Sprechern einer Sprache (Deutsch). Wir argumentieren, dass unsere Methode direkt verwandt mit der möglicherweise auditiv hervortretensten Komponente der menschlichen Stimme (das Stimmsignal) ist. Methodische Vorteile sind, dass die stimmlichen Dauercharakteristika verlässlich automatisch aufgrund des Stimmsignals berechnet werden können. Implikationen unserer Befunde zum Erwerb prosodischer Phänomene und zur Wahrnehmung von Sprache durch Neugeborene werden diskutiert.},\n\tjournal = {Travaux neuchatelois de linguistique},\n\tpublisher = {Université de Neuchâtel ; RERO},\n\tauthor = {Dellwo, Volker and Fourcin, Adrian},\n\tyear = {2013},\n\tpages = {87--107},\n}\n\n\n\n
\n
\n\n\n
\n Die vorliegende Studie untersuchte die Rolle von stimmhaften Intervallen (d.h. Intervalle laryngaler Aktivität) rhythmische Charakteristika im Sprachsignal zu kodieren. Die Dauercharakteristika stimmhafter und stimmloser intervalle (%VO, deltaUV, VarcoUV, VarcoVO, n-PVI_VO, r-PVI_UV) wurden analysiert. Aufgrund der untersuchten Sprachen konnten wir zeigen, dass stimmhafte Dauercharakteristika effektiv zu einer Klassifizierung von Sprachen führen, die einer auditorischen Klassifizierung der Sprachen in Rhythmusklassen (akzentzählend, silbenzählend) entspricht. Weiterhin fanden wir Variation zwischen den Sprechern einer Sprache (Deutsch). Wir argumentieren, dass unsere Methode direkt verwandt mit der möglicherweise auditiv hervortretensten Komponente der menschlichen Stimme (das Stimmsignal) ist. Methodische Vorteile sind, dass die stimmlichen Dauercharakteristika verlässlich automatisch aufgrund des Stimmsignals berechnet werden können. Implikationen unserer Befunde zum Erwerb prosodischer Phänomene und zur Wahrnehmung von Sprache durch Neugeborene werden diskutiert.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Rhythmische Variabilität bei synchronem Sprechen und ihre Bedeutung für die forensische Sprecheridentifizierung.\n \n \n \n\n\n \n Friedrichs, D.; and Dellwo, V.\n\n\n \n\n\n\n TRANEL - Travaux neuchâtelois de linguistique, 59: 149–166. 2013.\n \n\n\n\n
\n\n\n\n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Friedrichs2013,\n\ttitle = {Rhythmische {Variabilität} bei synchronem {Sprechen} und ihre {Bedeutung} für die forensische {Sprecheridentifizierung}},\n\tvolume = {59},\n\tissn = {1010-1705},\n\tdoi = {10.5167/uzh-91202},\n\tjournal = {TRANEL - Travaux neuchâtelois de linguistique},\n\tauthor = {Friedrichs, Daniel and Dellwo, Volker},\n\tyear = {2013},\n\tpages = {149--166},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n ( How ) can listeners identify the L1 in foreign accented L2 speech ?.\n \n \n \n\n\n \n Kolly, M.; and Dellwo, V.\n\n\n \n\n\n\n Travaux neuchatelois de linguistique, 148: 127–148. 2013.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Kolly2013,\n\ttitle = {( {How} ) can listeners identify the {L1} in foreign accented {L2} speech ?},\n\tvolume = {148},\n\tabstract = {Par son accent étranger, un locuteur/une locutrice révèle son origine, sa langue maternelle. Ainsi, la majorité des Suisses alémaniques seront reconnus comme tels, en parlant une langue seconde. A partir de cet accent 'suisse allemand', est-ce qu’on pourra aussi deviner la région dialectale d’où provient le locuteur/la locutrice? Et, si la perception humaine permet l’identification de ces subtilités, quels en sont les indices pertinents dans le signal linguistique? Les expériences de perception conduites avec nos sujets suisses alémaniques démontrent, dans un premier temps, que des différences d’accent dues à un dialecte particulier peuvent être perçues non seulement dans du matériel linguistique allemand standard, mais aussi quand lesdits locuteurs parlent français. Une expérience ultérieure explore l’importance des données temporelles, ou rythmiques, pour l’identification d’un accent étranger.},\n\tjournal = {Travaux neuchatelois de linguistique},\n\tauthor = {Kolly, Marie-josé and Dellwo, Volker},\n\tyear = {2013},\n\tpages = {127--148},\n}\n\n\n\n
\n
\n\n\n
\n Par son accent étranger, un locuteur/une locutrice révèle son origine, sa langue maternelle. Ainsi, la majorité des Suisses alémaniques seront reconnus comme tels, en parlant une langue seconde. A partir de cet accent 'suisse allemand', est-ce qu’on pourra aussi deviner la région dialectale d’où provient le locuteur/la locutrice? Et, si la perception humaine permet l’identification de ces subtilités, quels en sont les indices pertinents dans le signal linguistique? Les expériences de perception conduites avec nos sujets suisses alémaniques démontrent, dans un premier temps, que des différences d’accent dues à un dialecte particulier peuvent être perçues non seulement dans du matériel linguistique allemand standard, mais aussi quand lesdits locuteurs parlent français. Une expérience ultérieure explore l’importance des données temporelles, ou rythmiques, pour l’identification d’un accent étranger.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Sprachrhythmus bei bilingualen Sprechern.\n \n \n \n\n\n \n Schmid, S.; and Dellwo, V.\n\n\n \n\n\n\n Travaux neuchatelois de linguistique, 59(2): 109–126. 2013.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Schmid2013,\n\ttitle = {Sprachrhythmus bei bilingualen {Sprechern}},\n\tvolume = {59},\n\tnumber = {2},\n\tjournal = {Travaux neuchatelois de linguistique},\n\tauthor = {Schmid, Stephan and Dellwo, Volker},\n\tyear = {2013},\n\tpages = {109--126},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Audiovisuelle Sprechererkennung durch linguistisch naive Personen.\n \n \n \n \n\n\n \n Sutter, S.; and Dellwo, V.\n\n\n \n\n\n\n Travaux neuchatelois de linguistique, 59(2003): 167–181. 2013.\n \n\n\n\n
\n\n\n\n \n \n \"AudiovisuellePaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{sutterAudiovisuelleSprechererkennungDurch2013,\n\ttitle = {Audiovisuelle {Sprechererkennung} durch linguistisch naive {Personen}},\n\tvolume = {59},\n\tissn = {1010-1705},\n\turl = {https://doi.org/10.5167/uzh-91232},\n\tabstract = {Human speech perception is not only based on acoustic speech signals but also on visual cues like lip or jaw movements. Based on this assumption we used a between- subject design to test listeners’ speaker identification ability in a voice line-up after they were familiarized with a speaker under either of the following condition: (a) visual and degraded acoustic information, (b) degraded acoustic information only, and (c) visual information only. The results from this experiment indicate that listeners are able to perform the identification task to a considerable degree under all three experimental conditions. We conclude that listeners’ identification ability of speakers based on degraded acoustic material is about as good as their identification ability based on visual speech cues. The combination of acoustic and visual cues does not enhance listeners’ performance.},\n\tnumber = {2003},\n\tjournal = {Travaux neuchatelois de linguistique},\n\tpublisher = {Université de Neuchâtel ; RERO},\n\tauthor = {Sutter, Sibylle and Dellwo, Volker},\n\tyear = {2013},\n\tpages = {167--181},\n}\n\n\n\n
\n
\n\n\n
\n Human speech perception is not only based on acoustic speech signals but also on visual cues like lip or jaw movements. Based on this assumption we used a between- subject design to test listeners’ speaker identification ability in a voice line-up after they were familiarized with a speaker under either of the following condition: (a) visual and degraded acoustic information, (b) degraded acoustic information only, and (c) visual information only. The results from this experiment indicate that listeners are able to perform the identification task to a considerable degree under all three experimental conditions. We conclude that listeners’ identification ability of speakers based on degraded acoustic material is about as good as their identification ability based on visual speech cues. The combination of acoustic and visual cues does not enhance listeners’ performance.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2012\n \n \n (7)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Speaker-idiosyncratic temporal patterns in L2 speech.\n \n \n \n\n\n \n Kolly, M.; Dellwo, V.; and Leemann, A.\n\n\n \n\n\n\n ,1–2. 2012.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Kolly2012,\n\ttitle = {Speaker-idiosyncratic temporal patterns in {L2} speech},\n\tauthor = {Kolly, Marie-josé and Dellwo, Volker and Leemann, Adrian},\n\tyear = {2012},\n\tpages = {1--2},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Rhythmic variability in Swiss German dialects.\n \n \n \n \n\n\n \n Leemann, A.; Dellwo, V.; Kolly, M. M.; and Schmid, S.\n\n\n \n\n\n\n In 6th International Conference on Speech Prosody, pages 1–4, 2012. \n \n\n\n\n
\n\n\n\n \n \n \"RhythmicPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{Leemann2012,\n\ttitle = {Rhythmic variability in {Swiss} {German} dialects},\n\tisbn = {978-7-5608-4869-3},\n\turl = {http://www.scopus.com/inward/record.url?eid=2-s2.0-84902974589&partnerID=MN8TOARS},\n\tabstract = {Speech  rhythm  can  be  measured  acoustically  in  terms  of durational characteristics of consonantal and vocalic intervals . The  present  paper  investigated  how  acoustically  measurable rhythm  varies  across  dialects  of  Swiss  German.  Rhy thmic measurements (\\%V, ∆C, ∆V, varcoC, varcoV, rPVI-C, nPVI -C, nPVI-V) were carried out on four sentences of six speak ers from  eight  Swiss  dialects. Results  indicate  that  there  are significant  differences  acro ss  the  dialects  in  some  rhythm measures  but  not  in  others  and  that  dialects  can  be  grouped according to rhythmic characteristics},\n\tbooktitle = {6th {International} {Conference} on {Speech} {Prosody}},\n\tauthor = {Leemann, Adrian and Dellwo, Volker and Kolly, M.-J. Marie-josé and Schmid, Stephan},\n\tyear = {2012},\n\tpages = {1--4},\n}\n\n\n\n
\n
\n\n\n
\n Speech rhythm can be measured acoustically in terms of durational characteristics of consonantal and vocalic intervals . The present paper investigated how acoustically measurable rhythm varies across dialects of Swiss German. Rhy thmic measurements (%V, ∆C, ∆V, varcoC, varcoV, rPVI-C, nPVI -C, nPVI-V) were carried out on four sentences of six speak ers from eight Swiss dialects. Results indicate that there are significant differences acro ss the dialects in some rhythm measures but not in others and that dialects can be grouped according to rhythmic characteristics\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker idiosyncratic rhythmic features in the speech signal.\n \n \n \n \n\n\n \n Dellwo, V.; Leemann, A.; and Kolly, M.\n\n\n \n\n\n\n In Interspeech 2012, pages 1584–1587, September 2012. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"SpeakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwoSpeakerIdiosyncraticRhythmic2012,\n\ttitle = {Speaker idiosyncratic rhythmic features in the speech signal},\n\turl = {https://www.isca-archive.org/interspeech_2012/dellwo12_interspeech.html},\n\tdoi = {10.21437/Interspeech.2012-342},\n\tabstract = {Speakers' voices are to a high degree individual. In the present paper we report about an ongoing research project in which we study how temporal characteristics of human speech (e.g. segmental or prosodic timing patterns, speech rhythmic characteristics and durational patterns of voicing) contribute to speaker individuality. We report about the creation of the TEVOID-Corpus (Temporal Voice Idiosyncrasy) that we are currently creating in our lab at Zurich University. 8 speakers producing 16 spontaneous sentences each are currently in the database which is rapidly growing. The paper gives an overview of the general ideas for the data collection and first results showing that there are significant rhythmic differences (\\%V, \\%VO, VarcoPeak) in spontaneously produced sentences between speakers of Zurich German.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2012},\n\tpublisher = {ISCA},\n\tauthor = {Dellwo, Volker and Leemann, Adrian and Kolly, Marie-José},\n\tmonth = sep,\n\tyear = {2012},\n\tpages = {1584--1587},\n}\n\n\n\n
\n
\n\n\n
\n Speakers' voices are to a high degree individual. In the present paper we report about an ongoing research project in which we study how temporal characteristics of human speech (e.g. segmental or prosodic timing patterns, speech rhythmic characteristics and durational patterns of voicing) contribute to speaker individuality. We report about the creation of the TEVOID-Corpus (Temporal Voice Idiosyncrasy) that we are currently creating in our lab at Zurich University. 8 speakers producing 16 spontaneous sentences each are currently in the database which is rapidly growing. The paper gives an overview of the general ideas for the data collection and first results showing that there are significant rhythmic differences (%V, %VO, VarcoPeak) in spontaneously produced sentences between speakers of Zurich German.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker identification based on speech rhythm: the case of bilinguals.\n \n \n \n \n\n\n \n Dellwo, V.; Schmid, S.; Leemann, A.; Kolly, M.; and Müller, M.\n\n\n \n\n\n\n . July 2012.\n \n\n\n\n
\n\n\n\n \n \n \"SpeakerPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoSpeakerIdentificationBased2012,\n\ttitle = {Speaker identification based on speech rhythm: the case of bilinguals},\n\tshorttitle = {Speaker identification based on speech rhythm},\n\turl = {https://www.zora.uzh.ch/id/eprint/111811},\n\tdoi = {10.5167/UZH-111811},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Dellwo, Volker and Schmid, Stephan and Leemann, Adrian and Kolly, Marie-José and Müller, Mathias},\n\tmonth = jul,\n\tyear = {2012},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The Phonetics Lab and The Phonogram Archives at Zurich University, Switzerland.\n \n \n \n \n\n\n \n Dellwo, V.; and Studer-Joho, D.\n\n\n \n\n\n\n . 2012.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoPhoneticsLabPhonogram2012,\n\ttitle = {The {Phonetics} {Lab} and {The} {Phonogram} {Archives} at {Zurich} {University}, {Switzerland}},\n\turl = {https://www.zora.uzh.ch/id/eprint/113612},\n\tdoi = {10.5167/UZH-113612},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Society of Phonetic Sciences},\n\tauthor = {Dellwo, Volker and Studer-Joho, Dieter},\n\tyear = {2012},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Variability of speech rhythm in synchronous speech.\n \n \n \n \n\n\n \n Dellwo, V.; and Friedrichs, D.\n\n\n \n\n\n\n In Speech Prosody 2012, pages 539–542, May 2012. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"VariabilityPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwoVariabilitySpeechRhythm2012,\n\ttitle = {Variability of speech rhythm in synchronous speech},\n\turl = {https://www.isca-archive.org/speechprosody_2012/dellwo12_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2012-136},\n\tabstract = {Speakers are able to speak in synchrony to another speaker or to a recording of another speaker. The present research studied whether and, if yes, speakers change their speech rhythm when synchronizing to another speaker. We developed a measure (SRratio) which monitors on a scale between 0 and 1 whether the durational characteristics of a speaker’s synchronous speech are closer to his/her own read speech or closer to the characteristics of the speech of the speaker he/she is synchronizing to. Four speakers (synchronization speakers) synchronizing to twelve recorded sentences produced by four other speakers (target speakers) were studied. The durational characteristics we analyzed were \\%V and nPVI-v. Results for SRratio suggest that complex processes are going on with main effects for synchronization speakers and target speakers and interaction of the two factors.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2012},\n\tpublisher = {ISCA},\n\tauthor = {Dellwo, Volker and Friedrichs, Daniel},\n\tmonth = may,\n\tyear = {2012},\n\tpages = {539--542},\n}\n\n\n\n
\n
\n\n\n
\n Speakers are able to speak in synchrony to another speaker or to a recording of another speaker. The present research studied whether and, if yes, speakers change their speech rhythm when synchronizing to another speaker. We developed a measure (SRratio) which monitors on a scale between 0 and 1 whether the durational characteristics of a speaker’s synchronous speech are closer to his/her own read speech or closer to the characteristics of the speech of the speaker he/she is synchronizing to. Four speakers (synchronization speakers) synchronizing to twelve recorded sentences produced by four other speakers (target speakers) were studied. The durational characteristics we analyzed were %V and nPVI-v. Results for SRratio suggest that complex processes are going on with main effects for synchronization speakers and target speakers and interaction of the two factors.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Caratteristiche temporali del parlato italiano e tedesco: un confronto tra parlanti nativi, bilingui e non-nativi.\n \n \n \n \n\n\n \n Schmid, S.; and Dellwo, V.\n\n\n \n\n\n\n . 2012.\n \n\n\n\n
\n\n\n\n \n \n \"CaratteristichePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{schmidCaratteristicheTemporaliParlato2012,\n\ttitle = {Caratteristiche temporali del parlato italiano e tedesco: un confronto tra parlanti nativi, bilingui e non-nativi},\n\tshorttitle = {Caratteristiche temporali del parlato italiano e tedesco},\n\turl = {https://www.zora.uzh.ch/id/eprint/91064},\n\tdoi = {10.5167/UZH-91064},\n\tlanguage = {it},\n\turldate = {2025-04-18},\n\tpublisher = {Bulzoni},\n\tauthor = {Schmid, Stephan and Dellwo, Volker},\n\tyear = {2012},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2011\n \n \n (3)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Assessment of rhythm.\n \n \n \n\n\n \n White, L; Liss, J M; and Dellwo, V\n\n\n \n\n\n\n In Lowit, A.; and Kent, R. D., editor(s), Assessment of motor speech disorders, pages 213–252. Plural, San Diego, 2011.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@incollection{White2011,\n\taddress = {San Diego},\n\ttitle = {Assessment of rhythm},\n\tbooktitle = {Assessment of motor speech disorders},\n\tpublisher = {Plural},\n\tauthor = {White, L and Liss, J M and Dellwo, V},\n\teditor = {Lowit, A. and Kent, R. D.},\n\tyear = {2011},\n\tpages = {213--252},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n ‘Read speech normalization’ (RSN): a method to study prosodic variability in spontaneous speech.\n \n \n \n \n\n\n \n Zipp, L.; and Dellwo, V.\n\n\n \n\n\n\n . August 2011.\n \n\n\n\n
\n\n\n\n \n \n \"‘ReadPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{zippReadSpeechNormalization2011,\n\ttitle = {‘{Read} speech normalization’ ({RSN}): a method to study prosodic variability in spontaneous speech},\n\tshorttitle = {‘{Read} speech normalization’ ({RSN})},\n\turl = {https://www.zora.uzh.ch/id/eprint/49272},\n\tdoi = {10.5167/UZH-49272},\n\tabstract = {A method, the ‘read speech normalization’ method (RSN), is proposed by which the variability of prosodic parameters (rhythmic durational and intonation) can be compared across different conditions of spontaneous speech. In an experiment using a customized variety of the map task method, spontaneous speech was elicited from two speakers of English in a formal and an informal situation. Sentences from the spontaneously spoken formal and informal situations were afterwards read by the same speakers. The formal and informal conditions were then compared in terms of their differences to the read speech version (read speech normalization). Results showed that meaningful differences could be observed between formal and informal speech from the read speech normalized sentences that could not be obtained by comparing the two spontaneous speech conditions directly.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Phonetic Association},\n\tauthor = {Zipp, Lena and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2011},\n}\n\n\n\n
\n
\n\n\n
\n A method, the ‘read speech normalization’ method (RSN), is proposed by which the variability of prosodic parameters (rhythmic durational and intonation) can be compared across different conditions of spontaneous speech. In an experiment using a customized variety of the map task method, spontaneous speech was elicited from two speakers of English in a formal and an informal situation. Sentences from the spontaneously spoken formal and informal situations were afterwards read by the same speakers. The formal and informal conditions were then compared in terms of their differences to the read speech version (read speech normalization). Results showed that meaningful differences could be observed between formal and informal speech from the read speech normalized sentences that could not be obtained by comparing the two spontaneous speech conditions directly.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Influences of segmental content on the perception of word duration: a first approach towards a new perceptual model of speech rhythm.\n \n \n \n \n\n\n \n Dellwo, V.; and Hagmann, L.\n\n\n \n\n\n\n . August 2011.\n \n\n\n\n
\n\n\n\n \n \n \"InfluencesPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoInfluencesSegmentalContent2011,\n\ttitle = {Influences of segmental content on the perception of word duration: a first approach towards a new perceptual model of speech rhythm},\n\tshorttitle = {Influences of segmental content on the perception of word duration},\n\turl = {https://www.zora.uzh.ch/id/eprint/49270},\n\tdoi = {10.5167/UZH-49270},\n\tabstract = {The present research tested the segmental influences on the perception of monosyllabic word durations. 12 listeners of Swiss-German heard pairs of speech and non-speech sounds (monosyllabic words and rectangular-gated sinusoids). They were asked to change the duration of the second sound so that it would match the duration of the first one. Results showed that the Weber Fraction (∆T/T) for tones is normally distributed around 0 (perfect alignment) while the Weber Fraction for different monosyllabic word pairs can vary significantly from this. In particular quantitative vowel length in contrastive position was found to have an effect on the perception of word duration. Possible implications for models of speech rhythm are discussed.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Phonetic Association},\n\tauthor = {Dellwo, Volker and Hagmann, Lea},\n\tmonth = aug,\n\tyear = {2011},\n}\n\n\n\n
\n
\n\n\n
\n The present research tested the segmental influences on the perception of monosyllabic word durations. 12 listeners of Swiss-German heard pairs of speech and non-speech sounds (monosyllabic words and rectangular-gated sinusoids). They were asked to change the duration of the second sound so that it would match the duration of the first one. Results showed that the Weber Fraction (∆T/T) for tones is normally distributed around 0 (perfect alignment) while the Weber Fraction for different monosyllabic word pairs can vary significantly from this. In particular quantitative vowel length in contrastive position was found to have an effect on the perception of word duration. Possible implications for models of speech rhythm are discussed.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2010\n \n \n (4)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Influences of speech rate on the acoustic correlates of speech rhythm: An experimental phonetic study based on acoustic and perceptual evidence.\n \n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n University of Bonn, Faculty of Philosophy, Bonn, 2010.\n \n\n\n\n
\n\n\n\n \n \n \"InfluencesPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@book{Dellwo2010a,\n\taddress = {Bonn},\n\ttitle = {Influences of speech rate on the acoustic correlates of speech rhythm: {An} experimental phonetic study based on acoustic and perceptual evidence},\n\turl = {http://hss.ulb.uni-bonn.de/2010/2003/2003.htm},\n\tabstract = {Human listeners can distinguish languages of different rhythmic classes (rhythm class hypothesis) based on durational characteristics of consonantal and vocalic intervals (e.g. \\%V, deltaC, PVI). The present study investigated the influence of speech rate on measurements of speech rhythm and on the perception of rhythmic differences between languages. A database (BonnTempo-Corpus) was created comprising speech from 5 different languages (Czech, English, French, German, and Italian) that was characterised by high tempo variability within each language (very slow, slow, normal, fast and fastest possible speech). Results from the database revealed that measures of speech rate are equally reliable rhythm class indicators as measures of speech rhythm. As a consequence of this finding perception experiments were conducted to test to what extent listeners use rate as an indicator to rhythm class. In one experiment native listeners rated rhythmic regularity in delexicalized French and English sentences that varied in acoustically measurable rhythm and rate to natural degrees. Results revealed that rate correlated best with listeners’ perception of rhythmical events. In a second perceptual experiment listeners of English and French rated the pronunciation quality of speech in their native language produced by German native speakers. Results provided further evidence for rate measures being a correlate of pronunciation quality. No such evidence could be found for rhythm measures. Based on these findings a new theory was developed which claims that the auditory impression of languages falling into different rhythmic classes is to a high degree based on rate differences between these languages. To develop this argument it was further necessary to study the effects of speech rate on measurements of speech rhythm. It was demonstrated that some of the rhythm measures correlated negatively with speech rate (deltaC, rPVI). For this reason rate normalization methods for rhythmic measurements were either developed or refined. The study also includes two excursions, one in which the history of the rhythm class argument was studied and another one in which influences of linguistic/para-linguistic parameters on measurements of speech rhythm were analyzed.},\n\tpublisher = {University of Bonn, Faculty of Philosophy},\n\tauthor = {Dellwo, Volker},\n\tyear = {2010},\n}\n\n\n\n
\n
\n\n\n
\n Human listeners can distinguish languages of different rhythmic classes (rhythm class hypothesis) based on durational characteristics of consonantal and vocalic intervals (e.g. %V, deltaC, PVI). The present study investigated the influence of speech rate on measurements of speech rhythm and on the perception of rhythmic differences between languages. A database (BonnTempo-Corpus) was created comprising speech from 5 different languages (Czech, English, French, German, and Italian) that was characterised by high tempo variability within each language (very slow, slow, normal, fast and fastest possible speech). Results from the database revealed that measures of speech rate are equally reliable rhythm class indicators as measures of speech rhythm. As a consequence of this finding perception experiments were conducted to test to what extent listeners use rate as an indicator to rhythm class. In one experiment native listeners rated rhythmic regularity in delexicalized French and English sentences that varied in acoustically measurable rhythm and rate to natural degrees. Results revealed that rate correlated best with listeners’ perception of rhythmical events. In a second perceptual experiment listeners of English and French rated the pronunciation quality of speech in their native language produced by German native speakers. Results provided further evidence for rate measures being a correlate of pronunciation quality. No such evidence could be found for rhythm measures. Based on these findings a new theory was developed which claims that the auditory impression of languages falling into different rhythmic classes is to a high degree based on rate differences between these languages. To develop this argument it was further necessary to study the effects of speech rate on measurements of speech rhythm. It was demonstrated that some of the rhythm measures correlated negatively with speech rate (deltaC, rPVI). For this reason rate normalization methods for rhythmic measurements were either developed or refined. The study also includes two excursions, one in which the history of the rhythm class argument was studied and another one in which influences of linguistic/para-linguistic parameters on measurements of speech rhythm were analyzed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Speaker identification based on speech rhythm : the case of bilinguals.\n \n \n \n \n\n\n \n Dellwo, V.; Schmid, S.; Leemann, A.; Kolly, M.; and Müller, M.\n\n\n \n\n\n\n Perspectives on Rhythm and Timing (PoRT),5. July 2010.\n \n\n\n\n
\n\n\n\n \n \n \"SpeakerPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoSpeakerIdentificationBased2010,\n\ttitle = {Speaker identification based on speech rhythm : the case of bilinguals},\n\turl = {https://doi.org/10.5167/uzh-111811},\n\tabstract = {Voices are highly individual. The present study investigated how temporal characteristics of speech can contribute to speaker individuality for the same speaker speaking in different languages. By now there is a large body of evidence showing that measures based on temporal characteristics of consonantal and vocalic interval durations show drastic within language variability that is to a high degree a result of between speaker variability ([1], [2], [4]). Here we present results from an experiment on L2 and bilingual Italian/German speakers. Our assumption was that if speaker idiosyncratic rhythmic characteristics exist, then they should be present across utterances from different languages produced by the same speaker.},\n\tjournal = {Perspectives on Rhythm and Timing (PoRT)},\n\tpublisher = {s.n.},\n\tauthor = {Dellwo, Volker and Schmid, Stephan and Leemann, Adrian and Kolly, Marie-josé and Müller, Matthias},\n\tmonth = jul,\n\tyear = {2010},\n\tpages = {5},\n}\n\n\n\n
\n
\n\n\n
\n Voices are highly individual. The present study investigated how temporal characteristics of speech can contribute to speaker individuality for the same speaker speaking in different languages. By now there is a large body of evidence showing that measures based on temporal characteristics of consonantal and vocalic interval durations show drastic within language variability that is to a high degree a result of between speaker variability ([1], [2], [4]). Here we present results from an experiment on L2 and bilingual Italian/German speakers. Our assumption was that if speaker idiosyncratic rhythmic characteristics exist, then they should be present across utterances from different languages produced by the same speaker.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Expectations for discourse genre identification: a prosodic study.\n \n \n \n \n\n\n \n Obin, N.; Dellwo, V.; Lacheret, A.; and Rodet, X.\n\n\n \n\n\n\n In Interspeech 2010, pages 3070–3073, September 2010. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ExpectationsPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{obinExpectationsDiscourseGenre2010,\n\ttitle = {Expectations for discourse genre identification: a prosodic study},\n\tshorttitle = {Expectations for discourse genre identification},\n\turl = {https://www.isca-archive.org/interspeech_2010/obin10b_interspeech.html},\n\tdoi = {10.21437/Interspeech.2010-764},\n\tabstract = {Speech can be divided into discourse genres based on the contextual environment it occurs in (e.g. political speech, sport commentary speech, etc.). The present study investigated whether listeners can distinguish between speech from different discourse genres on the basis of acoustic prosodic cues only 1. In a perception experiment with delexicalized speech 70 listeners with varying experience in French (native speakers, nonnative speakers, and non-speakers) were asked to identify four different types of discourse genres (church service, political, journal, and sport commentary). Results revealed a fair identification ability with a significant increase in performance with increasing experience in French. Identification confusion was used to cluster discourse genres according to their perceptual similarity.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Interspeech 2010},\n\tpublisher = {ISCA},\n\tauthor = {Obin, Nicolas and Dellwo, Volker and Lacheret, Anne and Rodet, Xavier},\n\tmonth = sep,\n\tyear = {2010},\n\tpages = {3070--3073},\n}\n\n\n\n
\n
\n\n\n
\n Speech can be divided into discourse genres based on the contextual environment it occurs in (e.g. political speech, sport commentary speech, etc.). The present study investigated whether listeners can distinguish between speech from different discourse genres on the basis of acoustic prosodic cues only 1. In a perception experiment with delexicalized speech 70 listeners with varying experience in French (native speakers, nonnative speakers, and non-speakers) were asked to identify four different types of discourse genres (church service, political, journal, and sport commentary). Results revealed a fair identification ability with a significant increase in performance with increasing experience in French. Identification confusion was used to cluster discourse genres according to their perceptual similarity.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The role of speech rhythm in attending to one of two simultaneous speakers.\n \n \n \n \n\n\n \n Cushing, I. R.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2010, pages paper 039–0, May 2010. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{cushingRoleSpeechRhythm2010,\n\ttitle = {The role of speech rhythm in attending to one of two simultaneous speakers},\n\turl = {https://www.isca-archive.org/speechprosody_2010/cushing10_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2010-22},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2010},\n\tpublisher = {ISCA},\n\tauthor = {Cushing, Ian R. and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2010},\n\tpages = {paper 039--0},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2009\n \n \n (3)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n The influence of voice disguise on temporal characteristics of speech.\n \n \n \n\n\n \n Dellwo, V.; Ramyead, S.; and Dankovicova, J.\n\n\n \n\n\n\n Abstract presented at the …, 7(2007): 2008. 2009.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Dellwo2009,\n\ttitle = {The influence of voice disguise on temporal characteristics of speech},\n\tvolume = {7},\n\tnumber = {2007},\n\tjournal = {Abstract presented at the …},\n\tauthor = {Dellwo, Volker and Ramyead, Sheena and Dankovicova, Jana},\n\tyear = {2009},\n\tpages = {2008},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The development of measurable speech rhythm in Spanish speakers of English.\n \n \n \n \n\n\n \n Dellwo, V.; Gutiérrez Díez, F.; and Gavalda, N.\n\n\n \n\n\n\n . January 2009.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoDevelopmentMeasurableSpeech2009,\n\ttitle = {The development of measurable speech rhythm in {Spanish} speakers of {English}},\n\turl = {https://www.zora.uzh.ch/id/eprint/111804},\n\tdoi = {10.5167/UZH-111804},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Dellwo, Volker and Gutiérrez Díez, Francisco and Gavalda, Nuria},\n\tmonth = jan,\n\tyear = {2009},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Choosing the right rate normalization method for measurements of speech rhythm.\n \n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n . 2009.\n \n\n\n\n
\n\n\n\n \n \n \"ChoosingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoChoosingRightRate2009,\n\ttitle = {Choosing the right rate normalization method for measurements of speech rhythm},\n\turl = {https://www.zora.uzh.ch/id/eprint/45236},\n\tdoi = {10.5167/UZH-45236},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {EDK},\n\tauthor = {Dellwo, Volker},\n\tyear = {2009},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2008\n \n \n (4)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n The role of speech rate in perceiving speech rhythm.\n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n In pages 375–378, 2008. \n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{dellwoRoleSpeechRate2008,\n\ttitle = {The role of speech rate in perceiving speech rhythm},\n\tisbn = {978-0-616-22003-0},\n\tabstract = {Human listeners can distinguish between languages of different rhythmic classes (e.g. stress- and syllable-timed languages). The present study investigated the role of speech rate in this process. Acoustic data suggests (experiment I) that speech rate can distinguishes as reliable between stress- and syllable-timed languages as previously proposed correlates of speech rhythm (\\%V, VarcoC and nPVI). Behavioral data showed (experiment II) that listeners make use of rate differences when asked to assess rhythmic characteristics of stress- and syllable-timed languages in delexicalized speech. Results imply that speech rate is an important acoustic correlate for cross-language speech rhythm.},\n\tnumber = {experiment II},\n\tauthor = {Dellwo, Volker},\n\tyear = {2008},\n\tpages = {375--378},\n}\n\n\n\n
\n
\n\n\n
\n Human listeners can distinguish between languages of different rhythmic classes (e.g. stress- and syllable-timed languages). The present study investigated the role of speech rate in this process. Acoustic data suggests (experiment I) that speech rate can distinguishes as reliable between stress- and syllable-timed languages as previously proposed correlates of speech rhythm (%V, VarcoC and nPVI). Behavioral data showed (experiment II) that listeners make use of rate differences when asked to assess rhythmic characteristics of stress- and syllable-timed languages in delexicalized speech. Results imply that speech rate is an important acoustic correlate for cross-language speech rhythm.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n How speaker idiosyncratic is measurable speech rhythm.\n \n \n \n\n\n \n Dellwo, V.; and Koreman, J.\n\n\n \n\n\n\n Abstract presented at the annual IAFPA …. 2008.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{Dellwo2008b,\n\ttitle = {How speaker idiosyncratic is measurable speech rhythm},\n\tjournal = {Abstract presented at the annual IAFPA …},\n\tauthor = {Dellwo, Volker and Koreman, Jacques},\n\tyear = {2008},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n Rhythmic classification of languages based on voice timing.\n \n \n \n\n\n \n Fourcin, A; and Dellwo, V\n\n\n \n\n\n\n . 2008.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{fourcinRhythmicClassificationLanguages2008,\n\ttitle = {Rhythmic classification of languages based on voice timing},\n\tabstract = {Speech rhythm classes can be distinguished acoustically and perceptually from the variability of consonantal durations and the relative durations of vocalic intervals. The present research investigated whether this distinction can be made robustly, simply on the basis of voice timing, by measuring the durational characteristics of voiced and voiceless intervals in fluent speech. We show that voice patterns — in terms of vocal fold vibration — provide an effective basis for classification and that they can be automatically processed for large datasets. The possible implications that this finding can have on the ability of infants to distinguish between languages of different rhythmic classes are discussed.},\n\tlanguage = {en},\n\tauthor = {Fourcin, A and Dellwo, V},\n\tyear = {2008},\n}\n\n\n\n
\n
\n\n\n
\n Speech rhythm classes can be distinguished acoustically and perceptually from the variability of consonantal durations and the relative durations of vocalic intervals. The present research investigated whether this distinction can be made robustly, simply on the basis of voice timing, by measuring the durational characteristics of voiced and voiceless intervals in fluent speech. We show that voice patterns — in terms of vocal fold vibration — provide an effective basis for classification and that they can be automatically processed for large datasets. The possible implications that this finding can have on the ability of infants to distinguish between languages of different rhythmic classes are discussed.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Comparing native and non-native speech rhythm using acoustic rhythmic measures: Cantonese, Beijing Mandarin and English.\n \n \n \n \n\n\n \n Mok, P. P. K.; and Dellwo, V.\n\n\n \n\n\n\n In Speech Prosody 2008, pages 423–426, May 2008. ISCA\n \n\n\n\n
\n\n\n\n \n \n \"ComparingPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@inproceedings{mokComparingNativeNonnative2008,\n\ttitle = {Comparing native and non-native speech rhythm using acoustic rhythmic measures: {Cantonese}, {Beijing} {Mandarin} and {English}},\n\tshorttitle = {Comparing native and non-native speech rhythm using acoustic rhythmic measures},\n\turl = {https://www.isca-archive.org/speechprosody_2008/mok08_speechprosody.html},\n\tdoi = {10.21437/SpeechProsody.2008-93},\n\tabstract = {This study investigates the speech rhythm of Cantonese, Beijing Mandarin, Cantonese-accented English and Mandarinaccented English using acoustic rhythmic measures. They were compared with four languages in the BonnTempo corpus: German and English (stress-timed) and French and Italian (syllable-timed). Six Cantonese and six Beijing Mandarin native speakers were recorded reading the North Wind and the Sun story with a normal speech rate, telling the story semi-spontaneously and reading the English version of the story. Both raw and normalised rhythmic measures were calculated using vocalic, consonantal and syllabic durations (ΔC, ΔV, ΔS, \\%V, VarcoC, VarcoV, VarcoS, rPVI\\_C, rPVI\\_S, nPVI\\_V, nPVI\\_S). Results confirm the syllabletiming impression of Cantonese and Mandarin. Data of the two foreign English accents poses a challenge to the rhythmic measures because the two accents are syllable-timed impressionistically but were classified as stress-timed by some of the rhythmic measures (ΔC, rPVI\\_C, nPVI\\_V, ΔS, VarcoS, rPVI\\_S and nPVI\\_S). VarcoC and \\%V give the best classification of speech rhythm in this study.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speech {Prosody} 2008},\n\tpublisher = {ISCA},\n\tauthor = {Mok, Peggy P. K. and Dellwo, Volker},\n\tmonth = may,\n\tyear = {2008},\n\tpages = {423--426},\n}\n\n\n\n
\n
\n\n\n
\n This study investigates the speech rhythm of Cantonese, Beijing Mandarin, Cantonese-accented English and Mandarinaccented English using acoustic rhythmic measures. They were compared with four languages in the BonnTempo corpus: German and English (stress-timed) and French and Italian (syllable-timed). Six Cantonese and six Beijing Mandarin native speakers were recorded reading the North Wind and the Sun story with a normal speech rate, telling the story semi-spontaneously and reading the English version of the story. Both raw and normalised rhythmic measures were calculated using vocalic, consonantal and syllabic durations (ΔC, ΔV, ΔS, %V, VarcoC, VarcoV, VarcoS, rPVI_C, rPVI_S, nPVI_V, nPVI_S). Results confirm the syllabletiming impression of Cantonese and Mandarin. Data of the two foreign English accents poses a challenge to the rhythmic measures because the two accents are syllable-timed impressionistically but were classified as stress-timed by some of the rhythmic measures (ΔC, rPVI_C, nPVI_V, ΔS, VarcoS, rPVI_S and nPVI_S). VarcoC and %V give the best classification of speech rhythm in this study.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2007\n \n \n (3)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n Rhythmical classification of languages based on voice parameters.\n \n \n \n\n\n \n Dellwo, V.; Fourcin, A. J.; and Abberton, E. R.\n\n\n \n\n\n\n Proceedings of International Congress of Phonetic Sciences (ICPhS), (August): 1129–1132. 2007.\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoRhythmicalClassificationLanguages2007,\n\ttitle = {Rhythmical classification of languages based on voice parameters},\n\tabstract = {ABSTRACT It has been demonstrated that speech rhythm classes (eg stress-timed, syllable- timed) can be distinguished acoustically and perceptually on the basis of the variability of consonantal and vocalic interval durations. It has moreover been shown that even infants ...},\n\tnumber = {August},\n\tjournal = {Proceedings of International Congress of Phonetic Sciences (ICPhS)},\n\tauthor = {Dellwo, Volker and Fourcin, Adrian John and Abberton, Evelyn R.},\n\tyear = {2007},\n\tpages = {1129--1132},\n}\n\n\n\n
\n
\n\n\n
\n ABSTRACT It has been demonstrated that speech rhythm classes (eg stress-timed, syllable- timed) can be distinguished acoustically and perceptually on the basis of the variability of consonantal and vocalic interval durations. It has moreover been shown that even infants ...\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n How Is Individuality Expressed in Voice? An Introduction to Speech Production and Description for Speaker Classification.\n \n \n \n \n\n\n \n Dellwo, V.; Huckvale, M.; and Ashby, M.\n\n\n \n\n\n\n In Müller, C., editor(s), Speaker Classification I, volume 4343, pages 1–20. Springer Berlin Heidelberg, Berlin, Heidelberg, 2007.\n Series Title: Lecture Notes in Computer Science\n\n\n\n
\n\n\n\n \n \n \"HowPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@incollection{dellwoHowIndividualityExpressed2007a,\n\taddress = {Berlin, Heidelberg},\n\ttitle = {How {Is} {Individuality} {Expressed} in {Voice}? {An} {Introduction} to {Speech} {Production} and {Description} for {Speaker} {Classification}},\n\tvolume = {4343},\n\tisbn = {978-3-540-74186-2 978-3-540-74200-5},\n\tissn = {0302-9743, 1611-3349},\n\tshorttitle = {How {Is} {Individuality} {Expressed} in {Voice}?},\n\turl = {http://link.springer.com/10.1007/978-3-540-74200-5_1},\n\tdoi = {10.1007/978-3-540-74200-5_1},\n\tabstract = {As well as conveying a message in words and sounds, the speech signal carries information about the speaker's own anatomy, physiology, linguistic experience and mental state. These speaker characteristics are found in speech at all levels of description: from the spectral information in the sounds to the choice of words and utterances themselves. This chapter presents an introduction to speech production and to the phonetic description of speech to facilitate discussion of how speech can be a carrier for speaker characteristics as well as a carrier for messages. The chapter presents an overview of the physical structures of the human vocal tract used in speech, it introduces the standard phonetic classification system for the description of spoken gestures and it presents a catalogue of the different ways in which individuality can be expressed through speech. The chapter ends with a brief description of some applications which require access to information about speaker characteristics in speech.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tbooktitle = {Speaker {Classification} {I}},\n\tpublisher = {Springer Berlin Heidelberg},\n\tauthor = {Dellwo, Volker and Huckvale, Mark and Ashby, Michael},\n\teditor = {Müller, Christian},\n\tyear = {2007},\n\tnote = {Series Title: Lecture Notes in Computer Science},\n\tpages = {1--20},\n}\n\n\n\n
\n
\n\n\n
\n As well as conveying a message in words and sounds, the speech signal carries information about the speaker's own anatomy, physiology, linguistic experience and mental state. These speaker characteristics are found in speech at all levels of description: from the spectral information in the sounds to the choice of words and utterances themselves. This chapter presents an introduction to speech production and to the phonetic description of speech to facilitate discussion of how speech can be a carrier for speaker characteristics as well as a carrier for messages. The chapter presents an overview of the physical structures of the human vocal tract used in speech, it introduces the standard phonetic classification system for the description of spoken gestures and it presents a catalogue of the different ways in which individuality can be expressed through speech. The chapter ends with a brief description of some applications which require access to information about speaker characteristics in speech.\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n Czech speech rhythm and the rhythm class hypothesis.\n \n \n \n \n\n\n \n Dancovicova, J.; and Dellwo, V.\n\n\n \n\n\n\n . August 2007.\n \n\n\n\n
\n\n\n\n \n \n \"CzechPaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dancovicovaCzechSpeechRhythm2007,\n\ttitle = {Czech speech rhythm and the rhythm class hypothesis},\n\turl = {https://www.zora.uzh.ch/id/eprint/111796},\n\tdoi = {10.5167/UZH-111796},\n\tabstract = {While a number of languages have been classified as either syllable- or stress-timed, the case of Czech remains unclear. In this paper we make predictions about Czech rhythm on the basis of our analysis of syllable complexity in recorded samples of Czech. The results on syllable complexity show mixed features. This is reflected in the classification of Czech rhythm using rhythm measures based on durational variability of consonantal and vocalic intervals.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {International Phonetic Association},\n\tauthor = {Dancovicova, Jana and Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2007},\n}\n\n\n\n
\n
\n\n\n
\n While a number of languages have been classified as either syllable- or stress-timed, the case of Czech remains unclear. In this paper we make predictions about Czech rhythm on the basis of our analysis of syllable complexity in recorded samples of Czech. The results on syllable complexity show mixed features. This is reflected in the classification of Czech rhythm using rhythm measures based on durational variability of consonantal and vocalic intervals.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2006\n \n \n (1)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n The perception of intended speech rate in English, French, and German by French listeners.\n \n \n \n \n\n\n \n Dellwo, V.; Ferragne, E.; and Pellegrino, F.\n\n\n \n\n\n\n . May 2006.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoPerceptionIntendedSpeech2006,\n\ttitle = {The perception of intended speech rate in {English}, {French}, and {German} by {French} listeners},\n\turl = {https://www.zora.uzh.ch/id/eprint/111782},\n\tdoi = {10.5167/UZH-111782},\n\turldate = {2025-04-18},\n\tpublisher = {ISCA},\n\tauthor = {Dellwo, Volker and Ferragne, Emmanuel and Pellegrino, François},\n\tmonth = may,\n\tyear = {2006},\n}\n\n\n\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2004\n \n \n (1)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n BonnTempo Corpus.\n \n \n \n \n\n\n \n Dellwo, V.; Steiner, I.; Aschenberner, B.; Dankovi, J.; Wagner, P.; and Bonn, U.\n\n\n \n\n\n\n … of Interspeech 2004, (section 4): 1–4. 2004.\n \n\n\n\n
\n\n\n\n \n \n \"BonnTempoPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoBonnTempoCorpus2004,\n\ttitle = {{BonnTempo} {Corpus}},\n\tcopyright = {All rights reserved},\n\turl = {http://pub.uni-bielefeld.de/publication/1917026},\n\tabstract = {Work is currently being carried out on a speech database{\\textbackslash}nconstructed in order to study speech rhythm in connection with speech rate. The database, BonnTempo-Corpus, and the Praat based analysis tools, BonnTempo-Tools, are a powerful{\\textbackslash}ninstrument for examining various aspects of recently proposed rhythm measures (e.g. \\%V, C, nPVI, rPVI, etc.) in relation to speech rate among a wide range of languages and speakers.{\\textbackslash}nFirst observations pose new problems on traditionally not well classifiable languages like Czech.},\n\tnumber = {section 4},\n\tjournal = {… of Interspeech 2004},\n\tauthor = {Dellwo, Volker and Steiner, Ingmar and Aschenberner, Bianca and Dankovi, Jana and Wagner, Petra and Bonn, Universtität},\n\tyear = {2004},\n\tpages = {1--4},\n}\n\n\n\n\n\n\n\n\n\n\n\n
\n
\n\n\n
\n Work is currently being carried out on a speech database\\nconstructed in order to study speech rhythm in connection with speech rate. The database, BonnTempo-Corpus, and the Praat based analysis tools, BonnTempo-Tools, are a powerful\\ninstrument for examining various aspects of recently proposed rhythm measures (e.g. %V, C, nPVI, rPVI, etc.) in relation to speech rate among a wide range of languages and speakers.\\nFirst observations pose new problems on traditionally not well classifiable languages like Czech.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n 2003\n \n \n (2)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n \n Relations between language rhythm and speech rate.\n \n \n \n \n\n\n \n Dellwo, V.; and Wagner, P.\n\n\n \n\n\n\n The 15th International Congress of the Phonetic Sciences, Barcelona.,471–474. 2003.\n \n\n\n\n
\n\n\n\n \n \n \"RelationsPaper\n  \n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoRelationsLanguageRhythm2003a,\n\ttitle = {Relations between language rhythm and speech rate},\n\tcopyright = {All rights reserved},\n\tissn = {1876346485},\n\turl = {http://pub.uni-bielefeld.de/publication/1785384},\n\tabstract = {The interaction between speech rate and rhythm is a topic that has hardly been studied in respective models of language rhythm though its potential significance has recently been addressed: since both of these prosodic parameters are to a great extent dependent on speech timing they are suspected to interact to a great degree. The present research studies the influence of speech rate on the vocalic and intervocalic measures \\%V and ...},\n\tjournal = {The 15th International Congress of the Phonetic Sciences, Barcelona.},\n\tauthor = {Dellwo, Volker and Wagner, Petra},\n\tyear = {2003},\n\tpages = {471--474},\n}\n\n\n\n
\n
\n\n\n
\n The interaction between speech rate and rhythm is a topic that has hardly been studied in respective models of language rhythm though its potential significance has recently been addressed: since both of these prosodic parameters are to a great extent dependent on speech timing they are suspected to interact to a great degree. The present research studies the influence of speech rate on the vocalic and intervocalic measures %V and ...\n
\n\n\n
\n\n\n
\n \n\n \n \n \n \n \n \n The combined analysis of speech and gesture.\n \n \n \n \n\n\n \n Dellwo, V.\n\n\n \n\n\n\n . August 2003.\n \n\n\n\n
\n\n\n\n \n \n \"ThePaper\n  \n \n\n \n \n doi\n  \n \n\n \n link\n  \n \n\n bibtex\n \n\n \n  \n \n abstract \n \n\n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
@article{dellwoCombinedAnalysisSpeech2003,\n\ttitle = {The combined analysis of speech and gesture},\n\turl = {https://www.zora.uzh.ch/id/eprint/111778},\n\tdoi = {10.5167/UZH-111778},\n\tabstract = {The availability of fast computers and recent software developments does not only enable us to carry out complex digital speech analysis procedures on an everyday basis, it also allows us to analyze visual information of the communication act by processing annotated digital video data. This study presents a selection of computer tools with which an editing and analysis process of computer readable video data for communication research can be carried out on different levels. A demonstration of the software tools will be given on an example of a project work currently carried out by the author to study the influence of moods on speech and gestures.},\n\tlanguage = {en},\n\turldate = {2025-04-18},\n\tpublisher = {s.n.},\n\tauthor = {Dellwo, Volker},\n\tmonth = aug,\n\tyear = {2003},\n}\n\n\n\n
\n
\n\n\n
\n The availability of fast computers and recent software developments does not only enable us to carry out complex digital speech analysis procedures on an everyday basis, it also allows us to analyze visual information of the communication act by processing annotated digital video data. This study presents a selection of computer tools with which an editing and analysis process of computer readable video data for communication research can be carried out on different levels. A demonstration of the software tools will be given on an example of a project work currently carried out by the author to study the influence of moods on speech and gestures.\n
\n\n\n
\n\n\n\n\n\n
\n
\n\n
\n
\n  \n undefined\n \n \n (1)\n \n \n
\n
\n \n \n
\n \n\n \n \n \n \n \n .\n \n \n \n\n\n \n \n\n\n \n\n\n\n . .\n \n\n\n\n
\n\n\n\n \n\n \n\n \n link\n  \n \n\n bibtex\n \n\n \n\n \n  \n \n 9 downloads\n \n \n\n \n \n \n \n \n \n \n\n  \n \n \n\n\n\n
\n
\n
\n\n\n\n
\n\n\n\n\n\n
\n
\n\n\n\n\n
\n\n\n \n\n \n \n \n \n\n
\n"}; document.write(bibbase_data.data);