

de Recherche et d’Innovation
en Cybersécurité et Société
Moradi, A.; Falk, T. H.
Benchmarking Foundation Models for Cross-Domain Speaker Profiling Article d'actes
Dans: IEEE Conf. Artif. Intell., CAI, p. 78–84, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833156039-3 (ISBN), (Journal Abbreviation: IEEE Conf. Artif. Intell., CAI).
Résumé | Liens | BibTeX | Étiquettes: Benchmarking, Continuous speech recognition, Cross-domain, Forecasting, Formant frequency, Foundation models, Learning systems, Linguistics, Multi-attributes, Multi-task learning, Paralinguistic, Performance, Speaker identification, Speaker verification, Speech communication, State of the art, Verification task
@inproceedings{moradiBenchmarkingFoundationModels2026,
title = {Benchmarking Foundation Models for Cross-Domain Speaker Profiling},
author = {A. Moradi and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105042046048?origin=resultslist},
doi = {10.1109/CAI68641.2026.11536565},
isbn = {979-833156039-3 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Conf. Artif. Intell., CAI},
pages = {78–84},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Speech conveys both linguistic and paralinguistic content. While pre-trained speech foundation models have been widely explored for linguistic tasks, such as speech recognition, and for speaker identification and verification tasks, very limited work has been done to test their usefulness for multi-attribute speaker profiling, i.e., simultaneous prediction of biological sex, age, and height from speech. In this paper, we aim to benchmark the performance of four state-of-the-art self-supervised foundation models, namely, WavLM, Wav2vec2, HuBERT, and XLSR-53, under both within- and cross-domain conditions. Each model is employed as a frozen pre-trained feature extractor, with lightweight task-specific heads trained jointly in a multi-task learning framework, and their feature extraction latency is empirically analyzed to assess practical deployment considerations. Experiments on two datasets show that WavLM achieves consistently strong within- and cross-domain performance for biological sex prediction, while Wav2vec2 and XLSR-53 exhibit more consistent performance for the age and height regression tasks in cross-domain settings. Interpretability analysis based on correlations with key acoustic features show (1) WavLM internal representations correlating highly with pitch and formant frequencies, corroborating the improved performance on biological sex prediction, and (2) XLSR-53 correlating highly with the second formant frequency, corroborating the results obtained for age and height. Overall, our analysis shows that pre-trained speech foundation models could serve as useful tools for cross-domain speaker profiling tasks. While no model stood out as a clear winner across all tested physical traits, future work could explore the use of ensemble methods for improved generalizability. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Conf. Artif. Intell., CAI},
keywords = {Benchmarking, Continuous speech recognition, Cross-domain, Forecasting, Formant frequency, Foundation models, Learning systems, Linguistics, Multi-attributes, Multi-task learning, Paralinguistic, Performance, Speaker identification, Speaker verification, Speech communication, State of the art, Verification task},
pubstate = {published},
tppubtype = {inproceedings}
}
Zhu, Y.; Davoust, A.; Falk, T. H.
DeepSick: Deceiving Voice-Based Diagnostic Models with Synthetic Multilingual Pathological Speech Signals Article d'actes
Dans: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern., p. 69–74, Institute of Electrical and Electronics Engineers Inc., 2025, ISBN: 1062922X (ISSN); 979-833153358-8 (ISBN), (Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.).
Résumé | Liens | BibTeX | Étiquettes: COVID-19, Detection models, Diagnosis, Diagnostic model, Diagnostic systems, Generative model, Health assessments, Pathological conditions, Pathological speech signals, Scalable solution, Speech communication, Speech recognition, Speech synthesis, State of the art, Voice model
@inproceedings{zhuDeepSickDeceivingVoiceBased2025,
title = {DeepSick: Deceiving Voice-Based Diagnostic Models with Synthetic Multilingual Pathological Speech Signals},
author = {Y. Zhu and A. Davoust and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105033143787?origin=resultslist},
doi = {10.1109/SMC58881.2025.11343240},
isbn = {1062922X (ISSN); 979-833153358-8 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
pages = {69–74},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Voice-based diagnostic systems offer a scalable solution for remote health assessment. However, recent advances in generative voice models may enable malicious manipulation of voice samples to simulate or conceal disease-related speech characteristics, which poses new risks to diagnostic systems. This paper investigates the vulnerability of diagnostic and detection models to such types of "deepfake"attacks. We show that it is possible to train a generative model to convert between healthy voices and pathological ones, which in turn, can successfully deceive existing diagnostic systems. Here, focus is placed on COVID-19 infection and respiratory abnormalities, but the method can be applied across different pathological conditions affecting vocal attributes. We also benchmark four state-of-the-art synthesized voice detection models on both real and generated pathological speech from three datasets. Our results show that current synthetic voice detectors, typically trained on healthy speech data, perform poorly on generated pathological samples. While fine-tuning with real pathological voices improves detection, a substantial performance gap remains. This work provides initial insights on an emerging threat to remote voice diagnostic systems that needs further work. © 2025 IEEE.},
note = {Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
keywords = {COVID-19, Detection models, Diagnosis, Diagnostic model, Diagnostic systems, Generative model, Health assessments, Pathological conditions, Pathological speech signals, Scalable solution, Speech communication, Speech recognition, Speech synthesis, State of the art, Voice model},
pubstate = {published},
tppubtype = {inproceedings}
}
Zhu, Y.; Falk, T.
WavRx: A Disease-Agnostic, Generalizable, and Privacy-Preserving Speech Health Diagnostic Model Article de journal
Dans: IEEE Journal of Biomedical and Health Informatics, vol. 29, no 9, p. 6353–6365, 2025, ISSN: 21682194 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Agnostic, area under the curve, article, artificial neural network, asthma, autoencoder, Benchmarking, breathing, chronic obstructive lung disease, Computer-Assisted, controlled study, convolutional neural network, coronavirus disease 2019, Cross-domain, Databases, Diagnosis, Diagnostic, Diagnostic model, diagnostic test accuracy study, diagnostics, Differential privacy, Dynamics, dysarthria, Electronic health record, embedding, Embeddings, Factual, factual database, Generalizability, Health embedding, Health embeddings, Health monitoring, human, Humans, Machine learning, malignant neoplasm, model, Pathological speech, pathophysiology, physiology, pneumonia, Privacy, Privacy preserving, privacy preserving speech health diagnostic model, privacy-preserving, Privacy-preserving techniques, receiver operating characteristic, short time Fourier transform, Signal processing, speech, speech articulation, speech disorder, Speech Disorders, State of the art, temporal representation encoder, training, waveform
@article{zhuWavRxDiseaseAgnosticGeneralizable2025,
title = {WavRx: A Disease-Agnostic, Generalizable, and Privacy-Preserving Speech Health Diagnostic Model},
author = {Y. Zhu and T. Falk},
url = {https://www.scopus.com/pages/publications/85203439930?origin=resultslist},
doi = {10.1109/JBHI.2024.3454550},
issn = {21682194 (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {IEEE Journal of Biomedical and Health Informatics},
volume = {29},
number = {9},
pages = {6353–6365},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Speech is known to carry health-related attributes, which has emerged as a novel venue for remote and long-term health monitoring. However, existing models are usually tailored for a specific type of disease, and have been shown to lack generalizability across datasets. Furthermore, concerns have been raised recently towards the leakage of speaker identity from health embeddings. To mitigate these limitations, we propose WavRx, a speech health diagnostics model that captures the respiration and articulation related dynamics from a universal speech representation. Our in-domain and cross-domain experiments on six pathological speech datasets demonstrate WavRx as a new state-of-the-art health diagnostic model. Furthermore, we show that the amount of speaker identity entailed in the WavRx health embeddings is significantly reduced without extra guidance during training. An in-depth analysis of the model was performed, thus providing physiological interpretation of its improved generalizability and privacy-preserving ability. © 2013 IEEE.},
keywords = {Agnostic, area under the curve, article, artificial neural network, asthma, autoencoder, Benchmarking, breathing, chronic obstructive lung disease, Computer-Assisted, controlled study, convolutional neural network, coronavirus disease 2019, Cross-domain, Databases, Diagnosis, Diagnostic, Diagnostic model, diagnostic test accuracy study, diagnostics, Differential privacy, Dynamics, dysarthria, Electronic health record, embedding, Embeddings, Factual, factual database, Generalizability, Health embedding, Health embeddings, Health monitoring, human, Humans, Machine learning, malignant neoplasm, model, Pathological speech, pathophysiology, physiology, pneumonia, Privacy, Privacy preserving, privacy preserving speech health diagnostic model, privacy-preserving, Privacy-preserving techniques, receiver operating characteristic, short time Fourier transform, Signal processing, speech, speech articulation, speech disorder, Speech Disorders, State of the art, temporal representation encoder, training, waveform},
pubstate = {published},
tppubtype = {article}
}



