

de Recherche et d’Innovation
en Cybersécurité et Société
Moradi, A.; Zhu, Y.; Falk, T. H.
Towards Lightweight On-Device Audio Deepfake Detection Using Squeezeformers Article d'actes
Dans: K., Adi; O., Nguena Timo; N., Boulahia-Cuppens; D., Espes; N., Stakhanova; M., Omar (Ed.): Lect. Notes Comput. Sci., p. 376–389, Springer Science and Business Media Deutschland GmbH, 2026, ISBN: 03029743 (ISSN); 978-303220731-9 (ISBN), (Journal Abbreviation: Lect. Notes Comput. Sci.).
Résumé | Liens | BibTeX | Étiquettes: Audio DeepFake Detection, Detection mechanism, Edge Computing, Edge detection, Foundation models, High-accuracy, Large scale systems, Large-scale systems, Lightweight, Memory footprint, Performance, Real- time
@inproceedings{moradiLightweightOnDeviceAudio2026,
title = {Towards Lightweight On-Device Audio Deepfake Detection Using Squeezeformers},
author = {A. Moradi and Y. Zhu and T. H. Falk},
editor = {Adi K. and Nguena Timo O. and Boulahia-Cuppens N. and Espes D. and Stakhanova N. and Omar M.},
url = {https://www.scopus.com/pages/publications/105046137533?origin=resultslist},
doi = {10.1007/978-3-032-20732-6_24},
isbn = {03029743 (ISSN); 978-303220731-9 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Lect. Notes Comput. Sci.},
volume = {16295 LNCS},
pages = {376–389},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {The increasing threat of audio deepfakes necessitates detection mechanisms that can operate in real-time on resource-constrained edge devices. While large-scale systems, such as detectors based on speech foundation models, have demonstrated high accuracy, their computational and memory footprints make them ill-suited for on-device applications. This paper addresses this critical gap by investigating the key factors that influence the performance of lightweight deepfake detection models. We conduct a systematic comparison of model architectures, input feature choices, and data augmentation techniques, evaluating both deepfake detection accuracy and computational complexity across three datasets. Our findings show that with proper modeling choices, a lightweight model can achieve performance comparable to that of a much larger model while being approximately 100× smaller in size. This work provides actionable insights for developing efficient and effective audio deepfake detectors tailored for the constraints of edge computing. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2026.},
note = {Journal Abbreviation: Lect. Notes Comput. Sci.},
keywords = {Audio DeepFake Detection, Detection mechanism, Edge Computing, Edge detection, Foundation models, High-accuracy, Large scale systems, Large-scale systems, Lightweight, Memory footprint, Performance, Real- time},
pubstate = {published},
tppubtype = {inproceedings}
}
Temmar, D. E.; Hamadene, A.; Nallaguntla, V.; Fursule, A.; Allili, M. S.; Kshirsagar, S.; Avila, A. R.
Phonetic Analysis of Real and Synthetic Speech Using HuBERT Embeddings: Perspectives for Deepfake Detection Article d'actes
Dans: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern., p. 86–91, Institute of Electrical and Electronics Engineers Inc., 2025, ISBN: 1062922X (ISSN); 979-833153358-8 (ISBN), (Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.).
Résumé | Liens | BibTeX | Étiquettes: Artificial intelligence, Audio acoustics, Audio DeepFake Detection, Audio signal processing, Embeddings, Hu-BERT, KL-divergence, Linguistics, Phoneme and word Embedding, Phonetic analysis, Security systems, Self-Supervised Speech Representation, Speech analysis, Speech communication, Speech processing, Speech synthesis, Synthetic speech, Text to speech, Voice conversion
@inproceedings{temmarPhoneticAnalysisReal2025,
title = {Phonetic Analysis of Real and Synthetic Speech Using HuBERT Embeddings: Perspectives for Deepfake Detection},
author = {D. E. Temmar and A. Hamadene and V. Nallaguntla and A. Fursule and M. S. Allili and S. Kshirsagar and A. R. Avila},
url = {https://www.scopus.com/pages/publications/105033145913?origin=resultslist},
doi = {10.1109/SMC58881.2025.11343334},
isbn = {1062922X (ISSN); 979-833153358-8 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
pages = {86–91},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {The growing sophistication of speech generated by Artificial Intelligence (AI) has introduced new challenges in audio deepfake detection. Text-to-speech (TTS) and voice conversion (VC) technologies can now produce convincing synthetic speech with high quality and intelligibility. This poses a serious threat to voice biometric security systems, such as automatic speaker recognition. It also increases the risks associated to the spread of spoken disinformation, where synthetic voices can be used to disseminate malicious content. In this study, we conduct an analysis of real and synthetic speech at phonetic and word levels. For that, a parallel dataset comprising real and synthetic speech signals were developed based on a subset of the LibriSpeech ASR corpus. Synthetic speech samples were generated using two TTS and one VC systems: Coqui TTS, VITS TTS, and StarGANv2 VC. We adopted HuBERT, a self-supervised speech model, to extract speech embeddings. The motivation for using this model stems from its ability to recognize sound units corresponding to the so-called pseudo phonemes. Our analysis is based on the KL divergence (KLD) between the distributions of synthetic and real phonemes, which allowed us to rank synthetic phonemes based on their alignment with their real counterpart. We also trained several classifiers per phoneme to distinguish between real and synthetic samples. We then compute the correlations between KLD and accuracies per phoneme. Besides showing a list of phonemes that are more discriminative, our findings suggest that vowels correlate better with the classifiers' performance, suggesting that the KLD can be an indicator of the most distinguishable phonemes for deepfake detection. © 2025 IEEE.},
note = {Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
keywords = {Artificial intelligence, Audio acoustics, Audio DeepFake Detection, Audio signal processing, Embeddings, Hu-BERT, KL-divergence, Linguistics, Phoneme and word Embedding, Phonetic analysis, Security systems, Self-Supervised Speech Representation, Speech analysis, Speech communication, Speech processing, Speech synthesis, Synthetic speech, Text to speech, Voice conversion},
pubstate = {published},
tppubtype = {inproceedings}
}



