

de Recherche et d’Innovation
en Cybersécurité et Société
Temmar, D. E.; Hamadene, A.; Nallaguntla, V.; Fursule, A.; Allili, M. S.; Kshirsagar, S.; Avila, A. R.
Phonetic Analysis of Real and Synthetic Speech Using HuBERT Embeddings: Perspectives for Deepfake Detection Article d'actes
Dans: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern., p. 86–91, Institute of Electrical and Electronics Engineers Inc., 2025, ISBN: 1062922X (ISSN); 979-833153358-8 (ISBN), (Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.).
Résumé | Liens | BibTeX | Étiquettes: Artificial intelligence, Audio acoustics, Audio DeepFake Detection, Audio signal processing, Embeddings, Hu-BERT, KL-divergence, Linguistics, Phoneme and word Embedding, Phonetic analysis, Security systems, Self-Supervised Speech Representation, Speech analysis, Speech communication, Speech processing, Speech synthesis, Synthetic speech, Text to speech, Voice conversion
@inproceedings{temmarPhoneticAnalysisReal2025,
title = {Phonetic Analysis of Real and Synthetic Speech Using HuBERT Embeddings: Perspectives for Deepfake Detection},
author = {D. E. Temmar and A. Hamadene and V. Nallaguntla and A. Fursule and M. S. Allili and S. Kshirsagar and A. R. Avila},
url = {https://www.scopus.com/pages/publications/105033145913?origin=resultslist},
doi = {10.1109/SMC58881.2025.11343334},
isbn = {1062922X (ISSN); 979-833153358-8 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
pages = {86–91},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {The growing sophistication of speech generated by Artificial Intelligence (AI) has introduced new challenges in audio deepfake detection. Text-to-speech (TTS) and voice conversion (VC) technologies can now produce convincing synthetic speech with high quality and intelligibility. This poses a serious threat to voice biometric security systems, such as automatic speaker recognition. It also increases the risks associated to the spread of spoken disinformation, where synthetic voices can be used to disseminate malicious content. In this study, we conduct an analysis of real and synthetic speech at phonetic and word levels. For that, a parallel dataset comprising real and synthetic speech signals were developed based on a subset of the LibriSpeech ASR corpus. Synthetic speech samples were generated using two TTS and one VC systems: Coqui TTS, VITS TTS, and StarGANv2 VC. We adopted HuBERT, a self-supervised speech model, to extract speech embeddings. The motivation for using this model stems from its ability to recognize sound units corresponding to the so-called pseudo phonemes. Our analysis is based on the KL divergence (KLD) between the distributions of synthetic and real phonemes, which allowed us to rank synthetic phonemes based on their alignment with their real counterpart. We also trained several classifiers per phoneme to distinguish between real and synthetic samples. We then compute the correlations between KLD and accuracies per phoneme. Besides showing a list of phonemes that are more discriminative, our findings suggest that vowels correlate better with the classifiers' performance, suggesting that the KLD can be an indicator of the most distinguishable phonemes for deepfake detection. © 2025 IEEE.},
note = {Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
keywords = {Artificial intelligence, Audio acoustics, Audio DeepFake Detection, Audio signal processing, Embeddings, Hu-BERT, KL-divergence, Linguistics, Phoneme and word Embedding, Phonetic analysis, Security systems, Self-Supervised Speech Representation, Speech analysis, Speech communication, Speech processing, Speech synthesis, Synthetic speech, Text to speech, Voice conversion},
pubstate = {published},
tppubtype = {inproceedings}
}
Guimaraes, H. R.; Abdollahi, M.; Zhu, Y.; Maucourt, S.; Coallier, N.; Giovenazzo, P.; Falk, T. H.
Benchmarking Self-Supervised Audio Representations for IoT-Enabled Acoustic Beehive Monitoring Article de journal
Dans: IEEE Internet of Things Journal, vol. 12, no 21, p. 45000–45010, 2025, ISSN: 23274662 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Acoustics, Audio acoustics, Audio representation, Beehive monitoring, Benchmarking, Bioacoustics, Computer vision applications, Deep learning, Honeybee, honeybees, Internet of Things (IoT), IoT, Labeled data, Performance, Real time systems, Self-supervised learning, self-supervised learning (SSL), Societal benefits, Speech applications, Speech recognition, Supervised learning, Universal feature extractors
@article{guimaraesBenchmarkingSelfSupervisedAudio2025,
title = {Benchmarking Self-Supervised Audio Representations for IoT-Enabled Acoustic Beehive Monitoring},
author = {H. R. Guimaraes and M. Abdollahi and Y. Zhu and S. Maucourt and N. Coallier and P. Giovenazzo and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105013592730?origin=resultslist},
doi = {10.1109/JIOT.2025.3599483},
issn = {23274662 (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {IEEE Internet of Things Journal},
volume = {12},
number = {21},
pages = {45000–45010},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Self-supervised learning (SSL) has enabled the development of universal feature extractors that have redefined the performance envelope of computer vision and speech applications. Recent works have started to explore SSL in other domains, including bioacoustics, which could have significant societal benefits. Honeybees (Apis mellifera), for example, are crucial pollinators contributing to one-third of global food production. However, massive colony losses in recent years have raised concerns. Traditional hive monitoring methods rely on intrusive visual inspections by beekeepers, which can further disrupt colony dynamics. As such, Internet of Things (IoT)-based automated monitoring systems have emerged, integrating environmental and bioacoustic sensing to enable real-time, noninvasive hive assessment. In this work, we introduce a comprehensive evaluation and benchmarking of general-purpose and bioacoustic audio representations that generalize across various tasks in IoT-enabled acoustic beehive monitoring, even with limited labeled data. Herein, fourteen models are evaluated across four critical tasks: beehive state detection, beehive strength assessment, buzzing identification, and beekeeper voice activity detection. Reported results demonstrate the strong generalizability of existing representations, paving the way for advanced, scalable honeybee colony monitoring and preservation. © 2014 IEEE.},
keywords = {Acoustics, Audio acoustics, Audio representation, Beehive monitoring, Benchmarking, Bioacoustics, Computer vision applications, Deep learning, Honeybee, honeybees, Internet of Things (IoT), IoT, Labeled data, Performance, Real time systems, Self-supervised learning, self-supervised learning (SSL), Societal benefits, Speech applications, Speech recognition, Supervised learning, Universal feature extractors},
pubstate = {published},
tppubtype = {article}
}
Jalleli, O.; Zhu, Y.; Falk, T. H.
Audio-Visual Cross-Attention for Improved Deepfake Video Detection and Forgery Localization Article d'actes
Dans: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern., p. 75–79, Institute of Electrical and Electronics Engineers Inc., 2025, ISBN: 1062922X (ISSN); 979-833153358-8 (ISBN), (Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.).
Résumé | Liens | BibTeX | Étiquettes: Artificial intelligence, Audio acoustics, Audio signal processing, Audio-visual, Computer vision, Crossmodal attention, Deepfake detection, Forgery, Generative AI, Localisation, Multi-modal, Video detection, Video forgeries, Visual modalities
@inproceedings{jalleliAudioVisualCrossAttentionImproved2025,
title = {Audio-Visual Cross-Attention for Improved Deepfake Video Detection and Forgery Localization},
author = {O. Jalleli and Y. Zhu and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105033150623?origin=resultslist},
doi = {10.1109/SMC58881.2025.11342953},
isbn = {1062922X (ISSN); 979-833153358-8 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
pages = {75–79},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {With the emergence of multi-modal generative models, synthesized videos are becoming increasingly realistic, making the detection of deepfakes extremely challenging. While several video deepfake detection models have shown promising performance, their focus has been primarily on the visual modality. To overcome this limitation, we propose a dualstream framework that fuses visual and auditory information via cross-attention computed between embeddings extracted from pre-trained video and audio encoders. Additionally, we design a weakly-supervised forgery localization head that infers frame-level forgery scores from coarse segment-level labels, minimizing the need for fine-grained annotations and allowing for forgery location characterization. In this paper, we describe our preliminary results showing the proposed model outperforming state-of-the-art detectors on both frame-level localization and sequence-level deepfake detection tasks. Ongoing work focuses on investigating the complementarity between the visual and auditory modalities to improve model robustness and explainability. © 2025 IEEE.},
note = {Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
keywords = {Artificial intelligence, Audio acoustics, Audio signal processing, Audio-visual, Computer vision, Crossmodal attention, Deepfake detection, Forgery, Generative AI, Localisation, Multi-modal, Video detection, Video forgeries, Visual modalities},
pubstate = {published},
tppubtype = {inproceedings}
}
Abdollahi, M.; Zhu, Y.; Guimaraes, H. R.; Coallier, N.; Maucourt, S.; Giovenazzo, P.; Falk, T. H.
Audio Modulation Spectral Features for Improved Honeybee Colony Population Prediction Article de journal
Dans: IEEE Sensors Journal, vol. 25, no 24, p. 44378–44391, 2025, ISSN: 1530437X (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Apis mellifera, Audio acoustics, Audio recordings, Beehive acoustic, Beehive acoustics, Biodiversity, Chemical contamination, Climate variation, Ecology, Food security, Food supply, Honeybee, Honeybee colonies, honeybees, Modulation spectrogram, Parasite-, Population statistics, Spectral feature, Spectrograms
@article{abdollahiAudioModulationSpectral2025,
title = {Audio Modulation Spectral Features for Improved Honeybee Colony Population Prediction},
author = {M. Abdollahi and Y. Zhu and H. R. Guimaraes and N. Coallier and S. Maucourt and P. Giovenazzo and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105020704883?origin=resultslist},
doi = {10.1109/JSEN.2025.3625178},
issn = {1530437X (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {IEEE Sensors Journal},
volume = {25},
number = {24},
pages = {44378–44391},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Honeybees (Apis mellifera) are vital to agriculture and biodiversity, serving as primary pollinators for numerous crops and wild plants. However, the decline in bee populations due to factors such as pesticides, pathogens, parasites, and climate variations poses a serious threat to food security and ecological balance. This study introduces the use of modulation spectral features, extracted from beehive acoustic signals, to better predict the strength of the honeybee colony. Experiments conducted on the public urban beehive acoustics and phenotyping dataset (UrBAN), comprised of over 3000 h of beehive audio recordings, show the proposed features offering improved predictive performance compared with traditional audio features. This work underscores the potential of automated, noninvasive acoustic monitoring systems to support sustainable beekeeping and ecological preservation. © 2001-2012 IEEE.},
keywords = {Apis mellifera, Audio acoustics, Audio recordings, Beehive acoustic, Beehive acoustics, Biodiversity, Chemical contamination, Climate variation, Ecology, Food security, Food supply, Honeybee, Honeybee colonies, honeybees, Modulation spectrogram, Parasite-, Population statistics, Spectral feature, Spectrograms},
pubstate = {published},
tppubtype = {article}
}



