

de Recherche et d’Innovation
en Cybersécurité et Société
Moradi, A.; Falk, T. H.
Benchmarking Foundation Models for Cross-Domain Speaker Profiling Article d'actes
Dans: IEEE Conf. Artif. Intell., CAI, p. 78–84, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833156039-3 (ISBN), (Journal Abbreviation: IEEE Conf. Artif. Intell., CAI).
Résumé | Liens | BibTeX | Étiquettes: Benchmarking, Continuous speech recognition, Cross-domain, Forecasting, Formant frequency, Foundation models, Learning systems, Linguistics, Multi-attributes, Multi-task learning, Paralinguistic, Performance, Speaker identification, Speaker verification, Speech communication, State of the art, Verification task
@inproceedings{moradiBenchmarkingFoundationModels2026,
title = {Benchmarking Foundation Models for Cross-Domain Speaker Profiling},
author = {A. Moradi and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105042046048?origin=resultslist},
doi = {10.1109/CAI68641.2026.11536565},
isbn = {979-833156039-3 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Conf. Artif. Intell., CAI},
pages = {78–84},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Speech conveys both linguistic and paralinguistic content. While pre-trained speech foundation models have been widely explored for linguistic tasks, such as speech recognition, and for speaker identification and verification tasks, very limited work has been done to test their usefulness for multi-attribute speaker profiling, i.e., simultaneous prediction of biological sex, age, and height from speech. In this paper, we aim to benchmark the performance of four state-of-the-art self-supervised foundation models, namely, WavLM, Wav2vec2, HuBERT, and XLSR-53, under both within- and cross-domain conditions. Each model is employed as a frozen pre-trained feature extractor, with lightweight task-specific heads trained jointly in a multi-task learning framework, and their feature extraction latency is empirically analyzed to assess practical deployment considerations. Experiments on two datasets show that WavLM achieves consistently strong within- and cross-domain performance for biological sex prediction, while Wav2vec2 and XLSR-53 exhibit more consistent performance for the age and height regression tasks in cross-domain settings. Interpretability analysis based on correlations with key acoustic features show (1) WavLM internal representations correlating highly with pitch and formant frequencies, corroborating the improved performance on biological sex prediction, and (2) XLSR-53 correlating highly with the second formant frequency, corroborating the results obtained for age and height. Overall, our analysis shows that pre-trained speech foundation models could serve as useful tools for cross-domain speaker profiling tasks. While no model stood out as a clear winner across all tested physical traits, future work could explore the use of ensemble methods for improved generalizability. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Conf. Artif. Intell., CAI},
keywords = {Benchmarking, Continuous speech recognition, Cross-domain, Forecasting, Formant frequency, Foundation models, Learning systems, Linguistics, Multi-attributes, Multi-task learning, Paralinguistic, Performance, Speaker identification, Speaker verification, Speech communication, State of the art, Verification task},
pubstate = {published},
tppubtype = {inproceedings}
}
Nouboukpo, A.; Allili, M. S.
Semi-supervised flow-augmented Gaussian mixture VAEs with weighted contrastive learning for out-of-distribution detection Article de journal
Dans: Knowledge-Based Systems, vol. 347, 2026, ISSN: 09507051 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Auto encoders, Benchmarking, Classification (of information), Clusterings, Contrastive Learning, Flow based, Flow-based GMM clustering, Gaussian distribution, Gaussian-mixtures, Learning algorithms, Learning systems, Out-of-distribution detection, Semantics, Semi-supervised, Semi-supervised learning, uncertainty, Variational autoencoder, Variational autoencoders (VAEs)
@article{nouboukpoSemisupervisedFlowaugmentedGaussian2026,
title = {Semi-supervised flow-augmented Gaussian mixture VAEs with weighted contrastive learning for out-of-distribution detection},
author = {A. Nouboukpo and M. S. Allili},
url = {https://www.scopus.com/pages/publications/105040505017?origin=resultslist},
doi = {10.1016/j.knosys.2026.116312},
issn = {09507051 (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Knowledge-Based Systems},
volume = {347},
publisher = {Elsevier B.V.},
abstract = {Out-of-distribution (OOD) detection is essential for the safe deployment of machine learning systems, particularly in high-stakes domains where identifying inputs that deviate from the in-distribution (ID) is critical. However, in real-world scenarios, true OOD data are typically unavailable at training time, and only auxiliary datasets acting as imperfect proxies can be used. This makes fully supervised approaches impractical and potentially biased toward specific anomaly types. To address this limitation, we propose FLoW-ssGMVAE, a semi-supervised generative–discriminative framework that leverages limited proxy anomaly supervision together with abundant unlabeled data for robust OOD detection. Our model integrates a flow-augmented Gaussian Mixture VAE, enabling flexible latent modeling that captures nonlinear and multimodal intra-class distributions. Unlike standard VAEs with unimodal Gaussian priors, it uses class-conditional normalizing flows to better represent complex ID and OOD variability. To further refine the latent space and strengthen ID/OOD separation, our model integrates a weighted contrastive learning objective guided by sample-wise attention scores derived from likelihood uncertainty. This mechanism emphasizes ambiguous or hard-to-classify instances, reinforcing semantic boundaries and reducing false detections. By combining expressive generative modeling with uncertainty-aware discriminative training, our method constructs a structured and interpretable latent space that reliably identifies OOD samples under limited supervision. Experiments on standard benchmarks validate the effectiveness and scalability of our approach, demonstrating state-of-the-art performance with reasonable computational cost. Our implementation will be available at: FLoW-ssGMVAE. © 2026},
keywords = {Auto encoders, Benchmarking, Classification (of information), Clusterings, Contrastive Learning, Flow based, Flow-based GMM clustering, Gaussian distribution, Gaussian-mixtures, Learning algorithms, Learning systems, Out-of-distribution detection, Semantics, Semi-supervised, Semi-supervised learning, uncertainty, Variational autoencoder, Variational autoencoders (VAEs)},
pubstate = {published},
tppubtype = {article}
}
Malasi, J. -J. M.; Moudoud, H.; Missaoui, R.
A Lightweight Multimodal LLM-Based Intrusion Detection System for Open RAN Article d'actes
Dans: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833157731-5 (ISBN), (Journal Abbreviation: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW).
Résumé | Liens | BibTeX | Étiquettes: Benchmarking, Computational linguistics, Computer crime, Cyber security, Cybersecurity, Embedded systems, Embeddings, Intelligent controllers, Internet protocols, Intrusion Detection, intrusion detection system, Intrusion Detection System (IDS), Intrusion Detection Systems, Language model, Large language model, Large Language Models (LLMs), Mobile telecommunication systems, Multi-modal, Multi-modal learning, Multimodal Learning, Network security, Open RAN, Pipelines, RAN intelligent controller, RAN intelligent controller (RIC), Semantics
@inproceedings{malasiLightweightMultimodalLLMBased2026,
title = {A Lightweight Multimodal LLM-Based Intrusion Detection System for Open RAN},
author = {J. -J. M. Malasi and H. Moudoud and R. Missaoui},
url = {https://www.scopus.com/pages/publications/105043415034?origin=resultslist},
doi = {10.1109/WCNCW67598.2026.11555393},
isbn = {979-833157731-5 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Open Radio Access Network architectures introduce unprecedented openness and programmability through RAN Intelligent Controllers, expanding the attack surface beyond traditional volumetric threats. Most existing intrusion detection systems for fifth-generation mobile networks and ORAN rely primarily on numerical Key Performance Indicators (KPIs), often overlooking the security-relevant semantics embedded in logs and control messages. This paper proposes LLM4IDS, a semantic-aware and multimodal intrusion detection system that fuses lightweight Large Language Model (LLM) embeddings of textified events with traditional tabular KPIs in a compact Transformer-based classifier. Evaluated on the realworld OpenIreland O-RAN dataset and two classical IP network benchmarks (UNSW-NB15 and CICIDS2017), LLM4IDS consistently matches or surpasses strong tabular baselines while maintaining high efficiency. On OpenIreland, it reaches nearperfect detection (F1-score $textbackslashapprox 1.0$) and yields large gains for application-layer and volumetric attacks compared to KPI-only models. Across datasets, the classifier maintains an approximately 4 MB footprint and sub-millisecond CPU inference latency; the frozen sentence-LLM encoder can be shared across tasks, and its embeddings can be cached or precomputed when required by the deployment pipeline. These results indicate that integrating semantic context through multimodal fusion can significantly enhance intrusion detection in O-RAN while remaining competitive on standard IP-based IDS tasks. To the best of our knowledge, this work is among the first to study a complete multimodal LLM-based IDS pipeline for O-RAN data with cross-domain evaluation on both O-RAN and classical IP benchmarks. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW},
keywords = {Benchmarking, Computational linguistics, Computer crime, Cyber security, Cybersecurity, Embedded systems, Embeddings, Intelligent controllers, Internet protocols, Intrusion Detection, intrusion detection system, Intrusion Detection System (IDS), Intrusion Detection Systems, Language model, Large language model, Large Language Models (LLMs), Mobile telecommunication systems, Multi-modal, Multi-modal learning, Multimodal Learning, Network security, Open RAN, Pipelines, RAN intelligent controller, RAN intelligent controller (RIC), Semantics},
pubstate = {published},
tppubtype = {inproceedings}
}
Guimaraes, H. R.; Abdollahi, M.; Zhu, Y.; Maucourt, S.; Coallier, N.; Giovenazzo, P.; Falk, T. H.
Benchmarking Self-Supervised Audio Representations for IoT-Enabled Acoustic Beehive Monitoring Article de journal
Dans: IEEE Internet of Things Journal, vol. 12, no 21, p. 45000–45010, 2025, ISSN: 23274662 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Acoustics, Audio acoustics, Audio representation, Beehive monitoring, Benchmarking, Bioacoustics, Computer vision applications, Deep learning, Honeybee, honeybees, Internet of Things (IoT), IoT, Labeled data, Performance, Real time systems, Self-supervised learning, self-supervised learning (SSL), Societal benefits, Speech applications, Speech recognition, Supervised learning, Universal feature extractors
@article{guimaraesBenchmarkingSelfSupervisedAudio2025,
title = {Benchmarking Self-Supervised Audio Representations for IoT-Enabled Acoustic Beehive Monitoring},
author = {H. R. Guimaraes and M. Abdollahi and Y. Zhu and S. Maucourt and N. Coallier and P. Giovenazzo and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105013592730?origin=resultslist},
doi = {10.1109/JIOT.2025.3599483},
issn = {23274662 (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {IEEE Internet of Things Journal},
volume = {12},
number = {21},
pages = {45000–45010},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Self-supervised learning (SSL) has enabled the development of universal feature extractors that have redefined the performance envelope of computer vision and speech applications. Recent works have started to explore SSL in other domains, including bioacoustics, which could have significant societal benefits. Honeybees (Apis mellifera), for example, are crucial pollinators contributing to one-third of global food production. However, massive colony losses in recent years have raised concerns. Traditional hive monitoring methods rely on intrusive visual inspections by beekeepers, which can further disrupt colony dynamics. As such, Internet of Things (IoT)-based automated monitoring systems have emerged, integrating environmental and bioacoustic sensing to enable real-time, noninvasive hive assessment. In this work, we introduce a comprehensive evaluation and benchmarking of general-purpose and bioacoustic audio representations that generalize across various tasks in IoT-enabled acoustic beehive monitoring, even with limited labeled data. Herein, fourteen models are evaluated across four critical tasks: beehive state detection, beehive strength assessment, buzzing identification, and beekeeper voice activity detection. Reported results demonstrate the strong generalizability of existing representations, paving the way for advanced, scalable honeybee colony monitoring and preservation. © 2014 IEEE.},
keywords = {Acoustics, Audio acoustics, Audio representation, Beehive monitoring, Benchmarking, Bioacoustics, Computer vision applications, Deep learning, Honeybee, honeybees, Internet of Things (IoT), IoT, Labeled data, Performance, Real time systems, Self-supervised learning, self-supervised learning (SSL), Societal benefits, Speech applications, Speech recognition, Supervised learning, Universal feature extractors},
pubstate = {published},
tppubtype = {article}
}
Amamou, H.; Gagnon, S.; Davoust, A.; Avila, A. R.
Towards Robust Retrieval-Augmented Generation Based on Knowledge Graph: A Comparative Analysis Article d'actes
Dans: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern., p. 80–85, Institute of Electrical and Electronics Engineers Inc., 2025, ISBN: 1062922X (ISSN); 979-833153358-8 (ISBN), (Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.).
Résumé | Liens | BibTeX | Étiquettes: Benchmarking, Comparative analyzes, External sources, Generation systems, Information integration, Information retrieval, Knowledge graph, Knowledge graphs, Language model, Model response, Noise robustness, Pre-training, Prior-knowledge
@inproceedings{amamouRobustRetrievalAugmentedGeneration2025,
title = {Towards Robust Retrieval-Augmented Generation Based on Knowledge Graph: A Comparative Analysis},
author = {H. Amamou and S. Gagnon and A. Davoust and A. R. Avila},
url = {https://www.scopus.com/pages/publications/105033145604?origin=resultslist},
doi = {10.1109/SMC58881.2025.11343466},
isbn = {1062922X (ISSN); 979-833153358-8 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
pages = {80–85},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Retrieval-Augmented Generation (RAG) was first introduced to enhance the capabilities of Large Language Models (LLMs) beyond their encoded-prior knowledge. This is achieved by providing LLMs with an external source of knowledge, which helps to reduce factual hallucinations and enables the access to new information, typically not available during their pretraining phase. Despite its benefits, there is an increasing concern with the impact of inconsistent retrieved information towards LLMs' responses. Hence, the Retrieval-Augmented Generation Benchmark (RGB) was introduced as a new testbed for RAG evaluation, meant to assess the robustness of LLMs towards inconsistency in the retrieved information. In this work, we use the RGB corpus to evaluate LLMs in four scenarios: (1) noise robustness; (2) information integration; (3) negative rejection; and (4) counterfactual robustness. We perform a comparative analysis between the RAG baseline defined by the RGB and variations of GraphRAG, which is a RAG system based on a Knowledge Graph (KG) and developed to retrieve relevant information from large documents. We tested GraphRAG under three customization to improve its robustness. Our approach demonstrates improvements compared to the RGB baseline, providing insights on how to design more reliable RAG systems, tailored for real-world scenarios. © 2025 IEEE.},
note = {Journal Abbreviation: Conf. Proc. IEEE Int. Conf. Syst. Man Cybern.},
keywords = {Benchmarking, Comparative analyzes, External sources, Generation systems, Information integration, Information retrieval, Knowledge graph, Knowledge graphs, Language model, Model response, Noise robustness, Pre-training, Prior-knowledge},
pubstate = {published},
tppubtype = {inproceedings}
}
Zhu, Y.; Falk, T.
WavRx: A Disease-Agnostic, Generalizable, and Privacy-Preserving Speech Health Diagnostic Model Article de journal
Dans: IEEE Journal of Biomedical and Health Informatics, vol. 29, no 9, p. 6353–6365, 2025, ISSN: 21682194 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Agnostic, area under the curve, article, artificial neural network, asthma, autoencoder, Benchmarking, breathing, chronic obstructive lung disease, Computer-Assisted, controlled study, convolutional neural network, coronavirus disease 2019, Cross-domain, Databases, Diagnosis, Diagnostic, Diagnostic model, diagnostic test accuracy study, diagnostics, Differential privacy, Dynamics, dysarthria, Electronic health record, embedding, Embeddings, Factual, factual database, Generalizability, Health embedding, Health embeddings, Health monitoring, human, Humans, Machine learning, malignant neoplasm, model, Pathological speech, pathophysiology, physiology, pneumonia, Privacy, Privacy preserving, privacy preserving speech health diagnostic model, privacy-preserving, Privacy-preserving techniques, receiver operating characteristic, short time Fourier transform, Signal processing, speech, speech articulation, speech disorder, Speech Disorders, State of the art, temporal representation encoder, training, waveform
@article{zhuWavRxDiseaseAgnosticGeneralizable2025,
title = {WavRx: A Disease-Agnostic, Generalizable, and Privacy-Preserving Speech Health Diagnostic Model},
author = {Y. Zhu and T. Falk},
url = {https://www.scopus.com/pages/publications/85203439930?origin=resultslist},
doi = {10.1109/JBHI.2024.3454550},
issn = {21682194 (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {IEEE Journal of Biomedical and Health Informatics},
volume = {29},
number = {9},
pages = {6353–6365},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Speech is known to carry health-related attributes, which has emerged as a novel venue for remote and long-term health monitoring. However, existing models are usually tailored for a specific type of disease, and have been shown to lack generalizability across datasets. Furthermore, concerns have been raised recently towards the leakage of speaker identity from health embeddings. To mitigate these limitations, we propose WavRx, a speech health diagnostics model that captures the respiration and articulation related dynamics from a universal speech representation. Our in-domain and cross-domain experiments on six pathological speech datasets demonstrate WavRx as a new state-of-the-art health diagnostic model. Furthermore, we show that the amount of speaker identity entailed in the WavRx health embeddings is significantly reduced without extra guidance during training. An in-depth analysis of the model was performed, thus providing physiological interpretation of its improved generalizability and privacy-preserving ability. © 2013 IEEE.},
keywords = {Agnostic, area under the curve, article, artificial neural network, asthma, autoencoder, Benchmarking, breathing, chronic obstructive lung disease, Computer-Assisted, controlled study, convolutional neural network, coronavirus disease 2019, Cross-domain, Databases, Diagnosis, Diagnostic, Diagnostic model, diagnostic test accuracy study, diagnostics, Differential privacy, Dynamics, dysarthria, Electronic health record, embedding, Embeddings, Factual, factual database, Generalizability, Health embedding, Health embeddings, Health monitoring, human, Humans, Machine learning, malignant neoplasm, model, Pathological speech, pathophysiology, physiology, pneumonia, Privacy, Privacy preserving, privacy preserving speech health diagnostic model, privacy-preserving, Privacy-preserving techniques, receiver operating characteristic, short time Fourier transform, Signal processing, speech, speech articulation, speech disorder, Speech Disorders, State of the art, temporal representation encoder, training, waveform},
pubstate = {published},
tppubtype = {article}
}



