

de Recherche et d’Innovation
en Cybersécurité et Société
Shangwe, C. N.; Davoust, A.; Khoury, R.
Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities Article d'actes
Dans: K., Adi; O., Nguena Timo; N., Boulahia-Cuppens; D., Espes; N., Stakhanova; M., Omar (Ed.): Lect. Notes Comput. Sci., p. 19–34, Springer Science and Business Media Deutschland GmbH, 2026, ISBN: 03029743 (ISSN); 978-303220731-9 (ISBN), (Journal Abbreviation: Lect. Notes Comput. Sci.).
Résumé | Liens | BibTeX | Étiquettes: Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection
@inproceedings{shangweDetectionEvaluatingLLMs2026,
title = {Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities},
author = {C. N. Shangwe and A. Davoust and R. Khoury},
editor = {Adi K. and Nguena Timo O. and Boulahia-Cuppens N. and Espes D. and Stakhanova N. and Omar M.},
url = {https://www.scopus.com/pages/publications/105046105995?origin=resultslist},
doi = {10.1007/978-3-032-20732-6_2},
isbn = {03029743 (ISSN); 978-303220731-9 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Lect. Notes Comput. Sci.},
volume = {16295 LNCS},
pages = {19–34},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {As Large Language Models (LLMs) become increasingly integrated into software development and analysis workflows, a critical question arises: do these models truly understand the semantics of code, or do they merely excel at pattern matching? Our goal is to assess the extent to which LLMs can back their predictions in vulnerability detection by correctly attributing the identified vulnerabilities to the violation of particular rules as proof that their decision is based on actual code semantics understanding. We employed the SVEN dataset, composed of function-level code snippets, to conduct a series of experiments that evaluate both the model’s ability to detect vulnerabilities and attribute predictions to the correct violated rule and measure LLMs’ performance under varying experimental setups. Our findings reveal that while LLMs achieve reasonable accuracy in vulnerability detection, a significant drop in performance is observed when correct rule attribution is also required, exposing a gap between perceived accuracy and actual accuracy. The difference between actual and perceived accuracy offers critical insight into the depth of code semantics understanding of LLMs in vulnerability detection. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2026.},
note = {Journal Abbreviation: Lect. Notes Comput. Sci.},
keywords = {Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection},
pubstate = {published},
tppubtype = {inproceedings}
}
Elhajjout, A.; Houda, Z. A. E.; Moudoud, H.; Brik, B.; Jan, M. A.
Federated Large Language Models for A Trustworthy and Privacy-Preserving Healthcare: Applications, Challenges, and Future Research Directions Article de journal
Dans: IEEE Journal of Biomedical and Health Informatics, 2026, ISSN: 21682194 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Application research, Artificial intelligence, digital health, Distributed computer systems, Federated learning, Fine tuning, Health care, Health care application, Healthcare AI, human, Internet, Internet of medical thing, Internet of Medical Things, Language model, Large language model, large language models, Learning systems, male, Medical computing, natural language processing, Parameter-efficient fine-tuning, patient coding, Privacy, Privacy preservation, Privacy preserving, Privacy-preserving techniques, review, Sensitive data, trustworthiness
@article{elhajjoutFederatedLargeLanguage2026,
title = {Federated Large Language Models for A Trustworthy and Privacy-Preserving Healthcare: Applications, Challenges, and Future Research Directions},
author = {A. Elhajjout and Z. A. E. Houda and H. Moudoud and B. Brik and M. A. Jan},
url = {https://www.scopus.com/pages/publications/105034640126?origin=resultslist},
doi = {10.1109/JBHI.2026.3679612},
issn = {21682194 (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {IEEE Journal of Biomedical and Health Informatics},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Large-scale foundation models, especially Federated Large Language Models (FLLMs), aim to transform digital health by enabling clinically-grade natural language processing while keeping sensitive data local. However, their adoption is hindered by two main issues: (i) the computational and communication burden of parameter-rich models on resource-constrained Internet-of-Medical-Things (IoMT) devices, and (ii) performance degradation caused by Non-Independent and Identically Distributed (Non-IID) patient data. This paper presents a comprehensive survey of Federated Learning (FL) for LLMs in Healthcare (FedMed-LLMs). We review the foundations of FL and medical LLMs. Then, we present the FL-enabled LLMs applications in healthcare, and we examine their issues in terms of privacy, robustness, and trustworthiness. Finally, we present a set of core research problems and a comprehensive research agenda that identifies future directions for building robust and scalable FedMed-LLMs systems. © 2013 IEEE.},
keywords = {Application research, Artificial intelligence, digital health, Distributed computer systems, Federated learning, Fine tuning, Health care, Health care application, Healthcare AI, human, Internet, Internet of medical thing, Internet of Medical Things, Language model, Large language model, large language models, Learning systems, male, Medical computing, natural language processing, Parameter-efficient fine-tuning, patient coding, Privacy, Privacy preservation, Privacy preserving, Privacy-preserving techniques, review, Sensitive data, trustworthiness},
pubstate = {published},
tppubtype = {article}
}
Laamari, A.; Moudoud, H.; Houda, Z. A. El
Lightweight LLM Adaptation for Intrusion Detection via Token-Efficient Flow Representation Article d'actes
Dans: IEEE Conf. Artif. Intell., CAI, p. 2122–2127, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833156039-3 (ISBN), (Journal Abbreviation: IEEE Conf. Artif. Intell., CAI).
Résumé | Liens | BibTeX | Étiquettes: Classification (of information), Data flow analysis, Decoder-only large language model, Decoder-only LLMs, decoding, Flow classification, Intrusion Detection, Intrusion-Detection, Language model, Large language model, LLMs, LoRA, Low-rank adaptation, Network Flow Classification, Network intrusion, Network security, Networks flows, Qwen2.5, Signal encoding, T5-Small, Token-oriented object notation, Tokenization, TOON
@inproceedings{laamariLightweightLLMAdaptation2026,
title = {Lightweight LLM Adaptation for Intrusion Detection via Token-Efficient Flow Representation},
author = {A. Laamari and H. Moudoud and Z. A. El Houda},
url = {https://www.scopus.com/pages/publications/105042133342?origin=resultslist},
doi = {10.1109/CAI68641.2026.11536533},
isbn = {979-833156039-3 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Conf. Artif. Intell., CAI},
pages = {2122–2127},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Large language models (LLMs) are emerging as a promising approach for intrusion detection using structured network flow data. However, their practical deployment is constrained by context window limitations and the excessive token overhead introduced by conventional tabular serialization formats such as JSON. Verbose data representations inflate sequence lengths, often exceeding model input limits and causing feature truncation. Additionally, it remains unclear which LLM architecture is more suitable for structured intrusion detection tasks under limited training resources. To tackle this issue, we propose a novel representation-aware intrusion detection framework based on Token-Oriented Object Notation (TOON), a compact serialization format that maximizes token efficiency while preserving schema structure. Also, we integrate a Low-Rank Adaptation (LoRA) scheme to enable parameter-efficient fine-tuning. Finally, we evaluate the proposed framework as an encoder-decoder model (T5-Small) with a decoder-only model (Qwen2.5) on three benchmark datasets, including NSL-KDD, UNSW-NB15, and CIC-IDS2018 for binary and multi-class classification scenarios. The results show that our tokenizer-aligned, representation-aware preprocessing combined with lightweight encoder-decoder adaptation provides a practical and resource-efficient foundation for LLM-based intrusion detection. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Conf. Artif. Intell., CAI},
keywords = {Classification (of information), Data flow analysis, Decoder-only large language model, Decoder-only LLMs, decoding, Flow classification, Intrusion Detection, Intrusion-Detection, Language model, Large language model, LLMs, LoRA, Low-rank adaptation, Network Flow Classification, Network intrusion, Network security, Networks flows, Qwen2.5, Signal encoding, T5-Small, Token-oriented object notation, Tokenization, TOON},
pubstate = {published},
tppubtype = {inproceedings}
}
Elhajjout, A.; Jarir, Z.; Moudoud, H.; Davoust, A.; Houda, Z. A. El
Leveraging Large Language Models for Contextual Threat Hypothesis Generation in IoT Networks Article d'actes
Dans: Dig Tech Pap IEEE Int Conf Consum Electron, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 0747668X (ISSN); 979-833155343-2 (ISBN), (Journal Abbreviation: Dig Tech Pap IEEE Int Conf Consum Electron).
Résumé | Liens | BibTeX | Étiquettes: Alert Triage, Cybersecurity, Hypotheses generation, Internet of thing security, Internet of things, IoT Security, Language model, Large datasets, Large language model, large language models, Network security, Prompt Engineering, Security alerts, Security operation center, Security Operations, Security systems, Threat Hypothesis Generation
@inproceedings{elhajjoutLeveragingLargeLanguage2026,
title = {Leveraging Large Language Models for Contextual Threat Hypothesis Generation in IoT Networks},
author = {A. Elhajjout and Z. Jarir and H. Moudoud and A. Davoust and Z. A. El Houda},
url = {https://www.scopus.com/pages/publications/105037351870?origin=resultslist},
doi = {10.1109/ICCE67443.2026.11449914},
isbn = {0747668X (ISSN); 979-833155343-2 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Dig Tech Pap IEEE Int Conf Consum Electron},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Modern Security Operations Centers face numerous challenges in managing the volume and complexity of security alerts from Internet of Things (IoT) networks. While traditional rule-based systems excel at detection, they provide limited contextual reasoning to help analysts understand ambiguous alerts. Large Language Models (LLMs) have shown promise across various cybersecurity domains. However, their potential to generate rich, explanatory threat hypotheses for ambiguous IoT alerts remains largely unexplored. To address these issues, in this paper, we present a systematic investigation of Large Language Model-based threat hypothesis generation with three key contributions. First, we introduce a context-aware structured prompting framework capable of synthesizing heterogeneous device telemetry into coherent threat narratives. Second, we formalize the assessment of explainability in security through a rigorous multidimensional quality metric system. Third, we provide the first comparative analysis of five state-of-the-art models on real-world IoT incidents, revealing critical performance tradeoffs. Results show that GPT-4o-mini achieves 88.2% accuracy on diverse attack types, while Qwen 3-235B achieves 95.2% accuracy on network-focused attacks. Statistical analysis reveals significant model-dataset interactions. These findings show that properly guided Large Language Models can augment analyst capabilities by providing detailed, contextual explanations that facilitate efficient alert triage. © 2026 IEEE.},
note = {Journal Abbreviation: Dig Tech Pap IEEE Int Conf Consum Electron},
keywords = {Alert Triage, Cybersecurity, Hypotheses generation, Internet of thing security, Internet of things, IoT Security, Language model, Large datasets, Large language model, large language models, Network security, Prompt Engineering, Security alerts, Security operation center, Security Operations, Security systems, Threat Hypothesis Generation},
pubstate = {published},
tppubtype = {inproceedings}
}
Malasi, J. -J. M.; Moudoud, H.; Missaoui, R.
A Lightweight Multimodal LLM-Based Intrusion Detection System for Open RAN Article d'actes
Dans: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833157731-5 (ISBN), (Journal Abbreviation: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW).
Résumé | Liens | BibTeX | Étiquettes: Benchmarking, Computational linguistics, Computer crime, Cyber security, Cybersecurity, Embedded systems, Embeddings, Intelligent controllers, Internet protocols, Intrusion Detection, intrusion detection system, Intrusion Detection System (IDS), Intrusion Detection Systems, Language model, Large language model, Large Language Models (LLMs), Mobile telecommunication systems, Multi-modal, Multi-modal learning, Multimodal Learning, Network security, Open RAN, Pipelines, RAN intelligent controller, RAN intelligent controller (RIC), Semantics
@inproceedings{malasiLightweightMultimodalLLMBased2026,
title = {A Lightweight Multimodal LLM-Based Intrusion Detection System for Open RAN},
author = {J. -J. M. Malasi and H. Moudoud and R. Missaoui},
url = {https://www.scopus.com/pages/publications/105043415034?origin=resultslist},
doi = {10.1109/WCNCW67598.2026.11555393},
isbn = {979-833157731-5 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Open Radio Access Network architectures introduce unprecedented openness and programmability through RAN Intelligent Controllers, expanding the attack surface beyond traditional volumetric threats. Most existing intrusion detection systems for fifth-generation mobile networks and ORAN rely primarily on numerical Key Performance Indicators (KPIs), often overlooking the security-relevant semantics embedded in logs and control messages. This paper proposes LLM4IDS, a semantic-aware and multimodal intrusion detection system that fuses lightweight Large Language Model (LLM) embeddings of textified events with traditional tabular KPIs in a compact Transformer-based classifier. Evaluated on the realworld OpenIreland O-RAN dataset and two classical IP network benchmarks (UNSW-NB15 and CICIDS2017), LLM4IDS consistently matches or surpasses strong tabular baselines while maintaining high efficiency. On OpenIreland, it reaches nearperfect detection (F1-score $textbackslashapprox 1.0$) and yields large gains for application-layer and volumetric attacks compared to KPI-only models. Across datasets, the classifier maintains an approximately 4 MB footprint and sub-millisecond CPU inference latency; the frozen sentence-LLM encoder can be shared across tasks, and its embeddings can be cached or precomputed when required by the deployment pipeline. These results indicate that integrating semantic context through multimodal fusion can significantly enhance intrusion detection in O-RAN while remaining competitive on standard IP-based IDS tasks. To the best of our knowledge, this work is among the first to study a complete multimodal LLM-based IDS pipeline for O-RAN data with cross-domain evaluation on both O-RAN and classical IP benchmarks. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Wirel. Commun. Netw. Conf. Workshops, WCNCW},
keywords = {Benchmarking, Computational linguistics, Computer crime, Cyber security, Cybersecurity, Embedded systems, Embeddings, Intelligent controllers, Internet protocols, Intrusion Detection, intrusion detection system, Intrusion Detection System (IDS), Intrusion Detection Systems, Language model, Large language model, Large Language Models (LLMs), Mobile telecommunication systems, Multi-modal, Multi-modal learning, Multimodal Learning, Network security, Open RAN, Pipelines, RAN intelligent controller, RAN intelligent controller (RIC), Semantics},
pubstate = {published},
tppubtype = {inproceedings}
}
Moudoud, H.; Houda, Z. A. El; Brik, B.
Securing O-RAN with Zero Trust Architecture and Large Language Models Article d'actes
Dans: C., Iwendi; Z., Boulouard; N., Kryvinska (Ed.): Lect. Notes Networks Syst., p. 357–368, Springer Science and Business Media Deutschland GmbH, 2025, ISBN: 23673370 (ISSN); 978-303194619-6 (ISBN), (Journal Abbreviation: Lect. Notes Networks Syst.).
Résumé | Liens | BibTeX | Étiquettes: Access management, Access Management system, Architecture, Authentication, Block-chain, Blockchain, Computer architecture, Computer crime, Cryptography, Distributed computer systems, Intrusion Detection, Language model, Large language model, Management systems, Mobile security, Mobile telecommunication systems, Network architecture, Network security, O-RAN, Open radio access network, Radio access networks, Security systems, Security vulnerabilities, Trusted computing, Zero Trust
@inproceedings{moudoudSecuringORANZero2025,
title = {Securing O-RAN with Zero Trust Architecture and Large Language Models},
author = {H. Moudoud and Z. A. El Houda and B. Brik},
editor = {Iwendi C. and Boulouard Z. and Kryvinska N.},
url = {https://www.scopus.com/pages/publications/105011259647?origin=resultslist},
doi = {10.1007/978-3-031-94620-2_31},
isbn = {23673370 (ISSN); 978-303194619-6 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Lect. Notes Networks Syst.},
volume = {1312 LNNS},
pages = {357–368},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {The Open Radio Access Network (O-RAN) architecture is critical for the development of 6G networks, offering flexibility and interoperability through disaggregated components. However, this openness exposes O-RAN to new security vulnerabilities, including unauthorized access, data breaches, and malicious xApp deployments. To address these challenges, we propose DistillORAN, a novel Zero-Trust architecture designed specifically for O-RAN. DistillORAN features two core components: (1) a blockchain-based decentralized trust management system for secure verification, authentication, and dynamic access control of xApps, and (2) a lightweight intrusion detection module powered by DistilBERT, a transformer-based model optimized for resource-constrained environments. DistilBERT’s ability to analyze network activities and detect anomalies in real-time allows it to identify complex security threats and multi-step attack scenarios within the O-RAN ecosystem. Its lightweight nature makes it ideal for O-RAN’s distributed infrastructure, where computational resources may be limited. By combining blockchain technology for trust management with DistilBERT’s powerful pattern recognition for intrusion detection, DistillORAN enforces a Zero-Trust security model, ensuring continuous monitoring and verification of all network components. This comprehensive solution enhances the security and resilience of O-RAN networks, aligning with the dynamic needs of next-generation mobile infrastructures. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2025.},
note = {Journal Abbreviation: Lect. Notes Networks Syst.},
keywords = {Access management, Access Management system, Architecture, Authentication, Block-chain, Blockchain, Computer architecture, Computer crime, Cryptography, Distributed computer systems, Intrusion Detection, Language model, Large language model, Management systems, Mobile security, Mobile telecommunication systems, Network architecture, Network security, O-RAN, Open radio access network, Radio access networks, Security systems, Security vulnerabilities, Trusted computing, Zero Trust},
pubstate = {published},
tppubtype = {inproceedings}
}



