

de Recherche et d’Innovation
en Cybersécurité et Société
Shangwe, C. N.; Davoust, A.; Khoury, R.
Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities Article d'actes
Dans: K., Adi; O., Nguena Timo; N., Boulahia-Cuppens; D., Espes; N., Stakhanova; M., Omar (Ed.): Lect. Notes Comput. Sci., p. 19–34, Springer Science and Business Media Deutschland GmbH, 2026, ISBN: 03029743 (ISSN); 978-303220731-9 (ISBN), (Journal Abbreviation: Lect. Notes Comput. Sci.).
Résumé | Liens | BibTeX | Étiquettes: Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection
@inproceedings{shangweDetectionEvaluatingLLMs2026,
title = {Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities},
author = {C. N. Shangwe and A. Davoust and R. Khoury},
editor = {Adi K. and Nguena Timo O. and Boulahia-Cuppens N. and Espes D. and Stakhanova N. and Omar M.},
url = {https://www.scopus.com/pages/publications/105046105995?origin=resultslist},
doi = {10.1007/978-3-032-20732-6_2},
isbn = {03029743 (ISSN); 978-303220731-9 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Lect. Notes Comput. Sci.},
volume = {16295 LNCS},
pages = {19–34},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {As Large Language Models (LLMs) become increasingly integrated into software development and analysis workflows, a critical question arises: do these models truly understand the semantics of code, or do they merely excel at pattern matching? Our goal is to assess the extent to which LLMs can back their predictions in vulnerability detection by correctly attributing the identified vulnerabilities to the violation of particular rules as proof that their decision is based on actual code semantics understanding. We employed the SVEN dataset, composed of function-level code snippets, to conduct a series of experiments that evaluate both the model’s ability to detect vulnerabilities and attribute predictions to the correct violated rule and measure LLMs’ performance under varying experimental setups. Our findings reveal that while LLMs achieve reasonable accuracy in vulnerability detection, a significant drop in performance is observed when correct rule attribution is also required, exposing a gap between perceived accuracy and actual accuracy. The difference between actual and perceived accuracy offers critical insight into the depth of code semantics understanding of LLMs in vulnerability detection. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2026.},
note = {Journal Abbreviation: Lect. Notes Comput. Sci.},
keywords = {Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection},
pubstate = {published},
tppubtype = {inproceedings}
}



