

de Recherche et d’Innovation
en Cybersécurité et Société
Damadi, M. S.; Davoust, A.
Fairness in social machines: a systematic review Article de journal
Dans: Journal of Information, Communication and Ethics in Society, p. 1–40, 2026, ISSN: 1477996X (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Cultural bias, Discrimination, Fairness, Gender bias, Geographic bias, Social machines
@article{damadi_fairness_2026,
title = {Fairness in social machines: a systematic review},
author = {M. S. Damadi and A. Davoust},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-105030531101&doi=10.1108%2FJICES-01-2025-0002&partnerID=40&md5=63e2f87ee3852ffe2d49c514e38cba1c},
doi = {10.1108/JICES-01-2025-0002},
issn = {1477996X (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Journal of Information, Communication and Ethics in Society},
pages = {1–40},
abstract = {Purpose – The purpose of the paper is to provide a systematic review of biases in social machines to better understand the general problem of fairness in these systems. It aims to identify and categorize phenomena described as biases toward specific demographic groups, frame them normatively as harmful and relate them to established fairness concepts originally defined for algorithmic systems. Design/methodology/approach – The phenomenon of algorithmic bias refers to systematic biases against identifiable demographic groups that occur in automated decisions systems. Such biases have mostly been studied in the context of black-box decision systems built using machine learning (ML). However, similar problems have also been reported in complex socio-technical systems such as Wikipedia and Airbnb, known more generally as social machines, where the observed biases cannot necessarily be attributed to specific automated decision systems. Instead, the biases may emerge as a result of complex processes involving numerous users and a computational infrastructure. To gain a better understanding of fairness in social machines, the authors select a representative sample of social machines from six distinct categories, and systematically review the literature reporting biases in these systems, covering 196 papers. The authors classify the reported bias phenomena, identify the affected demographic groups and relate the phenomena to established notions of harm from algorithmic fairness research. Finally, the authors identify the normative expectations of fairness associated with the different problems and discuss the applicability of existing criteria proposed for ML-driven decision systems. The analysis highlights the conceptual similarity of bias phenomena between algorithmic systems and social machines, allowing for a shared vocabulary to describe and compare phenomena across a broad class of systems. Findings – The paper identifies two key biases in social machines: representational harm, from underrepresentation or biased portrayal of disadvantaged groups, and allocative harm, from unfair decision processes, measurable via metrics like demographic parity. Gender bias is prevalent and easier to detect due to explicit markers, offering insights for identifying other biases. Unique biases arise from user categorizations, creating unintended discrimination linked to protected characteristics. These biases result from complex user interactions, not isolated algorithms. Addressing them requires redesigning social machines, focusing on computational infrastructure and interaction norms, such as visibility settings, to mitigate harmful outcomes. Originality/value – The paper’s originality lies in its systematic review of biases in social machines, offering a novel perspective on fairness in these systems. Unlike prior studies focusing solely on algorithmic fairness, this work examines the broader socio-technical interactions within social machines, identifying biases that emerge from user interactions and design choices. By linking these biases to established fairness concepts like demographic parity and representational harm, the paper bridges the gap between algorithmic fairness and social dynamics. © 2025 Emerald Publishing Limited},
keywords = {Cultural bias, Discrimination, Fairness, Gender bias, Geographic bias, Social machines},
pubstate = {published},
tppubtype = {article}
}
Shangwe, C. N.; Davoust, A.; Khoury, R.
Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities Article d'actes
Dans: K., Adi; O., Nguena Timo; N., Boulahia-Cuppens; D., Espes; N., Stakhanova; M., Omar (Ed.): Lect. Notes Comput. Sci., p. 19–34, Springer Science and Business Media Deutschland GmbH, 2026, ISBN: 03029743 (ISSN); 978-303220731-9 (ISBN), (Journal Abbreviation: Lect. Notes Comput. Sci.).
Résumé | Liens | BibTeX | Étiquettes: Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection
@inproceedings{shangweDetectionEvaluatingLLMs2026,
title = {Beyond Detection: Evaluating LLMs’ Semantic Understanding of Code Vulnerabilities},
author = {C. N. Shangwe and A. Davoust and R. Khoury},
editor = {Adi K. and Nguena Timo O. and Boulahia-Cuppens N. and Espes D. and Stakhanova N. and Omar M.},
url = {https://www.scopus.com/pages/publications/105046105995?origin=resultslist},
doi = {10.1007/978-3-032-20732-6_2},
isbn = {03029743 (ISSN); 978-303220731-9 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Lect. Notes Comput. Sci.},
volume = {16295 LNCS},
pages = {19–34},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {As Large Language Models (LLMs) become increasingly integrated into software development and analysis workflows, a critical question arises: do these models truly understand the semantics of code, or do they merely excel at pattern matching? Our goal is to assess the extent to which LLMs can back their predictions in vulnerability detection by correctly attributing the identified vulnerabilities to the violation of particular rules as proof that their decision is based on actual code semantics understanding. We employed the SVEN dataset, composed of function-level code snippets, to conduct a series of experiments that evaluate both the model’s ability to detect vulnerabilities and attribute predictions to the correct violated rule and measure LLMs’ performance under varying experimental setups. Our findings reveal that while LLMs achieve reasonable accuracy in vulnerability detection, a significant drop in performance is observed when correct rule attribution is also required, exposing a gap between perceived accuracy and actual accuracy. The difference between actual and perceived accuracy offers critical insight into the depth of code semantics understanding of LLMs in vulnerability detection. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2026.},
note = {Journal Abbreviation: Lect. Notes Comput. Sci.},
keywords = {Analysis workflow, Code Semantics, Codes (symbols), Critical questions, Development workflow, Interpretability, Language model, Large language model, Large Language Models (LLMs), Model semantics, Pattern matching, Semantics, Semantics understanding, Software design, Vulnerability detection},
pubstate = {published},
tppubtype = {inproceedings}
}
Tiwari, A.; Arrabito, R.; Davoust, A.; Falk, T. H.
Bias in Physiology-Based Cognitive State Detection for Human–Autonomy Teaming: A Comprehensive Survey Article d'actes
Dans: R.A., Sottilare; J., Schwarz (Ed.): Lect. Notes Comput. Sci., p. 139–158, Springer Science and Business Media Deutschland GmbH, 2026, ISBN: 03029743 (ISSN); 978-303230014-0 (ISBN), (Journal Abbreviation: Lect. Notes Comput. Sci.).
Résumé | Liens | BibTeX | Étiquettes: Behavioral research, Bias, Biases, Biomedical signal processing, Cognitive state, Cognitive systems, Human Autonomy Teaming, Instructional system, Intelligent systems, Learning pathway, Learning systems, Personalized learning, Physiological models, Physiological signals, Population statistics, Psychophysiology, State Detection, Student feedback, Surveying, Wearable devices, Wearable technology
@inproceedings{tiwariBiasPhysiologyBasedCognitive2026,
title = {Bias in Physiology-Based Cognitive State Detection for Human–Autonomy Teaming: A Comprehensive Survey},
author = {A. Tiwari and R. Arrabito and A. Davoust and T. H. Falk},
editor = {Sottilare R.A. and Schwarz J.},
url = {https://www.scopus.com/pages/publications/105044000677?origin=resultslist},
doi = {10.1007/978-3-032-30015-7_9},
isbn = {03029743 (ISSN); 978-303230014-0 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Lect. Notes Comput. Sci.},
volume = {16737 LNCS},
pages = {139–158},
publisher = {Springer Science and Business Media Deutschland GmbH},
abstract = {Advances in artificial intelligence (AI) are changing the role of adaptive instructional systems (AIS) from passive tools, that deliver personalized learning pathways by adapting to student feedback, to active team members and co-learners that dynamically adapt alongside human partners. In this emerging landscape of human–autonomy teaming (HAT), assessing the cognitive states of individuals interacting with highly intelligent systems has become a critical design consideration for improving team performance, collaboration, and learning outcomes. Physiological signals have emerged as a popular method for measurement of cognitive states in the past few decades. However, these signals can be strongly influenced by different user demographics, including age and biological sex. If unaccounted for, these differences could introduce systematic biases into AIS based learning pathways for different demographic groups. These biases can lead to discriminative performance of instructional systems and may also leave them vulnerable to adversarial exploitation. Aside from physiological signal-induced biases, there may also be major behavioural differences between different sex and/or age groups when interacting with and teaming with AIS, thus further confounding cognitive state monitoring. In this survey, we review emerging human-autonomy teaming studies that incorporate physiological data and analyze where biological sex- and age-related physiological or behavioral differences were documented and/or analyzed. We highlight evidence demonstrating that differences do exist during interaction with autonomous systems. We then discuss how biases can accumulate across study design and analysis modelling pipelines, and provide practical guidelines for mitigating these biases in future human-autonomy teaming research and applications. © The Author(s), under exclusive license to Springer Nature Switzerland AG 2026.},
note = {Journal Abbreviation: Lect. Notes Comput. Sci.},
keywords = {Behavioral research, Bias, Biases, Biomedical signal processing, Cognitive state, Cognitive systems, Human Autonomy Teaming, Instructional system, Intelligent systems, Learning pathway, Learning systems, Personalized learning, Physiological models, Physiological signals, Population statistics, Psychophysiology, State Detection, Student feedback, Surveying, Wearable devices, Wearable technology},
pubstate = {published},
tppubtype = {inproceedings}
}
Damadi, M. S.; Davoust, A.
Fairness in social machines: a systematic review Article de journal
Dans: Journal of Information, Communication and Ethics in Society, vol. 24, no 3, 2026, ISSN: 1477996X (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Cultural bias, Discrimination, Fairness, Gender bias, Geographic bias, Social machines
@article{damadiFairnessSocialMachines2026,
title = {Fairness in social machines: a systematic review},
author = {M. S. Damadi and A. Davoust},
url = {https://www.scopus.com/pages/publications/105030531101?origin=resultslist},
doi = {10.1108/JICES-01-2025-0002},
issn = {1477996X (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Journal of Information, Communication and Ethics in Society},
volume = {24},
number = {3},
publisher = {Emerald Publishing},
abstract = {Purpose – The purpose of the paper is to provide a systematic review of biases in social machines to better understand the general problem of fairness in these systems. It aims to identify and categorize phenomena described as biases toward specific demographic groups, frame them normatively as harmful and relate them to established fairness concepts originally defined for algorithmic systems. Design/methodology/approach – The phenomenon of algorithmic bias refers to systematic biases against identifiable demographic groups that occur in automated decisions systems. Such biases have mostly been studied in the context of black-box decision systems built using machine learning (ML). However, similar problems have also been reported in complex socio-technical systems such as Wikipedia and Airbnb, known more generally as social machines, where the observed biases cannot necessarily be attributed to specific automated decision systems. Instead, the biases may emerge as a result of complex processes involving numerous users and a computational infrastructure. To gain a better understanding of fairness in social machines, the authors select a representative sample of social machines from six distinct categories, and systematically review the literature reporting biases in these systems, covering 196 papers. The authors classify the reported bias phenomena, identify the affected demographic groups and relate the phenomena to established notions of harm from algorithmic fairness research. Finally, the authors identify the normative expectations of fairness associated with the different problems and discuss the applicability of existing criteria proposed for ML-driven decision systems. The analysis highlights the conceptual similarity of bias phenomena between algorithmic systems and social machines, allowing for a shared vocabulary to describe and compare phenomena across a broad class of systems. Findings – The paper identifies two key biases in social machines: representational harm, from underrepresentation or biased portrayal of disadvantaged groups, and allocative harm, from unfair decision processes, measurable via metrics like demographic parity. Gender bias is prevalent and easier to detect due to explicit markers, offering insights for identifying other biases. Unique biases arise from user categorizations, creating unintended discrimination linked to protected characteristics. These biases result from complex user interactions, not isolated algorithms. Addressing them requires redesigning social machines, focusing on computational infrastructure and interaction norms, such as visibility settings, to mitigate harmful outcomes. Originality/value – The paper’s originality lies in its systematic review of biases in social machines, offering a novel perspective on fairness in these systems. Unlike prior studies focusing solely on algorithmic fairness, this work examines the broader socio-technical interactions within social machines, identifying biases that emerge from user interactions and design choices. By linking these biases to established fairness concepts like demographic parity and representational harm, the paper bridges the gap between algorithmic fairness and social dynamics. © 2025 Emerald Publishing Limited},
keywords = {Cultural bias, Discrimination, Fairness, Gender bias, Geographic bias, Social machines},
pubstate = {published},
tppubtype = {article}
}
Elhajjout, A.; Jarir, Z.; Moudoud, H.; Davoust, A.; Houda, Z. A. El
Leveraging Large Language Models for Contextual Threat Hypothesis Generation in IoT Networks Article d'actes
Dans: Dig Tech Pap IEEE Int Conf Consum Electron, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 0747668X (ISSN); 979-833155343-2 (ISBN), (Journal Abbreviation: Dig Tech Pap IEEE Int Conf Consum Electron).
Résumé | Liens | BibTeX | Étiquettes: Alert Triage, Cybersecurity, Hypotheses generation, Internet of thing security, Internet of things, IoT Security, Language model, Large datasets, Large language model, large language models, Network security, Prompt Engineering, Security alerts, Security operation center, Security Operations, Security systems, Threat Hypothesis Generation
@inproceedings{elhajjoutLeveragingLargeLanguage2026,
title = {Leveraging Large Language Models for Contextual Threat Hypothesis Generation in IoT Networks},
author = {A. Elhajjout and Z. Jarir and H. Moudoud and A. Davoust and Z. A. El Houda},
url = {https://www.scopus.com/pages/publications/105037351870?origin=resultslist},
doi = {10.1109/ICCE67443.2026.11449914},
isbn = {0747668X (ISSN); 979-833155343-2 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {Dig Tech Pap IEEE Int Conf Consum Electron},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Modern Security Operations Centers face numerous challenges in managing the volume and complexity of security alerts from Internet of Things (IoT) networks. While traditional rule-based systems excel at detection, they provide limited contextual reasoning to help analysts understand ambiguous alerts. Large Language Models (LLMs) have shown promise across various cybersecurity domains. However, their potential to generate rich, explanatory threat hypotheses for ambiguous IoT alerts remains largely unexplored. To address these issues, in this paper, we present a systematic investigation of Large Language Model-based threat hypothesis generation with three key contributions. First, we introduce a context-aware structured prompting framework capable of synthesizing heterogeneous device telemetry into coherent threat narratives. Second, we formalize the assessment of explainability in security through a rigorous multidimensional quality metric system. Third, we provide the first comparative analysis of five state-of-the-art models on real-world IoT incidents, revealing critical performance tradeoffs. Results show that GPT-4o-mini achieves 88.2% accuracy on diverse attack types, while Qwen 3-235B achieves 95.2% accuracy on network-focused attacks. Statistical analysis reveals significant model-dataset interactions. These findings show that properly guided Large Language Models can augment analyst capabilities by providing detailed, contextual explanations that facilitate efficient alert triage. © 2026 IEEE.},
note = {Journal Abbreviation: Dig Tech Pap IEEE Int Conf Consum Electron},
keywords = {Alert Triage, Cybersecurity, Hypotheses generation, Internet of thing security, Internet of things, IoT Security, Language model, Large datasets, Large language model, large language models, Network security, Prompt Engineering, Security alerts, Security operation center, Security Operations, Security systems, Threat Hypothesis Generation},
pubstate = {published},
tppubtype = {inproceedings}
}
Sadfi, Y.; Amamou, H.; Davoust, A.; Avila, A. R.
Comparative Analysis of Machine Learning, LLMs, and RAG for Fake News Detection Article d'actes
Dans: IEEE Conf. Artif. Intell., CAI, p. 2128–2133, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833156039-3 (ISBN), (Journal Abbreviation: IEEE Conf. Artif. Intell., CAI).
Résumé | Liens | BibTeX | Étiquettes: Classification (of information), Comparative analyzes, Comprehensive assessment, decision making, Decision-making process, Fake detection, Generalisation, High-accuracy, Language model, Learning algorithms, Learning systems, Machine learning, Machine learning algorithms, Machine-learning, Network platforms, Performance
@inproceedings{sadfiComparativeAnalysisMachine2026,
title = {Comparative Analysis of Machine Learning, LLMs, and RAG for Fake News Detection},
author = {Y. Sadfi and H. Amamou and A. Davoust and A. R. Avila},
url = {https://www.scopus.com/pages/publications/105042025800?origin=resultslist},
doi = {10.1109/CAI68641.2026.11536343},
isbn = {979-833156039-3 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Conf. Artif. Intell., CAI},
pages = {2128–2133},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {The rapid spread of misinformation on social network platforms poses a serious threat to the decision-making processes in democratic societies. To mitigate the problem and cope with new misinformation scenarios several datasets have been proposed, followed by the emergence of new methods for detecting fake news. While several comparative analyses have been proposed to benchmark these solutions, the ever-growing development of new methods requires up-to-date analysis. For instance, recent approaches based on retrieval-augmented generation (RAG) have yet to be assessed and compared to previous methods used for fake news detection. In this study, we aim to fill this gap by presenting a comprehensive assessment of classical machine learning (ML) algorithms, Transformer-based models, such as BERT, large language models (LLMs), and RAG-based classifiers. We adopt four traditional fake news datasets, including a Portuguese one. Results indicate that classical ML and BERT deliver strong performance on in-domain tasks, demonstrating efficiency and high accuracy. RAG-augmented LLMs, on the other hand, offer enhanced generalization and robustness in cross-domain and cross-dataset settings, but are outperformed by traditional approaches for in-domain evaluation. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Conf. Artif. Intell., CAI},
keywords = {Classification (of information), Comparative analyzes, Comprehensive assessment, decision making, Decision-making process, Fake detection, Generalisation, High-accuracy, Language model, Learning algorithms, Learning systems, Machine learning, Machine learning algorithms, Machine-learning, Network platforms, Performance},
pubstate = {published},
tppubtype = {inproceedings}
}
Tiwari, A.; Arrabito, R.; Davoust, A.; Falk, T. H.
Quantifying Biological Sex Leakage in Electroencephalography-Based Mental Workload Measurement and Its Impact on Model Performance Article d'actes
Dans: IEEE Int. Conf. Hum.-Mach. Syst., ICHMS, p. 392–397, Institute of Electrical and Electronics Engineers Inc., 2026, ISBN: 979-833154511-6 (ISBN), (Journal Abbreviation: IEEE Int. Conf. Hum.-Mach. Syst., ICHMS).
Résumé | Liens | BibTeX | Étiquettes: Biomedical signal processing, Demographic information, Designing systems, Economic and social effects, Electroencephalography, Electrophysiology, Human operator, Mental workload, Modeling performance, Optimal performance, Performance, Population statistics, Real- time, Well being, Workload measurements
@inproceedings{tiwariQuantifyingBiologicalSex2026,
title = {Quantifying Biological Sex Leakage in Electroencephalography-Based Mental Workload Measurement and Its Impact on Model Performance},
author = {A. Tiwari and R. Arrabito and A. Davoust and T. H. Falk},
url = {https://www.scopus.com/pages/publications/105045575255?origin=resultslist},
doi = {10.1109/ICHMS69701.2026.11602273},
isbn = {979-833154511-6 (ISBN)},
year = {2026},
date = {2026-01-01},
booktitle = {IEEE Int. Conf. Hum.-Mach. Syst., ICHMS},
pages = {392–397},
publisher = {Institute of Electrical and Electronics Engineers Inc.},
abstract = {Detection of mental workload is essential for designing systems that can monitor and adapt to the needs of human operators, thereby ensuring their optimal performance and well-being. Features derived from neurophysiological signals, particularly electroencephalograms (EEG), provide a realtime, unobtrusive, and objective indicator of mental workload. However, EEG signals are often correlated with demographic variables, including biological sex and age, which can introduce systematic performance differences across demographic groups, while unintentionally encoding demographic information into models trained on EEG data. In this work, we investigate and quantify the influence of user demographics on mental workload detection models using fairness analysis. Furthermore, we propose domain-adversarial training, in which an adversary penalizes the model for retaining sex-related information, to mitigate leakage of private demographic information (biological sex). The proposed approach simultaneously increases the odds ratio for sex from 0.83 to 0.90 (values closer to 1 indicate reduced demographic disparity) while reducing statistical significance. This approach, however, also leads to a reduction in mental workload detection performance, with a 12.7% decrease in balanced accuracy (BACC), suggesting an inherent trade-off between performance and privacy protection. © 2026 IEEE.},
note = {Journal Abbreviation: IEEE Int. Conf. Hum.-Mach. Syst., ICHMS},
keywords = {Biomedical signal processing, Demographic information, Designing systems, Economic and social effects, Electroencephalography, Electrophysiology, Human operator, Mental workload, Modeling performance, Optimal performance, Performance, Population statistics, Real- time, Well being, Workload measurements},
pubstate = {published},
tppubtype = {inproceedings}
}
Souza, J. V.; Amamou, H.; Chen, R.; Salari, E.; Gubelmann, R.; Niklaus, C.; Serpa, T.; Lima, M. M. F.; Pinto, P. T.; Kshirsagar, S.; Davoust, A.; Handschuh, S.; Avila, A. R.
Cross-Lingual Keyword Extraction for Pesticide Terminology in Brazilian Portuguese and English Article de journal
Dans: Journal of the Brazilian Computer Society, vol. 31, no 1, p. 973–990, 2025, ISSN: 01046500 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Agriculture, BERT embedding, BERT embeddings, Cross-lingual, Embeddings, extraction, Food consumption, Keywords extraction, Labelings, Low resource languages, Multilingual extraction, Pesticides, Technical terms, Terminology, Word alignment
@article{de_souza_cross-lingual_2025,
title = {Cross-Lingual Keyword Extraction for Pesticide Terminology in Brazilian Portuguese and English},
author = {J. V. Souza and H. Amamou and R. Chen and E. Salari and R. Gubelmann and C. Niklaus and T. Serpa and M. M. F. Lima and P. T. Pinto and S. Kshirsagar and A. Davoust and S. Handschuh and A. R. Avila},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-105019700300&doi=10.5753%2Fjbcs.2025.5815&partnerID=40&md5=85ee75baf4550666a307310cd04d1c83},
doi = {10.5753/jbcs.2025.5815},
issn = {01046500 (ISSN)},
year = {2025},
date = {2025-01-01},
journal = {Journal of the Brazilian Computer Society},
volume = {31},
number = {1},
pages = {973–990},
abstract = {Agriculture plays a crucial role in Brazil’s economy. As the country intensifies its activities in the sector, the use of pesticides also increases. Hence, the risks associated with pesticide-laden food consumption have become a concern for chemistry researchers. An issue affecting regulatory standardization of pesticides in Brazil is the difficulty in translating pesticide names, particularly from English. For example, the word malathion can be translated from English to Portuguese as malatiom or malatião, resulting in inconsistent labeling. This issue extends to the broader problem of translating highly technical terms between languages, in particular for low-resource languages. In this work, we investigate terminological variation in the chemistry of organophosphorus pesticides. Our goal is to study strategies for domain-specific multilingual keyword extraction. To that end, two corpora were built based on pesticide-related scientific documents in Brazilian Portuguese and English, which led to a total of 84 and 210 texts, respectively, representing the low-and high-resource languages in this study. We then assessed 6 methods for keyword extraction: Simple Maths, TF-IDF, YAKE, TextRank, MultipartiteRank, and KeyBERT. We relied on a multilingual contextual BERT embedding to retrieve corresponding pesticide names in the target language. Finetuning was also explored to improve the multilingual representation further. Moreover, we evaluated the use of large language models (LLMs) combined with the recent retrieval-augmented generation (RAG) framework. As a result, we found that the contextual approach, combined with fine-tuning, provided the best results, contributing to enhancing Pesticide Terminology Extraction in a multilingual scenario. © 2025, Brazilian Computing Society. All rights reserved.},
keywords = {Agriculture, BERT embedding, BERT embeddings, Cross-lingual, Embeddings, extraction, Food consumption, Keywords extraction, Labelings, Low resource languages, Multilingual extraction, Pesticides, Technical terms, Terminology, Word alignment},
pubstate = {published},
tppubtype = {article}
}
Ngouanfouo, C.; Davoust, A.
Detecting Machine-Generated Text using Grammatical Features Article d'actes
Dans: Proc. Int. Conf. Tools Artif. Intell. ICTAI, p. 843–848, IEEE Computer Society, 2025, ISBN: 10823409 (ISSN); 979-833154919-0 (ISBN).
Résumé | Liens | BibTeX | Étiquettes: AI Text Detection, CNN, Computational grammars, Detection methods, Language model, Machine-generated texts, Natural language generation, Natural language processing systems, Neural encoding, Neural modelling, Part Of Speech, Part-of Speech, Speech communication, Text detection, Written texts
@inproceedings{ngouanfouo_detecting_2025,
title = {Detecting Machine-Generated Text using Grammatical Features},
author = {C. Ngouanfouo and A. Davoust},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-105031903675&doi=10.1109%2FICTAI66417.2025.00123&partnerID=40&md5=5783b8797a3425f9dfa737343ee757d2},
doi = {10.1109/ICTAI66417.2025.00123},
isbn = {10823409 (ISSN); 979-833154919-0 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Proc. Int. Conf. Tools Artif. Intell. ICTAI},
pages = {843–848},
publisher = {IEEE Computer Society},
abstract = {Large Language Models (LLMs) have advanced natural language generation but pose ethical and practical challenges, making it crucial to detect machine-generated texts. Traditional detection methods rely on complex, hard-to-interpret neural encodings and model-specific features like perplexity. This study explores whether grammatical patterns-specifically sequences of parts of speech (POS), including punctuation and symbols-can distinguish machine-written texts from human ones. Using a CNN classifier on POS sequences, the approach achieves nearly 90 % accuracy on a benchmark dataset. Combining POS-based features with neural embeddings improves performance, and the model shows robustness against adversarial attacks, though it is less effective on short texts. © 2025 IEEE.},
keywords = {AI Text Detection, CNN, Computational grammars, Detection methods, Language model, Machine-generated texts, Natural language generation, Natural language processing systems, Neural encoding, Neural modelling, Part Of Speech, Part-of Speech, Speech communication, Text detection, Written texts},
pubstate = {published},
tppubtype = {inproceedings}
}
Ngouanfouo, C.; Davoust, A.
Detecting Machine-Generated Text using Grammatical Features Article d'actes
Dans: Proc. Int. Conf. Tools Artif. Intell. ICTAI, p. 843–848, IEEE Computer Society, 2025, ISBN: 10823409 (ISSN); 979-833154919-0 (ISBN), (Journal Abbreviation: Proc. Int. Conf. Tools Artif. Intell. ICTAI).
Résumé | Liens | BibTeX | Étiquettes: AI Text Detection, CNN, Computational grammars, Detection methods, Language model, Machine-generated texts, Natural language generation, Natural language processing systems, Neural encoding, Neural modelling, Part Of Speech, Part-of Speech, Speech communication, Text detection, Written texts
@inproceedings{ngouanfouoDetectingMachineGeneratedText2025,
title = {Detecting Machine-Generated Text using Grammatical Features},
author = {C. Ngouanfouo and A. Davoust},
url = {https://www.scopus.com/pages/publications/105031903675?origin=resultslist},
doi = {10.1109/ICTAI66417.2025.00123},
isbn = {10823409 (ISSN); 979-833154919-0 (ISBN)},
year = {2025},
date = {2025-01-01},
booktitle = {Proc. Int. Conf. Tools Artif. Intell. ICTAI},
pages = {843–848},
publisher = {IEEE Computer Society},
abstract = {Large Language Models (LLMs) have advanced natural language generation but pose ethical and practical challenges, making it crucial to detect machine-generated texts. Traditional detection methods rely on complex, hard-to-interpret neural encodings and model-specific features like perplexity. This study explores whether grammatical patterns-specifically sequences of parts of speech (POS), including punctuation and symbols-can distinguish machine-written texts from human ones. Using a CNN classifier on POS sequences, the approach achieves nearly 90 % accuracy on a benchmark dataset. Combining POS-based features with neural embeddings improves performance, and the model shows robustness against adversarial attacks, though it is less effective on short texts. © 2025 IEEE.},
note = {Journal Abbreviation: Proc. Int. Conf. Tools Artif. Intell. ICTAI},
keywords = {AI Text Detection, CNN, Computational grammars, Detection methods, Language model, Machine-generated texts, Natural language generation, Natural language processing systems, Neural encoding, Neural modelling, Part Of Speech, Part-of Speech, Speech communication, Text detection, Written texts},
pubstate = {published},
tppubtype = {inproceedings}
}



