

de Recherche et d’Innovation
en Cybersécurité et Société
Ameyoud, S. Mohamed; Allili, M. Saïd
Multi-modal malware classification with hierarchical consistency and saliency-constrained adversarial training Article de journal
Dans: Journal of Information Security and Applications, vol. 99, 2026, ISSN: 22142134 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Adversarial training, Capability of detection, Classification (of information), Convolution, convolutional neural network, Convolutional neural networks, Detection system, Hierarchical consistency, Hierarchical systems, Malware, Malware classification, Malware classifications, Malware families, Malwares, Multi-modal, Multi-modal learning, Semantics, Vision transformer, Vision transformers
@article{mohamed_ameyoud_multi-modal_2026,
title = {Multi-modal malware classification with hierarchical consistency and saliency-constrained adversarial training},
author = {S. Mohamed Ameyoud and M. Saïd Allili},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-105031186108&doi=10.1016%2Fj.jisa.2026.104429&partnerID=40&md5=2425da4ab40f9043ba4e67d223a1bdd9},
doi = {10.1016/j.jisa.2026.104429},
issn = {22142134 (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Journal of Information Security and Applications},
volume = {99},
abstract = {The increasing complexity of malware, including polymorphic, obfuscated, and adversarial variants, continues to outpace the capabilities of detection systems. Here, we introduce a robust multi-modal hierarchical framework that jointly leverages visual and code-level semantics to enhance malware family and type classification. Our architecture fuses convolutional and transformer-based encoders to extract complementary representations from raw malware binaries and decompiled control-flow functions, enabling a rich, cross-modal understanding of malicious behavior. The classification pipeline follows a two-stage hierarchical protocol, where the predicted malware type informs the family-level classification. This enforces ontological consistency between type and family prediction levels. To further bolster robustness against adversarial and obfuscated malware, we integrate a novel adversarial training strategy that generates plausible perturbations guided by attention distributions. Evaluation on multiple large-scale benchmarks including BODMAS, Malimg, Microsoft BIG 2015, and a curated set of from MalwareBazaar, demonstrate that our framework consistently outperforms state-of-the-art baselines, including ResNet, Swin Transformer, and MalBERTv2, across both malware type and family prediction tasks. Notably, our model exhibits outstanding generalization to unpacked, obfuscated, and previously unseen samples, with minimal performance degradation. It achieves accuracy gains of +3-6% over leading methods and exhibits superior resilience under adversarial threat models. These results highlight the effectiveness of hierarchical conditioning, adversarial robustness, and multi-modal fusion in tackling the evolving landscape of malware. The proposed framework thus offers a scalable and generalizable approach for next-generation malware classification in real-world cybersecurity environments. © 2026 Elsevier Ltd.},
keywords = {Adversarial training, Capability of detection, Classification (of information), Convolution, convolutional neural network, Convolutional neural networks, Detection system, Hierarchical consistency, Hierarchical systems, Malware, Malware classification, Malware classifications, Malware families, Malwares, Multi-modal, Multi-modal learning, Semantics, Vision transformer, Vision transformers},
pubstate = {published},
tppubtype = {article}
}
Ameyoud, S. Mohamed; Allili, M. Saïd
Multi-modal malware classification with hierarchical consistency and saliency-constrained adversarial training Article de journal
Dans: Journal of Information Security and Applications, vol. 99, 2026, ISSN: 22142134 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Adversarial training, Capability of detection, Classification (of information), Convolution, convolutional neural network, Convolutional neural networks, Detection system, Hierarchical consistency, Hierarchical systems, Malware, Malware classification, Malware classifications, Malware families, Malwares, Multi-modal, Multi-modal learning, Semantics, Vision transformer, Vision transformers
@article{mohamedameyoudMultimodalMalwareClassification2026,
title = {Multi-modal malware classification with hierarchical consistency and saliency-constrained adversarial training},
author = {S. Mohamed Ameyoud and M. Saïd Allili},
url = {https://www.scopus.com/pages/publications/105031186108?origin=resultslist},
doi = {10.1016/j.jisa.2026.104429},
issn = {22142134 (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Journal of Information Security and Applications},
volume = {99},
publisher = {Elsevier Ltd},
abstract = {The increasing complexity of malware, including polymorphic, obfuscated, and adversarial variants, continues to outpace the capabilities of detection systems. Here, we introduce a robust multi-modal hierarchical framework that jointly leverages visual and code-level semantics to enhance malware family and type classification. Our architecture fuses convolutional and transformer-based encoders to extract complementary representations from raw malware binaries and decompiled control-flow functions, enabling a rich, cross-modal understanding of malicious behavior. The classification pipeline follows a two-stage hierarchical protocol, where the predicted malware type informs the family-level classification. This enforces ontological consistency between type and family prediction levels. To further bolster robustness against adversarial and obfuscated malware, we integrate a novel adversarial training strategy that generates plausible perturbations guided by attention distributions. Evaluation on multiple large-scale benchmarks including BODMAS, Malimg, Microsoft BIG 2015, and a curated set of from MalwareBazaar, demonstrate that our framework consistently outperforms state-of-the-art baselines, including ResNet, Swin Transformer, and MalBERTv2, across both malware type and family prediction tasks. Notably, our model exhibits outstanding generalization to unpacked, obfuscated, and previously unseen samples, with minimal performance degradation. It achieves accuracy gains of +3-6% over leading methods and exhibits superior resilience under adversarial threat models. These results highlight the effectiveness of hierarchical conditioning, adversarial robustness, and multi-modal fusion in tackling the evolving landscape of malware. The proposed framework thus offers a scalable and generalizable approach for next-generation malware classification in real-world cybersecurity environments. © 2026 Elsevier Ltd.},
keywords = {Adversarial training, Capability of detection, Classification (of information), Convolution, convolutional neural network, Convolutional neural networks, Detection system, Hierarchical consistency, Hierarchical systems, Malware, Malware classification, Malware classifications, Malware families, Malwares, Multi-modal, Multi-modal learning, Semantics, Vision transformer, Vision transformers},
pubstate = {published},
tppubtype = {article}
}
Gapski, M. C. B.; Kawai, V. A. S.; Leticio, G. R.; Valem, L. P.; Pedronette, D. C. G.; Allili, M. S.
Graph Neural Networks for Semi-Supervised Image Classification with Multi-Feature Aggregation Article de journal
Dans: Journal of the Brazilian Computer Society, vol. 32, no 1, p. 1731–1754, 2026, ISSN: 01046500 (ISSN).
Résumé | Liens | BibTeX | Étiquettes: Classification accuracy, Convolution, Convolutional neural networks, extraction, Feature extraction, Feature extractor, Feature Fusion, Feature representation, Features fusions, Graph Neural Networks, Graph representation, Graph structures, Graph theory, Graphic methods, Image classification, Labeled data, Rank Aggregation, Semi-supervised, Semi-supervised image classification, Supervised image classifications, Textures
@article{gapskiGraphNeuralNetworks2026,
title = {Graph Neural Networks for Semi-Supervised Image Classification with Multi-Feature Aggregation},
author = {M. C. B. Gapski and V. A. S. Kawai and G. R. Leticio and L. P. Valem and D. C. G. Pedronette and M. S. Allili},
url = {https://www.scopus.com/pages/publications/105045179876?origin=resultslist},
doi = {10.5753/jbcs.2026.5880},
issn = {01046500 (ISSN)},
year = {2026},
date = {2026-01-01},
journal = {Journal of the Brazilian Computer Society},
volume = {32},
number = {1},
pages = {1731–1754},
publisher = {Brazilian Computing Society},
abstract = {Feature extraction involves the identification and extraction of salient characteristics or patterns, including edges, textures, shapes, and color attributes. Contemporary feature extractors predominantly leverage deep learning architectures, such as Convolutional Neural Networks (CNNs) and Vision Transformers (VITs). The availability of diverse feature extractors in the literature provides a wide range of feature representations. Features extracted from an image depend on the specific application, the chosen extractor, and its configuration. Therefore, integrating complementary information by combining distinct extractors offers a promising way to enhance performance. Graph Neural Networks (GNNs), particularly Graph Convolutional Networks (GCNs), have emerged as powerful and widely adopted approaches for semi-supervised image classification, as they effectively leverage both labeled and unlabeled data while exploiting the underlying graph structures that capture relationships among samples. This study proposes a novel approach for GNNs in scenarios where labeled data is scarce, by integrating diverse sets of feature and graph representations derived from various extractors in classification scenarios. Experimental investigations were conducted, encompassing combinations of distinct feature and graph extractors, as well as rank aggregation strategies. The primary contributions of this work are underscored by the experimental findings, which demonstrate that the strategic combination of feature and graph representations, coupled with the application of manifold learning for graph processing, leads to significant improvements in classification accuracy across the majority of experimental conditions. Furthermore, the utilization of rank aggregation techniques to integrate features from different extractors was shown to enhance classification accuracy. © 2026, Brazilian Computing Society. All rights reserved.},
keywords = {Classification accuracy, Convolution, Convolutional neural networks, extraction, Feature extraction, Feature extractor, Feature Fusion, Feature representation, Features fusions, Graph Neural Networks, Graph representation, Graph structures, Graph theory, Graphic methods, Image classification, Labeled data, Rank Aggregation, Semi-supervised, Semi-supervised image classification, Supervised image classifications, Textures},
pubstate = {published},
tppubtype = {article}
}
Yapi, D.; Nouboukpo, A.; Allili, M. S.; Member, IEEE
Mixture of multivariate generalized Gaussians for multi-band texture modeling and representation Article de journal
Dans: Signal Processing, vol. 209, 2023, ISSN: 01651684, (Publisher: Elsevier B.V.).
Résumé | Liens | BibTeX | Étiquettes: Color texture retrieval, Content-based, Content-based color-texture retrieval, Convolution, convolutional neural network, Gaussians, Image retrieval, Image texture, Mixture of multivariate generalized gaussians, Multi-scale Decomposition, Subbands, Texture representation, Textures
@article{yapi_mixture_2023,
title = {Mixture of multivariate generalized Gaussians for multi-band texture modeling and representation},
author = {D. Yapi and A. Nouboukpo and M. S. Allili and IEEE Member},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-85151300047&doi=10.1016%2fj.sigpro.2023.109011&partnerID=40&md5=3bf98e9667eb7b60cb3f59ed1dcb029c},
doi = {10.1016/j.sigpro.2023.109011},
issn = {01651684},
year = {2023},
date = {2023-01-01},
journal = {Signal Processing},
volume = {209},
publisher = {Elsevier B.V.},
abstract = {We present a unified statistical model for multivariate and multi-modal texture representation. This model is based on the formalism of finite mixtures of multivariate generalized Gaussians (MoMGG) which enables to build a compact and accurate representation of texture images using multi-resolution texture transforms. The MoMGG model enables to describe the joint statistics of subbands in different scales and orientations, as well as between adjacent locations within the same subband, providing a precise description of the texture layout. It can also combine different multi-scale transforms to build a richer and more representative texture signature for image similarity measurement. We tested our model on both traditional texture transforms (e.g., wavelets, contourlets, maximum response filter) and convolution neural networks (CNNs) features (e.g., ResNet, SqueezeNet). Experiments on color-texture image retrieval have demonstrated the performance of our approach comparatively to state-of-the-art methods. © 2023},
note = {Publisher: Elsevier B.V.},
keywords = {Color texture retrieval, Content-based, Content-based color-texture retrieval, Convolution, convolutional neural network, Gaussians, Image retrieval, Image texture, Mixture of multivariate generalized gaussians, Multi-scale Decomposition, Subbands, Texture representation, Textures},
pubstate = {published},
tppubtype = {article}
}
Laib, L.; Allili, M. S.; Ait-Aoudia, S.
A probabilistic topic model for event-based image classification and multi-label annotation Article de journal
Dans: Signal Processing: Image Communication, vol. 76, p. 283–294, 2019, ISSN: 09235965 (ISSN), (Publisher: Elsevier B.V.).
Résumé | Liens | BibTeX | Étiquettes: Annotation performance, Classification (of information), Convolution, Convolution neural network, Convolutional neural nets, Event classification, Event recognition, Image annotation, Image Enhancement, Latent Dirichlet allocation, Multi-label annotation, Neural networks, Probabilistic topic models, Semantics, Statistics, Topic Modeling
@article{laib_probabilistic_2019,
title = {A probabilistic topic model for event-based image classification and multi-label annotation},
author = {L. Laib and M. S. Allili and S. Ait-Aoudia},
url = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-85067936924&doi=10.1016%2fj.image.2019.05.012&partnerID=40&md5=a617885b93f3a931c6b6ce1a165f940b},
doi = {10.1016/j.image.2019.05.012},
issn = {09235965 (ISSN)},
year = {2019},
date = {2019-01-01},
journal = {Signal Processing: Image Communication},
volume = {76},
pages = {283–294},
publisher = {Elsevier B.V.},
abstract = {We propose an enhanced latent topic model based on latent Dirichlet allocation and convolutional neural nets for event classification and annotation in images. Our model builds on the semantic structure relating events, objects and scenes in images. Based on initial labels extracted from convolution neural networks (CNNs), and possibly user-defined tags, we estimate the event category and final annotation of an image through a refinement process based on the expectation–maximization (EM)algorithm. The EM steps allow to progressively ascertain the class category and refine the final annotation of the image. Our model can be thought of as a two-level annotation system, where the first level derives the image event from CNN labels and image tags and the second level derives the final annotation consisting of event-related objects/scenes. Experimental results show that the proposed model yields better classification and annotation performance in the two standard datasets: UIUC-Sports and WIDER. © 2019 Elsevier B.V.},
note = {Publisher: Elsevier B.V.},
keywords = {Annotation performance, Classification (of information), Convolution, Convolution neural network, Convolutional neural nets, Event classification, Event recognition, Image annotation, Image Enhancement, Latent Dirichlet allocation, Multi-label annotation, Neural networks, Probabilistic topic models, Semantics, Statistics, Topic Modeling},
pubstate = {published},
tppubtype = {article}
}



