@inproceedings{2e857bd88923483ea573c0b4db50c8b2,
title = "Nahw: A Comprehensive Benchmark of Arabic Grammar Understanding, Error Detection, Correction, and Explanation",
abstract = "Grammar comprehension is a critical capability for large language models (LLMs) to achieve fluency in a target language. In low-resource settings, such as the case with Arabic, limited availability of high-quality data can lead to significant gaps in grammatical understanding, making systematic evaluation essential. We introduce Nahw, a comprehensive benchmark for Arabic grammar that covers both theoretical knowledge and practical applications, including grammatical error detection, correction, and explanation. We evaluate a range of LLMs on these tasks and find that many models still exhibit substantial deficiencies in Arabic grammar comprehension, with GPT-4o achieving a score of 67\% on average over all tasks, while the best performing Arabic model in our experiment (ALLaM-7B) achieving 42\%. Our experiments also demonstrate that while fine-tuning with synthetic data can improve performance, it does not match the effectiveness of training on natural, high-quality data.",
author = "Hamdy Mubarak and Majd Hawasly and Abubakr Mohamed",
note = "Publisher Copyright: {\textcopyright} 2026 Association for Computational Linguistics.; 19th Conference of the European Chapter of the Association for Computational Linguistics, EACL 2026 ; Conference date: 24-03-2026 Through 29-03-2026",
year = "2026",
doi = "10.18653/v1/2026.eacl-long.296",
language = "English",
series = "EACL 2026 - 19th Conference of the European Chapter of the Association for Computational Linguistics, Proceedings of the Conference, Vol. 1 - (Long Papers)",
publisher = "Association for Computational Linguistics (ACL)",
pages = "6310--6328",
editor = "Vera Demberg and Kentaro Inui and \{Marquez Villodre\}, Lluis",
booktitle = "Long Papers",
address = "United States",
}