{ "Name": "Mizan", "Dialect Subsets": [ { "Name": "Iraqi Track", "Volume": 265.0, "Unit": "sentences", "Dialect": "Iraq" }, { "Name": "MSA Track", "Volume": 75.0, "Unit": "sentences", "Dialect": "Modern Standard Arabic" } ], "HF Link": "https://huggingface.co/datasets/nawaralseelawi/mizan-iraqi-arabic-benchmark", "Link": "https://huggingface.co/datasets/nawaralseelawi/mizan-iraqi-arabic-benchmark", "License": "CC BY 4.0", "Year": 2026, "Language": "ar", "Dialect": "mixed", "Source": [ "manual construction" ], "Domain": [ "politics", "culture", "society", "geography", "history", "law" ], "Form": "text", "Annotation Style": [ "human annotation", "human validation" ], "Description": "National benchmark for evaluating LLMs on Iraqi Arabic and civic context.", "Volume": 340.0, "Unit": "sentences", "Provider": [ "University of Misan" ], "Derived From": [], "Paper Title": "Mizan: A National Benchmark for Evaluating Large Language Models on Iraqi Arabic and the Iraqi Civic Context", "Paper Link": "https://arxiv.org/pdf/2609.13980.pdf", "Script": "Arab", "Tokenized": false, "Host": "other", "Access": "Free", "Cost": "", "Has Splits": false, "Partial": false, "Tasks": [ "question answering", "multiple choice question answering", "machine translation", "information extraction" ], "Venue Title": "arXiv", "Venue Type": "preprint", "Venue Name": "arXiv", "Authors": [ "Nawar S. Alseelawi", "Mustafa S. Aljumaily" ], "Affiliations": [ "University of Misan", "Missan Oil Company" ], "Abstract": "Arabic large-language-model (LLM) evaluation has matured around Modern Standard Arabic (MSA). Dialectal Arabic - the language Iraqis actually speak - remains nearly invisible to this infrastructure. We introduce Mizan (\u201dthe balance\u201d), Iraq\u2019s national benchmark for evaluating LLMs on Iraqi Arabic and the Iraqi civic context: an MSA baseline track paired with an Iraqi track across six axes - dialect comprehension, dialect generation, bidirectional MSA-Iraqi translation, Iraq-specific knowledge, official-document field extraction, and safety - built from 340 originally authored, dually reviewed items with statistically audited answer positions and Wilson intervals. A pilot evaluation of 27 systems yields four findings. The MSA track saturates while the Iraqi track discriminates. Official-document extraction confines every system to 32-56. Arabic-focused specialization behaves as MSA specialization. The safety-hardened tier of the newest model family deterministically refuses innocuous dialect-comprehension items as policy violations.", "Added By": "Zaid Alyafeai" }