{
  "generated_at": "2026-10-05",
  "count": 3302,
  "entries": [
    {
      "id": "xtts-v2",
      "name": "XTTS-v2",
      "type": "tts",
      "country": "INTL",
      "org": "Coqui",
      "license": "coqui-public-model-license",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/coqui/XTTS-v2"
      },
      "notes": "Coqui's multilingual voice cloning, Arabic support, 6s cloning",
      "metrics": {
        "downloads": 6665636,
        "likes": 3846,
        "lastModified": "2023-12-11"
      }
    },
    {
      "id": "whisper-large-v3-turbo",
      "name": "whisper-large-v3-turbo",
      "type": "asr",
      "country": "INTL",
      "org": "OpenAI",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/openai/whisper-large-v3-turbo"
      },
      "notes": "4x faster Whisper, Arabic support, MIT license",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 6295038,
        "likes": 3412,
        "lastModified": "2024-10-04"
      }
    },
    {
      "id": "openai-whisper-large-v3",
      "name": "openai/whisper-large-v3",
      "type": "asr",
      "country": "INTL",
      "org": "OpenAI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/openai/whisper-large-v3"
      },
      "notes": "Supports Arabic among many languages",
      "metrics": {
        "downloads": 4050188,
        "likes": 6561,
        "lastModified": "2024-08-12"
      }
    },
    {
      "id": "multilingual-chatterbox",
      "name": "Multilingual Chatterbox",
      "type": "tts",
      "country": "INTL",
      "org": "Resemble AI",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/ResembleAI/chatterbox"
      },
      "year": 2025,
      "notes": "Resemble AI Chatterbox multilingual TTS model covering 22 languages including Arabic.",
      "metrics": {
        "downloads": 1668253,
        "likes": 1823,
        "lastModified": "2026-06-10"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-53-arabic",
      "name": "wav2vec2-large-xlsr-53-arabic",
      "type": "asr",
      "country": "INTL",
      "org": "jonatasgrosman",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/jonatasgrosman/wav2vec2-large-xlsr-53-arabic"
      },
      "notes": "Fine-tuned on Common Voice & Arabic Speech Corpus",
      "metrics": {
        "downloads": 1514391,
        "likes": 55,
        "lastModified": "2022-12-14"
      }
    },
    {
      "id": "arabertv02",
      "name": "AraBERTv02",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-base-arabertv02"
      },
      "notes": "Improved tokenization (135M params, 12M+ downloads)",
      "metrics": {
        "downloads": 557389,
        "likes": 51,
        "lastModified": "2024-03-26"
      }
    },
    {
      "id": "phi-4",
      "name": "Phi-4",
      "type": "llm",
      "country": "INTL",
      "org": "Microsoft",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/microsoft/phi-4"
      },
      "size": "4B",
      "notes": "Multilingual with Arabic, compact & efficient",
      "metrics": {
        "downloads": 449588,
        "likes": 2316,
        "lastModified": "2026-07-14"
      }
    },
    {
      "id": "gemma4-e4b-claims-comparison",
      "name": "gemma4 e4b claims comparison",
      "type": "llm",
      "country": "INTL",
      "org": "k-chirkunov",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "nli",
        "claims-comparison"
      ],
      "links": {
        "hf": "https://huggingface.co/k-chirkunov/gemma4-e4b-claims-comparison"
      },
      "year": 2026,
      "notes": "Gemma 4 E4B LoRA fine-tune for Arabic dialectal claims comparison (NLI).",
      "base_model": [
        "google/gemma-4-e4b-it"
      ],
      "metrics": {
        "downloads": 437351,
        "likes": 2,
        "lastModified": "2026-08-05"
      }
    },
    {
      "id": "llama-3-3",
      "name": "Llama 3.3",
      "type": "llm",
      "country": "INTL",
      "org": "Meta",
      "license": "llama3.3",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct"
      },
      "size": "70B",
      "notes": "Strong Arabic performance",
      "base_model": [
        "meta-llama/llama-3.1-70b"
      ],
      "metrics": {
        "downloads": 399695,
        "likes": 3099,
        "lastModified": "2024-12-21"
      }
    },
    {
      "id": "mms-1b-all",
      "name": "MMS-1b-all",
      "type": "asr",
      "country": "INTL",
      "org": "Meta",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/mms-1b-all"
      },
      "notes": "Meta's Massively Multilingual Speech, ASR for 1100+ languages",
      "metrics": {
        "downloads": 292617,
        "likes": 207,
        "lastModified": "2023-06-15"
      }
    },
    {
      "id": "seamlessm4t-v2",
      "name": "SeamlessM4T v2",
      "type": "asr",
      "country": "INTL",
      "org": "Meta",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/seamless-m4t-v2-large"
      },
      "notes": "Meta's all-in-one ASR + translation, ~100 languages inc. Arabic",
      "metrics": {
        "downloads": 286661,
        "likes": 1024,
        "lastModified": "2024-01-04"
      }
    },
    {
      "id": "prophet-mosque-library",
      "name": "prophet mosque library",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/prophet-mosque-library"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Prophet’s Mosque Library is one of the primary resources for Islamic books.",
      "metrics": {
        "downloads": 284301,
        "likes": 6,
        "lastModified": "2025-05-14"
      }
    },
    {
      "id": "fanar-2-27b-instruct",
      "name": "Fanar-2-27B-Instruct",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/Fanar-2-27B-Instruct",
        "paper": "https://arxiv.org/abs/2603.16397"
      },
      "base_model": [
        "google/gemma-3-27b-pt"
      ],
      "size": "27B",
      "year": 2026,
      "notes": "Fanar 2 Arabic-English instruct model built on Gemma 3 27B, released March 2026.",
      "metrics": {
        "downloads": 204831,
        "likes": 16,
        "lastModified": "2026-03-25"
      }
    },
    {
      "id": "voxtral-mini",
      "name": "Voxtral Mini",
      "type": "asr",
      "country": "INTL",
      "org": "Mistral AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507"
      },
      "notes": "Mistral's speech model, 3B, Arabic support, Apache 2.0",
      "metrics": {
        "downloads": 175023,
        "likes": 675,
        "lastModified": "2025-07-28"
      }
    },
    {
      "id": "opus-mt-ar-en",
      "name": "opus-mt-ar-en",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-ar-en"
      },
      "notes": "Translation AR→EN - Helsinki-NLP (12.4M+ downloads)",
      "metrics": {
        "downloads": 132239,
        "likes": 50,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "waqfeya-library",
      "name": "Waqfeya Library",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining",
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/waqfeya-library"
      },
      "size": "10K-100K books",
      "year": 2025,
      "dialects": [
        "classical",
        "msa"
      ],
      "notes": "Book texts and PDFs from the Waqfeya Islamic library, 10K-100K files.",
      "metrics": {
        "downloads": 129453,
        "likes": 12,
        "lastModified": "2025-05-14"
      }
    },
    {
      "id": "arapoembert",
      "name": "AraPoemBERT",
      "type": "llm",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/faisalq/bert-base-arapoembert"
      },
      "year": 2024,
      "notes": "BERT model pretrained on Arabic poetry for meter, rhyme, sentiment classification.",
      "metrics": {
        "downloads": 109004,
        "likes": 2,
        "lastModified": "2024-05-23"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-sentiment",
      "name": "bert-base-arabic-camelbert-mix-sentiment",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-sentiment"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-Mix fine-tuned for Arabic sentiment analysis.",
      "metrics": {
        "downloads": 102574,
        "likes": 8,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "marbertv2",
      "name": "MARBERTv2",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/MARBERTv2"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Updated with improved dialectal coverage",
      "metrics": {
        "downloads": 102274,
        "likes": 14,
        "lastModified": "2022-03-30"
      }
    },
    {
      "id": "sentimentareng",
      "name": "SentimentArEng",
      "type": "llm",
      "country": "INTL",
      "org": "qandos0",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/qandos0/SentimentArEng"
      },
      "year": 2023,
      "notes": "Fine-tune of cardiffnlp twitter-xlm-roberta-base-sentiment for Arabic-English sentiment classification.",
      "base_model": [
        "cardiffnlp/twitter-xlm-roberta-base-sentiment"
      ],
      "metrics": {
        "downloads": 96525,
        "likes": 0,
        "lastModified": "2023-12-14"
      }
    },
    {
      "id": "shamela-waqfeya-library",
      "name": "shamela waqfeya library",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/shamela-waqfeya-library"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Shamela Waqfeya is one of the primary online resources for Islamic books, similar to Shamela.",
      "metrics": {
        "downloads": 90500,
        "likes": 4,
        "lastModified": "2025-05-14"
      }
    },
    {
      "id": "sard",
      "name": "SARD",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "cc-by-nc-nd-4.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/SARD"
      },
      "notes": "Synthetic Arabic OCR dataset for recognition training",
      "metrics": {
        "downloads": 56805,
        "likes": 14,
        "lastModified": "2026-05-20"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-da-sentiment",
      "name": "bert-base-arabic-camelbert-da-sentiment",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-da-sentiment"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-DA fine-tuned for Arabic sentiment analysis.",
      "metrics": {
        "downloads": 51077,
        "likes": 51,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "cohere-transcribe-arabic-07-2026",
      "name": "cohere-transcribe-arabic-07-2026",
      "type": "asr",
      "country": "INTL",
      "org": "Cohere Labs",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/CohereLabs/cohere-transcribe-arabic-07-2026"
      },
      "size": "2.1B",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Cohere Transcribe specialised for Arabic speech recognition.",
      "base_model": [
        "coherelabs/cohere-transcribe-03-2026"
      ],
      "metrics": {
        "downloads": 47128,
        "likes": 212,
        "lastModified": "2026-07-13"
      }
    },
    {
      "id": "opus-mt-tc-big-ar-en",
      "name": "opus mt tc big ar en",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-ar-en"
      },
      "year": 2023,
      "notes": "Neural machine translation model for translating from Arabic (ar) to English (en).",
      "metrics": {
        "downloads": 41224,
        "likes": 22,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "arabic-books",
      "name": "Arabic Books",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-books"
      },
      "size": "8.5k books",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "8,500 rows of full Arabic book texts.",
      "metrics": {
        "downloads": 33751,
        "likes": 3,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "xnli",
      "name": "XNLI",
      "type": "dataset",
      "country": "INTL",
      "org": "Facebook AI Research",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "github": "https://github.com/facebookresearch/XNLI",
        "hf": "https://huggingface.co/datasets/facebook/xnli",
        "paper": "https://arxiv.org/pdf/1809.05053.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "7,500 sentences",
      "year": 2018,
      "tags": [
        "multilingual"
      ],
      "notes": "Evaluation set for XLU by extending the development and test sets of the Multi-Genre Natural Language Inference Corpus (MultiNLI) to 15 languages,",
      "metrics": {
        "downloads": 31749,
        "likes": 73,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "shamela4-full-db",
      "name": "Shamela4 Full DB",
      "type": "dataset",
      "country": "INTL",
      "org": "AuthenticIlm",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AuthenticIlm/Shamela4_Full_DB"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "A complete extraction of al-Maktaba al-Shamela (الشاملة) v4, containing 8,589 books across 40 categories of classical Islamic sciences.",
      "metrics": {
        "downloads": 30637,
        "likes": 30,
        "lastModified": "2026-05-19"
      }
    },
    {
      "id": "wikiann",
      "name": "wikiann",
      "type": "dataset",
      "country": "INTL",
      "org": "Rensselaer Polytechnic Institute",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/unimelb-nlp/wikiann",
        "paper": "https://aclanthology.org/P17-1178.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "185,000 tokens",
      "year": 2017,
      "tags": [
        "multilingual"
      ],
      "notes": "Name tagging and linking for 282 languages from Wikipedia, with an Arabic subset.",
      "metrics": {
        "downloads": 30175,
        "likes": 125,
        "lastModified": "2024-02-22"
      }
    },
    {
      "id": "opus-mt-en-ar",
      "name": "opus-mt-en-ar",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-en-ar"
      },
      "notes": "Translation EN→AR - Helsinki-NLP (3.5M+ downloads)",
      "metrics": {
        "downloads": 28001,
        "likes": 48,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "whisper-large-v3-turbo-ar-quran",
      "name": "whisper-large-v3-turbo-ar-quran",
      "type": "asr",
      "country": "INTL",
      "org": "Naazim",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/naazimsnh02/whisper-large-v3-turbo-ar-quran"
      },
      "size": "809M",
      "year": 2025,
      "dialects": [
        "classical"
      ],
      "notes": "Whisper large-v3-turbo fine-tuned on Quran recitation.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 26512,
        "likes": 2,
        "lastModified": "2025-12-08"
      }
    },
    {
      "id": "global-mmlu",
      "name": "Global-MMLU",
      "type": "benchmark",
      "country": "INTL",
      "org": "Cohere For AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CohereForAI/Global-MMLU",
        "paper": "https://doi.org/10.18653/v1/2025.acl-long.919"
      },
      "dialects": [
        "msa"
      ],
      "size": "14,285 sentences",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Multilingual MMLU-style benchmark of 42 languages (Arabic included) with culturally sensitive and culturally agnostic subsets.",
      "metrics": {
        "downloads": 25560,
        "likes": 163,
        "lastModified": "2025-08-14"
      }
    },
    {
      "id": "flores-101",
      "name": "FLORES-101",
      "type": "benchmark",
      "country": "INTL",
      "org": "Facebook AI Research",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation"
      ],
      "links": {
        "github": "https://github.com/facebookresearch/flores",
        "hf": "https://huggingface.co/datasets/gsarti/flores_101",
        "paper": "https://arxiv.org/pdf/2106.03193.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,100,000 tokens",
      "year": 2021,
      "tags": [
        "multilingual"
      ],
      "notes": "Low-resource machine translation benchmark of 101 languages, with Arabic as one of them.",
      "metrics": {
        "downloads": 24985,
        "likes": 33,
        "lastModified": "2022-10-27"
      }
    },
    {
      "id": "qwen3-asr-arabic-uae",
      "name": "qwen3-asr-arabic-uae",
      "type": "asr",
      "country": "AE",
      "org": "Vadim Belsky",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/qwen3-asr-arabic-uae"
      },
      "size": "2B",
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Qwen3-ASR fine-tuned for Emirati Arabic.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 23661,
        "likes": 1,
        "lastModified": "2026-04-04"
      }
    },
    {
      "id": "whisper-quran",
      "name": "Whisper Quran",
      "type": "asr",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/tarteel-ai/whisper-base-ar-quran"
      },
      "dialects": [
        "classical"
      ],
      "notes": "Whisper fine-tuned for Quranic recitation recognition",
      "metrics": {
        "downloads": 23470,
        "likes": 185,
        "lastModified": "2022-12-13"
      }
    },
    {
      "id": "aya-dataset",
      "name": "Aya Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Cohere For AI Community",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CohereForAI/aya_dataset",
        "paper": "https://doi.org/10.18653/v1/2024.acl-long.620"
      },
      "dialects": [
        "mixed"
      ],
      "size": "14,250 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "The Aya Dataset is a multilingual instruction fine-tuning dataset curated by an open-science community via Aya Annotation Platform from Cohere For AI.",
      "metrics": {
        "downloads": 23214,
        "likes": 370,
        "lastModified": "2025-04-15"
      }
    },
    {
      "id": "arabic-sbert-100k",
      "name": "Arabic SBERT 100K",
      "type": "embedding",
      "country": "INTL",
      "org": "akhooli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/Arabic-SBERT-100K"
      },
      "year": 2024,
      "notes": "This is a sentence-transformers model finetuned from aubmindlab/bert-base-arabertv02.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 17433,
        "likes": 17,
        "lastModified": "2024-07-27"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-msa-ner",
      "name": "bert-base-arabic-camelbert-msa-ner",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-msa-ner"
      },
      "year": 2022,
      "dialects": [
        "msa"
      ],
      "notes": "CAMeLBERT-MSA fine-tuned for Arabic named entity recognition.",
      "metrics": {
        "downloads": 17108,
        "likes": 8,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "global-mmlu-lite",
      "name": "Global-MMLU Lite",
      "type": "dataset",
      "country": "INTL",
      "org": "Cohere For AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CohereLabs/Global-MMLU-Lite",
        "paper": "https://doi.org/10.18653/v1/2025.acl-long.919"
      },
      "dialects": [
        "msa"
      ],
      "size": "685 sentences",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Lite version of Global-MMLU with 200 culturally sensitive and 200 culturally agnostic samples per language across 16 languages, including Arabic.",
      "metrics": {
        "downloads": 15223,
        "likes": 43,
        "lastModified": "2026-06-08"
      }
    },
    {
      "id": "gate-arabert-v1",
      "name": "GATE-AraBert-v1",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/GATE-AraBert-v1"
      },
      "notes": "SOTA on MTEB Arabic STS",
      "base_model": [
        "omartificial-intelligence-space/arabic-triplet-matryoshka-v2"
      ],
      "metrics": {
        "downloads": 14758,
        "likes": 20,
        "lastModified": "2025-09-07"
      }
    },
    {
      "id": "muaalem-model-v3-2",
      "name": "muaalem model v3 2",
      "type": "asr",
      "country": "INTL",
      "org": "obadx",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/obadx/muaalem-model-v3_2"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "tags": [
        "variants:2"
      ],
      "notes": "Arabic automatic speech recognition model fine-tuned from facebook/w2v-bert-2.0 trained on obadx/muaalem-annotated-v3.",
      "base_model": [
        "facebook/w2v-bert-2.0"
      ],
      "metrics": {
        "downloads": 14381,
        "likes": 11,
        "lastModified": "2025-09-04"
      }
    },
    {
      "id": "quranic-recitation-data",
      "name": "Quranic Recitation Data",
      "type": "dataset",
      "country": "INTL",
      "org": "zaibihassan",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/zaibihassan/Quranic-Recitation-Data"
      },
      "size": "100K-1M clips",
      "year": 2026,
      "dialects": [
        "classical"
      ],
      "notes": "Large Quran recitation audio dataset with 100K-1M clips.",
      "metrics": {
        "downloads": 13953,
        "likes": 6,
        "lastModified": "2026-10-05"
      }
    },
    {
      "id": "arabic-common-voice",
      "name": "Arabic Common Voice",
      "type": "dataset",
      "country": "INTL",
      "org": "Mozilla",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://datacollective.mozillafoundation.org/datasets?q=common+voice",
        "hf": "https://huggingface.co/datasets/legacy-datasets/common_voice",
        "paper": "https://arxiv.org/pdf/1912.06670.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "85 hours",
      "year": 2020,
      "notes": "An open source, multi-language dataset of voices that anyone can use to train speech-enabled applications.",
      "metrics": {
        "downloads": 13022,
        "likes": 148,
        "lastModified": "2024-08-22"
      }
    },
    {
      "id": "f5tts-algerian-darja",
      "name": "f5tts-algerian-darja",
      "type": "tts",
      "country": "DZ",
      "org": "Touati Kamel",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/touati-kamel/f5tts-algerian-darja"
      },
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "F5-TTS fine-tuned on Algerian Darja; Algeria noted.",
      "base_model": [
        "ibrahimsalah/arabic-f5-tts-v2"
      ],
      "metrics": {
        "downloads": 12541,
        "likes": 0,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "asafaya-bert-base-arabic",
      "name": "asafaya/bert-base-arabic",
      "type": "embedding",
      "country": "INTL",
      "org": "asafaya",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/asafaya/bert-base-arabic"
      },
      "notes": "BERT-based Arabic embeddings",
      "metrics": {
        "downloads": 12443,
        "likes": 40,
        "lastModified": "2023-03-17"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-ner",
      "name": "bert-base-arabic-camelbert-mix-ner",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-ner"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-Mix fine-tuned for Arabic named entity recognition.",
      "metrics": {
        "downloads": 12165,
        "likes": 15,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "bert-base-arabertv2",
      "name": "bert-base-arabertv2",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-base-arabertv2"
      },
      "size": "136M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraBERT v2 base with Farasa-segmented input.",
      "metrics": {
        "downloads": 12118,
        "likes": 46,
        "lastModified": "2023-08-03"
      }
    },
    {
      "id": "fanar-1-9b",
      "name": "Fanar-1-9B",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/Fanar-1-9B-Instruct"
      },
      "base_model": [
        "google/gemma-2-9b"
      ],
      "size": "9B",
      "dialects": [
        "msa"
      ],
      "notes": "Arabic-English LLM",
      "metrics": {
        "downloads": 11929,
        "likes": 36,
        "lastModified": "2025-07-15"
      }
    },
    {
      "id": "falcon-arabic",
      "name": "Falcon Arabic",
      "type": "llm",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "falcon-llm-license",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/tiiuae/Falcon3-7B-Instruct"
      },
      "base_model": [
        "tiiuae/falcon3-7b-base"
      ],
      "size": "7B",
      "notes": "First Arabic model in Falcon series, top of OALL",
      "metrics": {
        "downloads": 11713,
        "likes": 80,
        "lastModified": "2025-05-31"
      }
    },
    {
      "id": "allam",
      "name": "ALLaM",
      "type": "llm",
      "country": "SA",
      "org": "SDAIA & IBM",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/ALLaM-AI/ALLaM-7B-Instruct-preview"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "7B",
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "Saudi's sovereign model, enterprise-focused",
      "metrics": {
        "downloads": 11639,
        "likes": 195,
        "lastModified": "2025-07-14"
      }
    },
    {
      "id": "allam-7b-instruct-preview",
      "name": "ALLaM-7B-Instruct-preview",
      "type": "llm",
      "country": "SA",
      "org": "HUMAIN",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/humain-ai/ALLaM-7B-Instruct-preview"
      },
      "base_model": [
        "from-scratch"
      ],
      "dialects": [
        "gulf"
      ],
      "size": "7B",
      "on_device": false,
      "year": 2025,
      "notes": "ALLaM is a series of powerful language models designed to advance Arabic Language Technology.",
      "metrics": {
        "downloads": 11639,
        "likes": 195,
        "lastModified": "2025-07-14"
      }
    },
    {
      "id": "coda-llm-data",
      "name": "coda llm data",
      "type": "dataset",
      "country": "INTL",
      "org": "mohameddalii",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohameddalii/coda-llm-data"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "This repository contains the full end-to-end dataset, fine-tuning scripts, evaluation suites, load testing harness, and proxy architecture for Coda LLM.",
      "metrics": {
        "downloads": 10702,
        "likes": 0,
        "lastModified": "2026-10-05"
      }
    },
    {
      "id": "mmmlu",
      "name": "MMMLU",
      "type": "dataset",
      "country": "INTL",
      "org": "OpenAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/openai/MMMLU"
      },
      "dialects": [
        "msa"
      ],
      "size": "14,000 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "Translated the MMLU’s test set into 14 languages using professional human translators.",
      "metrics": {
        "downloads": 10489,
        "likes": 526,
        "lastModified": "2024-10-16"
      }
    },
    {
      "id": "xstorycloze",
      "name": "XStoryCloze",
      "type": "dataset",
      "country": "INTL",
      "org": "MetaAI",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "commonsense"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/juletxara/xstory_cloze",
        "paper": "https://doi.org/10.18653/v1/2022.emnlp-main.616"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,870 sentences",
      "year": 2022,
      "tags": [
        "multilingual"
      ],
      "notes": "XStoryCloze consists of the professionally translated version of the English StoryCloze dataset (Spring 2016 version) to 10 non-English languages.",
      "metrics": {
        "downloads": 8986,
        "likes": 16,
        "lastModified": "2025-07-23"
      }
    },
    {
      "id": "massive",
      "name": "MASSIVE",
      "type": "dataset",
      "country": "INTL",
      "org": "Amazon",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "intent-detection",
        "slot-filling"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/qanastek/MASSIVE",
        "paper": "https://doi.org/10.18653/v1/2023.acl-long.235"
      },
      "dialects": [
        "msa"
      ],
      "size": "19,521 sentences",
      "year": 2022,
      "tags": [
        "multilingual"
      ],
      "notes": "1M parallel labeled virtual-assistant utterances in 51 languages, with Arabic as a subset.",
      "metrics": {
        "downloads": 8972,
        "likes": 28,
        "lastModified": "2022-12-23"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix",
      "name": "bert-base-arabic-camelbert-mix",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-Mix pretrained on MSA, dialectal and Classical Arabic.",
      "metrics": {
        "downloads": 8704,
        "likes": 19,
        "lastModified": "2021-09-14"
      }
    },
    {
      "id": "wav2vec2-quran-phonetics",
      "name": "wav2vec2 quran phonetics",
      "type": "asr",
      "country": "INTL",
      "org": "TBOGamer22",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/TBOGamer22/wav2vec2-quran-phonetics"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Unlike standard Arabic ASR systems that output orthographic Arabic text, this model outputs a phonetic (sound-level) representation, making it suitable.",
      "base_model": [
        "facebook/wav2vec2-base"
      ],
      "metrics": {
        "downloads": 8488,
        "likes": 9,
        "lastModified": "2026-01-21"
      }
    },
    {
      "id": "arabic-handwritten-ocr-4bit-qwen2-5-vl-3b-v2",
      "name": "Arabic-handwritten-OCR-4bit-Qwen2.5-VL-3B-v2",
      "type": "ocr",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/Arabic-handwritten-OCR-4bit-Qwen2.5-VL-3B-v2"
      },
      "size": "3.9B",
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "4-bit Qwen2.5-VL 3B for Arabic handwritten OCR.",
      "base_model": [
        "qwen/qwen2.5-vl-3b-instruct"
      ],
      "metrics": {
        "downloads": 8446,
        "likes": 12,
        "lastModified": "2025-12-29"
      }
    },
    {
      "id": "tydiqa",
      "name": "TYDIQA",
      "type": "dataset",
      "country": "INTL",
      "org": "Google Research",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/google-research-datasets/tydiqa",
        "hf": "https://huggingface.co/datasets/google-research-datasets/tydiqa",
        "paper": "https://aclanthology.org/2020.tacl-1.30.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "25,893 sentences",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "Question answering dataset covering 11 typologically diverse languages with 200K question-answer pairs",
      "metrics": {
        "downloads": 7550,
        "likes": 38,
        "lastModified": "2024-08-08"
      }
    },
    {
      "id": "arabic-triplet-matryoshka-v2",
      "name": "Arabic-Triplet-Matryoshka-V2",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-Triplet-Matryoshka-V2"
      },
      "notes": "Matryoshka representation",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 6958,
        "likes": 27,
        "lastModified": "2025-09-07"
      }
    },
    {
      "id": "facebook-mms-tts-ara",
      "name": "facebook/mms-tts-ara",
      "type": "tts",
      "country": "INTL",
      "org": "Meta",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/mms-tts-ara"
      },
      "on_device": true,
      "notes": "Facebook's Massively Multilingual Speech",
      "metrics": {
        "downloads": 6815,
        "likes": 21,
        "lastModified": "2023-09-01"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-53-arabic-quran-v-final",
      "name": "wav2vec2 large xlsr 53 arabic quran v final",
      "type": "asr",
      "country": "INTL",
      "org": "rabah2026",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/rabah2026/wav2vec2-large-xlsr-53-arabic-quran-v_final"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Ce modèle est une version fine-tunée de jonatasgrosman/wav2vec2-large-xlsr-53-arabic sur le dataset rabah2026/Quran-Ayah-Corpus.",
      "metrics": {
        "downloads": 6448,
        "likes": 7,
        "lastModified": "2025-12-21"
      }
    },
    {
      "id": "xquad",
      "name": "xquad",
      "type": "benchmark",
      "country": "INTL",
      "org": "HiTZ Center",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "github": "https://github.com/deepmind/xquad",
        "hf": "https://huggingface.co/datasets/google/xquad",
        "paper": "https://aclanthology.org/2020.acl-main.421.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,190 documents",
      "year": 2019,
      "tags": [
        "multilingual"
      ],
      "notes": "Cross-lingual QA benchmark of 240 SQuAD paragraphs and 1190 question-answer pairs translated into ten languages including Arabic.",
      "metrics": {
        "downloads": 6365,
        "likes": 42,
        "lastModified": "2024-01-04"
      }
    },
    {
      "id": "arabic-sts-matryoshka-v2",
      "name": "Arabic-STS-Matryoshka-V2",
      "type": "embedding",
      "country": "EG",
      "org": "Omar Elshehy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/omarelshehy/Arabic-STS-Matryoshka-V2"
      },
      "size": "135M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic STS Matryoshka embedding model.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 6240,
        "likes": 4,
        "lastModified": "2024-12-29"
      }
    },
    {
      "id": "jais-2-8b-chat",
      "name": "Jais 2 8B Chat",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/Jais-2-8B-Chat",
        "paper": "https://arxiv.org/abs/2608.13580"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "8B",
      "year": 2025,
      "notes": "Smaller Jais 2 Arabic-English chat model with GGUF builds.",
      "metrics": {
        "downloads": 5822,
        "likes": 28,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "nemotron-3-5-quran-dual-v5",
      "name": "nemotron 3.5 quran dual v5",
      "type": "asr",
      "country": "INTL",
      "org": "tamm5y5m5",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/tamm5y5m5/nemotron-3.5-quran-dual-v5"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Nemotron 3.5 streaming ASR (0.6B) fine-tuned for full-Quran recognition with tajweed marks; CER 2.76%.",
      "size": "0.6B",
      "on_device": true,
      "metrics": {
        "downloads": 5805,
        "likes": 2,
        "lastModified": "2026-09-02"
      }
    },
    {
      "id": "universal-dependencies",
      "name": "Universal Dependencies",
      "type": "dataset",
      "country": "INTL",
      "org": "Universal Dependencies(UD)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos",
        "morphology"
      ],
      "links": {
        "website": "https://github.com/UniversalDependencies",
        "hf": "https://huggingface.co/datasets/universal-dependencies/universal_dependencies"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,042,000 sentences",
      "year": 2020,
      "notes": "Universal Dependencies is a project that seeks to develop cross-linguistically consistent treebank annotation for many languages.",
      "metrics": {
        "downloads": 5742,
        "likes": 8,
        "lastModified": "2026-09-18"
      }
    },
    {
      "id": "quranic-universal-ayahs",
      "name": "quranic universal ayahs",
      "type": "dataset",
      "country": "INTL",
      "org": "QUD-Technologies",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "alignment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QUD-Technologies/quranic-universal-ayahs"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Qur'anic ayah recitation audio with forced-alignment word and letter timestamps for ASR and segmentation.",
      "metrics": {
        "downloads": 5584,
        "likes": 7,
        "lastModified": "2026-10-04"
      }
    },
    {
      "id": "lahgtna-v2",
      "name": "Dialectal Arabic Lahgtna v2",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/dialectal-arabic-lahgtna-v2"
      },
      "size": "3000h",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Multi-dialect Arabic speech across 13 dialects, 3,000+ hours.",
      "metrics": {
        "downloads": 5530,
        "likes": 29,
        "lastModified": "2026-08-04"
      }
    },
    {
      "id": "audar-asr-v1-turbo",
      "name": "Audar ASR V1 Turbo",
      "type": "asr",
      "country": "INTL",
      "org": "Audar AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/audarai/Audar-ASR-V1-Turbo"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Audar's Arabic-first speech-recognition model — leaderboard-grade, dialect-aware.",
      "metrics": {
        "downloads": 5469,
        "likes": 17,
        "lastModified": "2026-08-20"
      }
    },
    {
      "id": "arabic-pp-ocrv5-mobile-rec",
      "name": "arabic_PP-OCRv5_mobile_rec",
      "type": "ocr",
      "country": "INTL",
      "org": "PaddlePaddle",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/PaddlePaddle/arabic_PP-OCRv5_mobile_rec"
      },
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "PP-OCRv5 mobile recognition model for Arabic script.",
      "metrics": {
        "downloads": 5105,
        "likes": 7,
        "lastModified": "2025-10-16"
      }
    },
    {
      "id": "quranic-translation-audio-data",
      "name": "Quranic Translation Audio Data",
      "type": "dataset",
      "country": "INTL",
      "org": "zaibihassan",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/zaibihassan/Quranic-Translation-Audio-Data"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Complete Quran translation audio and commentary recitations across 51 translation directories in 14 languages.",
      "metrics": {
        "downloads": 5082,
        "likes": 1,
        "lastModified": "2026-10-05"
      }
    },
    {
      "id": "arabic-retrieval-v1-0",
      "name": "Arabic-Retrieval-v1.0",
      "type": "embedding",
      "country": "EG",
      "org": "Omar Elshehy",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/omarelshehy/Arabic-Retrieval-v1.0"
      },
      "size": "135M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic retrieval embedding model.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 5061,
        "likes": 6,
        "lastModified": "2024-12-19"
      }
    },
    {
      "id": "prophet-mosque-library-compressed",
      "name": "prophet mosque library compressed",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/prophet-mosque-library-compressed"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Prophet’s Mosque Library is one of the primary resources for Islamic books.",
      "metrics": {
        "downloads": 4957,
        "likes": 0,
        "lastModified": "2025-05-08"
      }
    },
    {
      "id": "quranic-word-by-word-audio-data",
      "name": "Quranic Word By Word Audio Data",
      "type": "dataset",
      "country": "INTL",
      "org": "zaibihassan",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/zaibihassan/Quranic-Word-By-Word-Audio-Data"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Two word-by-word Quran recitation sets, Muallim (teacher style) and Mujawwad (tajweed style), for ASR and TTS.",
      "metrics": {
        "downloads": 4876,
        "likes": 4,
        "lastModified": "2026-05-23"
      }
    },
    {
      "id": "waqfeya-library-compressed",
      "name": "waqfeya library compressed",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/waqfeya-library-compressed"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Waqfeya is one of the primary online resources for Islamic books, similar to Shamela.",
      "metrics": {
        "downloads": 4875,
        "likes": 6,
        "lastModified": "2025-04-25"
      }
    },
    {
      "id": "arwiki",
      "name": "arwiki",
      "type": "dataset",
      "country": "INTL",
      "org": "CALM",
      "license": "['unknown']",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CALM/arwiki"
      },
      "dialects": [
        "msa"
      ],
      "size": "10M–100M rows",
      "year": 2022,
      "notes": "This dataset is extracted using wikiextractor tool, from Wikipedia Arabic pages.",
      "metrics": {
        "downloads": 4509,
        "likes": 5,
        "lastModified": "2022-08-01"
      }
    },
    {
      "id": "bert-base-arabertv02-twitter",
      "name": "bert-base-arabertv02-twitter",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-base-arabertv02-twitter"
      },
      "size": "135M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "mixed"
      ],
      "notes": "AraBERT v02 base further pretrained on tweets including dialects.",
      "metrics": {
        "downloads": 4198,
        "likes": 8,
        "lastModified": "2023-03-23"
      }
    },
    {
      "id": "miracl",
      "name": "MIRACL",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/miracl/miracl-corpus"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,061,414 documents",
      "year": 2022,
      "tags": [
        "multilingual"
      ],
      "notes": "MIRACL multilingual retrieval corpus spanning 18 languages, with Arabic as one documented subset.",
      "metrics": {
        "downloads": 4116,
        "likes": 54,
        "lastModified": "2023-01-05"
      }
    },
    {
      "id": "everyayah-tarteel",
      "name": "Tarteel EveryAyah",
      "type": "dataset",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarteel-ai/everyayah"
      },
      "year": 2022,
      "dialects": [
        "classical"
      ],
      "notes": "Official Tarteel EveryAyah dataset of Quran recitations by multiple reciters.",
      "metrics": {
        "downloads": 4016,
        "likes": 42,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "ontonotes-5-0",
      "name": "OntoNotes 5.0",
      "type": "dataset",
      "country": "INTL",
      "org": "Raytheon BBN Technologies",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "ner",
        "coreference"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2013T19",
        "hf": "https://huggingface.co/datasets/ontonotes/conll2012_ontonotesv5",
        "paper": "https://catalog.ldc.upenn.edu/docs/LDC2013T19/OntoNotes-Release-5.0.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "300,000 tokens",
      "year": 2012,
      "tags": [
        "multilingual",
        "variants:2"
      ],
      "notes": "Arabic portion of OntoNotes 5.0, with 300K words of newswire annotated by LDC.",
      "metrics": {
        "downloads": 3958,
        "likes": 46,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "arbertv2",
      "name": "ARBERTv2",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/ARBERTv2"
      },
      "size": "164M",
      "year": 2023,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "ARBERT v2, MSA BERT from UBC.",
      "metrics": {
        "downloads": 3906,
        "likes": 9,
        "lastModified": "2024-04-24"
      }
    },
    {
      "id": "arat5",
      "name": "AraT5",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/AraT5-base"
      },
      "notes": "T5 for Arabic summarization, translation, paraphrasing",
      "metrics": {
        "downloads": 3854,
        "likes": 22,
        "lastModified": "2024-05-16"
      }
    },
    {
      "id": "qurantts",
      "name": "QuranTTS",
      "type": "dataset",
      "country": "INTL",
      "org": "Quran Lab",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "restoration"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Quran-Lab/QuranTTS"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "QuranTTS v4: ear-verified Quranic recitation corpus at 24-bit/48 kHz for TTS and speech restoration; non-commercial.",
      "metrics": {
        "downloads": 3830,
        "likes": 4,
        "lastModified": "2026-09-21"
      }
    },
    {
      "id": "arabic-stem-lexicon",
      "name": "arabic stem lexicon",
      "type": "dataset",
      "country": "INTL",
      "org": "TigreGotico",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "tts",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TigreGotico/arabic-stem-lexicon"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "An undiacritized Arabic surface form → its most frequent diacritized stem.",
      "metrics": {
        "downloads": 3773,
        "likes": 0,
        "lastModified": "2026-07-14"
      }
    },
    {
      "id": "tunisianmmlu",
      "name": "TunisianMMLU",
      "type": "benchmark",
      "country": "INTL",
      "org": "LINAGORA",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/linagora/TunisianMMLU"
      },
      "size": "10K-100K rows",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "MMLU translated into Tunisian Derja, LINAGORA (France); evaluated with lighteval.",
      "metrics": {
        "downloads": 3724,
        "likes": 0,
        "lastModified": "2026-09-21"
      }
    },
    {
      "id": "whisper-large-v3-ar",
      "name": "whisper large v3 ar",
      "type": "asr",
      "country": "INTL",
      "org": "Dr-AliGomaa",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/Dr-AliGomaa/whisper-large-v3-ar"
      },
      "dialects": [
        "egy",
        "classical",
        "msa"
      ],
      "year": 2026,
      "notes": "openai/whisper-large-v3 fine-tuned for Quran and Hadith transcription without giving up general Modern Standard Arabic.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 3719,
        "likes": 0,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-msa",
      "name": "bert base arabic camelbert msa",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-msa"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2021,
      "notes": "We release pre-trained language models for Modern Standard Arabic (MSA), dialectal Arabic (DA), and classical Arabic.",
      "metrics": {
        "downloads": 3648,
        "likes": 10,
        "lastModified": "2021-09-14"
      }
    },
    {
      "id": "arabicmmlu",
      "name": "ArabicMMLU",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ArabicMMLU"
      },
      "notes": "Multi-task language understanding from school exams",
      "metrics": {
        "downloads": 3623,
        "likes": 39,
        "lastModified": "2024-09-17"
      }
    },
    {
      "id": "common-voice-arabic",
      "name": "Common Voice Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "Mozilla Foundation",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mozilla-foundation/common_voice_17_0",
        "website": "https://commonvoice.mozilla.org/ar/datasets"
      },
      "year": 2017,
      "dialects": [
        "mixed"
      ],
      "notes": "Crowdsourced Arabic read speech in Common Voice, standard ASR training and evaluation set.",
      "metrics": {
        "downloads": 3532,
        "likes": 52,
        "lastModified": "2025-10-24"
      }
    },
    {
      "id": "muaalem-annotated-v3",
      "name": "muaalem annotated v3",
      "type": "dataset",
      "country": "INTL",
      "org": "obadx",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/obadx/muaalem-annotated-v3"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "Quran recitations from expert reciters with phonetic transcripts encoding tajweed rules, for recitation-error detection models.",
      "metrics": {
        "downloads": 3524,
        "likes": 7,
        "lastModified": "2025-09-04"
      }
    },
    {
      "id": "ara-reranker-v1",
      "name": "ARA-Reranker-V1",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reranking"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/ARA-Reranker-V1"
      },
      "size": "568M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic cross-encoder reranker.",
      "metrics": {
        "downloads": 3391,
        "likes": 4,
        "lastModified": "2025-04-03"
      }
    },
    {
      "id": "unnamed-split-ar",
      "name": "unnamed split ar",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "cc-by-3.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/unnamed-split-ar"
      },
      "year": 2026,
      "notes": "The data/ar/ partition of espnet/yodas3, cut from long recordings (median 5 minutes) into clips of 5-45 seconds, each with its own transcript.",
      "metrics": {
        "downloads": 3247,
        "likes": 0,
        "lastModified": "2026-10-05"
      }
    },
    {
      "id": "bert-large-arabertv02",
      "name": "bert-large-arabertv02",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-large-arabertv02"
      },
      "size": "371M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraBERT v02 large encoder.",
      "metrics": {
        "downloads": 3246,
        "likes": 10,
        "lastModified": "2023-08-03"
      }
    },
    {
      "id": "menaspeechbank",
      "name": "MENASpeechBank",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/MenaSpeechBank"
      },
      "size": "26.5h",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "MENA speech resource with reference voice bank and synthetic multi-turn persona dialogues.",
      "metrics": {
        "downloads": 3184,
        "likes": 4,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "qwen3-5-9b-saudi-dialect",
      "name": "Qwen3.5 9B saudi dialect",
      "type": "llm",
      "country": "INTL",
      "org": "AyoubChLin",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/AyoubChLin/Qwen3.5-9B-saudi-dialect"
      },
      "dialects": [
        "gulf"
      ],
      "size": "9B",
      "on_device": false,
      "year": 2026,
      "tags": [
        "variants:2"
      ],
      "notes": "This repository contains merged full weights for Saudi-dialect chat generation, not just LoRA adapters. (also: 2 variants)",
      "base_model": [
        "unsloth/qwen3.5-9b"
      ],
      "metrics": {
        "downloads": 3177,
        "likes": 0,
        "lastModified": "2026-03-22"
      }
    },
    {
      "id": "egyptian-arabic-wav2vec2-xlsr-53",
      "name": "egyptian-arabic-wav2vec2-xlsr-53",
      "type": "asr",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimAmin/egyptian-arabic-wav2vec2-xlsr-53"
      },
      "size": "315M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "egy"
      ],
      "notes": "wav2vec2 XLSR-53 fine-tuned on Egyptian Arabic.",
      "base_model": [
        "omarxadel/wav2vec2-large-xlsr-53-arabic-egyptian"
      ],
      "metrics": {
        "downloads": 3130,
        "likes": 6,
        "lastModified": "2025-05-11"
      }
    },
    {
      "id": "opus-mt-tc-big-en-ar",
      "name": "opus mt tc big en ar",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-en-ar"
      },
      "year": 2023,
      "notes": "Neural machine translation model for translating from English (en) to Arabic (ar).",
      "metrics": {
        "downloads": 3086,
        "likes": 29,
        "lastModified": "2023-10-10"
      }
    },
    {
      "id": "masri-100h",
      "name": "Masri 100h",
      "type": "dataset",
      "country": "EG",
      "org": "Ehab Negm",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ehabnegm/100-hour-Egyptian-dataset-single-speaker"
      },
      "size": "100h",
      "year": 2026,
      "dialects": [
        "egy"
      ],
      "notes": "100h Egyptian Arabic single-narrator speech corpus, 15.6k released clips.",
      "metrics": {
        "downloads": 3078,
        "likes": 11,
        "lastModified": "2026-08-07"
      }
    },
    {
      "id": "yemeni-arabic-assistant",
      "name": "Gemma-4 E2B Yemeni Arabic Assistant",
      "type": "llm",
      "country": "YE",
      "org": "Yemeni AI Lab",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/yemeni-ai-lab/gemma-4-e2b-yemeni-arabic-assistant-GGUF"
      },
      "dialects": [
        "yemeni"
      ],
      "notes": "Small Gemma-based chat assistant for Yemeni dialect, GGUF (Yemen).",
      "base_model": [
        "unsloth/gemma-4-e2b-it",
        "google/gemma-4-e2b-it"
      ],
      "metrics": {
        "downloads": 3058,
        "likes": 0,
        "lastModified": "2026-05-04"
      }
    },
    {
      "id": "aragpt2-base",
      "name": "aragpt2-base",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/aragpt2-base"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "148M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraGPT2 base, GPT-2 style Arabic generator from AUB.",
      "metrics": {
        "downloads": 2927,
        "likes": 34,
        "lastModified": "2023-10-30"
      }
    },
    {
      "id": "masribert-v4",
      "name": "MASRIBERT v4",
      "type": "llm",
      "country": "EG",
      "org": "T0KII",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/T0KII/MASRIBERTV4"
      },
      "size": "240M",
      "dialects": [
        "egy"
      ],
      "notes": "BERT-style encoder for Egyptian Arabic (Masri) text; community model.",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 2914,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "belebele-fleurs",
      "name": "Belebele-Fleurs",
      "type": "benchmark",
      "country": "INTL",
      "org": "University of Würzburg",
      "license": "cc-by-sa-4.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/WueNLP/belebele-fleurs",
        "paper": "https://arxiv.org/pdf/2501.06117"
      },
      "dialects": [
        "msa"
      ],
      "size": "387 sentences",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Belebele-Fleurs extends BeleBele into a spoken SLU benchmark with multilingual multiple-choice listening comprehension QA from speech.",
      "metrics": {
        "downloads": 2888,
        "likes": 9,
        "lastModified": "2024-12-12"
      }
    },
    {
      "id": "qari-ocr-v0-3-vl-2b-instruct",
      "name": "Qari-OCR-v0.3-VL-2B-Instruct",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Qari-OCR-v0.3-VL-2B-Instruct"
      },
      "size": "2.2B",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Qari OCR v0.3, 2B Qwen2-VL based Arabic OCR with diacritics support.",
      "metrics": {
        "downloads": 2831,
        "likes": 22,
        "lastModified": "2025-06-10"
      }
    },
    {
      "id": "darija-mmlu",
      "name": "DarijaMMLU",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/DarijaMMLU",
        "paper": "https://arxiv.org/abs/2409.17912"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "MMLU translated into Moroccan Darija, 22k+ multiple-choice questions.",
      "metrics": {
        "downloads": 2807,
        "likes": 8,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "lahgtna-libyan-tts-dataset",
      "name": "Lahgtna Libyan TTS dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/lahgtna-libyan-tts"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Libyan Arabic text-to-speech dataset from the Lahgtna multi-dialect TTS project.",
      "metrics": {
        "downloads": 2702,
        "likes": 0,
        "lastModified": "2026-08-31"
      }
    },
    {
      "id": "arabic-aya",
      "name": "Arabic Aya (2A)",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Arabic_Aya"
      },
      "size": "1M-10M rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic subset and translations of the Aya collection for instruction tuning.",
      "metrics": {
        "downloads": 2692,
        "likes": 16,
        "lastModified": "2024-03-15"
      }
    },
    {
      "id": "araelectra-base-artydiqa",
      "name": "araelectra-base-artydiqa",
      "type": "llm",
      "country": "LB",
      "org": "Wissam Antoun (AUB)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "question-answering"
      ],
      "links": {
        "hf": "https://huggingface.co/wissamantoun/araelectra-base-artydiqa"
      },
      "size": "135M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraELECTRA base fine-tuned on Arabic TyDiQA.",
      "metrics": {
        "downloads": 2686,
        "likes": 12,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "mixed-arabic-datasets-mad",
      "name": "Mixed Arabic Datasets (MAD)",
      "type": "dataset",
      "country": "INTL",
      "org": "M-A-D",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/M-A-D/Mixed-Arabic-Datasets-Repo"
      },
      "notes": "Community-driven diverse Arabic texts",
      "metrics": {
        "downloads": 2664,
        "likes": 38,
        "lastModified": "2023-10-16"
      }
    },
    {
      "id": "dialectal-arabic-voices",
      "name": "Dialectal Arabic Voices",
      "type": "dataset",
      "country": "INTL",
      "org": "moaead",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/moaead/dialectal-arabic-voices"
      },
      "year": 2026,
      "dialects": [
        "lev"
      ],
      "notes": "60,775 untranscribed Arabic recordings, currently labelled Palestinian Arabic.",
      "metrics": {
        "downloads": 2630,
        "likes": 0,
        "lastModified": "2026-09-26"
      }
    },
    {
      "id": "araelectra",
      "name": "AraELECTRA",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/araelectra-base-discriminator"
      },
      "notes": "ELECTRA for Arabic",
      "metrics": {
        "downloads": 2540,
        "likes": 5,
        "lastModified": "2024-10-29"
      }
    },
    {
      "id": "arabic-sts",
      "name": "arabic sts",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-sts"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "In the time of writing this Dataset Card, 31,112 civilians has been killed in Gaza (2/3 of them are women, elderly and children).",
      "metrics": {
        "downloads": 2480,
        "likes": 6,
        "lastModified": "2024-03-17"
      }
    },
    {
      "id": "ar-stablelm-2-chat",
      "name": "Ar-stablelm-2-chat",
      "type": "llm",
      "country": "SA",
      "org": "Stability AI",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/stabilityai/stablelm-2-1_6b-chat"
      },
      "size": "1.6B",
      "notes": "Small Arabic chat model",
      "on_device": true,
      "metrics": {
        "downloads": 2469,
        "likes": 34,
        "lastModified": "2024-06-03"
      }
    },
    {
      "id": "101-billion-arabic-words",
      "name": "101 Billion Arabic Words",
      "type": "dataset",
      "country": "INTL",
      "org": "ClusterlabAi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ClusterlabAi/101_billion_arabic_words_dataset"
      },
      "notes": "Massive Arabic web corpus",
      "metrics": {
        "downloads": 2416,
        "likes": 73,
        "lastModified": "2024-06-16"
      }
    },
    {
      "id": "ara-prompt-guard-v0",
      "name": "Ara-Prompt-Guard_V0",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Ara-Prompt-Guard_V0"
      },
      "size": "279M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic prompt-injection and jailbreak classifier from NAMAA.",
      "base_model": [
        "meta-llama/llama-prompt-guard-2-86m"
      ],
      "metrics": {
        "downloads": 2405,
        "likes": 11,
        "lastModified": "2026-03-08"
      }
    },
    {
      "id": "ocr-arabic-books",
      "name": "ocr arabic books",
      "type": "dataset",
      "country": "INTL",
      "org": "freococo",
      "license": "cc-by-nc-nd-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/freococo/ocr_arabic_books"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "This repository is a structurally aligned Arabic Optical Character Recognition (OCR) dataset.",
      "metrics": {
        "downloads": 2380,
        "likes": 5,
        "lastModified": "2026-07-09"
      }
    },
    {
      "id": "2m-belebele",
      "name": "2M-BELEBELE",
      "type": "dataset",
      "country": "INTL",
      "org": "Meta",
      "license": "cc-by-sa-4.0",
      "modality": "speech",
      "tasks": [
        "speech-comprehension"
      ],
      "links": {
        "github": "https://github.com/facebookresearch/belebele",
        "hf": "https://huggingface.co/datasets/facebook/2M-Belebele",
        "paper": "https://doi.org/10.18653/v1/2025.findings-acl.569"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,000 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "Multilingual speech and ASL comprehension dataset extending BELEBELE to 75 languages.",
      "metrics": {
        "downloads": 2355,
        "likes": 13,
        "lastModified": "2024-12-17"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-ca",
      "name": "bert-base-arabic-camelbert-ca",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-ca"
      },
      "year": 2022,
      "dialects": [
        "classical"
      ],
      "notes": "CAMeLBERT-CA pretrained on Classical Arabic.",
      "metrics": {
        "downloads": 2342,
        "likes": 13,
        "lastModified": "2021-09-14"
      }
    },
    {
      "id": "mlqa",
      "name": "MLQA",
      "type": "dataset",
      "country": "INTL",
      "org": "Facebook AI Research",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/facebookresearch/mlqa",
        "hf": "https://huggingface.co/datasets/facebook/mlqa",
        "paper": "https://arxiv.org/pdf/1910.07475.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,852 documents",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "5K extractive QA instances (12K in English) in SQuAD format in seven languages - English, Arabic, German, Spanish, Hindi, Vietnamese and Simplified Chinese.",
      "metrics": {
        "downloads": 2305,
        "likes": 44,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-da",
      "name": "bert-base-arabic-camelbert-da",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-da"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-DA pretrained on dialectal Arabic.",
      "metrics": {
        "downloads": 2300,
        "likes": 28,
        "lastModified": "2021-09-14"
      }
    },
    {
      "id": "masc-arabic",
      "name": "MASC Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/MASC-Arabic"
      },
      "size": "1000h",
      "year": 2023,
      "dialects": [
        "mixed"
      ],
      "notes": "Massive Arabic Speech Corpus, about 1,000h of YouTube speech across dialects, on HF.",
      "metrics": {
        "downloads": 2258,
        "likes": 8,
        "lastModified": "2026-04-06"
      }
    },
    {
      "id": "mizan-rerank-v2",
      "name": "Mizan Rerank V2",
      "type": "embedding",
      "country": "INTL",
      "org": "ALJIACHI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/ALJIACHI/Mizan-Rerank-V2"
      },
      "year": 2026,
      "tags": [
        "variants:2"
      ],
      "notes": "A cross-encoder model for reranking Arabic long texts, fine-tuned from Alibaba-NLP/gte-multilingual-reranker-base. (also: 2 variants)",
      "base_model": [
        "alibaba-nlp/gte-multilingual-reranker-base"
      ],
      "metrics": {
        "downloads": 2128,
        "likes": 0,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "cohere-command-a",
      "name": "Cohere Command-A",
      "type": "llm",
      "country": "INTL",
      "org": "Cohere",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/CohereForAI/c4ai-command-a-03-2025"
      },
      "size": "111B",
      "notes": "Optimized for RAG",
      "metrics": {
        "downloads": 2124,
        "likes": 395,
        "lastModified": "2025-10-30"
      }
    },
    {
      "id": "arat5v2-base-1024",
      "name": "AraT5v2-base-1024",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/AraT5v2-base-1024"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "AraT5 v2 base, Arabic text-to-text transformer with 1024 context.",
      "metrics": {
        "downloads": 2041,
        "likes": 32,
        "lastModified": "2024-05-16"
      }
    },
    {
      "id": "aramix-hq",
      "name": "AraMix-HQ",
      "type": "dataset",
      "country": "INTL",
      "org": "AdaMLLab",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AdaMLLab/AraMix-HQ"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "AraMix-HQ: 53.7M Arabic text rows from AraMix filtered with a model quality score (mmbert_score).",
      "metrics": {
        "downloads": 2038,
        "likes": 2,
        "lastModified": "2026-01-30"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-moroccan-darija",
      "name": "wav2vec2-large-xlsr-moroccan-darija",
      "type": "asr",
      "country": "MA",
      "org": "Boumehdi",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/boumehdi/wav2vec2-large-xlsr-moroccan-darija"
      },
      "year": 2023,
      "dialects": [
        "magh"
      ],
      "notes": "wav2vec2 XLSR fine-tuned on Moroccan Darija.",
      "base_model": [
        "facebook/wav2vec2-large-xlsr-53"
      ],
      "metrics": {
        "downloads": 2031,
        "likes": 32,
        "lastModified": "2024-04-12"
      }
    },
    {
      "id": "sada22",
      "name": "SADA22",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/SADA22"
      },
      "dialects": [
        "gulf",
        "mixed"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "The SADA dataset (Saudi Audio Dataset for Arabic) is a large-scale Arabic speech corpus designed to support the development.",
      "metrics": {
        "downloads": 2029,
        "likes": 30,
        "lastModified": "2025-05-03"
      }
    },
    {
      "id": "systematic-review-and-meta-analysis-of-2024-2025-studies-on",
      "name": "Systematic Review and Meta-Analysis of 2024–2025 Studies on AI Arabic Translation, Linguistics and Pedagogy",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "translation"
      ],
      "links": {
        "paper": "https://doi.org/10.32996/jcsts.2026.5.1.2"
      },
      "year": 2026,
      "venue": "Frontiers in Computer Science and Artificial Intelligence",
      "citations": 30,
      "notes": "This study aims to conduct a systematic review (SR) and meta-analysis (MA) of twenty articles by the author published between 2024–2025 on the use of AI models",
      "metrics": null
    },
    {
      "id": "a-holistic-assessment-of-the-carbon-footprint-of-noor-a-very",
      "name": "A Holistic Assessment of the Carbon Footprint of Noor, a Very Large Arabic Language Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2610.00223"
      },
      "year": 2026,
      "venue": "BIGSCIENCE",
      "citations": 26,
      "notes": "As ever larger language models grow more ubiquitous, it is crucial to consider their environmental impact.",
      "metrics": null
    },
    {
      "id": "a-comparative-study-of-pretrained-transformer-models-for-quranic-asr-s",
      "name": "A Comparative Study of Pretrained Transformer Models for Quranic ASR: Speech Representations, Label Formats, and Dataset Composition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "pretraining",
        "speech",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.19747"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents a systematic empirical study of domain-specific fine-tuning of pretrained Transformer-based models for Quranic ASR.",
      "metrics": null
    },
    {
      "id": "a-corpus-aligned-uthmani-to-standard-quranic-word-mapping-and-a-determ",
      "name": "A Corpus-Aligned Uthmani-to-Standard Quranic Word Mapping and a Deterministic Recitation Validator",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.14967"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "On top of the normalized text, we build a deterministic, LLM-free Quranic recitation validator using a four-layer verse-matching search.",
      "metrics": null
    },
    {
      "id": "a-human-in-the-loop-label-error-detection-framework-applied-to-arabic",
      "name": "A Human-in-the-Loop Label Error Detection Framework Applied to Arabic-Script HTR Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "asr",
        "ocr",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.16713"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "To help closing this gap, we propose a two-stage framework (CER-HV) for detecting label errors.",
      "metrics": null
    },
    {
      "id": "abjad-kids-an-arabic-speech-classification-dataset-for-primary-educati",
      "name": "Abjad-Kids: An Arabic Speech Classification Dataset for Primary Education",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.20255"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "However, children speech research remains limited due to the lack of publicly available datasets.",
      "metrics": null
    },
    {
      "id": "adab-arabic-dataset-for-automated-politeness-benchmarking-a-large-scal",
      "name": "ADAB: Arabic Dataset for Automated Politeness Benchmarking -- A Large-Scale Resource for Computational Sociopragmatics",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.13870"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce ADAB (Arabic Politeness Dataset), a new annotated Arabic dataset collected from four online platforms, including social media, e-commerce.",
      "metrics": null
    },
    {
      "id": "aha-memes-a-fine-grained-multimodal-benchmark-for-understanding-hate-i",
      "name": "AHA-Memes: A Fine-Grained Multimodal Benchmark for Understanding Hate in Arabic Memes",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2607.27393"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce AHA-Memes (Arabic HAteful Memes), which is, to our knowledge, the first large-scale Arabic hateful meme benchmark with fine-grained.",
      "metrics": null
    },
    {
      "id": "alexandria-a-multi-domain-dialectal-arabic-machine-translation-dataset",
      "name": "Alexandria: A Multi-Domain Dialectal Arabic Machine Translation Dataset for Culturally Inclusive and Linguistically Diverse LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.13099",
        "github": "https://github.com/UBC-NLP/Alexandria"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce Alexandria, a large-scale, community-driven, human-translated dataset designed to bridge this gap.",
      "metrics": null
    },
    {
      "id": "alexandriax-2026-the-first-shared-task-on-dialectal-arabic-machine-tra",
      "name": "AlexandriaX 2026: The First Shared Task on Dialectal Arabic Machine Translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding",
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.22796",
        "website": "https://alexandriax.dlnlp.ai"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present the AlexandriaX 2026 Shared Task on Dialectal Arabic MT, which addresses these challenges through three complementary subtasks.",
      "metrics": null
    },
    {
      "id": "almieyar-oryx-bloombench-a-bilingual-multimodal-benchmark-for-cognitiv",
      "name": "Almieyar-Oryx-BloomBench: A Bilingual Multimodal Benchmark for Cognitively Informed Evaluation of Vision-Language Models",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.05531",
        "github": "https://github.com/qcri/Almieyar-Oryx-BloomBench"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce BloomBench, part of the Almieyar benchmarking series, the first cognitively human-grounded, bilingual.",
      "metrics": null
    },
    {
      "id": "almieyar-a-culturally-grounded-benchmark-for-multi-dialect-arabic-spee",
      "name": "Almieyar: A Culturally Grounded Benchmark for Multi-Dialect Arabic Speech Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.35564"
      },
      "dialects": [
        "lev",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce ALMIEYAR, a culturally grounded ASR benchmark covering 17 Arabic dialects across six families.",
      "metrics": null
    },
    {
      "id": "an-end-to-end-hybrid-framework-for-rumour-detection-in-low-resources-a",
      "name": "An End-to-End Hybrid Framework for Rumour Detection in Low-Resources Algerian Dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.13411"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents an end-to-end rumour detection hybrid framework for Algerian dialect social media content.",
      "metrics": null
    },
    {
      "id": "arabdiscrim-a-decade-long-arabic-facebook-corpus-on-racism-and-discrim",
      "name": "ArabDiscrim: A Decade-Long Arabic Facebook Corpus on Racism and Discrimination",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.22081"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present ArabDiscrim, a decade-long lexical resource and corpus of 293K public Arabic Facebook posts (2014--2024) discussing racism and discrimination.",
      "metrics": null
    },
    {
      "id": "arabic-prompts-with-english-tools-a-benchmark",
      "name": "Arabic Prompts with English Tools: A Benchmark",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.05101"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Large Language Models (LLMs) are now integral to numerous industries, increasingly serving as the core reasoning engine for autonomous agents.",
      "metrics": null
    },
    {
      "id": "arabicdialecthub-a-cross-dialectal-arabic-learning-resource-and-platfo",
      "name": "ArabicDialectHub: A Cross-Dialectal Arabic Learning Resource and Platform",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.22987",
        "website": "https://arabic-dialect-hub.netlify.app"
      },
      "dialects": [
        "gulf",
        "lev",
        "magh",
        "msa"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present ArabicDialectHub, a cross-dialectal Arabic learning resource comprising 552 phrases across six varieties.",
      "metrics": null
    },
    {
      "id": "arabicdialectsafety-a-dialect-aware-benchmark-for-arabic-content-safet",
      "name": "ArabicDialectSafety: A Dialect-Aware Benchmark for Arabic Content Safety Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "safety",
        "dialects",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.01291"
      },
      "dialects": [
        "msa",
        "egy",
        "lev",
        "magh"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "ArabicDialectSafety: 25,071 human-curated safety prompts across MSA and four dialect groups, with seven harm categories and a dual-task evaluation.",
      "metrics": null
    },
    {
      "id": "aradetox-a-multi-dialect-arabic-detoxification-dataset",
      "name": "AraDetox: A Multi-Dialect Arabic Detoxification Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "arabicnlp-uk",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.22894",
        "github": "https://github.com/ArabicNLP-UK/AraDetox"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce AraDetox, a multi-dialect Arabic detoxification dataset comprising 10,500 harmful social-media posts and 84,000 detoxified rewrites.",
      "metrics": null
    },
    {
      "id": "arafa-an-llm-generated-arabic-fact-checking-dataset",
      "name": "ARAFA: An LLM-Generated Arabic Fact-Checking Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.25833"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce Arafa, a new large-scale dataset for fact-checking in Modern Standard Arabic, constructed through an automated framework.",
      "metrics": null
    },
    {
      "id": "aragenre-2026-a-hierarchical-definition-guided-arabic-genre-classifica",
      "name": "AraGenre 2026: A Hierarchical Definition-Guided Arabic Genre Classification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.27387"
      },
      "dialects": [
        "classical",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "AraGenre is a shared task on hierarchical, definition-guided Arabic genre classification, motivated by the limited availability of annotated data in Arabic.",
      "metrics": null
    },
    {
      "id": "arahopecorpus-annotation-guidelines-and-dataset-for-hope-speech-in-ara",
      "name": "AraHopeCorpus: Annotation Guidelines and Dataset for Hope Speech in Arabic Social Media Crisis Discourse",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "sentiment",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.23325"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper introduces AraHopeCorpus, the first annotated dataset of Arabic hope speech collected from ten thousand YouTube comments related to the war.",
      "metrics": null
    },
    {
      "id": "arams-28k-the-largest-publicly-released-line-level-dataset-of-historic",
      "name": "AraMS-28k: The Largest Publicly Released Line-Level Dataset of Historical Arabic Manuscripts with Margin and Insertion-Anchor Annotations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "ocr",
        "diacritization",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.26921"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce AraMS-28k, the largest publicly released line-level dataset of genuine historical Arabic manuscripts, comprising 14 books, 3,043 pages.",
      "metrics": null
    },
    {
      "id": "arcade-a-city-scale-corpus-for-fine-grained-arabic-dialect-tagging",
      "name": "ARCADE: A City-Scale Corpus for Fine-Grained Arabic Dialect Tagging",
      "type": "paper",
      "country": "SA",
      "org": "riotu-lab",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "dialect-id",
        "sentiment",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.02209",
        "hf": "https://huggingface.co/datasets/riotu-lab/ARCADE-full"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present ARCADE (Arabic Radio Corpus for Audio Dialect Evaluation), the first Arabic speech dataset designed explicitly with city-level dialect granularity.",
      "metrics": {
        "downloads": 117,
        "likes": 5,
        "lastModified": "2026-01-12"
      }
    },
    {
      "id": "are-arabic-benchmarks-reliable-qimma-s-quality-first-approach-to-llm-e",
      "name": "Are Arabic Benchmarks Reliable? QIMMA's Quality-First Approach to LLM Evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.03395"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present QIMMA, a quality-assured Arabic LLM leaderboard that places systematic benchmark validation at its core.",
      "metrics": null
    },
    {
      "id": "arpomeme-an-annotated-arabic-multimodal-dataset-for-political-ideology",
      "name": "ArPoMeme: An Annotated Arabic Multimodal Dataset for Political Ideology and Polarization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.20967"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents ArPoMeme, a large-scale dataset of approximately 7,300 Arabic political memes categorized by ideological orientation.",
      "metrics": null
    },
    {
      "id": "audience-engagement-with-arabic-women-s-social-empowerment-and-wellbei",
      "name": "Audience Engagement with Arabic Women's Social Empowerment and Wellbeing: A Decadal Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.22204"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents the Arabic Women and Society Corpus, a ten year collection of 252,487 public Arabic Facebook posts related to women's empowerment.",
      "metrics": null
    },
    {
      "id": "babeljudge-measuring-llm-as-a-judge-reliability-across-languages-and-a",
      "name": "BabelJudge: Measuring LLM-as-a-Judge Reliability Across Languages and Agent Trajectories",
      "type": "paper",
      "country": "INTL",
      "org": "shreyaskc",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.22329",
        "github": "https://github.com/Shreyaskc/BabelJudge"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce BabelJudge, an open-source benchmark and reliability audit framework that measures all four failure modes.",
      "metrics": null
    },
    {
      "id": "benchmarking-arabic-russian-machine-translation-a-comparison-of-fine-t",
      "name": "Benchmarking Arabic--Russian Machine Translation: A Comparison of Fine-tuned NMT and Few-shot LLMs under Rich Morphology and Low Lexical Overlap",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.29559"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Arabic-Russian machine translation (MT) remains under-explored due to the rich morphology of Arabic and low lexical overlap between the two languages.",
      "metrics": null
    },
    {
      "id": "benchmarking-commercial-asr-systems-on-code-switching-speech-arabic-pe",
      "name": "Benchmarking Commercial ASR Systems on Code-Switching Speech: Arabic, Persian, and German",
      "type": "paper",
      "country": "INTL",
      "org": "perle-ai",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "embedding",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.19069",
        "hf": "https://huggingface.co/datasets/Perle-ai/ASR_Code_Switch"
      },
      "dialects": [
        "egy",
        "gulf"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present a benchmark evaluating five commercial ASR providers across four language pairs: Egyptian Arabic--English, Saudi Arabic.",
      "metrics": {
        "downloads": 923,
        "likes": 12,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "benchmarking-frontier-llms-on-arabic-cultural-and-sociolinguistic-know",
      "name": "Benchmarking Frontier LLMs on Arabic Cultural and Sociolinguistic Knowledge: A Cross-Evaluation Framework with Human SME Ground Truth",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2607.00139"
      },
      "dialects": [
        "egy",
        "iraqi"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "The cost of human expert evaluation is a principal bottleneck to deploying language models in specialized, high-stakes domains.",
      "metrics": null
    },
    {
      "id": "beyond-fluency-a-rubric-based-benchmark-for-evaluating-saudi-dialect-a",
      "name": "Beyond Fluency: A Rubric-Based Benchmark for Evaluating Saudi Dialect and Cultural Competence in Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.29990"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present a rubric-based benchmark for the Saudi dialect, comprising 31 expert-authored prompts spanning idiomatic, pragmatic, lexical.",
      "metrics": null
    },
    {
      "id": "bolbosh-script-aware-flow-matching-for-kashmiri-text-to-speech",
      "name": "Bolbosh: Script-Aware Flow Matching for Kashmiri Text-to-Speech",
      "type": "paper",
      "country": "INTL",
      "org": "gaash-lab",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "diacritization",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.07513",
        "github": "https://github.com/gaash-lab/Bolbosh"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "In this work, we present the first dedicated open-source neural TTS system designed for Kashmiri.",
      "metrics": null
    },
    {
      "id": "bridging-scientific-heritage-an-arabic-russian-parallel-corpus-and-llm",
      "name": "Bridging Scientific Heritage: An Arabic--Russian Parallel Corpus and LLM Benchmark for Sustainable Knowledge Transfer",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.30943"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present a benchmark for Arabic--Russian scientific translation.",
      "metrics": null
    },
    {
      "id": "bulbul-a-dataset-for-dialectal-arabic-speech-recognition",
      "name": "Bulbul: A Dataset for Dialectal Arabic Speech Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.21950"
      },
      "dialects": [
        "classical",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present BULBUL, a multi-dialect Arabic ASR dataset collected from 275 speakers in 11 Arab countries.",
      "metrics": null
    },
    {
      "id": "candle-ctc-based-arabic-noisy-character-deduplication-using-a-lightwei",
      "name": "CANDLE: CTC-based Arabic Noisy-character Deduplication using a Lightweight Encoder",
      "type": "paper",
      "country": "INTL",
      "org": "Abjad AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.24758",
        "github": "https://github.com/abjadai/candle"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present CANDLE, a lightweight system for character-level Arabic noise deduplication that addresses this challenge without relying on handcrafted rules.",
      "metrics": null
    },
    {
      "id": "cardamom-a-micro-dialectal-arabic-speech-dataset-for-asr",
      "name": "CARDAMOM: A Micro-Dialectal Arabic Speech Dataset for ASR",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialects",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.34481"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Cardamom: about 40 hours of transcribed YouTube speech across 21 micro-dialects for fine-grained Arabic ASR evaluation.",
      "metrics": null
    },
    {
      "id": "cohesion-6k-an-arabic-dataset-for-analyzing-social-cohesion-and-confli",
      "name": "Cohesion-6K: An Arabic Dataset for Analyzing Social Cohesion and Conflict in Online Discourse",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "discourse",
        "social-media",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.22447"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Cohesion-6K: 6,000 Arabic Facebook posts on the Israeli occupation of Palestine, labelled on a conflict-to-cohesion discourse scale.",
      "metrics": null
    },
    {
      "id": "cross-lingual-learning-within-arabic-script-for-low-resource-htr",
      "name": "Cross-Lingual Learning within Arabic Script for Low-Resource HTR",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.02089"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Handwritten Text Recognition (HTR) with limited labeled data remains a challenging problem, particularly for Arabic-script languages.",
      "metrics": null
    },
    {
      "id": "crosshallu-do-hallucination-signals-generalize-across-languages-and-do",
      "name": "CrossHallu: Do Hallucination Signals Generalize Across Languages and Domains in Large Language Model's Internals?",
      "type": "paper",
      "country": "INTL",
      "org": "aishaalansari57",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2607.04029",
        "github": "https://github.com/aishaalansari57/CrossHal"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present CrossHallu, the first study to evaluate the cross-lingual and cross-domain generalization of hallucination detection.",
      "metrics": null
    },
    {
      "id": "cultural-benchmarking-of-llms-in-standard-and-dialectal-arabic-dialogu",
      "name": "Cultural Benchmarking of LLMs in Standard and Dialectal Arabic Dialogues",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.00119"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce ArabCulture-Dialogue, a culturally grounded conversational dataset covering 13 Arabic-speaking countries.",
      "metrics": null
    },
    {
      "id": "curriculum-learning-and-pseudo-labeling-improve-the-generalization-of",
      "name": "Curriculum Learning and Pseudo-Labeling Improve the Generalization of Multi-Label Arabic Dialect Identification Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.12937",
        "website": "https://mohamedalaa9.github.io/lahjatbert"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We construct a multi-label dataset by generating automatic multi-label annotations using GPT-4o and binary dialect acceptability classifiers.",
      "metrics": null
    },
    {
      "id": "cv-18-ner-augmented-common-voice-for-named-entity-recognition-from-ara",
      "name": "CV-18 NER: Augmented Common Voice for Named Entity Recognition from Arabic Speech",
      "type": "paper",
      "country": "INTL",
      "org": "elyadata",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "ner",
        "pretraining",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.02209",
        "hf": "https://huggingface.co/datasets/Elyadata/CV18-NER"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce CV-18 NER, the first publicly available dataset for NER from Arabic speech, created by augmenting the Arabic Common Voice 18 corpus.",
      "metrics": {
        "downloads": 99,
        "likes": 1,
        "lastModified": "2026-05-19"
      }
    },
    {
      "id": "do-llms-fabricate-legal-citations-a-bilingual-benchmark-on-saudi-data",
      "name": "Do LLMs Fabricate Legal Citations? A Bilingual Benchmark on Saudi Data Protection Law and the GDPR",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2607.11127"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce a bilingual benchmark of 120 questions probing whether freely accessible LLMs fabricate article citations for two data-protection instruments.",
      "metrics": null
    },
    {
      "id": "e-conan-entailment-contradition-and-neutral-benchmarks-arabic-textual",
      "name": "E-CONAN (Entailment, CONtradition And Neutral) Benchmarks: Arabic Textual Entailment and Natural Inference Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.11334"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper introduces E-CONAN benchmarks that are composed of sentences pairs from various sources: (1) automatically-translated pairs.",
      "metrics": null
    },
    {
      "id": "e-conan-entailment-contradition-and-neutral-diagnostics-dataset-invest",
      "name": "E-CONAN (Entailment, CONtradition And Neutral) Diagnostics Dataset Investigating Linguistic Phenomena in Arabic Natural Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.33530"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "To overcome this gap, we propose an initial hierarchy for Cross-Lingual NLU error analysis.",
      "metrics": null
    },
    {
      "id": "edrac-benchmarking-arabic-dialect-reading-comprehension",
      "name": "EDRAC: Benchmarking Arabic Dialect Reading Comprehension",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "qa",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.01113"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce EDRAC, the first large-scale benchmark for dialectal Arabic machine reading comprehension (MRC) and generative QA, covering five major dialects.",
      "metrics": null
    },
    {
      "id": "efficient-multilingual-neural-machine-translation-via-corpus-driven-vo",
      "name": "Efficient Multilingual Neural Machine Translation via Corpus-Driven Vocabulary Pruning: An English-Arabic Case Study",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.03480"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We propose in this paper a general optimization framework that combines a vocabulary pruning method with a targeted fine-tuning protocol for MNMT models.",
      "metrics": null
    },
    {
      "id": "fanar-sadiq-a-multi-agent-architecture-for-grounded-islamic-qa",
      "name": "Fanar-Sadiq: A Multi-Agent Architecture for Grounded Islamic QA",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "qa",
        "evaluation",
        "quran",
        "hadith"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.08501",
        "website": "https://api.fanar.qa/docs"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "To address these challenges, we present Fanar-Sadiq, a bilingual Arabic-English Islamic QA system built on a multi-agent, tool-augmented architecture.",
      "metrics": null
    },
    {
      "id": "grounding-arabic-llms-in-the-doha-historical-dictionary-retrieval-augm",
      "name": "Grounding Arabic LLMs in the Doha Historical Dictionary: Retrieval-Augmented Understanding of Quran and Hadith",
      "type": "paper",
      "country": "QA",
      "org": "somayaeltanbouly",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "rag",
        "quran",
        "lexicography"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.23972",
        "github": "https://github.com/somayaeltanbouly/Doha-Dictionary-RAG"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "RAG framework grounded in the Doha Historical Dictionary of Arabic that lifts Fanar and ALLaM above 85% on Quran and Hadith questions.",
      "metrics": null
    },
    {
      "id": "habibi-laying-the-open-source-foundation-of-unified-dialectal-arabic-s",
      "name": "Habibi: Laying the Open-Source Foundation of Unified-Dialectal Arabic Speech Synthesis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "diacritization",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.13802",
        "website": "https://SWivid.github.io/Habibi"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present Habibi, a unified-dialectal Arabic TTS framework that addresses all three.",
      "metrics": null
    },
    {
      "id": "hallutruthqa-4k-a-fine-grained-corpus-and-annotation-process-for-arabi",
      "name": "HalluTruthQA-4K: A Fine-Grained Corpus and Annotation Process for Arabic Hallucination Detection and Truth Verification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.03966"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present HalluTruthQA-4K, an expanded version of the HalluTruthQA resource containing 4,000 expert-curated Arabic question-answering instances.",
      "metrics": null
    },
    {
      "id": "hallutruthqa-a-fine-grained-benchmark-for-hallucination-detection-loca",
      "name": "HalluTruthQA: A Fine-Grained Benchmark for Hallucination Detection, Localization, and Explanation in Arabic Question Answering",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2607.20219"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce HalluTruthQA, a fine-grained benchmark for hallucination evaluation in Arabic question answering.",
      "metrics": null
    },
    {
      "id": "halluverse-m-3-a-multitask-multilingual-benchmark-for-hallucination-in",
      "name": "Halluverse-M^3: A multitask multilingual benchmark for hallucination in LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "sabdalja",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.06920",
        "hf": "https://huggingface.co/datasets/sabdalja/HalluVerse-M3"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce Halluverse-M^3, a dataset designed to enable systematic analysis of hallucinations across multiple languages, multiple generation tasks.",
      "metrics": {
        "downloads": 72,
        "likes": 2,
        "lastModified": "2026-05-04"
      }
    },
    {
      "id": "instruction-guided-poetry-generation-in-arabic-and-its-dialects",
      "name": "Instruction-Guided Poetry Generation in Arabic and Its Dialects",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "poetry"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.27766",
        "github": "https://github.com/mbzuai-nlp/instructpoet-ar"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Specifically, we present a large-scale, carefully curated instruction-based dataset in Modern Standard Arabic (MSA) and various Arabic dialects.",
      "metrics": null
    },
    {
      "id": "islamicmmlu-a-benchmark-for-evaluating-llms-on-islamic-knowledge",
      "name": "IslamicMMLU: A Benchmark for Evaluating LLMs on Islamic Knowledge",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation",
        "quran",
        "hadith"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.23750"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce IslamicMMLU, a benchmark of 10,013 multiple-choice questions spanning three tracks: Quran (2,013 questions), Hadith (4,000 questions), and Fiqh.",
      "metrics": null
    },
    {
      "id": "jobarabi-an-arabic-corpus-and-analysis-of-job-announcements-from-socia",
      "name": "JobArabi: An Arabic Corpus and Analysis of Job Announcements from Social Media",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.20960"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper introduces JobArabi, a large-scale corpus of Arabic job announcements collected from social media between January 2024 and October 2025.",
      "metrics": null
    },
    {
      "id": "linear-semantic-segmentation-for-low-resource-spoken-dialects",
      "name": "Linear Semantic Segmentation for Low-Resource Spoken Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.06276"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce a new multi-genre benchmark (more than 1000 samples) for semantic segmentation in conversational Arabic, focusing on dialectal discourse.",
      "metrics": null
    },
    {
      "id": "lqm-linguistically-motivated-multidimensional-quality-metrics-for-mach",
      "name": "LQM: Linguistically Motivated Multidimensional Quality Metrics for Machine Translation",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.18490",
        "github": "https://github.com/UBC-NLP/LQM_MT"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "yemeni",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "However, they often fail to capture dialect- and culture-specific errors in diglossic languages (e.g., Arabic).",
      "metrics": null
    },
    {
      "id": "macaron-controlled-human-written-benchmark-for-multilingual-and-multic",
      "name": "Macaron: Controlled, Human-Written Benchmark for Multilingual and Multicultural Reasoning via Template-Filling",
      "type": "paper",
      "country": "INTL",
      "org": "alaaahmed2444",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.10732",
        "hf": "https://huggingface.co/datasets/AlaaAhmed2444/Macaron"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We propose Macaron, a template-first benchmark that factorizes reasoning type and cultural aspect across question languages.",
      "metrics": {
        "downloads": 98,
        "likes": 2,
        "lastModified": "2026-02-14"
      }
    },
    {
      "id": "mawarith-a-dataset-and-benchmark-for-legal-inheritance-reasoning-with",
      "name": "MAWARITH: A Dataset and Benchmark for Legal Inheritance Reasoning with LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.07539",
        "website": "https://gitlab.com/nlpresearcher/mawarith"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "MAWARITH: 12,500 annotated Arabic Islamic inheritance cases for training and evaluating step-by-step inheritance reasoning.",
      "metrics": null
    },
    {
      "id": "mawqif-xt-an-arabic-benchmark-dataset-for-cross-target-stance-detectio",
      "name": "Mawqif-XT: An Arabic Benchmark Dataset for Cross-Target Stance Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.09539"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents the Mawqif-XT, consisting of 996 manually annotated Arabic tweets collected from three public targets: Women Driving, E-Cars.",
      "metrics": null
    },
    {
      "id": "medarabench-large-scale-arabic-medical-question-answering-dataset-and",
      "name": "MedAraBench: Large-Scale Arabic Medical Question Answering Dataset and Benchmark",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.01714"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "In this paper, we introduce MedAraBench, a large-scale dataset consisting of Arabic multiple-choice question-answer pairs across various medical specialties.",
      "metrics": null
    },
    {
      "id": "mederrbench-a-fine-grained-multilingual-benchmark-for-medical-error-de",
      "name": "MedErrBench: A Fine-Grained Multilingual Benchmark for Medical Error Detection and Correction with Clinical Expert Annotations",
      "type": "paper",
      "country": "INTL",
      "org": "congboma",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.05692",
        "github": "https://github.com/congboma/MedErrBench"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce MedErrBench, the first multilingual benchmark for error detection, localization.",
      "metrics": null
    },
    {
      "id": "mizan-a-national-benchmark-for-evaluating-large-language-models-on-ira",
      "name": "Mizan: A National Benchmark for Evaluating Large Language Models on Iraqi Arabic and the Iraqi Civic Context",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialects",
        "evaluation",
        "safety"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.13980"
      },
      "dialects": [
        "iraqi",
        "msa"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Mizan: national benchmark of 340 items evaluating LLMs on Iraqi Arabic and Iraqi civic context across six axes plus an MSA baseline track.",
      "metrics": null
    },
    {
      "id": "mudawansn-a-gold-standard-wolof-arabic-parallel-corpus-for-machine-tra",
      "name": "MudawanSn: A Gold-Standard Wolof-Arabic Parallel Corpus for Machine Translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.17539"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present MudawanSn, a gold-standard resource of 1,271 sentence-aligned pairs manually translated from Wolof into Modern Standard Arabic (MSA).",
      "metrics": null
    },
    {
      "id": "multi-task-instruction-tuning-via-data-scheduling-for-low-resource-ara",
      "name": "Multi-Task Instruction Tuning via Data Scheduling for Low-Resource Arabic SpeechLLMs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id",
        "sentiment",
        "summarization",
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.12494"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present a controlled study of multi-task instruction tuning for an Arabic-centric audio LLM across generative tasks.",
      "metrics": null
    },
    {
      "id": "multilingual-multi-label-emotion-classification-at-scale-with-syntheti",
      "name": "Multilingual Multi-Label Emotion Classification at Scale with Synthetic Data",
      "type": "paper",
      "country": "INTL",
      "org": "tabularisai",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.12633",
        "hf": "https://huggingface.co/tabularisai/multilingual-emotion-classification"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Emotion classification in multilingual settings remains constrained by the scarcity of annotated data.",
      "metrics": {
        "downloads": 2762,
        "likes": 27,
        "lastModified": "2026-07-31"
      }
    },
    {
      "id": "murad-a-large-scale-multi-domain-unified-reverse-arabic-dictionary-dat",
      "name": "MURAD: A Large-Scale Multi-Domain Unified Reverse Arabic Dictionary Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.21512"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present MURAD (Multi-domain Unified Reverse Arabic Dictionary), an open lexical dataset with 96,243 word-definition pairs.",
      "metrics": null
    },
    {
      "id": "nadi-2026-the-second-multidialectal-arabic-speech-processing-shared-ta",
      "name": "NADI 2026: The Second Multidialectal Arabic Speech Processing Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "tts",
        "translation",
        "dialect-id",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.27086"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "NADI 2026 is the seventh edition of the Nuanced Arabic Dialect Identification.",
      "metrics": null
    },
    {
      "id": "obscuring-data-contamination-through-translation-evidence-from-arabic",
      "name": "Obscuring Data Contamination Through Translation: Evidence from Arabic Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.14994"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We propose Translation-Aware Contamination Detection, which identifies contamination by comparing signals.",
      "metrics": null
    },
    {
      "id": "once-correct-still-wrong-counterfactual-hallucination-in-multilingual",
      "name": "Once Correct, Still Wrong: Counterfactual Hallucination in Multilingual Vision-Language Models",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2602.05437",
        "hf": "https://huggingface.co/datasets/QCRI/M2CQA"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce M$^2$CQA, a culturally grounded multimodal benchmark built from images spanning 17 MENA countries.",
      "metrics": {
        "downloads": 662,
        "likes": 2,
        "lastModified": "2026-06-12"
      }
    },
    {
      "id": "qias-2026-overview-of-the-shared-task-on-islamic-inheritance-reasoning",
      "name": "QIAS 2026: Overview of the Shared Task on Islamic Inheritance Reasoning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "qa",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.13756"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "This paper presents a comprehensive overview of the QIAS 2026 shared task, organized as part of the OSACT7 Workshop and co-located with LREC 2026.",
      "metrics": null
    },
    {
      "id": "quran-md-a-fine-grained-multilingual-multimodal-dataset-of-the-quran",
      "name": "Quran-MD: A Fine-Grained Multilingual Multimodal Dataset of the Quran",
      "type": "paper",
      "country": "INTL",
      "org": "buraaq",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "tts",
        "embedding",
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2601.17880",
        "hf": "https://huggingface.co/datasets/Buraaq/quran-audio-text-dataset"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present Quran MD, a comprehensive multimodal dataset of the Quran that integrates textual, linguistic, and audio dimensions at the verse and word levels.",
      "metrics": {
        "downloads": 132,
        "likes": 16,
        "lastModified": "2026-05-24"
      }
    },
    {
      "id": "quranicmmlu-a-cognitively-aware-benchmark-for-evaluating-generative-ai",
      "name": "QuranicMMLU: A Cognitively-Aware Benchmark for Evaluating Generative AI Solutions on Quranic Linguistic Knowledge",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "qa",
        "evaluation",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.22038"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce QuranicMMLU, a benchmark for evaluating generative AI on Quranic Arabic across multiple dimensions of linguistic complexity.",
      "metrics": null
    },
    {
      "id": "ramsa-a-large-sociolinguistically-rich-emirati-arabic-speech-corpus-fo",
      "name": "Ramsa: A Large Sociolinguistically Rich Emirati Arabic Speech Corpus for ASR and TTS",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.08125"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Ramsa is a developing 41-hour speech corpus of Emirati Arabic designed to support sociolinguistic research and low-resource language technologies.",
      "metrics": null
    },
    {
      "id": "rightnow-arabic-0-5b-turbo-an-open-sub-1b-arabic-language-model-via-vo",
      "name": "RightNow-Arabic-0.5B-Turbo: An Open Sub-1B Arabic Language Model via Vocabulary Injection and Edge-First Deployment",
      "type": "paper",
      "country": "INTL",
      "org": "rightnowai",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2605.28827",
        "hf": "https://huggingface.co/RightNowAI/RightNow-Arabic-0.5B-Turbo"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present RightNow-Arabic-0.5B-Turbo, a 518M-parameter Arabic-specialized decoder LLM built on Qwen2.5-0.5B.",
      "metrics": {
        "downloads": 513,
        "likes": 6,
        "lastModified": "2026-04-10"
      }
    },
    {
      "id": "sahm-a-benchmark-for-arabic-financial-and-shari-ah-compliant-reasoning",
      "name": "SAHM: A Benchmark for Arabic Financial and Shari'ah-Compliant Reasoning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "qa",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.19098"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce Sahm, the first Arabic financial benchmark spanning seven tasks.",
      "metrics": null
    },
    {
      "id": "shams-an-audio-grounded-pronunciation-benchmark-for-levantine-arabic",
      "name": "SHAMS: An Audio-Grounded Pronunciation Benchmark for Levantine Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "pronunciation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2610.01427",
        "website": "https://shams-nlp.github.io"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "SHAMS: 1,300-utterance benchmark for Levantine Arabic speech, evaluating pronunciation under opaque, non-standardized orthography.",
      "metrics": null
    },
    {
      "id": "syrisign-a-parallel-corpus-for-arabic-text-to-syrian-arabic-sign-langu",
      "name": "SyriSign: A Parallel Corpus for Arabic Text to Syrian Arabic Sign Language Translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "embedding",
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.29219"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "To overcome this gap, we introduce SyriSign, a dataset comprising 1500 video samples across 150 unique lexical signs.",
      "metrics": null
    },
    {
      "id": "tarab-a-multi-dialect-corpus-of-arabic-lyrics-and-poetry",
      "name": "Tarab: A Multi-Dialect Corpus of Arabic Lyrics and Poetry",
      "type": "paper",
      "country": "INTL",
      "org": "drelhaj",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.16601",
        "hf": "https://huggingface.co/datasets/drelhaj/Tarab"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "iraqi",
        "sudanese",
        "classical",
        "msa",
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce the Tarab Corpus, a large-scale cultural and linguistic resource that brings together Arabic song lyrics.",
      "metrics": {
        "downloads": 181,
        "likes": 4,
        "lastModified": "2026-02-25"
      }
    },
    {
      "id": "terminal-bench-lilt-multilingual-agentic-coding-benchmark-grounded-in",
      "name": "Terminal-Bench-LILT: Multilingual Agentic Coding Benchmark Grounded in Language, Region, and Culture",
      "type": "paper",
      "country": "INTL",
      "org": "lilt",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.28641",
        "github": "https://github.com/lilt/terminal-bench-lilt"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present Terminal-Bench-LILT, a suite of 300 authentic coding tasks in ten languages.",
      "metrics": null
    },
    {
      "id": "the-generator-eraser-paradox-community-guidelines-for-responsible-llm",
      "name": "The Generator-Eraser Paradox: Community Guidelines for Responsible LLM-Assisted Dialect Resource Creation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.06004"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Dialect resources occupy a unique position at the intersection of scientific description, cultural preservation, and computational infrastructure.",
      "metrics": null
    },
    {
      "id": "toxirex-a-dataset-on-toxic-reasoning-in-context",
      "name": "ToxiREX: A Dataset on Toxic REasoning in ConteXt",
      "type": "paper",
      "country": "INTL",
      "org": "cltl",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2606.27981",
        "github": "https://github.com/cltl/toxirex"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce a new, contextual, multilingual dataset called ToxiREX: Toxic REasoning in ConteXt.",
      "metrics": null
    },
    {
      "id": "ttlab-at-alexandriax-2026-a-fine-tuned-surface-tagger-for-arabic-machi",
      "name": "TTLab at AlexandriaX-2026: A Fine-Tuned Surface Tagger for Arabic Machine-Translation Error-Span Detection and Classification",
      "type": "paper",
      "country": "INTL",
      "org": "entailab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.29633",
        "github": "https://github.com/ENTAILab/arabic-dialectal-mt-error-span-detection"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present TTLab's submission to the AlexandriaX-2026 Subtask3 on Arabic MT error span detection and classification.",
      "metrics": null
    },
    {
      "id": "ttlab-at-daleel-2026-star-ar-sequence-tagging-for-argument-recognition",
      "name": "TTLab at Daleel 2026: STAR-Ar, Sequence Tagging for Argument Recognition in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "entailab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.39385",
        "github": "https://github.com/ENTAILab/daleel_2026_Arabic-Argumentative-Discourse-Mining"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Argument Mining (AM) is a critical NLP task that remains significantly under-resourced in Arabic.",
      "metrics": null
    },
    {
      "id": "tutlait-v1-a-crowdsourced-moroccan-tamazight-speech-dataset-with-arabi",
      "name": "TutlAit v1: a crowdsourced Moroccan Tamazight speech dataset with Arabic transcriptions and regional accent labels",
      "type": "paper",
      "country": "MA",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "tamazight"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.38219"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "TutlAit v1: crowdsourced Moroccan Tamazight speech paired with Arabic text and regional variety labels (Atlas, Souss, Rif).",
      "metrics": null
    },
    {
      "id": "what-really-controls-temporal-reasoning-in-large-language-models-token",
      "name": "What Really Controls Temporal Reasoning in Large Language Models: Tokenisation or Representation of Time?",
      "type": "paper",
      "country": "INTL",
      "org": "gagan3012",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2603.19017",
        "github": "https://github.com/gagan3012/mtb"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We present MultiTempBench, a multilingual temporal reasoning benchmark spanning three tasks, date arithmetic, time zone conversion.",
      "metrics": null
    },
    {
      "id": "when-do-vlms-help-arabic-manuscript-ocr-a-cross-dataset-study",
      "name": "When Do VLMs Help Arabic Manuscript OCR? A Cross-Dataset Study",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "diacritization",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2608.22366"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "Vision-language models (VLMs) are increasingly being used for document understanding, yet their role in Arabic and Islamic manuscript recognition remains.",
      "metrics": null
    },
    {
      "id": "yallamorph-a-benchmark-for-evaluating-arabic-morphological-generation",
      "name": "YallaMorph: A Benchmark for Evaluating Arabic Morphological Generation in Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2609.10153"
      },
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "We introduce YallaMorph, a large-scale benchmark for Arabic morphological generation covering verbs, nouns, adjectives, their cliticized forms.",
      "metrics": null
    },
    {
      "id": "ymir-a-new-benchmark-dataset-and-model-for-arabic-yemeni-music-genre-c",
      "name": "YMIR: A new Benchmark Dataset and Model for Arabic Yemeni Music Genre Classification Using Convolutional Neural Networks",
      "type": "paper",
      "country": "YE",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "music",
        "genre-classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.05011"
      },
      "dialects": [
        "yemeni"
      ],
      "year": 2026,
      "venue": "arXiv 2026",
      "notes": "YMIR: 1,475 audio clips of five Yemeni music genres labelled by experts, plus the YMCM CNN classifier.",
      "metrics": null
    },
    {
      "id": "fanar-an-arabic-centric-multimodal-generative-ai-platform",
      "name": "Fanar: An Arabic-Centric Multimodal Generative AI Platform",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.13944"
      },
      "year": 2025,
      "venue": "arXiv.org",
      "citations": 91,
      "notes": "We present Fanar, a platform for Arabic-centric multimodal generative AI systems, that supports language, speech and image generation tasks.",
      "metrics": null
    },
    {
      "id": "arabic-natural-language-processing-nlp-a-comprehensive-revie",
      "name": "Arabic Natural Language Processing (NLP): A Comprehensive Review of Challenges, Techniques, and Emerging Trends",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/computers14110497"
      },
      "year": 2025,
      "venue": "De Computis",
      "citations": 31,
      "notes": "Arabic natural language processing (NLP) has garnered significant attention in recent years due to the growing demand for automated text and Arabic-based",
      "metrics": null
    },
    {
      "id": "abstractive-text-summarization-in-arabic-like-script-using-m",
      "name": "Abstractive Text Summarization in Arabic-Like Script Using Multi-Encoder Architecture and Semantic Extraction Techniques",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2025.3575610"
      },
      "year": 2025,
      "venue": "IEEE Access",
      "citations": 29,
      "notes": "In the field of Natural Language Processing (NLP), the task of text summarization plays a vital role in understanding textual content and producing concise",
      "metrics": null
    },
    {
      "id": "evaluation-of-ai-generated-reading-comprehension-materials-f",
      "name": "Evaluation of AI-generated reading comprehension materials for Arabic language teaching",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1080/09588221.2025.2474037"
      },
      "year": 2025,
      "venue": "Computer Assisted Language Learning",
      "citations": 29,
      "notes": "This study investigates the effectiveness of AI tools in generating Arabic reading comprehension materials, focusing on",
      "metrics": null
    },
    {
      "id": "chinese-generative-ai-models-deepseek-and-qwen-rival-chatgpt",
      "name": "Chinese generative AI models (DeepSeek and Qwen) rival ChatGPT-4 in ophthalmology queries with excellent performance in Arabic and English",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.52225/narra.v5i1.2371"
      },
      "year": 2025,
      "venue": "Narra J",
      "citations": 27,
      "notes": "The rapid evolution of generative artificial intelligence (genAI) has ushered in a new era of digital medical consultations, with patients turning to AI-driven",
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-using-hybrid-contextual-features",
      "name": "Arabic fake news detection using hybrid contextual features",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.11591/ijece.v15i1.pp836-845"
      },
      "year": 2025,
      "venue": "International Journal of Electrical and Computer Engineering (IJECE)",
      "citations": 25,
      "notes": "Technology has advanced and social media users have grown dramatically in the last decade.",
      "metrics": null
    },
    {
      "id": "towards-inclusive-arabic-llms-a-culturally-aligned-benchmark",
      "name": "Towards Inclusive Arabic LLMs: A Culturally Aligned Benchmark in Arabic Large Language Model Evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://www.semanticscholar.org/paper/1ca6c715cd6664d25d8e9d5deda50f067359558b"
      },
      "year": 2025,
      "venue": "COLING Workshops",
      "citations": 22,
      "metrics": null
    },
    {
      "id": "a-scoping-review-of-arabic-natural-language-processing-for-m",
      "name": "A Scoping Review of Arabic Natural Language Processing for Mental Health",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/healthcare13090963"
      },
      "year": 2025,
      "venue": "Healthcare",
      "citations": 21,
      "notes": "Mental health disorders represent a substantial global health concern, impacting millions and placing a significant burden on public health systems.",
      "metrics": null
    },
    {
      "id": "command-r7b-arabic-a-small-enterprise-focused-multilingual-a",
      "name": "Command R7B Arabic: A Small, Enterprise Focused, Multilingual, and Culturally Aware Arabic LLM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2503.14603"
      },
      "year": 2025,
      "venue": "Proceedings of the Sixth Workshop on African Natural Language Processing (AfricaNLP 2025)",
      "citations": 21,
      "notes": "In this work, we present a data synthesis and refinement strategy to help address this problem, namely, by leveraging synthetic",
      "metrics": null
    },
    {
      "id": "arabic-transliteration-of-borrowed-english-nouns-with-g-by-a",
      "name": "Arabic Transliteration of Borrowed English Nouns with /g/ by Artificial Intelligence (AI",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "paper": "https://doi.org/10.32996/jcsts.2025.7.9.29"
      },
      "year": 2025,
      "venue": "Journal of Computer Science and Technology Studies",
      "citations": 20,
      "notes": "This study sought to explore the transliteration of English nouns containing /g/ by Microsoft Copilot (MC) and Google Translate (GT), and find out which Arabic",
      "metrics": null
    },
    {
      "id": "improving-mispronunciation-detection-and-diagnosis-for-non-n",
      "name": "Improving mispronunciation detection and diagnosis for non- native learners of the Arabic language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10791-024-09489-8"
      },
      "year": 2025,
      "venue": "Discover Computing",
      "citations": 20,
      "notes": "Mispronunciation detection and diagnosis (MDD) is a core component of computer-assisted pronunciation training (CAPT), which aims to provide opportunities for",
      "metrics": null
    },
    {
      "id": "the-dialects-gap-a-multi-task-learning-approach-for-enhancin",
      "name": "The dialects gap: A multi-task learning approach for enhancing hate speech detection in Arabic dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.eswa.2025.128584"
      },
      "year": 2025,
      "venue": "Expert systems with applications",
      "citations": 20,
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-using-neural-networks-concepts-lit",
      "name": "Arabic speech recognition using neural networks: concepts, literature review and challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "survey",
        "asr"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s43994-025-00213-w"
      },
      "year": 2025,
      "venue": "Journal of Umm Al-Qura University for Applied Sciences",
      "citations": 18,
      "notes": "The ability to recognize and translate human speech has grown in importance.",
      "metrics": null
    },
    {
      "id": "radar-an-ensemble-approach-for-radicalization-detection-in-a",
      "name": "RADAR#: An Ensemble Approach for Radicalization Detection in Arabic Social Media Using Hybrid Deep Learning and Transformer Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/info16070522"
      },
      "year": 2025,
      "venue": "Inf.",
      "citations": 17,
      "notes": "RADAR#, a deep ensemble approach for the detection of radicalization in Arabic tweets, is introduced in this paper.",
      "metrics": null
    },
    {
      "id": "translation-of-arabic-expressions-of-impossibility-by-ai-and",
      "name": "Translation of Arabic Expressions of Impossibility by AI and Student-Translators: A Comparative Study",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.32996/jcsts.2025.7.8.33"
      },
      "year": 2025,
      "venue": "Journal of Computer Science and Technology Studies",
      "citations": 17,
      "notes": "Expressions of impossibility (EIs) refer to events that never or rarely happen, things that are impossible to find, tasks that are difficult or impossible to",
      "metrics": null
    },
    {
      "id": "a-survey-of-code-switched-arabic-nlp-progress-challenges-and",
      "name": "A Survey of Code-switched Arabic NLP: Progress, Challenges, and Future Directions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.13419"
      },
      "year": 2025,
      "venue": "International Conference on Computational Linguistics",
      "citations": 16,
      "notes": "Language in the Arab world presents a complex diglossic and multilingual setting, involving the use of Modern Standard Arabic, various dialects and",
      "metrics": null
    },
    {
      "id": "arabic-fake-news-dataset-development-humans-and-ai-generated",
      "name": "Arabic Fake News Dataset Development: Humans and AI-Generated Contributions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2025.3556376"
      },
      "year": 2025,
      "venue": "IEEE Access",
      "citations": 16,
      "notes": "The extensive use of social media platforms has promoted the rapid spread of fake news on the internet, such as fake reviews, rumors, and propaganda.",
      "metrics": null
    },
    {
      "id": "evaluating-translation-quality-a-qualitative-and-quantitativ",
      "name": "Evaluating Translation Quality: A Qualitative and Quantitative Assessment of Machine and LLM-Driven Arabic-English Translations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/info16060440"
      },
      "year": 2025,
      "venue": "Inf.",
      "citations": 16,
      "notes": "This study investigates translation quality between Arabic and English, comparing traditional rule-based machine translation systems, modern neural machine",
      "metrics": null
    },
    {
      "id": "a-comprehensive-survey-on-arabic-text-augmentation-approache",
      "name": "A comprehensive survey on Arabic text augmentation: approaches, challenges, and applications",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s00521-025-11020-z"
      },
      "year": 2025,
      "venue": "Neural computing & applications (Print)",
      "citations": 15,
      "notes": "Arabic is a linguistically complex language with a rich structure and valuable syntax that pose unique challenges for natural language processing (NLP)",
      "metrics": null
    },
    {
      "id": "determining-the-meter-of-classical-arabic-poetry-using-deep",
      "name": "Determining the meter of classical Arabic poetry using deep learning: a performance analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.3389/frai.2025.1523336"
      },
      "year": 2025,
      "venue": "Frontiers Artif. Intell.",
      "citations": 15,
      "notes": "In this study, a deep learning-based approach was developed to accurately determine the meter of",
      "metrics": null
    },
    {
      "id": "how-well-can-llms-grade-essays-in-arabic",
      "name": "How well can LLMs Grade Essays in Arabic?",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.16516"
      },
      "year": 2025,
      "venue": "Computers and Education: Artificial Intelligence",
      "citations": 15,
      "notes": "This research assesses the effectiveness of state-of-the-art large language models (LLMs), including ChatGPT, Llama, Aya, Jais, and ACEGPT, in the task of",
      "metrics": null
    },
    {
      "id": "leveraging-chatgpt-for-enhancing-arabic-nlp-application-for",
      "name": "Leveraging ChatGPT for Enhancing Arabic NLP: Application for Semantic Role Labeling and Cross-Lingual Annotation Projection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2025.3525493"
      },
      "year": 2025,
      "venue": "IEEE Access",
      "citations": 15,
      "notes": "Semantic role labeling involves assigning semantic roles to sentence arguments, providing rich information for various NLP tasks and applications.",
      "metrics": null
    },
    {
      "id": "topic-modeling-and-sentiment-analysis-of-arabic-news-headlin",
      "name": "Topic Modeling and Sentiment Analysis of Arabic News Headlines for a Societal Well-Being Scoring and Monitoring System: Moroccan Use Case",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2025.3538888"
      },
      "year": 2025,
      "venue": "IEEE Access",
      "citations": 15,
      "notes": "In today’s information-rich society, the rapid dissemination of news content across digital platforms plays a pivotal role in shaping public perception and",
      "metrics": null
    },
    {
      "id": "msa-at-arahealthqa-2025-shared-task-enhancing-llm-performance-for-arab",
      "name": "!MSA at AraHealthQA 2025 Shared Task: Enhancing LLM Performance for Arabic Clinical Question Answering through Prompt Engineering and Ensemble Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.11365"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present our systems for Track 2 (General Arabic Health QA, MedArabiQ) of the AraHealthQA-2025 shared task.",
      "metrics": null
    },
    {
      "id": "msa-at-barec-shared-task-2025-ensembling-arabic-transformers-for-reada",
      "name": "!MSA at BAREC Shared Task 2025: Ensembling Arabic Transformers for Readability Assessment",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.10040"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present MSAs winning system for the BAREC 2025 Shared Task on fine-grained Arabic readability assessment, achieving first place in six of six tracks.",
      "metrics": null
    },
    {
      "id": "3lm-bridging-arabic-stem-and-code-through-benchmarking",
      "name": "3LM: Bridging Arabic, STEM, and Code through Benchmarking",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.15850"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To help bridge this gap, we present 3LM, a suite of three benchmarks designed specifically for Arabic.",
      "metrics": null
    },
    {
      "id": "a-culturally-diverse-multilingual-multimodal-video-benchmark-model",
      "name": "A Culturally-diverse Multilingual Multimodal Video Benchmark & Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "translation",
        "qa",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.07032",
        "website": "https://mbzuai-oryx.github.io/ViMUL"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In pursuit of more inclusive video LMMs, we introduce a multilingual Video LMM benchmark, named ViMUL-Bench, to evaluate Video LMMs across 14 languages.",
      "metrics": null
    },
    {
      "id": "a-large-and-balanced-corpus-for-fine-grained-arabic-readability-assess",
      "name": "A Large and Balanced Corpus for Fine-grained Arabic Readability Assessment",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.13520"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper introduces the Balanced Arabic Readability Evaluation Corpus (BAREC), a large-scale, fine-grained dataset for Arabic readability assessment.",
      "metrics": null
    },
    {
      "id": "adapting-falcon3-7b-language-model-for-arabic-methods-challenges-and-o",
      "name": "Adapting Falcon3-7B Language Model for Arabic: Methods, Challenges, and Outcomes",
      "type": "paper",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "llm-adaptation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.1/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "Describes adapting the Falcon3-7B language model to Arabic, from data collection through training and outcomes.",
      "metrics": null
    },
    {
      "id": "adi-20-arabic-dialect-identification-dataset-and-models",
      "name": "ADI-20: Arabic Dialect Identification dataset and models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "asr",
        "dialect-id",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.10070"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present ADI-20, an extension of the previously published ADI-17 Arabic Dialect Identification (ADI) dataset.",
      "metrics": null
    },
    {
      "id": "advancing-arabic-reverse-dictionary-systems-a-transformer-based-approa",
      "name": "Advancing Arabic Reverse Dictionary Systems: A Transformer-Based Approach with Dataset Construction Guidelines",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.21475"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present a novel transformer-based approach with a semi-encoder neural network architecture featuring geometrically decreasing layers.",
      "metrics": null
    },
    {
      "id": "ahasis-shared-task-on-sentiment-analysis-for-arabic-dialects",
      "name": "AHaSIS: Shared Task on Sentiment Analysis for Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.13335"
      },
      "dialects": [
        "gulf",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "The hospitality industry in the Arab world increasingly relies on customer feedback to shape services.",
      "metrics": null
    },
    {
      "id": "alarb-an-arabic-legal-argument-reasoning-benchmark",
      "name": "ALARB: An Arabic Legal Argument Reasoning Benchmark",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "legal",
        "reasoning",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.00694"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "ALARB: 13K Saudi commercial court cases for benchmarking Arabic LLM legal reasoning, with verdict prediction and regulation retrieval tasks.",
      "metrics": null
    },
    {
      "id": "algerian-dialect",
      "name": "Algerian Dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2512.19543",
        "website": "https://doi.org/10.17632/zzwg3nnhsz.2"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present Algerian Dialect, a large-scale sentiment-annotated dataset consisting of 45,000 YouTube comments written in Algerian Arabic dialect.",
      "metrics": null
    },
    {
      "id": "alhd-a-large-scale-and-multigenre-benchmark-dataset-for-arabic-llm-gen",
      "name": "ALHD: A Large-Scale and Multigenre Benchmark Dataset for Arabic LLM-Generated Text Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.03502"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce ALHD, the first large-scale comprehensive Arabic dataset explicitly designed to distinguish between human- and LLM-generated texts.",
      "metrics": null
    },
    {
      "id": "alignar-generative-sentence-alignment-for-arabic-english-parallel-corp",
      "name": "AlignAR: Generative Sentence Alignment for Arabic-English Parallel Corpora of Legal and Literary Texts",
      "type": "paper",
      "country": "INTL",
      "org": "xxx",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2512.21842",
        "github": "https://github.com/XXX"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present AlignAR, a generative sentence alignment method, and a new Arabic-English dataset comprising simple legal and complex literary parallel texts.",
      "metrics": null
    },
    {
      "id": "amcrawl-an-arabic-web-scale-dataset-of-interleaved-image-text-document",
      "name": "AMCrawl: An Arabic Web-Scale Dataset of Interleaved Image-Text Documents and Image-Text Pairs",
      "type": "paper",
      "country": "SA",
      "org": "SDAIA",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.37/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "First native Arabic multimodal web dataset of interleaved image-text documents and image-text pairs, filtered from Common Crawl for quality and safety.",
      "metrics": null
    },
    {
      "id": "an-annotated-corpus-of-arabic-tweets-for-hate-speech-analysis",
      "name": "An Annotated Corpus of Arabic Tweets for Hate Speech Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.11969"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Identifying hate speech content in the Arabic language is challenging due to the rich quality of dialectal variations.",
      "metrics": null
    },
    {
      "id": "an-exploration-of-knowledge-editing-for-arabic",
      "name": "An Exploration of Knowledge Editing for Arabic",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.34/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present the first study of Arabic KE.",
      "metrics": null
    },
    {
      "id": "an-open-research-dataset-of-the-1932-cairo-congress-of-arab-music",
      "name": "An Open Research Dataset of the 1932 Cairo Congress of Arab Music",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "music",
        "audio",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.14503"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "ORD-CC32: open dataset from the 1932 Cairo Congress of Arab Music recordings with maqam and iqa tags, tonic labels and acoustic features.",
      "metrics": null
    },
    {
      "id": "ara-hope-human-centric-post-editing-evaluation-for-dialectal-arabic-to",
      "name": "Ara-HOPE: Human-Centric Post-Editing Evaluation for Dialectal Arabic to Modern Standard Arabic Translation",
      "type": "paper",
      "country": "INTL",
      "org": "abdullahalabdullah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2512.21787",
        "github": "https://github.com/abdullahalabdullah/Ara-HOPE"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper introduces Ara-HOPE, a human-centric post-editing evaluation framework designed to systematically address these challenges.",
      "metrics": null
    },
    {
      "id": "arabemonet-a-lightweight-hybrid-2d-cnn-bilstm-model-with-attention-for",
      "name": "ArabEmoNet: A Lightweight Hybrid 2D CNN-BiLSTM Model with Attention for Robust Arabic Speech Emotion Recognition",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "emotion"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.17/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We introduce ArabEmoNet, a lightweight architecture designed to overcome these limitations and deliver state-of-the-art performance.",
      "metrics": null
    },
    {
      "id": "arabic-asr-on-the-sada-large-scale-arabic-speech-corpus-with-transform",
      "name": "Arabic ASR on the SADA Large-Scale Arabic Speech Corpus with Transformer-Based Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.12968"
      },
      "dialects": [
        "gulf",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We explore the performance of several state-of-the-art automatic speech recognition (ASR) models on a large-scale Arabic speech dataset, the SADA.",
      "metrics": null
    },
    {
      "id": "arabic-little-stt-arabic-children-speech-recognition-dataset",
      "name": "Arabic Little STT: Arabic Children Speech Recognition Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.23319"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present our created dataset, Arabic Little STT, a dataset of Levantine Arabic child speech recorded in classrooms.",
      "metrics": null
    },
    {
      "id": "arabic-multimodal-machine-learning-datasets-applications-approaches-an",
      "name": "Arabic Multimodal Machine Learning: Datasets, Applications, Approaches, and Challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "embedding",
        "sentiment",
        "survey",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.12227"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Multimodal Machine Learning (MML) aims to integrate and analyze information from diverse modalities, such as text, audio, and visuals.",
      "metrics": null
    },
    {
      "id": "arabic-tool-calling-data-fanar",
      "name": "Arabic tool-calling data (Fanar)",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "tool-calling"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.20957"
      },
      "year": 2025,
      "notes": "Paper on data strategies and instruction tuning for tool calling in Arabic LLMs (Fanar).",
      "metrics": null
    },
    {
      "id": "arabjobs-a-multinational-corpus-of-arabic-job-ads",
      "name": "ArabJobs: A Multinational Corpus of Arabic Job Ads",
      "type": "paper",
      "country": "INTL",
      "org": "drelhaj",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "jobs",
        "bias",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.22589",
        "github": "https://github.com/drelhaj/ArabJobs"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "ArabJobs: 8,500 Arabic job ads from Egypt, Jordan, Saudi Arabia and the UAE for gender-bias, profession and salary analysis.",
      "metrics": null
    },
    {
      "id": "arafinnews-arabic-financial-summarisation-with-domain-adapted-llms",
      "name": "AraFinNews: Arabic Financial Summarisation with Domain-Adapted LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "arabicnlp-uk",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.01265",
        "github": "https://github.com/ArabicNLP-uk/AraFinNews"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce AraFinNews, the largest publicly available Arabic financial news dataset to date.",
      "metrics": null
    },
    {
      "id": "arahallueval-a-fine-grained-hallucination-evaluation-framework-for-ara",
      "name": "AraHalluEval: A Fine-grained Hallucination Evaluation Framework for Arabic LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "aishaalansari57",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "summarization",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.04656",
        "github": "https://github.com/aishaalansari57/AraHalluEval"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper presents the first comprehensive hallucination evaluation of Arabic and multilingual LLMs.",
      "metrics": null
    },
    {
      "id": "arahealthqa-2025-the-first-shared-task-on-arabic-health-question-answe",
      "name": "AraHealthQA 2025: The First Shared Task on Arabic Health Question Answering",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.20047"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce AraHealthQA 2025, the Comprehensive Arabic Health Question Answering Shared Task, held in conjunction with ArabicNLP 2025.",
      "metrics": null
    },
    {
      "id": "aralingbench-a-human-annotated-benchmark-for-evaluating-arabic-linguis",
      "name": "AraLingBench A Human-Annotated Benchmark for Evaluating Arabic Linguistic Capabilities of Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.14295"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present AraLingBench: a fully human annotated benchmark for evaluating the Arabic linguistic competence of large language models (LLMs).",
      "metrics": null
    },
    {
      "id": "arareasoner-evaluating-reasoning-based-llms-for-arabic-nlp",
      "name": "AraReasoner: Evaluating Reasoning-Based LLMs for Arabic NLP",
      "type": "paper",
      "country": "INTL",
      "org": "gufransabri",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.08768",
        "github": "https://github.com/gufranSabri/deepseek-evals"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper presents a comprehensive benchmarking study of multiple reasoning-focused LLMs.",
      "metrics": null
    },
    {
      "id": "aratable-benchmarking-llms-reasoning-and-understanding-of-arabic-tabul",
      "name": "AraTable: Benchmarking LLMs' Reasoning and Understanding of Arabic Tabular Data",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.18442"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present AraTable, a novel and comprehensive benchmark designed to evaluate the reasoning and understanding capabilities of LLMs when applied.",
      "metrics": null
    },
    {
      "id": "arb-a-comprehensive-arabic-multimodal-reasoning-benchmark",
      "name": "ARB: A Comprehensive Arabic Multimodal Reasoning Benchmark",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.17021",
        "github": "https://github.com/mbzuai-oryx/ARB"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce the Comprehensive Arabic Multimodal Reasoning Benchmark.",
      "metrics": null
    },
    {
      "id": "arfake-a-robust-framework-for-multi-dialect-arabic-speech-spoofing-det",
      "name": "ArFake: A Robust Framework for Multi-Dialect Arabic Speech Spoofing Detection Benchmark",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "spoof-detection",
        "speech",
        "tts"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.22808"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "ArFake: first multi-dialect Arabic spoofed-speech dataset built from several TTS models, with a spoof-detection evaluation pipeline.",
      "metrics": null
    },
    {
      "id": "arvoice-a-multi-speaker-dataset-for-arabic-speech-synthesis",
      "name": "ArVoice: A Multi-Speaker Dataset for Arabic Speech Synthesis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "diacritization",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.20506"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce ArVoice, a multi-speaker Modern Standard Arabic.",
      "metrics": null
    },
    {
      "id": "arzen-multigenre-an-aligned-parallel-dataset-of-egyptian-arabic-song-l",
      "name": "ArzEn-MultiGenre: An aligned parallel dataset of Egyptian Arabic song lyrics, novels, and subtitles, with English translations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.01411"
      },
      "dialects": [
        "egy"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "ArzEn-MultiGenre is a parallel dataset of Egyptian Arabic song lyrics, novels.",
      "metrics": null
    },
    {
      "id": "autoarabic-a-three-stage-framework-for-localizing-video-text-retrieval",
      "name": "AutoArabic: A Three-Stage Framework for Localizing Video-Text Retrieval Benchmarks",
      "type": "paper",
      "country": "INTL",
      "org": "tahaalshatiri",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.16438",
        "github": "https://github.com/Tahaalshatiri/AutoArabic"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce a three-stage framework, AutoArabic, utilizing state-of-the-art large language models.",
      "metrics": null
    },
    {
      "id": "automatic-pronunciation-error-detection-and-correction-of-the-holy-qur",
      "name": "Automatic Pronunciation Error Detection and Correction of the Holy Quran's Learners Using Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "evaluation",
        "speech",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.00094",
        "website": "https://obadx.github.io/quran-muaalem/en"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We release our work as open-source: https://obadx.github.io/quran-muaalem/en/",
      "metrics": null
    },
    {
      "id": "balsam-a-platform-for-benchmarking-arabic-large-language-models",
      "name": "BALSAM: A Platform for Benchmarking Arabic Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.22603"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In particular, we introduce BALSAM, a comprehensive, community-driven benchmark aimed at advancing Arabic LLM development and evaluation.",
      "metrics": null
    },
    {
      "id": "benchmarking-the-legal-reasoning-of-llms-in-arabic-islamic-inheritance",
      "name": "Benchmarking the Legal Reasoning of LLMs in Arabic Islamic Inheritance Cases",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.15796"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Islamic inheritance domain holds significant importance for Muslims to ensure fair distribution of shares between heirs.",
      "metrics": null
    },
    {
      "id": "benchmarking-the-medical-understanding-and-reasoning-of-large-language",
      "name": "Benchmarking the Medical Understanding and Reasoning of Large Language Models in Arabic Healthcare Tasks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.15797"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Recent progress in large language models (LLMs) has showcased impressive proficiency in numerous Arabic natural language processing (NLP) applications.",
      "metrics": null
    },
    {
      "id": "beyond-mcq-an-open-ended-arabic-cultural-qa-benchmark-with-dialect-var",
      "name": "Beyond MCQ: An Open-Ended Arabic Cultural QA Benchmark with Dialect Variants",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.24328"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We propose a comprehensive method that (i) translates Modern Standard Arabic (MSA) multiple-choice questions.",
      "metrics": null
    },
    {
      "id": "building-and-aligning-comparable-corpora",
      "name": "Building and Aligning Comparable Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.02555"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In this paper, we present a method to build comparable corpora from Wikipedia encyclopedia and EURONEWS website in English, French and Arabic languages.",
      "metrics": null
    },
    {
      "id": "busted-at-arageneval-shared-task-a-comparative-study-of-transformer-ba",
      "name": "BUSTED at AraGenEval Shared Task: A Comparative Study of Transformer-Based Models for Arabic AI-Generated Text Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.20610"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper details our submission to the AraGenEval Shared Task on Arabic AI-generated text detection, where our team, BUSTED, secured 5th place.",
      "metrics": null
    },
    {
      "id": "capturing-intra-dialectal-variation-in-qatari-arabic-a-corpus-of-cultu",
      "name": "Capturing Intra-Dialectal Variation in Qatari Arabic: A Corpus of Cultural and Gender Dimensions",
      "type": "paper",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "asr",
        "pronunciation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.18/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present the first publicly available, multidimensional corpus of Qatari Arabic that captures intra-dialectal variation across Urban and Bedouin speakers.",
      "metrics": null
    },
    {
      "id": "carma-comprehensive-automatically-annotated-reddit-mental-health-datas",
      "name": "CARMA: Comprehensive Automatically-annotated Reddit Mental Health Dataset for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.03102"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present CARMA, the first automatically annotated large-scale dataset of Arabic Reddit posts.",
      "metrics": null
    },
    {
      "id": "comparative-approaches-to-sentiment-analysis-using-datasets-in-major-e",
      "name": "Comparative Approaches to Sentiment Analysis Using Datasets in Major European and Arabic Languages",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.12540"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This study explores transformer-based models such as BERT, mBERT, and XLM-R for multi-lingual sentiment analysis across diverse linguistic structures.",
      "metrics": null
    },
    {
      "id": "cross-lingual-synthdocs-a-large-scale-synthetic-corpus-for-any-to-arab",
      "name": "Cross-Lingual SynthDocs: A Large-Scale Synthetic Corpus for Any to Arabic OCR and Document Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr",
        "diacritization",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.04699"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Cross-Lingual SynthDocs is a large-scale synthetic corpus designed to address the scarcity of Arabic resources for Optical Character Recognition.",
      "metrics": null
    },
    {
      "id": "cs-fleurs-a-massively-multilingual-and-code-switched-speech-dataset",
      "name": "CS-FLEURS: A Massively Multilingual and Code-Switched Speech Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "byan",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.14161",
        "hf": "https://huggingface.co/datasets/byan/cs-fleurs"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present CS-FLEURS, a new dataset for developing and evaluating code-switched speech recognition and translation systems beyond high-resourced languages.",
      "metrics": {
        "downloads": 802,
        "likes": 19,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "cvpd-at-qias-2025-shared-task-an-efficient-encoder-based-approach-for",
      "name": "CVPD at QIAS 2025 Shared Task: An Efficient Encoder-Based Approach for Islamic Inheritance Reasoning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.00457"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present a lightweight framework for solving multiple-choice inheritance questions using a specialised Arabic text encoder and Attentive Relevance Scoring.",
      "metrics": null
    },
    {
      "id": "dialect2sql-a-novel-text-to-sql-dataset-for-arabic-dialects-with-a-foc",
      "name": "Dialect2SQL: A Novel Text-to-SQL Dataset for Arabic Dialects with a Focus on Moroccan Darija",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.11498"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In this work, we introduce Dialect2SQL, the first large-scale, cross-domain text-to-SQL dataset in an Arabic dialect.",
      "metrics": null
    },
    {
      "id": "dialectalarabicmmlu-benchmarking-dialectal-capabilities-in-arabic-and",
      "name": "DialectalArabicMMLU: Benchmarking Dialectal Capabilities in Arabic and Multilingual Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.27543"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present DialectalArabicMMLU, a new benchmark for evaluating the performance of large language models (LLMs) across Arabic dialects.",
      "metrics": null
    },
    {
      "id": "dialg2p-dialectal-grapheme-to-phoneme-arabic-as-a-case-study",
      "name": "DialG2P: Dialectal Grapheme-to-Phoneme. Arabic as a Case Study",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "g2p"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.38/",
        "github": "https://github.com/qcri/DialG2P"
      },
      "dialects": [
        "egy"
      ],
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We introduce an end-to-end dialectal G2P for Egyptian Arabic, a dialect without standard orthography.",
      "metrics": null
    },
    {
      "id": "emohopespeech-an-annotated-dataset-of-emotions-and-hope-speech-in-engl",
      "name": "EmoHopeSpeech: An Annotated Dataset of Emotions and Hope Speech in English and Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.11959"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This research introduces a bilingual dataset comprising 23,456 entries for Arabic and 10,036 entries for English.",
      "metrics": null
    },
    {
      "id": "enhanced-arabic-text-retrieval-with-attentive-relevance-scoring",
      "name": "Enhanced Arabic Text Retrieval with Attentive Relevance Scoring",
      "type": "paper",
      "country": "INTL",
      "org": "bekhouche",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "diacritization",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.23404",
        "github": "https://github.com/Bekhouche/APR"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In this paper, we present an enhanced Dense Passage Retrieval (DPR) framework developed specifically for Arabic.",
      "metrics": null
    },
    {
      "id": "evaluating-arabic-llms-a-survey-of-benchmarks-methods-and",
      "name": "Evaluating Arabic LLMs: A Survey of Benchmarks, Methods, and Gaps",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2025,
      "venue": "arXiv 2025",
      "tasks": [
        "survey",
        "benchmark",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.13430"
      },
      "notes": "Survey of Arabic LLM benchmarks, evaluation methods and gaps.",
      "metrics": null
    },
    {
      "id": "evaluating-prompt-relevance-in-arabic-automatic-essay-scoring-insights",
      "name": "Evaluating Prompt Relevance in Arabic Automatic Essay Scoring: Insights from Synthetic and Real-World Data",
      "type": "paper",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "essay-scoring"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.13/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present the first systematic study of binary prompt-essay relevance classification, supporting both AES scoring and dataset annotation.",
      "metrics": null
    },
    {
      "id": "fann-or-flop-a-multigenre-multiera-benchmark-for-arabic-poetry-underst",
      "name": "Fann or Flop: A Multigenre, Multiera Benchmark for Arabic Poetry Understanding in LLMs",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "poetry"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.18152",
        "github": "https://github.com/mbzuai-oryx/FannOrFlop"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Arabic poetry is one of the richest and most culturally rooted forms of expression in the Arabic language, known for its layered meanings.",
      "metrics": null
    },
    {
      "id": "feature-engineering-is-not-dead-a-step-towards-state-of-the-art-for-ar",
      "name": "Feature Engineering is not Dead: A Step Towards State of the Art for Arabic Automated Essay Scoring",
      "type": "paper",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "essay-scoring"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.19/",
        "github": "https://github.com/Maroibo/AES_features"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "Experiments are conducted on a dataset of 620 Arabic essays, each annotated with both holistic and trait-specific scores.",
      "metrics": null
    },
    {
      "id": "gate-general-arabic-text-embedding-for-enhanced-sts",
      "name": "GATE: General Arabic Text Embedding for Enhanced STS",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2025,
      "venue": "arXiv 2025",
      "tasks": [
        "embedding",
        "sts"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.24581"
      },
      "notes": "Matryoshka-style general Arabic text embeddings for semantic textual similarity.",
      "metrics": null
    },
    {
      "id": "hybrid-deep-learning-and-signal-processing-for-arabic-dialect-recognit",
      "name": "Hybrid Deep Learning and Signal Processing for Arabic Dialect Recognition in Low-Resource Settings",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.21386"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Arabic dialect recognition presents a significant challenge in speech technology due to the linguistic diversity of Arabic and the scarcity.",
      "metrics": null
    },
    {
      "id": "iqra-eval-a-shared-task-on-qur-anic-pronunciation-assessment",
      "name": "Iqra’Eval: A Shared Task on Qur’anic Pronunciation Assessment",
      "type": "paper",
      "country": "INTL",
      "org": "University of Sheffield",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pronunciation",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.61/",
        "hf": "https://huggingface.co/spaces/IqraEval/"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present the findings of the first shared task on Qur’anic pronunciation assessment, which focuses on addressing the unique challenges of evaluating.",
      "metrics": null
    },
    {
      "id": "jawaher-a-multidialectal-dataset-of-arabic-proverbs-for-llm-benchmarki",
      "name": "Jawaher: A Multidialectal Dataset of Arabic Proverbs for LLM Benchmarking",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2503.00231"
      },
      "dialects": [
        "lev",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To address this, we introduce Jawaher, a benchmark designed to assess LLMs' capacity to comprehend and interpret Arabic proverbs.",
      "metrics": null
    },
    {
      "id": "kitab-bench-a-comprehensive-multi-domain-benchmark-for-arabic-ocr-and",
      "name": "KITAB-Bench: A Comprehensive Multi-Domain Benchmark for Arabic OCR and Document Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "embedding",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.14949"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present KITAB-Bench, a comprehensive Arabic OCR benchmark that fills the gaps in current evaluation systems.",
      "metrics": null
    },
    {
      "id": "konooz-multi-domain-multi-dialect-corpus-for-named-entity-recognition",
      "name": "Konooz: Multi-domain Multi-dialect Corpus for Named Entity Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.12615",
        "website": "https://sina.birzeit.edu/wojood/#download"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce Konooz, a novel multi-dimensional corpus covering 16 Arabic dialects across 10 domains, resulting in 160 distinct corpora.",
      "metrics": null
    },
    {
      "id": "laila-a-large-trait-based-dataset-for-arabic-automated-essay-scoring",
      "name": "LAILA: A Large Trait-Based Dataset for Arabic Automated Essay Scoring",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2512.24235"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce LAILA, the largest publicly available Arabic AES dataset to date.",
      "metrics": null
    },
    {
      "id": "linto-audio-and-textual-datasets-to-train-and-evaluate-automatic-speec",
      "name": "LinTO Audio and Textual Datasets to Train and Evaluate Automatic Speech Recognition in Tunisian Arabic Dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.02604"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We propose the LinTO audio and textual datasets -- comprehensive resources that capture phonological and lexical features of Tunisian Arabic Dialect.",
      "metrics": null
    },
    {
      "id": "llmvox-autoregressive-streaming-text-to-speech-model-for-any-llm",
      "name": "LLMVoX: Autoregressive Streaming Text-to-Speech Model for Any LLM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "chat",
        "tts",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2503.04724",
        "website": "https://mbzuai-oryx.github.io/LLMVoX"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In contrast, we propose LLMVoX, a lightweight 30M-parameter, LLM-agnostic, autoregressive streaming TTS system that generates high-quality speech.",
      "metrics": null
    },
    {
      "id": "macosworld-a-multilingual-interactive-benchmark-for-gui-agents",
      "name": "macOSWorld: A Multilingual Interactive Benchmark for GUI Agents",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.04135",
        "website": "https://macos-world.github.io"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To bridge the gaps, we present macOSWorld, the first comprehensive benchmark for evaluating GUI agents.",
      "metrics": null
    },
    {
      "id": "maproc-at-ahasis-shared-task-few-shot-and-sentence-transformer-for-sen",
      "name": "MAPROC at AHaSIS Shared Task: Few-Shot and Sentence Transformer for Sentiment Analysis of Arabic Hotel Reviews",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.15291"
      },
      "dialects": [
        "gulf",
        "magh",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper describes our approach to the AHaSIS shared task, which focuses on sentiment analysis on Arabic dialects in the hospitality domain.",
      "metrics": null
    },
    {
      "id": "masrad-arabic-terminology-management-corpora-with-semi-automatic-const",
      "name": "MASRAD: Arabic Terminology Management Corpora with Semi-Automatic Construction",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2503.19211"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper presents MASRAD, a terminology dataset for Arabic terminology management, and a method with supporting tools for its semi-automatic construction.",
      "metrics": null
    },
    {
      "id": "medarabiq-benchmarking-large-language-models-on-arabic-medical-tasks",
      "name": "MedArabiQ: Benchmarking Large Language Models on Arabic Medical Tasks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.03427"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Large Language Models (LLMs) have demonstrated significant promise for various applications in healthcare.",
      "metrics": null
    },
    {
      "id": "memeintel-explainable-detection-of-propagandistic-and-hateful-memes",
      "name": "MemeIntel: Explainable Detection of Propagandistic and Hateful Memes",
      "type": "paper",
      "country": "INTL",
      "org": "mohamedbayan",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.16612",
        "github": "https://github.com/MohamedBayan/MemeIntel"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce MemeXplain, an explanation-enhanced dataset for propagandistic memes in Arabic and hateful memes in English.",
      "metrics": null
    },
    {
      "id": "mind-the-gap-a-review-of-arabic-post-training-datasets-and-their-limit",
      "name": "Mind the Gap: A Review of Arabic Post-Training Datasets and Their Limitations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "summarization",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.14688"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper presents a review of publicly available Arabic post-training datasets on the Hugging Face Hub, organized along four key dimensions.",
      "metrics": null
    },
    {
      "id": "mix-minhash-and-match-cross-source-agreement-for-multilingual-pretrain",
      "name": "Mix, MinHash, and Match: Cross-Source Agreement for Multilingual Pretraining Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2512.18834",
        "hf": "https://huggingface.co/collections/AdaMLLab/mixminmatch"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In this work, we propose to use this wasteful redundancy as a quality signal to create high-quality pretraining datasets.",
      "metrics": null
    },
    {
      "id": "mizanqa-benchmarking-large-language-models-on-moroccan-legal-question",
      "name": "MizanQA: Benchmarking Large Language Models on Moroccan Legal Question Answering",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.16357"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper introduces MizanQA (pronounced Mizan, meaning \"scale\" in Arabic, a universal symbol of justice).",
      "metrics": null
    },
    {
      "id": "mole-metadata-extraction-and-validation-in-scientific-papers-using-llm",
      "name": "MOLE: Metadata Extraction and Validation in Scientific Papers Using LLMs",
      "type": "paper",
      "country": "SA",
      "org": "IVUL-KAUST",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "metadata-extraction",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.19800",
        "github": "https://github.com/IVUL-KAUST/MOLE",
        "hf": "https://huggingface.co/datasets/IVUL-KAUST/MOLE"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "MOLE: LLM framework that extracts and validates metadata attributes from papers on non-Arabic language datasets, with a new benchmark.",
      "metrics": {
        "downloads": 49,
        "likes": 1,
        "lastModified": "2025-09-20"
      }
    },
    {
      "id": "morphbpe-a-morpho-aware-tokenizer-bridging-linguistic-complexity-for-e",
      "name": "MorphBPE: A Morpho-Aware Tokenizer Bridging Linguistic Complexity for Efficient LLM Training Across Morphologies",
      "type": "paper",
      "country": "INTL",
      "org": "llm-lab-org",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.00894",
        "github": "https://github.com/llm-lab-org/MorphBPE"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce MorphBPE, a morphology-aware extension of BPE that integrates linguistic structure into subword tokenization.",
      "metrics": null
    },
    {
      "id": "mudric-multi-dialect-reasoning-for-arabic-commonsense-validation",
      "name": "MuDRiC: Multi-Dialect Reasoning for Arabic Commonsense Validation",
      "type": "paper",
      "country": "INTL",
      "org": "kareemelozeiri",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.13130",
        "github": "https://github.com/KareemElozeiri/MuDRiC"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce MuDRiC, an extended Arabic commonsense dataset incorporating multiple dialects.",
      "metrics": null
    },
    {
      "id": "multi-agent-interactive-question-generation-framework-for-long-documen",
      "name": "Multi-Agent Interactive Question Generation Framework for Long Document Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "wangk0b",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "qa",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.20145"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We propose a fully automated, multi-agent interactive framework to generate long-context questions efficiently.",
      "metrics": null
    },
    {
      "id": "multi-lingual-cyber-threat-detection-in-tweets-x-using-ml-dl-and-llm-a",
      "name": "Multi-Lingual Cyber Threat Detection in Tweets/X Using ML, DL, and LLM: A Comparative Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "mmurrad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.04346",
        "github": "https://github.com/Mmurrad/Tweet-Data-Classification.git"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Cyber threat detection has become an important area of focus in today's digital age due to the growing spread of fake information and harmful content.",
      "metrics": null
    },
    {
      "id": "multiprose-a-multi-label-arabic-dataset-for-propaganda-sentiment-and-e",
      "name": "MultiProSE: A Multi-label Arabic Dataset for Propaganda, Sentiment, and Emotion Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.08319"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Propaganda is a form of persuasion that has been used throughout history with the intention goal of influencing people's opinions through rhetorical.",
      "metrics": null
    },
    {
      "id": "munsit-at-nadi-2025-shared-task-2-pushing-the-boundaries-of-multidiale",
      "name": "Munsit at NADI 2025 Shared Task 2: Pushing the Boundaries of Multidialectal Arabic ASR with Weakly Supervised Pretraining and Continual Supervised Fine-tuning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id",
        "pretraining",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.08912"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "In this work, we present a scalable training pipeline that combines weakly supervised learning with supervised fine-tuning to develop a robust Arabic ASR model.",
      "metrics": null
    },
    {
      "id": "nilechat-towards-linguistically-diverse-and-culturally-aware-llms-for",
      "name": "NileChat: Towards Linguistically Diverse and Culturally Aware LLMs for Local Communities",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.18383",
        "github": "https://github.com/UBC-NLP/nilechat"
      },
      "dialects": [
        "egy",
        "magh",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This work proposes a methodology to create both synthetic and retrieval-based pre-training data tailored to a specific community, considering.",
      "metrics": null
    },
    {
      "id": "oasis-a-multilingual-and-multimodal-dataset-for-culturally-grounded-sp",
      "name": "OASIS: A Multilingual and Multimodal Dataset for Culturally Grounded Spoken Visual QA",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "qa",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.06371",
        "hf": "https://huggingface.co/datasets/QCRI/OASIS"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce OASIS, a large-scale culturally grounded multimodal QA dataset covering images, text, and speech.",
      "metrics": {
        "downloads": 50,
        "likes": 0,
        "lastModified": "2026-05-16"
      }
    },
    {
      "id": "on-the-origin-of-cultural-biases-in-language-models-from-pre-training",
      "name": "On The Origin of Cultural Biases in Language Models: From Pre-training Data to Linguistic Phenomena",
      "type": "paper",
      "country": "INTL",
      "org": "tareknaous",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.04662",
        "github": "https://github.com/tareknaous/camel2"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce CAMeL-2, a parallel Arabic-English benchmark of 58,086 entities associated with Arab and Western cultures and 367 masked natural contexts.",
      "metrics": null
    },
    {
      "id": "palm-a-culturally-inclusive-and-linguistically-diverse-dataset-for-ara",
      "name": "Palm: A Culturally Inclusive and Linguistically Diverse Dataset for Arabic LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "culture",
        "instructions",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2503.00151"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "PALM: community-built instruction dataset covering all 22 Arab countries in MSA and dialects to evaluate LLM cultural inclusivity.",
      "metrics": null
    },
    {
      "id": "palmx-2025-the-first-shared-task-on-benchmarking-llms-on-arabic-and-is",
      "name": "PalmX 2025: The First Shared Task on Benchmarking LLMs on Arabic and Islamic Culture",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2509.02550"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce PalmX 2025, the first shared task designed to benchmark the cultural competence of LLMs in these specific domains.",
      "metrics": null
    },
    {
      "id": "peach-a-sentence-aligned-parallel-english-arabic-corpus-for-healthcare",
      "name": "PEACH: A sentence-aligned Parallel English-Arabic Corpus for Healthcare",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.05722"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper introduces PEACH, a sentence-aligned parallel English-Arabic corpus of healthcare texts encompassing patient information leaflets.",
      "metrics": null
    },
    {
      "id": "pearl-a-multimodal-culturally-aware-arabic-instruction-dataset",
      "name": "Pearl: A Multimodal Culturally-Aware Arabic Instruction Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "instruction-tuning",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.21979"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To address this gap, we introduce PEARL, a large-scale Arabic multimodal dataset and benchmark explicitly designed for cultural understanding.",
      "metrics": null
    },
    {
      "id": "phoneme-level-mispronunciation-detection-in-quranic-recitation-using-s",
      "name": "Phoneme-level mispronunciation detection in Quranic recitation using ShallowTransformer",
      "type": "paper",
      "country": "TN",
      "org": "University of Tunis",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "pronunciation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.63/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present ShallowTransformer, a lightweight and computationally efficient transformer model leveraging Wav2vec2.0 features and trained with CTC loss.",
      "metrics": null
    },
    {
      "id": "poem-meter-classification-of-recited-arabic-poetry-integrating-high-re",
      "name": "Poem Meter Classification of Recited Arabic Poetry: Integrating High-Resource Systems for a Low-Resource Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "poetry",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.12172"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We propose a state-of-the-art framework to identify the poem meters of recited Arabic poetry.",
      "metrics": null
    },
    {
      "id": "proper-noun-diacritization-for-arabic-wikipedia-a-benchmark-dataset",
      "name": "Proper Noun Diacritization for Arabic Wikipedia: A Benchmark Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "ner",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.02656"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce a new manually diacritized dataset of Arabic proper nouns of various origins with their English Wikipedia equivalent glosses.",
      "metrics": null
    },
    {
      "id": "propxplain-can-llms-enable-explainable-propaganda-detection",
      "name": "PropXplain: Can LLMs Enable Explainable Propaganda Detection?",
      "type": "paper",
      "country": "INTL",
      "org": "firojalam",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2502.16550",
        "github": "https://github.com/firojalam/PropXplain"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To address this issue, we propose a multilingual (i.e., Arabic and English) explanation-enhanced dataset, the first of its kind.",
      "metrics": null
    },
    {
      "id": "qias-2025-overview-of-the-shared-task-on-islamic-inheritance-reasoning",
      "name": "QIAS 2025: Overview of the Shared Task on Islamic Inheritance Reasoning and Knowledge Assessment",
      "type": "paper",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "reasoning",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.117/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "This paper provides a comprehensive overview of the QIAS 2025 shared task, organized as part of the ArabicNLP 2025 conference and co-located with EMNLP 2025.",
      "metrics": null
    },
    {
      "id": "quranmorph-morphologically-annotated-quranic-corpus",
      "name": "QuranMorph: Morphologically Annotated Quranic Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.18148",
        "website": "https://sina.birzeit.edu/quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present the QuranMorph corpus, a morphologically annotated corpus for the Quran (77,429 tokens).",
      "metrics": null
    },
    {
      "id": "sage-spliced-audio-generated-data-for-enhancing-foundational-models-in",
      "name": "SAGE: Spliced-Audio Generated Data for Enhancing Foundational Models in Low-Resource Arabic-English Code-Switched Speech Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.22143"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "This paper investigates the performance of various speech SSL models on dialectal Arabic (DA) and Arabic-English code-switched (CS) speech.",
      "metrics": null
    },
    {
      "id": "sard-a-large-scale-synthetic-arabic-ocr-dataset-for-book-style-text-re",
      "name": "SARD: A Large-Scale Synthetic Arabic OCR Dataset for Book-Style Text Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.24600"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "To address this significant gap, we introduce SARD (Large-Scale Synthetic Arabic OCR Dataset).",
      "metrics": null
    },
    {
      "id": "saudi-alignment-benchmark-assessing-llms-alignment-with-cultural-norms",
      "name": "Saudi-Alignment Benchmark: Assessing LLMs Alignment with Cultural Norms and Domain Knowledge in the Saudi Context",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.11/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "To address this gap and the challenge LLMs face with non-Western cultural nuance, this study introduces the Saudi-Alignment Benchmark.",
      "metrics": null
    },
    {
      "id": "senwave-a-fine-grained-multi-language-sentiment-analysis-dataset-sourc",
      "name": "SenWave: A Fine-Grained Multi-Language Sentiment Analysis Dataset Sourced from COVID-19 Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "gitdevqiang",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "sentiment",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2510.08214",
        "github": "https://github.com/gitdevqiang/SenWave"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We introduce SenWave, a novel fine-grained multi-language sentiment analysis dataset specifically designed for analyzing COVID-19 tweets.",
      "metrics": null
    },
    {
      "id": "shawarma-chats-a-benchmark-exact-dialogue-evaluation-platter-in-egypti",
      "name": "Shawarma Chats: A Benchmark Exact Dialogue & Evaluation Platter in Egyptian, Maghrebi & Modern Standard Arabic—A Triple-Dialect Feast for Hungry Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "University of Siena",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialogue",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.39/"
      },
      "dialects": [
        "egy",
        "magh",
        "msa"
      ],
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "Benchmark of 30,000 six-turn Wikipedia-grounded dialogues in Egyptian, Maghrebi and Modern Standard Arabic.",
      "metrics": null
    },
    {
      "id": "tedxtn-a-three-way-speech-translation-corpus-for-code-switched-tunisia",
      "name": "TEDxTN: A Three-way Speech Translation Corpus for Code-Switched Tunisian Arabic - English",
      "type": "paper",
      "country": "TN",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech-translation",
        "asr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.10780"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "TEDxTN: first public Tunisian Arabic to English speech translation corpus, 108 code-switched TEDx talks with about 25 hours of speech.",
      "metrics": null
    },
    {
      "id": "tell-me-habibi-is-it-real-or-fake",
      "name": "Tell me Habibi, is it Real or Fake?",
      "type": "paper",
      "country": "INTL",
      "org": "kartik060702",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "tts",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.22581",
        "hf": "https://huggingface.co/datasets/kartik060702/ArEnAV-Full"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Deepfake generation methods are evolving fast, making fake media harder to detect and raising serious societal concerns.",
      "metrics": {
        "downloads": 32,
        "likes": 7,
        "lastModified": "2025-06-01"
      }
    },
    {
      "id": "the-arageneval-shared-task-on-arabic-authorship-style-transfer-and-ai",
      "name": "The AraGenEval Shared Task on Arabic Authorship Style Transfer and AI Generated Text Detection",
      "type": "paper",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.1/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present an overview of the AraGenEval shared task, organized as part of the ArabicNLP 2025 conference.",
      "metrics": null
    },
    {
      "id": "the-cross-lingual-cost-retrieval-biases-in-rag-over-arabic-english-cor",
      "name": "The Cross-Lingual Cost: Retrieval Biases in RAG over Arabic-English Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2507.07543"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "Finally, we propose two simple retrieval strategies that address this source of failure by enforcing equal retrieval from both languages.",
      "metrics": null
    },
    {
      "id": "the-landscape-of-arabic-large-language-models",
      "name": "The Landscape of Arabic Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2025,
      "venue": "arXiv 2025",
      "tasks": [
        "survey",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/html/2506.01340v1"
      },
      "notes": "Survey of Arabic LLMs: models, data, and evaluation to date.",
      "metrics": null
    },
    {
      "id": "tool-calling-for-arabic-llms-data-strategies-and-instruction-tuning",
      "name": "Tool Calling for Arabic LLMs: Data Strategies and Instruction Tuning",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "tool-calling"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.28/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "Tool calling is a critical capability that allows Large Language Models (LLMs) to interact with external systems, significantly expanding their utility.",
      "metrics": null
    },
    {
      "id": "towards-a-unified-benchmark-for-arabic-pronunciation-assessment-qurani",
      "name": "Towards a Unified Benchmark for Arabic Pronunciation Assessment: Quranic Recitation as Case Study",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2506.07722"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present a unified benchmark for mispronunciation detection in Modern Standard Arabic (MSA) using Qur'anic recitation as a case study.",
      "metrics": null
    },
    {
      "id": "tunifra-a-tunisian-arabic-speech-corpus-with-orthographic-transcriptio",
      "name": "TuniFra: A Tunisian Arabic Speech Corpus with Orthographic Transcriptions and French Translations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.5/"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We introduce TuniFra, a novel and comprehensive corpus developed to advance research in Automatic Speech Recognition (ASR) and Speech-to-Text Translation.",
      "metrics": null
    },
    {
      "id": "voxlect-a-speech-foundation-model-benchmark-for-modeling-dialects-and",
      "name": "Voxlect: A Speech Foundation Model Benchmark for Modeling Dialects and Regional Languages Around the Globe",
      "type": "paper",
      "country": "INTL",
      "org": "tiantiaf0627",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2508.01691",
        "github": "https://github.com/tiantiaf0627/voxlect"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present Voxlect, a novel benchmark for modeling dialects and regional languages worldwide using speech foundation models.",
      "metrics": null
    },
    {
      "id": "wasm-a-pipeline-for-constructing-structured-arabic-interleaved-multimo",
      "name": "Wasm: A Pipeline for Constructing Structured Arabic Interleaved Multimodal Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "pretraining",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2511.07080"
      },
      "year": 2025,
      "venue": "arXiv 2025",
      "notes": "We present our pipeline Wasm for processing the Common Crawl dataset to create a new Arabic multimodal dataset that uniquely provides markdown output.",
      "metrics": null
    },
    {
      "id": "zero-shot-and-fine-tuned-evaluation-of-generative-llms-for-arabic-word",
      "name": "Zero-Shot and Fine-Tuned Evaluation of Generative LLMs for Arabic Word Sense Disambiguation",
      "type": "paper",
      "country": "SD",
      "org": "University of Khartoum",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-main.24/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "This paper benchmarks large generative language models (LLMs) for Arabic Word Sense Disambiguation (WSD) under both zero-shot and fine-tuning conditions.",
      "metrics": null
    },
    {
      "id": "arabicmmlu-assessing-massive-multitask-language-understandin",
      "name": "ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.12840"
      },
      "year": 2024,
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "citations": 120,
      "notes": "The focus of language model evaluation has transitioned towards reasoning and knowledge-intensive tasks, driven by advancements in pretraining large models.",
      "metrics": null
    },
    {
      "id": "real-time-arabic-sign-language-recognition-using-a-hybrid-de",
      "name": "Real-Time Arabic Sign Language Recognition Using a Hybrid Deep Learning Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/s24113683"
      },
      "year": 2024,
      "venue": "Italian National Conference on Sensors",
      "citations": 67,
      "notes": "Sign language is an essential means of communication for individuals with hearing disabilities.",
      "metrics": null
    },
    {
      "id": "darijabert-a-step-forward-in-nlp-for-the-written-moroccan-di",
      "name": "DarijaBERT: a step forward in NLP for the written Moroccan dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s41060-023-00498-2"
      },
      "year": 2024,
      "venue": "International Journal of Data Science and Analysis",
      "citations": 58,
      "notes": "The established performance of existing transformer-based language models, delivering state-of-the-art results on numerous downstream tasks, is noteworthy.",
      "metrics": null
    },
    {
      "id": "sentiment-analysis-of-arabic-social-media-texts-a-machine-le",
      "name": "Sentiment analysis of Arabic social media texts: A machine learning approach to deciphering customer perceptions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.heliyon.2024.e27863"
      },
      "year": 2024,
      "venue": "Heliyon",
      "citations": 53,
      "notes": "Sentiment analysis (SA) is a subfield of artificial intelligence that entails natural language processing.",
      "metrics": null
    },
    {
      "id": "improving-neural-machine-translation-for-low-resource-langua",
      "name": "Improving neural machine translation for low resource languages through non-parallel corpora: a case study of Egyptian dialect to modern standard Arabic translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1038/s41598-023-51090-4"
      },
      "year": 2024,
      "venue": "Scientific Reports",
      "citations": 46,
      "notes": "Machine translation for low-resource languages poses significant challenges, primarily due to the limited availability of data.",
      "metrics": null
    },
    {
      "id": "sentiment-analysis-for-arabic-call-center-notes-using-machin",
      "name": "Sentiment analysis for Arabic call center notes using machine learning techniques",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.32629/jai.v7i3.940"
      },
      "year": 2024,
      "venue": "Journal of Autonomous Intelligence",
      "citations": 44,
      "notes": "Call centers handle thousands of incoming calls daily, encompassing a diverse array of categories including product inquiries, complaints, and more.",
      "metrics": null
    },
    {
      "id": "arabic-sarcasm-detection-an-enhanced-fine-tuned-language-mod",
      "name": "Arabic sarcasm detection: An enhanced fine-tuned language model approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.asej.2024.102736"
      },
      "year": 2024,
      "venue": "Ain Shams Engineering Journal",
      "citations": 43,
      "metrics": null
    },
    {
      "id": "deberta-bilstm-a-multi-label-classification-model-of-arabic",
      "name": "DeBERTa-BiLSTM: A multi-label classification model of Arabic medical questions using pre-trained models and deep learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.compbiomed.2024.107921"
      },
      "year": 2024,
      "venue": "Comput. Biol. Medicine",
      "citations": 41,
      "notes": "It is wise to investigate past and present epidemics in the hopes of profiting from them and being better prepared for future ones.",
      "metrics": null
    },
    {
      "id": "unveiling-the-new-frontier-chatgpt-3-powered-translation-for",
      "name": "Unveiling the New Frontier: ChatGPT-3 Powered Translation for Arabic-English Language Pairs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.17507/tpls.1402.05"
      },
      "year": 2024,
      "venue": "Theory and Practice in Language Studies",
      "citations": 39,
      "notes": "This study evaluates the aptitude of ChatGPT for Arabic-English machine translation.",
      "metrics": null
    },
    {
      "id": "empowering-communication-a-deep-learning-framework-for-arabi",
      "name": "Empowering Communication: A Deep Learning Framework for Arabic Sign Language Recognition with an Attention Mechanism",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/computers13060153"
      },
      "year": 2024,
      "venue": "De Computis",
      "citations": 38,
      "notes": "This article emphasises the urgent need for appropriate communication tools for communities of people who are deaf or hard-of-hearing, with a specific emphasis",
      "metrics": null
    },
    {
      "id": "from-text-to-insight-an-integrated-cnn-bilstm-gru-model-for",
      "name": "From Text to Insight: An Integrated CNN-BiLSTM-GRU Model for Arabic Cyberbullying Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2024.3431939"
      },
      "year": 2024,
      "venue": "IEEE Access",
      "citations": 38,
      "notes": "Several research on cyberbullying detection have employed different deep learning and machine learning methodologies to achieve promising outcomes.",
      "metrics": null
    },
    {
      "id": "adocrnet-a-deep-learning-ocr-for-arabic-documents-recognitio",
      "name": "ADOCRNet: A Deep Learning OCR for Arabic Documents Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2024.3379530"
      },
      "year": 2024,
      "venue": "IEEE Access",
      "citations": 37,
      "notes": "In recent years, Optical character recognition (OCR) has experienced a resurgence of interest especially for contemporary Arabic data.",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis-of-monkeypox-using-deep-neural-net",
      "name": "Arabic sentiment analysis of Monkeypox using deep neural network and optimized hyperparameters of machine learning algorithms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13278-023-01188-4"
      },
      "year": 2024,
      "venue": "Social Network Analysis and Mining",
      "citations": 37,
      "metrics": null
    },
    {
      "id": "advancements-and-challenges-in-arabic-sentiment-analysis-a-d",
      "name": "Advancements and challenges in Arabic sentiment analysis: A decade of methodologies, applications, and resource development",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.heliyon.2024.e39786"
      },
      "year": 2024,
      "venue": "Heliyon",
      "citations": 36,
      "notes": "The exponential growth of digital information, particularly user-generated content on social media and blogging platforms, has underscored the importance of",
      "metrics": null
    },
    {
      "id": "language-discrepancies-in-the-performance-of-generative-arti",
      "name": "Language discrepancies in the performance of generative artificial intelligence models: an examination of infectious disease queries in English and Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1186/s12879-024-09725-y"
      },
      "year": 2024,
      "venue": "BMC Infectious Diseases",
      "citations": 35,
      "notes": "This study aimed to compare AI model efficiency in English and Arabic for infectious disease queries.",
      "metrics": null
    },
    {
      "id": "translating-classical-arabic-verse-human-translation-vs-ai-l",
      "name": "Translating classical Arabic verse: human translation vs. AI large language models (Gemini and ChatGPT)",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1080/23311886.2024.2410998"
      },
      "year": 2024,
      "venue": "Cogent Social Sciences",
      "citations": 35,
      "notes": "This paper examines the translation of 15 individual Classical Arabic verses by comparing the English renditions provided by a human translator and two",
      "metrics": null
    },
    {
      "id": "sada-saudi-audio-dataset-for-arabic",
      "name": "SADA: Saudi Audio Dataset for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ICASSP48485.2024.10446243"
      },
      "year": 2024,
      "venue": "IEEE International Conference on Acoustics, Speech, and Signal Processing",
      "citations": 34,
      "notes": "This paper introduces SADA, the Saudi Audio Dataset for Arabic, with 668 hours of high-quality audio suitable for supervised training.",
      "metrics": null
    },
    {
      "id": "ai-generated-text-detector-for-arabic-language-using-encoder",
      "name": "AI-Generated Text Detector for Arabic Language Using Encoder-Based Transformer Architecture",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/bdcc8030032"
      },
      "year": 2024,
      "venue": "Big Data and Cognitive Computing",
      "citations": 32,
      "notes": "This study introduces a novel AI text classifier designed specifically for Arabic, tackling the distinct challenges inherent in processing this language.",
      "metrics": null
    },
    {
      "id": "user-satisfaction-with-arabic-covid-19-apps-sentiment-analys",
      "name": "User satisfaction with Arabic COVID-19 apps: Sentiment analysis of users' reviews using machine learning techniques",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.ipm.2024.103644"
      },
      "year": 2024,
      "venue": "Information Processing & Management",
      "citations": 32,
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-in-social-media-context-using-wor",
      "name": "Arabic Fake News Detection in Social Media Context Using Word Embeddings and Pre-trained Transformers",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13369-024-08959-x"
      },
      "year": 2024,
      "venue": "The Arabian journal for science and engineering",
      "citations": 30,
      "metrics": null
    },
    {
      "id": "hate-speech-detection-with-adhar-a-multi-dialectal-hate-spee",
      "name": "Hate speech detection with ADHAR: a multi-dialectal hate speech corpus in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.3389/frai.2024.1391472"
      },
      "year": 2024,
      "venue": "Frontiers Artif. Intell.",
      "citations": 30,
      "notes": "Hate speech detection in Arabic poses a complex challenge due to the dialectal diversity across the Arab world.",
      "metrics": null
    },
    {
      "id": "a-tinydl-model-for-gesture-based-air-handwriting-arabic-numb",
      "name": "A TinyDL Model for Gesture-Based Air Handwriting Arabic Numbers and Simple Arabic Letters Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2024.3406631"
      },
      "year": 2024,
      "venue": "IEEE Access",
      "citations": 29,
      "notes": "The application of tiny machine learning (TinyML) in human-computer interaction is revolutionizing gesture recognition technologies.",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis-for-chatgpt-using-machine-learning",
      "name": "Arabic Sentiment Analysis for ChatGPT Using Machine Learning Classification Algorithms: A Hyperparameter Optimization Technique",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1145/3638285"
      },
      "year": 2024,
      "venue": "ACM Trans. Asian Low Resour. Lang. Inf. Process.",
      "citations": 29,
      "notes": "This study centers on ChatGPT, a popular machine learning model engaging in dialogues with users, garnering attention for its exceptional performance and",
      "metrics": null
    },
    {
      "id": "systematic-review-of-english-arabic-machine-translation-post",
      "name": "Systematic Review of English/Arabic Machine Translation Postediting: Implications for AI Application in Translation Research and Pedagogy",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "translation"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/informatics11020023"
      },
      "year": 2024,
      "venue": "Informatics",
      "citations": 29,
      "notes": "The twenty-first century has witnessed an extensive evolution in translation practice thanks to the accelerated progress in machine translation tools and",
      "metrics": null
    },
    {
      "id": "the-performance-of-openai-chatgpt-4-and-google-gemini-in-vir",
      "name": "The performance of OpenAI ChatGPT-4 and Google Gemini in virology multiple-choice questions: a comparative analysis of English and Arabic responses",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1186/s13104-024-06920-7"
      },
      "year": 2024,
      "venue": "BMC Research Notes",
      "citations": 29,
      "notes": "The integration of artificial intelligence (AI) in healthcare education is inevitable.",
      "metrics": null
    },
    {
      "id": "bert-based-model-for-aspect-based-sentiment-analysis-for-ana",
      "name": "BERT-Based Model for Aspect-Based Sentiment Analysis for Analyzing Arabic Open-Ended Survey Responses: A Case Study",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2023.3348342"
      },
      "year": 2024,
      "venue": "IEEE Access",
      "citations": 28,
      "notes": "Educational institutions typically gather feedback from beneficiaries through formal surveys.",
      "metrics": null
    },
    {
      "id": "dallah-a-dialect-aware-multimodal-large-language-model-for-a",
      "name": "Dallah: A Dialect-Aware Multimodal Large Language Model for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.18129"
      },
      "year": 2024,
      "venue": "ARABICNLP",
      "citations": 28,
      "notes": "Recent advancements have significantly enhanced the capabilities of Multimodal Large Language Models (MLLMs) in generating and understanding image-to-text",
      "metrics": null
    },
    {
      "id": "detection-of-arabic-offensive-language-in-social-media-using",
      "name": "Detection of Arabic offensive language in social media using machine learning models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.iswa.2024.200376"
      },
      "year": 2024,
      "venue": "Intelligent Systems with Applications",
      "citations": 28,
      "metrics": null
    },
    {
      "id": "arabic-sign-language-letters-recognition-using-vision-transf",
      "name": "Arabic sign language letters recognition using Vision Transformer",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s11042-024-18681-3"
      },
      "year": 2024,
      "venue": "Multimedia tools and applications",
      "citations": 27,
      "metrics": null
    },
    {
      "id": "evaluating-chatgpt-performance-in-arabic-dialects-a-comparat",
      "name": "Evaluating ChatGPT performance in Arabic dialects: A comparative study showing defects in responding to Jordanian and Tunisian general health prompts",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.58496/mjaih/2024/001"
      },
      "year": 2024,
      "venue": "Mesopotamian Journal of Artificial Intelligence in Healthcare",
      "citations": 27,
      "notes": "Background: The role of artificial intelligence (AI) is increasingly recognized to enhance digital health literacy.",
      "metrics": null
    },
    {
      "id": "linguistic-feature-fusion-for-arabic-fake-news-detection-and",
      "name": "Linguistic feature fusion for Arabic fake news detection and named entity recognition using reinforcement learning and swarm optimization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.neucom.2024.128078"
      },
      "year": 2024,
      "venue": "Neurocomputing",
      "citations": 26,
      "metrics": null
    },
    {
      "id": "a-tinyml-model-for-gesture-based-air-handwriting-arabic-numb",
      "name": "A TinyML Model for Gesture-Based Air Handwriting Arabic Numbers Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2024.05.070"
      },
      "year": 2024,
      "venue": "Procedia Computer Science",
      "citations": 25,
      "metrics": null
    },
    {
      "id": "enhancing-arabic-dialect-detection-on-social-media-a-hybrid",
      "name": "Enhancing Arabic Dialect Detection on Social Media: A Hybrid Model with an Attention Mechanism",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/info15060316"
      },
      "year": 2024,
      "venue": "Inf.",
      "citations": 25,
      "notes": "Recently, the widespread use of social media and easy access to the Internet have brought about a significant transformation in the type of textual data",
      "metrics": null
    },
    {
      "id": "hatformer-historic-handwritten-arabic-text-recognition-with",
      "name": "HATFormer: Historic Handwritten Arabic Text Recognition with Transformers",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.02179"
      },
      "year": 2024,
      "venue": "arXiv.org",
      "citations": 25,
      "notes": "Arabic handwritten text recognition (HTR) is challenging, especially for historical texts, due to diverse writing styles and the intrinsic features of Arabic",
      "metrics": null
    },
    {
      "id": "a-comprehensive-analysis-of-various-tokenizers-for-arabic-la",
      "name": "A Comprehensive Analysis of Various Tokenizers for Arabic Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app14135696"
      },
      "year": 2024,
      "venue": "Applied Sciences",
      "citations": 24,
      "notes": "Pretrained language models have achieved great success in various natural language understanding (NLU) tasks due to their capacity to capture deep",
      "metrics": null
    },
    {
      "id": "arabbert-lstm-improving-arabic-sentiment-analysis-based-on-t",
      "name": "ArabBert-LSTM: improving Arabic sentiment analysis based on transformer model and Long Short-Term Memory",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3389/frai.2024.1408845"
      },
      "year": 2024,
      "venue": "Frontiers Artif. Intell.",
      "citations": 24,
      "notes": "Sentiment analysis also referred to as opinion mining, plays a significant role in automating the identification of negative, positive, or neutral sentiments",
      "metrics": null
    },
    {
      "id": "efhamni-a-deep-learning-based-saudi-sign-language-recognitio",
      "name": "Efhamni: A Deep Learning-Based Saudi Sign Language Recognition Application",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/s24103112"
      },
      "year": 2024,
      "venue": "Italian National Conference on Sensors",
      "citations": 24,
      "notes": "Deaf and hard-of-hearing people mainly communicate using sign language, which is a set of signs made using hand gestures combined with facial expressions to",
      "metrics": null
    },
    {
      "id": "a-machine-learning-approach-to-cyberbullying-detection-in-ar",
      "name": "A Machine Learning Approach to Cyberbullying Detection in Arabic Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.32604/cmc.2024.048003"
      },
      "year": 2024,
      "venue": "Comput. Mater. Continua",
      "citations": 23,
      "notes": "With the rapid growth of internet usage, a new situation has been created that enables practicing bullying.",
      "metrics": null
    },
    {
      "id": "error-analysis-of-pretrained-language-models-plms-in-english",
      "name": "Error Analysis of Pretrained Language Models (PLMs) in English-to-Arabic Machine Translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s44230-024-00061-7"
      },
      "year": 2024,
      "venue": "Human-Centric Intelligent Systems",
      "citations": 23,
      "notes": "Advances in neural machine translation utilizing pretrained language models (PLMs) have shown promise in improving the translation quality between diverse",
      "metrics": null
    },
    {
      "id": "multiscale-cascaded-domain-based-approach-for-arabic-fake-re",
      "name": "Multiscale cascaded domain-based approach for Arabic fake reviews detection in e-commerce platforms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2024.101926"
      },
      "year": 2024,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 23,
      "notes": "This study addresses this gap by introducing a full-gold standard dataset, the",
      "metrics": null
    },
    {
      "id": "toward-robust-arabic-ai-generated-text-detection-tackling-di",
      "name": "Toward Robust Arabic AI-Generated Text Detection: Tackling Diacritics Challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/info15070419"
      },
      "year": 2024,
      "venue": "Inf.",
      "citations": 23,
      "notes": "This study introduces robust Arabic text detection models using Transformer-based",
      "metrics": null
    },
    {
      "id": "aratrust-an-evaluation-of-trustworthiness-for-llms-in-arabic",
      "name": "AraTrust: An Evaluation of Trustworthiness for LLMs in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.09017"
      },
      "year": 2024,
      "venue": "International Conference on Computational Linguistics",
      "citations": 22,
      "notes": "The swift progress and widespread acceptance of artificial intelligence (AI) systems highlight a pressing requirement to comprehend both the capabilities and",
      "metrics": null
    },
    {
      "id": "enhancedbert-a-feature-rich-ensemble-model-for-arabic-word-s",
      "name": "EnhancedBERT: A feature-rich ensemble model for Arabic word sense disambiguation with statistical analysis and optimized data collection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2023.101911"
      },
      "year": 2024,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 22,
      "notes": "Accurate assignment of meaning to a word based on its context, known as Word Sense Disambiguation (WSD), remains challenging across languages.",
      "metrics": null
    },
    {
      "id": "evaluating-arabic-emotion-recognition-task-using-chatgpt-mod",
      "name": "Evaluating Arabic Emotion Recognition Task Using ChatGPT Models: A Comparative Analysis between Emotional Stimuli Prompt, Fine-Tuning, and In-Context Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/jtaer19020058"
      },
      "year": 2024,
      "venue": "Journal of Theoretical and Applied Electronic Commerce Research",
      "citations": 22,
      "notes": "Textual emotion recognition (TER) has significant commercial potential since it can be used as an excellent tool to monitor a brand/business reputation",
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-advancement-and-challenges",
      "name": "Arabic Speech Recognition: Advancement and Challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2024.3376237"
      },
      "year": 2024,
      "venue": "IEEE Access",
      "citations": 21,
      "notes": "Speech recognition is a captivating process that revolutionizes human-computer interactions, allowing us to interact and control machines through spoken",
      "metrics": null
    },
    {
      "id": "cnn-based-methods-for-offline-arabic-handwriting-recognition",
      "name": "CNN-based Methods for Offline Arabic Handwriting Recognition: A Review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "survey",
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s11063-024-11544-w"
      },
      "year": 2024,
      "venue": "Neural Processing Letters",
      "citations": 21,
      "notes": "Arabic Handwriting Recognition (AHR) is a complex task involving the transformation of handwritten Arabic text from image format into machine-readable data",
      "metrics": null
    },
    {
      "id": "tayseer-a-novel-ai-powered-arabic-chatbot-framework-for-tech",
      "name": "Tayseer: A Novel AI-Powered Arabic Chatbot Framework for Technical and Vocational Student Helpdesk Services and Enhancing Student Interactions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "benchmark",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app14062547"
      },
      "year": 2024,
      "venue": "Applied Sciences",
      "citations": 21,
      "notes": "The rise of conversational agents (CAs) like chatbots in education has increased the demand for advisory services.",
      "metrics": null
    },
    {
      "id": "a-brief-review-on-preprocessing-text-in-arabic-language-data",
      "name": "A Brief Review on Preprocessing Text in Arabic Language Dataset: Techniques and Challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "preprocessing",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.58496/bjai/2024/007"
      },
      "year": 2024,
      "venue": "Babylonian Journal of Artificial Intelligence",
      "citations": 20,
      "notes": "The preprocessing of Arabic text still presents unique challenges due to the language's rich morphology, complex grammar, and",
      "metrics": null
    },
    {
      "id": "breaking-language-barriers-with-chatgpt-enhancing-low-resour",
      "name": "Breaking language barriers with ChatGPT: enhancing low-resource machine translation between algerian arabic and MSA",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s41870-024-01926-7"
      },
      "year": 2024,
      "venue": "International journal of information technology",
      "citations": 20,
      "metrics": null
    },
    {
      "id": "embedding-search-for-quranic-texts-based-on-large-language-m",
      "name": "Embedding search for quranic texts based on large language models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "embedding",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.34028/21/2/7"
      },
      "year": 2024,
      "venue": "˜The œinternational Arab journal of information technology",
      "citations": 20,
      "notes": "This paper is introduced in order to explore the use of large language models for semantic search of Quranic texts.",
      "metrics": null
    },
    {
      "id": "neural-multi-task-learning-for-end-to-end-arabic-aspect-base",
      "name": "Neural multi-task learning for end-to-end Arabic aspect-based sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.csl.2024.101683"
      },
      "year": 2024,
      "venue": "Computer Speech and Language",
      "citations": 20,
      "metrics": null
    },
    {
      "id": "word-embedding-as-a-semantic-feature-extraction-technique-in",
      "name": "Word embedding as a semantic feature extraction technique in arabic natural language processing: an overview",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "embedding"
      ],
      "links": {
        "paper": "https://doi.org/10.34028/21/2/13"
      },
      "year": 2024,
      "venue": "˜The œinternational Arab journal of information technology",
      "citations": 20,
      "notes": "Feature extraction has transformed the field of Natural Language Processing (NLP) by providing an effective way to represent linguistic features.",
      "metrics": null
    },
    {
      "id": "a-comprehensive-review-on-arabic-offensive-language-and-hate",
      "name": "A comprehensive review on Arabic offensive language and hate speech detection on social media: methods, challenges and solutions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13278-024-01258-1"
      },
      "year": 2024,
      "venue": "Social Network Analysis and Mining",
      "citations": 19,
      "metrics": null
    },
    {
      "id": "alignment-at-pre-training-towards-native-alignment-for-arabi",
      "name": "Alignment at Pre-training! Towards Native Alignment for Arabic LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.03253"
      },
      "year": 2024,
      "venue": "Neural Information Processing Systems",
      "citations": 19,
      "notes": "Traditional approaches focus on aligning models during the instruction tuning or reinforcement learning stages, referred to in this paper as `post alignment'.",
      "metrics": null
    },
    {
      "id": "aracovtexfinder-leveraging-the-transformer-based-language-mo",
      "name": "AraCovTexFinder: Leveraging the transformer-based language model for Arabic COVID-19 text identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.engappai.2024.107987"
      },
      "year": 2024,
      "venue": "Engineering applications of artificial intelligence",
      "citations": 19,
      "metrics": null
    },
    {
      "id": "code-mixing-unveiled-enhancing-the-hate-speech-detection-in",
      "name": "Code-mixing unveiled: Enhancing the hate speech detection in Arabic dialect tweets using machine learning models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1371/journal.pone.0305657"
      },
      "year": 2024,
      "venue": "PLoS ONE",
      "citations": 19,
      "notes": "Technological developments over the past few decades have changed the way people communicate, with platforms like social media and blogs becoming vital channels",
      "metrics": null
    },
    {
      "id": "exploring-retrieval-augmented-generation-in-arabic",
      "name": "Exploring Retrieval Augmented Generation in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "benchmark",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2408.07425"
      },
      "year": 2024,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 19,
      "notes": "Recently, Retrieval Augmented Generation (RAG) has emerged as a powerful technique in natural language processing, combining the strengths of retrieval-based",
      "metrics": null
    },
    {
      "id": "low-resource-arabic-dialects-transformer-neural-machine-tran",
      "name": "Low Resource Arabic Dialects Transformer Neural Machine Translation Improvement through Incremental Transfer of Shared Linguistic Features",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13369-023-08543-9"
      },
      "year": 2024,
      "venue": "The Arabian journal for science and engineering",
      "citations": 19,
      "metrics": null
    },
    {
      "id": "natural-language-processing-for-arabic-sentiment-analysis-a",
      "name": "Natural Language Processing for Arabic Sentiment Analysis: A Systematic Literature Review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/TBDATA.2024.3366083"
      },
      "year": 2024,
      "venue": "IEEE Transactions on Big Data",
      "citations": 19,
      "notes": "Sentiment analysis involves using computational methods to identify and classify opinions expressed in text, with the goal of determining whether the writer's",
      "metrics": null
    },
    {
      "id": "novel-approach-for-arabic-fake-news-classification-using-emb",
      "name": "Novel approach for Arabic fake news classification using embedding from large language features with CNN-LSTM ensemble model and explainable AI",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1038/s41598-024-82111-5"
      },
      "year": 2024,
      "venue": "Scientific Reports",
      "citations": 19,
      "notes": "This study addresses this critical issue by advancing fake news detection in Arabic and overcoming limitations in existing approaches.",
      "metrics": null
    },
    {
      "id": "optimizing-large-language-models-for-arabic-healthcare-commu",
      "name": "Optimizing Large Language Models for Arabic Healthcare Communication: A Focus on Patient-Centered NLP Applications",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/bdcc8110157"
      },
      "year": 2024,
      "venue": "Big Data and Cognitive Computing",
      "citations": 19,
      "notes": "Recent studies have highlighted the growing integration of Natural Language Processing (NLP) techniques and Large Language Models (LLMs) in healthcare.",
      "metrics": null
    },
    {
      "id": "al-qasida-analyzing-llm-quality-and-accuracy-systematically",
      "name": "AL-QASIDA: Analyzing LLM Quality and Accuracy Systematically in Dialectal Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.04193"
      },
      "year": 2024,
      "venue": "arXiv.org",
      "citations": 18,
      "notes": "Dialectal Arabic (DA) varieties are under-served by language technologies, particularly large language models (LLMs).",
      "metrics": null
    },
    {
      "id": "an-arabic-visual-speech-recognition-framework-with-cnn-and-v",
      "name": "An arabic visual speech recognition framework with CNN and vision transformers for lipreading",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "asr",
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s11042-024-18237-5"
      },
      "year": 2024,
      "venue": "Multimedia tools and applications",
      "citations": 18,
      "metrics": null
    },
    {
      "id": "arabic-automatic-speech-recognition-challenges-and-progress",
      "name": "Arabic Automatic Speech Recognition: Challenges and Progress",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.specom.2024.103110"
      },
      "year": 2024,
      "venue": "Speech Communication",
      "citations": 18,
      "metrics": null
    },
    {
      "id": "arabic-synonym-bert-based-adversarial-examples-for-text-clas",
      "name": "Arabic Synonym BERT-based Adversarial Examples for Text Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.03477"
      },
      "year": 2024,
      "venue": "Conference of the European Chapter of the Association for Computational Linguistics",
      "citations": 18,
      "notes": "Text classification systems have been proven vulnerable to adversarial text examples, modified versions of the original text examples that are often unnoticed",
      "metrics": null
    },
    {
      "id": "deep-learning-ensemble-and-supervised-machine-learning-for-a",
      "name": "Deep Learning, Ensemble and Supervised Machine Learning for Arabic Speech Emotion Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.48084/etasr.7134"
      },
      "year": 2024,
      "venue": "Engineering, Technology &amp; Applied Science Research",
      "citations": 18,
      "notes": "Today, automatic emotion recognition in speech is one of the most important areas of research in signal processing.",
      "metrics": null
    },
    {
      "id": "exploring-challenges-in-audiovisual-translation-a-comparativ",
      "name": "Exploring challenges in audiovisual translation: A comparative analysis of human- and AI-generated Arabic subtitles in Birdman",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1371/journal.pone.0311020"
      },
      "year": 2024,
      "venue": "PLoS ONE",
      "citations": 18,
      "notes": "Movies often use allusions to add depth, create connections, and enrich the storytelling.",
      "metrics": null
    },
    {
      "id": "optimised-cnn-architectures-for-handwritten-arabic-character",
      "name": "Optimised CNN Architectures for Handwritten Arabic Character Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.32604/cmc.2024.052016"
      },
      "year": 2024,
      "venue": "Computers, Materials &amp; Continua",
      "citations": 18,
      "notes": "Handwritten character recognition is considered challenging compared with machine-printed characters due to the different human writing styles.",
      "metrics": null
    },
    {
      "id": "stanceeval-2024-the-first-arabic-stance-detection-shared-tas",
      "name": "StanceEval 2024: The First Arabic Stance Detection Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.88/"
      },
      "year": 2024,
      "venue": "ARABICNLP",
      "citations": 18,
      "notes": "Recently, there has been a growing interest in analyzing user-generated text to understand opinions expressed on social media.",
      "metrics": null
    },
    {
      "id": "a-bidirectional-arabic-sign-language-framework-using-deep-le",
      "name": "A Bidirectional Arabic Sign Language Framework Using Deep Learning and Fuzzy Matching Score",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/math12081155"
      },
      "year": 2024,
      "venue": "Mathematics",
      "citations": 17,
      "notes": "Sign language is widely used to facilitate the communication process between deaf people and their surrounding environment.",
      "metrics": null
    },
    {
      "id": "a-systematic-literature-review-of-hate-speech-identification",
      "name": "A systematic literature review of hate speech identification on Arabic Twitter data: research challenges and future directions",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "hate-speech"
      ],
      "links": {
        "paper": "https://doi.org/10.7717/peerj-cs.1966"
      },
      "year": 2024,
      "venue": "PeerJ Computer Science",
      "citations": 17,
      "notes": "The automatic speech identification in Arabic tweets has generated substantial attention among academics in the fields of text mining and natural language",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification-in-social-media-a-hybrid-model",
      "name": "Arabic dialect identification in social media: A hybrid model with transformer models and BiLSTM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.heliyon.2024.e36280"
      },
      "year": 2024,
      "venue": "Heliyon",
      "citations": 17,
      "notes": "Therefore, this study aims to use transformers to address the issue of ADI on social media.",
      "metrics": null
    },
    {
      "id": "bert-based-arabic-diacritization-a-state-of-the-art-approach",
      "name": "BERT-Based Arabic Diacritization: A state-of-the-art approach for improving text accuracy and pronunciation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.eswa.2024.123416"
      },
      "year": 2024,
      "venue": "Expert systems with applications",
      "citations": 17,
      "metrics": null
    },
    {
      "id": "artificial-intelligence-generated-arabic-subtitles-insights",
      "name": "Artificial intelligence-generated Arabic subtitles: insights from Veed.io’s automatic speech recognition system of Jordanian Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "paper": "https://doi.org/10.1590/1983-3652.2024.46952"
      },
      "year": 2024,
      "venue": "Texto Livre",
      "citations": 16,
      "notes": "This paper examines the errors that the automatic speech recognition (ASR) system of Veed.io produces when transcribing utterances spoken in Jordanian Arabic",
      "metrics": null
    },
    {
      "id": "enhancing-arabic-sentiment-analysis-of-consumer-reviews-mach",
      "name": "Enhancing Arabic Sentiment Analysis of Consumer Reviews: Machine Learning and Deep Learning Methods Based on NLP",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/a17110495"
      },
      "year": 2024,
      "venue": "Algorithms",
      "citations": 16,
      "notes": "Sentiment analysis utilizes Natural Language Processing (NLP) techniques to extract opinions from text, which is critical for businesses looking to refine",
      "metrics": null
    },
    {
      "id": "integrated-multi-layer-perceptron-neural-network-and-novel-f",
      "name": "Integrated multi-layer perceptron neural network and novel feature extraction for handwritten Arabic recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.5267/j.ijdns.2024.3.015"
      },
      "year": 2024,
      "venue": "International Journal of Data and Network Science",
      "citations": 16,
      "notes": "Arabic handwritten script recognition presents an energetic area of study.",
      "metrics": null
    },
    {
      "id": "pre-trained-language-model-ensemble-for-arabic-fake-news-det",
      "name": "Pre-Trained Language Model Ensemble for Arabic Fake News Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/math12182941"
      },
      "year": 2024,
      "venue": "Mathematics",
      "citations": 16,
      "notes": "Fake news detection (FND) remains a challenge due to its vast and varied sources, especially on social media platforms.",
      "metrics": null
    },
    {
      "id": "a-hybrid-combination-of-cnn-attention-with-optimized-random",
      "name": "A hybrid combination of CNN Attention with optimized random forest with grey wolf optimizer to discriminate between Arabic hateful, abusive tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2024.101961"
      },
      "year": 2024,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 15,
      "notes": "Arabic hateful speech recognition has long been a major area of focus in Natural Language Processing (NLP) research.",
      "metrics": null
    },
    {
      "id": "a-robust-classification-approach-to-enhance-clinic-identific",
      "name": "A robust classification approach to enhance clinic identification from Arabic health text",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s00521-024-09453-z"
      },
      "year": 2024,
      "venue": "Neural computing & applications (Print)",
      "citations": 15,
      "metrics": null
    },
    {
      "id": "an-analysis-of-customer-perception-using-lexicon-based-senti",
      "name": "An analysis of customer perception using lexicon-based sentiment analysis of Arabic Texts framework",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.heliyon.2024.e30320"
      },
      "year": 2024,
      "venue": "Heliyon",
      "citations": 15,
      "notes": "Sentiment Analysis (SA) employing Natural Language Processing (NLP) is pivotal in determining the positivity and negativity of customer feedback.",
      "metrics": null
    },
    {
      "id": "enhancing-arabic-cyberbullying-detection-with-end-to-end-tra",
      "name": "Enhancing Arabic Cyberbullying Detection with End-to-End Transformer Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.32604/cmes.2024.052291"
      },
      "year": 2024,
      "venue": "Computer Modeling in Engineering &amp; Sciences",
      "citations": 15,
      "notes": "To tackle this challenge, our study introduces a new approach employing Bidirectional Encoder Representations from the",
      "metrics": null
    },
    {
      "id": "enhancing-communication-deep-learning-for-arabic-sign-langua",
      "name": "Enhancing communication: Deep learning for Arabic sign language translation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "translation",
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1515/eng-2024-0025"
      },
      "year": 2024,
      "venue": "Open Engineering",
      "citations": 15,
      "notes": "This study explores the field of sign language recognition through machine learning, focusing on the development and comparative evaluation of various",
      "metrics": null
    },
    {
      "id": "enhancing-semantic-similarity-understanding-in-arabic-nlp-wi",
      "name": "Enhancing Semantic Similarity Understanding in Arabic NLP with Nested Embedding Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.21139"
      },
      "year": 2024,
      "venue": "arXiv.org",
      "citations": 15,
      "notes": "This work presents a novel framework for training Arabic nested embedding models through Matryoshka Embedding Learning, leveraging multilingual",
      "metrics": null
    },
    {
      "id": "machine-learning-approach-for-arabic-handwritten-recognition",
      "name": "Machine Learning Approach for Arabic Handwritten Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app14199020"
      },
      "year": 2024,
      "venue": "Applied Sciences",
      "citations": 15,
      "notes": "Text recognition is an important area of the pattern recognition field.",
      "metrics": null
    },
    {
      "id": "mtl-arabert-an-enhanced-multi-task-learning-model-for-arabic",
      "name": "MTL-AraBERT: An Enhanced Multi-Task Learning Model for Arabic Aspect-Based Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/computers13040098"
      },
      "year": 2024,
      "venue": "De Computis",
      "citations": 15,
      "notes": "Aspect-based sentiment analysis (ABSA) is a fine-grained type of sentiment analysis; it works on an aspect level.",
      "metrics": null
    },
    {
      "id": "sentiment-analysis-of-imbalanced-arabic-data-using-sampling",
      "name": "Sentiment analysis of imbalanced Arabic data using sampling techniques and classification algorithms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.11591/eei.v13i1.5886"
      },
      "year": 2024,
      "venue": "Bulletin of Electrical Engineering and Informatics",
      "citations": 15,
      "notes": "Sentiment analysis is a popular natural language processing task that recognizes the opinions or feelings of a piece of text.",
      "metrics": null
    },
    {
      "id": "transformer-models-in-education-summarizing-science-textbook",
      "name": "Transformer Models in Education: Summarizing Science Textbooks with AraBART, MT5, AraT5, and mBART",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.07692"
      },
      "year": 2024,
      "venue": "arXiv.org",
      "citations": 15,
      "notes": "Given this challenge, we have developed an advanced text summarization system targeting Arabic textbooks.",
      "metrics": null
    },
    {
      "id": "101-billion-arabic-words-dataset",
      "name": "101 Billion Arabic Words Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.01590"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In recent years, Large Language Models have revolutionized the field of natural language processing.",
      "metrics": null
    },
    {
      "id": "a-context-contrastive-inference-approach-to-partial-diacritization",
      "name": "A Context-Contrastive Inference Approach To Partial Diacritization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.8/"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "In this light, we introduce Context-Contrastive Partial Diacritization (‘CCPD‘)—a novel approach to ‘PD‘ which integrates seamlessly with existing Arabic.",
      "metrics": null
    },
    {
      "id": "a-new-benchmark-for-evaluating-automatic-speech-recognition-in-the-ara",
      "name": "A New Benchmark for Evaluating Automatic Speech Recognition in the Arabic Call Domain",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.04280"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This work is an attempt to introduce a comprehensive benchmark for Arabic speech recognition.",
      "metrics": null
    },
    {
      "id": "a-survey-of-llms-for-arabic-language-and-its-dialects",
      "name": "A Survey of LLMs for Arabic Language and its Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "survey",
        "llm",
        "dialects"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.20238"
      },
      "notes": "Survey of LLMs for Arabic and its dialects.",
      "metrics": null
    },
    {
      "id": "alclam-arabic-dialectal-language-model",
      "name": "AlcLaM: Arabic Dialectal Language Model",
      "type": "paper",
      "country": "INTL",
      "org": "amurtadha",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.13097",
        "github": "https://github.com/amurtadha/Alclam",
        "hf": "https://huggingface.co/rahbi"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To tackle this, we construct an Arabic dialectal corpus comprising 3.4M sentences gathered from social media platforms.",
      "metrics": null
    },
    {
      "id": "allam-large-language-models-for-arabic-and-english",
      "name": "ALLaM: Large Language Models for Arabic and English",
      "type": "paper",
      "country": "SA",
      "org": "SDAIA & IBM",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "pretraining",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.15390"
      },
      "notes": "ALLaM: Arabic and English LLMs via second-language acquisition-style training.",
      "metrics": null
    },
    {
      "id": "arabiangpt-native-arabic-gpt-based-llm",
      "name": "ArabianGPT: Native Arabic GPT-based LLM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "pretraining",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.15313"
      },
      "notes": "Native Arabic GPT-style LLM family.",
      "metrics": null
    },
    {
      "id": "arabic-dataset-for-llm-safeguard-evaluation",
      "name": "Arabic Dataset for LLM Safeguard Evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.17040"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In particular, we present an Arab-region-specific safety evaluation dataset consisting of 5,799 questions, including direct attacks, indirect attacks.",
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-of-zero-resourced-languages-a-case-of-shehri",
      "name": "Arabic Speech Recognition of zero-resourced Languages: A case of Shehri (Jibbali) Language",
      "type": "paper",
      "country": "INTL",
      "org": "College of Computer and Information Scienc",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.10/"
      },
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "We collected a Shehri (Jibbali) speech corpus and utilized transfer learning by fine-tuning pre-trained ASR models on this dataset.",
      "metrics": null
    },
    {
      "id": "arabic-text-sentiment-analysis-reinforcing-human-performed-surveys-wit",
      "name": "Arabic Text Sentiment Analysis: Reinforcing Human-Performed Surveys with Wider Topic Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.01921"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Sentiment analysis (SA) has been, and is still, a thriving research area.",
      "metrics": null
    },
    {
      "id": "arabic-nougat-fine-tuning-vision-transformers-for-arabic-ocr-and-markd",
      "name": "Arabic-Nougat: Fine-Tuning Vision Transformers for Arabic OCR and Markdown Extraction",
      "type": "paper",
      "country": "INTL",
      "org": "mohamedalirashad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.17835",
        "github": "https://github.com/MohamedAliRashad/arabic-nougat"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present Arabic-Nougat, a suite of OCR models for converting Arabic book pages into structured Markdown text.",
      "metrics": null
    },
    {
      "id": "arabicnlu-2024-the-first-arabic-natural-language-understanding-shared",
      "name": "ArabicNLU 2024: The First Arabic Natural Language Understanding Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.20663"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper presents an overview of the Arabic Natural Language Understanding (ArabicNLU 2024) shared task, focusing on two subtasks.",
      "metrics": null
    },
    {
      "id": "arablegaleval-a-multitask-benchmark-for-assessing-arabic-legal-knowled",
      "name": "ArabLegalEval: A Multitask Benchmark for Assessing Arabic Legal Knowledge in Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "thiqah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2408.07983",
        "github": "https://github.com/Thiqah/ArabLegalEval"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To address this gap, we introduce ArabLegalEval, a multitask benchmark dataset for assessing the Arabic legal knowledge of LLMs.",
      "metrics": null
    },
    {
      "id": "arafinnlp-2024-the-first-arabic-financial-nlp-shared-task",
      "name": "AraFinNLP 2024: The First Arabic Financial NLP Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.09818"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "The expanding financial markets of the Arab world require sophisticated Arabic NLP tools.",
      "metrics": null
    },
    {
      "id": "araieval-shared-task-propagandistic-techniques-detection-in-unimodal-a",
      "name": "ArAIEval Shared Task: Propagandistic Techniques Detection in Unimodal and Multimodal Arabic Content",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.04247",
        "website": "https://araieval.gitlab.io"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present an overview of the second edition of the ArAIEval shared task, organized as part of the ArabicNLP 2024 conference co-located with ACL 2024.",
      "metrics": null
    },
    {
      "id": "arapoembert-a-pretrained-language-model-for-arabic-poetry-analysis",
      "name": "AraPoemBERT: A Pretrained Language Model for Arabic Poetry Analysis",
      "type": "paper",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "pretraining",
        "poetry"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.12392",
        "hf": "https://huggingface.co/faisalq"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In this paper, we introduce AraPoemBERT, an Arabic language model pretrained exclusively on Arabic poetry text.",
      "metrics": null
    },
    {
      "id": "arastem-a-native-arabic-multiple-choice-question-benchmark-for-evaluat",
      "name": "AraSTEM: A Native Arabic Multiple Choice Question Benchmark for Evaluating LLMs Knowledge In STEM Subjects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2501.00559"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To address this issue, we introduce AraSTEM, a new Arabic multiple-choice question dataset aimed at evaluating LLMs knowledge in STEM subjects.",
      "metrics": null
    },
    {
      "id": "aratar-a-corpus-to-support-the-fine-grained-detection-of-hate-speech-t",
      "name": "AraTar: A Corpus to Support the Fine-grained Detection of Hate Speech Targets in the Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "The University of Manchester",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.1/"
      },
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "To comprehensively address this problem, we have combined and re-annotated hate speech tweets from existing publicly available corpora, resulting.",
      "metrics": null
    },
    {
      "id": "areeg-chars-dataset-for-envisioned-speech-recognition-using-eeg-for-ar",
      "name": "ArEEG_Chars: Dataset for Envisioned Speech Recognition using EEG for Arabic Characters",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.15733"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce ArEEGChars, a novel EEG dataset for Arabic 31 characters collected from 30 participants.",
      "metrics": null
    },
    {
      "id": "areeg-words-dataset-for-envisioned-speech-recognition-using-eeg-for-ar",
      "name": "ArEEG_Words: Dataset for Envisioned Speech Recognition using EEG for Arabic Words",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.18888"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce in this paper ArEEGWords dataset, a novel EEG dataset recorded from 22 participants with mean age of 22 years.",
      "metrics": null
    },
    {
      "id": "arzen-llm-code-switched-egyptian-arabic-english-translation-and-speech",
      "name": "ArzEn-LLM: Code-Switched Egyptian Arabic-English Translation and Speech Recognition Using LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "ahmedheakl",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.18120",
        "github": "http://github.com/ahmedheakl/arazn-llm",
        "hf": "http://huggingface.co/collections/ahmedheakl/arazn-llm-662ceaf12777656607b9524e"
      },
      "dialects": [
        "egy"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Motivated by the widespread increase in the phenomenon of code-switching between Egyptian Arabic and English in recent times.",
      "metrics": null
    },
    {
      "id": "athar-a-high-quality-and-diverse-dataset-for-classical-arabic-to-engli",
      "name": "ATHAR: A High-Quality and Diverse Dataset for Classical Arabic to English Translation",
      "type": "paper",
      "country": "INTL",
      "org": "mohamed-khalil",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.19835",
        "hf": "https://huggingface.co/datasets/mohamed-khalil/ATHAR"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present the ATHAR dataset, which comprises 66,000 high-quality classical Arabic to English translation samples that cover a wide array of topics.",
      "metrics": {
        "downloads": 82,
        "likes": 13,
        "lastModified": "2024-08-04"
      }
    },
    {
      "id": "automated-essay-scoring-in-arabic-a-dataset-and-analysis-of-a-bert-bas",
      "name": "Automated essay scoring in Arabic: a dataset and analysis of a BERT-based system",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.11212"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Automated Essay Scoring (AES) holds significant promise in the field of education, helping educators to mark larger volumes of essays.",
      "metrics": null
    },
    {
      "id": "benchmarking-llama-3-on-arabic-language-generation-tasks",
      "name": "Benchmarking LLaMA-3 on Arabic Language Generation Tasks",
      "type": "paper",
      "country": "INTL",
      "org": "The University of British Col",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.24/"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We focus to bridge this gap by evaluating LLaMA-3-70B on a diverse set of Arabic natural language generation (NLG) benchmarks.",
      "metrics": null
    },
    {
      "id": "bimedix2-bio-medical-expert-lmm-for-diverse-medical-modalities",
      "name": "BiMediX2: Bio-Medical EXpert LMM for Diverse Medical Modalities",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "chat",
        "qa",
        "summarization",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.07769",
        "github": "https://github.com/mbzuai-oryx/BiMediX2"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce BiMediX2, a bilingual (Arabic-English) Bio-Medical EXpert Large Multimodal Model that supports text-based and image-based medical interactions.",
      "metrics": null
    },
    {
      "id": "bimedix-bilingual-medical-mixture-of-experts-llm",
      "name": "BiMediX: Bilingual Medical Mixture of Experts LLM",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "qa",
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.13253",
        "github": "https://github.com/mbzuai-oryx/BiMediX"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In this paper, we introduce BiMediX, the first bilingual medical mixture of experts LLM designed for seamless interaction in both English and Arabic.",
      "metrics": null
    },
    {
      "id": "cafe-a-novel-code-switching-dataset-for-algerian-dialect-french-and-en",
      "name": "CAFE A Novel Code switching Dataset for Algerian Dialect French and English",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "code-switching",
        "asr",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.13424"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "CAFE: first code-switching speech dataset mixing Algerian dialect, French and English, from spontaneous human-human conversations.",
      "metrics": null
    },
    {
      "id": "camel-bench-a-comprehensive-arabic-lmm-benchmark",
      "name": "CAMEL-Bench: A Comprehensive Arabic LMM Benchmark",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.18976"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Recent years have witnessed a significant interest in developing large multimodal models.",
      "metrics": null
    },
    {
      "id": "cameleval-advancing-culturally-aligned-arabic-language-models-and-benc",
      "name": "CamelEval: Advancing Culturally Aligned Arabic Language Models and Benchmarks",
      "type": "paper",
      "country": "INTL",
      "org": "elmrc",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2409.12623",
        "hf": "https://huggingface.co/elmrc"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper introduces Juhaina, a Arabic-English bilingual LLM specifically designed to align with the values and preferences of Arabic speakers.",
      "metrics": null
    },
    {
      "id": "catt-character-based-arabic-tashkeel-transformer",
      "name": "CATT: Character-based Arabic Tashkeel Transformer",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.03236",
        "github": "https://github.com/abjadai/catt"
      },
      "notes": "Character-based Transformer for Arabic diacritization (tashkeel).",
      "metrics": null
    },
    {
      "id": "cidar-culturally-relevant-instruction-dataset-for-arabic",
      "name": "CIDAR: Culturally Relevant Instruction Dataset For Arabic",
      "type": "paper",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.03177",
        "github": "https://github.com/ARBML/CIDAR"
      },
      "dialects": [
        "lev"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Instruction tuning has emerged as a prominent methodology for teaching Large Language Models (LLMs) to follow instructions.",
      "metrics": null
    },
    {
      "id": "cross-attention-fusion-of-visual-and-geometric-features-for-large-voca",
      "name": "Cross-Attention Fusion of Visual and Geometric Features for Large Vocabulary Arabic Lipreading",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.11520",
        "website": "https://crns-smartvision.github.io/lrwar"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To address this challenge, firstly, we propose a cross-attention fusion-based approach for large lexicon Arabic vocabulary to predict spoken words in videos.",
      "metrics": null
    },
    {
      "id": "darijabanking-a-new-resource-for-overcoming-language-barriers-in-banki",
      "name": "DarijaBanking: A New Resource for Overcoming Language Barriers in Banking Intent Detection for Moroccan Arabic Speakers",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.16482"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Navigating the complexities of language diversity is a central challenge in developing robust natural language processing systems.",
      "metrics": null
    },
    {
      "id": "dialectal-pretraining-for-arabic-asr-mbzuai",
      "name": "Dialectal pretraining for Arabic ASR (MBZUAI)",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.05872"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Study and models showing dialectal pretraining improves Arabic automatic speech recognition.",
      "metrics": null
    },
    {
      "id": "egybert-bert-pretrained-on-egyptian-dialect-corpora",
      "name": "EgyBERT: BERT Pretrained on Egyptian Dialect Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "pretraining",
        "encoder",
        "dialects"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2408.03524"
      },
      "notes": "BERT pretrained on Egyptian dialect corpora.",
      "metrics": null
    },
    {
      "id": "estimating-the-level-of-dialectness-predicts-interannotator-agreement",
      "name": "Estimating the Level of Dialectness Predicts Interannotator Agreement in Multi-dialect Arabic Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.11282"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "On annotating multi-dialect Arabic datasets, it is common to randomly assign the samples across a pool of native Arabic speakers.",
      "metrics": null
    },
    {
      "id": "event-arguments-extraction-corpus-and-modeling-using-bert-for-arabic",
      "name": "Event-Arguments Extraction Corpus and Modeling using BERT for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.21153",
        "website": "https://sina.birzeit.edu/wojood"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To fill this gap, we introduce the \\hadath corpus ($550$k tokens) as an extension of Wojood, enriched with event-argument annotations.",
      "metrics": null
    },
    {
      "id": "exploiting-dialect-identification-in-automatic-dialectal-text-normaliz",
      "name": "Exploiting Dialect Identification in Automatic Dialectal Text Normalization",
      "type": "paper",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.4/"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We benchmark newly developed pretrained sequence-to-sequence models on the task of CODAfication.",
      "metrics": null
    },
    {
      "id": "fineweb-edu-ar-machine-translated-corpus-to-support-arabic-small-langu",
      "name": "Fineweb-Edu-Ar: Machine-translated Corpus to Support Arabic Small Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.06402"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This is especially true for multilingual LLMs, where the scarcity of high-quality and readily available data online has led to a multitude.",
      "metrics": null
    },
    {
      "id": "from-nile-sands-to-digital-hands-machine-translation-of-coptic-texts",
      "name": "From Nile Sands to Digital Hands: Machine Translation of Coptic Texts",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.25/",
        "github": "https://github.com/UBC-NLP/copticmt"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We have also developed the first neural machine translation system between Coptic, English, and Arabic.",
      "metrics": null
    },
    {
      "id": "gazelle-an-instruction-dataset-for-arabic-writing-assistance",
      "name": "Gazelle: An Instruction Dataset for Arabic Writing Assistance",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.18163"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "To address these issues, we present Gazelle, a comprehensive dataset for Arabic writing assistance.",
      "metrics": null
    },
    {
      "id": "glare-google-apps-arabic-reviews-dataset",
      "name": "GLARE: Google Apps Arabic Reviews Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.15259"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper introduces GLARE an Arabic Apps Reviews dataset collected from Saudi Google PlayStore.",
      "metrics": null
    },
    {
      "id": "hate-speech-detection-in-arabic-corpus-design-and-evaluation",
      "name": "Hate speech detection in Arabic: corpus design and evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "Frontiers in AI 2024",
      "tasks": [
        "hate-speech",
        "dataset"
      ],
      "links": {
        "paper": "https://www.frontiersin.org/journals/artificial-intelligence/articles/10.3389/frai.2024.1345445/full"
      },
      "notes": "Arabic hate speech detection: corpus design and evaluation.",
      "metrics": null
    },
    {
      "id": "humvi-a-multilingual-dataset-for-detecting-violent-incidents-impacting",
      "name": "HumVI: A Multilingual Dataset for Detecting Violent Incidents Impacting Humanitarian Aid",
      "type": "paper",
      "country": "INTL",
      "org": "dataminr-ai",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.06370",
        "github": "https://github.com/dataminr-ai/humvi-dataset"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "HumVI: English, French and Arabic news articles labelled for violent incidents affecting humanitarian aid.",
      "metrics": null
    },
    {
      "id": "leveraging-corpus-metadata-to-detect-template-based-translation-an-exp",
      "name": "Leveraging Corpus Metadata to Detect Template-based Translation: An Exploratory Case Study of the Egyptian Arabic Wikipedia Edition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2404.00565"
      },
      "dialects": [
        "egy",
        "magh"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Wikipedia articles (content pages) are commonly used corpora in Natural Language Processing.",
      "metrics": null
    },
    {
      "id": "llamalens-specialized-multilingual-llm-for-analyzing-news-and-social-m",
      "name": "LlamaLens: Specialized Multilingual LLM for Analyzing News and Social Media Content",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "news",
        "social-media",
        "multilingual"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.15308",
        "hf": "https://huggingface.co/collections/QCRI/llamalens-672f7e0604a0498c6a2f0fe9"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "LlamaLens: specialised multilingual LLM for news and social media analysis, tested on 18 tasks and 52 Arabic, English and Hindi datasets.",
      "metrics": null
    },
    {
      "id": "llm-based-mt-data-creation-dialectal-to-msa-translation-shared-task",
      "name": "LLM-based MT Data Creation: Dialectal to MSA Translation Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "aiXplain Inc.",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.14/"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "This paper presents our approach to the Dialect to Modern Standard Arabic (MSA) Machine Translation shared task, conducted as part of the sixth Workshop.",
      "metrics": null
    },
    {
      "id": "mememind-at-araieval-shared-task-spotting-persuasive-spans-in-arabic-t",
      "name": "MemeMind at ArAIEval Shared Task: Spotting Persuasive Spans in Arabic Text with Persuasion Techniques Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2408.04540"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Using attention masks, we created uniform lengths for each span and assigned BIO tags to each token based on the provided labels.",
      "metrics": null
    },
    {
      "id": "mentalqa-an-annotated-arabic-corpus-for-questions-and-answers-of-menta",
      "name": "MentalQA: An Annotated Arabic Corpus for Questions and Answers of Mental Healthcare",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "sentiment",
        "qa",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.12619"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce MentalQA, a novel Arabic dataset featuring conversational-style question-and-answer (QA) interactions.",
      "metrics": null
    },
    {
      "id": "muharaf-manuscripts-of-handwritten-arabic-dataset-for-cursive-text-rec",
      "name": "Muharaf: Manuscripts of Handwritten Arabic Dataset for Cursive Text Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "asr",
        "ocr",
        "vision",
        "poetry"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.09630"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present the Manuscripts of Handwritten Arabic(Muharaf) dataset.",
      "metrics": null
    },
    {
      "id": "nadi-2024-fifth-nuanced-arabic-dialect-identification",
      "name": "NADI 2024: Fifth Nuanced Arabic Dialect Identification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "tasks": [
        "dialect-id",
        "shared-task"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.04910"
      },
      "notes": "Overview of the fifth Nuanced Arabic Dialect Identification shared task.",
      "metrics": null
    },
    {
      "id": "natural-language-processing-for-dialects-of-a-language-a-survey",
      "name": "Natural Language Processing for Dialects of a Language: A Survey",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "dialect-id",
        "sentiment",
        "summarization",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2401.05632"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Survey of NLP for dialects of a language, covering datasets and approaches for dialectal NLU and NLG, Arabic dialects included.",
      "metrics": null
    },
    {
      "id": "nullpointer-at-araieval-shared-task-arabic-propagandist-technique-dete",
      "name": "Nullpointer at ArAIEval Shared Task: Arabic Propagandist Technique Detection with Token-to-Word Mapping in Sequence Tagging",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.01360"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper investigates the optimization of propaganda technique detection in Arabic text, including tweets \\& news paragraphs, from ArAIEval shared task 1.",
      "metrics": null
    },
    {
      "id": "on-the-importance-of-data-scale-in-pretraining-arabic-language-models",
      "name": "On the importance of Data Scale in Pretraining Arabic Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "huawei-noah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2401.07760",
        "github": "https://github.com/huawei-noah/Pretrained-Language-Model/tree/master/JABER-PyTorch"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Pretraining monolingual language models have been proven to be vital for performance in Arabic Natural Language Processing (NLP) tasks.",
      "metrics": null
    },
    {
      "id": "optimized-quran-passage-retrieval-using-an-expanded-qa-dataset-and-fin",
      "name": "Optimized Quran Passage Retrieval Using an Expanded QA Dataset and Fine-Tuned Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "qa",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.11431"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Understanding the deep meanings of the Qur'an and bridging the language gap between modern standard Arabic and classical Arabic is essential.",
      "metrics": null
    },
    {
      "id": "osact6-dialect-to-msa-translation-shared-task-overview",
      "name": "OSACT6 Dialect to MSA Translation Shared Task Overview",
      "type": "paper",
      "country": "INTL",
      "org": "aiXplain Inc.",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.11/"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "This paper presents the Dialectal Arabic (DA) to Modern Standard Arabic (MSA) Machine Translation (MT) shared task in the sixth Workshop on Open-Source.",
      "metrics": null
    },
    {
      "id": "palo-a-polyglot-large-multimodal-model-for-5b-people",
      "name": "PALO: A Polyglot Large Multimodal Model for 5B People",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "translation",
        "instruction-tuning",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2402.14818",
        "github": "https://github.com/mbzuai-oryx/PALO"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In pursuit of more inclusive Vision-Language Models (VLMs), this study introduces a Large Multilingual Multimodal Model called PALO.",
      "metrics": null
    },
    {
      "id": "peacock-a-family-of-arabic-multimodal-large-language-models-and-benchm",
      "name": "Peacock: A Family of Arabic Multimodal Large Language Models and Benchmarks",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.01031",
        "github": "https://github.com/UBC-NLP/peacock"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Multimodal large language models (MLLMs) have proven effective in a wide range of tasks requiring complex reasoning and linguistic comprehension.",
      "metrics": null
    },
    {
      "id": "performance-analysis-of-speech-encoders-for-low-resource-slu-and-asr-i",
      "name": "Performance Analysis of Speech Encoders for Low-Resource SLU and ASR in Tunisian Dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "pretraining",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.04533"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Speech encoders pretrained through self-supervised learning (SSL) have demonstrated remarkable performance in various downstream tasks.",
      "metrics": null
    },
    {
      "id": "propaganda-to-hate-a-multimodal-analysis-of-arabic-memes-with-multi-ag",
      "name": "Propaganda to Hate: A Multimodal Analysis of Arabic Memes with Multi-Agent LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "firojalam",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2409.07246",
        "github": "https://github.com/firojalam/propaganda-and-hateful-memes"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In the past decade, social media platforms have been used for information dissemination and consumption.",
      "metrics": null
    },
    {
      "id": "qabas-an-open-source-arabic-lexicographic-database",
      "name": "Qabas: An Open-Source Arabic Lexicographic Database",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.06598",
        "website": "https://sina.birzeit.edu/qabas"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present Qabas, a novel open-source Arabic lexicon designed for NLP applications.",
      "metrics": null
    },
    {
      "id": "quranic-audio-dataset-crowdsourced-and-labeled-recitation-from-non-ara",
      "name": "Quranic Audio Dataset: Crowdsourced and Labeled Recitation from Non-Arabic Speakers",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.02675"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper addresses the challenge of learning to recite the Quran for non-Arabic speakers.",
      "metrics": null
    },
    {
      "id": "receiptsense-beyond-traditional-ocr-a-dataset-for-receipt-understandin",
      "name": "ReceiptSense: Beyond Traditional OCR -- A Dataset for Receipt Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "update-for-integrated-business-ai",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "qa",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.04493",
        "github": "https://github.com/Update-For-Integrated-Business-AI/CORU"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce \\dataset, a comprehensive dataset designed for Arabic-English receipt understanding comprising 20,000 annotated receipts.",
      "metrics": null
    },
    {
      "id": "resource-aware-arabic-llm-creation-model-adaptation-integration-and-mu",
      "name": "Resource-Aware Arabic LLM Creation: Model Adaptation, Integration, and Multi-Domain Testing",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "dialect-id",
        "qa",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.17548"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This paper presents a novel approach to fine-tuning the Qwen2-1.5B model for Arabic language processing using Quantized Low-Rank Adaptation.",
      "metrics": null
    },
    {
      "id": "sambalingo-teaching-llms-new-languages",
      "name": "SambaLingo: Teaching LLMs New Languages",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "pretraining",
        "multilingual",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2404.05829"
      },
      "notes": "Recipe for adapting LLMs to new languages including Arabic.",
      "metrics": null
    },
    {
      "id": "saudibert-bert-pretrained-on-saudi-dialect-corpora",
      "name": "SaudiBERT: BERT Pretrained on Saudi Dialect Corpora",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "pretraining",
        "encoder",
        "dialects"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.06239"
      },
      "notes": "BERT pretrained on Saudi dialect corpora.",
      "metrics": null
    },
    {
      "id": "sina-at-fignews-2024-multilingual-datasets-annotated-with-bias-and-pro",
      "name": "Sina at FigNews 2024: Multilingual Datasets Annotated with Bias and Propaganda",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.09327",
        "website": "https://sina.birzeit.edu/fada"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "The proliferation of bias and propaganda on social media is an increasingly significant concern.",
      "metrics": null
    },
    {
      "id": "sinatools-open-source-toolkit-for-arabic-natural-language-processing",
      "name": "SinaTools: Open Source Toolkit for Arabic Natural Language Processing",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "ner",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.01523",
        "website": "https://sina.birzeit.edu/sinatools"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce SinaTools, an open-source Python package for Arabic natural language processing and understanding.",
      "metrics": null
    },
    {
      "id": "strategies-for-arabic-readability-modeling",
      "name": "Strategies for Arabic Readability Modeling",
      "type": "paper",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.5/"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We present a set of experimental results on Arabic readability assessment using a diverse range of approaches, from rule-based methods to Arabic pretrained.",
      "metrics": null
    },
    {
      "id": "swan-and-arabicmteb-dialect-aware-cross-lingual-language",
      "name": "Swan and ArabicMTEB: Dialect-Aware, Cross-Lingual Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2024,
      "venue": "arXiv 2024",
      "tasks": [
        "embedding",
        "benchmark"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.01192"
      },
      "notes": "Dialect-aware Swan embedding models and the ArabicMTEB benchmark.",
      "metrics": null
    },
    {
      "id": "synthetic-arabic-medical-dialogues-using-advanced-multi-agent-llm-tech",
      "name": "Synthetic Arabic Medical Dialogues Using Advanced Multi-Agent LLM Techniques",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "medical",
        "dialogue"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.2/"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We introduce a novel Multi-Agent LLM approach capable of generating synthetic Arabic medical dialogues from patient notes, regardless of the original language.",
      "metrics": null
    },
    {
      "id": "tafsirextractor-text-preprocessing-pipeline-preparing-classical-arabic",
      "name": "TafsirExtractor: Text Preprocessing Pipeline preparing Classical Arabic Literature for Machine Learning Applications",
      "type": "paper",
      "country": "INTL",
      "org": "Goethe University Frankfurt",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.8/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "We present a comprehensive tool of preprocessing Classical Arabic (CA) literature in the field of historical exegetical studies for machine learning (ML)",
      "metrics": null
    },
    {
      "id": "the-evolution-of-darija-open-dataset-introducing-version-2",
      "name": "The Evolution of Darija Open Dataset: Introducing Version 2",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2405.13016"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Darija Open Dataset (DODa) represents an open-source project aimed at enhancing Natural Language Processing capabilities for the Moroccan dialect, Darija.",
      "metrics": null
    },
    {
      "id": "the-fignews-shared-task-on-news-media-narratives",
      "name": "The FIGNEWS Shared Task on News Media Narratives",
      "type": "paper",
      "country": "QA",
      "org": "Northwestern University in Qatar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.56/"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We present an overview of the FIGNEWSshared task, organized as part of the Arabic-NLP 2024 conference co-located with ACL2024.",
      "metrics": null
    },
    {
      "id": "the-qiyas-benchmark-measuring-chatgpt-mathematical-and-language-unders",
      "name": "The Qiyas Benchmark: Measuring ChatGPT Mathematical and Language Understanding in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "reasoning",
        "math",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.00146"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Qiyas benchmarks: Arabic mathematical reasoning and language understanding questions from Saudi's GAT exam, tested on ChatGPT.",
      "metrics": null
    },
    {
      "id": "the-samer-arabic-text-simplification-corpus",
      "name": "The SAMER Arabic Text Simplification Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2404.18615"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present the SAMER Corpus, the first manually annotated Arabic parallel corpus for text simplification targeting school-aged learners.",
      "metrics": null
    },
    {
      "id": "the-translation-of-circumlocution-in-arabic-short-stories-into-english",
      "name": "The Translation of Circumlocution in Arabic Short Stories into English",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.02887",
        "website": "https://ntu.edu.iq"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This study investigates the translation of circumlocution from Arabic to English in a corpus of short stories by renowned Arabic authors.",
      "metrics": null
    },
    {
      "id": "tibyan-corpus-balanced-and-comprehensive-error-coverage-corpus-using-c",
      "name": "Tibyan Corpus: Balanced and Comprehensive Error Coverage Corpus Using ChatGPT for Arabic Grammatical Error Correction",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.04588"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Natural language processing (NLP) utilizes text data augmentation to overcome sample size constraints.",
      "metrics": null
    },
    {
      "id": "to-distill-or-not-to-distill-on-the-robustness-of-robust-knowledge-dis",
      "name": "To Distill or Not to Distill? On the Robustness of Robust Knowledge Distillation",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.04512"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Arabic is known to present unique challenges for Automatic Speech Recognition (ASR).",
      "metrics": null
    },
    {
      "id": "towards-global-ai-inclusivity-a-large-scale-multilingual-terminology-d",
      "name": "Towards Global AI Inclusivity: A Large-Scale Multilingual Terminology Dataset (GIST)",
      "type": "paper",
      "country": "INTL",
      "org": "jerry999",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2412.18367",
        "hf": "https://huggingface.co/datasets/Jerry999/multilingual-terminology"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce GIST, a large-scale multilingual AI terminology dataset containing 5K terms extracted from top AI conference papers spanning 2000 to 2023.",
      "metrics": {
        "downloads": 72,
        "likes": 1,
        "lastModified": "2025-05-31"
      }
    },
    {
      "id": "towards-zero-shot-text-to-speech-for-arabic-dialects",
      "name": "Towards Zero-Shot Text-To-Speech for Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "dialect-id",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.16751",
        "website": "https://docs.coqui.ai/en/latest/models/xtts.html"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "Zero-shot multi-speaker text-to-speech (ZS-TTS) systems have advanced for English, however, it still lags behind due to insufficient resources.",
      "metrics": null
    },
    {
      "id": "visper-multilingual-audio-visual-speech-recognition",
      "name": "ViSpeR: Multilingual Audio-Visual Speech Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "yasserdahouml",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2406.00038",
        "github": "https://github.com/YasserdahouML/visper"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "This work presents an extensive and detailed study on Audio-Visual Speech Recognition (AVSR) for five widely spoken languages.",
      "metrics": null
    },
    {
      "id": "wojoodner-2024-the-second-arabic-named-entity-recognition-shared-task",
      "name": "WojoodNER 2024: The Second Arabic Named Entity Recognition Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.09936"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present WojoodNER-2024, the second Arabic Named Entity Recognition (NER) Shared Task.",
      "metrics": null
    },
    {
      "id": "xmodel-1-5-an-1b-scale-multilingual-llm",
      "name": "Xmodel-1.5: An 1B-scale Multilingual LLM",
      "type": "paper",
      "country": "INTL",
      "org": "xiaoduoailab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.10083",
        "github": "https://github.com/XiaoduoAILab/XmodelLM-1.5"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We introduce Xmodel-1.5, a 1-billion-parameter multilingual large language model pretrained on 2 trillion tokens.",
      "metrics": null
    },
    {
      "id": "xtrust-on-the-multilingual-trustworthiness-of-large-language-models",
      "name": "XTRUST: On the Multilingual Trustworthiness of Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "lluckyyh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2409.15762",
        "github": "https://github.com/LluckyYH/XTRUST"
      },
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "In response to the growing global deployment of LLMs, we introduce XTRUST, the first comprehensive multilingual trustworthiness benchmark.",
      "metrics": null
    },
    {
      "id": "zaebuc-spoken-a-multilingual-multidialectal-arabic-english-speech-corp",
      "name": "ZAEBUC-Spoken: A Multilingual Multidialectal Arabic-English Speech Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "asr",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2403.18182"
      },
      "dialects": [
        "egy",
        "gulf",
        "msa",
        "mixed"
      ],
      "year": 2024,
      "venue": "arXiv 2024",
      "notes": "We present ZAEBUC-Spoken, a multilingual multidialectal Arabic-English speech corpus.",
      "metrics": null
    },
    {
      "id": "gptaraeval-a-comprehensive-evaluation-of-chatgpt-on-arabic-n",
      "name": "GPTAraEval: A Comprehensive Evaluation of ChatGPT on Arabic NLP",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.14976"
      },
      "year": 2023,
      "venue": "Conference on Empirical Methods in Natural Language Processing",
      "citations": 108,
      "notes": "ChatGPT's emergence heralds a transformative phase in NLP, particularly demonstrated through its excellent performance on many English benchmarks.",
      "metrics": null
    },
    {
      "id": "isolated-arabic-sign-language-recognition-using-a-transforme",
      "name": "Isolated Arabic Sign Language Recognition Using a Transformer-based Model and Landmark Keypoints",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1145/3584984"
      },
      "year": 2023,
      "venue": "ACM Trans. Asian Low Resour. Lang. Inf. Process.",
      "citations": 95,
      "notes": "This article presents a framework for isolated Arabic sign language recognition using hand and face keypoints.",
      "metrics": null
    },
    {
      "id": "a-survey-of-ocr-in-arabic-language-applications-techniques-a",
      "name": "A Survey of OCR in Arabic Language: Applications, Techniques, and Challenges",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "survey",
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app13074584"
      },
      "year": 2023,
      "venue": "Applied Sciences",
      "citations": 77,
      "notes": "Optical character recognition (OCR) is the process of extracting handwritten or printed text from a scanned or printed image and converting it to a",
      "metrics": null
    },
    {
      "id": "arabic-abstractive-text-summarization-using-rnn-based-and-tr",
      "name": "Arabic abstractive text summarization using RNN-based and transformer-based architectures",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.ipm.2022.103227"
      },
      "year": 2023,
      "venue": "Information Processing & Management",
      "citations": 75,
      "metrics": null
    },
    {
      "id": "arabic-tweets-based-sentiment-analysis-to-investigate-the-im",
      "name": "Arabic Tweets-Based Sentiment Analysis to Investigate the Impact of COVID-19 in KSA: A Deep Learning Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/bdcc7010016"
      },
      "year": 2023,
      "venue": "Big Data and Cognitive Computing",
      "citations": 70,
      "notes": "The World Health Organization (WHO) declared the outbreak of Coronavirus disease 2019 (COVID-19) a pandemic on 11 March 2020.",
      "metrics": null
    },
    {
      "id": "prediction-of-the-customers-interests-using-sentiment-analys",
      "name": "Prediction of the customers' interests using sentiment analysis in e-commerce data for comparison of Arabic, English, and Turkish languages",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2023.02.017"
      },
      "year": 2023,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 67,
      "notes": "In the business world, large companies that can achieve continuity in innovation gain a significant competitive advantage.",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis-of-youtube-comments-nlp-based-mach",
      "name": "Arabic Sentiment Analysis of YouTube Comments: NLP-Based Machine Learning Approaches for Content Evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/bdcc7030127"
      },
      "year": 2023,
      "venue": "Big Data and Cognitive Computing",
      "citations": 61,
      "notes": "YouTube is a popular video-sharing platform that offers a diverse range of content.",
      "metrics": null
    },
    {
      "id": "comparative-performance-of-ensemble-machine-learning-for-ara",
      "name": "Comparative performance of ensemble machine learning for Arabic cyberbullying and offensive language detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10579-023-09683-y"
      },
      "year": 2023,
      "venue": "Language Resources and Evaluation",
      "citations": 53,
      "notes": "Since cyberbullying impacts both individual victims and entire society, research on abusive language and its detection has attracted attention in recent years.",
      "metrics": null
    },
    {
      "id": "tarjamat-evaluation-of-bard-and-chatgpt-on-machine-translati",
      "name": "TARJAMAT: Evaluation of Bard and ChatGPT on Machine Translation of Ten Arabic Varieties",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2308.03051"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 52,
      "notes": "Considering this constraint, we present a thorough assessment of Bard and ChatGPT",
      "metrics": null
    },
    {
      "id": "beyond-english-evaluating-llms-for-arabic-grammatical-error",
      "name": "Beyond English: Evaluating LLMs for Arabic Grammatical Error Correction",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2312.08400"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 35,
      "notes": "Large language models (LLMs) finetuned to follow human instruction have recently exhibited significant capabilities in various English NLP tasks.",
      "metrics": null
    },
    {
      "id": "qur-an-qa-2023-shared-task-overview-of-passage-retrieval-and",
      "name": "Qur’an QA 2023 Shared Task: Overview of Passage Retrieval and Reading Comprehension Tasks over the Holy Qur’an",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "quran",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.76/"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 31,
      "notes": "Motivated by the need for intelligent question answering (QA) systems on the Holy Qur’an and the success of the first Qur’an Question Answering shared task",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification-under-scrutiny-limitations-of",
      "name": "Arabic Dialect Identification under Scrutiny: Limitations of Single-label Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.13661"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 28,
      "notes": "Automatic Arabic Dialect Identification (ADI) of text has gained great popularity since it was introduced in the early 2010s.",
      "metrics": null
    },
    {
      "id": "a-parameter-efficient-learning-approach-to-arabic-dialect-id",
      "name": "A Parameter-Efficient Learning Approach to Arabic Dialect Identification with Pre-Trained General-Purpose Speech Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.11244"
      },
      "year": 2023,
      "venue": "Interspeech",
      "citations": 26,
      "notes": "In this work, we explore Parameter-Efficient-Learning (PEL) techniques to repurpose a General-Purpose-Speech (GSM) model for Arabic dialect identification",
      "metrics": null
    },
    {
      "id": "artst-arabic-text-and-speech-transformer",
      "name": "ArTST: Arabic Text and Speech Transformer",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.16621"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 26,
      "notes": "We present ArTST, a pre-trained Arabic text and speech transformer for supporting open-source speech technologies for the Arabic language.",
      "metrics": null
    },
    {
      "id": "evaluating-chatgpt-and-bard-ai-on-arabic-sentiment-analysis",
      "name": "Evaluating ChatGPT and Bard AI on Arabic Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.27/"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 24,
      "notes": "Large Language Models (LLMs) such as ChatGPT and Bard AI have gained much attention due to their outstanding performance on a range of NLP tasks.",
      "metrics": null
    },
    {
      "id": "octopus-a-multitask-model-and-toolkit-for-arabic-natural-lan",
      "name": "Octopus: A Multitask Model and Toolkit for Arabic Natural Language Generation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.16127"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 24,
      "notes": "Understanding Arabic text and generating human-like responses is a challenging task.",
      "metrics": null
    },
    {
      "id": "on-the-robustness-of-arabic-speech-dialect-identification",
      "name": "On the Robustness of Arabic Speech Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.03789"
      },
      "year": 2023,
      "venue": "Interspeech",
      "citations": 18,
      "notes": "As these pipelines require application of ADI tools to potentially out-of-domain data, we aim to investigate how vulnerable the tools",
      "metrics": null
    },
    {
      "id": "camelparser2-0-a-state-of-the-art-dependency-parser-for-arab",
      "name": "CamelParser2.0: A State-of-the-Art Dependency Parser for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.15/"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 17,
      "notes": "We present CamelParser2.0, an open-source Python-based Arabic dependency parser targeting two popular Arabic dependency formalisms, the Columbia Arabic Treebank",
      "metrics": null
    },
    {
      "id": "analyzing-multilingual-competency-of-llms-in-multi-turn-inst",
      "name": "Analyzing Multilingual Competency of LLMs in Multi-Turn Instruction Following: A Case Study of Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.14819"
      },
      "year": 2023,
      "venue": "ARABICNLP",
      "citations": 15,
      "notes": "While significant progress has been made in benchmarking Large Language Models (LLMs) across various tasks, there is a lack of comprehensive evaluation of their",
      "metrics": null
    },
    {
      "id": "a-benchmark-and-scoring-algorithm-for-enriching-arabic-synonyms",
      "name": "A Benchmark and Scoring Algorithm for Enriching Arabic Synonyms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2302.02232",
        "website": "https://portal.sina.birzeit.edu/synonyms"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present twofold contributions: an algorithm and a benchmark dataset.",
      "metrics": null
    },
    {
      "id": "afrisenti-a-twitter-sentiment-analysis-benchmark-for-african-languages",
      "name": "AfriSenti: A Twitter Sentiment Analysis Benchmark for African Languages",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2302.08956",
        "website": "https://afrisenti-semeval.github.io"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We introduce AfriSenti, a sentiment analysis benchmark that contains a total of >110,000 tweets in 14 African languages.",
      "metrics": null
    },
    {
      "id": "amurd-annotated-arabic-english-receipt-dataset-for-key-information-ext",
      "name": "AMuRD: Annotated Arabic-English Receipt Dataset for Key Information Extraction and Classification",
      "type": "paper",
      "country": "INTL",
      "org": "update-for-integrated-business-ai",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2309.09800",
        "github": "https://github.com/Update-For-Integrated-Business-AI/AMuRD"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "In this paper, we present AMuRD, a novel multilingual human-annotated dataset specifically designed for information extraction from receipts.",
      "metrics": null
    },
    {
      "id": "aner-arabic-and-arabizi-named-entity-recognition-using-transformer-bas",
      "name": "ANER: Arabic and Arabizi Named Entity Recognition using Transformer-Based Approach",
      "type": "paper",
      "country": "INTL",
      "org": "boda",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2308.14669",
        "hf": "https://huggingface.co/boda/ANER"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present ANER, a web-based named entity recognizer for the Arabic, and Arabizi languages.",
      "metrics": {
        "downloads": 43,
        "likes": 4,
        "lastModified": "2023-09-18"
      }
    },
    {
      "id": "arabic-fine-grained-entity-recognition",
      "name": "Arabic Fine-Grained Entity Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.17333",
        "website": "https://sina.birzeit.edu/wojood"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Traditional NER systems are typically trained to recognize coarse-grained entities.",
      "metrics": null
    },
    {
      "id": "arabic-handwritten-text-line-dataset",
      "name": "Arabic Handwritten Text Line Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2312.07573"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "In this paper, we present a new dataset specifically designed for historical Arabic script in which we annotate position in word level.",
      "metrics": null
    },
    {
      "id": "arabic-mini-climategpt-a-climate-change-and-sustainability-tailored-ar",
      "name": "Arabic Mini-ClimateGPT : A Climate Change and Sustainability Tailored Arabic LLM",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding",
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2312.09366",
        "github": "https://github.com/mbzuai-oryx/ClimateGPT"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We propose a light-weight Arabic Mini-ClimateGPT that is built on an open-source LLM and is specifically fine-tuned.",
      "metrics": null
    },
    {
      "id": "arabicros-ai-powered-arabic-crossword-puzzle-generation-for-educationa",
      "name": "ArabIcros: AI-Powered Arabic Crossword Puzzle Generation for Educational Applications",
      "type": "paper",
      "country": "INTL",
      "org": "University of Siena",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "crossword",
        "education"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.23/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "First Arabic crossword puzzle generator driven by LLMs including GPT-4, GPT-3 variants and BERT, aimed at education.",
      "metrics": null
    },
    {
      "id": "araieval-shared-task-persuasion-techniques-and-disinformation-detectio",
      "name": "ArAIEval Shared Task: Persuasion Techniques and Disinformation Detection in Arabic Text",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2311.03179",
        "website": "https://araieval.gitlab.io"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present an overview of the ArAIEval shared task, organized as part of the first ArabicNLP 2023 conference co-located with EMNLP 2023.",
      "metrics": null
    },
    {
      "id": "arbanking77-intent-detection-neural-model-and-a-new-dataset-in-modern",
      "name": "ArBanking77: Intent Detection Neural Model and a New Dataset in Modern and Dialectical Arabic",
      "type": "paper",
      "country": "PS",
      "org": "SinaLab, Birzeit University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "intent-detection",
        "banking"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.19034",
        "website": "https://sina.birzeit.edu/arbanking77"
      },
      "dialects": [
        "msa",
        "lev"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "ArBanking77: 31,404 banking intent queries in MSA and Palestinian dialect, arabized from Banking77, plus a neural intent-detection model.",
      "metrics": null
    },
    {
      "id": "arcoq-arabic-closest-opposite-questions-dataset",
      "name": "ARCOQ: Arabic Closest Opposite Questions Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.14384"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "This paper presents a dataset for closest opposite questions in Arabic language.",
      "metrics": null
    },
    {
      "id": "arpanemo-an-open-source-dataset-for-fine-grained-emotion-recognition-i",
      "name": "ArPanEmo: An Open-Source Dataset for Fine-Grained Emotion Recognition in Arabic Online Content during COVID-19 Pandemic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.17580"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "This paper presents the ArPanEmo dataset, a novel dataset for fine-grained emotion recognition of online posts in Arabic.",
      "metrics": null
    },
    {
      "id": "artrivia-harvesting-arabic-wikipedia-to-build-a-new-arabic-question-an",
      "name": "ArTrivia: Harvesting Arabic Wikipedia to Build A New Arabic Question Answering Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "University of Delaware",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.17/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "We present ArTrivia, a new Arabic question-answering dataset consisting of more than 10,000 question-answer pairs along with relevant passages, covering.",
      "metrics": null
    },
    {
      "id": "clartts-an-open-source-classical-arabic-text-to-speech-corpus",
      "name": "ClArTTS: An Open-Source Classical Arabic Text-to-Speech Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2303.00069"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "In a move towards filling this gap in resources, we present a speech corpus for Classical Arabic Text-to-Speech.",
      "metrics": null
    },
    {
      "id": "cultural-alignment-in-large-language-models-an-explanatory-analysis-ba",
      "name": "Cultural Alignment in Large Language Models: An Explanatory Analysis Based on Hofstede's Cultural Dimensions",
      "type": "paper",
      "country": "INTL",
      "org": "reemim",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2309.12342",
        "github": "https://github.com/reemim/Hofstedes_CAT"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "The deployment of large language models (LLMs) raises concerns regarding their cultural misalignment and potential ramifications on individuals.",
      "metrics": null
    },
    {
      "id": "dolphin-a-challenging-and-diverse-benchmark-for-arabic-nlg",
      "name": "Dolphin: A Challenging and Diverse Benchmark for Arabic NLG",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.14989"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present Dolphin, a novel benchmark that addresses the need for a natural language generation.",
      "metrics": null
    },
    {
      "id": "domain-adaptation-for-arabic-machine-translation-the-case-of-financial",
      "name": "Domain Adaptation for Arabic Machine Translation: The Case of Financial Texts",
      "type": "paper",
      "country": "INTL",
      "org": "Asas AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2309.12863",
        "hf": "https://huggingface.co/asas-ai"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Neural machine translation (NMT) has shown impressive performance when trained on large-scale corpora.",
      "metrics": null
    },
    {
      "id": "enriching-the-narabizi-treebank-a-multifaceted-approach-to-supporting",
      "name": "Enriching the NArabizi Treebank: A Multifaceted Approach to Supporting an Under-Resourced Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.14866"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We introduce an enriched version of NArabizi Treebank (Seddah et al., 2020) with three main contributions: the addition of two novel annotation layers.",
      "metrics": null
    },
    {
      "id": "evaluating-emotion-arcs-across-languages-bridging-the-global-divide-in",
      "name": "Evaluating Emotion Arcs Across Languages: Bridging the Global Divide in Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "dteodore",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.02213",
        "github": "https://github.com/dteodore/EmotionArcs"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Emotion arcs capture how an individual (or a population) feels over time.",
      "metrics": null
    },
    {
      "id": "gari-graph-attention-for-relative-isomorphism-of-arabic-word-embedding",
      "name": "GARI: Graph Attention for Relative Isomorphism of Arabic Word Embeddings",
      "type": "paper",
      "country": "INTL",
      "org": "asif6827",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.13068",
        "github": "https://github.com/asif6827/GARI"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "To address this, we propose GARI that combines the distributional training objectives with multiple isomorphism losses guided by the graph attention network.",
      "metrics": null
    },
    {
      "id": "having-beer-after-prayer-measuring-cultural-bias-in-large-language-mod",
      "name": "Having Beer after Prayer? Measuring Cultural Bias in Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "tareknaous",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "sentiment",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.14456",
        "github": "https://github.com/tareknaous/camel"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We introduce CAMeL, a novel resource of 628 naturally-occurring prompts and 20,368 entities spanning eight types that contrast Arab and Western cultures.",
      "metrics": null
    },
    {
      "id": "hicma-the-handwriting-identification-for-calligraphy-and-manuscripts-i",
      "name": "HICMA: The Handwriting Identification for Calligraphy and Manuscripts in Arabic Dataset",
      "type": "paper",
      "country": "LB",
      "org": "Lebanese American University",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.3/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "We present the Handwriting Identification of Manuscripts and Calligraphy in Arabic (HICMA) dataset as the first publicly available dataset with real-world.",
      "metrics": null
    },
    {
      "id": "idrisi-d-arabic-and-english-datasets-and-benchmarks-for-location-menti",
      "name": "IDRISI-D: Arabic and English Datasets and Benchmarks for Location Mention Disambiguation over Disaster Microblogs",
      "type": "paper",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.14/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "We introduce IDRISI-D, the largest to date English and the first Arabic public LMD datasets.",
      "metrics": null
    },
    {
      "id": "impact-of-subword-pooling-strategy-on-cross-lingual-event-detection",
      "name": "Impact of Subword Pooling Strategy on Cross-lingual Event Detection",
      "type": "paper",
      "country": "INTL",
      "org": "isi-boston",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2302.11365",
        "github": "https://github.com/isi-boston/ed-pooling"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Pre-trained multilingual language models (e.g., mBERT, XLM-RoBERTa) have significantly advanced the state-of-the-art.",
      "metrics": null
    },
    {
      "id": "jais-and-jais-chat-arabic-centric-foundation-and",
      "name": "Jais and Jais-chat: Arabic-Centric Foundation and Instruction-Tuned LLMs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "apache-2.0",
      "modality": "text",
      "year": 2023,
      "venue": "arXiv 2023",
      "tasks": [
        "pretraining",
        "instruction-tuning",
        "llm"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2308.16149",
        "hf": "https://huggingface.co/inceptionai/jais-13b-chat"
      },
      "notes": "Arabic-centric bilingual LLMs and chat models trained from scratch.",
      "metrics": {
        "downloads": 373,
        "likes": 168,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "larabench-benchmarking-arabic-ai-with-large-language-models",
      "name": "LAraBench: Benchmarking Arabic AI with Large Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "asr",
        "tts",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.14982"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Recent advancements in Large Language Models (LLMs) have significantly influenced the landscape of language and speech research.",
      "metrics": null
    },
    {
      "id": "mavericks-at-nadi-2023-shared-task-unravelling-regional-nuances-throug",
      "name": "Mavericks at NADI 2023 Shared Task: Unravelling Regional Nuances through Dialect Identification using Transformer-based Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "dialect-id",
        "pretraining",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2311.18739"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "In this paper, we present our approach for the \"Nuanced Arabic Dialect Identification (NADI) Shared Task 2023\".",
      "metrics": null
    },
    {
      "id": "myvoice-arabic-speech-resource-collaboration-platform",
      "name": "MyVoice: Arabic Speech Resource Collaboration Platform",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2308.02503"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We introduce MyVoice, a crowdsourcing platform designed to collect Arabic speech to enhance dialectal speech technologies.",
      "metrics": null
    },
    {
      "id": "n-shot-benchmarking-of-whisper-on-diverse-arabic-speech-recognition",
      "name": "N-Shot Benchmarking of Whisper on Diverse Arabic Speech Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "whisper",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.02902"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Evaluates Whisper on diverse Arabic speech recognition benchmarks across dialects and conditions, with n-shot prompting experiments.",
      "metrics": null
    },
    {
      "id": "nadi-2023-the-fourth-nuanced-arabic-dialect-identification-shared-task",
      "name": "NADI 2023: The Fourth Nuanced Arabic Dialect Identification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.16117"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We describe the findings of the fourth Nuanced Arabic Dialect Identification Shared Task (NADI 2023).",
      "metrics": null
    },
    {
      "id": "noor-ghateh-a-benchmark-dataset-for-evaluating-arabic-word-segmenters",
      "name": "Noor-Ghateh: A Benchmark Dataset for Evaluating Arabic Word Segmenters in Hadith Domain",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "hadith"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2307.09630"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present a standard dataset for analyzing the Arabic segmentation tools.",
      "metrics": null
    },
    {
      "id": "osn-mdad-machine-translation-dataset-for-arabic-multi-dialectal-conver",
      "name": "OSN-MDAD: Machine Translation Dataset for Arabic Multi-Dialectal Conversations on Online Social Media",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2309.12137"
      },
      "dialects": [
        "gulf",
        "lev",
        "iraqi",
        "yemeni",
        "msa",
        "mixed"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "While resources for English language are fairly sufficient to understand content on social media, similar resources in Arabic are still immature.",
      "metrics": null
    },
    {
      "id": "pali-a-language-identification-benchmark-for-perso-arabic-scripts",
      "name": "PALI: A Language Identification Benchmark for Perso-Arabic Scripts",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2304.01322"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "The Perso-Arabic scripts are a family of scripts that are widely adopted and used by various linguistic communities around the globe.",
      "metrics": null
    },
    {
      "id": "readme-benchmarking-multilingual-language-models-for-multi-domain-read",
      "name": "ReadMe++: Benchmarking Multilingual Language Models for Multi-Domain Readability Assessment",
      "type": "paper",
      "country": "INTL",
      "org": "tareknaous",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2305.14463",
        "github": "https://github.com/tareknaous/readme"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present a comprehensive evaluation of large language models for multilingual readability assessment.",
      "metrics": null
    },
    {
      "id": "rgb-arabic-alphabets-sign-language-dataset",
      "name": "RGB Arabic Alphabets Sign Language Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2301.11932"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "This paper introduces the RGB Arabic Alphabet Sign Language (AASL) dataset.",
      "metrics": null
    },
    {
      "id": "salma-arabic-sense-annotated-corpus-and-wsd-benchmarks",
      "name": "SALMA: Arabic Sense-Annotated Corpus and WSD Benchmarks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.19029",
        "website": "https://sina.birzeit.edu/salma"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "SALMA, the first Arabic sense-annotated corpus, consists of 34K tokens, which are all sense-annotated.",
      "metrics": null
    },
    {
      "id": "semeval-2023-task-12-sentiment-analysis-for-african-languages-afrisent",
      "name": "SemEval-2023 Task 12: Sentiment Analysis for African Languages (AfriSenti-SemEval)",
      "type": "paper",
      "country": "INTL",
      "org": "afrisenti-semeval",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2304.06845",
        "github": "https://github.com/afrisenti-semeval/afrisent-semeval-2023"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "AfriSenti-SemEval: tweet sentiment shared task in 14 African languages, including Algerian and Moroccan Arabic.",
      "metrics": null
    },
    {
      "id": "sentiment-analysis-dataset-in-moroccan-dialect-bridging-the-gap-betwee",
      "name": "Sentiment Analysis Dataset in Moroccan Dialect: Bridging the Gap Between Arabic and Latin Scripted dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2303.15987"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Sentiment analysis, the automated process of determining emotions or opinions expressed in text.",
      "metrics": null
    },
    {
      "id": "taqyim-evaluating-arabic-nlp-tasks-using-chatgpt-models",
      "name": "Taqyim: Evaluating Arabic NLP Tasks Using ChatGPT Models",
      "type": "paper",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "diacritization",
        "sentiment",
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.16322",
        "github": "https://github.com/ARBML/Taqyim"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Additionally, we introduce a new Python interface https://github.com/ARBML/Taqyim that facilitates the evaluation of these tasks effortlessly.",
      "metrics": null
    },
    {
      "id": "the-saudi-privacy-policy-dataset",
      "name": "The Saudi Privacy Policy Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "privacy",
        "legal",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2304.02757"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Saudi Privacy Policy Dataset: 1,000 Saudi websites' Arabic privacy policies annotated against the ten PDPL principles.",
      "metrics": null
    },
    {
      "id": "towards-arabic-multimodal-dataset-for-sentiment-analysis",
      "name": "Towards Arabic Multimodal Dataset for Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2306.06322"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "Multimodal Sentiment Analysis (MSA) has recently become a centric research direction for many real-world applications.",
      "metrics": null
    },
    {
      "id": "toxic-language-detection-a-systematic-review-of-arabic-datasets",
      "name": "Toxic language detection: a systematic review of Arabic datasets",
      "type": "paper",
      "country": "INTL",
      "org": "imene1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2312.07228",
        "github": "https://github.com/Imene1/Arabic-toxic-language"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "The detection of toxic language in the Arabic language has emerged as an active area of research in recent years.",
      "metrics": null
    },
    {
      "id": "using-lstm-and-gru-with-a-new-dataset-for-named-entity-recognition-in",
      "name": "Using LSTM and GRU With a New Dataset for Named Entity Recognition in the Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2304.03399"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "In addition, this work proposes long short term memory (LSTM) units and Gated Recurrent Units.",
      "metrics": null
    },
    {
      "id": "violet-a-vision-language-model-for-arabic-image-captioning-with-gemini",
      "name": "Violet: A Vision-Language Model for Arabic Image Captioning with Gemini Decoder",
      "type": "paper",
      "country": "INTL",
      "org": "The University of British C",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "image-captioning"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.1/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "Vision-language model for Arabic image captioning that uses a Gemini decoder.",
      "metrics": null
    },
    {
      "id": "wojoodner-2023-the-first-arabic-named-entity-recognition-shared-task",
      "name": "WojoodNER 2023: The First Arabic Named Entity Recognition Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.16153"
      },
      "year": 2023,
      "venue": "arXiv 2023",
      "notes": "We present WojoodNER-2023, the first Arabic Named Entity Recognition (NER) Shared Task.",
      "metrics": null
    },
    {
      "id": "yet-another-model-for-arabic-dialect-identification",
      "name": "Yet Another Model for Arabic Dialect Identification",
      "type": "paper",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.37/"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "We describe a spoken Arabic dialect identification (ADI) model for Arabic that consistently outperforms previously published results on two benchmark.",
      "metrics": null
    },
    {
      "id": "bert-models-for-arabic-text-classification-a-systematic-revi",
      "name": "BERT Models for Arabic Text Classification: A Systematic Review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app12115720"
      },
      "year": 2022,
      "venue": "Applied Sciences",
      "citations": 143,
      "notes": "Bidirectional Encoder Representations from Transformers (BERT) has gained increasing attention from researchers and practitioners as it has proven to be an",
      "metrics": null
    },
    {
      "id": "semeval-2022-task-6-isarcasmeval-intended-sarcasm-detection",
      "name": "SemEval-2022 Task 6: iSarcasmEval, Intended Sarcasm Detection in English and Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.semeval-1.111/"
      },
      "year": 2022,
      "venue": "International Workshop on Semantic Evaluation",
      "citations": 111,
      "notes": "iSarcasmEval is the first shared task to target intended sarcasm detection: the data for this task was provided and labelled by the authors of the texts",
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-based-on-deep-contextualized-embe",
      "name": "Arabic fake news detection based on deep contextualized embedding models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2205.03114"
      },
      "year": 2022,
      "venue": "Neural computing & applications (Print)",
      "citations": 84,
      "notes": "Social media is becoming a source of news for many people due to its ease and freedom of use.",
      "metrics": null
    },
    {
      "id": "arabart-a-pretrained-arabic-sequence-to-sequence-model-for-a",
      "name": "AraBART: a Pretrained Arabic Sequence-to-Sequence Model for Abstractive Summarization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2203.10945"
      },
      "year": 2022,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 83,
      "notes": "Like most natural language understanding and generation tasks, state-of-the-art models for summarization are transformer-based sequence-to-sequence",
      "metrics": null
    },
    {
      "id": "sign-language-recognition-for-arabic-alphabets-using-transfe",
      "name": "Sign Language Recognition for Arabic Alphabets Using Transfer Learning Technique",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2022/4567989"
      },
      "year": 2022,
      "venue": "Computational Intelligence and Neuroscience",
      "citations": 82,
      "notes": "Sign language is essential for deaf and mute people to communicate with normal people and themselves.",
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-based-on-textual-analysis",
      "name": "Arabic Fake News Detection Based on Textual Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13369-021-06449-y"
      },
      "year": 2022,
      "venue": "The Arabian journal for science and engineering",
      "citations": 78,
      "notes": "Over the years, social media has had a considerable impact on the way we share information and send messages.",
      "metrics": null
    },
    {
      "id": "abstractive-arabic-text-summarization-based-on-deep-learning",
      "name": "Abstractive Arabic Text Summarization Based on Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2022/1566890"
      },
      "year": 2022,
      "venue": "Computational Intelligence and Neuroscience",
      "citations": 77,
      "notes": "Text summarization (TS) is considered one of the most difficult tasks in natural language processing (NLP).",
      "metrics": null
    },
    {
      "id": "a-comprehensive-review-of-arabic-text-summarization",
      "name": "A Comprehensive Review of Arabic Text Summarization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2022.3163292"
      },
      "year": 2022,
      "venue": "IEEE Access",
      "citations": 72,
      "notes": "The explosion of online and offline data has changed how we gather, evaluate, and understand data.",
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-using-deep-learning",
      "name": "Arabic Fake News Detection Using Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.32604/cmc.2022.021449"
      },
      "year": 2022,
      "venue": "Computers Materials & Continua",
      "citations": 71,
      "notes": ": Nowadays, an unprecedented number of users interact through social media platforms and generate a massive amount of content due to the explosion of online",
      "metrics": null
    },
    {
      "id": "heterogeneous-ensemble-deep-learning-model-for-enhanced-arab",
      "name": "Heterogeneous Ensemble Deep Learning Model for Enhanced Arabic Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/s22103707"
      },
      "year": 2022,
      "venue": "Italian National Conference on Sensors",
      "citations": 66,
      "notes": "Sentiment analysis was nominated as a hot research topic a decade ago for its increasing importance in analyzing the people’s opinions extracted from social",
      "metrics": null
    },
    {
      "id": "mawqif-a-multi-label-arabic-dataset-for-target-specific-stan",
      "name": "Mawqif: A Multi-label Arabic Dataset for Target-specific Stance Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.16/"
      },
      "year": 2022,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 53,
      "notes": "Social media platforms are becoming inherent parts of people’s daily life to express opinions and stances toward topics of varying polarities.",
      "metrics": null
    },
    {
      "id": "zaebuc-an-annotated-arabic-english-bilingual-writer-corpus",
      "name": "ZAEBUC: An Annotated Arabic-English Bilingual Writer Corpus",
      "type": "paper",
      "country": "AE",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.lrec-1.9/"
      },
      "year": 2022,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 53,
      "notes": "We present ZAEBUC, an annotated Arabic-English bilingual writer corpus comprising short essays by first-year university students at Zayed University in the",
      "metrics": null
    },
    {
      "id": "armis-the-arabic-misogyny-and-sexism-corpus-with-annotator-s",
      "name": "ArMIS - The Arabic Misogyny and Sexism Corpus with Annotator Subjective Disagreements",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.lrec-1.244/"
      },
      "year": 2022,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 42,
      "notes": "The use of misogynistic and sexist language has increased in recent years in social media, and is increasing in the Arabic world in reaction to reforms",
      "metrics": null
    },
    {
      "id": "hierarchical-aggregation-of-dialectal-data-for-arabic-dialec",
      "name": "Hierarchical Aggregation of Dialectal Data for Arabic Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.lrec-1.489/"
      },
      "year": 2022,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 26,
      "notes": "Arabic is a collection of dialectal variants that are historically related but significantly different.",
      "metrics": null
    },
    {
      "id": "aradepsu-detecting-depression-and-suicidal-ideation-in-arabi",
      "name": "AraDepSu: Detecting Depression and Suicidal Ideation in Arabic Tweets Using Transformers",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.28/"
      },
      "year": 2022,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 23,
      "notes": "Among mental health diseases, depression is one of the most severe, as it often leads to suicide which is the fourth leading cause of death in the Middle East.",
      "metrics": null
    },
    {
      "id": "camel-treebank-an-open-multi-genre-arabic-dependency-treeban",
      "name": "Camel Treebank: An Open Multi-genre Arabic Dependency Treebank",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.lrec-1.286/"
      },
      "year": 2022,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 19,
      "notes": "We present the Camel Treebank (CAMELTB), a 188K word open-source dependency treebank of Modern Standard and Classical Arabic.",
      "metrics": null
    },
    {
      "id": "the-effect-of-arabic-dialect-familiarity-on-data-annotation",
      "name": "The Effect of Arabic Dialect Familiarity on Data Annotation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.39/"
      },
      "year": 2022,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 18,
      "notes": "Data annotation is the foundation of most natural language processing (NLP) tasks.",
      "metrics": null
    },
    {
      "id": "armath-a-dataset-for-solving-arabic-math-word-problems",
      "name": "ArMATH: a Dataset for Solving Arabic Math Word Problems",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.lrec-1.37/",
        "github": "https://github.com/reem-codes/ArMATH"
      },
      "year": 2022,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 16,
      "notes": "This paper studies solving Arabic Math Word Problems by deep learning.",
      "metrics": null
    },
    {
      "id": "emoji-sentiment-roles-for-sentiment-analysis-a-case-study-in",
      "name": "Emoji Sentiment Roles for Sentiment Analysis: A Case Study in Arabic Texts",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.32/"
      },
      "year": 2022,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 15,
      "notes": "Emoji (digital pictograms) are crucial features for textual sentiment analysis.",
      "metrics": null
    },
    {
      "id": "a-benchmark-study-of-contrastive-learning-for-arabic-social-meaning",
      "name": "A Benchmark Study of Contrastive Learning for Arabic Social Meaning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2210.12314"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "In this work, we present a comprehensive benchmark study of state-of-the-art supervised CL methods on a wide array of Arabic social meaning tasks.",
      "metrics": null
    },
    {
      "id": "a-deep-cnn-architecture-with-novel-pooling-layer-applied-to-two-sudane",
      "name": "A Deep CNN Architecture with Novel Pooling Layer Applied to Two Sudanese Arabic Sentiment Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2201.12664"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "sudanese",
        "msa",
        "mixed"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "Arabic sentiment analysis has become an important research field in recent years.",
      "metrics": null
    },
    {
      "id": "a-large-and-diverse-arabic-corpus-for-language-modeling",
      "name": "A Large and Diverse Arabic Corpus for Language Modeling",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2201.09227"
      },
      "dialects": [
        "lev"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "Language models (LMs) have introduced a major paradigm shift in Natural Language Processing.",
      "metrics": null
    },
    {
      "id": "a-pilot-study-on-the-collection-and-computational-analysis-of-linguist",
      "name": "A Pilot Study on the Collection and Computational Analysis of Linguistic Differences Amongst Men and Women in a Kuwaiti Arabic WhatsApp Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "University of Sheffield",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialogue"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.35/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "This study focuses on the collection and computational analysis of Kuwaiti Arabic (KA), which is considered a low resource dialect, to test different.",
      "metrics": null
    },
    {
      "id": "adversarial-text-to-speech-for-low-resource-languages",
      "name": "Adversarial Text-to-Speech for low-resource languages",
      "type": "paper",
      "country": "INTL",
      "org": "African Institute for Mathematical Sciences",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.8/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We propose a new method for training adversarial text-to-speech (TTS) models for low-resource languages using auxiliary data.",
      "metrics": null
    },
    {
      "id": "an-end-to-end-ocr-framework-for-robust-arabic-handwriting-recognition",
      "name": "An End-to-End OCR Framework for Robust Arabic-Handwriting Recognition using a Novel Transformers-based Model and an Innovative 270 Million-Words Multi-Font Corpus of Classical Arabic with Diacritics",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "diacritization",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2208.11484"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "Notably, we propose an end-to-end text recognition approach using Vision Transformers as an encoder, namely BEIT.",
      "metrics": null
    },
    {
      "id": "arabgend-gender-analysis-and-inference-on-arabic-twitter",
      "name": "ArabGend: Gender Analysis and Inference on Arabic Twitter",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2203.00271",
        "website": "http://anonymous.com"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "Gender analysis of Twitter can reveal important socio-cultural differences between male and female users.",
      "metrics": null
    },
    {
      "id": "arabsign-a-multi-modality-dataset-and-benchmark-for-continuous-arabic",
      "name": "ArabSign: A Multi-modality Dataset and Benchmark for Continuous Arabic Sign Language Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2210.03951"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "To benchmark this dataset, we propose an encoder-decoder model for Continuous ArSL recognition.",
      "metrics": null
    },
    {
      "id": "aranpcc-the-arabic-newspaper-covid-19-corpus",
      "name": "AraNPCC: The Arabic Newspaper COVID-19 Corpus",
      "type": "paper",
      "country": "SA",
      "org": "The National Center for Data Analytics and Artificial Intelligence",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "corpus",
        "covid"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.osact-1.4/"
      },
      "year": 2022,
      "venue": "OSACT 2022",
      "notes": "Arabic newspaper corpus on COVID-19 covering 2019-2021 from 12 Arab countries, with over 2 billion words.",
      "metrics": null
    },
    {
      "id": "arasas-the-open-source-arabic-semantic-tagger",
      "name": "AraSAS: The Open Source Arabic Semantic Tagger",
      "type": "paper",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "semantic-tagging"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.osact-1.3/",
        "github": "https://github.com/UCREL/AraSAS"
      },
      "year": 2022,
      "venue": "OSACT 2022",
      "notes": "First open-source Arabic semantic analysis tagging system, built on the UCREL Semantic Analysis System.",
      "metrics": null
    },
    {
      "id": "arzen-st-a-three-way-speech-translation-corpus-for-code-switched-egypt",
      "name": "ArzEn-ST: A Three-way Speech Translation Corpus for Code-Switched Egyptian Arabic - English",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2211.12000"
      },
      "dialects": [
        "egy"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "We present our work on collecting ArzEn-ST, a code-switched Egyptian Arabic - English Speech Translation Corpus.",
      "metrics": null
    },
    {
      "id": "beyond-arabic-software-for-perso-arabic-script-manipulation",
      "name": "Beyond Arabic: Software for Perso-Arabic Script Manipulation",
      "type": "paper",
      "country": "INTL",
      "org": "Google Research",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.36/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "This paper presents an open-source software library that provides a set of finite-state transducer (FST) components and corresponding utilities.",
      "metrics": null
    },
    {
      "id": "cross-lingual-transfer-for-low-resource-arabic-language-understanding",
      "name": "Cross-lingual transfer for low-resource Arabic language understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.21/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We adopt a BERT-based architecture and pretrain three models using open-source Wikipedia data and large-scale commercial datasets: monolingual:Arabic.",
      "metrics": null
    },
    {
      "id": "cs-um6p-at-semeval-2022-task-6-transformer-based-models-for-intended-s",
      "name": "CS-UM6P at SemEval-2022 Task 6: Transformer-based Models for Intended Sarcasm Detection in English and Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "abdelkadermh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2206.08415",
        "github": "https://github.com/AbdelkaderMH/iSarcasmEval"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "In this paper, we present our participating system to the intended sarcasm detection task in English and Arabic languages.",
      "metrics": null
    },
    {
      "id": "dialect-sentiment-identification-in-nuanced-arabic-tweets-using-an-ens",
      "name": "Dialect & Sentiment Identification in Nuanced Arabic Tweets Using an Ensemble of Prompt-based, Fine-tuned, and Multitask BERT-Based Models",
      "type": "paper",
      "country": "EG",
      "org": "Cairo University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.48/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We present our findings and results in the Nuanced Arabic Dialect Identification Shared Task (NADI 2022) for country-level dialect identification.",
      "metrics": null
    },
    {
      "id": "exaasc-a-general-target-based-stance-detection-corpus-in-arabic-langua",
      "name": "ExaASC: A General Target-Based Stance Detection Corpus in Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2204.13979"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "This work proposes a new method toward Target-based Stance detection by using the stance of replies toward a most important and arguing target in source tweet.",
      "metrics": null
    },
    {
      "id": "gulf-arabic-diacritization-guidelines-initial-dataset-and-results",
      "name": "Gulf Arabic Diacritization: Guidelines, Initial Dataset, and Results",
      "type": "paper",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.33/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We introduce a new Gulf Arabic diacritization dataset composed of 19,850 words based on a subset of the Gumar corpus.",
      "metrics": null
    },
    {
      "id": "harnessing-multilingual-resources-to-question-answering-in-arabic",
      "name": "Harnessing Multilingual Resources to Question Answering in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "quran"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2205.08024"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "The goal of the paper is to predict answers to questions given a passage of Qur'an.",
      "metrics": null
    },
    {
      "id": "lans-large-scale-arabic-news-summarization-corpus",
      "name": "LANS: Large-scale Arabic News Summarization Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2210.13600"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "We build, LANS, a large-scale and diverse dataset for Arabic Text Summarization task.",
      "metrics": null
    },
    {
      "id": "masader-plus-a-new-interface-for-exploring-500-arabic-nlp-datasets",
      "name": "Masader Plus: A New Interface for Exploring +500 Arabic NLP Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2208.00932",
        "website": "https://arbml.github.io/masader"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "In this paper, we introduce Masader Plus, a web interface for users to browse Masader.",
      "metrics": null
    },
    {
      "id": "nadi-2022-the-third-nuanced-arabic-dialect-identification-shared-task",
      "name": "NADI 2022: The Third Nuanced Arabic Dialect Identification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2210.09582"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "We describe findings of the third Nuanced Arabic Dialect Identification Shared Task (NADI 2022).",
      "metrics": null
    },
    {
      "id": "natiq-an-end-to-end-text-to-speech-system-for-arabic",
      "name": "NatiQ: An End-to-end Text-to-Speech System for Arabic",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.38/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "End-to-end Arabic text-to-speech system using Tacotron-based and faster transformer models with encoder-decoder attention.",
      "metrics": null
    },
    {
      "id": "new-results-for-the-text-recognition-of-arabic-maghrib-manuscripts-man",
      "name": "New Results for the Text Recognition of Arabic Maghrib{ī} Manuscripts -- Managing an Under-resourced Script",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "asr",
        "ocr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2211.16147"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "HTR models development has become a conventional step for digital humanities projects.",
      "metrics": null
    },
    {
      "id": "offensive-language-detection-in-under-resourced-algerian-dialectal-ara",
      "name": "Offensive Language Detection in Under-resourced Algerian Dialectal Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2203.10024"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "This paper addresses the problem of detecting the offensive and abusive content in Facebook comments.",
      "metrics": null
    },
    {
      "id": "on-the-arabic-dialects-identification-overcoming-challenges-of-geograp",
      "name": "On The Arabic Dialects’ Identification: Overcoming Challenges of Geographical Similarities Between Arabic dialects and Imbalanced Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "School of Computer Science",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.49/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We present a solution to tackle subtask 1 (Country-level dialect identification) of the Nuanced Arabic Dialect Identification (NADI) shared task 2022.",
      "metrics": null
    },
    {
      "id": "orca-a-challenging-benchmark-for-arabic-language-understanding",
      "name": "ORCA: A Challenging Benchmark for Arabic Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2212.10758"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "In this work, we introduce ORCA, a publicly available benchmark for Arabic language understanding evaluation.",
      "metrics": null
    },
    {
      "id": "overview-of-osact5-shared-task-on-arabic-offensive-language-and-hate-s",
      "name": "Overview of OSACT5 Shared Task on Arabic Offensive Language and Hate Speech Detection",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "offensive-language",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.osact-1.20/"
      },
      "year": 2022,
      "venue": "OSACT 2022",
      "notes": "This paper provides an overview of the shard task on detecting offensive language, hate speech, and fine-grained hate speech at the fifth workshop.",
      "metrics": null
    },
    {
      "id": "overview-of-the-wanlp-2022-shared-task-on-propaganda-detection-in-arab",
      "name": "Overview of the WANLP 2022 Shared Task on Propaganda Detection in Arabic",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "propaganda",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.11/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We propose a shared task on detecting propaganda techniques for Arabic textual content.",
      "metrics": null
    },
    {
      "id": "sa-7r-a-saudi-dialect-irony-dataset",
      "name": "Sa‘7r: A Saudi Dialect Irony Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "College of Computer and Information Sciences",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.osact-1.7/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2022,
      "venue": "OSACT 2022",
      "notes": "We present Sa‘7r ساخرthe Saudi irony dataset, and describe our efforts in constructing it.",
      "metrics": null
    },
    {
      "id": "the-arabic-ontology-an-arabic-wordnet-with-ontologically-clean-content",
      "name": "The Arabic Ontology -- An Arabic Wordnet with Ontologically Clean Content",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2205.09664",
        "website": "https://ontology.birzeit.edu"
      },
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "We present a formal Arabic wordnet built on the basis of a carefully designed ontology hereby referred to as the Arabic Ontology.",
      "metrics": null
    },
    {
      "id": "towards-arabic-sentence-simplification-via-classification-and-generati",
      "name": "Towards Arabic Sentence Simplification via Classification and Generative Approaches",
      "type": "paper",
      "country": "INTL",
      "org": "nouran-khallaf",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2204.09292"
      },
      "dialects": [
        "msa"
      ],
      "year": 2022,
      "venue": "arXiv 2022",
      "notes": "This paper presents an attempt to build a Modern Standard Arabic (MSA) sentence-level simplification system.",
      "metrics": null
    },
    {
      "id": "weakly-and-semi-supervised-learning-for-arabic-text-classification-usi",
      "name": "Weakly and Semi-Supervised Learning for Arabic Text Classification using Monodialectal Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Center for Integrative Petroleum Research (CIPR)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.24/"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "To that end, we propose a suite of language-agnostic techniques for large-scale data collection, automatic data annotation, and language model training.",
      "metrics": null
    },
    {
      "id": "wojood-nested-arabic-named-entity-corpus-and-recognition",
      "name": "Wojood: Nested Arabic Named Entity Corpus and Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2022,
      "venue": "arXiv 2022",
      "tasks": [
        "ner",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2205.09651"
      },
      "notes": "Nested named entity corpus in MSA and dialect, with a recognition baseline.",
      "metrics": null
    },
    {
      "id": "hyperparameter-tuning-for-machine-learning-algorithms-used-f",
      "name": "Hyperparameter Tuning for Machine Learning Algorithms Used for Arabic Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/informatics8040079"
      },
      "year": 2021,
      "venue": "Informatics",
      "citations": 392,
      "notes": "In this paper, a comprehensive comparative analysis of various",
      "metrics": null
    },
    {
      "id": "bert-for-arabic-topic-modeling-an-experimental-study-on-bert",
      "name": "BERT for Arabic Topic Modeling: An Experimental Study on BERTopic Technique",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2021.05.096"
      },
      "year": 2021,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 176,
      "notes": "Topic modeling is an unsupervised machine learning technique for finding abstract topics in a large collection of documents.",
      "metrics": null
    },
    {
      "id": "pre-training-bert-on-arabic-tweets-practical-considerations",
      "name": "Pre-Training BERT on Arabic Tweets: Practical Considerations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2102.10684"
      },
      "year": 2021,
      "venue": "arXiv.org",
      "citations": 157,
      "notes": "We pretrained 5 BERT models that differ in the size of their training sets, mixture of formal and informal Arabic, and linguistic preprocessing.",
      "metrics": null
    },
    {
      "id": "arabic-aspect-based-sentiment-analysis-using-bidirectional-g",
      "name": "Arabic aspect based sentiment analysis using bidirectional GRU based models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2101.10539"
      },
      "year": 2021,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 118,
      "notes": "Aspect-based Sentiment analysis (ABSA) accomplishes a fine-grained analysis that defines the aspects of a given document or sentence and the sentiments conveyed",
      "metrics": null
    },
    {
      "id": "a-comparative-study-of-effective-approaches-for-arabic-senti",
      "name": "A comparative study of effective approaches for Arabic sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.ipm.2020.102438"
      },
      "year": 2021,
      "venue": "Information Processing & Management",
      "citations": 99,
      "notes": "Sentiment analysis (SA) is a natural language processing (NLP) application that aims to analyse and identify sentiment within a piece of text.",
      "metrics": null
    },
    {
      "id": "detection-of-hate-speech-in-arabic-tweets-using-deep-learnin",
      "name": "Detection of hate speech in Arabic tweets using deep learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s00530-020-00742-w"
      },
      "year": 2021,
      "venue": "Multimedia Systems",
      "citations": 99,
      "metrics": null
    },
    {
      "id": "arabic-fake-news-detection-comparative-study-of-neural-netwo",
      "name": "Arabic Fake News Detection: Comparative Study of Neural Networks and Transformer-Based Approaches",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2021/5516945"
      },
      "year": 2021,
      "venue": "Complex",
      "citations": 94,
      "notes": "Fake news detection (FND) involves predicting the likelihood that a particular news article (news report, editorial, expose, etc.) is intentionally deceptive.",
      "metrics": null
    },
    {
      "id": "a-survey-of-offensive-language-detection-for-the-arabic-lang",
      "name": "A Survey of Offensive Language Detection for the Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1145/3421504"
      },
      "year": 2021,
      "venue": "ACM Trans. Asian Low Resour. Lang. Inf. Process.",
      "citations": 91,
      "notes": "The use of offensive language in user-generated content is a serious problem that needs to be addressed with the latest technology.",
      "metrics": null
    },
    {
      "id": "multi-label-arabic-text-classification-in-online-social-netw",
      "name": "Multi-label Arabic text classification in Online Social Networks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.IS.2021.101785"
      },
      "year": 2021,
      "venue": "Information Systems",
      "citations": 82,
      "metrics": null
    },
    {
      "id": "arabic-offensive-and-hate-speech-detection-using-a-cross-cor",
      "name": "Arabic Offensive and Hate Speech Detection Using a Cross-Corpora Multi-Task Learning Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/informatics8040069"
      },
      "year": 2021,
      "venue": "Informatics",
      "citations": 81,
      "notes": "As social media platforms offer a medium for opinion expression, social phenomena such as hatred, offensive language, racism, and all forms of verbal violence",
      "metrics": null
    },
    {
      "id": "benchmarking-transformer-based-language-models-for-arabic-se",
      "name": "Benchmarking Transformer-based Language Models for Arabic Sentiment and Sarcasm Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.3/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 80,
      "notes": "In this paper, we evaluate the performance of 24 of these models on Arabic sentiment and sarcasm detection",
      "metrics": null
    },
    {
      "id": "fake-news-detection-in-arabic-tweets-during-the-covid-19-pan",
      "name": "Fake News Detection in Arabic Tweets during the COVID-19 Pandemic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.14569/IJACSA.2021.0120691"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "citations": 80,
      "notes": "In March 2020, the World Health Organization declared the COVID-19 outbreak to be a pandemic.",
      "metrics": null
    },
    {
      "id": "a-morphologically-annotated-corpus-and-a-morphological-analy",
      "name": "A Morphologically Annotated Corpus and a Morphological Analyzer for Egyptian Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2021.05.084"
      },
      "year": 2021,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 77,
      "metrics": null
    },
    {
      "id": "enhancing-arabic-aspect-based-sentiment-analysis-using-deep",
      "name": "Enhancing Arabic aspect-based sentiment analysis using deep learning models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.CSL.2021.101224"
      },
      "year": 2021,
      "venue": "Computer Speech and Language",
      "citations": 77,
      "metrics": null
    },
    {
      "id": "preprocessing-arabic-text-on-social-media",
      "name": "Preprocessing Arabic text on social media",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.heliyon.2021.e06191"
      },
      "year": 2021,
      "venue": "Heliyon",
      "citations": 77,
      "notes": "Currently, social media plays an important role in daily life and routine.",
      "metrics": null
    },
    {
      "id": "dziribert-a-pre-trained-language-model-for-the-algerian-dial",
      "name": "DziriBERT: a Pre-trained Language Model for the Algerian Dialect",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2109.12346"
      },
      "year": 2021,
      "venue": "arXiv.org",
      "citations": 73,
      "notes": "Pre-trained transformers are now the de facto models in Natural Language Processing given their state-of-the-art results in many tasks and languages.",
      "metrics": null
    },
    {
      "id": "arabic-question-answering-system-a-survey",
      "name": "Arabic question answering system: a survey",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "qa"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10462-021-10031-1"
      },
      "year": 2021,
      "venue": "Artificial Intelligence Review",
      "citations": 71,
      "metrics": null
    },
    {
      "id": "aracovid19-mfh-arabic-covid-19-multi-label-fake-news-hate-sp",
      "name": "AraCOVID19-MFH: Arabic COVID-19 Multi-label Fake News & Hate Speech Detection Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2021.05.086"
      },
      "year": 2021,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 70,
      "notes": "Along with the COVID-19 pandemic, an \"infodemic\" of false and misleading information has emerged and has complicated the COVID-19 response efforts.",
      "metrics": null
    },
    {
      "id": "arabic-aspect-based-sentiment-analysis-a-systematic-literatu",
      "name": "Arabic Aspect-Based Sentiment Analysis: A Systematic Literature Review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2021.3127140"
      },
      "year": 2021,
      "venue": "IEEE Access",
      "citations": 69,
      "notes": "Recently sentiment analysis in Arabic has attracted much attention from researchers.",
      "metrics": null
    },
    {
      "id": "arabic-natural-language-processing-for-qur-anic-research-a-s",
      "name": "Arabic natural language processing for Qur’anic research: a systematic review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "quran"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10462-022-10313-2"
      },
      "year": 2021,
      "venue": "Artificial Intelligence Review",
      "citations": 69,
      "notes": "The Qur’an is a fourteen centuries old divine book in Arabic language that is read and followed by almost two billion Muslims globally as their sacred religious",
      "metrics": null
    },
    {
      "id": "unsupervised-neural-networks-for-automatic-arabic-text-summa",
      "name": "Unsupervised neural networks for automatic Arabic text summarization using document clustering and topic modeling",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.eswa.2021.114652"
      },
      "year": 2021,
      "venue": "Expert systems with applications",
      "citations": 69,
      "metrics": null
    },
    {
      "id": "deep-learning-for-emotion-analysis-in-arabic-tweets",
      "name": "Deep learning for emotion analysis in Arabic tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1186/s40537-021-00523-w"
      },
      "year": 2021,
      "venue": "Journal of Big Data",
      "citations": 64,
      "notes": "Currently, expressing feelings through social media requires great consideration as an essential part of our lives; besides sharing ideas and thoughts, we share",
      "metrics": null
    },
    {
      "id": "evaluating-various-tokenizers-for-arabic-text-classification",
      "name": "Evaluating Various Tokenizers for Arabic Text Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing",
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.07540"
      },
      "year": 2021,
      "venue": "Neural Processing Letters",
      "citations": 64,
      "notes": "The first step in any NLP pipeline is to split the text into individual tokens.",
      "metrics": null
    },
    {
      "id": "systematic-literature-review-of-dialectal-arabic-identificat",
      "name": "Systematic Literature Review of Dialectal Arabic: Identification and Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2021.3059504"
      },
      "year": 2021,
      "venue": "IEEE Access",
      "citations": 64,
      "notes": "This study comes to chart the field by conducting a systematic literature review that is intended to give insight into the most and least popular research",
      "metrics": null
    },
    {
      "id": "towards-arabic-aspect-based-sentiment-analysis-a-transfer-le",
      "name": "Towards Arabic aspect-based sentiment analysis: a transfer learning-based approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13278-021-00794-4"
      },
      "year": 2021,
      "venue": "Social Network Analysis and Mining",
      "citations": 64,
      "metrics": null
    },
    {
      "id": "a-deep-learning-framework-for-automatic-detection-of-hate-sp",
      "name": "A Deep Learning Framework for Automatic Detection of Hate Speech Embedded in Arabic Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13369-021-05383-3"
      },
      "year": 2021,
      "venue": "The Arabian journal for science and engineering",
      "citations": 63,
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-by-end-to-end-modular-systems-and",
      "name": "Arabic Speech Recognition by End-to-End, Modular Systems and Human",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2101.08454"
      },
      "year": 2021,
      "venue": "Computer Speech and Language",
      "citations": 63,
      "notes": "Recent advances in automatic speech recognition (ASR) have achieved accuracy levels comparable to human transcribers, which led researchers to debate if the",
      "metrics": null
    },
    {
      "id": "novel-deep-convolutional-neural-network-based-contextual-rec",
      "name": "Novel Deep Convolutional Neural Network-Based Contextual Recognition of Arabic Handwritten Scripts",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/e23030340"
      },
      "year": 2021,
      "venue": "Entropy",
      "citations": 63,
      "notes": "Offline Arabic Handwriting Recognition (OAHR) has recently become instrumental in the areas of pattern recognition and image processing due to its application",
      "metrics": null
    },
    {
      "id": "recognizing-arabic-handwritten-characters-using-deep-learnin",
      "name": "Recognizing arabic handwritten characters using deep learning and genetic algorithms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s11042-021-11185-4"
      },
      "year": 2021,
      "venue": "Multimedia tools and applications",
      "citations": 63,
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-using-end-to-end-deep-learning",
      "name": "Arabic speech recognition using end-to-end deep learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://doi.org/10.1049/SIL2.12057"
      },
      "year": 2021,
      "venue": "IET Signal Processing",
      "citations": 62,
      "notes": "In this work, the application of state-of-the-art end-to-end deep learning approaches is inves-tigated to build a",
      "metrics": null
    },
    {
      "id": "part-of-speech-tagging-for-arabic-tweets-using-crf-and-bi-ls",
      "name": "Part-of-speech tagging for Arabic tweets using CRF and Bi-LSTM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.csl.2020.101138"
      },
      "year": 2021,
      "venue": "Computer Speech and Language",
      "citations": 61,
      "notes": "Over the past few years, Twitter has experienced massive growth and the volume of its online content has increased rapidly.",
      "metrics": null
    },
    {
      "id": "towards-one-model-to-rule-all-multilingual-strategy-for-dial",
      "name": "Towards One Model to Rule All: Multilingual Strategy for Dialectal Code-Switching Arabic ASR",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2105.14779"
      },
      "year": 2021,
      "venue": "Interspeech",
      "citations": 59,
      "notes": "With the advent of globalization, there is an increasing demand for multilingual automatic speech recognition (ASR), handling language and dialectal variation",
      "metrics": null
    },
    {
      "id": "sarcasm-and-sentiment-detection-in-arabic-tweets-using-bert",
      "name": "Sarcasm and Sentiment Detection In Arabic Tweets Using BERT-based Models and Data Augmentation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.38/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 52,
      "notes": "In this paper, we describe our efforts on the shared task of sarcasm and sentiment detection in Arabic (Abu Farha et al., 2021)",
      "metrics": null
    },
    {
      "id": "deep-multi-task-model-for-sarcasm-detection-and-sentiment-an",
      "name": "Deep Multi-Task Model for Sarcasm Detection and Sentiment Analysis in Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.12488"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 50,
      "notes": "While previous research works tackle SA and sarcasm detection separately, this paper introduces an end-to-end deep Multi-Task Learning (MTL) model, allowing",
      "metrics": null
    },
    {
      "id": "alue-arabic-language-understanding-evaluation",
      "name": "ALUE: Arabic Language Understanding Evaluation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.18/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 48,
      "notes": "We strongly believe thatmany NLU problems in Arabic are especiallypoised to reap the benefits of such models",
      "metrics": null
    },
    {
      "id": "domain-adaptation-for-arabic-cross-domain-and-cross-dialect",
      "name": "Domain Adaptation for Arabic Cross-Domain and Cross-Dialect Sentiment Analysis from Contextualized Word Embedding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "embedding"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.naacl-main.226/"
      },
      "year": 2021,
      "venue": "North American Chapter of the Association for Computational Linguistics",
      "citations": 42,
      "notes": "Finetuning deep pre-trained language models has shown state-of-the-art performances on a wide range of Natural Language Processing (NLP) applications.",
      "metrics": null
    },
    {
      "id": "bert-transformer-model-for-detecting-arabic-gpt2-auto-genera",
      "name": "Bert Transformer model for Detecting Arabic GPT2 Auto-Generated Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2101.09345"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 34,
      "notes": "During the last two decades, we have progressively turned to the Internet and social media to find news, entertain conversations and share opinion.",
      "metrics": null
    },
    {
      "id": "asad-arabic-social-media-analytics-and-understanding-2021",
      "name": "ASAD: Arabic Social media Analytics and unDerstanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "hate-speech"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.eacl-demos.14/"
      },
      "year": 2021,
      "venue": "Conference of the European Chapter of the Association for Computational Linguistics",
      "citations": 32,
      "notes": "This system demonstration paper describes ASAD: Arabic Social media Analysis and unDerstanding, a suite of seven individual modules that allows users to",
      "metrics": null
    },
    {
      "id": "multi-task-learning-using-a-combination-of-contextualised-an",
      "name": "Multi-task Learning Using a Combination of Contextualised and Static Word Embeddings for Arabic Sarcasm Detection and Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "embedding",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.39/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 29,
      "notes": "In this study, we exploited this relationship to enhance both tasks by proposing a multi-task learning approach using a combination of static and",
      "metrics": null
    },
    {
      "id": "leveraging-offensive-language-for-sarcasm-and-sentiment-dete",
      "name": "Leveraging Offensive Language for Sarcasm and Sentiment Detection in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.47/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 28,
      "notes": "We propose two systems that harness knowledge from multiple tasks to improve the performance of the classifier",
      "metrics": null
    },
    {
      "id": "a-contextual-word-embedding-for-arabic-sarcasm-detection-wit",
      "name": "A Contextual Word Embedding for Arabic Sarcasm Detection with Random Forests",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "embedding",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.43/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 24,
      "notes": "In this paper, we propose a new approach for improving Arabic sarcasm detection",
      "metrics": null
    },
    {
      "id": "combining-context-free-and-contextualized-representations-fo",
      "name": "Combining Context-Free and Contextualized Representations for Arabic Sarcasm Detection and Sentiment Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2103.05683"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 24,
      "notes": "Since their inception, transformer-based language models have led to impressive performance gains across multiple natural language processing tasks.",
      "metrics": null
    },
    {
      "id": "empathetic-bert2bert-conversational-model-learning-arabic-la",
      "name": "Empathetic BERT2BERT Conversational Model: Learning Arabic Language Generation with Little Data",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "benchmark",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2103.04353"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 23,
      "notes": "Enabling empathetic behavior in Arabic dialogue agents is an important aspect of building human-like conversational models.",
      "metrics": null
    },
    {
      "id": "adapting-marbert-for-improved-arabic-dialect-identification",
      "name": "Adapting MARBERT for Improved Arabic Dialect Identification: Submission to the NADI 2021 Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2103.01065"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 22,
      "notes": "In this paper, we tackle the Nuanced Arabic Dialect Identification (NADI) shared task (Abdul-Mageed et al., 2021) and demonstrate state-of-the-art results on",
      "metrics": null
    },
    {
      "id": "bert-based-multi-task-model-for-country-and-province-level-m",
      "name": "BERT-based Multi-Task Model for Country and Province Level MSA and Dialectal Arabic Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.12495"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 21,
      "notes": "In this paper, we present our deep learning-based system, submitted to the second NADI shared task for country-level and province-level identification of Modern",
      "metrics": null
    },
    {
      "id": "improving-arabic-diacritization-with-regularized-decoding-an",
      "name": "Improving Arabic Diacritization with Regularized Decoding and Adversarial Training",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.acl-short.68/"
      },
      "year": 2021,
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "citations": 20,
      "notes": "Arabic diacritization is a fundamental task for Arabic language processing.",
      "metrics": null
    },
    {
      "id": "the-idc-system-for-sentiment-classification-and-sarcasm-dete",
      "name": "The IDC System for Sentiment Classification and Sarcasm Detection in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.48/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 18,
      "notes": "In this paper we present designated solutions for sentiment classification and sarcasm detection tasks that were introduced as part of a shared task by Abu",
      "metrics": null
    },
    {
      "id": "arsarcasm-shared-task-an-ensemble-bert-model-for-sarcasmdete",
      "name": "ArSarcasm Shared Task: An Ensemble BERT Model for SarcasmDetection in Arabic Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.40/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 17,
      "notes": "In this work, we present our submission of the sub-task1 of the shared task on sarcasm and sentiment detection in Arabic organized by the 6th Workshop for",
      "metrics": null
    },
    {
      "id": "machine-learning-based-approach-for-arabic-dialect-identific",
      "name": "Machine Learning-Based Approach for Arabic Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.34/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 17,
      "notes": "This paper describes our systems submitted to the Second Nuanced Arabic Dialect Identification Shared Task (NADI 2021)",
      "metrics": null
    },
    {
      "id": "wanlp-2021-shared-task-towards-irony-and-sentiment-detection",
      "name": "WANLP 2021 Shared-Task: Towards Irony and Sentiment Detection in Arabic Tweets using Multi-headed-LSTM-CNN-GRU and MaRBERT",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.37/"
      },
      "year": 2021,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 15,
      "notes": "This paper presents results and main findings in WANLP 2021 shared tasks one and two",
      "metrics": null
    },
    {
      "id": "an-enhanced-corpus-for-arabic-newspapers-comments",
      "name": "An Enhanced Corpus for Arabic Newspapers Comments",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2102.09965"
      },
      "dialects": [
        "magh"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "In this paper, we propose our enhanced approach to create a dedicated corpus for Algerian Arabic newspapers comments.",
      "metrics": null
    },
    {
      "id": "an-open-access-nlp-dataset-for-arabic-dialects-data-collection-labelin",
      "name": "An open access NLP dataset for Arabic dialects : Data collection, labeling, and model construction",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2102.11000"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "In this work, we present an open data set of social data content in several Arabic dialects.",
      "metrics": null
    },
    {
      "id": "arabic-offensive-language-on-twitter-analysis-and-experiments",
      "name": "Arabic Offensive Language on Twitter: Analysis and Experiments",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.13/"
      },
      "year": 2021,
      "venue": "WANLP 2021",
      "notes": "We focus on building a large Arabic offensive tweet dataset.",
      "metrics": null
    },
    {
      "id": "arabic-speech-emotion-recognition-employing-wav2vec2-0-and-hubert-base",
      "name": "Arabic Speech Emotion Recognition Employing Wav2vec2.0 and HuBERT Based on BAVED Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "sentiment",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2110.04425"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "This paper introduces a deep learning constructed emotional recognition model for Arabic speech dialogues.",
      "metrics": null
    },
    {
      "id": "aracovid19-mfh-arabic-covid-19-multi-label-fake-news-and-hate-speech-d",
      "name": "AraCOVID19-MFH: Arabic COVID-19 Multi-label Fake News and Hate Speech Detection Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2105.03143"
      },
      "dialects": [
        "lev"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "This paper releases \"AraCOVID19-MFH\" a manually annotated multi-label Arabic COVID-19 fake news and hate speech detection dataset.",
      "metrics": null
    },
    {
      "id": "aracovid19-ssd-arabic-covid-19-sentiment-and-sarcasm-detection-dataset",
      "name": "AraCOVID19-SSD: Arabic COVID-19 Sentiment and Sarcasm Detection Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2110.01948"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "Motivated by the emerging need for annotated datasets that tackle these kinds of problems in the context of COVID-19.",
      "metrics": null
    },
    {
      "id": "arastance-a-multi-country-and-multi-domain-dataset-of-arabic-stance-de",
      "name": "AraStance: A Multi-Country and Multi-Domain Dataset of Arabic Stance Detection for Fact Checking",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2104.13559"
      },
      "dialects": [
        "lev"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We present our new Arabic Stance Detection dataset (AraStance) of 4,063 claim--article pairs from a diverse set of sources.",
      "metrics": null
    },
    {
      "id": "arat5-text-to-text-transformers-for-arabic-language-generation",
      "name": "AraT5: Text-to-Text Transformers for Arabic Language Generation",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2109.12068",
        "github": "https://github.com/UBC-NLP/araT5"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "For evaluation, we introduce a novel benchmark for ARabic language GENeration (ARGEN), covering seven important tasks.",
      "metrics": null
    },
    {
      "id": "arbert-marbert-deep-bidirectional-transformers-for-arabic",
      "name": "ARBERT & MARBERT: Deep Bidirectional Transformers for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2021,
      "venue": "ACL 2021",
      "tasks": [
        "pretraining",
        "encoder",
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.acl-long.551/"
      },
      "notes": "ARBERT and MARBERT: MSA and dialect-heavy encoders, plus the ARLUE benchmark.",
      "metrics": null
    },
    {
      "id": "asmdd-arabic-speech-mispronunciation-detection-dataset",
      "name": "ASMDD: Arabic Speech Mispronunciation Detection Dataset",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2111.01136"
      },
      "dialects": [
        "egy"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "The largest dataset of Arabic speech mispronunciation detections in Egyptian dialogues is introduced.",
      "metrics": null
    },
    {
      "id": "calliar-an-online-handwritten-dataset-for-arabic-calligraphy",
      "name": "Calliar: An Online Handwritten Dataset for Arabic Calligraphy",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.10745"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "Calligraphy is an essential part of the Arabic heritage and culture.",
      "metrics": null
    },
    {
      "id": "exploratory-arabic-offensive-language-dataset-analysis",
      "name": "Exploratory Arabic Offensive Language Dataset Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2101.11434"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "This paper adding more insights towards resources and datasets used in Arabic offensive language research.",
      "metrics": null
    },
    {
      "id": "investigations-on-speech-recognition-systems-for-low-resource-dialecta",
      "name": "Investigations on Speech Recognition Systems for Low-Resource Dialectal Arabic-English Code-Switching Speech",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "asr",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2108.12881"
      },
      "dialects": [
        "egy"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "In this paper, we present our work on code-switched Egyptian Arabic-English automatic speech recognition (ASR).",
      "metrics": null
    },
    {
      "id": "let-mi-an-arabic-levantine-twitter-dataset-for-misogynistic-language",
      "name": "Let-Mi: An Arabic Levantine Twitter Dataset for Misogynistic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2103.10195"
      },
      "dialects": [
        "lev"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "In this paper, we introduce an Arabic Levantine Twitter dataset for Misogynistic language (LeT-Mi) to be the first benchmark dataset for Arabic misogyny.",
      "metrics": null
    },
    {
      "id": "masader-metadata-sourcing-for-arabic-text-and-speech-data-resources",
      "name": "Masader: Metadata Sourcing for Arabic Text and Speech Data Resources",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2110.06744"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "The NLP pipeline has evolved dramatically in the last few years.",
      "metrics": null
    },
    {
      "id": "nadi-2021-the-second-nuanced-arabic-dialect-identification-shared-task",
      "name": "NADI 2021: The Second Nuanced Arabic Dialect Identification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2103.08466"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We present the findings and results of the Second Nuanced Arabic Dialect Identification Shared Task (NADI 2021).",
      "metrics": null
    },
    {
      "id": "new-arabic-medical-dataset-for-diseases-classification",
      "name": "New Arabic Medical Dataset for Diseases Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.15236"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We introduce a new Arab medical dataset, which includes two thousand medical documents collected from several Arabic medical websites.",
      "metrics": null
    },
    {
      "id": "overview-of-the-arabic-sentiment-analysis-2021-competition-at-kaust",
      "name": "Overview of the Arabic Sentiment Analysis 2021 Competition at KAUST",
      "type": "paper",
      "country": "SA",
      "org": "KAUST",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "shared-task"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2109.14456",
        "website": "https://www.kaggle.com/c/arabic-sentiment-analysis-2021-kaust"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "Overview of the KAUST Arabic Sentiment Analysis Challenge on tweets, built on 55K ASAD tweets with three sentiment classes.",
      "metrics": null
    },
    {
      "id": "overview-of-the-wanlp-2021-shared-task-on-sarcasm-and-sentiment-detect",
      "name": "Overview of the WANLP 2021 Shared Task on Sarcasm and Sentiment Detection in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "The University of Edinburgh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "sarcasm",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.36/"
      },
      "year": 2021,
      "venue": "WANLP 2021",
      "notes": "This paper provides an overview of the WANLP 2021 shared task on sarcasm and sentiment detection in Arabic.",
      "metrics": null
    },
    {
      "id": "qasr-qcri-aljazeera-speech-resource-a-large-scale-annotated-arabic-spe",
      "name": "QASR: QCRI Aljazeera Speech Resource -- A Large Scale Annotated Arabic Speech Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id",
        "ner",
        "evaluation",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2106.13000"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We introduce the largest transcribed Arabic speech corpus, QASR, collected from the broadcast domain.",
      "metrics": null
    },
    {
      "id": "self-training-pre-trained-language-models-for-zero-and-few-shot-multi",
      "name": "Self-Training Pre-Trained Language Models for Zero- and Few-Shot Multi-Dialectal Arabic Sequence Labeling",
      "type": "paper",
      "country": "INTL",
      "org": "mohammadkhalifa",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2101.04758",
        "github": "https://github.com/mohammadKhalifa/zero-shot-arabic-dialects"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We propose to self-train pre-trained language models in zero- and few-shot scenarios to improve performance on data-scarce varieties using only resources.",
      "metrics": null
    },
    {
      "id": "sexism-detection-the-first-corpus-in-algerian-dialect-with-a-code-swit",
      "name": "Sexism detection: The first corpus in Algerian dialect with a code-switching in Arabic/ French and English",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2104.01443"
      },
      "dialects": [
        "magh"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "In this paper, an approach for hate speech detection against women in Arabic community on social media (e.g.",
      "metrics": null
    },
    {
      "id": "the-arabic-parallel-gender-corpus-2-0-extensions-and-analyses",
      "name": "The Arabic Parallel Gender Corpus 2.0: Extensions and Analyses",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2110.09216"
      },
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "We introduce a new corpus for gender identification and rewriting in contexts involving one or two target users (I and/or You).",
      "metrics": null
    },
    {
      "id": "the-interplay-of-variant-size-and-task-type-in-arabic-pre",
      "name": "The Interplay of Variant, Size, and Task Type in Arabic Pre-trained Language Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2021,
      "venue": "WANLP 2021",
      "tasks": [
        "pretraining",
        "encoder"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.10/"
      },
      "notes": "Studies how variant, size and task type shape Arabic pretrained encoders (CAMeLBERT).",
      "metrics": null
    },
    {
      "id": "ul2c-mapping-user-locations-to-countries-on-arabic-twitter",
      "name": "UL2C: Mapping User Locations to Countries on Arabic Twitter",
      "type": "paper",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.15/"
      },
      "year": 2021,
      "venue": "WANLP 2021",
      "notes": "We present the largest manually labeled dataset for mapping user locations on Arabic Twitter to their corresponding countries.",
      "metrics": null
    },
    {
      "id": "zero-resource-multi-dialectal-arabic-natural-language-understanding",
      "name": "Zero-Resource Multi-Dialectal Arabic Natural Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "sentiment",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2104.06591"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2021,
      "venue": "arXiv 2021",
      "notes": "To remedy such performance drop, we propose self-training with unlabeled DA data and apply it in the context of named entity recognition.",
      "metrics": null
    },
    {
      "id": "arabic-text-classification-using-deep-learning-models",
      "name": "Arabic text classification using deep learning models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.ipm.2019.102121"
      },
      "year": 2020,
      "venue": "Information Processing & Management",
      "citations": 238,
      "notes": "Text classification or categorization is the process of automatically tagging a textual document with most relevant labels or categories.",
      "metrics": null
    },
    {
      "id": "a-review-of-sentiment-analysis-research-in-arabic-language",
      "name": "A review of sentiment analysis research in Arabic language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2005.12240"
      },
      "year": 2020,
      "venue": "Future generations computer systems",
      "citations": 204,
      "notes": "Sentiment analysis is a task of natural language processing that has recently attracted increasing attention.",
      "metrics": null
    },
    {
      "id": "araelectra-pre-training-text-discriminators-for-arabic-langu",
      "name": "AraELECTRA: Pre-Training Text Discriminators for Arabic Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2012.15516"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 176,
      "notes": "Advances in English language representation enabled a more sample-efficient pre-training task by Efficiently Learning an Encoder that Classifies Token",
      "metrics": null
    },
    {
      "id": "deep-learning-cnn-lstm-framework-for-arabic-sentiment-analys",
      "name": "Deep learning CNN–LSTM framework for Arabic sentiment analysis using textual information shared in social networks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13278-020-00668-1"
      },
      "year": 2020,
      "venue": "Social Network Analysis and Mining",
      "citations": 162,
      "metrics": null
    },
    {
      "id": "aragpt2-pre-trained-transformer-for-arabic-language-generati",
      "name": "AraGPT2: Pre-Trained Transformer for Arabic Language Generation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2012.15520"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 161,
      "notes": "Recently, pre-trained transformer-based architectures have proven to be very efficient at language modeling and understanding, given that they are trained on a",
      "metrics": null
    },
    {
      "id": "a-panoramic-survey-of-natural-language-processing-in-the-ara",
      "name": "A panoramic survey of natural language processing in the Arab world",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2011.12631"
      },
      "year": 2020,
      "venue": "Communications of the ACM",
      "citations": 155,
      "notes": "The term natural language refers to any system of symbolic communication (spoken, signed or written) without intentional human planning and design.",
      "metrics": null
    },
    {
      "id": "nadi-2020-the-first-nuanced-arabic-dialect-identification-sh",
      "name": "NADI 2020: The First Nuanced Arabic Dialect Identification Shared Task",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2010.11334"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 146,
      "notes": "We present the results and findings of the First Nuanced Arabic Dialect Identification Shared Task (NADI).",
      "metrics": null
    },
    {
      "id": "deeparslr-a-novel-signer-independent-deep-learning-framework",
      "name": "DeepArSLR: A Novel Signer-Independent Deep Learning Framework for Isolated Arabic Sign Language Gestures Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2020.2990699"
      },
      "year": 2020,
      "venue": "IEEE Access",
      "citations": 127,
      "notes": "Hand gesture recognition has attracted the attention of many researchers due to its wide applications in robotics, games, virtual reality, sign language and",
      "metrics": null
    },
    {
      "id": "a-deep-learning-approach-for-automatic-hate-speech-detection",
      "name": "A Deep Learning Approach for Automatic Hate Speech Detection in the Saudi Twittersphere",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/app10238614"
      },
      "year": 2020,
      "venue": "Applied Sciences",
      "citations": 119,
      "notes": "With the rise of hate speech phenomena in the Twittersphere, significant research efforts have been undertaken in order to provide automatic solutions for",
      "metrics": null
    },
    {
      "id": "arabic-sign-language-recognition-and-generating-arabic-speec",
      "name": "Arabic Sign Language Recognition and Generating Arabic Speech Using Convolutional Neural Network",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2020/3685614"
      },
      "year": 2020,
      "venue": "Wireless Communications and Mobile Computing",
      "citations": 117,
      "notes": "Sign language encompasses the movement of the arms and hands as a means of communication for people with hearing disabilities.",
      "metrics": null
    },
    {
      "id": "qadi-arabic-dialect-identification-in-the-wild",
      "name": "QADI: Arabic Dialect Identification in the Wild",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2005.06557"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 108,
      "notes": "In this paper, we present a method for rapidly constructing a tweet dataset containing a wide range of country-level Arabic dialects —covering 18 different",
      "metrics": null
    },
    {
      "id": "intelligent-detection-of-hate-speech-in-arabic-social-networ",
      "name": "Intelligent detection of hate speech in Arabic social network: A machine learning approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1177/0165551520917651"
      },
      "year": 2020,
      "venue": "Journal of information science",
      "citations": 106,
      "notes": "Nowadays, cyber hate speech is increasingly growing, which forms a serious problem worldwide by threatening the cohesion of civil societies.",
      "metrics": null
    },
    {
      "id": "hate-and-offensive-speech-detection-on-arabic-social-media",
      "name": "Hate and offensive speech detection on Arabic social media",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.osnem.2020.100096"
      },
      "year": 2020,
      "venue": "Online Soc. Networks Media",
      "citations": 102,
      "notes": "We are witnessing an increasing proliferation of hate speech on social media targeting individuals for their protected characteristics.",
      "metrics": null
    },
    {
      "id": "deep-learning-for-arabic-subjective-sentiment-analysis-chall",
      "name": "Deep learning for Arabic subjective sentiment analysis: Challenges and research opportunities",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.asoc.2020.106836"
      },
      "year": 2020,
      "venue": "Applied Soft Computing",
      "citations": 101,
      "notes": "The fields of machine learning and Web technologies have witnessed significant development in the last years.",
      "metrics": null
    },
    {
      "id": "extractive-arabic-text-summarization-using-modified-pagerank",
      "name": "Extractive Arabic Text Summarization Using Modified PageRank Algorithm",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.eij.2019.11.001"
      },
      "year": 2020,
      "venue": "Egyptian Informatics Journal",
      "citations": 98,
      "notes": "This paper proposed an approach for Arabic text summarization.",
      "metrics": null
    },
    {
      "id": "deep-bidirectional-lstm-network-learning-based-sentiment-ana",
      "name": "Deep Bidirectional LSTM Network Learning-Based Sentiment Analysis for Arabic Text",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1515/jisys-2020-0021"
      },
      "year": 2020,
      "venue": "Journal of Intelligent Systems",
      "citations": 96,
      "notes": "Sentiment analysis aims to predict sentiment polarities (positive, negative or neutral) of a given piece of text.",
      "metrics": null
    },
    {
      "id": "arabic-sign-language-recognition-through-deep-neural-network",
      "name": "Arabic Sign Language Recognition through Deep Neural Networks Fine-Tuning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.3991/ijoe.v16i05.13087"
      },
      "year": 2020,
      "venue": "Int. J. Online Biomed. Eng.",
      "citations": 92,
      "notes": "Sign Language is considered the main communication tool for deaf or hearing impaired people.",
      "metrics": null
    },
    {
      "id": "egyptian-arabic-speech-emotion-recognition-using-prosodic-sp",
      "name": "Egyptian Arabic speech emotion recognition using prosodic, spectral and wavelet features",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.specom.2020.04.005"
      },
      "year": 2020,
      "venue": "Speech Communication",
      "citations": 90,
      "notes": "Speech emotion recognition (SER) has recently been receiving increased interest due to the rapid advancements in affective computing and human computer",
      "metrics": null
    },
    {
      "id": "a-sentiment-analysis-approach-to-predict-an-individual-s-awa",
      "name": "A Sentiment Analysis Approach to Predict an Individual’s Awareness of the Precautionary Procedures to Prevent COVID-19 Outbreaks in Saudi Arabia",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.3390/ijerph18010218"
      },
      "year": 2020,
      "venue": "International Journal of Environmental Research and Public Health",
      "citations": 85,
      "notes": "In March 2020, the World Health Organization (WHO) declared the outbreak of Coronavirus disease 2019 (COVID-19) as a pandemic, which affected all countries",
      "metrics": null
    },
    {
      "id": "impact-of-stemming-and-word-embedding-on-deep-learning-based",
      "name": "Impact of Stemming and Word Embedding on Deep Learning-Based Arabic Text Categorization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "preprocessing",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2020.3009217"
      },
      "year": 2020,
      "venue": "IEEE Access",
      "citations": 81,
      "notes": "Document classification is a classical problem in information retrieval, and plays an important role in a variety of applications.",
      "metrics": null
    },
    {
      "id": "comparative-performance-of-machine-learning-and-deep-learnin",
      "name": "Comparative Performance of Machine Learning and Deep Learning Algorithms for Arabic Hate Speech Detection in OSNs",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/978-3-030-44289-7_24"
      },
      "year": 2020,
      "venue": "International Conferences on Artificial Intelligence and Computer Vision",
      "citations": 76,
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis-a-systematic-literature-review",
      "name": "Arabic Sentiment Analysis: A Systematic Literature Review",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2020/7403128"
      },
      "year": 2020,
      "venue": "Applied Computational Intelligence and Soft Computing",
      "citations": 75,
      "notes": "This paper introduces a systematic review of the existing literature relevant to ASA.",
      "metrics": null
    },
    {
      "id": "arabic-text-summarization-using-deep-learning-approach",
      "name": "Arabic text summarization using deep learning approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1186/s40537-020-00386-7"
      },
      "year": 2020,
      "venue": "Journal of Big Data",
      "citations": 70,
      "notes": "Natural language processing has witnessed remarkable progress with the advent of deep learning techniques.",
      "metrics": null
    },
    {
      "id": "a-systematic-review-of-text-classification-research-based-on",
      "name": "A systematic review of text classification research based on deep learning models in Arabic language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.11591/IJECE.V10I6.PP6629-6643"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "citations": 69,
      "notes": "This paper undertakes a systematic review of the latest research in the field of the classification of Arabic texts.",
      "metrics": null
    },
    {
      "id": "hate-speech-detection-using-word-embedding-and-deep-learning",
      "name": "Hate Speech Detection using Word Embedding and Deep Learning in the Arabic Language Context",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "embedding",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.5220/0008954004530460"
      },
      "year": 2020,
      "venue": "International Conference on Pattern Recognition Applications and Methods",
      "citations": 68,
      "notes": ": Hate speech over online social networks is a worldwide problem that leads for diminishing the cohesion of civil societies.",
      "metrics": null
    },
    {
      "id": "detection-of-hate-speech-in-covid-19-related-tweets-in-the-a",
      "name": "Detection of Hate Speech in COVID-19–Related Tweets in the Arab Region: Deep Learning and Topic Modeling Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.2196/22609"
      },
      "year": 2020,
      "venue": "Journal of Medical Internet Research",
      "citations": 64,
      "notes": "The massive scale of social media platforms requires an automatic solution for detecting hate speech.",
      "metrics": null
    },
    {
      "id": "predicting-depression-symptoms-in-an-arabic-psychological-fo",
      "name": "Predicting Depression Symptoms in an Arabic Psychological Forum",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "benchmark"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2020.2981834"
      },
      "year": 2020,
      "venue": "IEEE Access",
      "citations": 64,
      "notes": "Recently, social media platforms have been widely used as a communication tool on social networks.",
      "metrics": null
    },
    {
      "id": "improving-sentiment-analysis-of-arabic-tweets-by-one-way-ano",
      "name": "Improving Sentiment Analysis of Arabic Tweets by One-way ANOVA",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2020.10.023"
      },
      "year": 2020,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 62,
      "notes": "Social media is an indispensable necessity for modern life.",
      "metrics": null
    },
    {
      "id": "recent-advances-in-nlp-the-case-of-arabic-language",
      "name": "Recent Advances in NLP: The Case of Arabic Language",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/978-3-030-34614-0"
      },
      "year": 2020,
      "venue": "Studies in Computational Intelligence",
      "citations": 62,
      "metrics": null
    },
    {
      "id": "nabiha-an-arabic-dialect-chatbot",
      "name": "Nabiha: An Arabic Dialect Chatbot",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.14569/ijacsa.2020.0110357"
      },
      "year": 2020,
      "venue": "International Journal of Advanced Computer Science and Applications",
      "citations": 60,
      "notes": "Nowadays, we are living in the era of technology and innovation that impact various fields, including sciences.",
      "metrics": null
    },
    {
      "id": "habibi-a-multi-dialect-multi-national-arabic-song-lyrics-cor",
      "name": "Habibi - a multi Dialect multi National Arabic Song Lyrics Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.lrec-1.165/"
      },
      "year": 2020,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 52,
      "notes": "This paper introduces Habibi the first Arabic Song Lyrics corpus",
      "metrics": null
    },
    {
      "id": "investigating-the-effects-of-gender-dialect-and-training-siz",
      "name": "Investigating the effects of gender, dialect, and training size on the performance of Arabic speech recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10579-020-09505-5"
      },
      "year": 2020,
      "venue": "Language Resources and Evaluation",
      "citations": 50,
      "notes": "Research in Arabic automatic speech recognition (ASR) is constrained by datasets of limited size, and of highly variable content and quality.",
      "metrics": null
    },
    {
      "id": "daict-a-dialectal-arabic-irony-corpus-extracted-from-twitter",
      "name": "DAICT: A Dialectal Arabic Irony Corpus Extracted from Twitter",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.lrec-1.768/"
      },
      "year": 2020,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 47,
      "notes": "We query Twitter using irony-related hashtags to collect ironic messages, which are then manually annotated by two linguists according to our working",
      "metrics": null
    },
    {
      "id": "empathy-driven-arabic-conversational-chatbot",
      "name": "Empathy-driven Arabic Conversational Chatbot",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.6/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 38,
      "notes": "To address these challenges, we create an Arabic conversational dataset that comprises empathetic responses",
      "metrics": null
    },
    {
      "id": "a-unified-model-for-arabizi-detection-and-transliteration-us",
      "name": "A Unified Model for Arabizi Detection and Transliteration using Sequence-to-Sequence Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.15/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 35,
      "notes": "In this paper, we present the first effort on a unified model for Arabizi detection and transliteration into a code-mixed output with consistent Arabic",
      "metrics": null
    },
    {
      "id": "language-resources-for-maghrebi-arabic-dialects-nlp-a-survey",
      "name": "Language resources for Maghrebi Arabic dialects’ NLP: a survey",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10579-020-09490-9"
      },
      "year": 2020,
      "venue": "Language Resources and Evaluation",
      "citations": 34,
      "metrics": null
    },
    {
      "id": "hate-speech-detection-in-saudi-twittersphere-a-deep-learning",
      "name": "Hate Speech Detection in Saudi Twittersphere: A Deep Learning Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.2/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 29,
      "notes": "This paper, therefore, aims to investigate several neural network models based on Convolutional Neural Network (CNN) and Recurrent Neural Networks (RNN) to",
      "metrics": null
    },
    {
      "id": "is-it-great-or-terrible-preserving-sentiment-in-neural-machi",
      "name": "Is it Great or Terrible? Preserving Sentiment in Neural Machine Translation of Arabic Reviews",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "translation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2010.13814"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 24,
      "notes": "Since the advent of Neural Machine Translation (NMT) approaches there has been a tremendous improvement in the quality of automatic translation.",
      "metrics": null
    },
    {
      "id": "morphological-analysis-and-disambiguation-for-gulf-arabic-th",
      "name": "Morphological Analysis and Disambiguation for Gulf Arabic: The Interplay between Resources and Methods",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.lrec-1.480/"
      },
      "year": 2020,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 24,
      "notes": "In this paper we present the first full morphological analysis and disambiguation system for Gulf Arabic",
      "metrics": null
    },
    {
      "id": "weighted-combination-of-bert-and-n-gram-features-for-nuanced",
      "name": "Weighted combination of BERT and N-GRAM features for Nuanced Arabic Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.27/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 24,
      "notes": "In this paper, we investigate the Arabic dialect identification task, from two perspectives: country-level dialect identification from 21 Arab countries, and",
      "metrics": null
    },
    {
      "id": "a-spelling-correction-corpus-for-multiple-arabic-dialects",
      "name": "A Spelling Correction Corpus for Multiple Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.lrec-1.508/"
      },
      "year": 2020,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 23,
      "notes": "In this paper, we present the MADAR CODA Corpus, a collection of 10,000 sentences from five Arabic city dialects (Beirut, Cairo, Doha, Rabat, and Tunis)",
      "metrics": null
    },
    {
      "id": "on-the-importance-of-tokenization-in-arabic-embedding-models",
      "name": "On the Importance of Tokenization in Arabic Embedding Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "preprocessing"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.11/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 23,
      "notes": "In this work, we propose two embedding strategies that modify the tokenization phase of traditional word embedding models (Word2Vec) and contextual word",
      "metrics": null
    },
    {
      "id": "effects-of-dialectal-code-switching-on-speech-modules-a-stud",
      "name": "Effects of Dialectal Code-Switching on Speech Modules: A Study Using Egyptian Arabic Broadcast Speech",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.21437/interspeech.2020-2271"
      },
      "year": 2020,
      "venue": "Interspeech",
      "citations": 22,
      "notes": "The intra-utterance code-switching (CS) is deﬁned as the alternation between two or more languages within the same ut-terance.",
      "metrics": null
    },
    {
      "id": "a-semi-supervised-bert-approach-for-arabic-named-entity-reco",
      "name": "A Semi-Supervised BERT Approach for Arabic Named Entity Recognition",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.5/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 21,
      "notes": "This paper proposes a semi-supervised learning approach to train a BERT-based NER model using labeled and semi-labeled datasets",
      "metrics": null
    },
    {
      "id": "deep-diacritization-efficient-hierarchical-recurrence-for-im",
      "name": "Deep Diacritization: Efficient Hierarchical Recurrence for Improved Arabic Diacritization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2011.00538"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 20,
      "notes": "We propose a novel architecture for labelling character sequences that achieves state-of-the-art results on the Tashkeela Arabic diacritization benchmark.",
      "metrics": null
    },
    {
      "id": "identifying-sentiments-in-algerian-code-switched-user-genera",
      "name": "Identifying Sentiments in Algerian Code-switched User-generated Comments",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.lrec-1.328/"
      },
      "year": 2020,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 18,
      "notes": "We present in this paper our work on Algerian language, an under-resourced North African colloquial Arabic variety, for which we built a comparably large",
      "metrics": null
    },
    {
      "id": "arabic-dialects-identification-for-all-arabic-countries",
      "name": "Arabic Dialects Identification for All Arabic countries",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.32/"
      },
      "year": 2020,
      "venue": "Workshop on Arabic Natural Language Processing",
      "citations": 16,
      "notes": "In this paper, several techniques with multiple algorithms are applied for Arabic dialects identification starting from removing noise till classification task",
      "metrics": null
    },
    {
      "id": "a-benchmark-arabic-dataset-for-commonsense-explanation",
      "name": "A Benchmark Arabic Dataset for Commonsense Explanation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2012.10251"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "In this paper, we present a benchmark Arabic dataset for commonsense explanation.",
      "metrics": null
    },
    {
      "id": "an-arabic-tweets-sentiment-analysis-dataset-atsad-using-distant-superv",
      "name": "An Arabic Tweets Sentiment Analysis Dataset (ATSAD) using Distant Supervision and Self Training",
      "type": "paper",
      "country": "INTL",
      "org": "University o",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.1/"
      },
      "year": 2020,
      "venue": "OSACT 2020",
      "notes": "We present an Arabic Sentiment Analysis Corpus collected from Twitter, which contains 36K tweets labelled into positive and negative.",
      "metrics": null
    },
    {
      "id": "an-empirical-study-of-pre-trained-transformers-for-arabic-information",
      "name": "An Empirical Study of Pre-trained Transformers for Arabic Information Extraction",
      "type": "paper",
      "country": "INTL",
      "org": "lanwuwei",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "pretraining"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2004.14519",
        "github": "https://github.com/lanwuwei/GigaBERT"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "Multilingual pre-trained Transformers, such as mBERT (Devlin et al., 2019) and XLM-RoBERTa.",
      "metrics": null
    },
    {
      "id": "araacom-arabic-algerian-corpus-for-opinion-mining",
      "name": "ARAACOM: ARAbic Algerian Corpus for Opinion Mining",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "vision"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2001.08010"
      },
      "dialects": [
        "magh"
      ],
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "In this paper, we propose our approach, for opinion mining in Arabic Algerian news paper.",
      "metrics": null
    },
    {
      "id": "arabert-transformer-based-model-for-arabic-language",
      "name": "AraBERT: Transformer-based Model for Arabic Language Understanding",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2020,
      "venue": "OSACT 2020",
      "tasks": [
        "pretraining",
        "encoder"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.2/",
        "github": "https://github.com/aub-mind/arabert"
      },
      "notes": "BERT-style Arabic encoder pretrained on a large MSA corpus.",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification-an-arabic-bert-model-with-data-augmentat",
      "name": "Arabic dialect identification: An Arabic-BERT model with data augmentation and ensembling strategy",
      "type": "paper",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.28/"
      },
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "ArabicProcessors team Arabic-BERT system with data augmentation and ensembling for NADI 2020 country- and province-level dialect identification.",
      "metrics": null
    },
    {
      "id": "araweat-multidimensional-analysis-of-biases-in-arabic-word-embeddings",
      "name": "AraWEAT: Multidimensional Analysis of Biases in Arabic Word Embeddings",
      "type": "paper",
      "country": "INTL",
      "org": "Data and Web Science Research Group",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.17/"
      },
      "dialects": [
        "egy",
        "msa"
      ],
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "We conduct an extensive analysis of biases in Arabic word embeddings by applying a range of recently introduced bias tests on a variety of embedding spaces.",
      "metrics": null
    },
    {
      "id": "arcov-19-the-first-arabic-covid-19-twitter-dataset-with-propagation-ne",
      "name": "ArCOV-19: The First Arabic COVID-19 Twitter Dataset with Propagation Networks",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2004.05861"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "We present ArCOV-19, an Arabic COVID-19 Twitter dataset that spans one year, covering the period from 27th of January 2020 till 31st of January 2021.",
      "metrics": null
    },
    {
      "id": "arcov19-rumors-arabic-covid-19-twitter-dataset-for-misinformation-dete",
      "name": "ArCOV19-Rumors: Arabic COVID-19 Twitter Dataset for Misinformation Detection",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2010.08768"
      },
      "dialects": [
        "lev"
      ],
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "In this paper we introduce ArCOV19-Rumors, an Arabic COVID-19 Twitter dataset for misinformation detection composed of tweets containing claims.",
      "metrics": null
    },
    {
      "id": "asad-a-twitter-based-benchmark-arabic-sentiment-analysis-dataset",
      "name": "ASAD: A Twitter-based Benchmark Arabic Sentiment Analysis Dataset",
      "type": "paper",
      "country": "SA",
      "org": "KAUST",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "benchmark"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2011.00578"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "ASAD: large annotated Twitter benchmark for Arabic sentiment analysis, launched as the KAUST-sponsored competition dataset.",
      "metrics": null
    },
    {
      "id": "automatic-arabic-dialect-identification-systems-for-written-texts-a-su",
      "name": "Automatic Arabic Dialect Identification Systems for Written Texts: A Survey",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "translation",
        "dialect-id",
        "evaluation",
        "survey",
        "speech"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2009.12622"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "In this paper, we present a comprehensive survey of Arabic dialect identification research in written texts.",
      "metrics": null
    },
    {
      "id": "contextual-embeddings-for-arabic-english-code-switched-data",
      "name": "Contextual Embeddings for Arabic-English Code-Switched Data",
      "type": "paper",
      "country": "EG",
      "org": "German University in Cairo",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "code-switching",
        "embedding"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.20/"
      },
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "We propose an open source trained bilingual contextual word embedding models of FLAIR, BERT, and ELECTRA.",
      "metrics": null
    },
    {
      "id": "dialex-a-benchmark-for-evaluating-multidialectal-arabic-word-embedding",
      "name": "DiaLex: A Benchmark for Evaluating Multidialectal Arabic Word Embeddings",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2011.10970"
      },
      "dialects": [
        "egy",
        "lev",
        "magh",
        "mixed"
      ],
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "We describe DiaLex, a benchmark for intrinsic evaluation of dialectal Arabic word embedding.",
      "metrics": null
    },
    {
      "id": "is-this-sentence-valid-an-arabic-dataset-for-commonsense-validation",
      "name": "Is this sentence valid? An Arabic Dataset for Commonsense Validation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2008.10873"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "We present a benchmark Arabic dataset for commonsense understanding and validation as well as a baseline research and models trained using the same dataset.",
      "metrics": null
    },
    {
      "id": "large-arabic-twitter-dataset-on-covid-19",
      "name": "Large Arabic Twitter Dataset on COVID-19",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2004.04315"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "In this work, we describe the first Arabic tweets dataset on COVID-19 that we have been collecting since January 1st, 2020.",
      "metrics": null
    },
    {
      "id": "machine-generation-and-detection-of-arabic-manipulated-and-fake-news",
      "name": "Machine Generation and Detection of Arabic Manipulated and Fake News",
      "type": "paper",
      "country": "INTL",
      "org": "Natural Language Processing Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.7/"
      },
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "We present a novel method for automatically generating Arabic manipulated (and potentially fake) news stories.",
      "metrics": null
    },
    {
      "id": "manorm-a-normalization-dictionary-for-moroccan-arabic-dialect-written",
      "name": "MANorm: A Normalization Dictionary for Moroccan Arabic Dialect Written in Latin Script",
      "type": "paper",
      "country": "MA",
      "org": "Mohammed V University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "lexicon"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.14/"
      },
      "dialects": [
        "magh"
      ],
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "This text, however, does not follow the standard rules of writing.",
      "metrics": null
    },
    {
      "id": "multi-dialect-arabic-bert-for-country-level-dialect-identification",
      "name": "Multi-dialect Arabic BERT for Country-level Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.10/"
      },
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "We present the experiments conducted, and the models developed by our competing team, Mawdoo3 AI, along the way to achieving our winning solution to subtask.",
      "metrics": null
    },
    {
      "id": "offensive-language-detection-in-arabic-using-ulmfit",
      "name": "Offensive language detection in Arabic using ULMFiT",
      "type": "paper",
      "country": "INTL",
      "org": "Rutgers University - Computer Science",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.13/"
      },
      "year": 2020,
      "venue": "OSACT 2020",
      "notes": "We approach the shared task OffenseEval 2020 by Mubarak et al.",
      "metrics": null
    },
    {
      "id": "osact4-shared-task-on-offensive-language-detection-intensive-preproces",
      "name": "OSACT4 Shared Task on Offensive Language Detection: Intensive Preprocessing-Based Approach",
      "type": "paper",
      "country": "KW",
      "org": "Kuwait University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language",
        "classification",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.8/"
      },
      "year": 2020,
      "venue": "OSACT 2020",
      "notes": "We apply intensive preprocessing techniques to the dataset before processing it further and feeding it into the classification model.",
      "metrics": null
    },
    {
      "id": "parallel-resources-for-tunisian-arabic-dialect-translation",
      "name": "Parallel resources for Tunisian Arabic Dialect Translation",
      "type": "paper",
      "country": "TN",
      "org": "University of Sfax",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.wanlp-1.18/"
      },
      "dialects": [
        "magh"
      ],
      "year": 2020,
      "venue": "WANLP 2020",
      "notes": "We present a data augmentation technique to create a parallel corpus for Tunisian Arabic dialect written in social media and standard Arabic in order.",
      "metrics": null
    },
    {
      "id": "understanding-and-detecting-dangerous-speech-in-social-media",
      "name": "Understanding and Detecting Dangerous Speech in Social Media",
      "type": "paper",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "offensive-language"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.6/"
      },
      "year": 2020,
      "venue": "OSACT 2020",
      "notes": "We report our efforts to build a labeled dataset for dangerous speech.",
      "metrics": null
    },
    {
      "id": "woli-at-semeval-2020-task-12-arabic-offensive-language-identification",
      "name": "WOLI at SemEval-2020 Task 12: Arabic Offensive Language Identification on Different Twitter Datasets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2009.05456"
      },
      "year": 2020,
      "venue": "arXiv 2020",
      "notes": "This paper presents the results and the main findings of SemEval-2020, Task 12 OffensEval Sub-task A Zampieri et al.",
      "metrics": null
    },
    {
      "id": "arabic-natural-language-processing-an-overview",
      "name": "Arabic natural language processing: An overview",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1903.02784"
      },
      "year": 2019,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 234,
      "notes": "Arabic is recognised as the 4th most used language of the Internet.",
      "metrics": null
    },
    {
      "id": "a-comprehensive-survey-of-arabic-sentiment-analysis",
      "name": "A comprehensive survey of arabic sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.IPM.2018.07.006"
      },
      "year": 2019,
      "venue": "Information Processing & Management",
      "citations": 201,
      "notes": "Sentiment analysis (SA) is a continuing field of research that lies at the intersection of many fields such as data mining, natural language processing and",
      "metrics": null
    },
    {
      "id": "neural-arabic-question-answering",
      "name": "Neural Arabic Question Answering",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1906.05394"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 177,
      "notes": "This paper tackles the problem of open domain factual Arabic question answering (QA) using Wikipedia as our knowledge source.",
      "metrics": null
    },
    {
      "id": "enhancing-aspect-based-sentiment-analysis-of-arabic-hotels-r",
      "name": "Enhancing Aspect-Based Sentiment Analysis of Arabic Hotels' reviews using morphological, syntactic and semantic features",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "morphology"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.IPM.2018.01.006"
      },
      "year": 2019,
      "venue": "Information Processing & Management",
      "citations": 169,
      "notes": "This research presents an enhanced approach for Aspect-Based Sentiment Analysis (ABSA) of Hotels’ Arabic reviews using supervised machine learning.",
      "metrics": null
    },
    {
      "id": "feature-selection-using-binary-grey-wolf-optimizer-with-elit",
      "name": "Feature selection using binary grey wolf optimizer with elite-based crossover for Arabic text classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s00521-019-04368-6"
      },
      "year": 2019,
      "venue": "Neural computing & applications (Print)",
      "citations": 167,
      "metrics": null
    },
    {
      "id": "the-madar-shared-task-on-arabic-fine-grained-dialect-identif",
      "name": "The MADAR Shared Task on Arabic Fine-Grained Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4622/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 150,
      "notes": "In this paper, we present the results and findings of the MADAR Shared Task on Arabic Fine-Grained Dialect Identification.",
      "metrics": null
    },
    {
      "id": "mazajak-an-online-arabic-sentiment-analyser",
      "name": "Mazajak: An Online Arabic Sentiment Analyser",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4621/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 141,
      "notes": "In this paper, we present “Mazajak”, an online system for Arabic SA.",
      "metrics": null
    },
    {
      "id": "sanad-single-label-arabic-news-articles-dataset-for-automati",
      "name": "SANAD: Single-label Arabic News Articles Dataset for automatic text categorization",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.dib.2019.104076"
      },
      "year": 2019,
      "venue": "Data in Brief",
      "citations": 106,
      "notes": "Text Classification is one of the most popular Natural Language Processing (NLP) tasks.",
      "metrics": null
    },
    {
      "id": "deep-learning-approaches-for-arabic-sentiment-analysis",
      "name": "Deep learning approaches for Arabic sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13278-019-0596-4"
      },
      "year": 2019,
      "venue": "Social Network Analysis and Mining",
      "citations": 100,
      "metrics": null
    },
    {
      "id": "detecting-arabic-depressed-users-from-twitter-data",
      "name": "Detecting Arabic Depressed Users from Twitter Data",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2019.12.107"
      },
      "year": 2019,
      "venue": "Procedia Computer Science",
      "citations": 94,
      "notes": "Depression is one of the most common health issues impacting the world.",
      "metrics": null
    },
    {
      "id": "idat-at-fire2019-overview-of-the-track-on-irony-detection-in",
      "name": "IDAT at FIRE2019: Overview of the Track on Irony Detection in Arabic Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1145/3368567.3368585"
      },
      "year": 2019,
      "venue": "Fire",
      "citations": 92,
      "notes": "This overview paper describes the first shared task on irony detection for the Arabic language.",
      "metrics": null
    },
    {
      "id": "a-survey-of-opinion-mining-in-arabic",
      "name": "A Survey of Opinion Mining in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey",
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1145/3295662"
      },
      "year": 2019,
      "venue": "ACM Trans. Asian Low Resour. Lang. Inf. Process.",
      "citations": 86,
      "notes": "Opinion-mining or sentiment analysis continues to gain interest in industry and academics.",
      "metrics": null
    },
    {
      "id": "detecting-offensive-language-on-arabic-social-media-using-de",
      "name": "Detecting Offensive Language on Arabic Social Media Using Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/SNAMS.2019.8931839"
      },
      "year": 2019,
      "venue": "International Conference on Social Networks Analysis, Management and Security",
      "citations": 86,
      "notes": "Offensive content on social media such as verbal attacks, demeaning comments or hate speech has many negative effects on its users.",
      "metrics": null
    },
    {
      "id": "arabic-natural-language-processing-and-machine-learning-base",
      "name": "Arabic Natural Language Processing and Machine Learning-Based Systems",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2018.2890076"
      },
      "year": 2019,
      "venue": "IEEE Access",
      "citations": 84,
      "notes": "Arabic natural language processing (ANLP) consists of developing techniques and tools that can utilize and analyze the Arabic language in both written and",
      "metrics": null
    },
    {
      "id": "hilatsa-a-hybrid-incremental-learning-approach-for-arabic-tw",
      "name": "HILATSA: A hybrid Incremental learning approach for Arabic tweets sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.EIJ.2019.03.002"
      },
      "year": 2019,
      "venue": "Egyptian Informatics Journal",
      "citations": 80,
      "notes": "A huge amount of data is generated since the evolution in technology and the tremendous growth of social networks.",
      "metrics": null
    },
    {
      "id": "the-mgb-5-challenge-recognition-and-dialect-identification-o",
      "name": "The MGB-5 Challenge: Recognition and Dialect Identification of Dialectal Arabic Speech",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ASRU46091.2019.9003960"
      },
      "year": 2019,
      "venue": "Automatic Speech Recognition & Understanding",
      "citations": 80,
      "notes": "This paper describes the fifth edition of the Multi-Genre Broadcast Challenge (MGB-5), an evaluation focused on Arabic speech recognition and dialect",
      "metrics": null
    },
    {
      "id": "arsentd-lev-a-multi-topic-corpus-for-target-based-sentiment",
      "name": "ArSentD-LEV: A Multi-Topic Corpus for Target-based Sentiment Analysis in Arabic Levantine Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1906.01830"
      },
      "year": 2019,
      "venue": "arXiv.org",
      "citations": 78,
      "notes": "Sentiment analysis is a highly subjective and challenging task.",
      "metrics": null
    },
    {
      "id": "a-study-of-the-effects-of-stemming-strategies-on-arabic-docu",
      "name": "A Study of the Effects of Stemming Strategies on Arabic Document Classification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2019.2903331"
      },
      "year": 2019,
      "venue": "IEEE Access",
      "citations": 70,
      "notes": "Stemming is one of the most effective techniques, which has been adopted in many applications, such as machine learning, machine translation, document",
      "metrics": null
    },
    {
      "id": "surface-and-deep-features-ensemble-for-sentiment-analysis-of",
      "name": "Surface and Deep Features Ensemble for Sentiment Analysis of Arabic Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2019.2924314"
      },
      "year": 2019,
      "venue": "IEEE Access",
      "citations": 70,
      "notes": "Sentiment analysis (SA) of Arabic tweets is a complex task due to the rich morphology of the Arabic language and the informal nature of language on Twitter.",
      "metrics": null
    },
    {
      "id": "an-efficient-single-document-arabic-text-summarization-using",
      "name": "An efficient single document Arabic text summarization using a combination of statistical and semantic features",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.jksuci.2019.03.010"
      },
      "year": 2019,
      "venue": "Journal of King Saud University: Computer and Information Sciences",
      "citations": 67,
      "notes": "In this paper, we propose an automatic, generic, and extractive Arabic single",
      "metrics": null
    },
    {
      "id": "hulmona-the-universal-language-model-in-arabic",
      "name": "hULMonA: The Universal Language Model in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4608/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 64,
      "notes": "Arabic is a complex language with limited resources which makes it challenging to produce accurate text classification tasks such as sentiment analysis.",
      "metrics": null
    },
    {
      "id": "highly-effective-arabic-diacritization-using-sequence-to-seq",
      "name": "Highly Effective Arabic Diacritization using Sequence to Sequence Modeling",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://aclanthology.org/N19-1248/"
      },
      "year": 2019,
      "venue": "North American Chapter of the Association for Computational Linguistics",
      "citations": 51,
      "notes": "Arabic text is typically written without short vowels (or diacritics).",
      "metrics": null
    },
    {
      "id": "arhnet-leveraging-community-interaction-for-detection-of-rel",
      "name": "ARHNet - Leveraging Community Interaction for Detection of Religious Hate Speech in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://aclanthology.org/P19-2038/"
      },
      "year": 2019,
      "venue": "Annual Meeting of the Association for Computational Linguistics",
      "citations": 35,
      "notes": "The rapid widespread of social media has lead to some undesirable consequences like the rapid increase of hateful content and offensive language.",
      "metrics": null
    },
    {
      "id": "no-army-no-navy-bert-semi-supervised-learning-of-arabic-dial",
      "name": "No Army, No Navy: BERT Semi-Supervised Learning of Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4637/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 34,
      "notes": "We present our deep leaning system submitted to MADAR shared task 2 focused on twitter user dialect identification.",
      "metrics": null
    },
    {
      "id": "adida-automatic-dialect-identification-for-arabic",
      "name": "ADIDA: Automatic Dialect Identification for Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/N19-4002/"
      },
      "year": 2019,
      "venue": "North American Chapter of the Association for Computational Linguistics",
      "citations": 26,
      "notes": "This demo paper describes ADIDA, a web-based system for automatic dialect identification for Arabic text.",
      "metrics": null
    },
    {
      "id": "morphologically-annotated-corpora-for-seven-arabic-dialects",
      "name": "Morphologically Annotated Corpora for Seven Arabic Dialects: Taizi, Sanaani, Najdi, Jordanian, Syrian, Iraqi and Moroccan",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4615/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 25,
      "notes": "We present a collection of morphologically annotated corpora for seven Arabic dialects: Taizi Yemeni, Sanaani Yemeni, Najdi, Jordanian, Syrian, Iraqi and",
      "metrics": null
    },
    {
      "id": "arabic-named-entity-recognition-what-works-and-what-s-next",
      "name": "Arabic Named Entity Recognition: What Works and What’s Next",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4607/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 23,
      "notes": "This paper presents the winning solution to the Arabic Named Entity Recognition challenge run by Topcoder.com.",
      "metrics": null
    },
    {
      "id": "arbengvec-arabic-english-cross-lingual-word-embedding-model",
      "name": "ArbEngVec : Arabic-English Cross-Lingual Word Embedding Model",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4605/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 19,
      "notes": "Word Embeddings (WE) are getting increasingly popular and widely applied in many Natural Language Processing (NLP) applications due to their effectiveness in",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification-for-travel-and-twitter-text",
      "name": "Arabic Dialect Identification for Travel and Twitter Text",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4628/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 17,
      "notes": "This paper presents the results of the experiments done as a part of MADAR Shared Task in WANLP 2019 on Arabic Fine-Grained Dialect Identification.",
      "metrics": null
    },
    {
      "id": "arabic-rule-based-named-entity-recognition-system-using-gate",
      "name": "Arabic Rule-based Named Entity Recognition System Using GATE",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://www.semanticscholar.org/paper/791020059cb28d5582104fa4e90478ba88c03f5f"
      },
      "year": 2019,
      "venue": "IAPR International Conference on Machine Learning and Data Mining in Pattern Recognition",
      "citations": 17,
      "metrics": null
    },
    {
      "id": "arabic-tweet-act-speech-act-recognition-for-arabic-asynchron",
      "name": "Arabic Tweet-Act: Speech Act Recognition for Arabic Asynchronous Conversations",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4620/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 16,
      "notes": "In this paper, we proposed speech act classification for asynchronous conversations on Twitter using multiple machine learning methods including SVM and deep",
      "metrics": null
    },
    {
      "id": "mawdoo3-ai-at-madar-shared-task-arabic-fine-grained-dialect",
      "name": "Mawdoo3 AI at MADAR Shared Task: Arabic Fine-Grained Dialect Identification with Ensemble Learning",
      "type": "paper",
      "country": "JO",
      "org": "Mawdoo3",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4630/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 16,
      "notes": "In this paper we discuss several models we used to classify 25 city-level Arabic dialects in addition to Modern Standard Arabic (MSA) as part of MADAR shared",
      "metrics": null
    },
    {
      "id": "st-madar-2019-shared-task-arabic-fine-grained-dialect-identi",
      "name": "ST MADAR 2019 Shared Task: Arabic Fine-Grained Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4635/"
      },
      "year": 2019,
      "venue": "WANLP@ACL 2019",
      "citations": 16,
      "notes": "This paper describes the solution that we propose on MADAR 2019 Arabic Fine-Grained Dialect Identification task.",
      "metrics": null
    },
    {
      "id": "the-madar-arabic-dialect-corpus-and-lexicon",
      "name": "The MADAR Arabic Dialect Corpus and Lexicon",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1535/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 324,
      "notes": "In this paper, we present two resources that were created as part of the Multi Arabic Dialect Applications and Resources (MADAR) project.",
      "metrics": null
    },
    {
      "id": "are-they-our-brothers-analysis-and-detection-of-religious-ha",
      "name": "Are They Our Brothers? Analysis and Detection of Religious Hate Speech in the Arabic Twittersphere",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://www.semanticscholar.org/paper/70ab0898e5325bf33330e96668a5771ef35d32d3"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 251,
      "metrics": null
    },
    {
      "id": "improved-whale-optimization-algorithm-for-feature-selection",
      "name": "Improved whale optimization algorithm for feature selection in Arabic sentiment analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10489-018-1334-8"
      },
      "year": 2018,
      "venue": "Applied intelligence (Boston)",
      "citations": 251,
      "metrics": null
    },
    {
      "id": "sentiment-analysis-of-arabic-tweets-using-deep-learning",
      "name": "Sentiment Analysis of Arabic Tweets using Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.PROCS.2018.10.466"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 245,
      "notes": "Sentiment analysis is the computational study of people’s opinions, attitudes and emotions toward entities, individuals, issues, events or topics.",
      "metrics": null
    },
    {
      "id": "using-long-short-term-memory-deep-neural-networks-for-aspect",
      "name": "Using long short-term memory deep neural networks for aspect-based sentiment analysis of Arabic reviews",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s13042-018-0799-4"
      },
      "year": 2018,
      "venue": "International Journal of Machine Learning and Cybernetics",
      "citations": 245,
      "metrics": null
    },
    {
      "id": "a-combined-cnn-and-lstm-model-for-arabic-sentiment-analysis",
      "name": "A Combined CNN and LSTM Model for Arabic Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1807.02911"
      },
      "year": 2018,
      "venue": "International Cross-Domain Conference on Machine Learning and Knowledge Extraction",
      "citations": 170,
      "notes": "Deep neural networks have shown good data modelling capabilities when dealing with challenging and large datasets from a wide range of application areas.",
      "metrics": null
    },
    {
      "id": "hotel-arabic-reviews-dataset-construction-for-sentiment-anal",
      "name": "Hotel Arabic-Reviews Dataset Construction for Sentiment Analysis Applications",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/978-3-319-67056-0_3"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 161,
      "metrics": null
    },
    {
      "id": "fine-grained-arabic-dialect-identification",
      "name": "Fine-Grained Arabic Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/C18-1113/"
      },
      "year": 2018,
      "venue": "International Conference on Computational Linguistics",
      "citations": 139,
      "notes": "This paper presents the first results on a fine-grained dialect classification task covering 25 specific cities from across the Arab World, in addition to",
      "metrics": null
    },
    {
      "id": "towards-accurate-detection-of-offensive-language-in-online-c",
      "name": "Towards Accurate Detection of Offensive Language in Online Communication in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/j.procs.2018.10.491"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 117,
      "notes": "We present the results of predictive modelling for the detection of anti-social behaviour in online communication in Arabic, such as comments which contain",
      "metrics": null
    },
    {
      "id": "automatic-arabic-dialect-classification-using-deep-learning",
      "name": "Automatic Arabic Dialect Classification Using Deep Learning Models",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.PROCS.2018.10.489"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 112,
      "notes": "Recently, the vast use of social media and the high availability of internet access have produced a considerably different textual data from the formal and",
      "metrics": null
    },
    {
      "id": "dataset-construction-for-the-detection-of-anti-social-behavi",
      "name": "Dataset Construction for the Detection of Anti-Social Behaviour in Online Communication in Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.PROCS.2018.10.473"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 112,
      "notes": "Warning: this paper contains a range of words which may cause offence.",
      "metrics": null
    },
    {
      "id": "arsas-an-arabic-speech-act-and-sentiment-corpus-of-tweets",
      "name": "ArSAS : An Arabic Speech-Act and Sentiment Corpus of Tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://www.semanticscholar.org/paper/d32d3bb226f1738f72c415c6b03b5ad66ff604a4"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 100,
      "metrics": null
    },
    {
      "id": "unified-guidelines-and-resources-for-arabic-dialect-orthogra",
      "name": "Unified Guidelines and Resources for Arabic Dialect Orthography",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1574/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 96,
      "metrics": null
    },
    {
      "id": "sedat-sentiment-and-emotion-detection-in-arabic-text-using-c",
      "name": "SEDAT: Sentiment and Emotion Detection in Arabic Text Using CNN-LSTM Deep Learning",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ICMLA.2018.00134"
      },
      "year": 2018,
      "venue": "International Conference on Machine Learning and Applications",
      "citations": 94,
      "notes": "Social media is growing as a communication medium where people can express online their feelings and opinions on a variety of topics in ways they rarely do in",
      "metrics": null
    },
    {
      "id": "a-survey-of-arabic-text-mining",
      "name": "A Survey of Arabic Text Mining",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/978-3-319-67056-0_20"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 89,
      "metrics": null
    },
    {
      "id": "improving-sentiment-analysis-in-arabic-using-word-representa",
      "name": "Improving Sentiment Analysis in Arabic Using Word Representation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1803.00124"
      },
      "year": 2018,
      "venue": "International Workshop on Arabic Script Analysis and Recognition",
      "citations": 88,
      "notes": "The complexities of Arabic language in morphology, orthography and dialects makes sentiment analysis for Arabic more challenging.",
      "metrics": null
    },
    {
      "id": "shami-a-corpus-of-levantine-arabic-dialects",
      "name": "Shami: A Corpus of Levantine Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1576/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 87,
      "notes": "Modern Standard Arabic (MSA) is the official language used in education and media across the Arab world both in writing and formal speech.",
      "metrics": null
    },
    {
      "id": "revisiting-k-means-and-topic-modeling-a-comparison-study-to",
      "name": "Revisiting K-Means and Topic Modeling, a Comparison Study to Cluster Arabic Documents",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ACCESS.2018.2852648"
      },
      "year": 2018,
      "venue": "IEEE Access",
      "citations": 86,
      "notes": "This paper uses a combined method to cluster Arabic text documents.",
      "metrics": null
    },
    {
      "id": "you-tweet-what-you-speak-a-city-level-dataset-of-arabic-dial",
      "name": "You Tweet What You Speak: A City-Level Dataset of Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1577/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 82,
      "notes": "Arabic has a wide range of varieties or dialects.",
      "metrics": null
    },
    {
      "id": "a-lexical-distance-study-of-arabic-dialects",
      "name": "A Lexical Distance Study of Arabic Dialects",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.PROCS.2018.10.456"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 80,
      "notes": "Diglossia is a very common phenomenon in Arabic-speaking communities, where the spoken language is different from both Classical Arabic (CA) and Modern Standard",
      "metrics": null
    },
    {
      "id": "a-hybrid-approach-for-arabic-text-summarization-using-domain",
      "name": "A Hybrid Approach for Arabic Text Summarization Using Domain Knowledge and Genetic Algorithms",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s12559-018-9547-z"
      },
      "year": 2018,
      "venue": "Cognitive Computation",
      "citations": 70,
      "metrics": null
    },
    {
      "id": "deep-models-for-arabic-dialect-identification-on-benchmarked",
      "name": "Deep Models for Arabic Dialect Identification on Benchmarked Data",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "paper": "https://aclanthology.org/W18-3930/"
      },
      "year": 2018,
      "venue": "VarDial@COLING 2018",
      "citations": 69,
      "notes": "We treat these two limitations:We (1) benchmark the data, and (2) empirically test6different deep learning methods on thetask, comparing peformance to several",
      "metrics": null
    },
    {
      "id": "automatic-arabic-image-captioning-using-rnn-lstm-based-langu",
      "name": "Automatic Arabic Image Captioning using RNN-LSTM-Based Language Model and CNN",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.14569/IJACSA.2018.090610"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 64,
      "notes": "The automatic generation of correct syntaxial and semantical image captions is an essential problem in Artificial Intelligence.",
      "metrics": null
    },
    {
      "id": "a-neural-machine-translation-model-for-arabic-dialects-that",
      "name": "A Neural Machine Translation Model for Arabic Dialects That Utilises Multitask Learning (MTL)",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "paper": "https://doi.org/10.1155/2018/7534712"
      },
      "year": 2018,
      "venue": "Computational Intelligence and Neuroscience",
      "citations": 63,
      "notes": "In this research article, we study the problem of employing a neural machine translation model to translate Arabic dialects to modern standard Arabic.",
      "metrics": null
    },
    {
      "id": "an-abstractive-arabic-text-summarizer-with-user-controlled-g",
      "name": "An abstractive Arabic text summarizer with user controlled granularity",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.IPM.2018.06.002"
      },
      "year": 2018,
      "venue": "Information Processing & Management",
      "citations": 63,
      "notes": "In the former we retain the more important sentences more or less in their original structure, while the latter requires a fusion of multiple",
      "metrics": null
    },
    {
      "id": "image-based-arabic-sign-language-recognition-system",
      "name": "Image based Arabic Sign Language Recognition System",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://doi.org/10.14569/IJACSA.2018.090327"
      },
      "year": 2018,
      "venue": "arXiv 2018",
      "citations": 62,
      "notes": "Through history, humans have used many ways of communication such as gesturing, sounds, drawing, writing, and speaking.",
      "metrics": null
    },
    {
      "id": "sentialg-automated-corpus-annotation-for-algerian-sentiment",
      "name": "SentiALG: Automated Corpus Annotation for Algerian Sentiment Analysis",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1808.05079"
      },
      "year": 2018,
      "venue": "International Conference on Advances in Brain Inspired Cognitive Systems",
      "citations": 61,
      "notes": "To sort a text into two classes, the very first thing we need is a good annotation guideline, establishing what is required to qualify for each class.",
      "metrics": null
    },
    {
      "id": "sentiment-lexicon-for-sentiment-analysis-of-saudi-dialect-tw",
      "name": "Sentiment lexicon for sentiment analysis of Saudi dialect tweets",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dataset"
      ],
      "links": {
        "paper": "https://doi.org/10.1016/J.PROCS.2018.10.494"
      },
      "year": 2018,
      "venue": "International Conference on Arabic Computational Linguistics",
      "citations": 61,
      "notes": "Twitter is one of the most widely used social media platforms in Saudi Arabia and is a rich source for mining the public’s attitude towards political, social",
      "metrics": null
    },
    {
      "id": "a-morphologically-annotated-corpus-of-emirati-arabic",
      "name": "A Morphologically Annotated Corpus of Emirati Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1607/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 55,
      "metrics": null
    },
    {
      "id": "comparison-of-pre-trained-word-vectors-for-arabic-text-class",
      "name": "Comparison of Pre-Trained Word Vectors for Arabic Text Classification Using Deep Learning Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "classification",
        "pretraining"
      ],
      "links": {
        "paper": "https://doi.org/10.1109/ICMLA.2018.00239"
      },
      "year": 2018,
      "venue": "International Conference on Machine Learning and Applications",
      "citations": 49,
      "notes": "In this paper, we describe an Arabic text sentiment analysis approach using a Deep Neural network,",
      "metrics": null
    },
    {
      "id": "multi-dialect-arabic-pos-tagging-a-crf-approach",
      "name": "Multi-Dialect Arabic POS Tagging: A CRF Approach",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1015/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 45,
      "notes": "This paper introduces a new dataset of POS-tagged Arabic tweets in four major dialects along with tagging guidelines.",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification-in-the-context-of-bivalency-an",
      "name": "Arabic Dialect Identification in the Context of Bivalency and Code-Switching",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1573/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 41,
      "notes": "In this paper we use a novel approach towards Arabic dialect identification using language bivalency and written code-switching.",
      "metrics": null
    },
    {
      "id": "collection-and-analysis-of-code-switch-egyptian-arabic-engli",
      "name": "Collection and Analysis of Code-switch Egyptian Arabic-English Speech Corpus",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1601/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 37,
      "notes": "Speech corpora are key components needed by both: linguists (in language analyses, research and teaching languages) and Natural Language Processing (NLP)",
      "metrics": null
    },
    {
      "id": "character-level-convolutional-neural-network-for-arabic-dial",
      "name": "Character Level Convolutional Neural Network for Arabic Dialect Identification",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/W18-3913/"
      },
      "year": 2018,
      "venue": "VarDial@COLING 2018",
      "citations": 35,
      "metrics": null
    },
    {
      "id": "part-of-speech-tagging-for-arabic-gulf-dialect-using-bi-lstm",
      "name": "Part-of-Speech Tagging for Arabic Gulf Dialect Using Bi-LSTM",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1620/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 31,
      "notes": "Part-of-speech (POS) tagging is one of the most important addressed areas in the natural language processing (NLP).",
      "metrics": null
    },
    {
      "id": "arabizi-sentiment-analysis-based-on-transliteration-and-auto",
      "name": "Arabizi sentiment analysis based on transliteration and automatic corpus annotation",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "preprocessing",
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/W18-6249/"
      },
      "year": 2018,
      "venue": "WASSA@EMNLP",
      "citations": 26,
      "notes": "Arabizi is a form of writing Arabic text which relies on Latin letters, numerals and punctuation rather than Arabic letters.",
      "metrics": null
    },
    {
      "id": "noise-robust-morphological-disambiguation-for-dialectal-arab",
      "name": "Noise-Robust Morphological Disambiguation for Dialectal Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "paper": "https://aclanthology.org/N18-1087/"
      },
      "year": 2018,
      "venue": "North American Chapter of the Association for Computational Linguistics",
      "citations": 25,
      "notes": "User-generated text tends to be noisy with many lexical and orthographic inconsistencies, making natural language processing (NLP) tasks more challenging.",
      "metrics": null
    },
    {
      "id": "unibuckernel-reloaded-first-place-in-arabic-dialect-identifi",
      "name": "UnibucKernel Reloaded: First Place in Arabic Dialect Identification for the Second Year in a Row",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/1805.04876"
      },
      "year": 2018,
      "venue": "VarDial@COLING 2018",
      "citations": 24,
      "notes": "We present a machine learning approach that ranked on the first place in the Arabic Dialect Identification (ADI) Closed Shared Tasks of the 2018 VarDial",
      "metrics": null
    },
    {
      "id": "a-leveled-reading-corpus-of-modern-standard-arabic",
      "name": "A Leveled Reading Corpus of Modern Standard Arabic",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dataset"
      ],
      "links": {
        "paper": "https://aclanthology.org/L18-1366/"
      },
      "year": 2018,
      "venue": "International Conference on Language Resources and Evaluation",
      "citations": 23,
      "notes": "We present a reading corpus in Modern Standard Arabic to enrich the sparse collection of resources that can be leveraged for educational applications.",
      "metrics": null
    },
    {
      "id": "deep-learning-for-arabic-nlp-a-survey",
      "name": "Deep Learning for Arabic NLP: A Survey",
      "type": "paper",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "year": 2018,
      "venue": "Journal of Computational Science 2018",
      "tasks": [
        "survey",
        "nlp"
      ],
      "links": {
        "paper": "https://www.sciencedirect.com/science/article/pii/S1877750317303757"
      },
      "notes": "Survey of deep learning methods across Arabic NLP tasks.",
      "metrics": null
    },
    {
      "id": "jais-family-590m",
      "name": "Jais-family-590M",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-family-590m"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "590M",
      "year": 2024,
      "on_device": true,
      "notes": "Smallest Jais-family Arabic-English model, suitable for edge and on-device use.",
      "metrics": {
        "downloads": 1985,
        "likes": 7,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "aramodernbert-base-v1-0",
      "name": "AraModernBert-Base-V1.0",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/AraModernBert-Base-V1.0"
      },
      "size": "150M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "ModernBERT-style Arabic encoder with long context from NAMAA.",
      "metrics": {
        "downloads": 1976,
        "likes": 15,
        "lastModified": "2026-03-30"
      }
    },
    {
      "id": "arabic-english-handwritten-ocr-v3",
      "name": "Arabic-English-handwritten-OCR-v3",
      "type": "ocr",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/Arabic-English-handwritten-OCR-v3"
      },
      "notes": "Qwen2.5-VL-3B, trained on 47K handwriting samples",
      "base_model": [
        "qwen/qwen2.5-vl-3b-instruct"
      ],
      "metrics": {
        "downloads": 1972,
        "likes": 14,
        "lastModified": "2025-12-29"
      }
    },
    {
      "id": "qwencleo-asr",
      "name": "QwenCleo-ASR",
      "type": "asr",
      "country": "EG",
      "org": "Mohammed Aly",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammedaly22/QwenCleo-ASR"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Arabic automatic speech recognition model fine-tuned from Qwen/Qwen3-ASR-1.7B.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 1955,
        "likes": 10,
        "lastModified": "2026-06-17"
      }
    },
    {
      "id": "synth-shamela-ocr-arabic-books",
      "name": "synth shamela ocr arabic books",
      "type": "dataset",
      "country": "INTL",
      "org": "freococo",
      "license": "cc-by-nc-nd-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/freococo/synth_shamela_ocr_arabic_books"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Structured book pages rendered dynamically with style, font, and degradation variations.",
      "metrics": {
        "downloads": 1945,
        "likes": 2,
        "lastModified": "2026-07-13"
      }
    },
    {
      "id": "whisper-large-v3-turbo-darija",
      "name": "whisper-large-v3-turbo-darija",
      "type": "asr",
      "country": "MA",
      "org": "Anas Zil",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/anaszil/whisper-large-v3-turbo-darija"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Whisper large-v3-turbo fine-tuned on Moroccan Darija.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 1927,
        "likes": 9,
        "lastModified": "2025-11-09"
      }
    },
    {
      "id": "silma-1-0",
      "name": "SILMA 1.0",
      "type": "llm",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/silma-ai/SILMA-9B-Instruct-v1.0"
      },
      "size": "9B",
      "dialects": [
        "msa"
      ],
      "notes": "Top-ranked Arabic LLM built on Gemma",
      "metrics": {
        "downloads": 1912,
        "likes": 89,
        "lastModified": "2025-06-25"
      }
    },
    {
      "id": "dziribert",
      "name": "DziriBERT",
      "type": "llm",
      "country": "DZ",
      "org": "alger-ia",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/alger-ia/dziribert"
      },
      "size": "124M",
      "year": 2021,
      "dialects": [
        "magh"
      ],
      "notes": "BERT for Algerian dialect incl. Arabizi (Algeria); first Algerian language model.",
      "metrics": {
        "downloads": 1909,
        "likes": 31,
        "lastModified": "2023-03-17"
      }
    },
    {
      "id": "mawrooth-allam-7b-lora",
      "name": "Mawrooth-ALLaM-7B-LoRA",
      "type": "llm",
      "country": "INTL",
      "org": "NorahAlobaied",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "poetry",
        "explanation"
      ],
      "links": {
        "hf": "https://huggingface.co/NorahAlobaied/Mawrooth-ALLaM-7B-LoRA"
      },
      "base_model": [
        "humain-ai/allam-7b-instruct-preview"
      ],
      "dialects": [
        "gulf"
      ],
      "size": "7B",
      "on_device": false,
      "year": 2026,
      "notes": "ALLaM-7B LoRA adapter that explains Nabati poetry verses in five structured sections.",
      "metrics": {
        "downloads": 1909,
        "likes": 0,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "warsh-segments-v3",
      "name": "warsh segments v3",
      "type": "dataset",
      "country": "INTL",
      "org": "Haitam03",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Haitam03/warsh-segments-v3"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Warsh (Rewayat Warsh A'n Nafi') Quran recitation, segmented at waqf with obadx/recitation-segmenter-v2.",
      "metrics": {
        "downloads": 1906,
        "likes": 0,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "alghafa-native",
      "name": "AlGhafa Arabic LLM Benchmark (Native)",
      "type": "benchmark",
      "country": "INTL",
      "org": "Open Arabic LLM Leaderboard (OALL)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OALL/AlGhafa-Arabic-LLM-Benchmark-Native"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Native-Arabic multiple-choice evaluation tasks used as OALL leaderboard tasks.",
      "metrics": {
        "downloads": 1846,
        "likes": 8,
        "lastModified": "2024-03-07"
      }
    },
    {
      "id": "jais-adapted",
      "name": "Jais-adapted (Llama-2 based)",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-adapted-13b-chat"
      },
      "base_model": [
        "inceptionai/jais-adapted-13b"
      ],
      "size": "7B-70B",
      "year": 2024,
      "notes": "Jais-adapted series (7B, 13B, 70B) continued-pretrained from Llama-2 with Arabic vocabulary extension.",
      "metrics": {
        "downloads": 1825,
        "likes": 7,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "recitation-segmenter-v2",
      "name": "recitation segmenter v2",
      "type": "asr",
      "country": "INTL",
      "org": "obadx",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/obadx/recitation-segmenter-v2"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of facebook/w2v-bert-2.0 for segmenting Holy Quran recitations based on pause points (waqf).",
      "base_model": [
        "facebook/w2v-bert-2.0"
      ],
      "metrics": {
        "downloads": 1777,
        "likes": 3,
        "lastModified": "2025-09-03"
      }
    },
    {
      "id": "fastconformer-quran",
      "name": "fastconformer-quran",
      "type": "asr",
      "country": "INTL",
      "org": "Muno459",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Muno459/fastconformer-quran"
      },
      "year": 2026,
      "dialects": [
        "classical"
      ],
      "notes": "FastConformer CTC model for Quran recitation.",
      "base_model": [
        "nvidia/stt_ar_fastconformer_hybrid_large_pcd_v1.0"
      ],
      "metrics": {
        "downloads": 1725,
        "likes": 26,
        "lastModified": "2026-08-15"
      }
    },
    {
      "id": "bert-large-arabertv2",
      "name": "bert-large-arabertv2",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-large-arabertv2"
      },
      "size": "371M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraBERT v2 large with Farasa-segmented input.",
      "metrics": {
        "downloads": 1700,
        "likes": 12,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "bert-base-qarib",
      "name": "bert base qarib",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/ahmedabdelali/bert-base-qarib"
      },
      "year": 2021,
      "notes": "QARiB: QCRI Arabic and Dialectal BERT trained on about 420M tweets and 180M sentences of text.",
      "dialects": [
        "mixed"
      ],
      "metrics": {
        "downloads": 1697,
        "likes": 9,
        "lastModified": "2021-05-20"
      }
    },
    {
      "id": "command-r7b-arabic",
      "name": "Command R7B Arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Cohere",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025"
      },
      "base_model": [
        "coherelabs/c4ai-command-r7b-12-2024"
      ],
      "size": "7B",
      "notes": "Arabic-optimized Command R variant",
      "metrics": {
        "downloads": 1673,
        "likes": 132,
        "lastModified": "2025-10-30"
      }
    },
    {
      "id": "arabic-emirati-female-piper",
      "name": "arabic-emirati-female-piper",
      "type": "tts",
      "country": "AE",
      "org": "Vadim Belsky",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/arabic-emirati-female-piper"
      },
      "year": 2025,
      "dialects": [
        "gulf"
      ],
      "notes": "Piper TTS voice for Emirati Arabic, female speaker.",
      "metrics": {
        "downloads": 1665,
        "likes": 1,
        "lastModified": "2025-12-01"
      }
    },
    {
      "id": "adi17",
      "name": "ADI17",
      "type": "dataset",
      "country": "INTL",
      "org": "Elyadata",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabicSpeech/ADI17"
      },
      "size": "3000h",
      "year": 2020,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic dialect identification speech dataset from YouTube covering 17 countries (Tunisia/Elyadata).",
      "metrics": {
        "downloads": 1645,
        "likes": 2,
        "lastModified": "2025-08-11"
      }
    },
    {
      "id": "jais-2-70b-chat",
      "name": "Jais 2 70B Chat",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/Jais-2-70B-Chat",
        "paper": "https://arxiv.org/abs/2608.13580"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "70B",
      "year": 2025,
      "notes": "Second-generation Jais Arabic-English chat model released Dec 2025.",
      "metrics": {
        "downloads": 1628,
        "likes": 27,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "cohere-speech-tashkeel-2b",
      "name": "Cohere-Speech-Tashkeel-2B",
      "type": "asr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Cohere-Speech-Tashkeel-2B"
      },
      "size": "2.1B",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Speech model that transcribes Arabic with tashkeel (diacritics).",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 1605,
        "likes": 28,
        "lastModified": "2026-07-13"
      }
    },
    {
      "id": "deepseek-ocr-arabic-v1",
      "name": "deepseek ocr arabic v1",
      "type": "dataset",
      "country": "INTL",
      "org": "FatimahEmadEldin",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "fine-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/deepseek-ocr-arabic-v1"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "550 tokenized Arabic OCR samples with image patches and labels for fine-tuning DeepSeek-OCR.",
      "metrics": {
        "downloads": 1603,
        "likes": 0,
        "lastModified": "2026-02-09"
      }
    },
    {
      "id": "rootformer-v14-sovereign-transmute",
      "name": "rootformer v14 sovereign transmute",
      "type": "llm",
      "country": "INTL",
      "org": "enver",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/enver/rootformer-v14-sovereign-transmute"
      },
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Rootformer v14.3 formalizes four generative algorithmic (also: 1 variants)",
      "metrics": {
        "downloads": 1537,
        "likes": 0,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "arabic-game-arabizations",
      "name": "arabic game arabizations",
      "type": "dataset",
      "country": "INTL",
      "org": "jynxzio5",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/jynxzio5/arabic-game-arabizations"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Catalog scraped from wolfgame-ar.site for use by an installer app.",
      "metrics": {
        "downloads": 1535,
        "likes": 0,
        "lastModified": "2026-08-04"
      }
    },
    {
      "id": "opus-mt-ar-fr",
      "name": "opus mt ar fr",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-ar-fr"
      },
      "year": 2023,
      "notes": "OPUS-MT transformer translation model from Arabic to French.",
      "metrics": {
        "downloads": 1535,
        "likes": 1,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "oall-arabic-mmlu",
      "name": "OALL Arabic MMLU",
      "type": "benchmark",
      "country": "INTL",
      "org": "Open Arabic LLM Leaderboard (OALL)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OALL/Arabic_MMLU"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "GPT-translated Arabic MMLU copy from FreedomIntelligence, used as an OALL v1 task.",
      "metrics": {
        "downloads": 1528,
        "likes": 3,
        "lastModified": "2024-09-05"
      }
    },
    {
      "id": "jais-family-590m-chat",
      "name": "jais family 590m chat",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-family-590m-chat"
      },
      "base_model": [
        "inceptionai/jais-family-590m"
      ],
      "size": "590M",
      "on_device": true,
      "year": 2024,
      "tags": [
        "variants:3"
      ],
      "notes": "The Jais family of models is a comprehensive series of bilingual English-Arabic large language models (LLMs). (also: 3 variants)",
      "metrics": {
        "downloads": 1527,
        "likes": 7,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "nvidia-fastconformer-arabic-diacritics",
      "name": "NVIDIA FastConformer Arabic (Diacritics)",
      "type": "asr",
      "country": "INTL",
      "org": "NVIDIA",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/nvidia/stt_ar_fastconformer_hybrid_large_pcd_v1.0"
      },
      "notes": "Arabic ASR with diacritical marks, 1100h training",
      "metrics": {
        "downloads": 1524,
        "likes": 42,
        "lastModified": "2025-10-21"
      }
    },
    {
      "id": "linto-dataset-audio-ar-tn",
      "name": "linto dataset audio ar tn",
      "type": "dataset",
      "country": "INTL",
      "org": "LINAGORA",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/linagora/linto-dataset-audio-ar-tn"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This is the first packaged version of the datasets used to train the Linto Tunisian dialect with code-switching STT (linagora/linto-asr-ar-tn).",
      "metrics": {
        "downloads": 1520,
        "likes": 22,
        "lastModified": "2025-04-11"
      }
    },
    {
      "id": "human-translated-arabic-mmlu",
      "name": "Human-Translated Arabic MMLU",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/human_translated_arabic_mmlu"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Human-translated MMLU into Arabic from MBZUAI, companion to ArabicMMLU.",
      "metrics": {
        "downloads": 1514,
        "likes": 4,
        "lastModified": "2024-09-17"
      }
    },
    {
      "id": "arabic-audio-collection-algerian-loubna-stories",
      "name": "arabic audio collection algerian loubna stories",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-algerian-loubna-stories"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Loubna Stories Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 237 hours of speech recordings.",
      "metrics": {
        "downloads": 1478,
        "likes": 0,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "fanar-2-oryx-ivu",
      "name": "Fanar-2-Oryx-IVU",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "image-understanding",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/Fanar-2-Oryx-IVU",
        "paper": "https://arxiv.org/abs/2603.16397"
      },
      "size": "7B",
      "year": 2026,
      "notes": "Arabic-English vision-language model on Qwen2.5-VL-7B with cultural understanding focus.",
      "base_model": [
        "qwen/qwen2.5-vl-7b-instruct"
      ],
      "metrics": {
        "downloads": 1463,
        "likes": 4,
        "lastModified": "2026-03-25"
      }
    },
    {
      "id": "cc-100",
      "name": "CC-100",
      "type": "dataset",
      "country": "INTL",
      "org": "Facebook AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://data.statmt.org/cc-100/",
        "hf": "https://huggingface.co/datasets/statmt/cc100",
        "paper": "https://aclanthology.org/2020.lrec-1.494.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "7,132,000 documents",
      "year": 2020,
      "notes": "Monolingual datasets from Common Crawl for a variety of languages",
      "metrics": {
        "downloads": 1462,
        "likes": 107,
        "lastModified": "2024-03-05"
      }
    },
    {
      "id": "mt5-m2o-arabic-crosssum",
      "name": "mT5_m2o_arabic_crossSum",
      "type": "llm",
      "country": "INTL",
      "org": "csebuetnlp",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/csebuetnlp/mT5_m2o_arabic_crossSum"
      },
      "year": 2023,
      "notes": "This repository contains the many-to-one (m2o) mT5 checkpoint finetuned on all cross-lingual pairs of the CrossSum dataset.",
      "metrics": {
        "downloads": 1462,
        "likes": 4,
        "lastModified": "2023-11-15"
      }
    },
    {
      "id": "wav2vec2-arabic-phoneme-asr",
      "name": "wav2vec2-arabic-phoneme-asr",
      "type": "asr",
      "country": "INTL",
      "org": "Mostafa Maroof",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MostafaMaroof/wav2vec2-arabic-phoneme-asr"
      },
      "size": "316M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "wav2vec2 phoneme recognizer for Arabic.",
      "base_model": [
        "facebook/wav2vec2-large-xlsr-53"
      ],
      "metrics": {
        "downloads": 1462,
        "likes": 1,
        "lastModified": "2026-07-06"
      }
    },
    {
      "id": "tlog",
      "name": "tlog",
      "type": "dataset",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarteel-ai/tlog"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "TLOG: audio recitations of Quranic ayahs paired with their Quranic texts.",
      "metrics": {
        "downloads": 1441,
        "likes": 19,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "noormontai",
      "name": "NoormontAI",
      "type": "llm",
      "country": "INTL",
      "org": "Noormont",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Noormont/NoormontAI"
      },
      "year": 2026,
      "notes": "تم حفظ النموذج عند الخطوة 20000.",
      "base_model": [
        "noormont/noormontai"
      ],
      "metrics": {
        "downloads": 1399,
        "likes": 0,
        "lastModified": "2026-09-09"
      }
    },
    {
      "id": "whisper-tiny-ar-quran",
      "name": "whisper-tiny-ar-quran",
      "type": "asr",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/tarteel-ai/whisper-tiny-ar-quran"
      },
      "year": 2022,
      "on_device": true,
      "dialects": [
        "classical"
      ],
      "notes": "Whisper tiny fine-tuned for Quran recitation transcription.",
      "metrics": {
        "downloads": 1398,
        "likes": 29,
        "lastModified": "2022-12-14"
      }
    },
    {
      "id": "arabart",
      "name": "AraBART",
      "type": "llm",
      "country": "INTL",
      "org": "Moussa Kamal Eddine",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "summarization",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/moussaKam/AraBART"
      },
      "year": 2022,
      "dialects": [
        "msa"
      ],
      "notes": "AraBART, first Arabic seq2seq model with pretrained encoder and decoder.",
      "metrics": {
        "downloads": 1391,
        "likes": 18,
        "lastModified": "2022-05-05"
      }
    },
    {
      "id": "aramix",
      "name": "AraMix",
      "type": "dataset",
      "country": "INTL",
      "org": "AdaMLLab",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AdaMLLab/AraMix"
      },
      "size": "100M–1B rows",
      "year": 2026,
      "notes": "AraMix: Arabic pretraining corpus of over 100M rows combining deduplicated sources, with a quality-matched split; see arXiv 2512.18834.",
      "metrics": {
        "downloads": 1373,
        "likes": 7,
        "lastModified": "2026-01-30"
      }
    },
    {
      "id": "whisper-large-v3-arabic-dialectal-v2",
      "name": "whisper-large-v3-arabic-dialectal-v2",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/whisper-large-v3-arabic-dialectal-v2"
      },
      "size": "1.5B",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Whisper large-v3 fine-tuned on Arabic dialects.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 1370,
        "likes": 2,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "emiratitts-smoke-samples",
      "name": "EmiratiTTS-smoke-samples",
      "type": "dataset",
      "country": "INTL",
      "org": "Alqayed2024",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Alqayed2024/EmiratiTTS-smoke-samples"
      },
      "dialects": [
        "gulf",
        "lev"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "These 10 audio clips are the stage 0.5 acceptance check for the EmiratiTTS project.",
      "metrics": {
        "downloads": 1366,
        "likes": 1,
        "lastModified": "2026-04-12"
      }
    },
    {
      "id": "quranlab-arabic-speech",
      "name": "QuranLab-Arabic-Speech",
      "type": "dataset",
      "country": "INTL",
      "org": "Muno459",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Muno459/QuranLab-Arabic-Speech"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Broad Arabic speech from 50,000+ hours of recordings, cut into 2-20 s segments for ASR training; access restricted.",
      "metrics": {
        "downloads": 1347,
        "likes": 0,
        "lastModified": "2026-10-05"
      }
    },
    {
      "id": "arabic-minilm-l12-v2-all-nli-triplet",
      "name": "Arabic-MiniLM-L12-v2-all-nli-triplet",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-MiniLM-L12-v2-all-nli-triplet"
      },
      "size": "118M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Small Arabic MiniLM sentence embedding model trained on NLI triplets.",
      "base_model": [
        "sentence-transformers/paraphrase-multilingual-minilm-l12-v2"
      ],
      "metrics": {
        "downloads": 1338,
        "likes": 4,
        "lastModified": "2025-06-10"
      }
    },
    {
      "id": "xtd-11",
      "name": "xtd 11",
      "type": "dataset",
      "country": "INTL",
      "org": "Arabic-Clip",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "embedding",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Arabic-Clip/xtd_11"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "The expanded XTD-11 dataset, now including Arabic, enhances the original XTD collection.",
      "metrics": {
        "downloads": 1326,
        "likes": 3,
        "lastModified": "2024-08-11"
      }
    },
    {
      "id": "unnamed-ar",
      "name": "unnamed ar",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "cc-by-3.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/unnamed-ar"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The data/ar/ partition of espnet/yodas3, repackaged as parquet with the audio inline so it loads without a script: Every row is one long-form recording.",
      "metrics": {
        "downloads": 1324,
        "likes": 0,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "karnak-40b-v1-0",
      "name": "Karnak 40B v1.0",
      "type": "llm",
      "country": "EG",
      "org": "Applied Innovation Center (MCIT Egypt)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Applied-Innovation-Center/Karnak-40B-v1.0"
      },
      "size": "40B",
      "on_device": false,
      "year": 2026,
      "notes": "Karnak is a powerful AI model that works in both Arabic and English.",
      "base_model": [
        "qwen/qwen3-30b-a3b-instruct-2507"
      ],
      "metrics": {
        "downloads": 1323,
        "likes": 72,
        "lastModified": "2026-04-24"
      }
    },
    {
      "id": "karnak-llm",
      "name": "Karnak LLM",
      "type": "llm",
      "country": "EG",
      "org": "Applied Innovation Center (MCIT Egypt)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "website": "https://itida.gov.eg/Arabic/PressReleases/Pages/egypt-national-ai-karnak-llm-launch-Ai-Everything-MEA-2026.aspx",
        "hf": "https://huggingface.co/Applied-Innovation-Center/Karnak"
      },
      "year": 2026,
      "notes": "Egypt national Arabic LLM; first national apps built on it: SIA tutor and a legal/regulatory guidance assistant.",
      "base_model": [
        "qwen/qwen3-30b-a3b-instruct-2507"
      ],
      "metrics": {
        "downloads": 1323,
        "likes": 72,
        "lastModified": "2026-04-24"
      }
    },
    {
      "id": "silma-kashif",
      "name": "SILMA Kashif",
      "type": "llm",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/silma-ai/SILMA-Kashif-2B-Instruct-v1.0"
      },
      "size": "2B",
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Lightweight RAG-optimized Arabic model",
      "metrics": {
        "downloads": 1315,
        "likes": 24,
        "lastModified": "2025-06-11"
      }
    },
    {
      "id": "flair-arabic-multi-ner",
      "name": "flair arabic multi ner",
      "type": "llm",
      "country": "INTL",
      "org": "megantosh",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/megantosh/flair-arabic-multi-ner"
      },
      "year": 2022,
      "notes": "Training was conducted over 94 epochs, using a linear decaying learning rate of 2e-05, starting from 0.225 and a batch size of 32 with GloVe.",
      "metrics": {
        "downloads": 1297,
        "likes": 5,
        "lastModified": "2022-03-09"
      }
    },
    {
      "id": "arabic-english-bge-m3",
      "name": "arabic english bge m3",
      "type": "embedding",
      "country": "INTL",
      "org": "sayed0am",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/sayed0am/arabic-english-bge-m3"
      },
      "year": 2025,
      "notes": "This model is a 36.2% smaller version of BAAI/bge-m3 for the Arabic language.",
      "base_model": [
        "baai/bge-m3"
      ],
      "metrics": {
        "downloads": 1282,
        "likes": 9,
        "lastModified": "2025-10-25"
      }
    },
    {
      "id": "athar-shamela4",
      "name": "Athar Shamela4",
      "type": "dataset",
      "country": "INTL",
      "org": "Kandil7",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Kandil7/Athar-Shamela4"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "A complete extraction of al-Maktaba al-Shamela (الشاملة) v4, containing 8,589 books across 40 categories of classical Islamic sciences.",
      "metrics": {
        "downloads": 1262,
        "likes": 1,
        "lastModified": "2026-06-02"
      }
    },
    {
      "id": "miracl-vision",
      "name": "MIRACL-VISION",
      "type": "benchmark",
      "country": "INTL",
      "org": "NVIDIA",
      "license": "cc-by-sa-4.0",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "retrieval",
        "rag"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nvidia/miracl-vision",
        "paper": "https://arxiv.org/pdf/2505.11651"
      },
      "dialects": [
        "msa"
      ],
      "size": "75,444 documents",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Multilingual visual document retrieval benchmark extending MIRACL with Wikipedia page images for 18 languages.",
      "metrics": {
        "downloads": 1255,
        "likes": 13,
        "lastModified": "2025-05-20"
      }
    },
    {
      "id": "qari-ocr-0-2-2-1-vl-2b-instruct",
      "name": "Qari OCR 0.2.2.1 VL 2B Instruct",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "chat",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Qari-OCR-0.2.2.1-VL-2B-Instruct"
      },
      "size": "2B",
      "on_device": true,
      "year": 2025,
      "notes": "This is the model described in the paper QARI-OCR: High-Fidelity Arabic Text Recognition through Multimodal Large Language Model Adaptation.",
      "base_model": [
        "unsloth/qwen2-vl-2b-instruct-unsloth-bnb-4bit"
      ],
      "metrics": {
        "downloads": 1250,
        "likes": 25,
        "lastModified": "2025-06-07"
      }
    },
    {
      "id": "moonshine-tiny-ar",
      "name": "moonshine tiny ar",
      "type": "asr",
      "country": "INTL",
      "org": "moonshine-ai",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/moonshine-ai/moonshine-tiny-ar"
      },
      "year": 2025,
      "notes": "This model is part of the Moonshine family of tiny specialized Automatic Speech Recognition (ASR) models for edge devices.",
      "metrics": {
        "downloads": 1249,
        "likes": 8,
        "lastModified": "2025-09-06"
      }
    },
    {
      "id": "bert-mini-arabic",
      "name": "bert mini arabic",
      "type": "llm",
      "country": "INTL",
      "org": "asafaya",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/asafaya/bert-mini-arabic"
      },
      "year": 2023,
      "tags": [
        "variants:2"
      ],
      "notes": "Arabic BERT Mini encoder pretrained on OSCAR and Wikipedia.",
      "on_device": true,
      "metrics": {
        "downloads": 1244,
        "likes": 3,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "marbert-all-nli-triplet-matryoshka",
      "name": "Marbert all nli triplet Matryoshka",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Marbert-all-nli-triplet-Matryoshka"
      },
      "year": 2025,
      "notes": "This is a sentence-transformers model finetuned from UBC-NLP/MARBERTv2 on the Omartificial-Intelligence-Space/arabic-nli-triplet dataset.",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 1223,
        "likes": 1,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "ar-quran-hadith14books-msa",
      "name": "ar quran hadith14books MSA",
      "type": "dataset",
      "country": "INTL",
      "org": "Dr-AliGomaa",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "chat",
        "summarization",
        "speech",
        "quran",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Dr-AliGomaa/ar-quran-hadith14books-MSA"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic speech for both primary sources of Islam — Quran and Hadith.",
      "metrics": {
        "downloads": 1219,
        "likes": 6,
        "lastModified": "2026-08-14"
      }
    },
    {
      "id": "audar-asr-v1-flash",
      "name": "Audar ASR V1 Flash",
      "type": "asr",
      "country": "INTL",
      "org": "Audar AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/audarai/Audar-ASR-V1-Flash"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Audar's Arabic-first ASR — the real-time, edge tier.",
      "metrics": {
        "downloads": 1216,
        "likes": 7,
        "lastModified": "2026-08-20"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-did-madar-corpus6",
      "name": "bert base arabic camelbert mix did madar corpus6",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-did-madar-corpus6"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "notes": "For the fine-tuning, we used the MADAR Corpus 6 dataset, which includes 6 labels.",
      "metrics": {
        "downloads": 1216,
        "likes": 1,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "arbert",
      "name": "ARBERT",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/ARBERT"
      },
      "year": 2022,
      "dialects": [
        "msa"
      ],
      "notes": "ARBERT, 61GB MSA BERT from UBC.",
      "metrics": {
        "downloads": 1208,
        "likes": 5,
        "lastModified": "2022-01-19"
      }
    },
    {
      "id": "oall-arabic-exams",
      "name": "Arabic EXAMS (OALL)",
      "type": "benchmark",
      "country": "INTL",
      "org": "Open Arabic LLM Leaderboard (OALL)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OALL/Arabic_EXAMS"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic subset of the EXAMS multilingual school-exam benchmark, an OALL task.",
      "metrics": {
        "downloads": 1207,
        "likes": 3,
        "lastModified": "2024-02-16"
      }
    },
    {
      "id": "speecht5-tts-clartts-ar",
      "name": "speecht5_tts_clartts_ar",
      "type": "tts",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/speecht5_tts_clartts_ar"
      },
      "dialects": [
        "classical"
      ],
      "notes": "SpeechT5 for Classical Arabic TTS",
      "metrics": {
        "downloads": 1198,
        "likes": 32,
        "lastModified": "2025-09-10"
      }
    },
    {
      "id": "muffakir-embedding-v2",
      "name": "Muffakir Embedding V2",
      "type": "embedding",
      "country": "INTL",
      "org": "mohamed2811",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/mohamed2811/Muffakir_Embedding_V2"
      },
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "Muffakir This is the second version of the MuffakirEmbedding model. (also: 1 variants)",
      "base_model": [
        "sayed0am/arabic-english-bge-m3"
      ],
      "metrics": {
        "downloads": 1194,
        "likes": 6,
        "lastModified": "2025-05-24"
      }
    },
    {
      "id": "algerian-youtube-comments",
      "name": "Algerian Youtube Comments",
      "type": "dataset",
      "country": "INTL",
      "org": "Touati Kamel",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/touati-kamel/Algerian-Youtube-Comments"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Algerian YouTube comments dataset with a comment-extraction pipeline, for Darija NLP and pretraining.",
      "metrics": {
        "downloads": 1193,
        "likes": 0,
        "lastModified": "2026-09-28"
      }
    },
    {
      "id": "arabic-flicker-8k",
      "name": "Arabic Flicker 8k",
      "type": "dataset",
      "country": "INTL",
      "org": "Arabic-Clip",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "image-captioning",
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Arabic-Clip/Arabic_Flicker_8k"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic version of Flickr8k images for Arabic image-captioning and CLIP training.",
      "metrics": {
        "downloads": 1189,
        "likes": 0,
        "lastModified": "2024-07-09"
      }
    },
    {
      "id": "arabic-dialects-gold20",
      "name": "arabic dialects gold20",
      "type": "dataset",
      "country": "INTL",
      "org": "TigreGotico",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "tts",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TigreGotico/arabic-dialects-gold20"
      },
      "dialects": [
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "660 sentences across 33 Arabic lects, each with diacritized dialectal spelling, gold IPA, English gloss and phonetic feature tags.",
      "metrics": {
        "downloads": 1188,
        "likes": 0,
        "lastModified": "2026-07-20"
      }
    },
    {
      "id": "qari-ocr-0-4-0-vl-4b-instruct",
      "name": "Qari-OCR-0.4.0-VL-4B-Instruct",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Qari-OCR-0.4.0-VL-4B-Instruct"
      },
      "size": "4B",
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Qari OCR 0.4.0, 4B vision-language Arabic OCR model.",
      "base_model": [
        "qwen/qwen3-vl-4b-instruct"
      ],
      "metrics": {
        "downloads": 1172,
        "likes": 10,
        "lastModified": "2026-02-01"
      }
    },
    {
      "id": "coru",
      "name": "CORU",
      "type": "dataset",
      "country": "INTL",
      "org": "DataScienceUIBK",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdoelsayed/CORU"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Multilingual OCR and information extraction from receipts remains challenging, particularly for complex scripts like Ar",
      "metrics": {
        "downloads": 1164,
        "likes": 15,
        "lastModified": "2025-06-19"
      }
    },
    {
      "id": "arams-restore",
      "name": "AraMS-Restore",
      "type": "dataset",
      "country": "INTL",
      "org": "Archatext",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Archatext/AraMS-Restore"
      },
      "size": "<1K rows",
      "year": 2026,
      "notes": "177 line images cropped from real damaged pages of a historical Arabic manuscript (book09), each with its transcription.",
      "metrics": {
        "downloads": 1157,
        "likes": 0,
        "lastModified": "2026-08-04"
      }
    },
    {
      "id": "mgb2-arabic",
      "name": "mgb2 arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/mgb2-arabic"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "The Arabic Multi-Genre Broadcast (MGB-2) dataset is a large-scale speech recognition corpus containing 1,200 hours of Arabic broadcast audio.",
      "metrics": {
        "downloads": 1156,
        "likes": 7,
        "lastModified": "2025-12-27"
      }
    },
    {
      "id": "arabic-documents-dataset-pdf",
      "name": "Arabic Documents Dataset PDF",
      "type": "dataset",
      "country": "INTL",
      "org": "HumynLabs",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "document-understanding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HumynLabs/Arabic_Documents_Dataset_PDF"
      },
      "size": "<1K rows",
      "year": 2025,
      "notes": "127 Arabic PDF documents (books, articles, reports) for document understanding and OCR research.",
      "metrics": {
        "downloads": 1136,
        "likes": 0,
        "lastModified": "2025-11-07"
      }
    },
    {
      "id": "arabic-all-nli-triplet-matryoshka",
      "name": "Arabic-all-nli-triplet-Matryoshka",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-all-nli-triplet-Matryoshka"
      },
      "size": "278M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic Matryoshka embeddings trained on translated all-nli triplets.",
      "base_model": [
        "sentence-transformers/paraphrase-multilingual-mpnet-base-v2"
      ],
      "metrics": {
        "downloads": 1124,
        "likes": 6,
        "lastModified": "2025-01-23"
      }
    },
    {
      "id": "lahgtna-v3-small",
      "name": "Lahgtna v3 Small",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/lahgtna-v3-small"
      },
      "size": "267h",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Dialect-balanced Arabic ASR corpus, 54,600 clips over 13 dialects.",
      "metrics": {
        "downloads": 1107,
        "likes": 7,
        "lastModified": "2026-07-17"
      }
    },
    {
      "id": "mms-300m-arabic-dialect-identifier",
      "name": "mms 300m arabic dialect identifier",
      "type": "asr",
      "country": "INTL",
      "org": "badrex",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/badrex/mms-300m-arabic-dialect-identifier"
      },
      "dialects": [
        "mixed"
      ],
      "size": "300M",
      "on_device": true,
      "year": 2025,
      "notes": "Arabic audio classification model fine-tuned from facebook/mms-300m.",
      "base_model": [
        "facebook/mms-300m"
      ],
      "metrics": {
        "downloads": 1080,
        "likes": 9,
        "lastModified": "2025-10-29"
      }
    },
    {
      "id": "nabra-82m-v0-1",
      "name": "Nabra-82M-v0.1",
      "type": "tts",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nabra-82M-v0.1"
      },
      "size": "82M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Nabra 82M compact Arabic text-to-speech.",
      "base_model": [
        "hexgrad/kokoro-82m"
      ],
      "metrics": {
        "downloads": 1075,
        "likes": 10,
        "lastModified": "2026-07-03"
      }
    },
    {
      "id": "nadi2024-baseline",
      "name": "NADI2024 baseline",
      "type": "llm",
      "country": "INTL",
      "org": "VARabi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/VARabi/NADI2024-baseline"
      },
      "year": 2024,
      "notes": "A BERT-based model fine-tuned to perform single-label Arabic Dialect Identification (ADI).",
      "metrics": {
        "downloads": 1072,
        "likes": 2,
        "lastModified": "2024-09-22"
      }
    },
    {
      "id": "stanza-ar",
      "name": "stanza ar",
      "type": "llm",
      "country": "INTL",
      "org": "stanfordnlp",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/stanfordnlp/stanza-ar"
      },
      "year": 2026,
      "notes": "Stanza is a collection of accurate and efficient tools for the linguistic analysis of many human languages.",
      "metrics": {
        "downloads": 1066,
        "likes": 0,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "medqa-darija-multilingual",
      "name": "MedQA-Darija-MultiLingual",
      "type": "dataset",
      "country": "INTL",
      "org": "Williamsanderson",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "medical",
        "qa",
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Williamsanderson/MedQA-Darija-MultiLingual"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Trilingual English, French and Moroccan Darija medical Q&A dataset with playable speech audio.",
      "metrics": {
        "downloads": 1063,
        "likes": 4,
        "lastModified": "2026-05-06"
      }
    },
    {
      "id": "quran-md-ayahs",
      "name": "quran md ayahs",
      "type": "dataset",
      "country": "INTL",
      "org": "Buraaq",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Buraaq/quran-md-ayahs"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Ayah-level split of Quran-MD, a multimodal Quran dataset aligning text, linguistic annotation and recitation audio.",
      "metrics": {
        "downloads": 1054,
        "likes": 20,
        "lastModified": "2026-01-27"
      }
    },
    {
      "id": "arabic-chat-llm-data",
      "name": "arabic chat llm data",
      "type": "dataset",
      "country": "INTL",
      "org": "amrykytt57",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/amrykytt57/arabic-chat-llm-data"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "Raw text from 4 sources (light cleaning: NFKC, diacritics/tatweel, URLs; exact de-duplication; no content classification), chat conversations, and token shards.",
      "metrics": {
        "downloads": 1048,
        "likes": 1,
        "lastModified": "2026-10-01"
      }
    },
    {
      "id": "everyayah-wav",
      "name": "everyayah wav",
      "type": "dataset",
      "country": "INTL",
      "org": "Ahmed Hany",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/dev-ahmedhany/everyayah-wav"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Full-mushaf Quranic recitation audio at 16 kHz mono 16-bit WAV, re-encoded from everyayah.com for ML / ASR research.",
      "metrics": {
        "downloads": 1043,
        "likes": 0,
        "lastModified": "2026-05-30"
      }
    },
    {
      "id": "whisper-large-v3-turbo-arabic-dialectal-v2",
      "name": "whisper-large-v3-turbo-arabic-dialectal-v2",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/whisper-large-v3-turbo-arabic-dialectal-v2"
      },
      "size": "809M",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Whisper large-v3-turbo fine-tuned on Arabic dialects.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 1038,
        "likes": 2,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "audar-tts-v1-flash",
      "name": "Audar TTS V1 Flash",
      "type": "tts",
      "country": "INTL",
      "org": "Audar AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/audarai/Audar-TTS-V1-Flash"
      },
      "year": 2026,
      "notes": "Open, Arabic-first, expressive zero-shot text-to-speech — quantized to run anywhere.",
      "metrics": {
        "downloads": 1037,
        "likes": 9,
        "lastModified": "2026-07-07"
      }
    },
    {
      "id": "gpt2-arabic-20k-lc",
      "name": "gpt2 arabic 20k lc",
      "type": "llm",
      "country": "INTL",
      "org": "aariciah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/aariciah/gpt2-arabic-20k-lc"
      },
      "year": 2026,
      "notes": "This model is a fine-tuned version of on the None dataset.",
      "metrics": {
        "downloads": 1012,
        "likes": 0,
        "lastModified": "2026-09-05"
      }
    },
    {
      "id": "egymmlu",
      "name": "EgyMMLU",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/EgyMMLU"
      },
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "MMLU translated into Egyptian Arabic from UBC (Canada), released with NileChat.",
      "metrics": {
        "downloads": 996,
        "likes": 0,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "fastconformer-quran-ar",
      "name": "fastconformer quran ar",
      "type": "asr",
      "country": "INTL",
      "org": "mohammed",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammed/fastconformer-quran-ar"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A fine-tuned NVIDIA FastConformer Hybrid Large model for Quranic Arabic speech recognition.",
      "metrics": {
        "downloads": 969,
        "likes": 18,
        "lastModified": "2026-06-07"
      }
    },
    {
      "id": "petraai",
      "name": "PetraAI",
      "type": "dataset",
      "country": "INTL",
      "org": "PetraAI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner",
        "qa",
        "translation",
        "summarization",
        "embedding",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/PetraAI/PetraAI"
      },
      "size": "1M–10M rows",
      "year": 2023,
      "notes": "PETRA is a multilingual dataset for training and evaluating AI systems on a diverse range of tasks across multiple modalities.",
      "metrics": {
        "downloads": 947,
        "likes": 22,
        "lastModified": "2023-09-14"
      }
    },
    {
      "id": "darijabert-arabizi",
      "name": "DarijaBERT-arabizi",
      "type": "llm",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/SI2M-Lab/DarijaBERT-arabizi"
      },
      "size": "255M",
      "dialects": [
        "magh"
      ],
      "notes": "DarijaBERT variant for Darija written in Latin script (Arabizi).",
      "metrics": {
        "downloads": 943,
        "likes": 10,
        "lastModified": "2024-09-25"
      }
    },
    {
      "id": "smolkalam-arabic-conversational-sft",
      "name": "smolkalam arabic conversational sft",
      "type": "dataset",
      "country": "INTL",
      "org": "AdaMLLab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AdaMLLab/smolkalam-arabic-conversational-sft"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "SmolKalam is a quality-filtered Arabic SFT dataset of 1,790,478 examples (2.45B tokens), built as an ensemble translation of SmolTalk2.",
      "metrics": {
        "downloads": 942,
        "likes": 3,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "nvidia-fastconformer-arabic",
      "name": "NVIDIA FastConformer Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "NVIDIA",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/nvidia/stt_ar_fastconformer_hybrid_large_pc_v1.0"
      },
      "notes": "SOTA Arabic ASR, 115M params, 760h training, CC-BY-4.0",
      "metrics": {
        "downloads": 941,
        "likes": 19,
        "lastModified": "2025-10-23"
      }
    },
    {
      "id": "arabic-whisper-codeswitching-edition",
      "name": "Arabic-Whisper-CodeSwitching-Edition",
      "type": "asr",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedRashad/Arabic-Whisper-CodeSwitching-Edition"
      },
      "size": "1.5B",
      "year": 2024,
      "dialects": [
        "mixed"
      ],
      "notes": "Whisper fine-tuned for Arabic with English code-switching.",
      "metrics": {
        "downloads": 939,
        "likes": 37,
        "lastModified": "2024-07-07"
      }
    },
    {
      "id": "openiti-vectors",
      "name": "openiti vectors",
      "type": "dataset",
      "country": "INTL",
      "org": "Maktabati",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Maktabati/openiti-vectors"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "This dataset contains the fully vectorized OpenITI RELEASE 2025-1-9 collection of classical Islamic texts, prepared for semantic search (RAG).",
      "metrics": {
        "downloads": 937,
        "likes": 0,
        "lastModified": "2026-05-30"
      }
    },
    {
      "id": "darijatts-v0-1-500m",
      "name": "DarijaTTS-v0.1-500M",
      "type": "tts",
      "country": "MA",
      "org": "Kandir Research",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/KandirResearch/DarijaTTS-v0.1-500M"
      },
      "size": "499M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "DarijaTTS 500M Moroccan Darija text-to-speech.",
      "base_model": [
        "outeai/outetts-0.2-500m"
      ],
      "metrics": {
        "downloads": 936,
        "likes": 0,
        "lastModified": "2025-11-26"
      }
    },
    {
      "id": "moulsot-v0-3",
      "name": "moulsot.v0.3",
      "type": "asr",
      "country": "MA",
      "org": "atlasia",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/moulsot.v0.3"
      },
      "size": "2B",
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Moulsot ASR for Moroccan Darija from Atlasia.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 936,
        "likes": 17,
        "lastModified": "2026-05-02"
      }
    },
    {
      "id": "tarteel-ea-ud",
      "name": "Tarteel EA-UD",
      "type": "dataset",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarteel-ai/EA-UD"
      },
      "year": 2022,
      "dialects": [
        "classical"
      ],
      "notes": "Tarteel evaluation set of user-submitted Quran recitations.",
      "metrics": {
        "downloads": 936,
        "likes": 2,
        "lastModified": "2022-07-15"
      }
    },
    {
      "id": "asr-code-switch",
      "name": "ASR Code Switch",
      "type": "benchmark",
      "country": "INTL",
      "org": "Perle-ai",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Perle-ai/ASR_Code_Switch"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "A curated benchmark of 1,200 code-switching utterances (300 per language pair) for evaluating commercial ASR systems on multilingual speech.",
      "metrics": {
        "downloads": 923,
        "likes": 12,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "marbertv2-arabic-written-dialect-classifier",
      "name": "marbertv2 arabic written dialect classifier",
      "type": "llm",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimAmin/marbertv2-arabic-written-dialect-classifier"
      },
      "dialects": [
        "egy",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of UBC-NLP/MARBERTv2 for Arabic written dialect classification.",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 923,
        "likes": 6,
        "lastModified": "2025-05-11"
      }
    },
    {
      "id": "namaa-saudi-tts-v2",
      "name": "NAMAA-Saudi-TTS-V2",
      "type": "tts",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/NAMAA-Saudi-TTS-V2"
      },
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Second-generation Saudi-dialect TTS from NAMAA.",
      "base_model": [
        "swivid/habibi-tts"
      ],
      "metrics": {
        "downloads": 922,
        "likes": 10,
        "lastModified": "2026-04-21"
      }
    },
    {
      "id": "asl-4b-v1",
      "name": "ASL 4B v1",
      "type": "llm",
      "country": "INTL",
      "org": "saai-sa",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/saai-sa/ASL-4B-v1"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "size": "4B",
      "on_device": false,
      "year": 2026,
      "notes": "ASL (أصل) is a 4-billion-parameter Arabic language model developed by SAAI.",
      "base_model": [
        "qwen3.5"
      ],
      "metrics": {
        "downloads": 920,
        "likes": 19,
        "lastModified": "2026-08-26"
      }
    },
    {
      "id": "opus-mt-fr-ar",
      "name": "opus mt fr ar",
      "type": "llm",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP/opus-mt-fr-ar"
      },
      "year": 2023,
      "notes": "Tatoeba-Challenge transformer translating French into Arabic, with several Arabic varieties as targets.",
      "metrics": {
        "downloads": 919,
        "likes": 0,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "artelingo-dummy",
      "name": "artelingo dummy",
      "type": "benchmark",
      "country": "INTL",
      "org": "youssef101",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/youssef101/artelingo-dummy"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "ArtELingo is a benchmark and dataset introduced in a research paper aimed at promoting work on diversity across languages and cultures.",
      "metrics": {
        "downloads": 912,
        "likes": 2,
        "lastModified": "2023-07-23"
      }
    },
    {
      "id": "quran-ayah-corpus",
      "name": "Quran Ayah Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "rabah2026",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/rabah2026/Quran-Ayah-Corpus"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "Ayah-Corpus is a large-scale, multi-reciter Arabic speech dataset meticulously curated for Automatic Speech Recognition (ASR) tasks.",
      "metrics": {
        "downloads": 912,
        "likes": 2,
        "lastModified": "2025-09-19"
      }
    },
    {
      "id": "masrygpt-chat-1-5b",
      "name": "MasryGPT-Chat-1.5B",
      "type": "llm",
      "country": "INTL",
      "org": "ISLAM-PO",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/ISLAM-PO/MasryGPT-Chat-1.5B"
      },
      "dialects": [
        "egy"
      ],
      "size": "1.5B",
      "on_device": true,
      "year": 2026,
      "notes": "80,000 Egyptian terms • 2,500 steps • Loss 0.079 • 2.9GB 16-bit Merged English | [بالمصري](#-بالمصري---الوثائق-الاح",
      "base_model": [
        "qwen/qwen2.5-1.5b-instruct"
      ],
      "metrics": {
        "downloads": 909,
        "likes": 1,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "arab-dialects-20-countries-3m",
      "name": "arab dialects 20 countries 3m",
      "type": "dataset",
      "country": "INTL",
      "org": "ISLAM-PO",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ISLAM-PO/arab-dialects-20-countries-3m"
      },
      "dialects": [
        "egy",
        "lev",
        "mixed"
      ],
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "The 3M target figure is a raw-repository claim and is not yet fully verified by the Hub index.",
      "metrics": {
        "downloads": 886,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "aragpt2-medium",
      "name": "aragpt2-medium",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/aragpt2-medium"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "394M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraGPT2 medium, 370M Arabic GPT-2 from AUB.",
      "metrics": {
        "downloads": 882,
        "likes": 11,
        "lastModified": "2023-10-30"
      }
    },
    {
      "id": "asjp-cerist",
      "name": "asjp cerist",
      "type": "dataset",
      "country": "INTL",
      "org": "DarjaCore",
      "license": "other",
      "modality": "text",
      "tasks": [
        "retrieval",
        "ocr",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/DarjaCore/asjp-cerist"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "37,807 academic PDFs (579,683 pages) with 21,604 metadata records from Algeria's ASJP platform in Arabic, French and English.",
      "metrics": {
        "downloads": 869,
        "likes": 1,
        "lastModified": "2026-09-30"
      }
    },
    {
      "id": "arabic-mmlu-10percent",
      "name": "Arabic MMLU 10percent",
      "type": "benchmark",
      "country": "INTL",
      "org": "arcee-globe",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "mmlu",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arcee-globe/Arabic_MMLU-10percent"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "10 percent subset of Arabic MMLU multiple-choice questions across subjects.",
      "metrics": {
        "downloads": 850,
        "likes": 0,
        "lastModified": "2024-08-13"
      }
    },
    {
      "id": "audar-tts-v1-turbo",
      "name": "Audar TTS V1 Turbo",
      "type": "tts",
      "country": "INTL",
      "org": "Audar AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/audarai/Audar-TTS-V1-Turbo"
      },
      "year": 2026,
      "notes": "Open, Arabic-first, expressive zero-shot text-to-speech — the balanced production default.",
      "metrics": {
        "downloads": 850,
        "likes": 5,
        "lastModified": "2026-07-07"
      }
    },
    {
      "id": "arabert-all-nli-triplet-matryoshka",
      "name": "Arabert all nli triplet Matryoshka",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabert-all-nli-triplet-Matryoshka"
      },
      "year": 2025,
      "notes": "This is a sentence-transformers model finetuned from aubmindlab/bert-base-arabertv02 on the Omartificial-Intelligence-Space/arabic-nli-triplet dataset.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 846,
        "likes": 11,
        "lastModified": "2025-01-23"
      }
    },
    {
      "id": "arabic-img2md",
      "name": "arabic-img2md",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-img2md"
      },
      "notes": "15K PDF pages paired with Markdown",
      "metrics": {
        "downloads": 840,
        "likes": 14,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "arabic-tts-spark",
      "name": "Arabic-TTS-Spark",
      "type": "tts",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "fair-noncommercial-research-license",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimSalah/Arabic-TTS-Spark"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Spark TTS fine-tuned on 300h clean Arabic, MSA with diacritics",
      "base_model": [
        "sparkaudio/spark-tts-0.5b"
      ],
      "metrics": {
        "downloads": 830,
        "likes": 30,
        "lastModified": "2025-11-27"
      }
    },
    {
      "id": "arabic-xvector-embeddings",
      "name": "arabic xvector embeddings",
      "type": "dataset",
      "country": "INTL",
      "org": "herwoww",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "embedding",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/herwoww/arabic_xvector_embeddings"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "There is one speaker embedding for each utterance in the validation set of both datasets.",
      "metrics": {
        "downloads": 820,
        "likes": 5,
        "lastModified": "2024-05-13"
      }
    },
    {
      "id": "linto-dataset-audio-ar-tn-augmented",
      "name": "linto dataset audio ar tn augmented",
      "type": "dataset",
      "country": "INTL",
      "org": "LINAGORA",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/linagora/linto-dataset-audio-ar-tn-augmented"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This is the augmented datasets used to train the Linto Tunisian dialect with code-switching STT linagora/linto-asr-ar-tn.",
      "metrics": {
        "downloads": 817,
        "likes": 7,
        "lastModified": "2025-04-11"
      }
    },
    {
      "id": "karsl-502-arabic-sign-language-v2",
      "name": "karsl 502 arabic sign language v2",
      "type": "dataset",
      "country": "INTL",
      "org": "FatimahEmadEldin",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/karsl-502-arabic-sign-language-v2"
      },
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "KArSL-502 Arabic sign language image frames organized by signer and sign folders.",
      "metrics": {
        "downloads": 816,
        "likes": 1,
        "lastModified": "2026-01-20"
      }
    },
    {
      "id": "karnak-6b",
      "name": "Karnak-6B-v1.0",
      "type": "llm",
      "country": "EG",
      "org": "Applied Innovation Center (MCIT Egypt)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Applied-Innovation-Center/Karnak-6B-v1.0"
      },
      "size": "5.9B",
      "year": 2026,
      "dialects": [
        "msa",
        "egy"
      ],
      "notes": "6B small sibling of the Karnak national Egyptian LLM from the Applied Innovation Center, MCIT.",
      "base_model": [
        "qwen/qwen3-4b-instruct-2507"
      ],
      "metrics": {
        "downloads": 815,
        "likes": 7,
        "lastModified": "2026-05-11"
      }
    },
    {
      "id": "ayncoding-qwen3-8b-slim",
      "name": "ayncoding qwen3 8b slim",
      "type": "llm",
      "country": "INTL",
      "org": "enver",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/enver/ayncoding-qwen3-8b-slim"
      },
      "size": "8B",
      "on_device": false,
      "year": 2026,
      "notes": "Unlike standard models trained on unstructured, noisy repositories, AynCoding-Qwen3-8B-Slim actively purges illogical pre-",
      "base_model": [
        "qwen/qwen3-8b"
      ],
      "metrics": {
        "downloads": 807,
        "likes": 0,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "faster-whisper-base-ar-quran",
      "name": "faster whisper base ar quran",
      "type": "asr",
      "country": "INTL",
      "org": "OdyAsh",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/OdyAsh/faster-whisper-base-ar-quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This model is a CTranslate2 version of tarteel-ai/whisper-base-ar-quran.",
      "base_model": [
        "tarteel-ai/whisper-base-ar-quran",
        "openai/whisper-base"
      ],
      "metrics": {
        "downloads": 806,
        "likes": 6,
        "lastModified": "2025-06-24"
      }
    },
    {
      "id": "mr-tydi",
      "name": "Mr. TyDi",
      "type": "dataset",
      "country": "INTL",
      "org": "David R. Cheriton School of Computer Science",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "github": "https://github.com/castorini/mr.tydi",
        "hf": "https://huggingface.co/datasets/castorini/mr-tydi",
        "paper": "https://arxiv.org/pdf/2108.08787.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "16,573 sentences",
      "year": 2021,
      "tags": [
        "multilingual"
      ],
      "notes": "Multilingual retrieval benchmark built from TyDi QA across eleven languages including Arabic.",
      "metrics": {
        "downloads": 806,
        "likes": 23,
        "lastModified": "2022-10-12"
      }
    },
    {
      "id": "qwen2-5-7b-instruct-arabic-yt-merged",
      "name": "qwen2.5 7b instruct arabic yt merged",
      "type": "llm",
      "country": "INTL",
      "org": "HarithSami",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HarithSami/qwen2.5-7b-instruct-arabic-yt-merged"
      },
      "size": "7B",
      "on_device": false,
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "This qwen2 model was trained 2x faster with Unsloth and Huggingface's TRL library. (also: 1 variants)",
      "base_model": [
        "unsloth/qwen2.5-7b-instruct"
      ],
      "metrics": {
        "downloads": 796,
        "likes": 0,
        "lastModified": "2026-09-16"
      }
    },
    {
      "id": "arabert-arabic-sentiment-analysis",
      "name": "AraBert-Arabic-Sentiment-Analysis",
      "type": "llm",
      "country": "INTL",
      "org": "PRAli22",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/PRAli22/AraBert-Arabic-Sentiment-Analysis"
      },
      "year": 2024,
      "notes": "Arabic text classification model.",
      "metrics": {
        "downloads": 789,
        "likes": 5,
        "lastModified": "2024-03-13"
      }
    },
    {
      "id": "arvoice",
      "name": "ArVoice",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ArVoice"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Multi-speaker MSA corpus, 83h, 11 voices, diacritized",
      "metrics": {
        "downloads": 781,
        "likes": 33,
        "lastModified": "2025-10-31"
      }
    },
    {
      "id": "arabictext-large",
      "name": "ArabicText-Large",
      "type": "dataset",
      "country": "INTL",
      "org": "Jr23xd23",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Jr23xd23/ArabicText-Large"
      },
      "notes": "743K articles for LLM training",
      "metrics": {
        "downloads": 773,
        "likes": 69,
        "lastModified": "2025-10-27"
      }
    },
    {
      "id": "clartts",
      "name": "ClArTTS",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ClArTTS"
      },
      "dialects": [
        "classical"
      ],
      "notes": "Classical Arabic TTS corpus by MBZUAI",
      "metrics": {
        "downloads": 769,
        "likes": 27,
        "lastModified": "2025-10-01"
      }
    },
    {
      "id": "abjad-kids",
      "name": "Abjad Kids",
      "type": "dataset",
      "country": "INTL",
      "org": "Aziz-snoubra",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "speech",
        "education"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Aziz-snoubra/Abjad-Kids"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Abjad-Kids: Arabic speech classification dataset of children's recordings of letters, numbers and colors for primary education.",
      "metrics": {
        "downloads": 761,
        "likes": 1,
        "lastModified": "2026-03-14"
      }
    },
    {
      "id": "silma-embedding-matryoshka-v0-1",
      "name": "silma-embedding-matryoshka-v0.1",
      "type": "embedding",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/silma-ai/silma-embedding-matryoshka-v0.1"
      },
      "size": "135M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "SILMA Arabic and English Matryoshka embedding model.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 755,
        "likes": 14,
        "lastModified": "2025-05-15"
      }
    },
    {
      "id": "arad",
      "name": "ArAD",
      "type": "benchmark",
      "country": "INTL",
      "org": "SpeechAntiSpoofingBenchmarks",
      "license": "odc-by",
      "modality": "speech",
      "tasks": [
        "speech",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SpeechAntiSpoofingBenchmarks/ArAD"
      },
      "dialects": [
        "lev"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Benchmark-ready packaging of the test split of the Arabic Audio Deepfake (ArAD) dataset: binary anti-spoofing on Arabic (primarily Levantine dialect) speech.",
      "metrics": {
        "downloads": 749,
        "likes": 1,
        "lastModified": "2026-06-27"
      }
    },
    {
      "id": "bojji",
      "name": "bojji",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "static-embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/bojji"
      },
      "year": 2025,
      "notes": "Lightweight, fast Arabic static embedding model built with Model2Vec.",
      "on_device": true,
      "metrics": {
        "downloads": 745,
        "likes": 2,
        "lastModified": "2025-06-16"
      }
    },
    {
      "id": "camelbert-msa-sentiment",
      "name": "CAMeLBERT-MSA-Sentiment",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-msa-sentiment"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Sentiment Analysis - Fine-tuned for MSA sentiment",
      "metrics": {
        "downloads": 742,
        "likes": 8,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "mbert2mbert-arabic-text-summarization",
      "name": "mbert2mbert arabic text summarization",
      "type": "llm",
      "country": "INTL",
      "org": "malmarjeh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/malmarjeh/mbert2mbert-arabic-text-summarization"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "notes": "BERT2BERT abstractive summarizer initialized from mBERT, fine-tuned on 84,764 Arabic paragraph-summary pairs.",
      "metrics": {
        "downloads": 737,
        "likes": 12,
        "lastModified": "2023-07-01"
      }
    },
    {
      "id": "openiti-chunked",
      "name": "openiti chunked",
      "type": "dataset",
      "country": "INTL",
      "org": "mittagessen",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mittagessen/openiti_chunked"
      },
      "size": "10M–100M rows",
      "year": 2025,
      "notes": "This dataset is derived from the 2023.1.8 release of the OpenITI corpus and is intended to pretrain small language models with short context lengths.",
      "metrics": {
        "downloads": 737,
        "likes": 1,
        "lastModified": "2025-06-16"
      }
    },
    {
      "id": "mtrini-svl-1-1-merged",
      "name": "Mtrini SVL 1.1 Merged",
      "type": "llm",
      "country": "INTL",
      "org": "CompiwerAI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/CompiwerAI/Mtrini-SVL-1.1-Merged"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Mtrini-SVL-1.1 is an open-weight 8B model developed by Compiwer AI, based on Qwen3-VL-8B-Instruct.",
      "base_model": [
        "qwen/qwen3-vl-8b-instruct"
      ],
      "metrics": {
        "downloads": 735,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "ivis-400m-gpu",
      "name": "ivis 400m gpu",
      "type": "llm",
      "country": "INTL",
      "org": "nepetai",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/nepetai/ivis-400m-gpu"
      },
      "size": "451M",
      "on_device": true,
      "year": 2026,
      "notes": "Hybrid LM of about 451M parameters mixing GQA Transformer, Mamba SSM and MoE layers, Arabic and English.",
      "metrics": {
        "downloads": 734,
        "likes": 0,
        "lastModified": "2026-08-23"
      }
    },
    {
      "id": "whisper-medium-finetuned-sada-asr",
      "name": "whisper medium finetuned sada asr",
      "type": "asr",
      "country": "INTL",
      "org": "wageehkhad",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/wageehkhad/whisper-medium-finetuned-sada-asr"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "openai/whisper-medium fine-tuned on the full SADA22 dataset (420 hours of Saudi Arabic speech) for Arabic automatic speech recognition (ASR).",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 730,
        "likes": 3,
        "lastModified": "2026-03-05"
      }
    },
    {
      "id": "adi20",
      "name": "ADI-20",
      "type": "dataset",
      "country": "INTL",
      "org": "Elyadata",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabicSpeech/ADI20"
      },
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic dialect identification dataset and models extending ADI-17 to 20 countries (Tunisia/Elyadata).",
      "metrics": {
        "downloads": 727,
        "likes": 0,
        "lastModified": "2026-07-04"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-did-madar-corpus26",
      "name": "bert-base-arabic-camelbert-mix-did-madar-corpus26",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-did-madar-corpus26"
      },
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "CAMeLBERT-Mix fine-tuned for 26-city dialect ID on MADAR.",
      "metrics": {
        "downloads": 723,
        "likes": 4,
        "lastModified": "2025-10-27"
      }
    },
    {
      "id": "whisper-large-v3-turbo-arabic-dialectal",
      "name": "whisper large v3 turbo arabic dialectal",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/whisper-large-v3-turbo-arabic-dialectal"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "Fine-tune of openai/whisper-large-v3-turbo (809M) for multi-dialect Arabic speech recognition (undiacritized output).",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 723,
        "likes": 4,
        "lastModified": "2026-07-09"
      }
    },
    {
      "id": "arabict5-17gb-base",
      "name": "ArabicT5-17GB-base",
      "type": "llm",
      "country": "INTL",
      "org": "sultan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-to-text"
      ],
      "links": {
        "hf": "https://huggingface.co/sultan/ArabicT5-17GB-base"
      },
      "year": 2023,
      "tags": [
        "variants:4"
      ],
      "notes": "ArabicT5 pre-trained on Arabic Wikipedia, Marefa, Hindawi books and news (17GB).",
      "metrics": {
        "downloads": 716,
        "likes": 4,
        "lastModified": "2023-11-09"
      }
    },
    {
      "id": "tunisian-proverbs-with-image-associations-a-cultural-and-linguistic-da",
      "name": "Tunisian Proverbs with Image Associations A Cultural and Linguistic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "HabibaAbderrahim",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HabibaAbderrahim/Tunisian-Proverbs-with-Image-Associations-A-Cultural-and-Linguistic-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2025,
      "notes": "Tunisian proverbs with explanations, English translations and AI-generated image associations.",
      "metrics": {
        "downloads": 714,
        "likes": 0,
        "lastModified": "2025-09-02"
      }
    },
    {
      "id": "arabart-qalb14-gec-ged-13",
      "name": "arabart-qalb14-gec-ged-13",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "grammar-correction"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/arabart-qalb14-gec-ged-13"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "AraBART fine-tuned for Arabic grammatical error correction on QALB-2014.",
      "metrics": {
        "downloads": 705,
        "likes": 3,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "ara-df-2026",
      "name": "ArA-DF-2026",
      "type": "dataset",
      "country": "INTL",
      "org": "Elyadata",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabicSpeech/ArA-DF-2026"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "ArA-DF-2026 is an Arabic speech deepfake detection dataset for binary audio classification.",
      "metrics": {
        "downloads": 694,
        "likes": 1,
        "lastModified": "2026-08-04"
      }
    },
    {
      "id": "50m-2048-emhotob",
      "name": "50M 2048 Emhotob",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/50M-2048-Emhotob"
      },
      "size": "50M",
      "on_device": true,
      "year": 2026,
      "tags": [
        "variants:10"
      ],
      "notes": "from scratch on 20 billion Arabic tokens with a 2048-token context window. (also: 10 variants)",
      "metrics": {
        "downloads": 689,
        "likes": 7,
        "lastModified": "2026-07-14"
      }
    },
    {
      "id": "alexandria-dialect",
      "name": "Alexandria",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/alexandria"
      },
      "size": "107k samples",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Community-driven dialectal Arabic dataset covering 13 Arab countries and 11 domains.",
      "metrics": {
        "downloads": 688,
        "likes": 15,
        "lastModified": "2026-06-24"
      }
    },
    {
      "id": "gpt2-small-arabic",
      "name": "gpt2 small arabic",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/gpt2-small-arabic"
      },
      "year": 2023,
      "notes": "GPT2 model from Arabic Wikipedia dataset based on gpt2-small (using Fastai2).",
      "metrics": {
        "downloads": 685,
        "likes": 19,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "quran-tajweed-phonetics",
      "name": "quran tajweed phonetics",
      "type": "dataset",
      "country": "INTL",
      "org": "Quran Lab",
      "license": "other",
      "modality": "text",
      "tasks": [
        "asr",
        "tts",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Quran-Lab/quran-tajweed-phonetics"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The complete phonetic layer of the Quran in the riwaya of Hafs 'an 'Asim via tariq al-Shatibiyyah.",
      "metrics": {
        "downloads": 684,
        "likes": 4,
        "lastModified": "2026-10-01"
      }
    },
    {
      "id": "fahadprimex-b27-v3",
      "name": "FahadPrimeX-b27-V3",
      "type": "llm",
      "country": "INTL",
      "org": "FahadPrimeX",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/FahadPrimeX/FahadPrimeX-b27-V3"
      },
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Base model: Qwen/Qwen3.8-27B (declared with thanks). (also: 1 variants)",
      "metrics": {
        "downloads": 678,
        "likes": 0,
        "lastModified": "2026-09-21"
      }
    },
    {
      "id": "arabic-speech-to-text",
      "name": "arabic speech to text",
      "type": "asr",
      "country": "INTL",
      "org": "maherghanem86",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/maherghanem86/arabic_speech_to_text"
      },
      "year": 2025,
      "notes": "تم تطوير وتدريب هذا النموذج للتعرف التلقائي على الكلام للغة العربية.",
      "metrics": {
        "downloads": 677,
        "likes": 1,
        "lastModified": "2025-09-07"
      }
    },
    {
      "id": "quranic-asr-benchmark",
      "name": "Quranic ASR Benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "Quran Lab",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Quran-Lab/quranic-asr-benchmark"
      },
      "size": "under 1K clips",
      "year": 2026,
      "dialects": [
        "classical"
      ],
      "notes": "Small benchmark set for evaluating ASR models on Quran recitation.",
      "metrics": {
        "downloads": 677,
        "likes": 5,
        "lastModified": "2026-08-26"
      }
    },
    {
      "id": "mmcqa-semeval27",
      "name": "MMCQA-SemEval27",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-4.0",
      "modality": "multimodal",
      "tasks": [
        "qa",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/MMCQA-SemEval27"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This is the dataset for MMCultureQA, the SemEval 2027 shared task on culturally grounded visual question answering in English and Arabic.",
      "metrics": {
        "downloads": 675,
        "likes": 12,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "egyptian-arabic-asr-clean",
      "name": "Egyptian Arabic ASR Clean",
      "type": "dataset",
      "country": "INTL",
      "org": "MAdel121",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MAdel121/arabic-egy-cleaned"
      },
      "dialects": [
        "egy"
      ],
      "notes": "~72 hours of Egyptian Arabic speech",
      "metrics": {
        "downloads": 668,
        "likes": 18,
        "lastModified": "2025-05-04"
      }
    },
    {
      "id": "alrage",
      "name": "ALRAGE",
      "type": "benchmark",
      "country": "INTL",
      "org": "Open Arabic LLM Leaderboard (OALL)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "rag"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OALL/ALRAGE"
      },
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic retrieval-augmented generation evaluation set used in OALL v2.",
      "metrics": {
        "downloads": 667,
        "likes": 4,
        "lastModified": "2024-11-14"
      }
    },
    {
      "id": "arabic-labse-matryoshka",
      "name": "Arabic-labse-Matryoshka",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-labse-Matryoshka"
      },
      "size": "471M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic Matryoshka sentence embeddings based on LaBSE.",
      "base_model": [
        "sentence-transformers/labse"
      ],
      "metrics": {
        "downloads": 667,
        "likes": 5,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "darija-asr-corpus",
      "name": "darija asr corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "abnajlae",
      "license": "['mit', 'cc-by-4.0', 'cc-by-sa-4.0']",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abnajlae/darija-asr-corpus"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabizi (Latin-script) transcriptions of Moroccan Darija speech, produced for a Whisper fine-tuning pipeline (paper not yet published -- citation forthcoming).",
      "metrics": {
        "downloads": 663,
        "likes": 0,
        "lastModified": "2026-09-07"
      }
    },
    {
      "id": "m2cqa",
      "name": "M2CQA",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "vision",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/M2CQA"
      },
      "dialects": [
        "egy",
        "lev",
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This release contains English, Modern Standard Arabic, Levantine Arabic, and Egyptian Arabic statements over a shared set of images.",
      "metrics": {
        "downloads": 662,
        "likes": 2,
        "lastModified": "2026-06-12"
      }
    },
    {
      "id": "vynis-0-1-2b-instant",
      "name": "Vynis 0.1 2b instant",
      "type": "llm",
      "country": "INTL",
      "org": "OpenWeightsAI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/OpenWeightsAI/Vynis-0.1-2b-instant"
      },
      "on_device": true,
      "year": 2026,
      "notes": "Arabic-English instruction-tuned chat model fine-tuned from Llama-3.2-1B-Instruct on alpaca-cleaned.",
      "base_model": [
        "unsloth/llama-3.2-1b-instruct"
      ],
      "metrics": {
        "downloads": 661,
        "likes": 0,
        "lastModified": "2026-09-15"
      }
    },
    {
      "id": "aratrust",
      "name": "AraTrust",
      "type": "benchmark",
      "country": "INTL",
      "org": "Asas AI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation",
        "safety"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/asas-ai/AraTrust"
      },
      "size": "522 questions",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic LLM trustworthiness benchmark across truthfulness, ethics, safety and privacy.",
      "metrics": {
        "downloads": 653,
        "likes": 4,
        "lastModified": "2024-05-07"
      }
    },
    {
      "id": "darijabert",
      "name": "DarijaBERT",
      "type": "llm",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/SI2M-Lab/DarijaBERT"
      },
      "size": "209M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "DarijaBERT, first BERT model for Moroccan Darija.",
      "metrics": {
        "downloads": 643,
        "likes": 37,
        "lastModified": "2024-09-25"
      }
    },
    {
      "id": "arabic-openhermes-2-5",
      "name": "Arabic-OpenHermes-2.5",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Arabic-OpenHermes-2.5"
      },
      "notes": "Arabic OpenHermes instruction dataset",
      "metrics": {
        "downloads": 641,
        "likes": 21,
        "lastModified": "2024-03-15"
      }
    },
    {
      "id": "arabic-sentiment-twitter-corpus-arbml",
      "name": "Arabic Sentiment Twitter Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Sentiment_Twitter_Corpus"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "58.7K Arabic tweets labelled for sentiment.",
      "metrics": {
        "downloads": 637,
        "likes": 2,
        "lastModified": "2024-03-30"
      }
    },
    {
      "id": "arabic-oscar",
      "name": "Arabic OSCAR",
      "type": "dataset",
      "country": "INTL",
      "org": "Inria",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://oscar-corpus.com/",
        "hf": "https://huggingface.co/datasets/oscar-corpus/oscar",
        "paper": "https://arxiv.org/pdf/2006.06202.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "8,117,162,828 tokens",
      "year": 2020,
      "notes": "A huge multilingual corpus obtained by language classification and filtering of the Common Crawl",
      "metrics": {
        "downloads": 632,
        "likes": 208,
        "lastModified": "2025-09-04"
      }
    },
    {
      "id": "nile-chat-4b",
      "name": "Nile-Chat-4B",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris/Nile-Chat-4B"
      },
      "base_model": [
        "google/gemma-3-4b-pt"
      ],
      "size": "3.9B",
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "Nile-Chat 4B, Egyptian dialect model handling both Arabic and Arabizi script.",
      "metrics": {
        "downloads": 617,
        "likes": 20,
        "lastModified": "2025-07-10"
      }
    },
    {
      "id": "fikr-7b-reasoning",
      "name": "Fikr 7B Reasoning",
      "type": "llm",
      "country": "INTL",
      "org": "Hatim2221",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Hatim2221/Fikr-7B-Reasoning"
      },
      "size": "7B",
      "on_device": false,
      "year": 2026,
      "notes": "Qwen2.5-7B-Instruct fine-tuned for structured chain-of-thought Arabic reasoning, evaluated on Arabic-GSM8K.",
      "base_model": [
        "qwen/qwen2.5-7b-instruct"
      ],
      "metrics": {
        "downloads": 600,
        "likes": 1,
        "lastModified": "2026-08-29"
      }
    },
    {
      "id": "livekit-turn-detector-arabic",
      "name": "livekit turn detector arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Moustafa3092",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Moustafa3092/livekit-turn-detector-arabic"
      },
      "year": 2025,
      "notes": "Fine-tuned Arabic End-of-Utterance (EOU) detection model for LiveKit voice agents.",
      "base_model": [
        "livekit/turn-detector"
      ],
      "metrics": {
        "downloads": 597,
        "likes": 3,
        "lastModified": "2025-12-14"
      }
    },
    {
      "id": "aradice-arabicmmlu-egy",
      "name": "AraDICE ArabicMMLU Egyptian",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AraDICE-ArabicMMLU-egy"
      },
      "year": 2024,
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian-dialect translation of ArabicMMLU from the QCRI AraDiCE benchmark suite.",
      "metrics": {
        "downloads": 591,
        "likes": 1,
        "lastModified": "2024-11-08"
      }
    },
    {
      "id": "jev-ar-bench",
      "name": "jev ar bench",
      "type": "benchmark",
      "country": "INTL",
      "org": "atmaneayoub",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atmaneayoub/jev-ar-bench"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Arabic intent-routing benchmarks for MSA, Emirati, Saudi and code-switched Arabic",
      "metrics": {
        "downloads": 591,
        "likes": 1,
        "lastModified": "2026-10-01"
      }
    },
    {
      "id": "arabic-dialects-gold20-code-switch",
      "name": "arabic dialects gold20 code switch",
      "type": "dataset",
      "country": "INTL",
      "org": "TigreGotico",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TigreGotico/arabic-dialects-gold20-code-switch"
      },
      "dialects": [
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "Code-switched Arabic sentences with IPA: 20 rows per lect across 33 Arabic lects (the same roster as the sibling TigreGotico/arabic-dialects-gold20).",
      "metrics": {
        "downloads": 582,
        "likes": 0,
        "lastModified": "2026-07-20"
      }
    },
    {
      "id": "spark-tts-normazlied-masri-mega",
      "name": "spark tts normazlied masri mega",
      "type": "llm",
      "country": "EG",
      "org": "Mohamed Gomaa",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "tts",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedGomaa30/spark-tts-normazlied-masri-mega"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "This model is a fine-tuned version of MohamedGomaa30/spark-tts-normazlied-masri-mega.",
      "base_model": [
        "mohamedgomaa30/spark-tts-normazlied-masri-mega"
      ],
      "metrics": {
        "downloads": 581,
        "likes": 0,
        "lastModified": "2026-02-03"
      }
    },
    {
      "id": "arabic-ner-pii2",
      "name": "Arabic NER PII2",
      "type": "llm",
      "country": "INTL",
      "org": "MutazYoune",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/MutazYoune/Arabic-NER-PII2"
      },
      "year": 2025,
      "notes": "This is an Arabic Named Entity Recognition (NER) model fine-tuned on BERT architecture specifically for Arabic text processing.",
      "metrics": {
        "downloads": 579,
        "likes": 1,
        "lastModified": "2025-06-12"
      }
    },
    {
      "id": "iraqi-sales-dialogue",
      "name": "Iraqi Arabic Sales Dialogue",
      "type": "dataset",
      "country": "IQ",
      "org": "Ameer Wisam",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ameer4wisam/iraqi-arabic-sales-dialogue-dataset"
      },
      "dialects": [
        "iraqi"
      ],
      "notes": "Iraqi Arabic sales conversation dataset (Iraq).",
      "metrics": {
        "downloads": 577,
        "likes": 0,
        "lastModified": "2026-07-30"
      }
    },
    {
      "id": "whisper-large-v3-egyptian-arabic",
      "name": "whisper-large-v3-egyptian-arabic",
      "type": "asr",
      "country": "EG",
      "org": "Abdelrahman Hassan",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/AbdelrahmanHassan/whisper-large-v3-egyptian-arabic"
      },
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "Whisper large-v3 fine-tuned on Egyptian Arabic.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 575,
        "likes": 12,
        "lastModified": "2025-08-04"
      }
    },
    {
      "id": "atlas-chat-9b",
      "name": "Atlas-Chat-9B",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris/Atlas-Chat-9B"
      },
      "base_model": [
        "google/gemma-2-9b-it"
      ],
      "size": "9.2B",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Atlas-Chat 9B, Gemma-2 based instruction model for Moroccan Darija.",
      "metrics": {
        "downloads": 571,
        "likes": 30,
        "lastModified": "2024-10-24"
      }
    },
    {
      "id": "covost-2",
      "name": "CoVoST 2",
      "type": "dataset",
      "country": "INTL",
      "org": "Facebook AI",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "github": "https://github.com/facebookresearch/covost",
        "hf": "https://huggingface.co/datasets/facebook/covost2",
        "paper": "https://arxiv.org/pdf/2007.10310.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6 hours",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "Multilingual speech translation corpus based on Common Voice, with Arabic among its languages.",
      "metrics": {
        "downloads": 567,
        "likes": 51,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "fa-en-ar-handwritten-ocr-v1",
      "name": "fa en ar handwritten ocr v1",
      "type": "benchmark",
      "country": "INTL",
      "org": "saeid1999",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "asr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/saeid1999/fa-en-ar-handwritten-ocr-v1"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "A large, clean, augmentation-rich synthetic handwriting dataset for training and benchmarking OCR / HTR models on Persian (fa), Arabic (ar) and English (en).",
      "metrics": {
        "downloads": 565,
        "likes": 0,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "badr-embedding-v0",
      "name": "Badr embedding v0",
      "type": "embedding",
      "country": "INTL",
      "org": "somayaeltanbouly",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "retrieval",
        "fiqh"
      ],
      "links": {
        "hf": "https://huggingface.co/somayaeltanbouly/Badr_embedding_v0"
      },
      "year": 2026,
      "notes": "Arabic dense-retrieval embedding model for fiqh (Islamic jurisprudence), fine-tuned from Muffakir_Embedding.",
      "base_model": [
        "mohamed2811/muffakir_embedding"
      ],
      "metrics": {
        "downloads": 556,
        "likes": 0,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "qween7-5-arabic-story-teller-2",
      "name": "qween7.5 arabic story teller 2",
      "type": "llm",
      "country": "INTL",
      "org": "Raido",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "story-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/Raido/qween7.5-arabic-story-teller-2"
      },
      "year": 2025,
      "notes": "Qwen2.5-7B-Instruct fine-tuned with Unsloth as an Arabic story-telling model.",
      "size": "7B",
      "base_model": [
        "unsloth/qwen2.5-7b-instruct-bnb-4bit"
      ],
      "metrics": {
        "downloads": 556,
        "likes": 0,
        "lastModified": "2025-01-07"
      }
    },
    {
      "id": "arabicbert-arabic-dialect-identification",
      "name": "arabicBert_arabic_dialect_identification",
      "type": "llm",
      "country": "INTL",
      "org": "lafifi-24",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "dialect-id",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/lafifi-24/arabicBert_arabic_dialect_identification"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "notes": "Our Arabic Dialect Identification models are trained to accurately identify spoken dialects in Arabic text.",
      "metrics": {
        "downloads": 555,
        "likes": 0,
        "lastModified": "2023-04-30"
      }
    },
    {
      "id": "voho-saudi-chat-4b",
      "name": "voho-saudi-chat-4b",
      "type": "llm",
      "country": "SA",
      "org": "Voho AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/VohoAI/voho-saudi-chat-4b"
      },
      "size": "4B",
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Voho Saudi chat 4B, Saudi-dialect conversational model.",
      "base_model": [
        "qwen/qwen3-4b-instruct-2507"
      ],
      "metrics": {
        "downloads": 555,
        "likes": 1,
        "lastModified": "2026-09-19"
      }
    },
    {
      "id": "lahgtna-omnivoice-v2",
      "name": "Lahgtna OmniVoice v2",
      "type": "tts",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/lahgtna-omnivoice-v2"
      },
      "size": "613M",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Multi-dialect Arabic TTS from the Lahgtna project, trained on dialectal speech.",
      "base_model": [
        "k2-fsa/omnivoice"
      ],
      "metrics": {
        "downloads": 554,
        "likes": 13,
        "lastModified": "2026-06-15"
      }
    },
    {
      "id": "alm-bench",
      "name": "ALM-Bench",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "multimodal",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ALM-Bench"
      },
      "year": 2024,
      "notes": "All Languages Matter multimodal cultural benchmark covering 100 languages including Arabic.",
      "metrics": {
        "downloads": 553,
        "likes": 20,
        "lastModified": "2025-02-28"
      }
    },
    {
      "id": "sfc-mini",
      "name": "SFC-mini",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/faisalq/SFC-mini"
      },
      "year": 2024,
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi forums corpus subset used to pretrain SaudiBERT.",
      "metrics": {
        "downloads": 551,
        "likes": 2,
        "lastModified": "2024-05-08"
      }
    },
    {
      "id": "qwen3-4b-algerian-darja",
      "name": "Qwen3 4B Algerian Darja",
      "type": "llm",
      "country": "INTL",
      "org": "DarjaCore",
      "license": "['apache-2.0']",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/DarjaCore/Qwen3-4B-Algerian-Darja"
      },
      "base_model": [
        "qwen/qwen3-4b-base"
      ],
      "dialects": [
        "magh"
      ],
      "size": "4B",
      "on_device": false,
      "year": 2026,
      "notes": "This model is an experimental Algerian Darja adaptation of Qwen/Qwen3-4B-Base.",
      "metrics": {
        "downloads": 548,
        "likes": 1,
        "lastModified": "2026-09-10"
      }
    },
    {
      "id": "dclm-pro-arabic",
      "name": "dclm pro arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/dclm-pro-arabic"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "Arabic translation of DCLM-Pro (global shards 01 and 05), translated with Seed-X-PPO-7B using greedy decoding.",
      "metrics": {
        "downloads": 547,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "madlad-arabic-clean",
      "name": "MADLAD Arabic Clean",
      "type": "dataset",
      "country": "INTL",
      "org": "ymoslem",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ymoslem/MADLAD-Arabic-Clean"
      },
      "size": "10M–100M rows",
      "year": 2024,
      "notes": "Cleaned Arabic subset of MADLAD-400 with 12.4M documents (about 77 GB of text).",
      "metrics": {
        "downloads": 547,
        "likes": 0,
        "lastModified": "2024-07-11"
      }
    },
    {
      "id": "jais-family-chat",
      "name": "Jais family chat models",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-family-13b-chat"
      },
      "base_model": [
        "inceptionai/jais-family-13b"
      ],
      "size": "590M-30B",
      "year": 2024,
      "notes": "Jais family of bilingual Arabic-English chat models from 590M to 30B trained from scratch.",
      "metrics": {
        "downloads": 541,
        "likes": 10,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "evarest",
      "name": "EvArEST",
      "type": "dataset",
      "country": "INTL",
      "org": "Melaraby",
      "license": "bsd-3-clause",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Melaraby/EvArEST-dataset-for-Arabic-scene-text-recognition"
      },
      "year": 2022,
      "dialects": [
        "msa"
      ],
      "notes": "Everyday Arabic-English scene text recognition dataset.",
      "metrics": {
        "downloads": 539,
        "likes": 2,
        "lastModified": "2025-11-09"
      }
    },
    {
      "id": "ayncoding-qwen2-5-coder-1-5b",
      "name": "ayncoding qwen2.5 coder 1.5b",
      "type": "llm",
      "country": "INTL",
      "org": "enver",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/enver/ayncoding-qwen2.5-coder-1.5b"
      },
      "size": "1.5B",
      "on_device": true,
      "year": 2026,
      "notes": "It serves as the ultra-fast Drafter model within the Hierarchical Symbolic-Neural MoE (H-MoE) architecture, generating high-velocity syntax.",
      "base_model": [
        "qwen/qwen2.5-coder-1.5b"
      ],
      "metrics": {
        "downloads": 535,
        "likes": 0,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "synthetic-bilingual-invoices-200",
      "name": "synthetic bilingual invoices 200",
      "type": "dataset",
      "country": "INTL",
      "org": "HV09",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HV09/synthetic-bilingual-invoices-200"
      },
      "size": "<1K rows",
      "year": 2026,
      "notes": "200 rendered invoice images and a matching 17-field ground-truth record for every one.",
      "metrics": {
        "downloads": 535,
        "likes": 0,
        "lastModified": "2026-08-01"
      }
    },
    {
      "id": "idiomx",
      "name": "IdiomX",
      "type": "benchmark",
      "country": "INTL",
      "org": "aymansharara",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/aymansharara/IdiomX"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Dataset: 190K+ examples • 12K+ idioms • 3 languages • 4 benchmark tasks It supports four benchmark tasks: 1.",
      "metrics": {
        "downloads": 534,
        "likes": 2,
        "lastModified": "2026-06-05"
      }
    },
    {
      "id": "muharaf-public",
      "name": "muharaf public",
      "type": "dataset",
      "country": "INTL",
      "org": "Muharaf Project",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/aamijar/muharaf-public"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This dataset contains 24,495 line images of Arabic handwriting and the corresponding text.",
      "metrics": {
        "downloads": 534,
        "likes": 21,
        "lastModified": "2025-01-24"
      }
    },
    {
      "id": "arabic-quran-asr-dataset",
      "name": "Arabic Quran ASR dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Sabri12blm",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Sabri12blm/Arabic-Quran-ASR-dataset"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Quranic recitation ASR dataset of 74,832 audio-transcript pairs.",
      "metrics": {
        "downloads": 531,
        "likes": 5,
        "lastModified": "2025-02-19"
      }
    },
    {
      "id": "qari-markdown-mixed",
      "name": "QARI Markdown Mixed Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/QariOCR-v0.3-markdown-mixed-dataset",
        "paper": "https://arxiv.org/abs/2506.02295"
      },
      "size": "10K-100K rows",
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Training data for Qari OCR: Arabic page images paired with markdown including diacritics.",
      "metrics": {
        "downloads": 530,
        "likes": 13,
        "lastModified": "2025-06-10"
      }
    },
    {
      "id": "hifzguide-muaalem-mini",
      "name": "hifzguide muaalem mini",
      "type": "asr",
      "country": "INTL",
      "org": "sysofwan",
      "license": "agpl-3.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/sysofwan/hifzguide-muaalem-mini"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A size-distilled, teacher-initialised student of the Muaalem Quran phoneme-recognition model.",
      "base_model": [
        "obadx/muaalem-model-v3_2"
      ],
      "metrics": {
        "downloads": 527,
        "likes": 0,
        "lastModified": "2026-09-22"
      }
    },
    {
      "id": "dr-ai-v2",
      "name": "DR AI V2",
      "type": "llm",
      "country": "INTL",
      "org": "ehab215",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/ehab215/DR-AI-V2"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "A self-contained 4B conversational medical assistant for Egyptian Arabic and English.",
      "base_model": [
        "google/medgemma-4b-it"
      ],
      "metrics": {
        "downloads": 519,
        "likes": 0,
        "lastModified": "2026-09-26"
      }
    },
    {
      "id": "king-saud-university",
      "name": "King Saud University",
      "type": "org",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/faisalq/SaudiBERT"
      },
      "notes": "SaudiBERT, Saudi dialect corpora (STMC, SFC)",
      "metrics": {
        "downloads": 518,
        "likes": 21,
        "lastModified": "2024-05-23"
      }
    },
    {
      "id": "mmlu-arabic-fi",
      "name": "MMLU Arabic (FreedomIntelligence)",
      "type": "benchmark",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/MMLU_Arabic"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "MMLU translated into Arabic by gpt-3.5-turbo.",
      "metrics": {
        "downloads": 518,
        "likes": 1,
        "lastModified": "2023-08-06"
      }
    },
    {
      "id": "saudibert",
      "name": "SaudiBERT",
      "type": "llm",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/faisalq/SaudiBERT"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi dialect BERT, trained on 141M tweets (STMC) + forums",
      "metrics": {
        "downloads": 518,
        "likes": 21,
        "lastModified": "2024-05-23"
      }
    },
    {
      "id": "mkqa",
      "name": "MKQA",
      "type": "dataset",
      "country": "INTL",
      "org": "Apple",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/apple/ml-mkqa",
        "hf": "https://huggingface.co/datasets/apple/mkqa",
        "paper": "https://doi.org/10.1162/tacl_a_00433"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "10k question-answer pairs aligned across 26 languages, including Arabic.",
      "metrics": {
        "downloads": 516,
        "likes": 42,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "noor-platform-hadith",
      "name": "noor platform hadith",
      "type": "dataset",
      "country": "INTL",
      "org": "hozifa1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hozifa1/noor-platform-hadith"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "JSON hadith books and chapters (e.g., Abu Dawud) from the Noor platform.",
      "metrics": {
        "downloads": 515,
        "likes": 0,
        "lastModified": "2026-09-03"
      }
    },
    {
      "id": "rightnow-arabic-0-5b-turbo",
      "name": "RightNow-Arabic-0.5B-Turbo",
      "type": "llm",
      "country": "INTL",
      "org": "RightNowAI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/RightNowAI/RightNow-Arabic-0.5B-Turbo"
      },
      "size": "0.5B",
      "on_device": true,
      "year": 2026,
      "notes": "RightNow-Arabic-0.5B-Turbo is a 518M-parameter Arabic-specialized language model built on top of Qwen2.5-0.5B via vocabulary injection.",
      "base_model": [
        "qwen/qwen2.5-0.5b"
      ],
      "metrics": {
        "downloads": 513,
        "likes": 6,
        "lastModified": "2026-04-10"
      }
    },
    {
      "id": "lahgtna-levantine-tts",
      "name": "Lahgtna Levantine TTS",
      "type": "dataset",
      "country": "EG",
      "org": "Mohammed Aly",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohammedaly22/lahgtna-levantine-tts"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Levantine Arabic TTS dataset from the Lahgtna project.",
      "metrics": {
        "downloads": 512,
        "likes": 4,
        "lastModified": "2026-05-30"
      }
    },
    {
      "id": "alghafa-translated",
      "name": "AlGhafa Arabic LLM Benchmark (Translated)",
      "type": "benchmark",
      "country": "INTL",
      "org": "Open Arabic LLM Leaderboard (OALL)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OALL/AlGhafa-Arabic-LLM-Benchmark-Translated"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Machine-translated companion tasks to AlGhafa native, used in OALL v1.",
      "metrics": {
        "downloads": 511,
        "likes": 2,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "habibi-tts-doda-darija",
      "name": "habibi-tts-doda-darija",
      "type": "tts",
      "country": "MA",
      "org": "Jip7e",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/Jip7e/habibi-tts-doda-darija"
      },
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Habibi TTS voice for Moroccan Darija.",
      "base_model": [
        "swivid/habibi-tts"
      ],
      "metrics": {
        "downloads": 510,
        "likes": 1,
        "lastModified": "2026-08-29"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-did-nadi",
      "name": "bert base arabic camelbert mix did nadi",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "dialect-id",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-did-nadi"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "notes": "For the fine-tuning, we used the NADI Coountry-level dataset, which includes 21 labels.",
      "metrics": {
        "downloads": 505,
        "likes": 0,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "hadith-datasets",
      "name": "hadith datasets",
      "type": "dataset",
      "country": "INTL",
      "org": "meeAtif",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/meeAtif/hadith_datasets"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "An open-source collection of authenticated Hadiths from the six major books of Sunnah, available in both JSON and CSV formats for research, study.",
      "metrics": {
        "downloads": 505,
        "likes": 14,
        "lastModified": "2026-02-09"
      }
    },
    {
      "id": "islamic-scholars-prosopography",
      "name": "islamic scholars prosopography",
      "type": "dataset",
      "country": "INTL",
      "org": "hozifa1",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hozifa1/islamic-scholars-prosopography"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "مستودع موحد مفتوح المصدر يضم أضخم قواعد البيانات التراجمية الأكاديمية المهيكلة لعلماء وأعلام العالم الإسلامي وحواضر الأندلس والمشرق والمغرب، والمستخرجة من.",
      "metrics": {
        "downloads": 501,
        "likes": 0,
        "lastModified": "2026-09-07"
      }
    },
    {
      "id": "arabic-pp-ocrv3-mobile-rec",
      "name": "arabic PP OCRv3 mobile rec",
      "type": "ocr",
      "country": "INTL",
      "org": "PaddlePaddle",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/PaddlePaddle/arabic_PP-OCRv3_mobile_rec"
      },
      "year": 2025,
      "notes": "arabicPP-OCRv3mobilerec is a text line recognition model within the PP-OCRv3rec series, developed by the PaddleOCR team.",
      "metrics": {
        "downloads": 500,
        "likes": 4,
        "lastModified": "2025-07-22"
      }
    },
    {
      "id": "documents-egyptian-arabic",
      "name": "documents Egyptian Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "ISLAM-PO",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "asr",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ISLAM-PO/documents-Egyptian-Arabic"
      },
      "dialects": [
        "egy"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "Source-specific configurations are explicitly declared in the dataset metadata; use the config that matches the schema",
      "metrics": {
        "downloads": 500,
        "likes": 2,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "qwen3-vl-2b-persian-arabic-ocr-v1-0",
      "name": "Qwen3-VL-2B-Persian-Arabic-Ocr-v1.0",
      "type": "ocr",
      "country": "INTL",
      "org": "Mohajesmaeili",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/mohajesmaeili/Qwen3-VL-2B-Persian-Arabic-Ocr-v1.0"
      },
      "size": "2.1B",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Qwen3-VL 2B OCR for Persian and Arabic scripts.",
      "base_model": [
        "qwen/qwen3-vl-2b-instruct"
      ],
      "metrics": {
        "downloads": 500,
        "likes": 20,
        "lastModified": "2025-12-23"
      }
    },
    {
      "id": "masc-massive-arabic-speech-corpus",
      "name": "MASC: Massive Arabic Speech Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Appswave",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://ieee-dataport.org/open-access/masc-massive-arabic-speech-corpus",
        "hf": "https://huggingface.co/datasets/pain/MASC",
        "paper": "https://ieeexplore.ieee.org/abstract/document/10022652"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,000 hours",
      "year": 2022,
      "notes": "This corpus is a dataset that contains 1,000 hours of speech sampled at 16kHz and crawled from over 700 YouTube channels.",
      "metrics": {
        "downloads": 498,
        "likes": 10,
        "lastModified": "2023-06-12"
      }
    },
    {
      "id": "covid-19-disinformation",
      "name": "COVID 19 disinformation",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/COVID-19-disinformation"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "This repository contains a multilingual dataset related to the COVID-19 infodemic, annotated with fine-grained labels.",
      "metrics": {
        "downloads": 495,
        "likes": 1,
        "lastModified": "2024-09-09"
      }
    },
    {
      "id": "common-voice-18-arabic",
      "name": "common voice 18 arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "cc",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/common-voice-18-arabic"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset is an unofficial Arabic-only extraction of Mozilla Common Voice Corpus 18.0, prepared for Automatic Speech Recognition.",
      "metrics": {
        "downloads": 488,
        "likes": 5,
        "lastModified": "2025-12-27"
      }
    },
    {
      "id": "arabic-legal-documents-ocr-1-0",
      "name": "arabic-legal-documents-ocr-1.0",
      "type": "ocr",
      "country": "EG",
      "org": "Bakrianoo",
      "license": "gemma",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/bakrianoo/arabic-legal-documents-ocr-1.0"
      },
      "size": "4.3B",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Gemma-based OCR for Arabic legal documents.",
      "base_model": [
        "google/gemma-3-4b-it"
      ],
      "metrics": {
        "downloads": 486,
        "likes": 48,
        "lastModified": "2026-02-04"
      }
    },
    {
      "id": "zipformer-p-arabic-v3",
      "name": "zipformer_p-arabic-v3",
      "type": "asr",
      "country": "INTL",
      "org": "Quran Lab",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Quran-Lab/zipformer_p-arabic-v3"
      },
      "year": 2026,
      "dialects": [
        "classical"
      ],
      "notes": "Zipformer ASR for Quranic Arabic.",
      "metrics": {
        "downloads": 474,
        "likes": 52,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "cohere-transcribe-arabic-07-2026-dialectal-v2",
      "name": "cohere-transcribe-arabic-07-2026-dialectal-v2",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/cohere-transcribe-arabic-07-2026-dialectal-v2"
      },
      "size": "2.1B",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Cohere Transcribe Arabic fine-tuned for dialectal speech.",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 473,
        "likes": 0,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "spokennativqa",
      "name": "SpokenNativQA",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "qa",
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/SpokenNativQA"
      },
      "size": "10K-100K rows",
      "year": 2025,
      "dialects": [
        "mixed"
      ],
      "notes": "Spoken everyday queries with manually curated answers in multiple languages including Arabic.",
      "metrics": {
        "downloads": 473,
        "likes": 3,
        "lastModified": "2025-05-29"
      }
    },
    {
      "id": "nile-chat-12b",
      "name": "Nile-Chat-12B",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris/Nile-Chat-12B"
      },
      "base_model": [
        "google/gemma-3-12b-pt"
      ],
      "size": "11.8B",
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "Nile-Chat 12B, Egyptian dialect model handling both Arabic and Arabizi script.",
      "metrics": {
        "downloads": 472,
        "likes": 15,
        "lastModified": "2025-07-10"
      }
    },
    {
      "id": "baseer-nakba",
      "name": "Baseer Nakba",
      "type": "ocr",
      "country": "SA",
      "org": "Misraj AI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/Misraj/Baseer__Nakba",
        "paper": "https://arxiv.org/abs/2509.18174"
      },
      "year": 2026,
      "notes": "Merged Baseer Arabic document OCR VLM variant on Qwen2.5-VL.",
      "base_model": [
        "qwen2.5vl"
      ],
      "metrics": {
        "downloads": 469,
        "likes": 7,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "openthoughts3-en-ar-midtrain",
      "name": "openthoughts3 en ar midtrain",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/openthoughts3-en-ar-midtrain"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic translation of the OpenThoughts31.2M split of smoltalk2 (config Mid).",
      "metrics": {
        "downloads": 466,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "nemotron-r1-en-ar-midtrain",
      "name": "nemotron r1 en ar midtrain",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/nemotron-r1-en-ar-midtrain"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic translation of the LlamaNemotronPostTrainingDatasetreasoningr1 split of smoltalk2 (config Mid, pinned revision fc6cc21).",
      "metrics": {
        "downloads": 463,
        "likes": 1,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "arabic-mpnet-base-all-nli-triplet",
      "name": "Arabic mpnet base all nli triplet",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-mpnet-base-all-nli-triplet"
      },
      "year": 2025,
      "notes": "This is a sentence-transformers model finetuned from tomaarsen/mpnet-base-all-nli-triplet on the Omartificial-Intelligence-Space/arabic-nli-triplet dataset.",
      "base_model": [
        "tomaarsen/mpnet-base-all-nli-triplet"
      ],
      "metrics": {
        "downloads": 462,
        "likes": 10,
        "lastModified": "2025-01-23"
      }
    },
    {
      "id": "cidar",
      "name": "CIDAR",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/CIDAR"
      },
      "notes": "Culturally relevant instruction dataset (10K pairs)",
      "metrics": {
        "downloads": 460,
        "likes": 57,
        "lastModified": "2025-07-01"
      }
    },
    {
      "id": "ar-sarcasm",
      "name": "ar sarcasm",
      "type": "dataset",
      "country": "INTL",
      "org": "iabufarha",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "sarcasm",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/iabufarha/ar_sarcasm"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "ArSarcasm: 10,547 tweets labelled for sarcasm, sentiment and dialect, built from SemEval 2017 and ASTD.",
      "metrics": {
        "downloads": 457,
        "likes": 18,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "silma-embedding-sts-v0-1",
      "name": "silma-embedding-sts-v0.1",
      "type": "embedding",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/silma-ai/silma-embedding-sts-v0.1"
      },
      "size": "135M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "SILMA embedding model tuned for Arabic semantic textual similarity.",
      "base_model": [
        "silma-ai/silma-embeddding-matryoshka-0.1",
        "silma-ai/silma-embedding-matryoshka-v0.1"
      ],
      "metrics": {
        "downloads": 455,
        "likes": 6,
        "lastModified": "2024-11-04"
      }
    },
    {
      "id": "wav2vec2-xls-r-300m-iqraeval",
      "name": "wav2vec2 xls r 300m iqraeval",
      "type": "asr",
      "country": "YE",
      "org": "Fatimah Emad Eldin",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "vision",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/FatimahEmadEldin/wav2vec2-xls-r-300m-iqraeval"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "size": "300M",
      "on_device": true,
      "year": 2026,
      "notes": "This model is a fine-tuned version of facebook/wav2vec2-xls-r-300m specifically adapted for the Iqra'Eval 2026 Shared Task.",
      "metrics": {
        "downloads": 454,
        "likes": 3,
        "lastModified": "2026-03-01"
      }
    },
    {
      "id": "madinahqa",
      "name": "MadinahQA",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/MadinahQA"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic language and grammar multiple-choice QA split released with ArabicMMLU.",
      "metrics": {
        "downloads": 453,
        "likes": 1,
        "lastModified": "2024-09-17"
      }
    },
    {
      "id": "arcd",
      "name": "ARCD",
      "type": "benchmark",
      "country": "INTL",
      "org": "Mozannar et al.",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hsseinmz/arcd"
      },
      "size": "1.4k questions",
      "year": 2019,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic Reading Comprehension Dataset of 1,395 Wikipedia-based questions from the SOQAL paper.",
      "metrics": {
        "downloads": 452,
        "likes": 12,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "common-voice-arabic-12-0-augmented",
      "name": "common voice Arabic 12.0 Augmented",
      "type": "dataset",
      "country": "INTL",
      "org": "Salama1429",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Salama1429/common_voice_Arabic_12.0_Augmented"
      },
      "size": "10K–100K rows",
      "year": 2022,
      "notes": "Augmented Common Voice 12.0 Arabic: 63,546 training audio-sentence pairs at 16 kHz.",
      "metrics": {
        "downloads": 449,
        "likes": 2,
        "lastModified": "2022-12-24"
      }
    },
    {
      "id": "imageeval-arabicnlp26",
      "name": "ImageEval-ArabicNLP26",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "qa",
        "evaluation",
        "speech",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ImageEval-ArabicNLP26"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "It covers both of the shared task's tasks: AynVQA (Task 1), a culturally grounded Arabic multimodal benchmark for spoken visual question answering.",
      "metrics": {
        "downloads": 449,
        "likes": 4,
        "lastModified": "2026-08-30"
      }
    },
    {
      "id": "quranic-audio-nonnative",
      "name": "Quranic Audio (non-Arabic speakers)",
      "type": "dataset",
      "country": "INTL",
      "org": "RetaSy",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "pronunciation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/RetaSy/quranic_audio_dataset"
      },
      "year": 2024,
      "dialects": [
        "classical"
      ],
      "notes": "Crowdsourced and labeled Quran recitations from non-Arabic speakers.",
      "metrics": {
        "downloads": 448,
        "likes": 12,
        "lastModified": "2024-05-14"
      }
    },
    {
      "id": "arabic-vlm-full-pearl",
      "name": "Arabic-VLM-Full-Pearl",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/Arabic-VLM-Full-Pearl"
      },
      "notes": "309K multimodal examples for VLM training",
      "metrics": {
        "downloads": 443,
        "likes": 10,
        "lastModified": "2025-12-08"
      }
    },
    {
      "id": "mmbert-base-arabic-nli",
      "name": "mmbert base arabic nli",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "nli"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/mmbert-base-arabic-nli"
      },
      "year": 2025,
      "notes": "Arabic sentence embedding model fine-tuned from mmBERT-base on 900K NLI triplets with GISTEmbedLoss (768-dim).",
      "base_model": [
        "jhu-clsp/mmbert-base"
      ],
      "metrics": {
        "downloads": 442,
        "likes": 1,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "hadra-tts-f5",
      "name": "Hadra-TTS-f5",
      "type": "tts",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/algerian-nlp/Hadra-TTS-f5"
      },
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "F5-based TTS for Algerian Darija; Algeria noted.",
      "base_model": [
        "ibrahimsalah/arabic-f5-tts-v2"
      ],
      "metrics": {
        "downloads": 441,
        "likes": 0,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "silma-tts",
      "name": "SILMA TTS v1",
      "type": "tts",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "voice-cloning"
      ],
      "links": {
        "github": "https://github.com/SILMA-AI/silma-tts",
        "hf": "https://huggingface.co/silma-ai/silma-tts"
      },
      "year": 2026,
      "notes": "Open 150M bilingual Arabic/English TTS on the F5-TTS architecture; voice cloning from 8 s of audio.",
      "size": "150M",
      "on_device": true,
      "metrics": {
        "downloads": 437,
        "likes": 35,
        "lastModified": "2026-08-23"
      }
    },
    {
      "id": "aradice-arabicmmlu-lev",
      "name": "AraDICE-ArabicMMLU-lev",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AraDICE-ArabicMMLU-lev"
      },
      "dialects": [
        "lev"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The AraDiCE dataset is crafted to assess the dialectal and cultural understanding of large language models (LLMs) within Arabic-speaking contexts.",
      "metrics": {
        "downloads": 425,
        "likes": 0,
        "lastModified": "2024-11-08"
      }
    },
    {
      "id": "gate-reranker-v1",
      "name": "GATE-Reranker-V1",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reranking"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/GATE-Reranker-V1"
      },
      "size": "135M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic reranker from NAMAA built on GATE-AraBERT.",
      "base_model": [
        "omartificial-intelligence-space/gate-arabert-v1"
      ],
      "metrics": {
        "downloads": 423,
        "likes": 11,
        "lastModified": "2025-04-03"
      }
    },
    {
      "id": "arabic-speech-corpus",
      "name": "Arabic Speech Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "halabi2016",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/halabi2016/arabic_speech_corpus"
      },
      "notes": "Nawar Halabi's Levantine Arabic TTS corpus (CC-BY-4.0)",
      "metrics": {
        "downloads": 417,
        "likes": 38,
        "lastModified": "2024-08-14"
      }
    },
    {
      "id": "msa-omnivoice-tts-v1",
      "name": "msa omnivoice tts v1",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/msa-omnivoice-tts-v1"
      },
      "dialects": [
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "It is intended for training and fine-tuning Arabic speech models, including Text-to-Speech (TTS), Automatic Speech Recognition.",
      "metrics": {
        "downloads": 412,
        "likes": 1,
        "lastModified": "2026-07-08"
      }
    },
    {
      "id": "al-kawakib-magazine-ocr",
      "name": "al kawakib magazine ocr",
      "type": "dataset",
      "country": "INTL",
      "org": "amrosama",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "instruction-tuning",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/amrosama/al-kawakib-magazine-ocr"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This dataset contains rendered Arabic magazine page images paired with page-level text and line-level bounding boxes.",
      "metrics": {
        "downloads": 411,
        "likes": 3,
        "lastModified": "2026-07-25"
      }
    },
    {
      "id": "ajgt-twitter-ar",
      "name": "ajgt twitter ar",
      "type": "dataset",
      "country": "INTL",
      "org": "komari6",
      "license": "['unknown']",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/komari6/ajgt_twitter_ar"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic Jordanian General Tweets: 1,800 tweets labelled positive or negative, in MSA or Jordanian dialect.",
      "dialects": [
        "msa",
        "lev"
      ],
      "metrics": {
        "downloads": 410,
        "likes": 4,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "bge-m3-law",
      "name": "bge m3 law",
      "type": "embedding",
      "country": "INTL",
      "org": "mhaseeb1604",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/mhaseeb1604/bge-m3-law"
      },
      "year": 2024,
      "notes": "This model is a fine-tuned version of the BAAI/bge-m3 model, which is specialized for sentence similarity tasks in Arabic legal texts in both Arabic.",
      "base_model": [
        "baai/bge-m3"
      ],
      "metrics": {
        "downloads": 404,
        "likes": 6,
        "lastModified": "2024-10-09"
      }
    },
    {
      "id": "labr",
      "name": "LABR",
      "type": "dataset",
      "country": "EG",
      "org": "Cairo University",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/mohamedadaly/LABR",
        "hf": "https://huggingface.co/datasets/mohamedadaly/labr",
        "paper": "https://aclanthology.org/P13-2088.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "63,257 sentences",
      "year": 2013,
      "notes": "The largest sentiment analysis dataset to-date for the Arabic language.",
      "metrics": {
        "downloads": 402,
        "likes": 3,
        "lastModified": "2024-08-08"
      }
    },
    {
      "id": "whisper-tunisian-dialect",
      "name": "whisper-tunisian-dialect",
      "type": "asr",
      "country": "TN",
      "org": "TuniSpeech-AI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/TuniSpeech-AI/whisper-tunisian-dialect"
      },
      "size": "1.5B",
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Whisper fine-tuned on Tunisian dialect.",
      "metrics": {
        "downloads": 402,
        "likes": 4,
        "lastModified": "2026-02-13"
      }
    },
    {
      "id": "moulsot-full",
      "name": "MoulSot-Full",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/MoulSot-Full"
      },
      "size": "1500h",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Large Moroccan Darija speech corpus of about 1,500 hours.",
      "metrics": {
        "downloads": 399,
        "likes": 12,
        "lastModified": "2026-05-05"
      }
    },
    {
      "id": "bert-large-arabertv02-twitter",
      "name": "bert large arabertv02 twitter",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/bert-large-arabertv02-twitter"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "notes": "AraBERTv0.2-Twitter-base/large are two new models for Arabic dialects and tweets.",
      "metrics": {
        "downloads": 394,
        "likes": 4,
        "lastModified": "2023-04-26"
      }
    },
    {
      "id": "shifaa-medical",
      "name": "Shifaa Medical",
      "type": "dataset",
      "country": "EG",
      "org": "Ahmed Selem",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Ahmed-Selem/Shifaa_Arabic_Medical_Consultations"
      },
      "notes": "Arabic medical consultation dataset",
      "metrics": {
        "downloads": 394,
        "likes": 13,
        "lastModified": "2025-02-28"
      }
    },
    {
      "id": "arabicweb24",
      "name": "ArabicWeb24",
      "type": "dataset",
      "country": "INTL",
      "org": "lightonai",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/lightonai/ArabicWeb24"
      },
      "notes": "39B+ tokens of high-quality Arabic web content",
      "metrics": {
        "downloads": 392,
        "likes": 24,
        "lastModified": "2024-09-23"
      }
    },
    {
      "id": "jabarti-llm-dataset",
      "name": "jabarti llm dataset",
      "type": "dataset",
      "country": "EG",
      "org": "Bakrianoo",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/bakrianoo/jabarti-llm-dataset"
      },
      "dialects": [
        "egy"
      ],
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Cleaned, section-chunked training corpus for a small bilingual LLM.",
      "metrics": {
        "downloads": 390,
        "likes": 24,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "arabic-ocr-synthetic-scans-faker-300k",
      "name": "arabic ocr synthetic scans faker 300k",
      "type": "dataset",
      "country": "INTL",
      "org": "loay",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/loay/arabic-ocr-synthetic-scans-faker-300k"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "A large-scale synthetic dataset of 300,000 Arabic book pages generated to mimic real-world scanning imperfections.",
      "metrics": {
        "downloads": 387,
        "likes": 7,
        "lastModified": "2026-02-12"
      }
    },
    {
      "id": "nawah-50m-rag-support-2k",
      "name": "Nawah 50M RAG Support 2K",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-50M-RAG-Support-2K"
      },
      "dialects": [
        "msa"
      ],
      "size": "50M",
      "on_device": true,
      "year": 2026,
      "notes": "A 51.8M-parameter Arabic (MSA) retrieval-augmented customer-support answerer, trained from scratch.",
      "base_model": [
        "oddadmix/50m-2048-emhotob"
      ],
      "metrics": {
        "downloads": 387,
        "likes": 5,
        "lastModified": "2026-08-19"
      }
    },
    {
      "id": "arabic-marbert-sentiment",
      "name": "arabic MARBERT sentiment",
      "type": "llm",
      "country": "INTL",
      "org": "Ammar-alhaj-ali",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/Ammar-alhaj-ali/arabic-MARBERT-sentiment"
      },
      "year": 2022,
      "notes": "MARBERT fine-tuned for three-class Arabic sentiment analysis (positive, negative, neutral).",
      "metrics": {
        "downloads": 385,
        "likes": 11,
        "lastModified": "2022-08-09"
      }
    },
    {
      "id": "laion-coco-5m-arabic-labels",
      "name": "laion coco 5m arabic labels",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "cc-by-sa-4.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/laion-coco-5m-arabic-labels"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "An Arabic-labelled, image-bearing derivative of CaptionEmporium/laion-coco-13m-molmo-d-7b.",
      "metrics": {
        "downloads": 385,
        "likes": 0,
        "lastModified": "2026-08-29"
      }
    },
    {
      "id": "darija-english-doda",
      "name": "Darija-English (DODa)",
      "type": "dataset",
      "country": "MA",
      "org": "Darija Open Dataset",
      "license": "cc",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/imomayiz/darija-english"
      },
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Darija-English pairs from the Darija Open Dataset (DODa) project.",
      "metrics": {
        "downloads": 384,
        "likes": 13,
        "lastModified": "2024-04-17"
      }
    },
    {
      "id": "kfupm-arabic-ai-text-detection",
      "name": "KFUPM Arabic AI Text Detection",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM-JRCAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KFUPM-JRCAI/arabic-generated-abstracts"
      },
      "notes": "Machine-generated Arabic text across multiple LLMs",
      "metrics": {
        "downloads": 383,
        "likes": 4,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "neoarabert",
      "name": "NeoAraBERT",
      "type": "llm",
      "country": "INTL",
      "org": "U4RASD (ACRPS)",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/U4RASD/NeoAraBERT"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2026,
      "notes": "NeoAraBERT_Mix: NeoBERT-based Arabic encoder from ACRPS U4RASD and AUB, ranking first on 18 of 23 tasks.",
      "base_model": [
        "u4rasd/neoarabert"
      ],
      "metrics": {
        "downloads": 382,
        "likes": 10,
        "lastModified": "2026-09-08"
      }
    },
    {
      "id": "whisper-medium-darija",
      "name": "whisper-medium-darija",
      "type": "asr",
      "country": "MA",
      "org": "Youssef Chafiqui",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/whisper-medium-darija"
      },
      "size": "764M",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "Whisper medium fine-tuned on Moroccan Darija.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 381,
        "likes": 7,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "waqf-ocr-hand-written-v1",
      "name": "waqf ocr hand written v1",
      "type": "ocr",
      "country": "EG",
      "org": "Waqf AI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "chat",
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/Waqf-AI/waqf-ocr-hand-written-v1"
      },
      "year": 2026,
      "notes": "Fine-tuned from PaddlePaddle/PaddleOCR-VL-1.6 on khatt-augmented-arabic-20k-s3.",
      "metrics": {
        "downloads": 380,
        "likes": 5,
        "lastModified": "2026-06-09"
      }
    },
    {
      "id": "wav2veclarge-quran-syllables-recognition",
      "name": "Wav2vecLarge_quran_syllables_recognition",
      "type": "asr",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimSalah/Wav2vecLarge_quran_syllables_recognition"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This is fine tuned wav2vec2 model to recognize quran syllables from speech.",
      "metrics": {
        "downloads": 378,
        "likes": 6,
        "lastModified": "2025-01-05"
      }
    },
    {
      "id": "rasam-1",
      "name": "RASAM 1",
      "type": "dataset",
      "country": "INTL",
      "org": "calfa-ai",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/calfa-ai/RASAM-1"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "RASAM 1 is a specialized dataset for Handwritten Text Recognit (also: 1 variants)",
      "metrics": {
        "downloads": 375,
        "likes": 3,
        "lastModified": "2026-09-11"
      }
    },
    {
      "id": "fineweb-edu-arabic",
      "name": "FineWeb-Edu Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/fineweb-edu-arabic"
      },
      "size": "10M-100M rows",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic translation of the FineWeb-Edu sample filtered by language score.",
      "metrics": {
        "downloads": 373,
        "likes": 1,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "jais-13b-chat",
      "name": "jais 13b chat",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-13b-chat"
      },
      "base_model": [
        "inception-mbzuai/jais-13b"
      ],
      "size": "13B",
      "on_device": false,
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "Instruction-tuned chat version of Jais-13B, a bilingual Arabic-English LLM (arXiv 2308.16149).",
      "metrics": {
        "downloads": 373,
        "likes": 168,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "arabic-morocco-speech-to-text",
      "name": "Arabic Morocco Speech To Text",
      "type": "asr",
      "country": "INTL",
      "org": "smerchi",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/smerchi/Arabic-Morocco-Speech_To_Text"
      },
      "year": 2024,
      "notes": "Whisper large-v3 fine-tuned on Cleverlytics voice data for Moroccan Arabic speech recognition.",
      "dialects": [
        "magh"
      ],
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 372,
        "likes": 17,
        "lastModified": "2024-04-02"
      }
    },
    {
      "id": "aramodernbert-base-sts",
      "name": "AraModernBert-Base-STS",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/AraModernBert-Base-STS"
      },
      "year": 2025,
      "notes": "This SentenceTransformer is fine-tuned from NAMAA-Space/AraModernBert-Base-V1.0, bringing strong arabic embeddings useful for a multiple.",
      "base_model": [
        "namaa-space/aramodernbert-base-v1.0"
      ],
      "metrics": {
        "downloads": 371,
        "likes": 7,
        "lastModified": "2025-03-09"
      }
    },
    {
      "id": "star-dataset-instructions",
      "name": "star dataset instructions",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM-JRCAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "translation",
        "summarization",
        "qa",
        "embedding",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KFUPM-JRCAI/star-dataset-instructions"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "STAR Instructions is a large-scale Arabic instruction-tuning dataset built by rendering the 355 STAR Jinja2 prompt templates against their 87 source.",
      "metrics": {
        "downloads": 371,
        "likes": 1,
        "lastModified": "2026-09-03"
      }
    },
    {
      "id": "arabic-tashkeel-flan-t5-small",
      "name": "arabic tashkeel flan t5 small",
      "type": "llm",
      "country": "INTL",
      "org": "Abdou",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "tashkeel"
      ],
      "links": {
        "hf": "https://huggingface.co/Abdou/arabic-tashkeel-flan-t5-small"
      },
      "year": 2024,
      "notes": "Flan-T5-small model that adds tashkeel (Arabic diacritics) to Arabic text.",
      "on_device": true,
      "metrics": {
        "downloads": 370,
        "likes": 4,
        "lastModified": "2024-10-22"
      }
    },
    {
      "id": "blip3-grounding-1m-arabic",
      "name": "blip3 grounding 1m arabic",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/blip3-grounding-1m-arabic"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic version of the first 1,000,000 usable rows of Salesforce/blip3-grounding-50m.",
      "metrics": {
        "downloads": 365,
        "likes": 0,
        "lastModified": "2026-08-23"
      }
    },
    {
      "id": "arabicner-wojood",
      "name": "ArabicNER-Wojood",
      "type": "llm",
      "country": "PS",
      "org": "SinaLab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/SinaLab/ArabicNER-Wojood"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic nested NER model trained on Wojood; Palestine.",
      "metrics": {
        "downloads": 363,
        "likes": 10,
        "lastModified": "2024-03-20"
      }
    },
    {
      "id": "hadra-asr-whisper-medium",
      "name": "Hadra-ASR-whisper-medium",
      "type": "asr",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/algerian-nlp/Hadra-ASR-whisper-medium"
      },
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Whisper medium for Algerian Darija; Algeria noted.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 362,
        "likes": 1,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "nilechat-3b",
      "name": "NileChat-3B",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "qwen-research",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/NileChat-3B"
      },
      "base_model": [
        "qwen/qwen2.5-3b"
      ],
      "size": "3B",
      "on_device": true,
      "dialects": [
        "egy",
        "magh"
      ],
      "notes": "Egyptian and Moroccan dialects",
      "metrics": {
        "downloads": 361,
        "likes": 23,
        "lastModified": "2026-05-11"
      }
    },
    {
      "id": "visper",
      "name": "visper",
      "type": "dataset",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "cc-by-nc-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tiiuae/visper"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This repository contains ViSpeR, a large-scale dataset and models for Visual Speech Recognition for Arabic, Chinese, French, Arabic and Spanish.",
      "metrics": {
        "downloads": 360,
        "likes": 6,
        "lastModified": "2025-04-17"
      }
    },
    {
      "id": "quranmb-v2",
      "name": "QuranMB.v2",
      "type": "benchmark",
      "country": "INTL",
      "org": "IqraEval",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IqraEval/QuranMB.v2"
      },
      "year": 2025,
      "dialects": [
        "classical"
      ],
      "notes": "IqraEval benchmark for Quranic mispronunciation detection.",
      "metrics": {
        "downloads": 350,
        "likes": 3,
        "lastModified": "2025-07-05"
      }
    },
    {
      "id": "egyptian-arabic-lectures",
      "name": "Egyptian Arabic Lectures",
      "type": "dataset",
      "country": "INTL",
      "org": "ismaeeelxd",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ismaeeelxd/Egyptian-Arabic-Lectures"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "The Egyptian Arabic Lectures dataset is a collection of transcribed audio clips.",
      "metrics": {
        "downloads": 347,
        "likes": 3,
        "lastModified": "2026-06-21"
      }
    },
    {
      "id": "tafsir-mcp-data",
      "name": "tafsir mcp data",
      "type": "dataset",
      "country": "INTL",
      "org": "Tafsir Center",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tafsircenter/tafsir-mcp-data"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Quran and tafsir database behind tafsir-mcp, an MCP server giving AI assistants offline access to Quranic scholarship.",
      "metrics": {
        "downloads": 346,
        "likes": 4,
        "lastModified": "2026-05-13"
      }
    },
    {
      "id": "khatt-v1-0-dataset",
      "name": "KHATT v1.0 dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "johnlockejrr",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "ocr",
        "htr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/johnlockejrr/KHATT_v1.0_dataset"
      },
      "year": 2024,
      "notes": "Line-level KHATT database of unconstrained handwritten Arabic text from 1000 writers, developed by a KFUPM research group.",
      "metrics": {
        "downloads": 345,
        "likes": 4,
        "lastModified": "2024-07-01"
      }
    },
    {
      "id": "open-arabicaqa",
      "name": "Open-ArabicaQA",
      "type": "dataset",
      "country": "INTL",
      "org": "DataScienceUIBK",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdoelsayed/Open-ArabicaQA"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Open-domain QA and retrieval splits of ArabicaQA on Hugging Face.",
      "metrics": {
        "downloads": 345,
        "likes": 10,
        "lastModified": "2024-03-27"
      }
    },
    {
      "id": "aramix-translation-scores",
      "name": "AraMix-Translation-Scores",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/AraMix-Translation-Scores"
      },
      "size": "100M–1B rows",
      "year": 2026,
      "notes": "AdaMLLab/AraMix (minhashdeduped subset, 178,883,241 rows) with a machine-translation-detection score added to every document.",
      "metrics": {
        "downloads": 344,
        "likes": 0,
        "lastModified": "2026-07-18"
      }
    },
    {
      "id": "algerian-darja-corpus",
      "name": "Algerian Darja Corpus",
      "type": "dataset",
      "country": "DZ",
      "org": "Touati Kamel",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/touati-kamel/algerian-darja-corpus"
      },
      "size": "10K-100K rows",
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Algerian Darja text corpus with Zenodo DOI (Algeria).",
      "metrics": {
        "downloads": 343,
        "likes": 6,
        "lastModified": "2026-09-18"
      }
    },
    {
      "id": "t5-arabic-summarization",
      "name": "t5-arabic-summarization",
      "type": "llm",
      "country": "INTL",
      "org": "malmarjeh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/malmarjeh/t5-arabic-text-summarization"
      },
      "notes": "Summarization - T5 for Arabic news summarization",
      "metrics": {
        "downloads": 343,
        "likes": 14,
        "lastModified": "2023-07-01"
      }
    },
    {
      "id": "aramix-native",
      "name": "AraMix-Native",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "translation",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/AraMix-Native"
      },
      "size": "100M–1B rows",
      "year": 2026,
      "notes": "A native-Arabic-filtered version of AdaMLLab/AraMix (minhashdeduped), derived from SultanR/AraMix-Translation-Scores.",
      "metrics": {
        "downloads": 341,
        "likes": 0,
        "lastModified": "2026-07-19"
      }
    },
    {
      "id": "openresearcher-corpus-ar",
      "name": "openresearcher corpus ar",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/openresearcher-corpus-ar"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic translation of OpenResearcher-Corpus, an 11B token curated web corpus.",
      "metrics": {
        "downloads": 341,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "ar-musa",
      "name": "Ar MUSA",
      "type": "dataset",
      "country": "INTL",
      "org": "Skhaled",
      "license": "afl-3.0",
      "modality": "multimodal",
      "tasks": [
        "speech",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Skhaled/Ar-MUSA"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The Ar-MUSA directory contains annotated datasets organized by batches and annotation teams.",
      "metrics": {
        "downloads": 338,
        "likes": 3,
        "lastModified": "2025-04-14"
      }
    },
    {
      "id": "aragen",
      "name": "AraGen",
      "type": "benchmark",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/inception42/AraGen"
      },
      "year": 2024,
      "notes": "Generative Arabic LLM benchmark with 3C3H scoring and dynamic periodic refresh, with MBZUAI.",
      "metrics": {
        "downloads": 337,
        "likes": 2,
        "lastModified": "2025-12-15"
      }
    },
    {
      "id": "arabic-qwen3-5-ocr-v4",
      "name": "Arabic-Qwen3.5-OCR-v4",
      "type": "ocr",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/Arabic-Qwen3.5-OCR-v4"
      },
      "size": "853M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Qwen3.5-based Arabic OCR model.",
      "base_model": [
        "qwen/qwen3.5-0.8b"
      ],
      "metrics": {
        "downloads": 336,
        "likes": 10,
        "lastModified": "2026-03-24"
      }
    },
    {
      "id": "arabic-audio-collection-moroccan-ameed",
      "name": "arabic audio collection moroccan ameed",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-moroccan-ameed"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Ameed Moroccan Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 176 hours of speech recordings.",
      "metrics": {
        "downloads": 335,
        "likes": 1,
        "lastModified": "2026-06-30"
      }
    },
    {
      "id": "tunbert",
      "name": "TunBERT",
      "type": "llm",
      "country": "TN",
      "org": "InstaDeep / iCompass",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/tunis-ai/TunBERT"
      },
      "size": "110M",
      "year": 2021,
      "dialects": [
        "magh"
      ],
      "notes": "First BERT for Tunisian dialect (Derja), from InstaDeep and iCompass.",
      "metrics": {
        "downloads": 335,
        "likes": 15,
        "lastModified": "2025-09-29"
      }
    },
    {
      "id": "arabiangpt",
      "name": "ArabianGPT",
      "type": "llm",
      "country": "SA",
      "org": "Prince Sultan University",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/riotu-lab/ArabianGPT-01B"
      },
      "size": "0.1B",
      "notes": "GPT-2 for Arabic, RIOTU Lab",
      "on_device": true,
      "metrics": {
        "downloads": 334,
        "likes": 13,
        "lastModified": "2024-02-27"
      }
    },
    {
      "id": "quran-recitations",
      "name": "Quran Recitations",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/Quran-Recitations"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "Quranic verses paired with audio recitations by multiple Qaris, 124,689 audio-text rows.",
      "metrics": {
        "downloads": 333,
        "likes": 63,
        "lastModified": "2025-03-30"
      }
    },
    {
      "id": "covost2-en-ar",
      "name": "CoVoST2-EN-AR",
      "type": "dataset",
      "country": "INTL",
      "org": "ymoslem",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "translation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ymoslem/CoVoST2-EN-AR"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "CoVoST 2 is a large-scale multilingual speech translation corpus based on Common Voice, developed by FAIR.",
      "metrics": {
        "downloads": 332,
        "likes": 5,
        "lastModified": "2024-12-04"
      }
    },
    {
      "id": "arabic-glm-ocr-v2",
      "name": "Arabic-GLM-OCR-v2",
      "type": "ocr",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/Arabic-GLM-OCR-v2"
      },
      "size": "1.1B",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "GLM-OCR fine-tuned for Arabic documents.",
      "base_model": [
        "zai-org/glm-ocr"
      ],
      "metrics": {
        "downloads": 331,
        "likes": 17,
        "lastModified": "2026-05-27"
      }
    },
    {
      "id": "cohere-transcribe-arabic-07-2026-dialectal",
      "name": "cohere transcribe arabic 07 2026 dialectal",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/cohere-transcribe-arabic-07-2026-dialectal"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "Full fine-tune of CohereLabs/cohere-transcribe-arabic-07-2026 (2.07B, cohereasr Conformer encoder-decoder) for multi-dialect Arabic speech recognition.",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 330,
        "likes": 2,
        "lastModified": "2026-07-09"
      }
    },
    {
      "id": "unimorph",
      "name": "UniMorph",
      "type": "dataset",
      "country": "INTL",
      "org": "Center for Language and Speech Processing Johns Hopkins University",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/unimorph/ara",
        "hf": "https://huggingface.co/datasets/unimorph/universal_morphologies",
        "paper": "https://unimorph.github.io/doc/unimorph-schema.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "140,003 tokens",
      "year": 2015,
      "notes": "167 languages have been annotated according to the UniMorph schema.",
      "metrics": {
        "downloads": 330,
        "likes": 20,
        "lastModified": "2023-06-08"
      }
    },
    {
      "id": "tamazight-speech-to-arabic-text-tamazight-nlp",
      "name": "Tamazight Speech to Arabic Text",
      "type": "dataset",
      "country": "INTL",
      "org": "Tamazight-NLP",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Tamazight-NLP/Tamazight-Speech-to-Arabic-Text"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This is the Tamazight-NLP organization-hosted version of the Tamazight-Arabic Speech Recognition Dataset.",
      "metrics": {
        "downloads": 324,
        "likes": 11,
        "lastModified": "2025-03-29"
      }
    },
    {
      "id": "egyptian-asr-mgb3",
      "name": "Egyptian ASR MGB-3",
      "type": "dataset",
      "country": "EG",
      "org": "Mightystudent",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MightyStudent/Egyptian-ASR-MGB-3"
      },
      "size": "1K-10K rows",
      "year": 2023,
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic dialect ASR set derived from MGB-3 YouTube data.",
      "metrics": {
        "downloads": 322,
        "likes": 23,
        "lastModified": "2024-09-03"
      }
    },
    {
      "id": "kde4",
      "name": "KDE4",
      "type": "dataset",
      "country": "INTL",
      "org": "Uppsala University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Helsinki-NLP/kde4",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2012/pdf/463_Paper.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "700,000 sentences",
      "year": 2012,
      "notes": "A parallel corpus of KDE4 localization files (v.2). 92 languages, 4,099 bitexts",
      "metrics": {
        "downloads": 322,
        "likes": 26,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "algerian-darija-text",
      "name": "Algerian Darija",
      "type": "dataset",
      "country": "DZ",
      "org": "ayoubkirouane",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ayoubkirouane/Algerian-Darija"
      },
      "size": "100K-1M rows",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Algerian Darija text collected from HF datasets, scraping and social media (Algeria).",
      "metrics": {
        "downloads": 321,
        "likes": 17,
        "lastModified": "2026-04-18"
      }
    },
    {
      "id": "quran-ocr",
      "name": "Quran OCR",
      "type": "benchmark",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/sherif1313/Quran-OCR"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Any text recognition or Optical Character Recognition (OCR) system requires a dataset to learn how to recognize the text.",
      "metrics": {
        "downloads": 320,
        "likes": 0,
        "lastModified": "2026-02-18"
      }
    },
    {
      "id": "ashaar-poetry",
      "name": "Ashaar",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/ashaar"
      },
      "size": "254k poems",
      "year": 2022,
      "dialects": [
        "classical"
      ],
      "notes": "Largest Arabic poetry dataset, 254k poems and 3.8M verses with meter and genre metadata.",
      "metrics": {
        "downloads": 317,
        "likes": 8,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "staro-ai-2-69-super",
      "name": "StarO-AI-2.69-Super",
      "type": "llm",
      "country": "INTL",
      "org": "C-a-Star-Technology-Official",
      "license": "creativeml-openrail-m",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/C-a-Star-Technology-Official/StarO-AI-2.69-Super"
      },
      "year": 2026,
      "notes": "Created by Hawa-Al-Akram, Update By C.a.",
      "metrics": {
        "downloads": 317,
        "likes": 3,
        "lastModified": "2026-06-30"
      }
    },
    {
      "id": "whisper-small-darija",
      "name": "whisper small darija",
      "type": "asr",
      "country": "MA",
      "org": "Youssef Chafiqui",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/whisper-small-darija"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "This model is a fine-tuned version of openai/whisper-small on the Darija Speech to Text dataset.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 317,
        "likes": 3,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "nawah-50m-rag-chat",
      "name": "Nawah-50M RAG Chat",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-50M-RAG-Chat-8K-GRPO"
      },
      "size": "52M",
      "dialects": [
        "msa"
      ],
      "notes": "Tiny 50M Arabic RAG chat model with 8K context, tuned with GRPO.",
      "base_model": [
        "oddadmix/nawah-50m-rag-chat-8k"
      ],
      "metrics": {
        "downloads": 316,
        "likes": 0,
        "lastModified": "2026-08-20"
      }
    },
    {
      "id": "arat5-base-title-generation",
      "name": "AraT5-base-title-generation",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "title-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/AraT5-base-title-generation"
      },
      "dialects": [
        "msa"
      ],
      "year": 2022,
      "notes": "AraT5-base fine-tuned for Arabic title generation, from the AraT5 paper.",
      "metrics": {
        "downloads": 315,
        "likes": 13,
        "lastModified": "2022-05-26"
      }
    },
    {
      "id": "mubsir-qwen-2b-vl",
      "name": "Mubsir Qwen 2B VL",
      "type": "ocr",
      "country": "INTL",
      "org": "Hatim2221",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "chat",
        "asr",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/Hatim2221/Mubsir-Qwen-2B-VL"
      },
      "size": "2B",
      "on_device": true,
      "year": 2026,
      "notes": "This repository contains a vision-language model fine-tuned specifically for Arabic Handwritten Text Recognition (HTR).",
      "base_model": [
        "qwen/qwen-vl"
      ],
      "metrics": {
        "downloads": 315,
        "likes": 5,
        "lastModified": "2026-08-24"
      }
    },
    {
      "id": "arabic-audio-collection-algerian-kahwa-postcast",
      "name": "arabic audio collection algerian kahwa postcast",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-algerian-kahwa-postcast"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Kahwa Postcast Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 110 hours of speech recordings.",
      "metrics": {
        "downloads": 313,
        "likes": 2,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "arabic-to-code-3m",
      "name": "Arabic-to-Code 8 Languages 3M",
      "type": "dataset",
      "country": "INTL",
      "org": "ISLAM-PO",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "code"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ISLAM-PO/arabic-to-code-8-langs-3m"
      },
      "size": "3M",
      "year": 2026,
      "notes": "3M Arabic instruction examples for code generation across 8 languages.",
      "metrics": {
        "downloads": 313,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "esprit-derja-qwen3-8b",
      "name": "ESPRIT-Derja-Qwen3-8B",
      "type": "llm",
      "country": "TN",
      "org": "ESPRIT Group",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/ESPRIT-Group/ESPRIT-Derja-Qwen3-8B-v2"
      },
      "base_model": [
        "qwen/qwen3-8b"
      ],
      "size": "8.2B",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Qwen3-8B fine-tuned for Tunisian Derja by ESPRIT.",
      "metrics": {
        "downloads": 313,
        "likes": 2,
        "lastModified": "2026-06-08"
      }
    },
    {
      "id": "ultradata-math-textbook-exercise-ar",
      "name": "ultradata math textbook exercise ar",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/ultradata-math-textbook-exercise-ar"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "Arabic translation of the English portion of UltraData-Math, config UltraData-Math-L3-Textbook-Exercise-Synthetic.",
      "metrics": {
        "downloads": 312,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "sard-extended",
      "name": "SARD Extended",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "cc-by-nc-nd-4.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/SARD-Extended"
      },
      "year": 2026,
      "notes": "SARD: large-scale synthetic Arabic OCR dataset rendered in many fonts (Amiri, Sakkal Majalla, Arial, Calibri and others).",
      "metrics": {
        "downloads": 311,
        "likes": 9,
        "lastModified": "2026-05-20"
      }
    },
    {
      "id": "whisper-small-egyptian-codeswitch",
      "name": "Whisper Small Egyptian Code-Switch",
      "type": "asr",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimAmin/code-switched-egyptian-arabic-whisper-small"
      },
      "size": "242M",
      "dialects": [
        "egy",
        "mixed"
      ],
      "notes": "Whisper-small tuned for Egyptian Arabic with English code-switching.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 310,
        "likes": 5,
        "lastModified": "2025-05-17"
      }
    },
    {
      "id": "ten2zero",
      "name": "Ten2Zero",
      "type": "dataset",
      "country": "SA",
      "org": "Umm Al-Qura University",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/gfbati/Ten2Zero",
        "paper": "https://doi.org/10.21608/JESAUN.2023.231628.1254"
      },
      "dialects": [
        "mixed"
      ],
      "size": "935 tokens",
      "year": 2023,
      "notes": "Balanced spoken Arabic digit dataset (numbers 0-10) for teaching machine learning without coding, including wav files and generated tabular.",
      "metrics": {
        "downloads": 306,
        "likes": 1,
        "lastModified": "2023-10-20"
      }
    },
    {
      "id": "dclm-edu-ar-500k",
      "name": "dclm edu ar 500k",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/dclm-edu-ar-500k"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "This is an Arabic translation of the HuggingFaceTB/dclm-edu dataset.",
      "metrics": {
        "downloads": 305,
        "likes": 0,
        "lastModified": "2026-02-01"
      }
    },
    {
      "id": "arabic-tweets",
      "name": "Arabic Tweets",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohammad Albarham",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/pain/Arabic-Tweets"
      },
      "notes": "41GB+ of Arabic tweets (~4B words)",
      "metrics": {
        "downloads": 304,
        "likes": 23,
        "lastModified": "2023-04-08"
      }
    },
    {
      "id": "darija-speech-to-text",
      "name": "Darija Speech to Text",
      "type": "dataset",
      "country": "MA",
      "org": "adiren7",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/adiren7/darija_speech_to_text"
      },
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan Darija speech with transcripts for ASR training.",
      "metrics": {
        "downloads": 304,
        "likes": 17,
        "lastModified": "2024-08-11"
      }
    },
    {
      "id": "ketaba-ocr-lora",
      "name": "Ketaba-OCR-LoRA",
      "type": "ocr",
      "country": "INTL",
      "org": "HassanB4",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/HassanB4/Ketaba-OCR-LoRA"
      },
      "year": 2026,
      "notes": "This repository contains the official models and results for Ketaba-OCR, our submission to the NakbaNLP 2026 Shared Task.",
      "base_model": [
        "sherif1313/arabic-english-handwritten-ocr-v3"
      ],
      "metrics": {
        "downloads": 304,
        "likes": 5,
        "lastModified": "2026-03-05"
      }
    },
    {
      "id": "mlma-hate-speech",
      "name": "MLMA hate speech",
      "type": "dataset",
      "country": "INTL",
      "org": "nedjmaou",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nedjmaou/MLMA_hate_speech"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Arabic text dataset (10K<N<100K rows).",
      "metrics": {
        "downloads": 304,
        "likes": 5,
        "lastModified": "2025-03-22"
      }
    },
    {
      "id": "camel-readability-arabertv02",
      "name": "CAMeL Readability (AraBERTv02)",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/readability-arabertv02-word-CE"
      },
      "year": 2024,
      "notes": "Word-level Arabic readability model from the BAREC readability work at NYUAD.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 299,
        "likes": 1,
        "lastModified": "2025-07-15"
      }
    },
    {
      "id": "memexplain",
      "name": "MemeXplain",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/MemeXplain"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "MemeXplain is a comprehensive multimodal dataset for detecting and explaining propagandistic and hateful content in memes.",
      "metrics": {
        "downloads": 298,
        "likes": 0,
        "lastModified": "2026-03-27"
      }
    },
    {
      "id": "whisper-l-v3-turbo-quran-lora-dataset-mix",
      "name": "whisper l v3 turbo quran lora dataset mix",
      "type": "asr",
      "country": "INTL",
      "org": "MaddoggProduction",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/MaddoggProduction/whisper-l-v3-turbo-quran-lora-dataset-mix"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "This is a specialized Automatic Speech Recognition (ASR) model for Quranic Recitation with tashkeel or diacritics.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 298,
        "likes": 21,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "atlas-chat-2b",
      "name": "Atlas-Chat-2B",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris/Atlas-Chat-2B"
      },
      "base_model": [
        "google/gemma-2-2b-it"
      ],
      "size": "2.6B",
      "year": 2024,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "Atlas-Chat 2B, small Moroccan Darija instruction model.",
      "metrics": {
        "downloads": 297,
        "likes": 27,
        "lastModified": "2025-03-28"
      }
    },
    {
      "id": "nabra-7m-distill",
      "name": "Nabra 7M Distill",
      "type": "tts",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nabra-7M-Distill"
      },
      "size": "7M",
      "year": 2026,
      "on_device": true,
      "notes": "Distilled 7M-parameter Arabic Kokoro-style TTS for phones.",
      "metrics": {
        "downloads": 297,
        "likes": 11,
        "lastModified": "2026-09-16"
      }
    },
    {
      "id": "yehia",
      "name": "Yehia",
      "type": "llm",
      "country": "INTL",
      "org": "Navid-AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Navid-AI/Yehia-7B"
      },
      "base_model": [
        "allam-ai/allam-7b-instruct-preview"
      ],
      "size": "7B",
      "notes": "Based on ALLaM, Arabic instruction-tuned",
      "metrics": {
        "downloads": 297,
        "likes": 25,
        "lastModified": "2026-01-06"
      }
    },
    {
      "id": "yehia-7b-preview",
      "name": "Yehia 7B preview",
      "type": "llm",
      "country": "INTL",
      "org": "Navid-AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Navid-AI/Yehia-7B-preview"
      },
      "base_model": [
        "allam-ai/allam-7b-instruct-preview"
      ],
      "size": "7B",
      "on_device": false,
      "year": 2026,
      "notes": "Arabic-English chat model fine-tuned from ALLaM-7B-Instruct-preview.",
      "metrics": {
        "downloads": 297,
        "likes": 25,
        "lastModified": "2026-01-06"
      }
    },
    {
      "id": "niletts-dataset",
      "name": "NileTTS-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "KickItLikeShika",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KickItLikeShika/NileTTS-dataset",
        "paper": "https://arxiv.org/abs/2602.15675"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "First large-scale public Egyptian Arabic TTS dataset: 38.1 hours, 9,521 utterances, 2 speakers, across diverse domains.",
      "metrics": {
        "downloads": 294,
        "likes": 3,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-ca-ner",
      "name": "bert base arabic camelbert ca ner",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-ca-ner"
      },
      "year": 2021,
      "notes": "For the fine-tuning, we used the ANERcorp dataset.",
      "metrics": {
        "downloads": 293,
        "likes": 3,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "bert-base-arabic-finetuned-emotion",
      "name": "bert base arabic finetuned emotion",
      "type": "llm",
      "country": "INTL",
      "org": "hatemnoaman",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/hatemnoaman/bert-base-arabic-finetuned-emotion"
      },
      "year": 2023,
      "notes": "This model is a fine-tuned version of asafaya/bert-base-arabic on the emotonear dataset.",
      "metrics": {
        "downloads": 293,
        "likes": 3,
        "lastModified": "2023-12-03"
      }
    },
    {
      "id": "naqta",
      "name": "Naqta",
      "type": "llm",
      "country": "INTL",
      "org": "Mostafa Maroof",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/MostafaMaroof/Naqta"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "notes": "Restores punctuation in Modern Standard Arabic.",
      "metrics": {
        "downloads": 292,
        "likes": 2,
        "lastModified": "2026-09-02"
      }
    },
    {
      "id": "nemotron-mc-en-ar-midtrain",
      "name": "nemotron mc en ar midtrain",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "pretraining",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/nemotron-mc-en-ar-midtrain"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "Arabic translation of the Nemotron-Pretraining-Multiple-Choice config of Nemotron-Pretraining-Specialized-v1.2 (pinned revision 807afc1).",
      "metrics": {
        "downloads": 292,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "persian-arabic-textline-image-ocr-medium",
      "name": "Persian_Arabic_TextLine_Image_Ocr_Medium",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohajesmaeili",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohajesmaeili/Persian_Arabic_TextLine_Image_Ocr_Medium"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "Text-line image OCR dataset (626K train, 165K test images) covering Persian, Arabic, Pashto and Urdu scripts.",
      "metrics": {
        "downloads": 292,
        "likes": 18,
        "lastModified": "2025-09-26"
      }
    },
    {
      "id": "arabic-audio-collection-mohamed-khairy",
      "name": "arabic audio collection mohamed khairy",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-mohamed-khairy"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Mohamed Khairy Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 430 hours of speech recordings.",
      "metrics": {
        "downloads": 290,
        "likes": 8,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "qwen3-asr-arabic-ksa",
      "name": "qwen3-asr-arabic-ksa",
      "type": "asr",
      "country": "SA",
      "org": "Vadim Belsky",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/qwen3-asr-arabic-ksa"
      },
      "size": "2B",
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Qwen3-ASR fine-tuned for Saudi Arabic.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 286,
        "likes": 2,
        "lastModified": "2026-04-06"
      }
    },
    {
      "id": "ar-res-reviews",
      "name": "ar res reviews",
      "type": "dataset",
      "country": "INTL",
      "org": "hadyelsahar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hadyelsahar/ar_res_reviews"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "ArRestReviews: 8,364 Arabic restaurant reviews from qaym.com for sentiment analysis.",
      "metrics": {
        "downloads": 285,
        "likes": 8,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "aslema-synth-tn",
      "name": "Aslema Synth TN",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "speech",
        "asr",
        "tts",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/Aslema-Synth-TN"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "understanding: speech annotated for intent and for slot filling.",
      "metrics": {
        "downloads": 282,
        "likes": 2,
        "lastModified": "2026-09-06"
      }
    },
    {
      "id": "qwen3-5-tts-emirati",
      "name": "qwen3.5 TTS Emirati",
      "type": "tts",
      "country": "AE",
      "org": "Vadim Belsky",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/qwen3.5-TTS-Emirati"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "🚀 Try the live demo on HuggingFace Spaces The base model (Qwen3-TTS-12Hz-1.7B-Base) ships with a fixed set of languages in its codec token vocabulary.",
      "base_model": [
        "qwen/qwen3-tts-12hz-1.7b-base"
      ],
      "metrics": {
        "downloads": 282,
        "likes": 3,
        "lastModified": "2026-03-15"
      }
    },
    {
      "id": "aragpt2-large",
      "name": "aragpt2-large",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/aragpt2-large"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "829M",
      "year": 2022,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "AraGPT2 large, 792M Arabic GPT-2 from AUB.",
      "metrics": {
        "downloads": 279,
        "likes": 10,
        "lastModified": "2024-05-29"
      }
    },
    {
      "id": "fanar-2-diwan",
      "name": "Fanar-2-Diwan",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/Fanar-2-Diwan"
      },
      "base_model": [
        "aubmindlab/aragpt2-mega"
      ],
      "notes": "Fanar-2 family model released by QCRI on Hugging Face.",
      "metrics": {
        "downloads": 279,
        "likes": 2,
        "lastModified": "2026-03-25"
      }
    },
    {
      "id": "wav2vec2-base-word-by-word-quran-asr",
      "name": "wav2vec2 base word by word quran asr",
      "type": "asr",
      "country": "INTL",
      "org": "HamzaSidhu786",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/HamzaSidhu786/wav2vec2-base-word-by-word-quran-asr"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "It achieves the following results on the",
      "base_model": [
        "facebook/wav2vec2-base"
      ],
      "metrics": {
        "downloads": 279,
        "likes": 5,
        "lastModified": "2024-07-23"
      }
    },
    {
      "id": "whisper-turbo-egyptian-codeswitch",
      "name": "Whisper Turbo Egyptian Code-Switch",
      "type": "asr",
      "country": "EG",
      "org": "Mohammed Aly",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammedaly22/whisper-large-v3-turbo-egyptian-code-switching"
      },
      "size": "809M",
      "dialects": [
        "egy",
        "mixed"
      ],
      "notes": "Whisper large-v3-turbo fine-tuned for Egyptian Arabic and English code-switched speech.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 279,
        "likes": 0,
        "lastModified": "2026-05-13"
      }
    },
    {
      "id": "darijadz",
      "name": "DarijaDz",
      "type": "dataset",
      "country": "INTL",
      "org": "nasrellahkharroubi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nasrellahkharroubi/DarijaDz"
      },
      "dialects": [
        "magh"
      ],
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "DarijaDZ: about 22.5M user comments (259M tokens) from Algerian YouTube and TikTok channels, mostly in Algerian Darija.",
      "metrics": {
        "downloads": 278,
        "likes": 12,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "muslim-names-dataset",
      "name": "muslim names dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "takiuddinahmed",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/takiuddinahmed/muslim-names-dataset"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "A comprehensive collection of Muslim names with meanings scraped from muslimnames.com.",
      "metrics": {
        "downloads": 277,
        "likes": 3,
        "lastModified": "2025-07-05"
      }
    },
    {
      "id": "whisper-small-egyptian-arabic",
      "name": "Whisper Small Egyptian Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "MAdel121",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MAdel121/whisper-small-egyptian-arabic"
      },
      "size": "242M",
      "dialects": [
        "egy"
      ],
      "notes": "Whisper-small fine-tuned on Egyptian Arabic speech; lighter sibling of the medium model.",
      "metrics": {
        "downloads": 274,
        "likes": 7,
        "lastModified": "2025-05-21"
      }
    },
    {
      "id": "arabculture",
      "name": "ArabCulture",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "commonsense"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ArabCulture",
        "paper": "https://arxiv.org/abs/2502.12788"
      },
      "size": "3.5k questions",
      "year": 2025,
      "dialects": [
        "mixed"
      ],
      "notes": "Culturally grounded commonsense reasoning benchmark across 13 Arab countries written natively.",
      "metrics": {
        "downloads": 273,
        "likes": 14,
        "lastModified": "2025-05-23"
      }
    },
    {
      "id": "mgb-3-arabic",
      "name": "MGB 3 Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/MGB-3-Arabic"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The MGB-3 Arabic dataset is a multi-genre collection of Egyptian Arabic speech extracted from YouTube videos.",
      "metrics": {
        "downloads": 273,
        "likes": 7,
        "lastModified": "2025-12-28"
      }
    },
    {
      "id": "arabic-english-code-switching",
      "name": "Arabic-English Code-Switching",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-english-code-switching"
      },
      "notes": "Code-switching speech from YouTube",
      "metrics": {
        "downloads": 272,
        "likes": 35,
        "lastModified": "2024-07-04"
      }
    },
    {
      "id": "mediaspeech",
      "name": "MediaSpeech",
      "type": "dataset",
      "country": "INTL",
      "org": "ymoslem",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ymoslem/MediaSpeech"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "MediaSpeech is a dataset of Arabic, French, Spanish, and Turkish media speech built with the purpose of testing Automated Speech Recognition.",
      "metrics": {
        "downloads": 271,
        "likes": 16,
        "lastModified": "2024-03-25"
      }
    },
    {
      "id": "islamic-sciences",
      "name": "islamic sciences",
      "type": "dataset",
      "country": "INTL",
      "org": "islamlab",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "quran",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/islamlab/islamic-sciences"
      },
      "dialects": [
        "classical"
      ],
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "The Islamic sciences other than Qur'an and hadith, as their authors wrote them.",
      "metrics": {
        "downloads": 270,
        "likes": 3,
        "lastModified": "2026-08-15"
      }
    },
    {
      "id": "jasmine",
      "name": "JASMINE",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/Jasmine-350M"
      },
      "size": "0.3B-6.7B",
      "notes": "Arabic GPT for few-shot learning, 400GB training data",
      "metrics": {
        "downloads": 270,
        "likes": 5,
        "lastModified": "2024-05-01"
      }
    },
    {
      "id": "katib-qwen3-5-0-8b-0-1",
      "name": "Katib-Qwen3.5-0.8B-0.1",
      "type": "ocr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Katib-Qwen3.5-0.8B-0.1"
      },
      "size": "0.8B",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Small Qwen3.5 0.8B Arabic OCR model.",
      "base_model": [
        "unsloth/qwen3.5-0.8b"
      ],
      "metrics": {
        "downloads": 270,
        "likes": 11,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "acegpt-13b-chat",
      "name": "AceGPT-13B-chat",
      "type": "llm",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/FreedomIntelligence/AceGPT-13B-chat"
      },
      "size": "13B",
      "on_device": false,
      "year": 2023,
      "tags": [
        "variants:6"
      ],
      "notes": "AceGPT is a fully fine-tuned generative text model collection based on LlaMA2, particularly in the Arabic language domain. (also: 6 variants)",
      "metrics": {
        "downloads": 268,
        "likes": 27,
        "lastModified": "2023-12-01"
      }
    },
    {
      "id": "ararest-arabic-restaurant-reviews-sentiment-analysis",
      "name": "AraRest-Arabic-Restaurant-Reviews-Sentiment-Analysis",
      "type": "llm",
      "country": "INTL",
      "org": "Abdu-GH",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/Abdu-GH/AraRest-Arabic-Restaurant-Reviews-Sentiment-Analysis"
      },
      "year": 2025,
      "notes": "This fine-tuned AraBERT model classifies Arabic restaurant reviews as Positive or Negative.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 267,
        "likes": 8,
        "lastModified": "2025-02-13"
      }
    },
    {
      "id": "cafe-algerian-codeswitch-speech",
      "name": "cafe algerian codeswitch speech",
      "type": "dataset",
      "country": "YE",
      "org": "Fatimah Emad Eldin",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/cafe-algerian-codeswitch-speech"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "This dataset contains Algerian Arabic and French code-switched speech.",
      "metrics": {
        "downloads": 265,
        "likes": 0,
        "lastModified": "2026-06-13"
      }
    },
    {
      "id": "aisa-ar-functioncall-think",
      "name": "AISA-AR-FunctionCall-Think",
      "type": "llm",
      "country": "INTL",
      "org": "TuwaiqAcademy",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/TuwaiqAcademy/AISA-AR-FunctionCall-Think"
      },
      "year": 2026,
      "notes": "A compact (270M-parameter) Arabic function-calling model t",
      "base_model": [
        "google/gemma-3-270m"
      ],
      "metrics": {
        "downloads": 264,
        "likes": 3,
        "lastModified": "2026-06-01"
      }
    },
    {
      "id": "qwen3-tts-ksa",
      "name": "qwen3 TTS KSA",
      "type": "tts",
      "country": "AE",
      "org": "Vadim Belsky",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/qwen3-TTS-KSA"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "A fine-tuned version of Qwen/Qwen3-TTS-12Hz-1.7B-Base for Saudi Arabian (Khaleeji/KSA) Arabic speech synthesis.",
      "base_model": [
        "qwen/qwen3-tts-12hz-1.7b-base"
      ],
      "metrics": {
        "downloads": 262,
        "likes": 6,
        "lastModified": "2026-03-17"
      }
    },
    {
      "id": "voho-saudi-speak-0-6b",
      "name": "voho-saudi-speak-0.6b",
      "type": "tts",
      "country": "SA",
      "org": "Voho AI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/VohoAI/voho-saudi-speak-0.6b"
      },
      "size": "596M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "gulf"
      ],
      "notes": "Voho 0.6B Saudi-dialect TTS.",
      "base_model": [
        "qwen/qwen3-0.6b"
      ],
      "metrics": {
        "downloads": 262,
        "likes": 0,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "openslr-quranic-asr",
      "name": "openslr quranic asr",
      "type": "dataset",
      "country": "INTL",
      "org": "Sadique5",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Sadique5/openslr_quranic_asr"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Quranic speech recognition dataset with 16 kHz audio, durations, transcripts and precomputed Whisper features.",
      "metrics": {
        "downloads": 261,
        "likes": 3,
        "lastModified": "2024-09-05"
      }
    },
    {
      "id": "whisperv3-tunisian-codeswitch",
      "name": "Whisperv3 tunisian codeswitch",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Whisperv3-tunisian-codeswitch"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "year": 2026,
      "notes": "Whisper-large-v3 full fine-tune for Tunisian Arabic ↔ French/English code-switched ASR (NADI 2026 shared task, subtask 1.3).",
      "base_model": [
        "oddadmix/whisper-large-v3-tunisian-codeswitch-asr-v2"
      ],
      "metrics": {
        "downloads": 260,
        "likes": 1,
        "lastModified": "2026-08-02"
      }
    },
    {
      "id": "arabic-guardrail",
      "name": "arabic guardrail",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-guardrail"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "message and which of 12 safety classes it belongs to.",
      "metrics": {
        "downloads": 259,
        "likes": 0,
        "lastModified": "2026-08-30"
      }
    },
    {
      "id": "fasttext-ar-vectors",
      "name": "fasttext ar vectors",
      "type": "llm",
      "country": "INTL",
      "org": "Meta",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/fasttext-ar-vectors"
      },
      "year": 2023,
      "notes": "fastText is an open-source, free, lightweight library that allows users to learn text representations and text classifiers.",
      "metrics": {
        "downloads": 259,
        "likes": 6,
        "lastModified": "2023-06-03"
      }
    },
    {
      "id": "masri-audio-mega",
      "name": "MasriAudio-Mega",
      "type": "dataset",
      "country": "EG",
      "org": "Mohamed Gomaa",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedGomaa30/MasriAudio-Mega-v0"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Large Egyptian Arabic speech corpus for ASR and TTS training.",
      "metrics": {
        "downloads": 259,
        "likes": 2,
        "lastModified": "2026-02-05"
      }
    },
    {
      "id": "nawah-dialect-bert-6m",
      "name": "Nawah Dialect BERT 6M",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-Dialect-BERT-6M"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "size": "6M",
      "on_device": true,
      "year": 2026,
      "notes": "A 5.98M-parameter Arabic dialect identifier.",
      "base_model": [
        "oddadmix/nawah-bert-6m-v2"
      ],
      "metrics": {
        "downloads": 259,
        "likes": 2,
        "lastModified": "2026-09-10"
      }
    },
    {
      "id": "gliner-arabic",
      "name": "GLiNER Arabic",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/gliner_arabic-v2.1"
      },
      "notes": "Named Entity Recognition - Plug-and-play Arabic NER (NAMAA-Space)",
      "base_model": [
        "urchade/gliner_multi-v2.1"
      ],
      "metrics": {
        "downloads": 257,
        "likes": 22,
        "lastModified": "2025-04-13"
      }
    },
    {
      "id": "egyptianmmlu",
      "name": "EgyptianMMLU",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/EgyptianMMLU"
      },
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "MMLU translated into Egyptian Arabic, released with Nile-Chat.",
      "metrics": {
        "downloads": 255,
        "likes": 0,
        "lastModified": "2025-06-02"
      }
    },
    {
      "id": "arabic-manuscript-collection",
      "name": "Arabic Manuscript Collection",
      "type": "dataset",
      "country": "INTL",
      "org": "TheSeniorTeam",
      "license": "mixed-per-subset",
      "modality": "vision",
      "tasks": [
        "ocr",
        "handwriting"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TheSeniorTeam/Arabic_Manuscript_Collection_Dataset"
      },
      "year": 2026,
      "notes": "Seven Arabic handwritten text recognition subsets unified in one format.",
      "metrics": {
        "downloads": 254,
        "likes": 0,
        "lastModified": "2026-09-12"
      }
    },
    {
      "id": "sada",
      "name": "SADA",
      "type": "dataset",
      "country": "SA",
      "org": "SDAIA",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/m6011/sada2022"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi Audio Dataset: 668h from 57 TV shows, multi-dialect (SDAIA)",
      "metrics": {
        "downloads": 254,
        "likes": 3,
        "lastModified": "2024-09-12"
      }
    },
    {
      "id": "dialectal-arabic-mmlu",
      "name": "Dialectal Arabic MMLU",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/Dialectal-Arabic-MMLU"
      },
      "size": "10K-100K rows",
      "year": 2025,
      "dialects": [
        "mixed"
      ],
      "notes": "Human-translated MMLU extended to five Arabic dialects plus MSA.",
      "metrics": {
        "downloads": 252,
        "likes": 2,
        "lastModified": "2026-09-28"
      }
    },
    {
      "id": "egypt-legal-corpus",
      "name": "egypt legal corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "dataflare",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/dataflare/egypt-legal-corpus"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "A comprehensive collection of Egyptian legal texts, meticulously extracted and tokenized for Natural Language Processing.",
      "metrics": {
        "downloads": 251,
        "likes": 4,
        "lastModified": "2026-01-19"
      }
    },
    {
      "id": "ultradata-math-qa-ar",
      "name": "ultradata math qa ar",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/ultradata-math-qa-ar"
      },
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "Arabic translation of the English portion of UltraData-Math, config UltraData-Math-L3-QA-Synthetic.",
      "metrics": {
        "downloads": 251,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "quran-nazimali",
      "name": "quran",
      "type": "dataset",
      "country": "INTL",
      "org": "nazimali",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "ner",
        "translation",
        "embedding",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nazimali/quran"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "The Quran with metadata, translations, and multiple Arabic text (can use specific types for embeddings, search, classification, and display).",
      "metrics": {
        "downloads": 250,
        "likes": 20,
        "lastModified": "2024-09-08"
      }
    },
    {
      "id": "sanad",
      "name": "SANAD",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/SANAD"
      },
      "year": 2019,
      "dialects": [
        "msa"
      ],
      "notes": "Single-label Arabic news articles dataset for text classification.",
      "metrics": {
        "downloads": 250,
        "likes": 3,
        "lastModified": "2024-05-07"
      }
    },
    {
      "id": "ufal-north-levantine",
      "name": "UFAL Parallel Corpus of North Levantine",
      "type": "dataset",
      "country": "INTL",
      "org": "UFAL Charles University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FrancophonIA/UFAL_Parallel_Corpus_of_North_Levantine_1.0"
      },
      "dialects": [
        "lev"
      ],
      "notes": "North Levantine to English parallel corpus (Levant).",
      "metrics": {
        "downloads": 250,
        "likes": 0,
        "lastModified": "2025-03-30"
      }
    },
    {
      "id": "duwatbench",
      "name": "DuwatBench",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/DuwatBench"
      },
      "year": 2026,
      "dialects": [
        "msa",
        "classical"
      ],
      "notes": "Benchmark on Arabic calligraphy images bridging language and visual heritage for multimodal models.",
      "metrics": {
        "downloads": 249,
        "likes": 6,
        "lastModified": "2026-01-28"
      }
    },
    {
      "id": "dziribert-sentiment",
      "name": "DziriBERT Sentiment",
      "type": "llm",
      "country": "DZ",
      "org": "alger-ia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/alger-ia/dziribert_sentiment"
      },
      "size": "124M",
      "dialects": [
        "magh"
      ],
      "notes": "DziriBERT fine-tuned for Algerian sentiment analysis (Algeria).",
      "metrics": {
        "downloads": 248,
        "likes": 9,
        "lastModified": "2023-04-02"
      }
    },
    {
      "id": "whisper-arabic-small",
      "name": "Whisper Arabic (small)",
      "type": "asr",
      "country": "INTL",
      "org": "ayoubkirouane",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/ayoubkirouane/whisper-small-ar"
      },
      "notes": "Whisper-small fine-tuned for Arabic language",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 247,
        "likes": 4,
        "lastModified": "2023-09-19"
      }
    },
    {
      "id": "admd",
      "name": "ADMD",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/ADMD"
      },
      "size": "<1K rows",
      "year": 2025,
      "notes": "The Arabic Depth Mini Dataset (ADMD) is a compact yet highly challenging dataset designed to evaluate Arabic language models across diverse domains.",
      "metrics": {
        "downloads": 245,
        "likes": 1,
        "lastModified": "2025-02-13"
      }
    },
    {
      "id": "aracast-text",
      "name": "aracast text",
      "type": "dataset",
      "country": "INTL",
      "org": "Asas AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "text-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/asas-ai/aracast_text"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "16.3K plain-text Arabic transcripts for text generation (135 MB).",
      "metrics": {
        "downloads": 245,
        "likes": 0,
        "lastModified": "2024-07-01"
      }
    },
    {
      "id": "aqar-fm-saudi-real-estate-listings",
      "name": "Aqar.fm Saudi Real Estate Listings",
      "type": "dataset",
      "country": "INTL",
      "org": "afaskar",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "real-estate",
        "tabular"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/afaskar/Aqar.fm-Saudi-Real-Estate-Listings"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "6,252 real estate listings (sales, rentals, auctions) scraped from Aqar.fm covering regions of Saudi Arabia.",
      "metrics": {
        "downloads": 243,
        "likes": 3,
        "lastModified": "2026-01-02"
      }
    },
    {
      "id": "bimedix-bi",
      "name": "BiMediX-Bi",
      "type": "llm",
      "country": "INTL",
      "org": "BiMediX",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "medical",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/BiMediX/BiMediX-Bi"
      },
      "year": 2024,
      "notes": "Bilingual English-Arabic medical Mixtral-8x7B MoE trained on BiMed1.3M for MCQA, closed QA and chat.",
      "metrics": {
        "downloads": 242,
        "likes": 6,
        "lastModified": "2024-04-10"
      }
    },
    {
      "id": "arabic-audio-collection-sudanese-sudan-podcast",
      "name": "arabic audio collection sudanese sudan podcast",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "chat",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-sudanese-sudan-podcast"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Sudan Podcast Arabic Speech Dataset is a large-scale Sudanese Arabic speech corpus containing approximately 132 hours of speech recordings.",
      "metrics": {
        "downloads": 241,
        "likes": 1,
        "lastModified": "2026-08-07"
      }
    },
    {
      "id": "arabic-msa-25k-saudi-tashkeel",
      "name": "Arabic MSA 25K Saudi Male Tashkeel",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/arabic-msa-25k-saudi-male-tashkeel"
      },
      "size": "25k pairs",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "25,000 fully diacritized MSA text-audio pairs rendered with a single Saudi male neural voice at 48 kHz.",
      "metrics": {
        "downloads": 241,
        "likes": 11,
        "lastModified": "2026-04-20"
      }
    },
    {
      "id": "nawah-math-reasoning",
      "name": "Nawah Math Reasoning",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-Math-Reasoning"
      },
      "year": 2026,
      "notes": "A 51.8M-parameter Arabic math reasoning model.",
      "base_model": [
        "oddadmix/50m-2048-emhotob"
      ],
      "metrics": {
        "downloads": 241,
        "likes": 2,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "darjacore-algerian-darja-forum-posts",
      "name": "DarjaCore Algerian Darja forum posts",
      "type": "dataset",
      "country": "INTL",
      "org": "DarjaCore",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/DarjaCore/algerian-darja-forum-posts"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Corpus of Algerian Darja forum posts released by the DarjaCore Algerian-language AI group.",
      "metrics": {
        "downloads": 239,
        "likes": 2,
        "lastModified": "2026-09-09"
      }
    },
    {
      "id": "semantic-ar-qwen-embed-0-6b",
      "name": "Semantic-Ar-Qwen-Embed-0.6B",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Semantic-Ar-Qwen-Embed-0.6B"
      },
      "size": "596M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic semantic embedding model built on Qwen3-Embedding 0.6B.",
      "base_model": [
        "qwen/qwen3-embedding-0.6b"
      ],
      "metrics": {
        "downloads": 239,
        "likes": 8,
        "lastModified": "2025-09-07"
      }
    },
    {
      "id": "arabic-marbert-dialect-identification-city",
      "name": "arabic MARBERT dialect identification city",
      "type": "llm",
      "country": "INTL",
      "org": "Ammar-alhaj-ali",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/Ammar-alhaj-ali/arabic-MARBERT-dialect-identification-city"
      },
      "year": 2022,
      "notes": "MARBERT fine-tuned for Arabic dialect identification at city level (26 city labels).",
      "dialects": [
        "mixed"
      ],
      "metrics": {
        "downloads": 238,
        "likes": 12,
        "lastModified": "2022-08-09"
      }
    },
    {
      "id": "llamalens-arabic",
      "name": "LlamaLens Arabic",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "classification",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/LlamaLens-Arabic"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic instruction data for 18 news and social media analysis tasks used to train LlamaLens.",
      "metrics": {
        "downloads": 238,
        "likes": 2,
        "lastModified": "2025-03-13"
      }
    },
    {
      "id": "nawah-bert-6m-bilingual",
      "name": "Nawah BERT 6M bilingual",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-BERT-6M-bilingual"
      },
      "size": "6M",
      "on_device": true,
      "year": 2026,
      "notes": "Same architecture as Nawah-BERT-6M-v2 (hidden 128, 8 layers, 2 heads, 5,993,600 params) but pretrained on a genuinely balanced bilingual corpus.",
      "metrics": {
        "downloads": 238,
        "likes": 0,
        "lastModified": "2026-09-06"
      }
    },
    {
      "id": "sudan-mm",
      "name": "Sudan-MM",
      "type": "dataset",
      "country": "SD",
      "org": "IndabaX Sudan",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IndabaXSudan/Sudan-MM"
      },
      "dialects": [
        "sudanese"
      ],
      "year": 2026,
      "notes": "First public multimodal dataset for Sudanese Arabic.",
      "metrics": {
        "downloads": 237,
        "likes": 2,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "hubert-arabic-spoken-dialect-classifier",
      "name": "hubert arabic spoken dialect classifier",
      "type": "asr",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimAmin/hubert-arabic-spoken-dialect-classifier"
      },
      "dialects": [
        "egy",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of facebook/hubert-base-ls960 for Arabic spoken dialect classification.",
      "base_model": [
        "facebook/hubert-base-ls960"
      ],
      "metrics": {
        "downloads": 236,
        "likes": 1,
        "lastModified": "2025-05-11"
      }
    },
    {
      "id": "arabic-pos-dialect",
      "name": "Arabic POS Dialect",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/arabic_pos_dialect"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "POS tagging in Arabic dialects",
      "metrics": {
        "downloads": 235,
        "likes": 12,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "multi-dialect-bert-base-arabic",
      "name": "multi dialect bert base arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Bashar Talafha",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/bashar-talafha/multi-dialect-bert-base-arabic"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2021,
      "notes": "This is a repository of Multi-dialect Arabic BERT model.",
      "metrics": {
        "downloads": 235,
        "likes": 8,
        "lastModified": "2021-05-19"
      }
    },
    {
      "id": "tunisian-derja-dataset",
      "name": "Tunisian Derja Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "LINAGORA",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/linagora/Tunisian_Derja_Dataset"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian dialect text documents used for continual pre-training of Jais towards Tunisian.",
      "metrics": {
        "downloads": 234,
        "likes": 5,
        "lastModified": "2025-09-15"
      }
    },
    {
      "id": "multinativqa",
      "name": "MultiNativQA",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/MultiNativQA"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "mixed"
      ],
      "notes": "Multilingual native culturally aligned natural queries with manual answers, includes Arabic.",
      "metrics": {
        "downloads": 233,
        "likes": 2,
        "lastModified": "2026-03-31"
      }
    },
    {
      "id": "laya-ara-rag",
      "name": "laya ara rag",
      "type": "llm",
      "country": "INTL",
      "org": "Wouze",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Wouze/laya-ara-rag"
      },
      "year": 2026,
      "notes": "Arabic short-list reranker — passage relevance and k≤12 ranking on laya-multilingual.",
      "base_model": [
        "convaiinnovations/laya-multilingual"
      ],
      "metrics": {
        "downloads": 232,
        "likes": 3,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "sofelia-tts",
      "name": "Sofelia TTS",
      "type": "tts",
      "country": "INTL",
      "org": "hamdallah",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/hamdallah/Sofelia-TTS"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Palestinian Arabic TTS model fine-tuned on top of YatharthS/MiraTTS.",
      "base_model": [
        "yatharths/miratts"
      ],
      "metrics": {
        "downloads": 232,
        "likes": 4,
        "lastModified": "2026-01-19"
      }
    },
    {
      "id": "gpt2-small-arabic-poetry",
      "name": "gpt2 small arabic poetry",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/gpt2-small-arabic-poetry"
      },
      "year": 2023,
      "notes": "Fine-tuned model of Arabic poetry dataset based on gpt2-small-arabic.",
      "metrics": {
        "downloads": 231,
        "likes": 5,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "arabic-glm-ocr-v1",
      "name": "Arabic GLM OCR v1",
      "type": "ocr",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "chat",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/Arabic-GLM-OCR-v1"
      },
      "year": 2026,
      "notes": "Arabic image text to text model fine-tuned from zai-org/GLM-OCR.",
      "base_model": [
        "zai-org/glm-ocr"
      ],
      "metrics": {
        "downloads": 230,
        "likes": 9,
        "lastModified": "2026-03-29"
      }
    },
    {
      "id": "aragpt2",
      "name": "AraGPT2",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/aubmindlab/aragpt2-mega"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "1.5B",
      "notes": "GPT-2 for Arabic text generation",
      "on_device": true,
      "metrics": {
        "downloads": 225,
        "likes": 12,
        "lastModified": "2024-10-24"
      }
    },
    {
      "id": "finewiki-ar-checked",
      "name": "finewiki ar checked",
      "type": "dataset",
      "country": "INTL",
      "org": "SultanR",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/finewiki-ar-checked"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic translation of the English subset of FineWiki (6.61M Wikipedia pages, extracted from HTML dumps with templates rendered and math and tables preserved).",
      "metrics": {
        "downloads": 223,
        "likes": 0,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "arabic-exams-redux",
      "name": "Arabic EXAMS-Redux",
      "type": "benchmark",
      "country": "AE",
      "org": "Inception Labs",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/inceptlabs/Arabic_EXAMS-Redux"
      },
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Corrected and text-repaired version of OALL/Arabic_EXAMS.",
      "metrics": {
        "downloads": 220,
        "likes": 1,
        "lastModified": "2026-05-04"
      }
    },
    {
      "id": "doda-audio",
      "name": "DODa Audio",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/DODa-audio-dataset"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Audio recordings of Darija Open Dataset sentences, gated on Hugging Face.",
      "metrics": {
        "downloads": 220,
        "likes": 25,
        "lastModified": "2025-02-07"
      }
    },
    {
      "id": "egyptalk-asr-v2",
      "name": "EgypTalk-ASR-v2",
      "type": "asr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/EgypTalk-ASR-v2"
      },
      "dialects": [
        "egy"
      ],
      "notes": "High-performance ASR for Egyptian Arabic",
      "base_model": [
        "nvidia/stt_ar_fastconformer_hybrid_large_pcd_v1.0"
      ],
      "metrics": {
        "downloads": 220,
        "likes": 17,
        "lastModified": "2025-08-09"
      }
    },
    {
      "id": "arabic-audio-collection-moroccan-wak3i",
      "name": "arabic audio collection moroccan wak3i",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-moroccan-wak3i"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Mak3i Moroccan Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 70 hours of speech recordings.",
      "metrics": {
        "downloads": 219,
        "likes": 0,
        "lastModified": "2026-06-30"
      }
    },
    {
      "id": "marefa-ner",
      "name": "marefa ner",
      "type": "llm",
      "country": "INTL",
      "org": "marefa-nlp",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/marefa-nlp/marefa-ner"
      },
      "year": 2021,
      "notes": "Marefa Arabic Named Entity Recognition Model",
      "metrics": {
        "downloads": 219,
        "likes": 25,
        "lastModified": "2021-12-04"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-da-ner",
      "name": "bert base arabic camelbert da ner",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-da-ner"
      },
      "year": 2021,
      "notes": "For the fine-tuning, we used the ANERcorp dataset.",
      "metrics": {
        "downloads": 213,
        "likes": 0,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "english-arabic-speech-translation",
      "name": "english arabic speech translation",
      "type": "dataset",
      "country": "INTL",
      "org": "ammagra",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech-translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ammagra/english-arabic-speech-translation"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "302K English-Arabic speech translation rows with audio, Arabic sentence and translation (Common Voice style).",
      "metrics": {
        "downloads": 213,
        "likes": 4,
        "lastModified": "2025-01-13"
      }
    },
    {
      "id": "astd",
      "name": "ASTD",
      "type": "dataset",
      "country": "EG",
      "org": "Cairo University",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/mahmoudnabil/ASTD",
        "hf": "https://huggingface.co/datasets/arbml/ASTD",
        "paper": "https://aclanthology.org/D15-1299.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,006 sentences",
      "year": 2015,
      "notes": "10k Arabic sentiment tweets classified into four classes subjective positive, subjective negative, subjective mixed, and objective",
      "metrics": {
        "downloads": 212,
        "likes": 2,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "t5-arabic-base",
      "name": "t5 arabic base",
      "type": "llm",
      "country": "INTL",
      "org": "bakrianoo",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "text-to-text"
      ],
      "links": {
        "hf": "https://huggingface.co/bakrianoo/t5-arabic-base"
      },
      "year": 2021,
      "tags": [
        "variants:2"
      ],
      "notes": "Smaller T5 base model targeting only Arabic and English, offered as an alternative to google/mt5-base.",
      "metrics": {
        "downloads": 212,
        "likes": 0,
        "lastModified": "2021-06-26"
      }
    },
    {
      "id": "arabic-billion-words",
      "name": "Arabic Billion Words",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-billion-words"
      },
      "notes": "Abu El-Khair corpus: 5M+ articles, 1.5B+ words",
      "metrics": {
        "downloads": 211,
        "likes": 12,
        "lastModified": "2023-12-16"
      }
    },
    {
      "id": "nemotron-3-5-asr-streaming-0-6b-jordanian",
      "name": "nemotron-3.5-asr-streaming-0.6b-jordanian",
      "type": "asr",
      "country": "JO",
      "org": "Rama Bashar",
      "license": "openmdw-1.1",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/RamaBashar22/nemotron-3.5-asr-streaming-0.6b-jordanian"
      },
      "size": "638M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "lev"
      ],
      "notes": "Streaming Nemotron ASR 0.6B tuned for Jordanian Arabic.",
      "base_model": [
        "nvidia/nemotron-3.5-asr-streaming-0.6b"
      ],
      "metrics": {
        "downloads": 211,
        "likes": 4,
        "lastModified": "2026-08-30"
      }
    },
    {
      "id": "qed-corpus-qcri-educational-domain",
      "name": "QED Corpus (QCRI Educational Domain)",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/qedcorpus/",
        "hf": "https://huggingface.co/datasets/Helsinki-NLP/qed_amara"
      },
      "year": 2014,
      "notes": "Open multilingual collection of subtitles for educational videos (formerly QCRI AMARA corpus); includes Arabic-English parallel data.",
      "metrics": {
        "downloads": 211,
        "likes": 5,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "egy-arabic-qwen3-tts-12hz-1-7b-base",
      "name": "Egy Arabic Qwen3 TTS 12Hz 1.7B Base",
      "type": "tts",
      "country": "INTL",
      "org": "itshamdi404",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/itshamdi404/Egy_Arabic_Qwen3-TTS-12Hz-1.7B-Base"
      },
      "dialects": [
        "egy"
      ],
      "size": "1.7B",
      "on_device": false,
      "year": 2026,
      "notes": "A fine-tuned Qwen3-TTS 1.7B model specialized in generating Egyptian Arabic (Masri) speech with a natural, conversational tone.",
      "base_model": [
        "qwen/qwen3-tts-12hz-1.7b-base"
      ],
      "metrics": {
        "downloads": 210,
        "likes": 4,
        "lastModified": "2026-03-10"
      }
    },
    {
      "id": "gutenberg-arabic-ocr-html-pages",
      "name": "Gutenberg Arabic OCR HTML Pages",
      "type": "dataset",
      "country": "YE",
      "org": "Fatimah Emad Eldin",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "translation",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/Gutenberg-Arabic-OCR-HTML-Pages"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The Gutenberg Arabic HTML-Page Dataset is a large-scale, synthetically generated dataset designed for training and evaluating document understanding.",
      "metrics": {
        "downloads": 210,
        "likes": 3,
        "lastModified": "2025-09-21"
      }
    },
    {
      "id": "aradice-openbookqa",
      "name": "AraDiCE-OpenBookQA",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AraDiCE-OpenBookQA"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "AraDiCE OpenBookQA: post-edited OpenBookQA test set in English, MSA and dialects for dialectal and cultural LLM evaluation.",
      "dialects": [
        "msa",
        "mixed"
      ],
      "metrics": {
        "downloads": 208,
        "likes": 0,
        "lastModified": "2024-11-03"
      }
    },
    {
      "id": "dehatebert-mono-arabic",
      "name": "dehatebert mono arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Hate-speech-CNERG",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/Hate-speech-CNERG/dehatebert-mono-arabic"
      },
      "year": 2021,
      "notes": "This model is used detecting hatespeech in Arabic language.",
      "metrics": {
        "downloads": 208,
        "likes": 4,
        "lastModified": "2021-09-25"
      }
    },
    {
      "id": "alpaca-gpt4-arabic",
      "name": "Alpaca-GPT4 Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/alpaca-gpt4-arabic"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic translation of Alpaca GPT-4 instruction data from the AceGPT team.",
      "metrics": {
        "downloads": 207,
        "likes": 12,
        "lastModified": "2023-08-06"
      }
    },
    {
      "id": "arabic-tashkeel-dataset",
      "name": "arabic tashkeel dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdou",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdou/arabic-tashkeel-dataset"
      },
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "1.5M Arabic sentences pairing non-vocalized and vocalized (tashkeel) text from Tashkeela, Shamela and other sources.",
      "metrics": {
        "downloads": 205,
        "likes": 7,
        "lastModified": "2024-10-28"
      }
    },
    {
      "id": "sadeeddiac-25",
      "name": "SadeedDiac-25",
      "type": "benchmark",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/SadeedDiac-25",
        "paper": "https://arxiv.org/abs/2504.21635"
      },
      "size": "1K-10K rows",
      "year": 2025,
      "dialects": [
        "msa",
        "classical"
      ],
      "notes": "Benchmark for Arabic diacritization covering MSA and classical text.",
      "metrics": {
        "downloads": 205,
        "likes": 7,
        "lastModified": "2025-05-20"
      }
    },
    {
      "id": "xor-tydi-qa",
      "name": "XOR-TyDi QA",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Washington",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://nlp.cs.washington.edu/xorqa/index.html",
        "hf": "https://huggingface.co/datasets/akariasai/xor_tydi_qa",
        "paper": "https://doi.org/10.18653/v1/2021.naacl-main.46"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,235 sentences",
      "year": 2021,
      "tags": [
        "multilingual"
      ],
      "notes": "XOR-TyDi QA brings together for the first time information-seeking questions, open-retrieval QA, and multilingual QA to create a multilingual open-retrieval.",
      "metrics": {
        "downloads": 205,
        "likes": 3,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "nadi-2026-adi20-micro",
      "name": "NADI 2026 ADI20 micro",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/NADI_2026_ADI20_micro"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This is a smaller version of the ADI-20/ADI-17 Arabic dialect identification dataset use for the NADI 2026 shared task.",
      "metrics": {
        "downloads": 204,
        "likes": 2,
        "lastModified": "2026-06-16"
      }
    },
    {
      "id": "camelbert-msa-qalb14-ged-13",
      "name": "camelbert msa qalb14 ged 13",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/camelbert-msa-qalb14-ged-13"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "notes": "For the fine-tuning, we used the QALB-2014 dataset.",
      "metrics": {
        "downloads": 203,
        "likes": 1,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "aradice-culture",
      "name": "AraDiCE-Culture",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AraDiCE-Culture",
        "paper": "https://aclanthology.org/2025.coling-main.283/"
      },
      "dialects": [
        "mixed",
        "egy",
        "lev",
        "gulf"
      ],
      "size": "180 sentences",
      "year": 2024,
      "notes": "Dataset of 180 culturally specific questions by hiring native an- notators from the Gulf, Egypt, and the Levant re- gions to generate seed questions.",
      "metrics": {
        "downloads": 202,
        "likes": 1,
        "lastModified": "2024-11-05"
      }
    },
    {
      "id": "darija-stt-mix",
      "name": "darija stt mix",
      "type": "dataset",
      "country": "INTL",
      "org": "ayoubkirouane",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ayoubkirouane/darija-stt-mix"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The Darija Speech To Text Dataset is a comprehensive collection designed to support speech recognition tasks for the Darija dialect.",
      "metrics": {
        "downloads": 201,
        "likes": 6,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "emotone-ar",
      "name": "emotone ar",
      "type": "dataset",
      "country": "INTL",
      "org": "emotone-ar-cicling2017",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "emotion"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/emotone-ar-cicling2017/emotone_ar"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "EmoTone-Ar: 10,065 Arabic tweets labelled for emotional tone.",
      "metrics": {
        "downloads": 201,
        "likes": 14,
        "lastModified": "2024-08-08"
      }
    },
    {
      "id": "jais-13b",
      "name": "jais 13b",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-13b"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "13B",
      "on_device": false,
      "year": 2024,
      "tags": [
        "variants:2"
      ],
      "notes": "Jais-13B: bilingual Arabic-English 13B decoder-only LLM from Inception (arXiv 2308.16149).",
      "metrics": {
        "downloads": 201,
        "likes": 183,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "metrec",
      "name": "MetRec",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/zaidalyafeai/MetRec",
        "hf": "https://huggingface.co/datasets/Zaid/metrec",
        "paper": "https://www.sciencedirect.com/science/article/pii/S2352340920313792"
      },
      "dialects": [
        "classical"
      ],
      "size": "47,124 sentences",
      "year": 2020,
      "notes": "More than 40K of verses with their meters",
      "metrics": {
        "downloads": 200,
        "likes": 4,
        "lastModified": "2024-08-08"
      }
    },
    {
      "id": "arabic-function-calling",
      "name": "Arabic_Function_Calling",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Arabic_Function_Calling"
      },
      "notes": "First Arabic function calling dataset (50K+ samples)",
      "metrics": {
        "downloads": 199,
        "likes": 60,
        "lastModified": "2025-12-14"
      }
    },
    {
      "id": "lub-saudi-arabic-intent",
      "name": "LUB Saudi Arabic Intent",
      "type": "dataset",
      "country": "SA",
      "org": "NABA AI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "intent",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NABA-AI/LUB-Saudi-Arabic-Intent"
      },
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Synthetic Saudi Arabic dataset for contextual intent, emotion and sarcasm.",
      "metrics": {
        "downloads": 199,
        "likes": 0,
        "lastModified": "2026-09-24"
      }
    },
    {
      "id": "saudi-dialect-conversations",
      "name": "Saudi Najdi Dialect Conversations",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/saudi-dialect-conversations"
      },
      "size": "3.5k conversations",
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "3,545 multi-turn conversations in Saudi Najdi dialect.",
      "metrics": {
        "downloads": 199,
        "likes": 20,
        "lastModified": "2026-02-18"
      }
    },
    {
      "id": "aradice-boolq",
      "name": "AraDiCE-BoolQ",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "reading-comprehension"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AraDiCE-BoolQ"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "AraDiCE BoolQ: post-edited BoolQ reading-comprehension test set in MSA and dialects for dialectal and cultural LLM evaluation.",
      "dialects": [
        "msa",
        "mixed"
      ],
      "metrics": {
        "downloads": 198,
        "likes": 0,
        "lastModified": "2024-11-03"
      }
    },
    {
      "id": "darja-gpt-50m",
      "name": "darja gpt 50m",
      "type": "llm",
      "country": "INTL",
      "org": "Touati Kamel",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/touati-kamel/darja-gpt-50m"
      },
      "dialects": [
        "magh"
      ],
      "size": "50M",
      "on_device": true,
      "year": 2026,
      "notes": "Domain-conditioned 50M causal LM for Algerian Darija with LoRA adapters and classifier-free guidance.",
      "metrics": {
        "downloads": 197,
        "likes": 0,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "egyptian-sft-mixture",
      "name": "Egyptian SFT Mixture",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/Egyptian-SFT-Mixture"
      },
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "Supervised fine-tuning samples for Nile-Chat in Arabic and Latin (Arabizi) script.",
      "metrics": {
        "downloads": 196,
        "likes": 6,
        "lastModified": "2025-06-17"
      }
    },
    {
      "id": "kasbahtts-v0",
      "name": "KasbahTTS-V0",
      "type": "tts",
      "country": "INTL",
      "org": "MenaVoice",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/MenaVoice/KasbahTTS-V0"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Zero-shot voice-cloning Algerian Arabic (Darija) TTS built on F5-TTS and Habibi-TTS.",
      "metrics": {
        "downloads": 195,
        "likes": 10,
        "lastModified": "2026-07-10"
      }
    },
    {
      "id": "aha-memes",
      "name": "AHA MEMES",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-4.0",
      "modality": "multimodal",
      "tasks": [
        "vision",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/AHA-MEMES"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Hateful memes carry their meaning in the interaction between an image and the text laid over it.",
      "metrics": {
        "downloads": 192,
        "likes": 0,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "commonvoice-12-0-arabic-voice-converted",
      "name": "commonvoice 12.0 arabic voice converted",
      "type": "dataset",
      "country": "INTL",
      "org": "xmodar",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/xmodar/commonvoice-12.0-arabic-voice-converted"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "This dataset is derived from the Common Voice Arabic Corpus 12.0 and includes automatically diacritized transcriptions and phoneme representations.",
      "metrics": {
        "downloads": 191,
        "likes": 8,
        "lastModified": "2024-12-17"
      }
    },
    {
      "id": "quran-md-words",
      "name": "quran md words",
      "type": "dataset",
      "country": "INTL",
      "org": "Buraaq",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Buraaq/quran-md-words"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Word-level split of Quran-MD, a multimodal Quran dataset aligning text, linguistic annotation and recitation audio.",
      "metrics": {
        "downloads": 190,
        "likes": 13,
        "lastModified": "2026-01-27"
      }
    },
    {
      "id": "acegpt-13b",
      "name": "AceGPT-13B",
      "type": "llm",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/FreedomIntelligence/AceGPT-13B"
      },
      "base_model": [
        "meta-llama/llama-2-13b-hf"
      ],
      "size": "13B",
      "on_device": false,
      "year": 2023,
      "tags": [
        "variants:6"
      ],
      "notes": "AceGPT is a fully fine-tuned generative text model collection based on LlaMA2, particularly in the Arabic language domain. (also: 6 variants)",
      "metrics": {
        "downloads": 189,
        "likes": 10,
        "lastModified": "2023-12-01"
      }
    },
    {
      "id": "hubert-egyptian-arabic",
      "name": "HuBERT Egyptian Arabic",
      "type": "asr",
      "country": "EG",
      "org": "Omar Adel, Alexandria University",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/omarxadel/hubert-large-arabic-egyptian"
      },
      "dialects": [
        "egy"
      ],
      "notes": "HuBERT fine-tuned for Egyptian dialect on MGB-3",
      "metrics": {
        "downloads": 189,
        "likes": 22,
        "lastModified": "2023-03-19"
      }
    },
    {
      "id": "qari-ocr-0-1-vl-2b-instruct",
      "name": "Qari-OCR-0.1-VL-2B-Instruct",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Qari-OCR-0.1-VL-2B-Instruct"
      },
      "notes": "Qwen2 VL 2B fine-tuned for Arabic OCR",
      "base_model": [
        "unsloth/qwen2-vl-2b-instruct-unsloth-bnb-4bit"
      ],
      "metrics": {
        "downloads": 189,
        "likes": 44,
        "lastModified": "2025-06-10"
      }
    },
    {
      "id": "al-lataif-al-musawwara-magazine-ocr",
      "name": "al lataif al musawwara magazine ocr",
      "type": "dataset",
      "country": "INTL",
      "org": "amrosama",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "instruction-tuning",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/amrosama/al-lataif-al-musawwara-magazine-ocr"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "This dataset contains rendered Arabic magazine page images from Al-Lataif Al-Musawwara paired with page-level text and line-level bounding boxes.",
      "metrics": {
        "downloads": 186,
        "likes": 9,
        "lastModified": "2026-07-25"
      }
    },
    {
      "id": "darija-hellaswag",
      "name": "DarijaHellaSwag",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation",
        "commonsense"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/DarijaHellaSwag"
      },
      "size": "1K-10K rows",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "HellaSwag translated into Moroccan Darija for commonsense evaluation.",
      "metrics": {
        "downloads": 186,
        "likes": 4,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "quran-speech-recognition-kaggle",
      "name": "Quran speech recognition kaggle",
      "type": "dataset",
      "country": "INTL",
      "org": "Nuwaisir",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Nuwaisir/Quran_speech_recognition_kaggle"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2022,
      "notes": "This dataset can be found in Kaggle",
      "metrics": {
        "downloads": 186,
        "likes": 5,
        "lastModified": "2022-02-20"
      }
    },
    {
      "id": "egyptian-arabic-400k",
      "name": "egyptian arabic 400k",
      "type": "dataset",
      "country": "INTL",
      "org": "geeeezx",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/geeeezx/egyptian-arabic-400k"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "408K Egyptian Arabic speech clips with transcript and speaker gender.",
      "metrics": {
        "downloads": 185,
        "likes": 4,
        "lastModified": "2025-08-10"
      }
    },
    {
      "id": "faseeh",
      "name": "Faseeh",
      "type": "llm",
      "country": "INTL",
      "org": "Abdulmohsena",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Abdulmohsena/Faseeh"
      },
      "year": 2026,
      "notes": "نموذج لغوي مصمم للترجمة إلى لسان عربي فصيح، لأن السائد حاليا في الترجمة هي العربية المستحدثة (العرنجية) لغة ظاهرها العربية، وباطنها الأفرنجية.",
      "metrics": {
        "downloads": 185,
        "likes": 5,
        "lastModified": "2026-09-22"
      }
    },
    {
      "id": "masrawy-translator",
      "name": "Masrawy Translator",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/masrawy-english-to-egyptian-arabic-translator-v2.9"
      },
      "dialects": [
        "egy"
      ],
      "notes": "EN→Egyptian Arabic - 150K+ rows, 10M+ tokens, Egyptian dialect",
      "base_model": [
        "helsinki-nlp/opus-mt-tc-big-en-ar"
      ],
      "metrics": {
        "downloads": 185,
        "likes": 20,
        "lastModified": "2025-01-27"
      }
    },
    {
      "id": "namaa-saudi-tts",
      "name": "NAMAA-Saudi-TTS",
      "type": "tts",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/NAMAA-Saudi-TTS"
      },
      "size": "536M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi-dialect TTS from NAMAA.",
      "base_model": [
        "resembleai/chatterbox"
      ],
      "metrics": {
        "downloads": 184,
        "likes": 51,
        "lastModified": "2026-01-29"
      }
    },
    {
      "id": "qadt",
      "name": "QADT",
      "type": "dataset",
      "country": "IQ",
      "org": "Mosul University",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/annealdahi/quran-recitation",
        "hf": "https://huggingface.co/datasets/obadx/qdat",
        "paper": "https://www.researchgate.net/publication/350785609_QDAT_A_data_set_for_Reciting_the_Quran"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,505 sentences",
      "year": 2021,
      "notes": "The audio files are manually annotated by expert to show the correctness of the Reciting the Quran with Tajwid according to three rules of recitation of Quran.",
      "metrics": {
        "downloads": 184,
        "likes": 0,
        "lastModified": "2025-09-09"
      }
    },
    {
      "id": "arabic-hadith",
      "name": "Arabic Hadith",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdo1Kamr",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdo1Kamr/Arabic_Hadith"
      },
      "dialects": [
        "classical"
      ],
      "year": 2021,
      "notes": "There are two files of Hadith, the first one for all hadith With Tashkil and Without Tashkel from the Nine Books that are 62,169 Hadith.",
      "metrics": {
        "downloads": 182,
        "likes": 7,
        "lastModified": "2021-08-21"
      }
    },
    {
      "id": "baybars",
      "name": "baybars",
      "type": "dataset",
      "country": "INTL",
      "org": "calfa-ai",
      "license": "etalab-2.0",
      "modality": "vision",
      "tasks": [
        "htr",
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/calfa-ai/baybars"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "BAYBARS: line-level handwritten text ground truth for Arabic historical manuscripts of the Sirat Baybars, with transcriptions.",
      "metrics": {
        "downloads": 182,
        "likes": 3,
        "lastModified": "2026-09-11"
      }
    },
    {
      "id": "egyptian-translation-dataset-2-9-openai-batch",
      "name": "egyptian translation dataset 2.9 openai batch",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/egyptian-translation-dataset-2.9-openai-batch"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "English-to-Egyptian-Arabic translation pairs, 135,282 rows.",
      "metrics": {
        "downloads": 182,
        "likes": 1,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "sadeed-tashkeela",
      "name": "Sadeed Tashkeela",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/Sadeed_Tashkeela"
      },
      "size": "1M-10M rows",
      "year": 2025,
      "dialects": [
        "classical"
      ],
      "notes": "Cleaned diacritization training corpus derived from Tashkeela, used to train Sadeed.",
      "metrics": {
        "downloads": 182,
        "likes": 16,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "araeurobert-210m",
      "name": "AraEuroBert-210M",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/AraEuroBert-210M"
      },
      "size": "212M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic embedding model fine-tuned from EuroBERT 210M.",
      "base_model": [
        "eurobert/eurobert-210m"
      ],
      "metrics": {
        "downloads": 181,
        "likes": 7,
        "lastModified": "2025-03-20"
      }
    },
    {
      "id": "arsentd-lev",
      "name": "ArSenTD-LEV",
      "type": "dataset",
      "country": "INTL",
      "org": "(1) MIT Computer Science and Artificial Intelligence Laboratory",
      "license": "other",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "http://oma-project.com/ArSenL/ArSenTD_Lev_Intro",
        "hf": "https://huggingface.co/datasets/ramybaly/arsentd_lev",
        "paper": "https://paperswithcode.com/paper/190601830"
      },
      "dialects": [
        "lev"
      ],
      "size": "4,000 sentences",
      "year": 2019,
      "notes": "ArSentD-LEV is a multi-topic corpus for target-based sentiment analysis in Arabic Levantine tweets",
      "metrics": {
        "downloads": 181,
        "likes": 5,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "darijabench",
      "name": "DarijaBench",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/DarijaBench"
      },
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Darija evaluation sets for summarization, translation and sentiment analysis, released with Atlas-Chat.",
      "metrics": {
        "downloads": 181,
        "likes": 4,
        "lastModified": "2025-02-17"
      }
    },
    {
      "id": "tarab",
      "name": "Tarab",
      "type": "dataset",
      "country": "INTL",
      "org": "drelhaj",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/Tarab"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "size": "10M–100M rows",
      "year": 2026,
      "notes": "It contains 2,557,311 verses and 13,509,336 tokens, spanning Classical Arabic, MSA, and six major regional dialect groups.",
      "metrics": {
        "downloads": 181,
        "likes": 4,
        "lastModified": "2026-02-25"
      }
    },
    {
      "id": "doda-darija-cosyvoice2",
      "name": "doda darija cosyvoice2",
      "type": "dataset",
      "country": "INTL",
      "org": "Jip7e",
      "license": "['cc-by-4.0']",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Jip7e/doda-darija-cosyvoice2"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "DODa Moroccan Darija speech dataset standardized and tokenized for CosyVoice2 TTS training, with Kaldi-style formats.",
      "metrics": {
        "downloads": 180,
        "likes": 1,
        "lastModified": "2026-08-29"
      }
    },
    {
      "id": "islamiceval2026-subtask2-submission",
      "name": "IslamicEval2026-Subtask2-Submission",
      "type": "benchmark",
      "country": "YE",
      "org": "Fatimah Emad Eldin",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/IslamicEval2026-Subtask2-Submission"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Code, notebooks, experiments and the system-description paper for the Namaa Community submission to Subtask 2 of IslamicEval 2026.",
      "metrics": {
        "downloads": 180,
        "likes": 0,
        "lastModified": "2026-08-15"
      }
    },
    {
      "id": "arcov-19",
      "name": "ArCOV-19",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University (bigIR)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/bigIR/ar_cov19"
      },
      "year": 2020,
      "notes": "Arabic COVID-19 Twitter dataset with rumor and propagation annotations from Qatar University bigIR.",
      "metrics": {
        "downloads": 177,
        "likes": 5,
        "lastModified": "2023-09-19"
      }
    },
    {
      "id": "barec-corpus",
      "name": "BAREC Corpus",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CAMeL-Lab/BAREC-Corpus-v1.0"
      },
      "notes": "Arabic Readability Assessment",
      "metrics": {
        "downloads": 177,
        "likes": 2,
        "lastModified": "2025-09-02"
      }
    },
    {
      "id": "oclar",
      "name": "OCLAR",
      "type": "dataset",
      "country": "INTL",
      "org": "Université de Poitiers",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "website": "http://archive.ics.uci.edu/ml/datasets/Opinion+Corpus+for+Lebanese+Arabic+Reviews+%28OCLAR%29#",
        "hf": "https://huggingface.co/datasets/community-datasets/oclar",
        "paper": "https://ieeexplore.ieee.org/abstract/document/8716394/"
      },
      "dialects": [
        "lev"
      ],
      "size": "3,916 sentences",
      "year": 2019,
      "notes": "Opinion Corpus for Lebanese Arabic Reviews",
      "metrics": {
        "downloads": 176,
        "likes": 1,
        "lastModified": "2024-06-26"
      }
    },
    {
      "id": "qasr-arabic-speech-continuations",
      "name": "qasr arabic speech continuations",
      "type": "dataset",
      "country": "LB",
      "org": "Wissam Antoun (AUB)",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech-continuation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/wissamantoun/qasr-arabic-speech-continuations"
      },
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "Arabic speech segments with transcripts and episode IDs for speech continuation.",
      "metrics": {
        "downloads": 176,
        "likes": 0,
        "lastModified": "2026-03-11"
      }
    },
    {
      "id": "shamela-waqfeya-library-compressed",
      "name": "shamela waqfeya library compressed",
      "type": "dataset",
      "country": "INTL",
      "org": "ieasybooks",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ieasybooks-org/shamela-waqfeya-library-compressed"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Shamela Waqfeya is one of the primary online resources for Islamic books, similar to Shamela.",
      "metrics": {
        "downloads": 176,
        "likes": 0,
        "lastModified": "2025-05-14"
      }
    },
    {
      "id": "khatt",
      "name": "KHATT",
      "type": "dataset",
      "country": "INTL",
      "org": "benhachem",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/benhachem/KHATT"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "The database contains handwritten Arabic text images and its ground-truth developed for research in the area of Arabic handwritten text.",
      "metrics": {
        "downloads": 175,
        "likes": 8,
        "lastModified": "2024-11-25"
      }
    },
    {
      "id": "dzair",
      "name": "DZAIR",
      "type": "llm",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/algerian-nlp/DZAIR"
      },
      "size": "105M",
      "dialects": [
        "magh"
      ],
      "notes": "Algerian Darja language model from the algerian-nlp community (Algeria).",
      "metrics": {
        "downloads": 174,
        "likes": 2,
        "lastModified": "2026-09-24"
      }
    },
    {
      "id": "kemetone",
      "name": "kemetone",
      "type": "tts",
      "country": "INTL",
      "org": "Rabe3",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/Rabe3/kemetone"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "A fine-tune of Nabra-82M , itself built on Kokoro-82M .",
      "base_model": [
        "oddadmix/nabra-82m-v0.1"
      ],
      "metrics": {
        "downloads": 174,
        "likes": 4,
        "lastModified": "2026-08-27"
      }
    },
    {
      "id": "alyah-emirati-benchmark",
      "name": "Alyah Emirati Benchmark",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tiiuae/alyah-emirati-benchmark"
      },
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Emirati dialect benchmark for Arabic LLMs released alongside Falcon-Arabic work.",
      "metrics": {
        "downloads": 173,
        "likes": 23,
        "lastModified": "2026-01-27"
      }
    },
    {
      "id": "arsas",
      "name": "ArSAS",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/ArSAS"
      },
      "size": "21k tweets",
      "year": 2018,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic tweets annotated for speech-act and sentiment.",
      "metrics": {
        "downloads": 173,
        "likes": 1,
        "lastModified": "2022-10-15"
      }
    },
    {
      "id": "jais-family-30b-8k",
      "name": "jais family 30b 8k",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inception42/jais-family-30b-8k"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "30B",
      "on_device": false,
      "year": 2024,
      "tags": [
        "variants:3"
      ],
      "notes": "The Jais family of models is a comprehensive series of bilingual English-Arabic large language models (LLMs). (also: 3 variants)",
      "metrics": {
        "downloads": 173,
        "likes": 10,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "llamalens-arabic-native",
      "name": "LlamaLens-Arabic-Native",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/LlamaLens-Arabic-Native"
      },
      "size": "1M–10M rows",
      "year": 2025,
      "notes": "LlamaLens is a specialized multilingual LLM designed for analyzing news and social media content.",
      "metrics": {
        "downloads": 173,
        "likes": 0,
        "lastModified": "2025-03-13"
      }
    },
    {
      "id": "easc-essex-arabic-summaries-corpus",
      "name": "EASC (Essex Arabic Summaries Corpus)",
      "type": "dataset",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/EASC"
      },
      "year": 2010,
      "notes": "153 Arabic documents with 765 human-made extractive summaries.",
      "metrics": {
        "downloads": 172,
        "likes": 0,
        "lastModified": "2025-11-28"
      }
    },
    {
      "id": "habibi",
      "name": "Habibi",
      "type": "dataset",
      "country": "INTL",
      "org": "SWivid",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SWivid/Habibi"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Paper: \"Habibi: Laying the Open-Source Foundation of Unified-Dialectal Arabic Speech Synthesis\" Github: https://github.com/SWivid/Habibi-TTS",
      "metrics": {
        "downloads": 172,
        "likes": 6,
        "lastModified": "2026-01-21"
      }
    },
    {
      "id": "marefa-mt-en-ar",
      "name": "marefa mt en ar",
      "type": "llm",
      "country": "INTL",
      "org": "marefa-nlp",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/marefa-nlp/marefa-mt-en-ar"
      },
      "year": 2021,
      "notes": "This is a model for translating English to Arabic.",
      "metrics": {
        "downloads": 172,
        "likes": 16,
        "lastModified": "2021-09-22"
      }
    },
    {
      "id": "sadeem-arabic-qna",
      "name": "Sadeem QnA",
      "type": "dataset",
      "country": "INTL",
      "org": "Sadeem AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/sadeem-ai/arabic-qna"
      },
      "size": "1K-10K rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic question-answer dataset from Wikipedia-style passages.",
      "metrics": {
        "downloads": 172,
        "likes": 5,
        "lastModified": "2024-02-05"
      }
    },
    {
      "id": "tunswitch",
      "name": "TunSwitch",
      "type": "dataset",
      "country": "TN",
      "org": "InstaDeep / iCompass",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tunis-ai/TunSwitch"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2024,
      "notes": "Tunisian Arabic code-switched speech data used to build and test a Tunisian ASR model (mirror of the Zenodo release).",
      "metrics": {
        "downloads": 172,
        "likes": 3,
        "lastModified": "2024-05-04"
      }
    },
    {
      "id": "medical-arabic-qa",
      "name": "medical arabic qa",
      "type": "dataset",
      "country": "INTL",
      "org": "MustafaIbrahim",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "medical",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MustafaIbrahim/medical-arabic-qa"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Arabic medical question-answer pairs with labels, 52,758 rows.",
      "metrics": {
        "downloads": 171,
        "likes": 3,
        "lastModified": "2025-03-15"
      }
    },
    {
      "id": "arabic-text-to-speech-andrewatef",
      "name": "Arabic Text to Speech",
      "type": "dataset",
      "country": "INTL",
      "org": "andrewatef",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/andrewatef/Arabic-Text-to-Speech"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "20.9K Arabic speech segments with transcripts and YouTube video metadata for TTS.",
      "metrics": {
        "downloads": 170,
        "likes": 4,
        "lastModified": "2025-04-27"
      }
    },
    {
      "id": "syrian-podcast-audio",
      "name": "Syrian Podcast Audio Collection",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-syrian-podcast"
      },
      "year": 2026,
      "dialects": [
        "lev"
      ],
      "notes": "Syrian podcast audio collection for speech training.",
      "metrics": {
        "downloads": 170,
        "likes": 2,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "arabic-audio-collection-algerian-rawi",
      "name": "arabic audio collection algerian rawi",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-algerian-rawi"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "The Rawi Postcast Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 51 hours of speech recordings.",
      "metrics": {
        "downloads": 168,
        "likes": 0,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-53-levantine-arabic",
      "name": "wav2vec2-large-xlsr-53-levantine-arabic",
      "type": "asr",
      "country": "INTL",
      "org": "Ali Elgeish",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/elgeish/wav2vec2-large-xlsr-53-levantine-arabic"
      },
      "year": 2022,
      "dialects": [
        "lev"
      ],
      "notes": "wav2vec2 XLSR-53 fine-tuned on Levantine Arabic.",
      "metrics": {
        "downloads": 168,
        "likes": 6,
        "lastModified": "2021-07-06"
      }
    },
    {
      "id": "marbertv2-finetuned-egyptian-hate-speech-detection",
      "name": "marbertv2 finetuned egyptian hate speech detection",
      "type": "llm",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimAmin/marbertv2-finetuned-egyptian-hate-speech-detection"
      },
      "dialects": [
        "egy"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of MARBERTv2.",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 166,
        "likes": 3,
        "lastModified": "2025-08-17"
      }
    },
    {
      "id": "rasaif",
      "name": "Rasaif",
      "type": "dataset",
      "country": "INTL",
      "org": "ImruQays",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ImruQays/Rasaif-Classical-Arabic-English-Parallel-texts"
      },
      "dialects": [
        "classical"
      ],
      "notes": "Classical Arabic-English parallel texts (24 books)",
      "metrics": {
        "downloads": 166,
        "likes": 8,
        "lastModified": "2024-03-22"
      }
    },
    {
      "id": "xlm-r-large-arabic-toxic",
      "name": "xlm r large arabic toxic",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/xlm-r-large-arabic-toxic"
      },
      "year": 2024,
      "notes": "See this Linkedin post This model: Toxic (hate speech) classification (Label0: non-toxic, Label1: toxic) of Arabic comments by fine-tuning XLM-Roberta-Large.",
      "metrics": {
        "downloads": 166,
        "likes": 6,
        "lastModified": "2024-10-08"
      }
    },
    {
      "id": "tinystories-algerian-darija",
      "name": "TinyStories-Algerian-Darija",
      "type": "dataset",
      "country": "INTL",
      "org": "Touati Kamel",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "text-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/touati-kamel/TinyStories-Algerian-Darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "TinyStories translated into Algerian Darija: parallel children's stories corpus with a cultural-adaptation pipeline.",
      "metrics": {
        "downloads": 165,
        "likes": 0,
        "lastModified": "2026-09-28"
      }
    },
    {
      "id": "whisper-small-libyan",
      "name": "whisper-small-libyan",
      "type": "asr",
      "country": "LY",
      "org": "Tass02",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Tass02/whisper-small-libyan"
      },
      "size": "242M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "Whisper small fine-tuned on Libyan Arabic; Libya noted.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 164,
        "likes": 1,
        "lastModified": "2026-09-21"
      }
    },
    {
      "id": "acegpt-v2-32b-chat",
      "name": "AceGPT-v2-32B-Chat",
      "type": "llm",
      "country": "INTL",
      "org": "Asas AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/asas-ai/AceGPT-v2-32B-Chat"
      },
      "base_model": [
        "qwen/qwen1.5-32b"
      ],
      "size": "32B",
      "on_device": false,
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "AceGPT is a fully fine-tuned generative text model collection, particularly focused on the Arabic language domain. (also: 1 variants)",
      "metrics": {
        "downloads": 161,
        "likes": 11,
        "lastModified": "2024-07-06"
      }
    },
    {
      "id": "quran-tafseer",
      "name": "Quran Tafseer Collection",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/Quran-Tafseer"
      },
      "year": 2024,
      "dialects": [
        "classical"
      ],
      "notes": "Collection of classical tafseer commentaries for each Quran verse.",
      "metrics": {
        "downloads": 161,
        "likes": 61,
        "lastModified": "2024-09-13"
      }
    },
    {
      "id": "context212-alhazen-ocr",
      "name": "context212 alhazen ocr",
      "type": "benchmark",
      "country": "INTL",
      "org": "context212",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/context212/context212-alhazen-ocr"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "alhazen-ocr is the training dataset behind context212/alhazen-ocr, an Arabic-first OCR vision-language model.",
      "metrics": {
        "downloads": 160,
        "likes": 3,
        "lastModified": "2026-08-24"
      }
    },
    {
      "id": "arabic-tashkeel-speech",
      "name": "arabic tashkeel speech",
      "type": "dataset",
      "country": "INTL",
      "org": "NahwAI",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NahwAI/arabic-tashkeel-speech"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "An open-source collection of 1,093 fully diacritized Arabic speech recordings, crowd-sourced from native speakers via Nahw.ai.",
      "metrics": {
        "downloads": 157,
        "likes": 3,
        "lastModified": "2026-04-21"
      }
    },
    {
      "id": "gigabert-v4-arabic-and-english",
      "name": "GigaBERT-v4-Arabic-and-English",
      "type": "llm",
      "country": "INTL",
      "org": "lanwuwei",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/lanwuwei/GigaBERT-v4-Arabic-and-English"
      },
      "year": 2021,
      "tags": [
        "variants:1"
      ],
      "notes": "GigaBERT-v4 is a continued pre-training of GigaBERT-v3 on code-switched data, showing improved zero-shot transfer performance from English to Arabic.",
      "metrics": {
        "downloads": 157,
        "likes": 5,
        "lastModified": "2021-05-19"
      }
    },
    {
      "id": "historical-arabic-handwritten-ocr",
      "name": "Historical Arabic Handwritten OCR",
      "type": "dataset",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "asr",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/sherif1313/Historical-Arabic-Handwritten-OCR"
      },
      "year": 2026,
      "notes": "Description A collection of rich historical Arabic text, spanning different geographies across centuries, is present in this dataset.",
      "metrics": {
        "downloads": 157,
        "likes": 1,
        "lastModified": "2026-02-14"
      }
    },
    {
      "id": "ogc-energy-arabic",
      "name": "OGC_Energy_Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "racine.ai",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/racineai/OGC_Energy_Arabic"
      },
      "dialects": [
        "msa"
      ],
      "size": "17,900 images",
      "year": 2025,
      "notes": "OGCEnergyArabic is a curated multimodal dataset focused on Arabic energy sector documents, including reports, financial statements, technical documentation.",
      "metrics": {
        "downloads": 157,
        "likes": 5,
        "lastModified": "2025-11-20"
      }
    },
    {
      "id": "tutlait-v1",
      "name": "Tutlait v1",
      "type": "dataset",
      "country": "INTL",
      "org": "Ma-OpenHub",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Ma-OpenHub/Tutlait-v1"
      },
      "dialects": [
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "21 hours of spoken Tamazight (Amazigh / Berber) from 118 speakers, each clip paired with an Arabic (MSA) translation.",
      "metrics": {
        "downloads": 157,
        "likes": 3,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "vdr-energy-arabic",
      "name": "VDR Energy Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "racineai",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/racineai/VDR_Energy_Arabic"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This dataset was created using our open-source tool VDRpdf-to-parquet.",
      "metrics": {
        "downloads": 157,
        "likes": 5,
        "lastModified": "2025-11-20"
      }
    },
    {
      "id": "arabic-doc-to-markdown",
      "name": "arabic doc to markdown",
      "type": "dataset",
      "country": "INTL",
      "org": "presightai",
      "license": "other",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "embedding",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/presightai/arabic_doc_to_markdown"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This dataset contains OCR image–markdown pairs specifically curated for document structure retrieval and reconstruction tasks.",
      "metrics": {
        "downloads": 156,
        "likes": 5,
        "lastModified": "2025-07-04"
      }
    },
    {
      "id": "emirates-dialect-speech-male",
      "name": "emirates dialect speech male",
      "type": "dataset",
      "country": "INTL",
      "org": "AhmedEladl",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "dialects"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AhmedEladl/emirates-dialect-speech-male"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "4,201 Emirati-dialect speech segments with base and dialectal transcriptions for ASR and TTS fine-tuning.",
      "dialects": [
        "gulf"
      ],
      "metrics": {
        "downloads": 156,
        "likes": 0,
        "lastModified": "2026-08-19"
      }
    },
    {
      "id": "fineweb2-najdi",
      "name": "FineWeb2 Najdi Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/FineWeb2-Najdi-Arabic"
      },
      "size": "10M-100M rows",
      "year": 2024,
      "dialects": [
        "gulf"
      ],
      "notes": "Najdi Arabic portion of FineWeb2.",
      "metrics": {
        "downloads": 156,
        "likes": 3,
        "lastModified": "2024-12-12"
      }
    },
    {
      "id": "sunnah-ar-en-dataset",
      "name": "sunnah ar en dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "gurgutan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "qa",
        "chat",
        "embedding",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/gurgutan/sunnah_ar_en_dataset"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This dataset contains a comprehensive bilingual (Arabic-English) collection of hadiths from 14 major authenticated books of Islamic tradition.",
      "metrics": {
        "downloads": 156,
        "likes": 3,
        "lastModified": "2025-03-13"
      }
    },
    {
      "id": "whisper-large-v3-tarteel",
      "name": "whisper large v3 Tarteel",
      "type": "asr",
      "country": "INTL",
      "org": "IJyad",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/IJyad/whisper-large-v3-Tarteel"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of OpenAI’s Whisper Large V3 model, adapted specifically for Arabic Quranic speech recognition.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 156,
        "likes": 6,
        "lastModified": "2025-06-04"
      }
    },
    {
      "id": "camel-bench",
      "name": "CAMEL-Bench",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "vqa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/CAMEL-Bench",
        "github": "https://github.com/mbzuai-oryx/CAMEL-Bench"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic large multimodal model benchmark across eight domains: OCR, charts, medical, remote sensing, video.",
      "metrics": {
        "downloads": 153,
        "likes": 0,
        "lastModified": "2026-05-09"
      }
    },
    {
      "id": "egyspeak",
      "name": "EGYSpeak",
      "type": "dataset",
      "country": "EG",
      "org": "Mohamed Gomaa",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedGomaa30/EGYSpeak"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "A curated dataset of 147,979 single-speaker Egyptian Arabic (pure dialect) audio clips with transcriptions.",
      "metrics": {
        "downloads": 153,
        "likes": 1,
        "lastModified": "2026-04-28"
      }
    },
    {
      "id": "qwen2-5-1-5b-amiya-palestinian",
      "name": "qwen2.5-1.5b-amiya-palestinian",
      "type": "llm",
      "country": "INTL",
      "org": "Khamad",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Khamad/qwen2.5-1.5b-amiya-palestinian"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "Qwen2.5-1.5B fine-tuned for Palestinian dialect generation in the AMIYA shared task.",
      "base_model": [
        "qwen/qwen2.5-1.5b-instruct"
      ],
      "metrics": {
        "downloads": 152,
        "likes": 0,
        "lastModified": "2026-01-12"
      }
    },
    {
      "id": "araeurobert-610m",
      "name": "AraEuroBert-610M",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/AraEuroBert-610M"
      },
      "size": "610M",
      "on_device": true,
      "year": 2025,
      "notes": "This model maps sentences and paragraphs to a 1152-dimensional dense vector space and Maximum Sequence Length: 8,192 tokens.",
      "base_model": [
        "eurobert/eurobert-610m"
      ],
      "metrics": {
        "downloads": 151,
        "likes": 4,
        "lastModified": "2025-03-20"
      }
    },
    {
      "id": "emhotob-25m-english-msa-v1",
      "name": "Emhotob 25M English MSA v1",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Emhotob-25M-English-MSA-v1"
      },
      "dialects": [
        "msa"
      ],
      "size": "25M",
      "on_device": true,
      "year": 2026,
      "tags": [
        "variants:4"
      ],
      "notes": "A 25.3M-parameter model that translates both ways between English and Modern Standard Arabic (الفصحى). (also: 4 variants)",
      "base_model": [
        "oddadmix/emhotob-25m"
      ],
      "metrics": {
        "downloads": 151,
        "likes": 0,
        "lastModified": "2026-07-18"
      }
    },
    {
      "id": "saudinewsnet",
      "name": "SaudiNewsNet",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "news"
      ],
      "links": {
        "github": "https://github.com/inparallel/SaudiNewsNet",
        "hf": "https://huggingface.co/datasets/inparallel/saudinewsnet"
      },
      "dialects": [
        "msa"
      ],
      "size": "31,030 documents",
      "year": 2015,
      "notes": "31,030 Arabic newspaper articles with metadata, extracted from online Saudi newspapers.",
      "metrics": {
        "downloads": 151,
        "likes": 9,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "wiki-qa-ar",
      "name": "wiki qa ar",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "['unknown']",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/qcri/wiki_qa_ar"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Arabic WikiQA: machine-translated WikiQA with crowdsourced selection of the best translation.",
      "metrics": {
        "downloads": 150,
        "likes": 3,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "arabic-audio-collection-sudanese-ahmed-gobara",
      "name": "arabic audio collection sudanese ahmed gobara",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-sudanese-ahmed-gobara"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "The Ahmed Gobara Sudanese Arabic Speech Dataset is a single-speaker Sudanese Arabic speech corpus containing approximately 19 hours of speech recordings.",
      "metrics": {
        "downloads": 149,
        "likes": 1,
        "lastModified": "2026-08-07"
      }
    },
    {
      "id": "fasih-tts-v1",
      "name": "Fasih TTS V1",
      "type": "tts",
      "country": "INTL",
      "org": "NightPrince",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/NightPrince/Fasih-TTS-V1"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "notes": "Fasih-TTS-V1 · فَصِيح Modern Standard Arabic (Fusha) text-to-speech with a professional male voice.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 149,
        "likes": 6,
        "lastModified": "2026-10-02"
      }
    },
    {
      "id": "arabic-wsd-benchmark",
      "name": "Arabic WSD Benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "Asas AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "wsd",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/asas-ai/Arabic_WSD_Benchmark"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Arabic word sense disambiguation benchmark of 31K rows with word, definition, example and label fields.",
      "metrics": {
        "downloads": 148,
        "likes": 0,
        "lastModified": "2024-05-06"
      }
    },
    {
      "id": "araseg-2026-shared-task-nopnx-pa",
      "name": "AraSeg-2026-Shared-Task-NoPnx-PA",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/AraSeg-2026-Shared-Task-NoPnx-PA"
      },
      "dialects": [
        "msa"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "The corpus is designed to support research on sentence segmentation in Modern Standard Arabic.",
      "metrics": {
        "downloads": 148,
        "likes": 3,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "fineweb2-moroccan",
      "name": "FineWeb2 Moroccan Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/FineWeb2-Moroccan-Arabic"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan Arabic subset extracted from FineWeb2.",
      "metrics": {
        "downloads": 148,
        "likes": 3,
        "lastModified": "2024-12-12"
      }
    },
    {
      "id": "mawqif",
      "name": "Mawqif",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stance-detection"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NoraAlt/Mawqif_Stance-Detection"
      },
      "year": 2022,
      "notes": "Multi-label Arabic dataset for target-specific stance, sentiment and sarcasm detection.",
      "metrics": {
        "downloads": 148,
        "likes": 6,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "tweets-ar-en-parallel",
      "name": "tweets ar en parallel",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI ALT",
      "license": "['apache-2.0']",
      "modality": "text",
      "tasks": [
        "translation",
        "twitter"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alt-qsri/tweets_ar_en_parallel"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Bilingual corpus of Arabic-English parallel tweets collected for machine translation research.",
      "metrics": {
        "downloads": 148,
        "likes": 4,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "al-maktabah-al-shamilah",
      "name": "Al-Maktabah Al-Shamilah",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/Al-Maktabah-Al-Shamilah"
      },
      "dialects": [
        "classical"
      ],
      "notes": "Classical Arabic books text from the Shamela library, for pretraining.",
      "metrics": {
        "downloads": 147,
        "likes": 14,
        "lastModified": "2025-09-11"
      }
    },
    {
      "id": "arabic-t5-small",
      "name": "arabic t5 small",
      "type": "llm",
      "country": "INTL",
      "org": "Flax Community",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-to-text"
      ],
      "links": {
        "hf": "https://huggingface.co/flax-community/arabic-t5-small"
      },
      "year": 2023,
      "notes": "T5 v1.1 small trained on Arabic Billion Words plus Arabic subsets of mC4 and OSCAR, covering about 10% of the data.",
      "metrics": {
        "downloads": 147,
        "likes": 10,
        "lastModified": "2023-11-29"
      }
    },
    {
      "id": "fineweb2-egyptian",
      "name": "FineWeb2 Egyptian Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/FineWeb2-Egyptian-Arabic"
      },
      "size": "10M-100M rows",
      "year": 2024,
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic portion of FineWeb2.",
      "metrics": {
        "downloads": 147,
        "likes": 2,
        "lastModified": "2024-12-12"
      }
    },
    {
      "id": "tsac",
      "name": "TSAC",
      "type": "dataset",
      "country": "TN",
      "org": "Fethi Bougares",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fbougares/tsac"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian Sentiment Analysis Corpus of Facebook comments in Tunisian dialect.",
      "metrics": {
        "downloads": 147,
        "likes": 3,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "aura-classification",
      "name": "AURA Classification",
      "type": "dataset",
      "country": "INTL",
      "org": "irfan-ahmad",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/irfan-ahmad/AURA-Classification"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "The AURA Classification dataset is available in two versions: Both versions contain the same 2,900 Arabic app reviews.",
      "metrics": {
        "downloads": 146,
        "likes": 3,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "journalists-questions",
      "name": "journalists_questions",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://www.dropbox.com/scl/fi/0r3mrx1npf1gi0d1zbhzu/ArQAT-JQ-Dataset-v1.0.zip?rlkey=fhigh3hpjvp4ptpyj780kqo9c&e=1&dl=0",
        "hf": "https://huggingface.co/datasets/community-datasets/journalists_questions",
        "paper": "https://www.semanticscholar.org/paper/What-Questions-Do-Journalists-Ask-on-Twitter-Hasanain-Bagdouri/d1b32df7e9f39e6fba912cc209054ae0256638eb"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2016,
      "notes": "Arabic Twitter dataset of 10K tweets with crowdsourced binary annotations indicating whether each tweet contains a genuine question.",
      "metrics": {
        "downloads": 146,
        "likes": 0,
        "lastModified": "2024-06-26"
      }
    },
    {
      "id": "arabic-audio-collection-moroccan-noone-stories",
      "name": "arabic audio collection moroccan noone stories",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-moroccan-noone-stories"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Noone Stories Moroccan Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 156 hours.",
      "metrics": {
        "downloads": 145,
        "likes": 0,
        "lastModified": "2026-06-30"
      }
    },
    {
      "id": "arabicragb",
      "name": "ArabicRAGB",
      "type": "benchmark",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/ArabicRAGB"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic RAG Benchmark (multi-dialect)",
      "metrics": {
        "downloads": 145,
        "likes": 13,
        "lastModified": "2025-12-15"
      }
    },
    {
      "id": "arabic-audio-collection-sudanese-nuuar",
      "name": "arabic audio collection sudanese nuuar",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-sudanese-nuuar"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Nuuar Sudanese Arabic Speech Dataset is a single-speaker Sudanese Arabic speech corpus containing approximately 75 hours of speech recordings.",
      "metrics": {
        "downloads": 144,
        "likes": 0,
        "lastModified": "2026-08-07"
      }
    },
    {
      "id": "aralingbench",
      "name": "AraLingBench",
      "type": "benchmark",
      "country": "INTL",
      "org": "hammh0a",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hammh0a/AraLingBench"
      },
      "size": "<1K rows",
      "year": 2025,
      "notes": "AraLingBench: 150 human-authored Arabic multiple-choice questions testing grammar, morphology, spelling, comprehension and syntax.",
      "metrics": {
        "downloads": 144,
        "likes": 12,
        "lastModified": "2025-11-19"
      }
    },
    {
      "id": "araseg-2026-shared-task-np",
      "name": "AraSeg-2026-Shared-Task-NP",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/AraSeg-2026-Shared-Task-NP"
      },
      "dialects": [
        "msa"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "The corpus is designed to support research on sentence segmentation in Modern Standard Arabic.",
      "metrics": {
        "downloads": 144,
        "likes": 1,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "whisper-algerian-darja-medium",
      "name": "whisper algerian darja medium",
      "type": "asr",
      "country": "INTL",
      "org": "Touati Kamel",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/touati-kamel/whisper-algerian-darja-medium"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Whisper Medium (769M) fine-tuned with LoRA for Algerian Arabic (Darja) speech recognition.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 144,
        "likes": 1,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "alarb",
      "name": "ALARB",
      "type": "benchmark",
      "country": "INTL",
      "org": "THIQAH-RD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/THIQAH-RD/ALARB"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "ALARB includes a dataset of structured legal cases.",
      "metrics": {
        "downloads": 143,
        "likes": 8,
        "lastModified": "2025-10-15"
      }
    },
    {
      "id": "alpaca-arabic-gpt4",
      "name": "Alpaca Arabic GPT4",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Alpaca-Arabic-GPT4"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "The dataset is created by (1) translating alpaca English questions into Arabic using GPT4 and (2) requesting GPT4 to generate Arabic responses.",
      "metrics": {
        "downloads": 143,
        "likes": 1,
        "lastModified": "2023-09-06"
      }
    },
    {
      "id": "arapro",
      "name": "AraPro",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/AraPro"
      },
      "year": 2025,
      "notes": "5,001 Arabic multiple-choice questions on professional domains from the ALLaM evaluation suite.",
      "metrics": {
        "downloads": 143,
        "likes": 3,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "arabic-dexlit-corpus",
      "name": "ArabicDeXlit Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "mohammedaly22",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "transliteration",
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohammedaly22/ArabicDeXlit-Corpus"
      },
      "year": 2026,
      "notes": "Training data for undoing transliteration in code-switched Arabic ASR output.",
      "metrics": {
        "downloads": 142,
        "likes": 0,
        "lastModified": "2026-09-20"
      }
    },
    {
      "id": "cohere-transcribe-arabic-07-2026-int4",
      "name": "cohere transcribe arabic 07 2026 int4",
      "type": "asr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/cohere-transcribe-arabic-07-2026-int4"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "A memory-efficient INT4 (NF4) quantization of CohereLabs/cohere-transcribe-arabic-07-2026, Cohere Labs' 2B-parameter Arabic/English speech-recognition model.",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 142,
        "likes": 5,
        "lastModified": "2026-07-10"
      }
    },
    {
      "id": "quranicwhisperdataset",
      "name": "QURANICWhisperDataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ahishamm",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ahishamm/QURANICWhisperDataset"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "31K Quran recitation audio clips with text and harakat transcription for fine-tuning Whisper.",
      "metrics": {
        "downloads": 141,
        "likes": 6,
        "lastModified": "2024-04-02"
      }
    },
    {
      "id": "arabic-multidialect-emotional-speech-demo",
      "name": "arabic multidialect emotional speech demo",
      "type": "dataset",
      "country": "INTL",
      "org": "datahiveai",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech",
        "sentiment",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/datahiveai/arabic-multidialect-emotional-speech-demo"
      },
      "dialects": [
        "gulf",
        "lev",
        "magh",
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "A DataHive AI dataset: a stratified 1-hour demo sample from a full corpus of 50+ hours.",
      "metrics": {
        "downloads": 140,
        "likes": 3,
        "lastModified": "2026-05-07"
      }
    },
    {
      "id": "fineweb-edu-egypt",
      "name": "fineweb-edu-Egypt",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/nilechat-fw-edu-egy",
        "paper": "https://aclanthology.org/2025.emnlp-main.556.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "5,521,803 documents",
      "year": 2025,
      "notes": "Fineweb-edu-Egypt is a substantial dataset specifically developed to foster the creation and improvement of language models for the Egyptian Arabic dialect.",
      "metrics": {
        "downloads": 140,
        "likes": 3,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "misraj-dococr",
      "name": "Misraj-DocOCR",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/Misraj-DocOCR"
      },
      "year": 2025,
      "notes": "Public 400-page expert-checked Arabic document OCR benchmark (WER, CER, TEDS).",
      "metrics": {
        "downloads": 140,
        "likes": 10,
        "lastModified": "2025-10-12"
      }
    },
    {
      "id": "arabic-marbert-news-article-classification",
      "name": "arabic MARBERT news article classification",
      "type": "llm",
      "country": "INTL",
      "org": "Ammar-alhaj-ali",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Ammar-alhaj-ali/arabic-MARBERT-news-article-classification"
      },
      "year": 2022,
      "notes": "Arabic MARBERT News Article Classification Model",
      "metrics": {
        "downloads": 139,
        "likes": 4,
        "lastModified": "2022-08-12"
      }
    },
    {
      "id": "murad",
      "name": "MURAD",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "summarization",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/MURAD"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic is a linguistically and culturally rich",
      "metrics": {
        "downloads": 139,
        "likes": 2,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "qadi",
      "name": "QADI",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdelrahman-Rezk",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdelrahman-Rezk/Arabic_Dialect_Identification"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Twitter-based multi-class dialect classification",
      "metrics": {
        "downloads": 139,
        "likes": 12,
        "lastModified": "2022-05-17"
      }
    },
    {
      "id": "scc22",
      "name": "SCC22",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/SCC22"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The Saudilang Code-Switch Corpus (SCC) is a 5-hour transcribed audio dataset featuring informal Saudi Arabic speech with code-switching to English.",
      "metrics": {
        "downloads": 139,
        "likes": 8,
        "lastModified": "2025-05-03"
      }
    },
    {
      "id": "arabic-gec-v1",
      "name": "arabic-gec-v1",
      "type": "llm",
      "country": "INTL",
      "org": "alnnahwi",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/alnnahwi/gemma-3-1b-arabic-gec-v1"
      },
      "base_model": [
        "google/gemma-3-1b-pt"
      ],
      "notes": "Grammar Correction - Gemma-3-1b for Arabic GEC",
      "metrics": {
        "downloads": 137,
        "likes": 8,
        "lastModified": "2025-06-15"
      }
    },
    {
      "id": "atlasocr-data",
      "name": "AtlasOCR Darija Dataset",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/atlasOCR-data"
      },
      "size": "10K-100K rows",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan Darija OCR training images with transcripts used for AtlasOCR.",
      "metrics": {
        "downloads": 137,
        "likes": 3,
        "lastModified": "2025-09-16"
      }
    },
    {
      "id": "arabic-audio-mostafa-mahmoud",
      "name": "Mostafa Mahmoud Arabic Speech",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-mostafa-mahmoud"
      },
      "size": "187h",
      "year": 2026,
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic speech corpus of about 187 hours for ASR and TTS.",
      "metrics": {
        "downloads": 137,
        "likes": 13,
        "lastModified": "2026-06-18"
      }
    },
    {
      "id": "shifaa-mental-health",
      "name": "Shifaa Mental Health",
      "type": "dataset",
      "country": "EG",
      "org": "Ahmed Selem",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Ahmed-Selem/Shifaa_Arabic_Mental_Health_Consultations"
      },
      "notes": "Arabic mental health consultations",
      "metrics": {
        "downloads": 137,
        "likes": 14,
        "lastModified": "2025-03-08"
      }
    },
    {
      "id": "arabic-mms-speech-synthesis",
      "name": "Arabic MMS Speech Synthesis",
      "type": "tts",
      "country": "INTL",
      "org": "SeyedAli",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/SeyedAli/Arabic-Speech-synthesis-MMS"
      },
      "on_device": true,
      "notes": "VITS-based Arabic TTS from MMS (36M params)",
      "metrics": {
        "downloads": 136,
        "likes": 22,
        "lastModified": "2023-09-20"
      }
    },
    {
      "id": "arvoice-undiacritized",
      "name": "ArVoice-Undiacritized",
      "type": "dataset",
      "country": "SA",
      "org": "mabahboh",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mabahboh/ArVoice-Undiacritized"
      },
      "dialects": [
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "A derivative of MBZUAI/ArVoice with one added column, textundiacritized: the transcription transcripts with diacritics removed.",
      "metrics": {
        "downloads": 136,
        "likes": 0,
        "lastModified": "2026-09-28"
      }
    },
    {
      "id": "niletts-xtts",
      "name": "NileTTS-XTTS",
      "type": "tts",
      "country": "INTL",
      "org": "KickItLikeShika",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/KickItLikeShika/NileTTS-XTTS"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "This model was fine-tuned on the NileTTS dataset, comprising 38 hours of Egyptian Arabic speech across medical, sales, and general conversation domains.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 135,
        "likes": 4,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "lemura-arabic-asr-lite",
      "name": "Lemura Arabic ASR Lite",
      "type": "asr",
      "country": "INTL",
      "org": "Lemura Labs",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/lemuralabs/lemura-arabic-asr-lite"
      },
      "year": 2026,
      "dialects": [
        "msa",
        "gulf",
        "egy",
        "lev",
        "magh"
      ],
      "notes": "NeMo multi-dialect Arabic ASR covering MSA, Gulf, Egyptian, Levantine and Maghrebi.",
      "metrics": {
        "downloads": 134,
        "likes": 3,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "wasil",
      "name": "WASIL",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/WASIL"
      },
      "size": "9.3k turns",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "In-the-wild Arabic spoken interactions with an LLM assistant, ~9K turns from 93 users across dialects.",
      "metrics": {
        "downloads": 134,
        "likes": 2,
        "lastModified": "2026-05-19"
      }
    },
    {
      "id": "aramath",
      "name": "AraMath",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/AraMath"
      },
      "year": 2025,
      "notes": "605 Arabic math word problems in multiple-choice form adapted from ArMath.",
      "metrics": {
        "downloads": 133,
        "likes": 1,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "orca-parquet",
      "name": "orca parquet",
      "type": "dataset",
      "country": "LB",
      "org": "Wissam Antoun (AUB)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/wissamantoun/orca_parquet"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Arabic tabular/text dataset (100K<N<1M rows).",
      "metrics": {
        "downloads": 133,
        "likes": 0,
        "lastModified": "2026-03-01"
      }
    },
    {
      "id": "tashkeel-700m",
      "name": "Tashkeel 700M",
      "type": "llm",
      "country": "INTL",
      "org": "Etherll",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "diacritization",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/Etherll/Tashkeel-700M"
      },
      "size": "700M",
      "on_device": true,
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "نموذج بحجم 700 مليون بارامتر مخصص لتشكيل النصوص العربية. (also: 1 variants)",
      "base_model": [
        "liquidai/lfm2-700m"
      ],
      "metrics": {
        "downloads": 133,
        "likes": 6,
        "lastModified": "2025-08-22"
      }
    },
    {
      "id": "darija-omnivoice-kore-v1",
      "name": "darija omnivoice kore v1",
      "type": "tts",
      "country": "INTL",
      "org": "ai-ssam",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/ai-ssam/darija-omnivoice-kore-v1"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "A full fine-tune of k2-fsa/OmniVoice for Moroccan Arabic (Darija), targeting the Kore voice.",
      "base_model": [
        "k2-fsa/omnivoice"
      ],
      "metrics": {
        "downloads": 132,
        "likes": 3,
        "lastModified": "2026-09-16"
      }
    },
    {
      "id": "meraj-mini",
      "name": "Meraj Mini",
      "type": "llm",
      "country": "INTL",
      "org": "Arcee AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/arcee-ai/Meraj-Mini"
      },
      "year": 2026,
      "notes": "Following the release of Arcee Meraj, our enterprise's globally top-performing Arabic LLM, we are thrilled to unveil Arcee Meraj Mini.",
      "base_model": [
        "qwen/qwen2.5-7b-instruct"
      ],
      "metrics": {
        "downloads": 132,
        "likes": 19,
        "lastModified": "2026-01-16"
      }
    },
    {
      "id": "quran-audio-text-dataset",
      "name": "quran audio text dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Buraaq",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Buraaq/quran-audio-text-dataset"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "This collection provides comprehensive audio recordings of the complete Quran (Holy Book of Islam) with multiple recitations and granular annotations.",
      "metrics": {
        "downloads": 132,
        "likes": 16,
        "lastModified": "2026-05-24"
      }
    },
    {
      "id": "tunswitch-diacritized",
      "name": "TunSwitch (diacritized)",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/TunSwitch"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Diacritized transcriptions of the TunSwitch Tunisian Arabic-French code-switching speech corpus.",
      "metrics": {
        "downloads": 132,
        "likes": 3,
        "lastModified": "2025-09-04"
      }
    },
    {
      "id": "arabic-audio-deepfake",
      "name": "Arabic Audio Deepfake",
      "type": "dataset",
      "country": "INTL",
      "org": "DeepFake-Audio-Rangers",
      "license": "odc-by",
      "modality": "speech",
      "tasks": [
        "speech",
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/DeepFake-Audio-Rangers/Arabic_Audio_Deepfake"
      },
      "dialects": [
        "lev",
        "mixed"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "This dataset contains Arabic deepfake audio samples, focusing mainly on Levantine dialect with some examples in Standard Arabic.",
      "metrics": {
        "downloads": 131,
        "likes": 4,
        "lastModified": "2024-10-01"
      }
    },
    {
      "id": "arabic-english-sts-matryoshka-v2-0",
      "name": "arabic english sts matryoshka v2.0",
      "type": "embedding",
      "country": "EG",
      "org": "Omar Elshehy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/omarelshehy/arabic-english-sts-matryoshka-v2.0"
      },
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "Bilingual Arabic-English sentence embedding model with Matryoshka dimensions; v2.0 of omarelshehy/arabic-english-sts-matryoshka.",
      "base_model": [
        "facebookai/xlm-roberta-large"
      ],
      "metrics": {
        "downloads": 131,
        "likes": 3,
        "lastModified": "2024-10-25"
      }
    },
    {
      "id": "minidense-arabic-v1",
      "name": "miniDense_arabic_v1",
      "type": "embedding",
      "country": "INTL",
      "org": "prithivida",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/prithivida/miniDense_arabic_v1"
      },
      "year": 2025,
      "notes": "Table 1: Arabic retrieval performance on the MIRACL dev set (measured by nDCG@10) Table Of Contents",
      "metrics": {
        "downloads": 131,
        "likes": 7,
        "lastModified": "2025-05-01"
      }
    },
    {
      "id": "arb-multimodal-reasoning",
      "name": "ARB",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/ARB"
      },
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Comprehensive Arabic multimodal reasoning benchmark with step-by-step reasoning annotations.",
      "metrics": {
        "downloads": 130,
        "likes": 11,
        "lastModified": "2025-06-21"
      }
    },
    {
      "id": "armeme",
      "name": "ArMeme",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ArMeme"
      },
      "size": "1K-10K rows",
      "year": 2024,
      "dialects": [
        "mixed"
      ],
      "notes": "First multimodal Arabic memes dataset with text and images from social media.",
      "metrics": {
        "downloads": 130,
        "likes": 9,
        "lastModified": "2024-10-08"
      }
    },
    {
      "id": "egybert",
      "name": "EgyBERT",
      "type": "llm",
      "country": "SA",
      "org": "Faisal Qarah",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/faisalq/EgyBERT"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian dialect BERT, trained on 34M tweets + 44M forum sentences",
      "metrics": {
        "downloads": 130,
        "likes": 6,
        "lastModified": "2024-08-19"
      }
    },
    {
      "id": "egyptian-handwriting-dataset",
      "name": "Egyptian Handwriting Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "OmarMDiab",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/OmarMDiab/Egyptian-Handwriting-Dataset"
      },
      "dialects": [
        "egy"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "A dataset of 11k+ handwritten Arabic words from Egyptian writers, extracted and tightly cropped from scanned paper forms.",
      "metrics": {
        "downloads": 130,
        "likes": 4,
        "lastModified": "2026-01-11"
      }
    },
    {
      "id": "arabic-ocr-qwen2-5-vl-7b-vision",
      "name": "Arabic OCR Qwen2.5 VL 7B Vision",
      "type": "ocr",
      "country": "INTL",
      "org": "loay",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "chat",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/loay/Arabic-OCR-Qwen2.5-VL-7B-Vision"
      },
      "size": "7B",
      "on_device": false,
      "year": 2025,
      "notes": "This repository contains the float16 merged version of a Vision-Language Model (VLM).",
      "base_model": [
        "unsloth/qwen2.5-vl-7b-instruct-bnb-4bit"
      ],
      "metrics": {
        "downloads": 129,
        "likes": 5,
        "lastModified": "2025-07-18"
      }
    },
    {
      "id": "arabic-prompt-routing",
      "name": "arabic prompt routing",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-prompt-routing"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Categories are arbitrary Arabic — the point is a model that routes into a label set it has never seen.",
      "metrics": {
        "downloads": 129,
        "likes": 0,
        "lastModified": "2026-09-04"
      }
    },
    {
      "id": "araelectra-arabic-squadv2-qa",
      "name": "AraElectra-Arabic-SQuADv2-QA",
      "type": "llm",
      "country": "INTL",
      "org": "ZeyadAhmed",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/ZeyadAhmed/AraElectra-Arabic-SQuADv2-QA"
      },
      "year": 2022,
      "notes": "This is the AraElectra model, fine-tuned using the Arabic-SQuADv2.0 dataset.",
      "metrics": {
        "downloads": 129,
        "likes": 18,
        "lastModified": "2022-07-04"
      }
    },
    {
      "id": "fibonacci-2-14b",
      "name": "fibonacci 2 14B",
      "type": "llm",
      "country": "INTL",
      "org": "fibonacciai",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/fibonacciai/fibonacci-2-14B"
      },
      "size": "14B",
      "on_device": false,
      "year": 2025,
      "notes": "مدل Fibonacci-2-14b یک مدل زبانی بزرگ (LLM) مبتنی بر معماری Phi 4 است که با 14 میلیارد پارامتر طراحی شده است.",
      "base_model": [
        "fibonacciai/fibonacci-1-en-8b-chat.p1_5"
      ],
      "metrics": {
        "downloads": 129,
        "likes": 13,
        "lastModified": "2025-04-02"
      }
    },
    {
      "id": "noon",
      "name": "Noon",
      "type": "llm",
      "country": "SA",
      "org": "Naseej",
      "license": "bigscience-bloom-rail-1.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Naseej/noon-7b"
      },
      "size": "7B",
      "notes": "BLOOM-based Arabic LLM, instruction-tuned",
      "metrics": {
        "downloads": 129,
        "likes": 49,
        "lastModified": "2023-06-21"
      }
    },
    {
      "id": "whisper-large-v3-arabic-byne",
      "name": "whisper large v3 arabic",
      "type": "asr",
      "country": "INTL",
      "org": "Byne",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Byne/whisper-large-v3-arabic"
      },
      "year": 2025,
      "notes": "Whisper large-v3 fine-tuned for Arabic speech recognition, reaching 9.38 WER on its evaluation set.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 129,
        "likes": 6,
        "lastModified": "2025-04-20"
      }
    },
    {
      "id": "whisper-small-arabic-dialectal-v2",
      "name": "whisper small arabic dialectal v2",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/whisper-small-arabic-dialectal-v2"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "tags": [
        "variants:3"
      ],
      "notes": "multi-dialect Arabic on the dialect-balanced dataset oddadmix/lahgtna-v3-small (52k train / 2.6k test, seed 42, undiacritized output). (also: 3 variants)",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 128,
        "likes": 0,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "araifeval",
      "name": "AraIFEval",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/AraIFEval"
      },
      "year": 2025,
      "notes": "Arabic instruction-following benchmark, 535 instances with verifiable instructions, from the ALLaM team.",
      "metrics": {
        "downloads": 127,
        "likes": 1,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "elner-dz",
      "name": "ELNER DZ",
      "type": "dataset",
      "country": "INTL",
      "org": "HadjerHaninebgt7878",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HadjerHaninebgt7878/ELNER-DZ"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2025,
      "notes": "This dataset, titled ELNER-DZ, was created by Bouguettoucha Hadjer Hanine and Djouablia Ilhem as part of our Master’s thesis .",
      "metrics": {
        "downloads": 127,
        "likes": 4,
        "lastModified": "2025-07-12"
      }
    },
    {
      "id": "instar-500k",
      "name": "InstAr-500k",
      "type": "dataset",
      "country": "INTL",
      "org": "ClusterlabAi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ClusterlabAi/InstAr-500k"
      },
      "size": "500k rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "About 500k Arabic instruction-response pairs for LLM fine-tuning.",
      "metrics": {
        "downloads": 127,
        "likes": 15,
        "lastModified": "2024-07-30"
      }
    },
    {
      "id": "misraj-kitab-reviewed",
      "name": "KITAB PDF-to-Markdown (reviewed)",
      "type": "benchmark",
      "country": "SA",
      "org": "Misraj AI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/KITAB_pdf_to_markdown_reviewed"
      },
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Corrected version of the KITAB-Bench PDF-to-Markdown subset for Arabic document OCR.",
      "metrics": {
        "downloads": 126,
        "likes": 4,
        "lastModified": "2025-09-24"
      }
    },
    {
      "id": "mgb2-qcri",
      "name": "MGB-2 (QCRI)",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/mgb2"
      },
      "size": "~1200h",
      "year": 2016,
      "dialects": [
        "mixed"
      ],
      "notes": "Multi-Genre Broadcast 2 Arabic Aljazeera speech recognition corpus, gated on Hugging Face.",
      "metrics": {
        "downloads": 126,
        "likes": 4,
        "lastModified": "2025-10-13"
      }
    },
    {
      "id": "namaa-ara-reranker-v1",
      "name": "Namaa ARA Reranker V1",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Namaa-ARA-Reranker-V1"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "notes": "✨ This model is designed specifically for Arabic language reranking tasks, optimized to handle queries and passages.",
      "metrics": {
        "downloads": 125,
        "likes": 6,
        "lastModified": "2025-04-03"
      }
    },
    {
      "id": "probel-mtl",
      "name": "ProBel-MTL",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/ProBel-MTL"
      },
      "year": 2026,
      "notes": "The bilingual multi-task model from the ProBel paper (Mt-SFT).",
      "base_model": [
        "qwen/qwen2.5-7b-instruct"
      ],
      "metrics": {
        "downloads": 125,
        "likes": 0,
        "lastModified": "2026-08-26"
      }
    },
    {
      "id": "quran-hadith",
      "name": "Quran Hadith",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith",
        "similarity"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Quran_Hadith"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "8,144 Quran-Hadith verse pairs with a relatedness label.",
      "metrics": {
        "downloads": 125,
        "likes": 5,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "silma-ragqa-benchmark",
      "name": "SILMA RAGQA Benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/silma-ai/silma-rag-qa-benchmark-v1.0"
      },
      "notes": "Evaluates Arabic/English LMs in Extractive QA tasks",
      "metrics": {
        "downloads": 125,
        "likes": 7,
        "lastModified": "2025-06-11"
      }
    },
    {
      "id": "gemma-4-e2b-arabic-english-vision",
      "name": "gemma 4 e2b arabic english vision",
      "type": "llm",
      "country": "INTL",
      "org": "ml-intern-explorers",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "chat",
        "speech",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/ml-intern-explorers/gemma-4-e2b-arabic-english-vision"
      },
      "year": 2026,
      "notes": "A pruned version of google/gemma-4-e2b-it optimized for Arabic and English vision-language tasks.",
      "base_model": [
        "google/gemma-4-e2b-it"
      ],
      "metrics": {
        "downloads": 124,
        "likes": 4,
        "lastModified": "2026-04-28"
      }
    },
    {
      "id": "mgb-5",
      "name": "MGB-5",
      "type": "dataset",
      "country": "INTL",
      "org": "University Of Düsseldorf",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabicSpeech/MGB-5",
        "paper": "https://ieeexplore.ieee.org/document/9003960"
      },
      "dialects": [
        "magh"
      ],
      "size": "14 hours",
      "year": 2019,
      "notes": "Moroccan Arabic speech extracted from 93 YouTube videos distributed across seven genres: comedy, cooking, family/children, fashion, drama, sports.",
      "metrics": {
        "downloads": 124,
        "likes": 1,
        "lastModified": "2025-08-11"
      }
    },
    {
      "id": "namaa-egyptian-tts",
      "name": "NAMAA-Egyptian-TTS",
      "type": "tts",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/NAMAA-Egyptian-TTS"
      },
      "size": "536M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian-dialect TTS from NAMAA.",
      "base_model": [
        "resembleai/chatterbox"
      ],
      "metrics": {
        "downloads": 124,
        "likes": 15,
        "lastModified": "2026-03-04"
      }
    },
    {
      "id": "arabic-audio-collection-libyan-a7rar-podcast",
      "name": "arabic audio collection libyan a7rar podcast",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-libyan-a7rar-podcast"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "The Libyan Postcast Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 26 hours of speech recordings.",
      "metrics": {
        "downloads": 123,
        "likes": 1,
        "lastModified": "2026-06-21"
      }
    },
    {
      "id": "egyptian-customs-authority",
      "name": "egyptian customs authority",
      "type": "dataset",
      "country": "INTL",
      "org": "mostafatouny",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mostafatouny/egyptian-customs-authority"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Raw laws are downloaded from Egyptian Customs Authority.",
      "metrics": {
        "downloads": 123,
        "likes": 4,
        "lastModified": "2026-05-08"
      }
    },
    {
      "id": "aisa-arabicfc",
      "name": "AISA-ArabicFC",
      "type": "dataset",
      "country": "INTL",
      "org": "TuwaiqAcademy",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TuwaiqAcademy/AISA-ArabicFC"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic Function Calling for Agentic AI Systems",
      "metrics": {
        "downloads": 122,
        "likes": 8,
        "lastModified": "2026-07-25"
      }
    },
    {
      "id": "amina",
      "name": "AMINA",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedZayton",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedZayton/AMINA"
      },
      "year": 2024,
      "notes": "We are pleased to introduce the AMINA : An Arabic Multi-Purpose Integral News Articles Dataset.",
      "metrics": {
        "downloads": 122,
        "likes": 8,
        "lastModified": "2024-09-20"
      }
    },
    {
      "id": "arabic-dialects",
      "name": "Arabic Dialects",
      "type": "dataset",
      "country": "INTL",
      "org": "drelhaj",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/Arabic-Dialects"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "The Arabic Dialects Dataset is a specialised corpus designed for automatic dialect identification.",
      "metrics": {
        "downloads": 122,
        "likes": 4,
        "lastModified": "2025-11-28"
      }
    },
    {
      "id": "aratruthfulqa",
      "name": "AraTruthfulQA",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/AraTruthfulQA"
      },
      "year": 2025,
      "notes": "287 culturally adapted TruthfulQA questions testing truthfulness on misconceptions common in the Arab world.",
      "metrics": {
        "downloads": 122,
        "likes": 3,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "madis5",
      "name": "MADIS5",
      "type": "dataset",
      "country": "INTL",
      "org": "badrex",
      "license": "cc",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/badrex/MADIS5-spoken-arabic-dialects"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Spoken Arabic dialects",
      "metrics": {
        "downloads": 122,
        "likes": 0,
        "lastModified": "2025-06-02"
      }
    },
    {
      "id": "arabic-turath-ocr",
      "name": "arabic turath ocr",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ocr",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/arabic-turath-ocr"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Page-level text of Arabic Turath books with raw and cleaned text and Arabic-character ratio, from digitized PDFs.",
      "metrics": {
        "downloads": 121,
        "likes": 0,
        "lastModified": "2026-04-22"
      }
    },
    {
      "id": "arabic-small-nougat",
      "name": "arabic-small-nougat",
      "type": "ocr",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedRashad/arabic-small-nougat"
      },
      "notes": "Smaller Nougat variant for Arabic document OCR",
      "base_model": [
        "facebook/nougat-small"
      ],
      "metrics": {
        "downloads": 121,
        "likes": 26,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "arsarcasm-v2",
      "name": "ArSarcasm-v2",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sarcasm",
        "sentiment",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/ArSarcasm_v2"
      },
      "year": 2021,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic tweets annotated for sarcasm, sentiment and dialect.",
      "metrics": {
        "downloads": 121,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "egyptianhellaswag",
      "name": "EgyptianHellaSwag",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "['mit']",
      "modality": "text",
      "tasks": [
        "commonsense"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/EgyptianHellaSwag"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "Machine-translated Egyptian Arabic (Masri) HellaSwag benchmark for commonsense reasoning and reading comprehension.",
      "metrics": {
        "downloads": 121,
        "likes": 1,
        "lastModified": "2025-04-15"
      }
    },
    {
      "id": "qameleon",
      "name": "QAmeleon",
      "type": "dataset",
      "country": "INTL",
      "org": "Google Research",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/google-research-datasets/QAmeleon",
        "hf": "https://huggingface.co/datasets/imvladikon/QAmeleon",
        "paper": "https://aclanthology.org/2023.tacl-1.98"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,970 sentences",
      "year": 2023,
      "notes": "Synthetic Arabic QA pairs generated by prompting PaLM-540B with 5 gold examples",
      "metrics": {
        "downloads": 121,
        "likes": 1,
        "lastModified": "2023-08-13"
      }
    },
    {
      "id": "tedxtn",
      "name": "TEDxTN",
      "type": "dataset",
      "country": "TN",
      "org": "Fethi Bougares",
      "license": "cc-by-nc-nd-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fbougares/TEDxTN"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "size": "1M–10M rows",
      "year": 2025,
      "notes": "Contact person : fethi.bougares@elyadata.com We introduce TEDxTN, the first publicly available Tunisian Arabic to English speech translation dataset.",
      "metrics": {
        "downloads": 120,
        "likes": 12,
        "lastModified": "2025-11-27"
      }
    },
    {
      "id": "lahja-sa-ahmad-v1",
      "name": "lahja sa ahmad v1",
      "type": "tts",
      "country": "INTL",
      "org": "wasmdashai",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/wasmdashai/lahja-sa-ahmad-v1"
      },
      "year": 2026,
      "notes": "Designed for both research and enterprise applications, Lahja SA Ahmad V1 enables developers to integrate realistic Arabic speech synthesis.",
      "base_model": [
        "wasmdashai/vits-ar-sa-a"
      ],
      "metrics": {
        "downloads": 119,
        "likes": 5,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "muharaf-public-pages",
      "name": "muharaf public pages",
      "type": "dataset",
      "country": "INTL",
      "org": "TheRealOKAI",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/TheRealOKAI/muharaf-public-pages"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset contains 1,218 full page images of Arabic handwriting and the corresponding text.",
      "metrics": {
        "downloads": 119,
        "likes": 7,
        "lastModified": "2025-12-01"
      }
    },
    {
      "id": "3arablm-4b-islamic-v2",
      "name": "3arabLM-4B-islamic-v2",
      "type": "llm",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "quran",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/3arabLM-4B-islamic-v2"
      },
      "dialects": [
        "classical"
      ],
      "size": "4B",
      "on_device": false,
      "year": 2026,
      "notes": "Arabic text generation model fine-tuned from sherif1313/3arabLM-4B-Fiqh-v1.",
      "base_model": [
        "sherif1313/3arablm-4b-fiqh-v1"
      ],
      "metrics": {
        "downloads": 118,
        "likes": 1,
        "lastModified": "2026-08-26"
      }
    },
    {
      "id": "arabic-gsm8k",
      "name": "Arabic GSM8K",
      "type": "benchmark",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-gsm8k"
      },
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic translation of the GSM8K grade-school math word problems.",
      "metrics": {
        "downloads": 118,
        "likes": 3,
        "lastModified": "2025-08-11"
      }
    },
    {
      "id": "masri-podcast-300h",
      "name": "Masri Podcast 300h",
      "type": "dataset",
      "country": "EG",
      "org": "Ehab Negm",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ehabnegm/masri-podcast-300h-egyptian-tts"
      },
      "dialects": [
        "egy"
      ],
      "notes": "300 hours of Egyptian podcast speech prepared for TTS training.",
      "metrics": {
        "downloads": 118,
        "likes": 0,
        "lastModified": "2026-09-07"
      }
    },
    {
      "id": "nemotron-asr-arabic-dialectal-v2",
      "name": "Nemotron 3.5 ASR Arabic Dialectal v2",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/nemotron-3.5-asr-arabic-dialectal-v2"
      },
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Streaming dialectal Arabic ASR fine-tuned from NVIDIA Nemotron 3.5 ASR on Lahgtna.",
      "base_model": [
        "nvidia/nemotron-3.5-asr-streaming-0.6b"
      ],
      "metrics": {
        "downloads": 118,
        "likes": 2,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "arabic-sts-matryoshka",
      "name": "Arabic STS Matryoshka",
      "type": "embedding",
      "country": "EG",
      "org": "Omar Elshehy",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/omarelshehy/Arabic-STS-Matryoshka"
      },
      "year": 2024,
      "notes": "This is an Arabic only sentence-transformers model finetuned from FacebookAI/xlm-roberta-large.",
      "base_model": [
        "facebookai/xlm-roberta-large"
      ],
      "metrics": {
        "downloads": 117,
        "likes": 2,
        "lastModified": "2024-10-13"
      }
    },
    {
      "id": "arcade-full",
      "name": "ARCADE full",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "speech",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/ARCADE-full"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "ARCADE is a city-scale corpus of Arabic radio speech designed for fine-grained dialect identification.",
      "metrics": {
        "downloads": 117,
        "likes": 5,
        "lastModified": "2026-01-12"
      }
    },
    {
      "id": "cohere-jordanian-dialect",
      "name": "Cohere Jordanian Dialect",
      "type": "asr",
      "country": "INTL",
      "org": "Rlamas",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Rlamas/Cohere-Jordanian-Dialect"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "This repository is self-contained — it includes the fine-tuned weights plus all processor/tokenizer files needed to run it directly, with no depende",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 117,
        "likes": 3,
        "lastModified": "2026-08-31"
      }
    },
    {
      "id": "dialect-router-v0-1",
      "name": "dialect router v0.1",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "tts",
        "dialect-id",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/dialect-router-v0.1"
      },
      "year": 2026,
      "notes": "A lightweight Arabic dialect identification model that classifies input text into one of 11 Arabic dialect / language codes.",
      "base_model": [
        "asafaya/bert-mini-arabic"
      ],
      "metrics": {
        "downloads": 117,
        "likes": 9,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "sa-retrieval-embeddings-0-2b",
      "name": "SA Retrieval Embeddings 0.2B",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/SA-Retrieval-Embeddings-0.2B"
      },
      "size": "0.2B",
      "on_device": true,
      "year": 2025,
      "notes": "This model is a retrieval-optimized SentenceTransformer, fine-tuned from Omartificial-Intelligence-Space/SA-STS-Embeddings-0.2B.",
      "base_model": [
        "omartificial-intelligence-space/sa-sts-embeddings-0.2b"
      ],
      "metrics": {
        "downloads": 117,
        "likes": 7,
        "lastModified": "2025-12-23"
      }
    },
    {
      "id": "tunisian-dialectic-english-derja",
      "name": "Tunisian Dialectic English Derja",
      "type": "dataset",
      "country": "INTL",
      "org": "khaled123",
      "license": "creativeml-openrail-m",
      "modality": "text",
      "tasks": [
        "asr",
        "translation",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/khaled123/Tunisian_Dialectic_English_Derja"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "This dataset is a rich and extensive collection of Tunisian dialectic (Derja) and English translations from various sources, updated as of October 2024.",
      "metrics": {
        "downloads": 117,
        "likes": 9,
        "lastModified": "2024-10-26"
      }
    },
    {
      "id": "arabic-history-and-dialects",
      "name": "arabic history and dialects",
      "type": "dataset",
      "country": "INTL",
      "org": "ISLAM-PO",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ISLAM-PO/arabic-history-and-dialects"
      },
      "dialects": [
        "egy",
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "Arabic Multi-Dialect & Civilization Instruction Dataset",
      "metrics": {
        "downloads": 115,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "arabic-quotes",
      "name": "arabic quotes",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/arabic-quotes"
      },
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "The \"Arabic Quotes\" dataset contains a collection of Arabic quotes along with their corresponding authors and tags.",
      "metrics": {
        "downloads": 115,
        "likes": 7,
        "lastModified": "2023-07-16"
      }
    },
    {
      "id": "egyptian-arabic-fake-reviews",
      "name": "egyptian arabic fake reviews",
      "type": "dataset",
      "country": "INTL",
      "org": "IbrahimAmin",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "fake-reviews",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IbrahimAmin/egyptian-arabic-fake-reviews"
      },
      "dialects": [
        "egy"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "FREAD: Egyptian Arabic fake reviews dataset with original and translated reviews and ratings.",
      "metrics": {
        "downloads": 115,
        "likes": 2,
        "lastModified": "2026-07-19"
      }
    },
    {
      "id": "arabic-ifeval-inception",
      "name": "Arabic IFEval (Inception)",
      "type": "benchmark",
      "country": "AE",
      "org": "Inception AI (G42)",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/inception42/Arabic-IFEval"
      },
      "year": 2025,
      "notes": "Arabic instruction-following evaluation set released by Inception.",
      "metrics": {
        "downloads": 114,
        "likes": 5,
        "lastModified": "2025-04-05"
      }
    },
    {
      "id": "arabic-civitai-images",
      "name": "Arabic-CivitAi-Images",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "multimodal",
      "tasks": [
        "image-captioning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/Arabic-CivitAi-Images"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Top 2K+ CivitAI images described with Qwen-VL-Max and translated into Arabic with Command-R.",
      "metrics": {
        "downloads": 114,
        "likes": 5,
        "lastModified": "2024-03-22"
      }
    },
    {
      "id": "araseg-2026-shared-task-pa",
      "name": "AraSeg-2026-Shared-Task-PA",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/AraSeg-2026-Shared-Task-PA"
      },
      "dialects": [
        "msa"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "The corpus is designed to support research on sentence segmentation in Modern Standard Arabic.",
      "metrics": {
        "downloads": 114,
        "likes": 1,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "levanti",
      "name": "levanti",
      "type": "dataset",
      "country": "INTL",
      "org": "guymorlan",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/guymorlan/levanti"
      },
      "dialects": [
        "lev",
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Levanti: 500K Levantine colloquial Arabic sentences translated to English and Hebrew with diacritics and transliterations.",
      "metrics": {
        "downloads": 114,
        "likes": 10,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "saudipedia-arabic-qa",
      "name": "saudipedia arabic qa",
      "type": "dataset",
      "country": "INTL",
      "org": "AhmadHakami",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "culture"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AhmadHakami/saudipedia-arabic-qa"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "1,082 question-answer pairs scraped from Saudipedia about Saudi culture and topics.",
      "metrics": {
        "downloads": 114,
        "likes": 3,
        "lastModified": "2025-09-03"
      }
    },
    {
      "id": "arabiangpt-03b",
      "name": "ArabianGPT-03B",
      "type": "llm",
      "country": "SA",
      "org": "riotu-lab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/riotu-lab/ArabianGPT-03B"
      },
      "size": "03B",
      "on_device": true,
      "year": 2024,
      "tags": [
        "variants:2"
      ],
      "notes": "ArabianGPT-0.3B: raw pre-trained Arabic GPT model from Prince Sultan University's RIOTU Lab.",
      "metrics": {
        "downloads": 113,
        "likes": 28,
        "lastModified": "2024-02-27"
      }
    },
    {
      "id": "arabic-audio-collection-tunisian-deep-confessions",
      "name": "arabic audio collection tunisian deep confessions",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "diacritization",
        "sentiment",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-audio-collection-tunisian-deep-confessions"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "The Deep Confessions Podcast Arabic Speech Dataset is a large-scale, first-of-its-kind Arabic speech corpus containing approximately 175 hours.",
      "metrics": {
        "downloads": 113,
        "likes": 1,
        "lastModified": "2026-06-22"
      }
    },
    {
      "id": "arabic-wikipedia-20230101-nobots",
      "name": "Arabic_Wikipedia_20230101_nobots",
      "type": "dataset",
      "country": "INTL",
      "org": "Clarkson University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/Arabic_Wikipedia_20230101_nobots",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.19.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "847,000 documents",
      "year": 2023,
      "notes": "ArabicWikipedia20230101nobots is a dataset created using the Arabic Wikipedia articles, excluding the bot-generated articles, downloaded on the 1st.",
      "metrics": {
        "downloads": 113,
        "likes": 2,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "araseg-2026-shared-task-nopnx-np",
      "name": "AraSeg-2026-Shared-Task-NoPnx-NP",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "ner",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/AraSeg-2026-Shared-Task-NoPnx-NP"
      },
      "dialects": [
        "msa"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "The corpus is designed to support research on sentence segmentation in Modern Standard Arabic.",
      "metrics": {
        "downloads": 113,
        "likes": 1,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "gate-arabert-v0",
      "name": "GATE-AraBert-v0",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/GATE-AraBert-v0"
      },
      "year": 2025,
      "notes": "This is a General Arabic Text Embedding trained using SentenceTransformers in a multi-task setup.",
      "base_model": [
        "omartificial-intelligence-space/arabert-all-nli-triplet-matryoshka"
      ],
      "metrics": {
        "downloads": 113,
        "likes": 1,
        "lastModified": "2025-01-23"
      }
    },
    {
      "id": "islamicfaithqa",
      "name": "IslamicFaithQA",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/IslamicFaithQA"
      },
      "size": "1K-10K rows",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Bilingual Arabic/English generative Islamic QA benchmark for faithfulness evaluation.",
      "metrics": {
        "downloads": 113,
        "likes": 3,
        "lastModified": "2026-04-30"
      }
    },
    {
      "id": "jeem",
      "name": "JEEM",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "evaluation",
        "image-captioning",
        "vqa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/toloka/JEEM",
        "paper": "http://arxiv.org/pdf/2503.21910v1.pdf"
      },
      "dialects": [
        "mixed",
        "lev",
        "gulf",
        "egy",
        "magh"
      ],
      "size": "2,178 images",
      "year": 2025,
      "notes": "Benchmark for VLMs visual understanding across four Arabic dialects",
      "metrics": {
        "downloads": 113,
        "likes": 14,
        "lastModified": "2025-05-11"
      }
    },
    {
      "id": "probel",
      "name": "ProBel",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ProBel"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic and English news sentences and social-media posts, each annotated with a categories, technique-labeled character spans.",
      "metrics": {
        "downloads": 113,
        "likes": 0,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "saudi-dialect-speech-female",
      "name": "saudi dialect speech female",
      "type": "benchmark",
      "country": "SA",
      "org": "Ahmed Eladl",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AhmedEladl/saudi-dialect-speech-female"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "This repository contains cleaned, segmented, and dual-transcribed Arabic speech data intended for speech modeling, ASR benchmarking, and Text-to-Speech.",
      "metrics": {
        "downloads": 113,
        "likes": 1,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "algerian-arabic-english-50k",
      "name": "Algerian Arabic-English 50K",
      "type": "dataset",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/algerian-nlp/algerian-arabic-english-translation-50k"
      },
      "size": "50k",
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "50,000 aligned Algerian Darja to English sentence pairs; Algeria.",
      "metrics": {
        "downloads": 112,
        "likes": 0,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "tarbiyah-ai-v1-1",
      "name": "tarbiyah ai v1 1",
      "type": "asr",
      "country": "INTL",
      "org": "Habib-HF",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/Habib-HF/tarbiyah-ai-v1-1"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of OpenAI's whisper-small model, specifically adapted for Automatic Speech Recognition (ASR) of Quranic Arabic recitation.",
      "metrics": {
        "downloads": 112,
        "likes": 3,
        "lastModified": "2025-06-25"
      }
    },
    {
      "id": "nawah-asr-118m-v5",
      "name": "Nawah ASR 118M v5",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-ASR-118M-v5"
      },
      "dialects": [
        "egy"
      ],
      "size": "118M",
      "on_device": true,
      "year": 2026,
      "notes": "Arabic ASR with no third-party weights anywhere in the stack.",
      "metrics": {
        "downloads": 111,
        "likes": 2,
        "lastModified": "2026-09-08"
      }
    },
    {
      "id": "sitr-arabic-pii",
      "name": "Sitr Arabic PII",
      "type": "dataset",
      "country": "SA",
      "org": "mabahboh",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner",
        "pii"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mabahboh/sitr-arabic-pii"
      },
      "year": 2026,
      "dialects": [
        "gulf",
        "lev"
      ],
      "notes": "Arabic PII detection dataset with 43 personal-data types over 17,899 samples.",
      "metrics": {
        "downloads": 111,
        "likes": 0,
        "lastModified": "2026-09-19"
      }
    },
    {
      "id": "arabic-ljp",
      "name": "Arabic LJP",
      "type": "benchmark",
      "country": "INTL",
      "org": "mbayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mbayan/Arabic-LJP"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset is designed for Arabic Legal Judgment Prediction (LJP), collected and preprocessed from Saudi commercial court judgments.",
      "metrics": {
        "downloads": 110,
        "likes": 5,
        "lastModified": "2025-02-28"
      }
    },
    {
      "id": "fannorflop",
      "name": "FannOrFlop",
      "type": "benchmark",
      "country": "INTL",
      "org": "omkarthawakar",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/omkarthawakar/FannOrFlop"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Fann or Flop: multigenre, multi-era benchmark testing how well LLMs understand Arabic poetry.",
      "metrics": {
        "downloads": 110,
        "likes": 11,
        "lastModified": "2025-08-21"
      }
    },
    {
      "id": "arabic-kw-mdel",
      "name": "Arabic KW Mdel",
      "type": "embedding",
      "country": "INTL",
      "org": "medmediani",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/medmediani/Arabic-KW-Mdel"
      },
      "year": 2023,
      "notes": "This is a sentence-transformers model: It maps sentences & paragraphs to a 768 dimensional dense vector space and can be used.",
      "metrics": {
        "downloads": 109,
        "likes": 5,
        "lastModified": "2023-04-30"
      }
    },
    {
      "id": "darija-to-french-speech-to-text",
      "name": "darija to french speech to text",
      "type": "dataset",
      "country": "INTL",
      "org": "adiren7",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/adiren7/darija_to_french_speech_to_text"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2025,
      "notes": "94 Darija speech segments with start/end times and French transcriptions for speech-to-text.",
      "metrics": {
        "downloads": 109,
        "likes": 9,
        "lastModified": "2025-02-16"
      }
    },
    {
      "id": "dziralign",
      "name": "DziriAlign",
      "type": "dataset",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/algerian-nlp/DziriAlign"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Algerian Darja alignment dataset (Algeria).",
      "metrics": {
        "downloads": 109,
        "likes": 0,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "saudi-dialect-speech-male",
      "name": "Saudi Dialect Speech (Male)",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Saudi speech)",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AhmedEladl/saudi-dialect-speech-male"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Single-speaker male Saudi-dialect speech for TTS.",
      "metrics": {
        "downloads": 109,
        "likes": 0,
        "lastModified": "2025-08-23"
      }
    },
    {
      "id": "whisper-algerian-dialect",
      "name": "whisper algerian dialect",
      "type": "asr",
      "country": "INTL",
      "org": "MohammedNasri",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MohammedNasri/whisper-algerian-dialect"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of OpenAI's Whisper-tiny specifically for Algerian dialect automatic speech recognition (ASR).",
      "metrics": {
        "downloads": 109,
        "likes": 5,
        "lastModified": "2025-08-13"
      }
    },
    {
      "id": "acegpt-7b-chat-gguf-heshamharoon",
      "name": "AceGPT-7B-chat-GGUF",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/AceGPT-7B-chat-GGUF"
      },
      "size": "7B",
      "on_device": true,
      "year": 2024,
      "notes": "GGUF quantizations (Q4_K_M, Q5_K_M) of AceGPT-7B-chat for local CPU inference; the card gives no other details.",
      "metrics": {
        "downloads": 108,
        "likes": 1,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "arabic-punctuation",
      "name": "arabic punctuation",
      "type": "dataset",
      "country": "INTL",
      "org": "Asas AI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/asas-ai/arabic_punctuation"
      },
      "size": "10M–100M rows",
      "year": 2024,
      "notes": "This is a curated dataset, specifically designed to facilitate the study of punctuation.",
      "metrics": {
        "downloads": 108,
        "likes": 2,
        "lastModified": "2024-02-12"
      }
    },
    {
      "id": "peacock",
      "name": "Peacock",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "other",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/Peacock"
      },
      "size": "7B",
      "notes": "Arabic multimodal (InstructBLIP + AraLLaMA)",
      "metrics": {
        "downloads": 108,
        "likes": 2,
        "lastModified": "2024-11-25"
      }
    },
    {
      "id": "shako-iraqi-4b",
      "name": "Shako Iraqi 4B",
      "type": "llm",
      "country": "IQ",
      "org": "Anas Pro",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/anaspro/Shako-iraqi-4B-it"
      },
      "size": "8.4B",
      "dialects": [
        "iraqi"
      ],
      "notes": "Instruction-tuned 4B model for Iraqi dialect (Iraq).",
      "base_model": [
        "unsloth/gemma-3n-e4b-it"
      ],
      "metrics": {
        "downloads": 108,
        "likes": 2,
        "lastModified": "2025-11-14"
      }
    },
    {
      "id": "lahja-sa-huba-v1",
      "name": "lahja sa huba v1",
      "type": "tts",
      "country": "INTL",
      "org": "wasmdashai",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/wasmdashai/lahja-sa-huba-v1"
      },
      "year": 2026,
      "notes": "Designed for both research and enterprise applications, Lahja SA Huba V1 enables developers to integrate realistic Arabic speech synthesis.",
      "base_model": [
        "wasmdashai/vits-ar-sa-huba-v2"
      ],
      "metrics": {
        "downloads": 107,
        "likes": 3,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "leva-tts",
      "name": "leva tts",
      "type": "tts",
      "country": "INTL",
      "org": "Mohammed Aly",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammedaly22/leva-tts"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "XTTS-v2 fine-tune for low-latency code-switched Levantine Arabic and English TTS in conversational agents.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 107,
        "likes": 3,
        "lastModified": "2026-06-10"
      }
    },
    {
      "id": "mawps-ar",
      "name": "MaWPS-ar",
      "type": "dataset",
      "country": "EG",
      "org": "Omar Adel, Alexandria University",
      "license": "['mit']",
      "modality": "text",
      "tasks": [
        "math"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/omarxadel/MaWPS-ar"
      },
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "Arabic-English version of the MAWPS math word problem repository, with equations.",
      "metrics": {
        "downloads": 107,
        "likes": 1,
        "lastModified": "2022-07-12"
      }
    },
    {
      "id": "sambalingo-arabic-base",
      "name": "SambaLingo-Arabic-Base",
      "type": "llm",
      "country": "INTL",
      "org": "SambaNova",
      "license": "llama2",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/sambanovasystems/SambaLingo-Arabic-Base"
      },
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "SambaLingo-Arabic-Base is a pretrained Bi-lingual Arabic and English model that adapts Llama 2 to Arabic by training on 63 billion tokens.",
      "metrics": {
        "downloads": 107,
        "likes": 37,
        "lastModified": "2024-05-14"
      }
    },
    {
      "id": "ien-mcq",
      "name": "IEN-MCQ",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/IEN_MCQ"
      },
      "year": 2025,
      "notes": "About 10K Arabic school multiple-choice questions across subjects and grades from Saudi Ien platform.",
      "metrics": {
        "downloads": 106,
        "likes": 1,
        "lastModified": "2025-10-16"
      }
    },
    {
      "id": "nawah-router-bert-6m-bilingual",
      "name": "Nawah Router BERT 6M bilingual",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-Router-BERT-6M-bilingual"
      },
      "dialects": [
        "msa"
      ],
      "size": "6M",
      "on_device": true,
      "year": 2026,
      "notes": "Give it a text and any categories in plain English or Arabic; it scores all of them in one forward pass.",
      "base_model": [
        "oddadmix/nawah-bert-6m-v2"
      ],
      "metrics": {
        "downloads": 106,
        "likes": 1,
        "lastModified": "2026-09-06"
      }
    },
    {
      "id": "tadabur-whisper-small",
      "name": "tadabur Whisper Small",
      "type": "asr",
      "country": "INTL",
      "org": "FaisaI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/FaisaI/tadabur-Whisper-Small"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Tadabur-Whisper-Small A Whisper Small model fine-tuned on Tadabur for Qur'anic speech recognition. [![Base Model](https://img.shields.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 106,
        "likes": 18,
        "lastModified": "2026-04-24"
      }
    },
    {
      "id": "asfar",
      "name": "Asfar",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "ocr",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/asfar"
      },
      "size": "123k pages",
      "year": 2026,
      "dialects": [
        "classical",
        "msa"
      ],
      "notes": "Page-level corpus of classical Arabic heritage: 123k pages from 461 PDF volumes.",
      "metrics": {
        "downloads": 105,
        "likes": 7,
        "lastModified": "2026-04-22"
      }
    },
    {
      "id": "arabic-math-reasoning-synth",
      "name": "arabic math reasoning synth",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-math-reasoning-synth"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Generated with gemma-3-12b-it and Qwen3.8-27B-Uncensored-NVFP4 and own arithmetic does not check out were dropped.",
      "metrics": {
        "downloads": 104,
        "likes": 0,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "artst-asr-v2",
      "name": "artst asr v2",
      "type": "asr",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/artst_asr_v2"
      },
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "ArTST model finetuned for automatic speech recognition (speech-to-text) on MGB2. (also: 1 variants)",
      "metrics": {
        "downloads": 104,
        "likes": 2,
        "lastModified": "2025-09-10"
      }
    },
    {
      "id": "evol-instruct-arabic-gpt4",
      "name": "Evol-Instruct Arabic GPT4",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Evol-Instruct-Arabic-GPT4"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic translation of Evol-Instruct-70k with GPT-4 answers.",
      "metrics": {
        "downloads": 104,
        "likes": 4,
        "lastModified": "2023-12-06"
      }
    },
    {
      "id": "propxplain",
      "name": "PropXplain",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/PropXplain"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "PropXplain is a multilingual dataset for explainable propaganda detection in Arabic and English text.",
      "metrics": {
        "downloads": 104,
        "likes": 0,
        "lastModified": "2026-07-29"
      }
    },
    {
      "id": "50m-darija-english-v1",
      "name": "50M Darija English v1",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/50M-Darija-English-v1"
      },
      "dialects": [
        "magh"
      ],
      "size": "50M",
      "on_device": true,
      "year": 2026,
      "notes": "A 51.8M-parameter small language model that translates both ways between both directions; a direction-specific system prompt selects which way to translate.",
      "base_model": [
        "oddadmix/50m-2048-emhotob"
      ],
      "metrics": {
        "downloads": 103,
        "likes": 0,
        "lastModified": "2026-07-17"
      }
    },
    {
      "id": "barec-shared-task-2025-sent",
      "name": "BAREC Shared Task 2025 sent",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CAMeL-Lab/BAREC-Shared-Task-2025-sent"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "The dataset is annotated at the sentence level. (also: 1 variants)",
      "metrics": {
        "downloads": 103,
        "likes": 2,
        "lastModified": "2025-06-11"
      }
    },
    {
      "id": "fasee7-najdi-small",
      "name": "Fasee7-Najdi-Small",
      "type": "llm",
      "country": "SA",
      "org": "Wittify.ai",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Wittify/Fasee7-Najdi-Small"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Small Najdi-dialect Arabic chat model from Wittify.",
      "base_model": [
        "openbmb/voxcpm2"
      ],
      "metrics": {
        "downloads": 103,
        "likes": 4,
        "lastModified": "2026-06-28"
      }
    },
    {
      "id": "palm",
      "name": "palm",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-nd-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/palm"
      },
      "notes": "Human-created Arabic instruction dataset",
      "metrics": {
        "downloads": 103,
        "likes": 21,
        "lastModified": "2025-10-28"
      }
    },
    {
      "id": "saudi-dialect-asr-v1",
      "name": "Saudi Dialect ASR v1.0",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Saudi speech)",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/musabalosimi/saudi_dialect_asrv1.0"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi-dialect speech-transcription dataset.",
      "metrics": {
        "downloads": 103,
        "likes": 4,
        "lastModified": "2025-09-16"
      }
    },
    {
      "id": "stt-arabic-whisper-finetuned-diactires",
      "name": "stt arabic whisper finetuned diactires",
      "type": "asr",
      "country": "INTL",
      "org": "NightPrince",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/NightPrince/stt-arabic-whisper-finetuned-diactires"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Fine-tuned openai/whisper-small on tarteel-ai/everyayah for Automatic Speech Recognition of Quranic recitation with complete tashkeel.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 103,
        "likes": 3,
        "lastModified": "2026-03-18"
      }
    },
    {
      "id": "acva",
      "name": "ACVA",
      "type": "benchmark",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/ACVA-Arabic-Cultural-Value-Alignment"
      },
      "notes": "Arabic Cultural Value Alignment (8000+ questions, 58 areas)",
      "metrics": {
        "downloads": 102,
        "likes": 8,
        "lastModified": "2023-09-21"
      }
    },
    {
      "id": "arabic-base-all-nli-stsb-quora",
      "name": "Arabic base all nli stsb quora",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-base-all-nli-stsb-quora"
      },
      "year": 2024,
      "notes": "This is a sentence-transformers model finetuned from google-bert/bert-base-multilingual-cased on the all-nli-pair, all-nli-pair-class, all-nli-pair-score.",
      "base_model": [
        "google-bert/bert-base-multilingual-cased"
      ],
      "metrics": {
        "downloads": 102,
        "likes": 1,
        "lastModified": "2024-06-28"
      }
    },
    {
      "id": "modernbert-morocco-sentence-embeddings-v0-2-bs-32-lr-2e-05-ep-2-wp-0-0",
      "name": "ModernBERT-Morocco-Sentence-Embeddings-v0.2-bs-32-lr-2e-05-ep-2-wp-0.05-gacc-1-gnm-1.0-v0.3",
      "type": "embedding",
      "country": "INTL",
      "org": "BounharAbdelaziz",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/BounharAbdelaziz/ModernBERT-Morocco-Sentence-Embeddings-v0.2-bs-32-lr-2e-05-ep-2-wp-0.05-gacc-1-gnm-1.0-v0.3"
      },
      "year": 2025,
      "notes": "This is a sentence-transformers model finetuned from BounharAbdelaziz/ModernBERT-Morocco on the triplet, negationtriplet, pairscore and [englishnonenglish]",
      "base_model": [
        "bounharabdelaziz/modernbert-morocco"
      ],
      "metrics": {
        "downloads": 102,
        "likes": 0,
        "lastModified": "2025-02-20"
      }
    },
    {
      "id": "qasr",
      "name": "QASR",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/QASR"
      },
      "size": "2000h",
      "year": 2021,
      "dialects": [
        "mixed"
      ],
      "notes": "QCRI Aljazeera Speech Resource, ~2,000h of transcribed Arabic broadcast speech with multi-layer annotation.",
      "metrics": {
        "downloads": 102,
        "likes": 5,
        "lastModified": "2025-10-13"
      }
    },
    {
      "id": "sanadset",
      "name": "Sanadset",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Sanadset"
      },
      "year": 2021,
      "dialects": [
        "classical"
      ],
      "notes": "Hadith narration dataset with transmission chains (isnad) annotated for narrators.",
      "metrics": {
        "downloads": 102,
        "likes": 4,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "arabic-whisper-multidialect",
      "name": "arabic whisper multidialect",
      "type": "dataset",
      "country": "INTL",
      "org": "MadLook",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MadLook/arabic-whisper-multidialect"
      },
      "dialects": [
        "mixed"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "A comprehensive multi-dialect Arabic speech recognition dataset prepared for Whisper model fine-tuning.",
      "metrics": {
        "downloads": 101,
        "likes": 3,
        "lastModified": "2025-11-13"
      }
    },
    {
      "id": "arabic-large-nougat",
      "name": "arabic-large-nougat",
      "type": "ocr",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedRashad/arabic-large-nougat"
      },
      "notes": "End-to-end structured OCR for Arabic documents",
      "metrics": {
        "downloads": 101,
        "likes": 18,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "gemmaroc-27b-it",
      "name": "GemMaroc-27b-it",
      "type": "llm",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/AbderrahmanSkiredj1/GemMaroc-27b-it"
      },
      "base_model": [
        "google/gemma-3-27b-it"
      ],
      "dialects": [
        "magh"
      ],
      "size": "27B",
      "on_device": false,
      "year": 2025,
      "notes": "Unlocking Moroccan Darija proficiency in a state‑of‑the‑art large language model.",
      "metrics": {
        "downloads": 101,
        "likes": 3,
        "lastModified": "2025-06-18"
      }
    },
    {
      "id": "harrier-arabic-matryoshka-0-6b",
      "name": "Harrier Arabic Matryoshka 0.6B",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Harrier-Arabic-Matryoshka-0.6B"
      },
      "size": "0.6B",
      "on_device": true,
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "A 0.6B-parameter Arabic sentence embedding model based on microsoft/harrier-oss-v1-0.6b.",
      "base_model": [
        "microsoft/harrier-oss-v1-0.6b"
      ],
      "metrics": {
        "downloads": 101,
        "likes": 5,
        "lastModified": "2026-04-29"
      }
    },
    {
      "id": "mixed-arabic-dataset-main",
      "name": "Mixed Arabic Dataset Main",
      "type": "dataset",
      "country": "INTL",
      "org": "M-A-D",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/M-A-D/Mixed-Arabic-Dataset-Main"
      },
      "size": "100K–1M rows",
      "year": 2023,
      "tags": [
        "variants:1"
      ],
      "notes": "The Mixed Arabic Datasets (MAD) project provides a comprehensive collection of diverse Arabic-language datasets, sourced from various repositories.",
      "metrics": {
        "downloads": 101,
        "likes": 7,
        "lastModified": "2023-10-06"
      }
    },
    {
      "id": "personachat-arabic",
      "name": "personachat arabic",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/personachat-arabic"
      },
      "year": 2023,
      "notes": "Conversational model fine-tuned on a machine-translated Arabic subset of PersonaChat; limited training data.",
      "metrics": {
        "downloads": 101,
        "likes": 11,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "sada22-msa",
      "name": "SADA22 (MSA)",
      "type": "dataset",
      "country": "INTL",
      "org": "badrex",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/badrex/arabic-speech-SADA22-MSA"
      },
      "notes": "MSA subset of SADA, Khaliji speech",
      "metrics": {
        "downloads": 101,
        "likes": 2,
        "lastModified": "2025-05-12"
      }
    },
    {
      "id": "sharegpt-arabic",
      "name": "sharegpt arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/sharegpt-arabic"
      },
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "Arabic text dataset (1K<N<10K rows).",
      "metrics": {
        "downloads": 101,
        "likes": 5,
        "lastModified": "2023-08-13"
      }
    },
    {
      "id": "absher",
      "name": "Absher",
      "type": "benchmark",
      "country": "SA",
      "org": "King Khalid University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "github": "https://github.com/renad-01/Absher-Benchmark",
        "hf": "https://huggingface.co/datasets/Renad10/Absher-Benchmark",
        "paper": "https://arxiv.org/pdf/2507.10216.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "18,564 sentences",
      "year": 2025,
      "notes": "Absher is a culturally grounded benchmark designed to evaluate the ability of Large Language Models (LLMs) to understand Saudi Arabic dialects.",
      "metrics": {
        "downloads": 100,
        "likes": 1,
        "lastModified": "2026-01-24"
      }
    },
    {
      "id": "paxqa",
      "name": "PAXQA",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Pennsylvania",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/manestay/paxqa_val_test",
        "paper": "https://doi.org/10.18653/v1/2023.findings-emnlp.32"
      },
      "dialects": [
        "msa"
      ],
      "size": "593 sentences",
      "year": 2023,
      "tags": [
        "multilingual"
      ],
      "notes": "A synthetic Arabic-English cross-lingual QA dataset generated with parallel corpora and alignment-based translation.",
      "metrics": {
        "downloads": 100,
        "likes": 1,
        "lastModified": "2024-06-14"
      }
    },
    {
      "id": "whisperlevantine",
      "name": "WhisperLevantine",
      "type": "asr",
      "country": "INTL",
      "org": "HebArabNlpProject",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/HebArabNlpProject/WhisperLevantine"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "notes": "This model is a fine-tuned version of Whisper Larg v3 tailored specifically for transcribing Levantine Arabic, focusing on the Israeli dialect.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 100,
        "likes": 7,
        "lastModified": "2025-05-25"
      }
    },
    {
      "id": "f5-tts-arabic",
      "name": "F5-TTS-Arabic",
      "type": "tts",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimSalah/F5-TTS-Arabic"
      },
      "notes": "F5-TTS with regional diversity",
      "base_model": [
        "swivid/f5-tts"
      ],
      "metrics": {
        "downloads": 99,
        "likes": 27,
        "lastModified": "2025-11-13"
      }
    },
    {
      "id": "rasam",
      "name": "RASAM",
      "type": "dataset",
      "country": "INTL",
      "org": "johnlockejrr",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/johnlockejrr/RASAM"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "An Open Dataset for the Recognition and Analysis of Scripts in Arabic Maghrebi The paper has been presented during the ICDAR 2021 conference (ASAR workshop).",
      "metrics": {
        "downloads": 99,
        "likes": 5,
        "lastModified": "2024-07-02"
      }
    },
    {
      "id": "hindawi-books-dataset",
      "name": "Hindawi Books dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "alielfilali01",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alielfilali01/Hindawi-Books-dataset"
      },
      "dialects": [
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "Hindawi Books Dataset offers a rich and diverse collection of literary works, covering various topics and genres, all written in Modern Standard Arabic.",
      "metrics": {
        "downloads": 98,
        "likes": 14,
        "lastModified": "2023-08-03"
      }
    },
    {
      "id": "the-arabic-e-book-corpus",
      "name": "The Arabic E-Book Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "mohres",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohres/The_Arabic_E-Book_Corpus"
      },
      "notes": "1,745 books (81.5M words)",
      "metrics": {
        "downloads": 98,
        "likes": 3,
        "lastModified": "2024-06-09"
      }
    },
    {
      "id": "dzirieval",
      "name": "DziriEval",
      "type": "benchmark",
      "country": "DZ",
      "org": "Algerian NLP",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/algerian-nlp/DziriEval"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Evaluation set for Algerian Darja language models (Algeria).",
      "metrics": {
        "downloads": 97,
        "likes": 0,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "dahee7",
      "name": "Dahee7",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Dahee7"
      },
      "year": 2024,
      "notes": "Arabic speech-transcription dataset with about 7.5k training utterances.",
      "metrics": {
        "downloads": 96,
        "likes": 7,
        "lastModified": "2024-02-06"
      }
    },
    {
      "id": "madar-tun",
      "name": "MADAR-TUN",
      "type": "dataset",
      "country": "INTL",
      "org": "Univ. Grenoble Alpes",
      "license": "cc-by-nc-3.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/eligugliotta/MADAR-TUN",
        "hf": "https://huggingface.co/datasets/tunis-ai/MADAR-TUN",
        "paper": "https://aclanthology.org/2023.ldk-1.14.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "3,990 sentences",
      "year": 2023,
      "notes": "Linguistic annotations for the Tunisian part of the MADAR corpus (Bouamor et al., 2018)",
      "metrics": {
        "downloads": 96,
        "likes": 0,
        "lastModified": "2025-10-05"
      }
    },
    {
      "id": "arabic-medical-dialogue",
      "name": "arabic medical dialogue",
      "type": "dataset",
      "country": "INTL",
      "org": "Mars203020",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "medical",
        "dialogue"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Mars203020/arabic_medical_dialogue"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic medical dialogue dataset for text generation.",
      "metrics": {
        "downloads": 95,
        "likes": 3,
        "lastModified": "2024-06-29"
      }
    },
    {
      "id": "arabic-tts-wav-24k",
      "name": "arabic tts wav 24k",
      "type": "dataset",
      "country": "INTL",
      "org": "NeoBoy",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NeoBoy/arabic-tts-wav-24k"
      },
      "dialects": [
        "msa"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "A high-quality, open-source dataset for Arabic Text-to-Speech (TTS) research, containing paired audio and text samples from both male and female speakers.",
      "metrics": {
        "downloads": 95,
        "likes": 3,
        "lastModified": "2025-06-15"
      }
    },
    {
      "id": "atlasia-darija-english",
      "name": "AtlasIA Darija-English",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/darija_english"
      },
      "size": "100K-1M pairs",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Compilation of Darija-English sentence pairs curated by AtlasIA.",
      "metrics": {
        "downloads": 95,
        "likes": 14,
        "lastModified": "2024-05-16"
      }
    },
    {
      "id": "mgb-3",
      "name": "MGB 3",
      "type": "dataset",
      "country": "INTL",
      "org": "Elyadata",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabicSpeech/MGB-3"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "6,547 MGB-3 Egyptian-Arabic speech rows with audio and transcript.",
      "dialects": [
        "egy"
      ],
      "metrics": {
        "downloads": 95,
        "likes": 4,
        "lastModified": "2025-08-11"
      }
    },
    {
      "id": "saudispell-arat5",
      "name": "SaudiSpell-AraT5",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "spelling-correction"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/SaudiSpell-AraT5"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "AraT5 sequence-to-sequence spelling corrector fine-tuned on a balanced Saudi dialect corpus.",
      "base_model": [
        "ubc-nlp/arat5v2-base-1024"
      ],
      "metrics": {
        "downloads": 95,
        "likes": 6,
        "lastModified": "2026-01-26"
      }
    },
    {
      "id": "arabic-news",
      "name": "Arabic News",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "news",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_News"
      },
      "size": "1M–10M rows",
      "year": 2022,
      "notes": "7.3M Arabic news text rows for language modelling.",
      "metrics": {
        "downloads": 94,
        "likes": 2,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "egyptian-arabic-hate-speech",
      "name": "egyptian arabic hate speech",
      "type": "dataset",
      "country": "EG",
      "org": "Ibrahim Amin",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IbrahimAmin/egyptian-arabic-hate-speech"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset consists of 8,169 Egyptian-Arabic text samples manually labeled for offensive language and hate speech classificati",
      "metrics": {
        "downloads": 94,
        "likes": 2,
        "lastModified": "2025-08-17"
      }
    },
    {
      "id": "voho-saudi-stt-small",
      "name": "voho saudi stt small",
      "type": "asr",
      "country": "SA",
      "org": "Voho AI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/VohoAI/voho-saudi-stt-small"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "year": 2026,
      "notes": "Most Arabic speech models are trained on Modern Standard Arabic: the news, not a phone call from Riyadh.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 94,
        "likes": 2,
        "lastModified": "2026-09-14"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-msa-sixteenth",
      "name": "bert base arabic camelbert msa sixteenth",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-msa-sixteenth"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2021,
      "notes": "We release pre-trained language models for Modern Standard Arabic (MSA), dialectal Arabic (DA), and classical Arabic.",
      "metrics": {
        "downloads": 93,
        "likes": 4,
        "lastModified": "2021-09-14"
      }
    },
    {
      "id": "lc-eval",
      "name": "LC-Eval",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/LC-Eval"
      },
      "year": 2025,
      "notes": "Bilingual Arabic-English long-context benchmark with contexts from 4K to over 128K tokens.",
      "metrics": {
        "downloads": 93,
        "likes": 2,
        "lastModified": "2025-09-15"
      }
    },
    {
      "id": "arabic-quranic-asr",
      "name": "arabic quranic asr",
      "type": "dataset",
      "country": "INTL",
      "org": "Sadique5",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Sadique5/arabic_quranic_asr"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "This dataset contains quran recitations of every ayats or verses.",
      "metrics": {
        "downloads": 92,
        "likes": 4,
        "lastModified": "2024-07-04"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-mix-pos-egy",
      "name": "bert base arabic camelbert mix pos egy",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-mix-pos-egy"
      },
      "dialects": [
        "egy"
      ],
      "year": 2021,
      "notes": "For the fine-tuning, we used the ARZTB dataset .",
      "metrics": {
        "downloads": 92,
        "likes": 3,
        "lastModified": "2021-10-18"
      }
    },
    {
      "id": "detoxllm",
      "name": "DetoxLLM",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "detoxification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/DetoxLLM"
      },
      "year": 2024,
      "notes": "Detoxification dataset from the DetoxLLM framework paper, includes Arabic rewrites.",
      "metrics": {
        "downloads": 91,
        "likes": 1,
        "lastModified": "2024-11-12"
      }
    },
    {
      "id": "f5-tts-egyptian-arabic",
      "name": "f5-tts-egyptian-arabic",
      "type": "tts",
      "country": "INTL",
      "org": "MAdel121",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/MAdel121/f5-tts-egyptian-arabic"
      },
      "year": 2026,
      "dialects": [
        "egy"
      ],
      "notes": "F5-TTS fine-tuned for Egyptian Arabic.",
      "base_model": [
        "swivid/habibi-tts"
      ],
      "metrics": {
        "downloads": 91,
        "likes": 2,
        "lastModified": "2026-02-13"
      }
    },
    {
      "id": "cohere-transcribe-arabic-cpu",
      "name": "Cohere Transcribe Arabic CPU-Friendly",
      "type": "asr",
      "country": "INTL",
      "org": "sayedM",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/sayedM/cohere-transcribe-arabic-cpu-friendly"
      },
      "year": 2026,
      "notes": "Int8 CPU-oriented build of Cohere Transcribe Arabic with speculative decoding.",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 90,
        "likes": 5,
        "lastModified": "2026-09-06"
      }
    },
    {
      "id": "atlaset",
      "name": "Atlaset",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Atlaset"
      },
      "size": "1M-10M rows",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Large Moroccan Darija text corpus compiled by AtlasIA (1M-10M rows), gated.",
      "metrics": {
        "downloads": 89,
        "likes": 33,
        "lastModified": "2025-03-31"
      }
    },
    {
      "id": "egyptian-arabic-islamic-qa-dataset",
      "name": "Egyptian Arabic Islamic QA Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "islamic"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omar-youssef/islamic-qa-egyptian-arabic"
      },
      "dialects": [
        "egy"
      ],
      "size": "7,465 sentences",
      "year": 2025,
      "notes": "7,465 question-answer pairs in Egyptian Arabic covering Islamic studies topics.",
      "metrics": {
        "downloads": 89,
        "likes": 1,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "evol-instruct-arabic",
      "name": "evol instruct arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/evol-instruct-arabic"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "The dataset is used in the research related to MultilingualSIFT.",
      "metrics": {
        "downloads": 89,
        "likes": 2,
        "lastModified": "2023-08-06"
      }
    },
    {
      "id": "magpie-tts-saudi-arabic",
      "name": "Magpie-TTS-Saudi-Arabic",
      "type": "tts",
      "country": "SA",
      "org": "Ahmed Eladl",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedEladl/Magpie-TTS-Saudi-Arabic"
      },
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "Magpie TTS fine-tuned for Saudi Arabic.",
      "base_model": [
        "nvidia/magpie_tts_multilingual_357m"
      ],
      "metrics": {
        "downloads": 89,
        "likes": 3,
        "lastModified": "2026-08-24"
      }
    },
    {
      "id": "acegpt-v2-alignment",
      "name": "AceGPT-v2 AlignmentData",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "alignment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/AceGPT-v2-AlignmentData"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Data to train a small alignment model for native-Arabic filtering in AceGPT-v2.",
      "metrics": {
        "downloads": 88,
        "likes": 0,
        "lastModified": "2025-12-29"
      }
    },
    {
      "id": "aslad-190k",
      "name": "ASLAD-190K",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Constantine",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/2fgpn5dwgc/2",
        "hf": "https://huggingface.co/datasets/aboulesnane/ASLAD-190K",
        "paper": "https://osf.io/preprints/osf/n236q_v2"
      },
      "dialects": [
        "msa"
      ],
      "size": "190,000 images",
      "year": 2024,
      "notes": "A dataset of 190,000 annotated RGB images of 32 Arabic sign-language letters and gestures collected under varied lighting, distance, and background conditions.",
      "metrics": {
        "downloads": 88,
        "likes": 0,
        "lastModified": "2025-09-19"
      }
    },
    {
      "id": "quran-question-answer-context",
      "name": "quran question answer context",
      "type": "dataset",
      "country": "INTL",
      "org": "nazimali",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "translation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nazimali/quran-question-answer-context"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Translated the original dataset from Arabic to English and added the Surah ayahs to the context column.",
      "metrics": {
        "downloads": 88,
        "likes": 10,
        "lastModified": "2024-09-04"
      }
    },
    {
      "id": "sambalingo-arabic",
      "name": "SambaLingo-Arabic",
      "type": "llm",
      "country": "INTL",
      "org": "SambaNova",
      "license": "llama2",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/sambanovasystems/SambaLingo-Arabic-Chat"
      },
      "size": "7B, 70B",
      "notes": "Arabic-adapted Llama 2",
      "metrics": {
        "downloads": 88,
        "likes": 64,
        "lastModified": "2024-04-16"
      }
    },
    {
      "id": "whisper-largev3-medical",
      "name": "whisper largev3 medical",
      "type": "asr",
      "country": "INTL",
      "org": "yehiazak",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/yehiazak/whisper-largev3-medical"
      },
      "year": 2025,
      "notes": "This model is a fine-tuned version of openai/whisper-large-v3 on medical speech data.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 88,
        "likes": 8,
        "lastModified": "2025-07-12"
      }
    },
    {
      "id": "darijabert-mix",
      "name": "DarijaBERT-mix",
      "type": "llm",
      "country": "MA",
      "org": "SI2M Lab, INSEA",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/SI2M-Lab/DarijaBERT-mix"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Moroccan Darija BERT encoder, the mix checkpoint from SI2M Lab (INSEA).",
      "metrics": {
        "downloads": 87,
        "likes": 2,
        "lastModified": "2024-09-25"
      }
    },
    {
      "id": "mushkil",
      "name": "mushkil",
      "type": "llm",
      "country": "SA",
      "org": "riotu-lab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/riotu-lab/mushkil"
      },
      "year": 2024,
      "notes": "AraT5v2 fine-tuned for Arabic diacritization, framed as translation from undiacritized to diacritized text.",
      "metrics": {
        "downloads": 87,
        "likes": 2,
        "lastModified": "2024-05-02"
      }
    },
    {
      "id": "arabic-speech-commands-dataset",
      "name": "Arabic Speech Commands Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Syrian Virtual University",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/abdulkaderghandoura/arabic-speech-commands-dataset",
        "hf": "https://huggingface.co/datasets/arbml/Speech_Commands_Dataset",
        "paper": "https://www.sciencedirect.com/science/article/pii/S0952197621001147"
      },
      "dialects": [
        "msa"
      ],
      "size": "3 hours",
      "year": 2021,
      "notes": "This dataset is designed to help train simple machine learning models that serve educational and research purposes in the speech recognition domain",
      "metrics": {
        "downloads": 86,
        "likes": 1,
        "lastModified": "2022-11-01"
      }
    },
    {
      "id": "aradpr",
      "name": "AraDPR",
      "type": "llm",
      "country": "INTL",
      "org": "DataScienceUIBK",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/abdoelsayed/AraDPR"
      },
      "year": 2024,
      "notes": "AraDPR is a state-of-the-art dense passage retrieval model specifically designed for the Arabic language.",
      "metrics": {
        "downloads": 86,
        "likes": 2,
        "lastModified": "2024-03-27"
      }
    },
    {
      "id": "dimi-embedding",
      "name": "DIMI-embedding",
      "type": "embedding",
      "country": "INTL",
      "org": "AhmedZaky1",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedZaky1/DIMI-embedding-matryoshka-arabic"
      },
      "notes": "Matryoshka + AraBERT for NLI",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 86,
        "likes": 4,
        "lastModified": "2025-05-30"
      }
    },
    {
      "id": "qwen3-embedding-0-6b-arabic-ecom",
      "name": "Qwen3-Embedding-0.6B Arabic E-commerce",
      "type": "embedding",
      "country": "INTL",
      "org": "Presto AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/prestoai/qwen3-embedding-0.6b-arabic-ecom"
      },
      "size": "0.6B",
      "year": 2026,
      "on_device": true,
      "notes": "LoRA-tuned Arabic e-commerce search embedding with a companion benchmark.",
      "base_model": [
        "qwen/qwen3-embedding-0.6b"
      ],
      "metrics": {
        "downloads": 86,
        "likes": 1,
        "lastModified": "2026-06-28"
      }
    },
    {
      "id": "aosed",
      "name": "AOSED",
      "type": "benchmark",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/bayandashnan/AOSED",
        "paper": "https://doi.org/10.1145/3838729"
      },
      "dialects": [
        "msa"
      ],
      "size": "100 sentences",
      "year": 2026,
      "notes": "AOSED is the first evaluation benchmark for Arabic opinion summarization.",
      "metrics": {
        "downloads": 85,
        "likes": 0,
        "lastModified": "2026-08-03"
      }
    },
    {
      "id": "arabic-text-embedding-for-sts",
      "name": "Arabic text embedding for sts",
      "type": "embedding",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/AbderrahmanSkiredj1/Arabic_text_embedding_for_sts"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "This is a sentence-transformers model trained on the AbderrahmanSkiredj1/arabicquoraduplicatesstsbalueholyquranaranli900kanchorpositivenegative dataset.",
      "metrics": {
        "downloads": 85,
        "likes": 6,
        "lastModified": "2024-07-07"
      }
    },
    {
      "id": "fineweb-edu-ar",
      "name": "fineweb edu ar",
      "type": "dataset",
      "country": "SA",
      "org": "KAUST Generative AI",
      "license": "['cc-by-nc-4.0']",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/kaust-generative-ai/fineweb-edu-ar"
      },
      "size": "100M–1B rows",
      "year": 2024,
      "notes": "Machine-translated Arabic version of FineWeb-Edu (about 202B tokens, via NLLB-200 600M) for small language models.",
      "metrics": {
        "downloads": 85,
        "likes": 13,
        "lastModified": "2024-11-12"
      }
    },
    {
      "id": "al-atlas-0-5b",
      "name": "Al-Atlas-0.5B",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Al-Atlas-0.5B"
      },
      "base_model": [
        "qwen/qwen2.5-0.5b"
      ],
      "size": "494M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "magh"
      ],
      "notes": "Al-Atlas 0.5B, small Moroccan Darija language model.",
      "metrics": {
        "downloads": 84,
        "likes": 15,
        "lastModified": "2026-05-22"
      }
    },
    {
      "id": "darija-sft-mixture",
      "name": "Darija-SFT-Mixture",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/Darija-SFT-Mixture"
      },
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Darija instruction-tuning mixture used to train Atlas-Chat.",
      "metrics": {
        "downloads": 84,
        "likes": 18,
        "lastModified": "2025-05-02"
      }
    },
    {
      "id": "eg-legal-qa",
      "name": "eg legal qa",
      "type": "dataset",
      "country": "INTL",
      "org": "fr3on",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "chat",
        "instruction-tuning",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fr3on/eg-legal-qa"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Question-answering dataset for Arabic legal texts with instruction-following format for training conversational AI models.",
      "metrics": {
        "downloads": 84,
        "likes": 2,
        "lastModified": "2025-09-20"
      }
    },
    {
      "id": "mizan-iraqi-benchmark",
      "name": "Mizan Iraqi Arabic Benchmark",
      "type": "benchmark",
      "country": "IQ",
      "org": "Nawar Alseelawi",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nawaralseelawi/mizan-iraqi-arabic-benchmark"
      },
      "year": 2026,
      "dialects": [
        "iraqi"
      ],
      "notes": "Originally authored Iraqi Arabic LLM benchmark, pilot public dev set; Iraq.",
      "metrics": {
        "downloads": 84,
        "likes": 1,
        "lastModified": "2026-09-12"
      }
    },
    {
      "id": "python-assistant",
      "name": "python assistant",
      "type": "llm",
      "country": "INTL",
      "org": "jana-ashraf-ai",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/jana-ashraf-ai/python-assistant"
      },
      "year": 2026,
      "notes": "A fine-tuned version of Qwen2.5-1.5B-Instruct that answers Python programming questions in Arabic, with structured JSON output.",
      "base_model": [
        "qwen/qwen2.5-1.5b-instruct"
      ],
      "metrics": {
        "downloads": 84,
        "likes": 5,
        "lastModified": "2026-04-02"
      }
    },
    {
      "id": "quran-tafsir-tarteel-ai",
      "name": "quran tafsir",
      "type": "dataset",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarteel-ai/quran-tafsir"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "Quran tafsir and translations in many English editions, one column per translator.",
      "metrics": {
        "downloads": 84,
        "likes": 9,
        "lastModified": "2023-02-11"
      }
    },
    {
      "id": "arabic-nli-triplet",
      "name": "Arabic NLI Triplet",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-NLi-Triplet"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic SNLI and MultiNLI as anchor-positive-negative triplets used to train Arabic Matryoshka embeddings.",
      "metrics": {
        "downloads": 83,
        "likes": 6,
        "lastModified": "2024-08-02"
      }
    },
    {
      "id": "arzen-multigenre",
      "name": "ArzEn-MultiGenre",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/ArzEn-MultiGenre"
      },
      "size": "1K-10K pairs",
      "year": 2023,
      "dialects": [
        "egy"
      ],
      "notes": "Parallel Egyptian Arabic-English dataset across songs, novels and TV subtitles.",
      "metrics": {
        "downloads": 83,
        "likes": 12,
        "lastModified": "2023-12-31"
      }
    },
    {
      "id": "egyptian-punctuation-restoration",
      "name": "egyptian punctuation restoration",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "punctuation-restoration"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/egyptian-punctuation-restoration"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Restores punctuation in raw Egyptian Arabic text such as chat logs and transcribed speech.",
      "metrics": {
        "downloads": 83,
        "likes": 6,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "lemura-arabic-asr-qwen3",
      "name": "Lemura Arabic ASR Qwen3",
      "type": "asr",
      "country": "INTL",
      "org": "Lemura Labs",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/lemuralabs/lemura-arabic-asr-qwen3"
      },
      "year": 2026,
      "dialects": [
        "msa",
        "gulf",
        "egy"
      ],
      "notes": "Qwen3-ASR based multi-dialect Arabic speech recognizer from Lemura Labs.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 83,
        "likes": 2,
        "lastModified": "2026-08-21"
      }
    },
    {
      "id": "arabic-poem-gen",
      "name": "arabic poem gen",
      "type": "llm",
      "country": "INTL",
      "org": "usama98",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/usama98/arabic_poem_gen"
      },
      "year": 2022,
      "notes": "To save computation time the model used pretrained weights from another [model](https://huggingf",
      "metrics": {
        "downloads": 82,
        "likes": 4,
        "lastModified": "2022-05-31"
      }
    },
    {
      "id": "arabic-colbert-100k",
      "name": "Arabic-ColBERT-100K",
      "type": "embedding",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/Arabic-ColBERT-100K"
      },
      "year": 2024,
      "notes": "First version of Arabic ColBERT (better models are available now - see the 250k and 711k ones).",
      "base_model": [
        "aubmindlab/bert-base-arabertv02"
      ],
      "metrics": {
        "downloads": 82,
        "likes": 4,
        "lastModified": "2024-08-15"
      }
    },
    {
      "id": "arentail",
      "name": "ArEntail",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/ArEntail",
        "paper": "https://link.springer.com/article/10.1007/s10579-024-09731-1"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,000 sentences",
      "year": 2024,
      "notes": "Arabic NLI dataset called ArEntail, consisting of 6000 sentence pairs collected from news headlines and manually labeled to indicate whether an entailment.",
      "metrics": {
        "downloads": 82,
        "likes": 1,
        "lastModified": "2024-05-04"
      }
    },
    {
      "id": "asfar-grpo",
      "name": "asfar-grpo",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "question-answering"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/asfar-grpo"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Classical Arabic heritage (turath) tasks for GRPO/RLVR: QA, fill-mask and multiple choice, derived from the asfar corpus.",
      "metrics": {
        "downloads": 82,
        "likes": 1,
        "lastModified": "2026-04-24"
      }
    },
    {
      "id": "athar",
      "name": "ATHAR",
      "type": "dataset",
      "country": "INTL",
      "org": "mohamed-khalil",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohamed-khalil/ATHAR"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Welcome to ATHAR Dataset [ Paper ] GitHub ]",
      "metrics": {
        "downloads": 82,
        "likes": 13,
        "lastModified": "2024-08-04"
      }
    },
    {
      "id": "arabert-arabic-ner-conllpp",
      "name": "AraBert-Arabic-NER-CoNLLpp",
      "type": "llm",
      "country": "INTL",
      "org": "MostafaAhmed98",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/MostafaAhmed98/AraBert-Arabic-NER-CoNLLpp"
      },
      "year": 2024,
      "notes": "AraBERT fine-tuned for Arabic named-entity recognition on CoNLL++.",
      "metrics": {
        "downloads": 81,
        "likes": 3,
        "lastModified": "2024-06-20"
      }
    },
    {
      "id": "arafinnews",
      "name": "AraFinNews",
      "type": "dataset",
      "country": "INTL",
      "org": "Mo El-Haj (ArabicNLP.uk)",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "summarization",
        "news"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/AraFinNews"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "Arabic financial news dataset for summarisation, released by Mo El-Haj on Hugging Face.",
      "metrics": {
        "downloads": 81,
        "likes": 1,
        "lastModified": "2025-11-30"
      }
    },
    {
      "id": "hala-4-6m-sft",
      "name": "Hala 4.6M SFT",
      "type": "dataset",
      "country": "SA",
      "org": "KAUST",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hammh0a/Hala-4.6M-SFT"
      },
      "size": "1M–10M rows",
      "year": 2025,
      "notes": "Hala: 4.06M Arabic-centric instruction and translation conversations from the KAUST Hala technical report.",
      "metrics": {
        "downloads": 81,
        "likes": 5,
        "lastModified": "2025-09-18"
      }
    },
    {
      "id": "moroccan-darija-youtube-subtitles",
      "name": "moroccan darija youtube subtitles",
      "type": "dataset",
      "country": "INTL",
      "org": "bourbouh",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/bourbouh/moroccan-darija-youtube-subtitles"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2024,
      "notes": "248 transcripts of Moroccan Darija YouTube videos from popular channels.",
      "metrics": {
        "downloads": 81,
        "likes": 3,
        "lastModified": "2024-04-03"
      }
    },
    {
      "id": "qwen3-4b-oman-qlora",
      "name": "qwen3-4b-oman-qlora",
      "type": "llm",
      "country": "INTL",
      "org": "menasaat",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "rag"
      ],
      "links": {
        "hf": "https://huggingface.co/menasaat/qwen3-4b-oman-qlora"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Qwen3-4B QLoRA fine-tune focused on Oman, released by Menasaat.",
      "base_model": [
        "qwen/qwen3-4b"
      ],
      "metrics": {
        "downloads": 81,
        "likes": 0,
        "lastModified": "2026-09-21"
      }
    },
    {
      "id": "doda-10k",
      "name": "DODa 10K",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "['cc-by-nc-4.0']",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/DODa-10K"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "DODa-10K: Darija Open Dataset subset of translation quintuples between Darija (Arabic and Latin script) and other languages.",
      "dialects": [
        "magh"
      ],
      "metrics": {
        "downloads": 80,
        "likes": 3,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "dah-dataset-hassaniya",
      "name": "DAH (DAtaset Hassaniya)",
      "type": "dataset",
      "country": "INTL",
      "org": "Hassan-IA",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/Hassan-IA/DAH",
        "hf": "https://huggingface.co/datasets/hassan-IA/dah"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "First public bilingual Hassaniya-English translation dataset.",
      "metrics": {
        "downloads": 79,
        "likes": 6,
        "lastModified": "2026-03-26"
      }
    },
    {
      "id": "general-facts-in-english-arabic-egyptian-arabic",
      "name": "General Facts in English Arabic Egyptian Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "miscovery",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/miscovery/General_Facts_in_English_Arabic_Egyptian_Arabic"
      },
      "dialects": [
        "egy"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "The World Facts General Knowledge Dataset (v1.0) is a high-quality, human-reviewed Q&A resource by Miscovery.",
      "metrics": {
        "downloads": 79,
        "likes": 12,
        "lastModified": "2025-04-06"
      }
    },
    {
      "id": "masriswitch-gemma3n",
      "name": "MasriSwitch-Gemma3n",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/MasriSwitch-Gemma3n-Transcriber-v1"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic code-switching transcription",
      "base_model": [
        "unsloth/gemma-3n-e4b-it"
      ],
      "metrics": {
        "downloads": 79,
        "likes": 16,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "the-arabic-news-speech-corpus-dataset",
      "name": "The Arabic News speech Corpus Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "diacritization",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IbrahimSalah/The_Arabic_News_speech_Corpus_Dataset"
      },
      "dialects": [
        "msa"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "This dataset is an Arabic speech corpus that supports the development of syllable-based Arabic speech recognition using Wav2Vec-2 architecture.",
      "metrics": {
        "downloads": 79,
        "likes": 6,
        "lastModified": "2024-09-29"
      }
    },
    {
      "id": "arabguard-egyptian-v1",
      "name": "ArabGuard-Egyptian-V1",
      "type": "dataset",
      "country": "INTL",
      "org": "d12o6aa",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/d12o6aa/ArabGuard-Egyptian-V1"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Global safety guardrails often exhibit a \"Linguistic Blind Spot\" when faced with local cultural nuances, slang, or code-switching.",
      "metrics": {
        "downloads": 78,
        "likes": 5,
        "lastModified": "2026-03-09"
      }
    },
    {
      "id": "dz-emobert",
      "name": "Dz-EmoBERT",
      "type": "llm",
      "country": "INTL",
      "org": "Houdna-khilouf",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "sentiment",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/Houdna-khilouf/Dz-EmoBERT"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "To validate the quality of the Dz-Emotion dataset, we fine-tuned a transformer-based model, resulting in Dz-EmoBERT.",
      "base_model": [
        "alger-ia/dziribert"
      ],
      "metrics": {
        "downloads": 78,
        "likes": 3,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "arabic-books-and-research-dataset",
      "name": "Arabic books and research dataset",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/Arabic-books-and-research-dataset"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "About 10 GB of cleaned, previously unpublished Arabic Islamic research and book text extracted from 60K Word files.",
      "metrics": {
        "downloads": 77,
        "likes": 6,
        "lastModified": "2024-06-27"
      }
    },
    {
      "id": "arabic-multi-classification-dataset-amcd",
      "name": "Arabic-Multi-Classification-Dataset-AMCD",
      "type": "dataset",
      "country": "INTL",
      "org": "Waelyafooz",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "summarization"
      ],
      "links": {
        "github": "https://github.com/waelyafooz/Arabic-Multi-Classification-Dataset-AMCD",
        "hf": "https://huggingface.co/datasets/arbml/AMCD"
      },
      "dialects": [
        "mixed"
      ],
      "size": "8,046 sentences",
      "year": 2021,
      "notes": "Arabic Multi Classification Dataset (AMCD) of YouTube video metadata and comments for text mining, clustering, and classification.",
      "metrics": {
        "downloads": 77,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arabiccultural-qa",
      "name": "ArabicCulturalQA",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ArabicCulturalQA"
      },
      "size": "10K-100K rows",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Cross-dialectal cultural QA benchmark with parallel MCQ and open-ended formats across MSA and dialects.",
      "metrics": {
        "downloads": 77,
        "likes": 2,
        "lastModified": "2026-06-21"
      }
    },
    {
      "id": "egyptian-dpo-mixture",
      "name": "Egyptian DPO Mixture",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/Egyptian-DPO-Mixture"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset supports Direct Preference Optimization (DPO) fine-tuning using off-policy alignment signals to enhance stylistic control.",
      "metrics": {
        "downloads": 77,
        "likes": 3,
        "lastModified": "2025-07-07"
      }
    },
    {
      "id": "arabic-hate-speech",
      "name": "Arabic Hate Speech",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Hate_Speech"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "9,823 Arabic tweets labelled for offensive, hate, vulgar and violent content.",
      "metrics": {
        "downloads": 76,
        "likes": 8,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "arabic-rc",
      "name": "Arabic RC",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "reading-comprehension",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_RC"
      },
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "1,008 Arabic reading comprehension questions with passages and question class labels.",
      "metrics": {
        "downloads": 76,
        "likes": 5,
        "lastModified": "2022-10-05"
      }
    },
    {
      "id": "e5-base-mlqa-finetuned-arabic-for-rag",
      "name": "e5 base mlqa finetuned arabic for rag",
      "type": "embedding",
      "country": "INTL",
      "org": "OmarAlsaabi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/OmarAlsaabi/e5-base-mlqa-finetuned-arabic-for-rag"
      },
      "year": 2024,
      "notes": "This is a sentence-transformers model: It maps sentences & paragraphs to a 768 dimensional dense vector space and can be used.",
      "metrics": {
        "downloads": 76,
        "likes": 5,
        "lastModified": "2024-02-07"
      }
    },
    {
      "id": "egyptian-arabic-wikipedia-20230101",
      "name": "Egyptian Arabic Wikipedia 20230101",
      "type": "dataset",
      "country": "INTL",
      "org": "SaiedAlshahrani",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/Egyptian_Arabic_Wikipedia_20230101"
      },
      "dialects": [
        "egy"
      ],
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "This dataset is created using the Egyptian Arabic Wikipedia articles, downloaded on the 1st of January 2023, processed using Gensim Python library.",
      "metrics": {
        "downloads": 76,
        "likes": 7,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "namaa-saudi-asr-v1",
      "name": "NAMAA Saudi ASR V1",
      "type": "asr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/NAMAA-Saudi-ASR-V1"
      },
      "year": 2026,
      "dialects": [
        "gulf"
      ],
      "notes": "LoRA adapter specializing Cohere Transcribe Arabic for Saudi dialect speech.",
      "base_model": [
        "coherelabs/cohere-transcribe-arabic-07-2026"
      ],
      "metrics": {
        "downloads": 76,
        "likes": 2,
        "lastModified": "2026-09-16"
      }
    },
    {
      "id": "sib-200",
      "name": "SIB-200",
      "type": "benchmark",
      "country": "INTL",
      "org": "University College London",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "classification"
      ],
      "links": {
        "github": "https://github.com/dadelani/SIB-200",
        "hf": "https://huggingface.co/datasets/Davlan/sib200_14classes",
        "paper": "https://arxiv.org/pdf/2309.07445"
      },
      "dialects": [
        "mixed",
        "msa",
        "iraqi",
        "yemeni",
        "magh",
        "lev"
      ],
      "size": "11,440 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "A large-scale open-source benchmark dataset for topic classification in 200+ languages and dialects, based on Flores-200.",
      "metrics": {
        "downloads": 76,
        "likes": 0,
        "lastModified": "2025-05-27"
      }
    },
    {
      "id": "aragemma-embedding-300m",
      "name": "AraGemma-Embedding-300m",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/AraGemma-Embedding-300m"
      },
      "size": "303M",
      "year": 2025,
      "on_device": true,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic embedding model built on EmbeddingGemma 300M.",
      "base_model": [
        "google/embeddinggemma-300m"
      ],
      "metrics": {
        "downloads": 75,
        "likes": 15,
        "lastModified": "2025-09-07"
      }
    },
    {
      "id": "atlas-chat-27b",
      "name": "Atlas-Chat-27B",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris/Atlas-Chat-27B"
      },
      "base_model": [
        "google/gemma-2-27b-it"
      ],
      "size": "27.2B",
      "year": 2024,
      "dialects": [
        "magh"
      ],
      "notes": "Atlas-Chat 27B, largest Moroccan Darija instruction model of the family.",
      "metrics": {
        "downloads": 75,
        "likes": 16,
        "lastModified": "2024-10-24"
      }
    },
    {
      "id": "darija-dataset",
      "name": "Darija Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "JasperV13",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "language-modeling"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/JasperV13/Darija_Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "Moroccan Darija text corpus of 2.7M rows (about 1.4 GB) with a source column.",
      "metrics": {
        "downloads": 75,
        "likes": 5,
        "lastModified": "2024-07-29"
      }
    },
    {
      "id": "hubert-large-arabic",
      "name": "HuBERT-Large Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "asafaya",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/asafaya/hubert-large-arabic-transcribe"
      },
      "notes": "HuBERT fine-tuned on 2000h Arabic, 17.68% WER",
      "metrics": {
        "downloads": 75,
        "likes": 4,
        "lastModified": "2022-12-26"
      }
    },
    {
      "id": "whisper-large-arabic-dialects-v5",
      "name": "whisper-large-arabic-dialects-v5",
      "type": "asr",
      "country": "INTL",
      "org": "samil24",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/samil24/whisper-large-arabic-dialects-v5"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh"
      ],
      "year": 2025,
      "notes": "Whisper-large fine-tuned on multi-dialect Arabic speech.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 75,
        "likes": 2,
        "lastModified": "2025-09-03"
      }
    },
    {
      "id": "arabic-tts-xtts-v2",
      "name": "arabic tts xtts v2",
      "type": "tts",
      "country": "INTL",
      "org": "Moeeldouma",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/Moeeldouma/arabic-tts-xtts-v2"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "A systematic project to improve Coqui XTTS-v2 for high-quality Arabic text-to-speech generation.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 73,
        "likes": 5,
        "lastModified": "2026-04-30"
      }
    },
    {
      "id": "quran-bil-quran",
      "name": "quran bil quran",
      "type": "dataset",
      "country": "INTL",
      "org": "iqrossed",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/iqrossed/quran-bil-quran"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "A linguistic dataset connecting Quranic verses through their Arabic roots — a digital Mufharis (concordance) with semantic family layers.",
      "metrics": {
        "downloads": 73,
        "likes": 4,
        "lastModified": "2026-03-09"
      }
    },
    {
      "id": "silma-arabic-triplets-dataset-v1-0",
      "name": "silma arabic triplets dataset v1.0",
      "type": "dataset",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/silma-ai/silma-arabic-triplets-dataset-v1.0"
      },
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "The SILMA Arabic Triplets Dataset - v1.0 is a high-quality, diverse dataset specifically curated for training and training embedding models.",
      "metrics": {
        "downloads": 73,
        "likes": 4,
        "lastModified": "2024-10-17"
      }
    },
    {
      "id": "algerianme5",
      "name": "algerianME5",
      "type": "embedding",
      "country": "INTL",
      "org": "81melody",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/81melody/algerianME5"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Sentence-Transformer (768-dim) tuned on multilingual-e5 for Algerian car and real-estate search mixing Arabic, French and darja.",
      "base_model": [
        "intfloat/multilingual-e5-base"
      ],
      "metrics": {
        "downloads": 72,
        "likes": 3,
        "lastModified": "2026-04-19"
      }
    },
    {
      "id": "arabic-hate-speech-superset",
      "name": "arabic-hate-speech-superset",
      "type": "dataset",
      "country": "INTL",
      "org": "manueltonneau",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/manueltonneau/arabic-hate-speech-superset"
      },
      "notes": "Comprehensive hate speech detection",
      "metrics": {
        "downloads": 72,
        "likes": 8,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "fineweb2-msa",
      "name": "FineWeb2-MSA",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "odc-by",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/FineWeb2-MSA"
      },
      "dialects": [
        "egy",
        "msa"
      ],
      "size": "100M–1B rows",
      "year": 2024,
      "notes": "This is the MSA Arabic Portion of The FineWeb2 Dataset.",
      "metrics": {
        "downloads": 72,
        "likes": 2,
        "lastModified": "2024-12-15"
      }
    },
    {
      "id": "gemma-iraqi-finetune-v2",
      "name": "gemma iraqi finetune v2",
      "type": "llm",
      "country": "IQ",
      "org": "Ameer Wisam",
      "license": "gemma",
      "modality": "text",
      "tasks": [
        "chat",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/ameer4wisam/gemma-iraqi-finetune-v2"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2026,
      "tags": [
        "variants:3"
      ],
      "notes": "نموذج محادثة باللهجة العراقية مبني على google/gemma-4-E12B-it بتقنية LoRA ثم دُمج بالأوزان الأساسية (bf16). (also: 3 variants)",
      "base_model": [
        "google/gemma-4-12b-it"
      ],
      "metrics": {
        "downloads": 72,
        "likes": 2,
        "lastModified": "2026-07-29"
      }
    },
    {
      "id": "lahgtna-chatterbox-v1",
      "name": "Lahgtna Chatterbox v1",
      "type": "tts",
      "country": "EG",
      "org": "oddadmix",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "voice-cloning"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/lahgtna-chatterbox-v1"
      },
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Multi-dialect Arabic TTS on Chatterbox covering Egyptian, Saudi, Gulf, Iraqi and Moroccan.",
      "base_model": [
        "resembleai/chatterbox"
      ],
      "metrics": {
        "downloads": 72,
        "likes": 16,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "masriswitch-bench",
      "name": "MasriSwitch-Bench",
      "type": "benchmark",
      "country": "EG",
      "org": "Tarek737",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Tarek737/MasriSwitch-Bench"
      },
      "dialects": [
        "egy",
        "mixed"
      ],
      "notes": "Benchmark for Egyptian Arabic / English code-switched speech recognition.",
      "metrics": {
        "downloads": 72,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "ultimate-arabic-news",
      "name": "ultimate arabic news",
      "type": "dataset",
      "country": "INTL",
      "org": "khalidalt",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "news"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/khalidalt/ultimate_arabic_news"
      },
      "size": "100K–1M rows",
      "year": 2022,
      "notes": "Collection of single-label modern Arabic news-website and press-article texts for text classification.",
      "metrics": {
        "downloads": 72,
        "likes": 4,
        "lastModified": "2022-06-15"
      }
    },
    {
      "id": "ultrafeedback-arabic",
      "name": "ultrafeedback arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "alielfilali01",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preference",
        "rlhf"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alielfilali01/ultrafeedback-arabic"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "63K Arabic UltraFeedback preference pairs with chosen/rejected responses and scores.",
      "metrics": {
        "downloads": 72,
        "likes": 3,
        "lastModified": "2024-01-31"
      }
    },
    {
      "id": "arabic-speech-sada22-khaliji",
      "name": "arabic speech SADA22 Khaliji",
      "type": "dataset",
      "country": "INTL",
      "org": "badrex",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/badrex/arabic-speech-SADA22-Khaliji"
      },
      "dialects": [
        "gulf"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "This is only the portion of the SADA dataset where the speaker dialect is Khaliji.",
      "metrics": {
        "downloads": 71,
        "likes": 5,
        "lastModified": "2025-05-12"
      }
    },
    {
      "id": "eg-legal-rag",
      "name": "eg legal rag",
      "type": "dataset",
      "country": "INTL",
      "org": "fr3on",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fr3on/eg-legal-rag"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Retrieval-augmented generation optimized dataset with summaries, keywords, and cross-references for building legal search systems.",
      "metrics": {
        "downloads": 71,
        "likes": 4,
        "lastModified": "2025-09-20"
      }
    },
    {
      "id": "gazelle-benchmark",
      "name": "gazelle benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-nd-4.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/gazelle_benchmark"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "Writing has long been considered a hallmark of human intelligence and remains a pinnacle task for artificial intelligence.",
      "metrics": {
        "downloads": 71,
        "likes": 9,
        "lastModified": "2024-12-23"
      }
    },
    {
      "id": "lebanese-llama-3-1-8b",
      "name": "Lebanese Llama 3.1 8B",
      "type": "llm",
      "country": "LB",
      "org": "Assix Research",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/assix-research/lebanese-llama-3.1-8b"
      },
      "base_model": [
        "unsloth/meta-llama-3.1-8b-instruct-bnb-4bit"
      ],
      "size": "8.0B",
      "dialects": [
        "lev"
      ],
      "notes": "Llama-3.1-8B fine-tuned for Lebanese dialect.",
      "metrics": {
        "downloads": 71,
        "likes": 2,
        "lastModified": "2026-01-23"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-53-arabic-egyptian",
      "name": "wav2vec2-large-xlsr-53-arabic-egyptian",
      "type": "asr",
      "country": "SA",
      "org": "ARBML",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/arbml/wav2vec2-large-xlsr-53-arabic-egyptian"
      },
      "year": 2022,
      "dialects": [
        "egy"
      ],
      "notes": "wav2vec2 XLSR-53 fine-tuned on Egyptian Arabic.",
      "metrics": {
        "downloads": 71,
        "likes": 16,
        "lastModified": "2021-07-05"
      }
    },
    {
      "id": "doctr-model-v1-arabic",
      "name": "doctr model v1 arabic",
      "type": "ocr",
      "country": "INTL",
      "org": "rania-sr",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/rania-sr/doctr-model-v1-arabic"
      },
      "year": 2025,
      "notes": "docTR text-recognition model trained for Arabic (TensorFlow 2 and PyTorch).",
      "metrics": {
        "downloads": 70,
        "likes": 3,
        "lastModified": "2025-06-14"
      }
    },
    {
      "id": "hassaniya-stories-ocr",
      "name": "hassaniya stories ocr",
      "type": "dataset",
      "country": "INTL",
      "org": "hassan-IA",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hassan-IA/hassaniya-stories-ocr"
      },
      "size": "<1K rows",
      "year": 2026,
      "notes": "Image-to-text OCR pairs extracted from a Hassaniya Arabic stories corpus.",
      "metrics": {
        "downloads": 70,
        "likes": 4,
        "lastModified": "2026-06-01"
      }
    },
    {
      "id": "palmx-gc",
      "name": "PalmX-GC",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "other",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/palmx_2025_subtask1_culture",
        "paper": "https://doi.org/10.18653/v1/2025.arabicnlp-sharedtasks.107"
      },
      "dialects": [
        "msa"
      ],
      "size": "4,500 sentences",
      "year": 2025,
      "notes": "PalmX-GC evaluates a model’s grasp of general Arab culture—customs, history, geography, arts, cuisine, notable figures, and everyday life across the 22 Arab.",
      "metrics": {
        "downloads": 70,
        "likes": 1,
        "lastModified": "2025-10-07"
      }
    },
    {
      "id": "saudi-tts-v4",
      "name": "Saudi tts v4",
      "type": "tts",
      "country": "INTL",
      "org": "khalidhabbash",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/khalidhabbash/Saudi-tts-v4"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Saudi TTS V4 is a 162.7M-parameter, reference-conditioned F5-TTS/DiT acoustic model adapted from SILMA TTS v1.",
      "base_model": [
        "silma-ai/silma-tts"
      ],
      "metrics": {
        "downloads": 70,
        "likes": 3,
        "lastModified": "2026-09-13"
      }
    },
    {
      "id": "sudanese-dialect-speech",
      "name": "Sudanese Dialect Speech",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/sudanese_dialect_speech"
      },
      "year": 2022,
      "dialects": [
        "sudanese"
      ],
      "notes": "Sudanese Arabic dialect speech recordings with transcripts.",
      "metrics": {
        "downloads": 70,
        "likes": 6,
        "lastModified": "2022-12-04"
      }
    },
    {
      "id": "3lm-native-stem-arabic-benchmark",
      "name": "3LM Native STEM Arabic Benchmark",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "other",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "github": "https://github.com/tiiuae/3LM-benchmark",
        "hf": "https://huggingface.co/datasets/tiiuae/NativeQA",
        "paper": "https://aclanthology.org/2025.arabicnlp-main.4.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "865 sentences",
      "year": 2025,
      "notes": "The 3LM Native STEM dataset contains 865 multiple-choice questions (MCQs) curated from real Arabic educational sources.",
      "metrics": {
        "downloads": 69,
        "likes": 2,
        "lastModified": "2026-02-14"
      }
    },
    {
      "id": "alpaca-arabic",
      "name": "alpaca arabic",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/alpaca_arabic"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "tags": [
        "variants:3"
      ],
      "notes": "52K Arabic translation of the Alpaca instruction set with English originals kept alongside.",
      "metrics": {
        "downloads": 69,
        "likes": 4,
        "lastModified": "2023-08-11"
      }
    },
    {
      "id": "goat-llama3-1-v0-1",
      "name": "GOAT llama3.1 v0.1",
      "type": "llm",
      "country": "INTL",
      "org": "kshabana",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/kshabana/GOAT-llama3.1-v0.1"
      },
      "year": 2024,
      "notes": "we are happy to announce that we are starting our goat model family this is a finetune model of llama3.1 that can perform well.",
      "metrics": {
        "downloads": 69,
        "likes": 3,
        "lastModified": "2024-08-15"
      }
    },
    {
      "id": "gpt2-medium-arabic-poetry",
      "name": "gpt2 medium arabic poetry",
      "type": "llm",
      "country": "INTL",
      "org": "Ali Elgeish",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/elgeish/gpt2-medium-arabic-poetry"
      },
      "year": 2021,
      "notes": "Fine-tuned aubmindlab/aragpt2-medium on the Arabic Poetry Dataset (6th - 21st century) using 41,922 lines of poetry as the train split and 9,007.",
      "metrics": {
        "downloads": 69,
        "likes": 7,
        "lastModified": "2021-05-21"
      }
    },
    {
      "id": "hatecheck-arabic",
      "name": "hatecheck arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "Paul",
      "license": "['cc-by-4.0']",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Paul/hatecheck-arabic"
      },
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "Multilingual HateCheck (MHC) is a suite of functional tests for hate speech detection models in 10 different languages.",
      "metrics": {
        "downloads": 69,
        "likes": 5,
        "lastModified": "2022-07-05"
      }
    },
    {
      "id": "jawaher-benchmark",
      "name": "Jawaher",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/Jawaher-benchmark"
      },
      "year": 2025,
      "dialects": [
        "mixed"
      ],
      "notes": "Benchmark for LLM understanding and explanation of Arabic proverbs across dialects.",
      "metrics": {
        "downloads": 69,
        "likes": 3,
        "lastModified": "2025-05-27"
      }
    },
    {
      "id": "quran-speech-recognizer",
      "name": "Quran speech recognizer",
      "type": "asr",
      "country": "INTL",
      "org": "Nuwaisir",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "pretraining",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/Nuwaisir/Quran_speech_recognizer"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "This application will listen to the user's Quran recitation, and take the user to the position of the Quran from where the s/he had recited.",
      "metrics": {
        "downloads": 69,
        "likes": 16,
        "lastModified": "2022-08-20"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-ca-sentiment",
      "name": "bert base arabic camelbert ca sentiment",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-ca-sentiment"
      },
      "year": 2021,
      "notes": "For the fine-tuning, we used the ASTD, ArSAS, and SemEval datasets.",
      "metrics": {
        "downloads": 68,
        "likes": 3,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "bert-base-arabic-hate-speech",
      "name": "bert base arabic hate speech",
      "type": "llm",
      "country": "INTL",
      "org": "hossam87",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/hossam87/bert-base-arabic-hate-speech"
      },
      "year": 2025,
      "notes": "A fine-tuned BERT model to classify Arabic text into: Neutral, Offensive, Sexism, Religious Discrimination, or Racism.",
      "base_model": [
        "camel-lab/bert-base-arabic-camelbert-da-sentiment"
      ],
      "metrics": {
        "downloads": 68,
        "likes": 3,
        "lastModified": "2025-07-21"
      }
    },
    {
      "id": "egyptian-legal-v2",
      "name": "egyptian legal v2",
      "type": "dataset",
      "country": "INTL",
      "org": "tarekys5",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarekys5/egyptian_legal_v2"
      },
      "dialects": [
        "egy"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This dataset is a high-quality, structured collection of Egyptian Legal Question & Answer pairs.",
      "metrics": {
        "downloads": 68,
        "likes": 6,
        "lastModified": "2026-03-03"
      }
    },
    {
      "id": "emirati-dialect-shows-audio",
      "name": "Emirati Dialect Shows Audio Transcription",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Emirati speech)",
      "license": "afl-3.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/eabayed/EmiratiDialictShowsAudioTranscription"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Audio and transcripts from Emirati-dialect TV shows for ASR fine-tuning.",
      "metrics": {
        "downloads": 68,
        "likes": 3,
        "lastModified": "2022-05-30"
      }
    },
    {
      "id": "transliteration",
      "name": "Transliteration",
      "type": "dataset",
      "country": "INTL",
      "org": "Google",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/google/transliteration",
        "hf": "https://huggingface.co/datasets/arbml/google_transliteration",
        "paper": "https://arxiv.org/pdf/1610.09565.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "15,898 tokens",
      "year": 2016,
      "tags": [
        "multilingual"
      ],
      "notes": "Arabic-English transliteration dataset mined from Wikipedia.",
      "metrics": {
        "downloads": 68,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arabic-dialects-dataset",
      "name": "Arabic Dialects Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Dialects_Dataset"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "9,992 Arabic dialect texts with dialect labels for dialect identification.",
      "metrics": {
        "downloads": 67,
        "likes": 3,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "asas",
      "name": "ASAS",
      "type": "dataset",
      "country": "INTL",
      "org": "HebArabNlpProject",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HebArabNlpProject/ASAS"
      },
      "dialects": [
        "msa"
      ],
      "size": "388 sentences",
      "year": 2025,
      "notes": "ASAS: Arabic summarization dataset with sentence-level human validation and supporting evidence from the source.",
      "metrics": {
        "downloads": 67,
        "likes": 0,
        "lastModified": "2025-10-14"
      }
    },
    {
      "id": "fiqhqa",
      "name": "FiqhQA",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/FiqhQA"
      },
      "size": "under 1k rows",
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Islamic jurisprudence questions across madhabs testing LLM reliability and abstention.",
      "metrics": {
        "downloads": 67,
        "likes": 3,
        "lastModified": "2025-08-13"
      }
    },
    {
      "id": "bimed-v-1-6m",
      "name": "BiMed-V-1.6M",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "medical",
        "vqa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/BiMed-V-1.6M",
        "paper": "https://arxiv.org/abs/2412.07769"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "BiMed-V: Arabic-English multimodal biomedical dataset of 1.6M samples for visual QA and image-to-text, from BiMediX2.",
      "metrics": {
        "downloads": 66,
        "likes": 2,
        "lastModified": "2025-10-21"
      }
    },
    {
      "id": "namaa-reranker",
      "name": "Namaa-Reranker",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Namaa-Reranker-v1"
      },
      "year": 2024,
      "notes": "NAMAA-space releases Namaa-Reranker-v1, a high-performance model fine-tuned on unicamp-dl/mmarco to elevate Arabic document retrieval and ranking to new.",
      "base_model": [
        "omartificial-intelligence-space/arabic-triplet-matryoshka-v2"
      ],
      "metrics": {
        "downloads": 66,
        "likes": 1,
        "lastModified": "2025-04-03"
      }
    },
    {
      "id": "six-millions-instruction-dataset-for-arabic-llm-ft",
      "name": "six millions instruction dataset for arabic llm ft",
      "type": "dataset",
      "country": "INTL",
      "org": "akbargherbal",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/akbargherbal/six_millions_instruction_dataset_for_arabic_llm_ft"
      },
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "6.4M Arabic instruction, input and output rows for LLM fine-tuning.",
      "metrics": {
        "downloads": 66,
        "likes": 3,
        "lastModified": "2024-05-20"
      }
    },
    {
      "id": "tunisian-tts",
      "name": "Tunisian-TTS",
      "type": "tts",
      "country": "TN",
      "org": "Oussema Nouira",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/Nouira-Oussema/Tunisian-TTS"
      },
      "year": 2026,
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian dialect text-to-speech model.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 66,
        "likes": 5,
        "lastModified": "2026-06-19"
      }
    },
    {
      "id": "ara-prompt-guard-v1",
      "name": "Ara Prompt Guard V1",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Ara-Prompt-Guard_V1"
      },
      "year": 2026,
      "notes": "This model is a fine-tuned version of the meta-llama/Llama-Prompt-Guard-2-86M model.",
      "base_model": [
        "meta-llama/llama-prompt-guard-2-86m"
      ],
      "metrics": {
        "downloads": 65,
        "likes": 7,
        "lastModified": "2026-03-10"
      }
    },
    {
      "id": "facebook-darija-dataset",
      "name": "facebook darija dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "abdeljalilELmajjodi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdeljalilELmajjodi/facebook_darija_dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "This dataset consists of more than 5k public posts from Facebook.",
      "metrics": {
        "downloads": 65,
        "likes": 6,
        "lastModified": "2024-12-11"
      }
    },
    {
      "id": "tunisian-dialect-corpus",
      "name": "Tunisian Dialect Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialects",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Tunisian_Dialect_Corpus"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "49.9K Tunisian dialect tweets with labels.",
      "metrics": {
        "downloads": 65,
        "likes": 3,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "arabic-diacritized-tts",
      "name": "Arabic Diacritized TTS",
      "type": "dataset",
      "country": "INTL",
      "org": "Nourhann",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "diacritization",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Nourhann/Arabic-Diacritized-TTS"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The Arabic-Diacritized-TTS dataset contains Arabic audio samples and their corresponding text with full diacritization.",
      "metrics": {
        "downloads": 64,
        "likes": 9,
        "lastModified": "2025-02-20"
      }
    },
    {
      "id": "araclip",
      "name": "araclip",
      "type": "llm",
      "country": "INTL",
      "org": "Arabic-Clip",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "clip",
        "image-text"
      ],
      "links": {
        "hf": "https://huggingface.co/Arabic-Clip/araclip",
        "github": "https://github.com/Arabic-Clip/Araclip"
      },
      "year": 2025,
      "notes": "AraClip: Arabic CLIP model published with the araclip library for image-text matching.",
      "metrics": {
        "downloads": 64,
        "likes": 6,
        "lastModified": "2025-03-10"
      }
    },
    {
      "id": "gulf-arabic-tweets-2018-2020",
      "name": "Gulf Arabic Tweets 2018-2020",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Gulf)",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AhmedSSabir/Gulf-Arabic-Tweets-2018-2020"
      },
      "year": 2020,
      "dialects": [
        "gulf"
      ],
      "notes": "Collection of Gulf-region Arabic tweets from 2018 to 2020.",
      "metrics": {
        "downloads": 64,
        "likes": 1,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "khattvision-muse-glimmer-30b-lora",
      "name": "KhattVision-Muse-Glimmer-30B-LoRA",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "other",
      "modality": "vision",
      "tasks": [
        "chat",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/KhattVision-Muse-Glimmer-30B-LoRA"
      },
      "size": "30B",
      "on_device": false,
      "year": 2026,
      "notes": "A Muse Glimmer 30B LoRA for Arabic calligraphy understanding",
      "base_model": [
        "unsloth/muse-glimmer-30b-unsloth-bnb-4bit"
      ],
      "metrics": {
        "downloads": 64,
        "likes": 4,
        "lastModified": "2026-08-17"
      }
    },
    {
      "id": "afad",
      "name": "AFAD",
      "type": "dataset",
      "country": "INTL",
      "org": "Purdue University",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "deepfake-detection"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/elsayedissa/AFAD-MSA",
        "paper": "https://doi.org/10.3390/computation14010020"
      },
      "dialects": [
        "msa"
      ],
      "size": "23 hours",
      "year": 2026,
      "notes": "A curated corpus of authentic and synthetic Arabic speech designed to advance research on Arabic deepfake and spoofed-speech detection.",
      "metrics": {
        "downloads": 63,
        "likes": 0,
        "lastModified": "2026-07-10"
      }
    },
    {
      "id": "arabic-dialect-dpo",
      "name": "arabic-dialect-dpo",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "preference",
        "alignment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/arabic-dialect-dpo"
      },
      "dialects": [
        "egy",
        "gulf"
      ],
      "year": 2026,
      "notes": "Arabic dialect preference pairs for DPO, ORPO and GRPO, with Egyptian and Saudi subsets.",
      "metrics": {
        "downloads": 63,
        "likes": 0,
        "lastModified": "2026-02-21"
      }
    },
    {
      "id": "phoenix-arabic-manuscript-htr",
      "name": "phoenix arabic manuscript htr",
      "type": "ocr",
      "country": "INTL",
      "org": "factlogic",
      "license": "cc-by-nc-sa-2.0",
      "modality": "vision",
      "tasks": [
        "asr",
        "ocr",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/factlogic/phoenix-arabic-manuscript-htr"
      },
      "year": 2026,
      "notes": "It uses a CNN + BiLSTM + CTC architecture with approximately 4.99 million parameters and is designed to work across several Arabic handwriting domains.",
      "metrics": {
        "downloads": 63,
        "likes": 4,
        "lastModified": "2026-08-24"
      }
    },
    {
      "id": "vivo-c-v1",
      "name": "vivo c v1",
      "type": "llm",
      "country": "INTL",
      "org": "wasmdashai",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/wasmdashai/vivo-c-v1"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Arabic text generation model fine-tuned from Qwen/Qwen3-235B-A22B-Instruct-2507.",
      "base_model": [
        "qwen/qwen3-235b-a22b-instruct-2507"
      ],
      "metrics": {
        "downloads": 63,
        "likes": 3,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "arabartsummarization",
      "name": "arabartsummarization",
      "type": "llm",
      "country": "INTL",
      "org": "abdalrahmanshahrour",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/abdalrahmanshahrour/arabartsummarization"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "notes": "BERT2BERT Arabic summarization model built on AraBERT, also used for news title generation and paraphrasing.",
      "metrics": {
        "downloads": 62,
        "likes": 6,
        "lastModified": "2023-01-02"
      }
    },
    {
      "id": "arabic-base-nougat",
      "name": "arabic base nougat",
      "type": "ocr",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedRashad/arabic-base-nougat"
      },
      "year": 2024,
      "notes": "End-to-end structured OCR for Arabic books based on Nougat (arXiv 2411.17835).",
      "base_model": [
        "facebook/nougat-base"
      ],
      "metrics": {
        "downloads": 62,
        "likes": 2,
        "lastModified": "2024-11-28"
      }
    },
    {
      "id": "arabic-ecom-search-bench",
      "name": "arabic ecom search bench",
      "type": "benchmark",
      "country": "INTL",
      "org": "Presto AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/prestoai/arabic-ecom-search-bench"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic e-commerce search benchmark for evaluating retrieval systems in Modern Standard Arabic and Libyan dialect.",
      "metrics": {
        "downloads": 62,
        "likes": 5,
        "lastModified": "2026-07-09"
      }
    },
    {
      "id": "arabic-nli-pair-class",
      "name": "Arabic NLi Pair Class",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nli",
        "embedding-training"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-NLi-Pair-Class"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Arabic translation of SNLI and MultiNLI (pair-class subset: premise, hypothesis, label) for NLI and embedding training.",
      "metrics": {
        "downloads": 62,
        "likes": 2,
        "lastModified": "2024-08-03"
      }
    },
    {
      "id": "arabic-quran-nahj-sahife",
      "name": "arabic quran nahj sahife",
      "type": "llm",
      "country": "INTL",
      "org": "pourmand1376",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/pourmand1376/arabic-quran-nahj-sahife"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "A model which is jointly trained and fine-tuned on Quran, Saheefa and nahj-al-balaqa.",
      "metrics": {
        "downloads": 62,
        "likes": 5,
        "lastModified": "2022-06-09"
      }
    },
    {
      "id": "hadith-data",
      "name": "hadith data",
      "type": "dataset",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fawazahmed0/hadith-data"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Hadith collections with book name, hadith number, Arabic text and grading by scholars such as Al-Albani (CC0).",
      "metrics": {
        "downloads": 62,
        "likes": 7,
        "lastModified": "2024-10-30"
      }
    },
    {
      "id": "saudiirony",
      "name": "SaudiIrony",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "irony",
        "twitter"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/SaudiIrony"
      },
      "dialects": [
        "gulf"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "19.8K Saudi tweets labelled for irony.",
      "metrics": {
        "downloads": 62,
        "likes": 2,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "alexandria-backtranslated-pairs",
      "name": "AlexandriaX Back-Translated Pairs",
      "type": "dataset",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/alexandria-backtranslated-pairs"
      },
      "size": "349k",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "348,787 synthetic English to dialectal Arabic pairs over 14 varieties.",
      "metrics": {
        "downloads": 61,
        "likes": 0,
        "lastModified": "2026-08-21"
      }
    },
    {
      "id": "arabic-roots",
      "name": "arabic roots",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/arabic-roots"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The \"arabic-roots\" dataset is a comprehensive collection of Arabic root words along with their detailed definitions, sourced from classical Arabic lexicons.",
      "metrics": {
        "downloads": 61,
        "likes": 2,
        "lastModified": "2024-07-09"
      }
    },
    {
      "id": "arabjobs",
      "name": "ArabJobs",
      "type": "dataset",
      "country": "INTL",
      "org": "Mo El-Haj (ArabicNLP.uk)",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification",
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/ArabJobs"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "Arabic job-advertisement dataset supporting classification, summarisation, QA and token-classification tasks.",
      "metrics": {
        "downloads": 61,
        "likes": 0,
        "lastModified": "2025-11-28"
      }
    },
    {
      "id": "artst-asr-v3-qasr",
      "name": "artst asr v3 qasr",
      "type": "asr",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/artst_asr_v3_qasr"
      },
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "ArTST model finetuned for automatic speech recognition (speech-to-text) on QASR (best for Dialectal Arabic Variants) (also: 1 variants)",
      "metrics": {
        "downloads": 61,
        "likes": 4,
        "lastModified": "2025-09-10"
      }
    },
    {
      "id": "artst-asr-v3",
      "name": "artst_asr_v3",
      "type": "asr",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/artst_asr_v3"
      },
      "dialects": [
        "msa"
      ],
      "notes": "ArTST for ASR on MGB2 (best for MSA)",
      "metrics": {
        "downloads": 61,
        "likes": 0,
        "lastModified": "2025-09-10"
      }
    },
    {
      "id": "mfa-quran-hafs",
      "name": "mfa quran hafs",
      "type": "asr",
      "country": "INTL",
      "org": "Quran Lab",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "forced-alignment",
        "tajweed"
      ],
      "links": {
        "hf": "https://huggingface.co/Quran-Lab/mfa-quran-hafs"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Montreal Forced Aligner acoustic model for Hafs Quran recitation, built for phone-level tajweed measurement.",
      "metrics": {
        "downloads": 61,
        "likes": 4,
        "lastModified": "2026-08-06"
      }
    },
    {
      "id": "saudilang-code-switch-corpus",
      "name": "Saudilang Code Switch Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "SDAIA NCAI",
      "license": "cc-by-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SDAIANCAI/Saudilang-Code-Switch-Corpus"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "SCC: SDAIA NCAI corpus of transcribed code-switched Arabic-English conversations from the Thmanyah YouTube podcast.",
      "metrics": {
        "downloads": 61,
        "likes": 3,
        "lastModified": "2024-07-28"
      }
    },
    {
      "id": "afrd-arabic-fake-reviews-detection",
      "name": "AFRD: Arabic Fake Reviews Detection",
      "type": "dataset",
      "country": "SA",
      "org": "Qassim University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dialect-id",
        "classification",
        "gender-id"
      ],
      "links": {
        "github": "https://github.com/NoorAmer0/AFRD-arabic-fake-reviews-dataset",
        "hf": "https://huggingface.co/datasets/Noor0/AFRD_Arabic-Fake-Reviews-Detection",
        "paper": "https://www.sciencedirect.com/science/article/pii/S1319157824000156"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1,958 sentences",
      "year": 2024,
      "notes": "Arabic Fake Reviews Detection (AFRD) is the first gold-standard dataset comprised of three domains, namely, hotel, restaurant, and product domains.",
      "metrics": {
        "downloads": 60,
        "likes": 0,
        "lastModified": "2024-02-09"
      }
    },
    {
      "id": "apcd",
      "name": "APCD",
      "type": "dataset",
      "country": "EG",
      "org": "Nile university",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "website": "https://hci-lab.github.io/LearningMetersPoems/",
        "hf": "https://huggingface.co/datasets/arbml/APCD",
        "paper": "https://arxiv.org/pdf/1905.05700.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,831,770 sentences",
      "year": 2019,
      "notes": "APCD is a dataset of 1.8M Arabic poetry lines annotated with their poetic meters for meter classification research.",
      "metrics": {
        "downloads": 60,
        "likes": 2,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "quranexe",
      "name": "QuranExe",
      "type": "dataset",
      "country": "INTL",
      "org": "mustapha",
      "license": "['mit']",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mustapha/QuranExe"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2022,
      "notes": "This dataset contains the exegeses/tafsirs (تفسير القرآن) of the holy Quran in arabic by 8 exegetes.",
      "metrics": {
        "downloads": 60,
        "likes": 11,
        "lastModified": "2022-07-20"
      }
    },
    {
      "id": "rasaif-translations",
      "name": "rasaif translations",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedRashad/rasaif-translations"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "1,951 Arabic-English sentence pairs sourced from rasaif.com.",
      "metrics": {
        "downloads": 60,
        "likes": 6,
        "lastModified": "2024-03-19"
      }
    },
    {
      "id": "alpaca-arabic-instruct",
      "name": "Alpaca Arabic Instruct",
      "type": "dataset",
      "country": "INTL",
      "org": "Yasbok",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Yasbok/Alpaca_arabic_instruct"
      },
      "notes": "Arabic Alpaca instruction dataset",
      "metrics": {
        "downloads": 59,
        "likes": 27,
        "lastModified": "2024-04-21"
      }
    },
    {
      "id": "math-cot-arabic-english-reasoning",
      "name": "Math_CoT_Arabic_English_Reasoning",
      "type": "benchmark",
      "country": "INTL",
      "org": "miscovery",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/miscovery/Math_CoT_Arabic_English_Reasoning"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "A high-quality, bilingual (English & Arabic) dataset for Chain-of-Thought (COT) reasoning in mathematics and related disciplines, developed by Miscovery AI.",
      "metrics": {
        "downloads": 59,
        "likes": 17,
        "lastModified": "2025-05-12"
      }
    },
    {
      "id": "translate-ar-en-v1-0-hplt",
      "name": "translate ar en v1.0 hplt",
      "type": "llm",
      "country": "INTL",
      "org": "HPLT",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/HPLT/translate-ar-en-v1.0-hplt"
      },
      "year": 2024,
      "notes": "This repository contains the translation model for Arabic-English trained with HPLT data only.",
      "metrics": {
        "downloads": 59,
        "likes": 4,
        "lastModified": "2024-03-14"
      }
    },
    {
      "id": "darija",
      "name": "darija",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "7,325 Darija-English sentence pairs.",
      "metrics": {
        "downloads": 58,
        "likes": 4,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "habibi-lyrics-corpus",
      "name": "Habibi Lyrics Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "lyrics",
        "dialect"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/Habibi"
      },
      "year": 2020,
      "notes": "Multi-dialect corpus of 30,000+ Arabic song lyrics across 18 countries.",
      "dialects": [
        "mixed"
      ],
      "metrics": {
        "downloads": 58,
        "likes": 1,
        "lastModified": "2025-11-28"
      }
    },
    {
      "id": "hassaniya-mms-asr",
      "name": "hassaniya-mms-asr",
      "type": "asr",
      "country": "INTL",
      "org": "Hassen80",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Hassen80/hassaniya-mms-asr"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "MMS (wav2vec2) fine-tunes for Hassaniya Arabic speech recognition.",
      "metrics": {
        "downloads": 58,
        "likes": 0,
        "lastModified": "2026-08-02"
      }
    },
    {
      "id": "metro-asr-small",
      "name": "Metro ASR Small",
      "type": "asr",
      "country": "INTL",
      "org": "Mohammed Aly",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammedaly22/Metro-ASR-Small"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Non-autoregressive CTC Conformer ASR for Egyptian Arabic and Arabic-English code-switching, with a detachable n-gram language head.",
      "metrics": {
        "downloads": 58,
        "likes": 2,
        "lastModified": "2026-08-18"
      }
    },
    {
      "id": "oasst1-ar-threads",
      "name": "oasst1-ar-threads",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/oasst1-ar-threads"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "notes": "Arabic conversation threads from OpenAssistant oasst1 with train and validation splits; used to tune llama-2-7b-chat-ar.",
      "metrics": {
        "downloads": 58,
        "likes": 3,
        "lastModified": "2023-12-28"
      }
    },
    {
      "id": "zipformer-p-quran",
      "name": "zipformer p quran",
      "type": "asr",
      "country": "INTL",
      "org": "Muno459",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/Muno459/zipformer_p-quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Zipformer ASR model for Quran recitation; gated, free non-commercial use under NPL-1.1.",
      "base_model": [
        "muno459/zipformer_p-arabic"
      ],
      "metrics": {
        "downloads": 58,
        "likes": 14,
        "lastModified": "2026-08-15"
      }
    },
    {
      "id": "arabic-guanaco-oasst1",
      "name": "Arabic guanaco oasst1",
      "type": "dataset",
      "country": "INTL",
      "org": "alielfilali01",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alielfilali01/Arabic_guanaco_oasst1"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "This dataset is the openassistant-guanaco dataset a subset of the Open Assistant dataset translated to Arabic.",
      "metrics": {
        "downloads": 57,
        "likes": 8,
        "lastModified": "2023-06-12"
      }
    },
    {
      "id": "arabic-nli-pair",
      "name": "Arabic NLi Pair",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-NLi-Pair"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "The Arabic Version of SNLI and MultiNLI datasets.",
      "metrics": {
        "downloads": 57,
        "likes": 4,
        "lastModified": "2024-08-02"
      }
    },
    {
      "id": "arabic-orpo-llama3-8b",
      "name": "Arabic-Orpo-Llama-3-8B-Instruct",
      "type": "llm",
      "country": "INTL",
      "org": "MohamedRashad",
      "license": "llama3",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedRashad/Arabic-Orpo-Llama-3-8B-Instruct"
      },
      "base_model": [
        "meta-llama/meta-llama-3-8b-instruct"
      ],
      "size": "8.0B",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Llama-3-8B aligned for Arabic with ORPO.",
      "metrics": {
        "downloads": 57,
        "likes": 17,
        "lastModified": "2024-05-03"
      }
    },
    {
      "id": "arasl-database-grayscale",
      "name": "ArASL_Database_Grayscale",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohammad Albarham",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/pain/ArASL_Database_Grayscale"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "A new dataset consists of 54,049 images of ArSL alphabets performed by more than 40 people for 32 standard Arabic signs and alphabets.",
      "metrics": {
        "downloads": 57,
        "likes": 2,
        "lastModified": "2023-01-07"
      }
    },
    {
      "id": "darijabridge",
      "name": "DarijaBridge",
      "type": "dataset",
      "country": "MA",
      "org": "MAD Community",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/M-A-D/DarijaBridge"
      },
      "size": "1M-10M pairs",
      "year": 2023,
      "dialects": [
        "magh"
      ],
      "notes": "Darija-English parallel dataset for translation, 1M-10M rows.",
      "metrics": {
        "downloads": 57,
        "likes": 8,
        "lastModified": "2023-11-26"
      }
    },
    {
      "id": "fr-wolof-quran-corpus",
      "name": "fr wolof quran corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Lahad",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Lahad/fr_wolof_quran_corpus"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "It has been generated using this raw template.",
      "metrics": {
        "downloads": 57,
        "likes": 3,
        "lastModified": "2025-02-09"
      }
    },
    {
      "id": "jisr-align-29m",
      "name": "Jisr Align 29M",
      "type": "embedding",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Jisr-Align-29M"
      },
      "size": "29M",
      "on_device": true,
      "year": 2026,
      "notes": "The encoder half of the Jisr Arabic-English MT model, contrastively finetuned for sentence alignment. 29M parameters, 448-dim, 6 layers.",
      "metrics": {
        "downloads": 57,
        "likes": 2,
        "lastModified": "2026-08-16"
      }
    },
    {
      "id": "kallamni-4b-v1",
      "name": "kallamni 4b v1",
      "type": "llm",
      "country": "INTL",
      "org": "yasserrmd",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/yasserrmd/kallamni-4b-v1"
      },
      "dialects": [
        "gulf"
      ],
      "size": "4B",
      "on_device": false,
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "Chat model fine-tuned on natural spoken Emirati Arabic dialect.",
      "base_model": [
        "qwen/qwen3-4b"
      ],
      "metrics": {
        "downloads": 57,
        "likes": 3,
        "lastModified": "2025-10-05"
      }
    },
    {
      "id": "quora-arabic-gpt4",
      "name": "Quora Arabic GPT4",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Quora-Arabic-GPT4"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "The dataset is created by (1) collecting Arabic questions in Quora and (2) requesting GPT4 to generate responses.",
      "metrics": {
        "downloads": 57,
        "likes": 3,
        "lastModified": "2023-09-06"
      }
    },
    {
      "id": "ags-corpus",
      "name": "AGS Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "FahdSeddik",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FahdSeddik/AGS-Corpus"
      },
      "size": "100K–1M rows",
      "year": 2023,
      "notes": "AGS: Arabic GPT Summarization Corpus AGS is the first publicly accessible abstractive summarization dataset for Arabic.",
      "metrics": {
        "downloads": 56,
        "likes": 7,
        "lastModified": "2023-09-29"
      }
    },
    {
      "id": "arabic-stsb",
      "name": "Arabic stsb",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "sts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-stsb"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic version of the Semantic Textual Similarity Benchmark with human-annotated similarity scores from 1 to 5.",
      "metrics": {
        "downloads": 56,
        "likes": 4,
        "lastModified": "2024-08-02"
      }
    },
    {
      "id": "arabic-t5-small-question-paraphrasing",
      "name": "arabic t5 small question paraphrasing",
      "type": "llm",
      "country": "INTL",
      "org": "salti",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "paraphrasing"
      ],
      "links": {
        "hf": "https://huggingface.co/salti/arabic-t5-small-question-paraphrasing"
      },
      "year": 2023,
      "notes": "arabic-t5-small fine-tuned for Arabic question paraphrasing.",
      "metrics": {
        "downloads": 56,
        "likes": 3,
        "lastModified": "2023-11-21"
      }
    },
    {
      "id": "asr-wav2vec2-dvoice-darija",
      "name": "asr-wav2vec2-dvoice-darija",
      "type": "asr",
      "country": "INTL",
      "org": "SpeechBrain",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/speechbrain/asr-wav2vec2-dvoice-darija"
      },
      "year": 2022,
      "dialects": [
        "magh"
      ],
      "notes": "SpeechBrain wav2vec2 recipe trained on the DVoice Darija dataset.",
      "metrics": {
        "downloads": 56,
        "likes": 15,
        "lastModified": "2024-02-19"
      }
    },
    {
      "id": "mizanqa",
      "name": "MizanQA",
      "type": "benchmark",
      "country": "MA",
      "org": "UM6P",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/adlbh/MizanQA-v0",
        "paper": "https://aclanthology.org/2026.eacl-industry.10.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "1,769 sentences",
      "year": 2025,
      "notes": "MizanQA is the first benchmark designed to evaluate large language models (LLMs) on Moroccan legal question answering tasks.",
      "metrics": {
        "downloads": 56,
        "likes": 1,
        "lastModified": "2025-08-22"
      }
    },
    {
      "id": "arabic-triplets-1m-curated-sims-len",
      "name": "arabic triplets 1m curated sims len",
      "type": "dataset",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/akhooli/arabic-triplets-1m-curated-sims-len"
      },
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "This is a curated dataset to use in Arabic ColBERT and SBERT models (among other uses).",
      "metrics": {
        "downloads": 55,
        "likes": 10,
        "lastModified": "2024-07-27"
      }
    },
    {
      "id": "arabic-llama3",
      "name": "Arabic-llama3",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/Arabic-llama3"
      },
      "year": 2024,
      "notes": "Fine-tuned from Meta Llama 3",
      "base_model": [
        "unsloth/llama-3-8b-bnb-4bit"
      ],
      "metrics": {
        "downloads": 55,
        "likes": 15,
        "lastModified": "2024-04-21"
      }
    },
    {
      "id": "ariq",
      "name": "ARIQ",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "quotation-extraction"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/ARIQ"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "ARIQ: first manually annotated dataset for extracting and attributing Quran, Hadith and scholarly quotations in Arabic fatwas.",
      "metrics": {
        "downloads": 55,
        "likes": 2,
        "lastModified": "2026-09-30"
      }
    },
    {
      "id": "cidar-eval-100",
      "name": "CIDAR EVAL 100",
      "type": "benchmark",
      "country": "SA",
      "org": "ARBML",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/CIDAR-EVAL-100"
      },
      "dialects": [
        "lev"
      ],
      "size": "<1K rows",
      "year": 2024,
      "notes": "CIDAR-EVAL-100 contains 100 instructions about Arabic culture.",
      "metrics": {
        "downloads": 55,
        "likes": 2,
        "lastModified": "2024-02-14"
      }
    },
    {
      "id": "egtts-v0-1",
      "name": "EGTTS V0.1",
      "type": "tts",
      "country": "INTL",
      "org": "OmarSamir",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/OmarSamir/EGTTS-V0.1"
      },
      "dialects": [
        "egy"
      ],
      "year": 2025,
      "notes": "EGTTS V0.1 is a cutting-edge text-to-speech (TTS) model specifically designed for Egyptian Arabic.",
      "base_model": [
        "coqui/xtts-v2"
      ],
      "metrics": {
        "downloads": 55,
        "likes": 42,
        "lastModified": "2025-03-13"
      }
    },
    {
      "id": "english-moroccan-darija-v1",
      "name": "English-Moroccan-Darija-v1",
      "type": "llm",
      "country": "INTL",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/English-Moroccan-Darija-v1"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Translation model from English to Moroccan Darija.",
      "base_model": [
        "liquidai/lfm2-350m"
      ],
      "metrics": {
        "downloads": 55,
        "likes": 1,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "fine-tuning-gemma-2b-it-for-arabic",
      "name": "Fine Tuning Gemma 2b it for Arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Ruqiya",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/Ruqiya/Fine-Tuning-Gemma-2b-it-for-Arabic"
      },
      "size": "2B",
      "on_device": true,
      "year": 2024,
      "notes": "This model is a fine-tuned version of google/gemma-2b-it on arbml/CIDAR Arabic dataset.",
      "base_model": [
        "google/gemma-2b-it"
      ],
      "metrics": {
        "downloads": 55,
        "likes": 4,
        "lastModified": "2024-03-28"
      }
    },
    {
      "id": "moroccan-lawqa-dataset",
      "name": "moroccan-lawQA-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ilyassacha",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "legal",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ilyassacha/moroccan-lawQA-dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "4,999 Moroccan legal question-answering instruction rows.",
      "metrics": {
        "downloads": 55,
        "likes": 3,
        "lastModified": "2025-06-16"
      }
    },
    {
      "id": "tarjama-25",
      "name": "Tarjama-25",
      "type": "benchmark",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/Tarjama-25"
      },
      "year": 2025,
      "dialects": [
        "msa"
      ],
      "notes": "Bidirectional Arabic-English machine translation benchmark built to stress-test MT models.",
      "metrics": {
        "downloads": 55,
        "likes": 5,
        "lastModified": "2025-05-28"
      }
    },
    {
      "id": "aisa-ar-functioncall",
      "name": "AISA-AR-FunctionCall",
      "type": "dataset",
      "country": "INTL",
      "org": "AISA-Framework",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AISA-Framework/AISA-AR-FunctionCall"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "AISA-AR-FunctionCall is a large-scale Arabic dataset designed for training language models to convert natural language into structured executable tool calls.",
      "metrics": {
        "downloads": 54,
        "likes": 3,
        "lastModified": "2026-03-15"
      }
    },
    {
      "id": "aya-sambalingo-arabic-chat-dpo",
      "name": "Aya-SambaLingo.Arabic.Chat-DPO",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Aya-SambaLingo.Arabic.Chat-DPO"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Dataset Sources & Infos",
      "metrics": {
        "downloads": 54,
        "likes": 3,
        "lastModified": "2024-05-25"
      }
    },
    {
      "id": "darija-stories-dataset",
      "name": "Darija Stories Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "alielfilali01",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alielfilali01/Darija-Stories-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "Darija (Moroccan Arabic) Stories Dataset contains a diverse range of stories that provide insights into Moroccan culture, traditions, and everyday life.",
      "metrics": {
        "downloads": 54,
        "likes": 10,
        "lastModified": "2023-07-29"
      }
    },
    {
      "id": "darija-wiki-audio",
      "name": "Moroccan Darija Wiki Audio Dataset",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Moroccan-Darija-Wiki-Audio-Dataset"
      },
      "size": "1K-10K rows",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Recorded reading of Moroccan Darija Wikipedia sentences.",
      "metrics": {
        "downloads": 54,
        "likes": 17,
        "lastModified": "2025-02-14"
      }
    },
    {
      "id": "shami-corpus",
      "name": "Shami",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Shami"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Levantine dialect corpus (Jordanian, Lebanese, Palestinian, Syrian) from social media.",
      "metrics": {
        "downloads": 54,
        "likes": 0,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "alfloos",
      "name": "AlFloos",
      "type": "dataset",
      "country": "SA",
      "org": "Umm Al-Qura University",
      "license": "gpl-3.0",
      "modality": "vision",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/gfbati/alfloos",
        "hf": "https://huggingface.co/datasets/gfbati/alfloos",
        "paper": "https://link.springer.com/content/pdf/10.1007/s43995-024-00067-z.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1,330 images",
      "year": 2024,
      "notes": "Balanced image dataset of all seven banknote denominations of the sixth issue of Saudi Arabian currency.",
      "metrics": {
        "downloads": 53,
        "likes": 1,
        "lastModified": "2025-09-06"
      }
    },
    {
      "id": "alsanaa-emirati-asr",
      "name": "Alsanaa Emirati Arabic ASR",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Emirati speech)",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/alsanaa-emirati-arabic-asr"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Emirati Arabic speech-transcription dataset for fine-tuning ASR.",
      "metrics": {
        "downloads": 53,
        "likes": 0,
        "lastModified": "2026-06-13"
      }
    },
    {
      "id": "arabic-reranker",
      "name": "arabic reranker",
      "type": "embedding",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/arabic-reranker"
      },
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "This is an Arabic reranker model, fine-tuned from the Omartificial-Intelligence-Space/Arabic-Triplet-Matryoshka-V2, which itself is based on… (also: 1 variants)",
      "base_model": [
        "omartificial-intelligence-space/arabic-triplet-matryoshka-v2"
      ],
      "metrics": {
        "downloads": 53,
        "likes": 10,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "arquad",
      "name": "ArQuAD",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/RashaMObeidat/ArQuAD",
        "hf": "https://huggingface.co/datasets/arbml/ArQuAD",
        "paper": "https://link.springer.com/article/10.1007/s12559-024-10248-6"
      },
      "dialects": [
        "msa"
      ],
      "size": "16,020 sentences",
      "year": 2024,
      "notes": "A large MRC dataset for the Arabic language.",
      "metrics": {
        "downloads": 53,
        "likes": 0,
        "lastModified": "2024-05-04"
      }
    },
    {
      "id": "etec-benchmark",
      "name": "ETEC",
      "type": "benchmark",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/humain-ai/Etec"
      },
      "year": 2025,
      "notes": "1,887 Arabic multiple-choice questions from Saudi Education and Training Evaluation Commission exams.",
      "metrics": {
        "downloads": 53,
        "likes": 2,
        "lastModified": "2025-09-17"
      }
    },
    {
      "id": "minireranker-arabic-v1",
      "name": "miniReranker_arabic_v1",
      "type": "llm",
      "country": "INTL",
      "org": "prithivida",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/prithivida/miniReranker_arabic_v1"
      },
      "year": 2025,
      "notes": "This model is a Arabic equivalent of ms-marco-MiniLM-L-12-v2 reranker.",
      "metrics": {
        "downloads": 53,
        "likes": 4,
        "lastModified": "2025-05-01"
      }
    },
    {
      "id": "arabic-reasoning-dataset",
      "name": "Arabic Reasoning Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic_Reasoning_Dataset"
      },
      "notes": "9.2K instruction-based reasoning QA pairs",
      "metrics": {
        "downloads": 52,
        "likes": 10,
        "lastModified": "2024-12-01"
      }
    },
    {
      "id": "arabic-wikipedia-20230101-bots",
      "name": "Arabic_Wikipedia_20230101_bots",
      "type": "dataset",
      "country": "INTL",
      "org": "Clarkson University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/Arabic_Wikipedia_20230101_bots",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.19.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,000,000 documents",
      "year": 2023,
      "notes": "ArabicWikipedia20230101bots is a dataset created using the Arabic Wikipedia articles, including the bot-generated articles, downloaded on the 1st of January.",
      "metrics": {
        "downloads": 52,
        "likes": 1,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "fineweb2-north-levantine",
      "name": "FineWeb2 North Levantine Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "odc-by",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/FineWeb2-North-Levantine-Arabic"
      },
      "dialects": [
        "lev"
      ],
      "notes": "North Levantine Arabic subset extracted from FineWeb2.",
      "metrics": {
        "downloads": 52,
        "likes": 2,
        "lastModified": "2024-12-12"
      }
    },
    {
      "id": "masrawy-bilingual-v1",
      "name": "Masrawy-BiLingual-v1",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Masrawy-BiLingual-v1"
      },
      "year": 2026,
      "notes": "The model was evaluated using held-out test sets consisting of diverse, high-q",
      "base_model": [
        "ubc-nlp/arat5v2-base-1024"
      ],
      "metrics": {
        "downloads": 52,
        "likes": 9,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "tarbiyah-ai-whisper-medium-merged",
      "name": "tarbiyah ai whisper medium merged",
      "type": "asr",
      "country": "INTL",
      "org": "Habib-HF",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/Habib-HF/tarbiyah-ai-whisper-medium-merged"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Modelrepo: Habib-HF/tarbiyah-ai-whisper-medium-merged This repository contains an Automatic Speech Recognition (ASR) model based on openai/whisper-medium.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 52,
        "likes": 6,
        "lastModified": "2025-11-26"
      }
    },
    {
      "id": "tawnza-tifinagh-arabic-dataset",
      "name": "TAWNZA TIFINAGH ARABIC DATASET",
      "type": "dataset",
      "country": "INTL",
      "org": "Tamazight-NLP",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Tamazight-NLP/TAWNZA_TIFINAGH_ARABIC_DATASET"
      },
      "dialects": [
        "mixed"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "This dataset is 70% of the total data I have of this show.",
      "metrics": {
        "downloads": 52,
        "likes": 8,
        "lastModified": "2026-04-20"
      }
    },
    {
      "id": "vits-ar",
      "name": "vits ar",
      "type": "tts",
      "country": "INTL",
      "org": "wasmdashai",
      "license": "afl-3.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/wasmdashai/vits-ar"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2024,
      "notes": "An advanced text-to-speech (TTS) system specifically designed for the Arabic language.",
      "metrics": {
        "downloads": 52,
        "likes": 18,
        "lastModified": "2024-09-05"
      }
    },
    {
      "id": "whisper-yemeni",
      "name": "Whisper Yemeni",
      "type": "asr",
      "country": "YE",
      "org": "Mansoor Saleh",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mansoorSaleh/whisper-yemeni"
      },
      "size": "242M",
      "dialects": [
        "yemeni"
      ],
      "notes": "Whisper fine-tuned for Yemeni dialect (Yemen).",
      "base_model": [
        "openai/whisper-large-v2"
      ],
      "metrics": {
        "downloads": 52,
        "likes": 0,
        "lastModified": "2026-08-19"
      }
    },
    {
      "id": "yemeni-speech-emotion",
      "name": "Yemeni Speech Emotion Dataset",
      "type": "dataset",
      "country": "YE",
      "org": "Fatimah Emad Eldin",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "emotion"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FatimahEmadEldin/Yemeni-Speech-Emotion-Dataset"
      },
      "dialects": [
        "yemeni"
      ],
      "notes": "Yemeni Arabic emotional speech dataset (Yemen).",
      "metrics": {
        "downloads": 52,
        "likes": 1,
        "lastModified": "2026-05-01"
      }
    },
    {
      "id": "ai-for-rim",
      "name": "AI for RIM",
      "type": "dataset",
      "country": "INTL",
      "org": "Emin009",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Emin009/AI-for-RIM"
      },
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "This repository contains open-source datasets designed for training and fine-tuning Large Language Models (LLMs) on the Hassaniya dialect of Mauritania.",
      "metrics": {
        "downloads": 51,
        "likes": 4,
        "lastModified": "2026-01-12"
      }
    },
    {
      "id": "arabart-qalb15-gec-ged-13",
      "name": "arabart qalb15 gec ged 13",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "gec"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/arabart-qalb15-gec-ged-13"
      },
      "year": 2024,
      "notes": "MSA grammatical error correction model, AraBART fine-tuned with morphology features on QALB-2015.",
      "dialects": [
        "msa"
      ],
      "metrics": {
        "downloads": 51,
        "likes": 2,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "financial-news",
      "name": "financial news",
      "type": "dataset",
      "country": "INTL",
      "org": "Asas AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "news",
        "finance"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/asas-ai/financial_news"
      },
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "7,560 source-target text pairs of Arabic financial news.",
      "metrics": {
        "downloads": 51,
        "likes": 2,
        "lastModified": "2023-11-01"
      }
    },
    {
      "id": "quran-tafseers",
      "name": "Quran Tafseers",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/Quran-Tafseers"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Quran tafsir texts for classical Arabic NLP and language modeling, from Prince Sultan University's RIOTU Lab.",
      "metrics": {
        "downloads": 51,
        "likes": 6,
        "lastModified": "2024-01-26"
      }
    },
    {
      "id": "roberta-large-eng-ara-128k",
      "name": "roberta large eng ara 128k",
      "type": "llm",
      "country": "INTL",
      "org": "jhu-clsp",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/jhu-clsp/roberta-large-eng-ara-128k"
      },
      "year": 2023,
      "notes": "roberta-large-eng-ara-128k is an English�Arabic bilingual encoders of 24-layer Transformers (d\\model= 1024), the same size as XLM-R large.",
      "metrics": {
        "downloads": 51,
        "likes": 5,
        "lastModified": "2023-06-24"
      }
    },
    {
      "id": "semeval-2016-absa-reviews-arabic",
      "name": "semeval 2016 absa reviews arabic",
      "type": "benchmark",
      "country": "INTL",
      "org": "srinivasbilla",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/srinivasbilla/semeval-2016-absa-reviews-arabic"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "Aspect based sentiment analysis dataset using hotel reviews in Arabic.",
      "metrics": {
        "downloads": 51,
        "likes": 4,
        "lastModified": "2023-06-07"
      }
    },
    {
      "id": "voxlect-arabic-dialect-whisper-large-v3",
      "name": "voxlect-arabic-dialect-whisper-large-v3",
      "type": "tool",
      "country": "INTL",
      "org": "tiantiaf (USC)",
      "license": "openrail",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/tiantiaf/voxlect-arabic-dialect-whisper-large-v3"
      },
      "dialects": [
        "egy",
        "gulf",
        "lev",
        "magh",
        "iraqi",
        "yemeni",
        "sudanese"
      ],
      "year": 2025,
      "notes": "Whisper-large-v3 Arabic dialect classifier from the Voxlect benchmark.",
      "metrics": {
        "downloads": 51,
        "likes": 2,
        "lastModified": "2025-08-10"
      }
    },
    {
      "id": "arabic-osact4-offensive-language-detection",
      "name": "Arabic OSACT4 : Offensive Language Detection",
      "type": "dataset",
      "country": "INTL",
      "org": "The University of Edinburgh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "github": "https://github.com/motazsaad/arabic-hatespeech-data/blob/master/OSACT4/README.md",
        "hf": "https://huggingface.co/datasets/arbml/OSACT4_hatespeech",
        "paper": "https://aclanthology.org/2020.osact-1.7.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "8,000 sentences",
      "year": 2020,
      "notes": "OSACT4 Shared Task on Offensive Language Detection",
      "metrics": {
        "downloads": 50,
        "likes": 1,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "author-attribution-tweets",
      "name": "Author Attribution Tweets",
      "type": "dataset",
      "country": "PS",
      "org": "Birzeit University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "authorship-attribution"
      ],
      "links": {
        "website": "https://fada.birzeit.edu/handle/20.500.11889/6743",
        "hf": "https://huggingface.co/datasets/arbml/Author_Attribution_Tweets",
        "paper": "https://fada.birzeit.edu/bitstream/20.500.11889/6787/1/AA_PAPER___ACM.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "71,397 sentences",
      "year": 2021,
      "notes": "Consists of 71,397 tweets for 45 authors for MSA collected from twitter.",
      "metrics": {
        "downloads": 50,
        "likes": 0,
        "lastModified": "2022-10-21"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-ca-pos-egy",
      "name": "bert base arabic camelbert ca pos egy",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-ca-pos-egy"
      },
      "dialects": [
        "egy"
      ],
      "year": 2021,
      "notes": "For the fine-tuning, we used the ARZTB dataset .",
      "metrics": {
        "downloads": 50,
        "likes": 3,
        "lastModified": "2021-10-18"
      }
    },
    {
      "id": "camelbert-msa-zaebuc-ged-13",
      "name": "camelbert msa zaebuc ged 13",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/camelbert-msa-zaebuc-ged-13"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "notes": "For the fine-tuning, we used a combination of the QALB-2014, QALB-2015, and ZAEBUC datasets.",
      "metrics": {
        "downloads": 50,
        "likes": 3,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "cidar-mcq-100",
      "name": "CIDAR MCQ 100",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/CIDAR-MCQ-100"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "CIDAR-MCQ-100 contains 100 multiple-choice questions and answers about the Arabic culture.",
      "metrics": {
        "downloads": 50,
        "likes": 4,
        "lastModified": "2024-04-02"
      }
    },
    {
      "id": "maknuune",
      "name": "Maknuune",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "lexicon"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Maknuune"
      },
      "year": 2022,
      "dialects": [
        "lev"
      ],
      "notes": "Large open Palestinian Arabic lexicon (PS) from NYU Abu Dhabi CAMeL Lab.",
      "metrics": {
        "downloads": 50,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "speechbrain-wav2vec2-arabic",
      "name": "SpeechBrain wav2vec2 Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "SpeechBrain",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/speechbrain/asr-wav2vec2-commonvoice-14-ar"
      },
      "notes": "Wav2Vec2 + CTC fine-tuned on CommonVoice 14 Arabic",
      "metrics": {
        "downloads": 50,
        "likes": 0,
        "lastModified": "2024-02-26"
      }
    },
    {
      "id": "whisper-large-libyan",
      "name": "whisper-large-libyan",
      "type": "asr",
      "country": "INTL",
      "org": "Rhaodgh",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Rhaodgh/whisper-large-libyan"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Whisper large-v3 fine-tuned for Libyan Arabic; a medium-size variant is also released.",
      "base_model": [
        "openai/whisper-large-v3"
      ],
      "metrics": {
        "downloads": 50,
        "likes": 0,
        "lastModified": "2026-07-16"
      }
    },
    {
      "id": "afnd",
      "name": "AFND",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/AFND"
      },
      "year": 2022,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic fake news dataset from Arab news sources for misinformation detection.",
      "metrics": {
        "downloads": 49,
        "likes": 0,
        "lastModified": "2022-10-31"
      }
    },
    {
      "id": "code-alpaca-arabic-gpt4",
      "name": "Code Alpaca Arabic GPT4",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Code-Alpaca-Arabic-GPT4"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "The dataset is created by (1) translating code-alpaca English questions into Arabic using GPT4 and (2) requesting GPT4 to generate Arabic responses.",
      "metrics": {
        "downloads": 49,
        "likes": 3,
        "lastModified": "2023-09-06"
      }
    },
    {
      "id": "fignews-2024",
      "name": "FIGNEWS 2024",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "annotation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CAMeL-Lab/FIGNEWS-2024"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Shared task data on bias and propaganda annotation of news on the Gaza war, ArabicNLP 2024.",
      "metrics": {
        "downloads": 49,
        "likes": 2,
        "lastModified": "2024-07-27"
      }
    },
    {
      "id": "nilechat-arabizi-egypt",
      "name": "NileChat Arabizi-Egypt",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/nilechat-arabizi-egy"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabizi dataset for LLM pre-training",
      "metrics": {
        "downloads": 49,
        "likes": 4,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "quranqa",
      "name": "quranqa",
      "type": "dataset",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "['cc-by-nd-4.0']",
      "modality": "text",
      "tasks": [
        "qa",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarteel-ai/quranqa"
      },
      "dialects": [
        "classical"
      ],
      "size": "<1K rows",
      "year": 2024,
      "notes": "Qur'anic Reading Comprehension Dataset (QRCD) from the Quran QA 2022 shared task.",
      "metrics": {
        "downloads": 49,
        "likes": 17,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "shami-mt",
      "name": "Shami-MT",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Shami-MT"
      },
      "size": "368M",
      "year": 2025,
      "dialects": [
        "lev"
      ],
      "notes": "Syrian dialect to MSA bidirectional translation model.",
      "base_model": [
        "ubc-nlp/arat5v2-base-1024"
      ],
      "metrics": {
        "downloads": 49,
        "likes": 1,
        "lastModified": "2025-08-06"
      }
    },
    {
      "id": "unlimited-ocr-quran-uthmani",
      "name": "unlimited ocr quran uthmani",
      "type": "ocr",
      "country": "INTL",
      "org": "AyoubChLin",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "ocr",
        "diacritization",
        "vision",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/AyoubChLin/unlimited-ocr-quran-uthmani"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "AyoubChLin/unlimited-ocr-quran-uthmani is a Quran-focused OCR fine-tune of baidu/Unlimited-OCR.",
      "base_model": [
        "baidu/unlimited-ocr"
      ],
      "metrics": {
        "downloads": 49,
        "likes": 5,
        "lastModified": "2026-07-02"
      }
    },
    {
      "id": "all-hadith",
      "name": "all hadith",
      "type": "dataset",
      "country": "INTL",
      "org": "M-AI-C",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/M-AI-C/all_hadith"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "Hadith corpus with source, chapter, hadith number, chain index and Arabic text columns.",
      "metrics": {
        "downloads": 48,
        "likes": 4,
        "lastModified": "2023-04-02"
      }
    },
    {
      "id": "arabic-dialects-question-and-answer",
      "name": "arabic dialects question and answer",
      "type": "dataset",
      "country": "INTL",
      "org": "CNTXTAI0",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "dialects",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CNTXTAI0/arabic_dialects_question_and_answer"
      },
      "dialects": [
        "msa",
        "gulf",
        "egy",
        "lev"
      ],
      "size": "<1K rows",
      "year": 2025,
      "notes": "500 reasoning questions in English, MSA, Emirati, Egyptian and Levantine columns, with hints and word counts.",
      "metrics": {
        "downloads": 48,
        "likes": 6,
        "lastModified": "2025-01-31"
      }
    },
    {
      "id": "arabic-poem-comprehensive-dataset-apcd",
      "name": "Arabic Poem Comprehensive Dataset APCD",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdelrahman-Rezk",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry",
        "meter"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdelrahman-Rezk/Arabic_Poem_Comprehensive_Dataset_APCD"
      },
      "size": "1M–10M rows",
      "year": 2022,
      "notes": "1.68M Arabic poem verses with era, poet, collection, rhyme and meter (bahr) metadata.",
      "metrics": {
        "downloads": 48,
        "likes": 3,
        "lastModified": "2022-05-27"
      }
    },
    {
      "id": "arabic-semantic-relevance",
      "name": "arabic semantic relevance",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "ner",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/arabic-semantic-relevance"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "This dataset provides high-quality Arabic query-context pairs with fine-grained sem",
      "metrics": {
        "downloads": 48,
        "likes": 2,
        "lastModified": "2026-01-13"
      }
    },
    {
      "id": "nn-auto-bench-ds",
      "name": "nn auto bench ds",
      "type": "benchmark",
      "country": "INTL",
      "org": "nanonets",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/nanonets/nn-auto-bench-ds"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "nn-auto-bench-ds is a dataset designed for key information extraction (KIE) and serves as a benchmark dataset for nn-auto-bench.",
      "metrics": {
        "downloads": 48,
        "likes": 5,
        "lastModified": "2025-03-13"
      }
    },
    {
      "id": "arab3m-triplets",
      "name": "Arab3M-Triplets",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arab3M-Triplets"
      },
      "size": "1M-10M rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Roughly three million Arabic triplets for contrastive embedding training.",
      "metrics": {
        "downloads": 47,
        "likes": 4,
        "lastModified": "2024-08-29"
      }
    },
    {
      "id": "arabert-summarization-goud",
      "name": "AraBERT Summarization Goud",
      "type": "llm",
      "country": "INTL",
      "org": "Goud",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/Goud/AraBERT-summarization-goud"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "year": 2022,
      "notes": "It is an encoder-decoder model that was initialized with bert-base-arabertv02-twitter checkpoint.",
      "metrics": {
        "downloads": 47,
        "likes": 1,
        "lastModified": "2022-04-29"
      }
    },
    {
      "id": "arabic-fake-news-dataset",
      "name": "Arabic fake news dataset",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Arabic_fake_news_dataset"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "This repository contains the Arabicfakenewsdataset, a collection of news articles scraped from the Egyptian platform متصدقش (Matsda2sh).",
      "metrics": {
        "downloads": 47,
        "likes": 2,
        "lastModified": "2023-07-23"
      }
    },
    {
      "id": "arbn",
      "name": "ArBN",
      "type": "dataset",
      "country": "INTL",
      "org": "Arab Center for Research and Policy Studies",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/dru-ac/ArBNTopic",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.32.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "19,784 sentences",
      "year": 2023,
      "notes": "Arabic dataset for topic classification",
      "metrics": {
        "downloads": 47,
        "likes": 3,
        "lastModified": "2023-10-17"
      }
    },
    {
      "id": "iadd",
      "name": "IADD",
      "type": "dataset",
      "country": "MA",
      "org": "Cadi Ayyad University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/JihadZa/IADD",
        "hf": "https://huggingface.co/datasets/evageon/IADD",
        "paper": "https://doi.org/10.1016/j.dib.2021.107777"
      },
      "dialects": [
        "mixed",
        "magh",
        "egy",
        "iraqi",
        "gulf",
        "msa"
      ],
      "size": "135,804 sentences",
      "year": 2022,
      "notes": "Integrated dataset for Arabic dialect identification with 135k texts from 5 regions and 9 countries.",
      "metrics": {
        "downloads": 47,
        "likes": 5,
        "lastModified": "2022-01-29"
      }
    },
    {
      "id": "nawah-parakeet-60m",
      "name": "Nawah Parakeet 60M",
      "type": "asr",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "pretraining",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/Nawah-Parakeet-60M"
      },
      "size": "60M",
      "on_device": true,
      "year": 2026,
      "notes": "A 62.7M-parameter Parakeet-style Token-and-Duration Transducer for Arabic speech recognition — FastConformer encoder, TDT objective.",
      "metrics": {
        "downloads": 47,
        "likes": 2,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "nileulex",
      "name": "NileULex",
      "type": "dataset",
      "country": "EG",
      "org": "Nile University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/NileTMRG/NileULex",
        "hf": "https://huggingface.co/datasets/arbml/NileULex",
        "paper": "https://aclanthology.org/L16-1463.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,953 sentences",
      "year": 2016,
      "notes": "Egyptian Arabic and Modern Standard Arabic sentiment words and their polarity",
      "metrics": {
        "downloads": 47,
        "likes": 0,
        "lastModified": "2024-07-21"
      }
    },
    {
      "id": "quranplus",
      "name": "QuranPlus",
      "type": "llm",
      "country": "INTL",
      "org": "justdeen",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/justdeen/QuranPlus"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This model is fine-tuned to answer questions about Islam and the Quran in English.",
      "metrics": {
        "downloads": 47,
        "likes": 4,
        "lastModified": "2025-08-03"
      }
    },
    {
      "id": "tashkeelav2",
      "name": "tashkeelav2",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/tashkeelav2"
      },
      "size": "100K–1M rows",
      "year": 2023,
      "notes": "580K Arabic sentences with diacritized and plain text pairs for tashkeel.",
      "metrics": {
        "downloads": 47,
        "likes": 6,
        "lastModified": "2023-04-09"
      }
    },
    {
      "id": "arabic-spam-and-ham-tweets",
      "name": "Arabic Spam and Ham Tweets",
      "type": "dataset",
      "country": "INTL",
      "org": "Zayed University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "spam-detection"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/86x733xkb8/2",
        "hf": "https://huggingface.co/datasets/arbml/arabic_spam_ham_twitter",
        "paper": "https://link.springer.com/article/10.1007/s00521-023-08614-w"
      },
      "dialects": [
        "mixed"
      ],
      "size": "13,241 sentences",
      "year": 2024,
      "notes": "The dataset contains 13241 records.",
      "metrics": {
        "downloads": 46,
        "likes": 0,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "argurad-task1",
      "name": "ArGuard Track A",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-4.0",
      "modality": "multimodal",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ArGuard-Task1"
      },
      "size": "1K-10K rows",
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic hateful-meme detection dataset for the ArGuard shared task.",
      "metrics": {
        "downloads": 46,
        "likes": 0,
        "lastModified": "2026-08-10"
      }
    },
    {
      "id": "asr-whisper-large-v2-commonvoice-ar",
      "name": "asr whisper large v2 commonvoice ar",
      "type": "asr",
      "country": "INTL",
      "org": "SpeechBrain",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/speechbrain/asr-whisper-large-v2-commonvoice-ar"
      },
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "This repository provides all the necessary tools to perform automatic speech recognition from an end-to-end whisper model fine-tuned on CommonVoice.",
      "metrics": {
        "downloads": 46,
        "likes": 3,
        "lastModified": "2025-01-03"
      }
    },
    {
      "id": "egyptian-arabic-english-translation-50k",
      "name": "egyptian arabic english translation 50k",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "translation",
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/egyptian-arabic-english-translation-50k"
      },
      "dialects": [
        "egy",
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "An open-source parallel corpus of 50,000 Egyptian Arabic sentences paired with their English translations.",
      "metrics": {
        "downloads": 46,
        "likes": 2,
        "lastModified": "2026-07-22"
      }
    },
    {
      "id": "quran-dataset-abdulbasit-clean",
      "name": "quran dataset abdulbasit clean",
      "type": "dataset",
      "country": "INTL",
      "org": "Nash-pAnDiTa",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Nash-pAnDiTa/quran_dataset_abdulbasit_clean"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "The Tanzil Project is an international initiative aimed at providing a highly accurate and verified Quranic text in Unicode.",
      "metrics": {
        "downloads": 46,
        "likes": 3,
        "lastModified": "2024-11-15"
      }
    },
    {
      "id": "spoken-arabic-digits",
      "name": "spoken arabic digits",
      "type": "dataset",
      "country": "INTL",
      "org": "mohnasgbr",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mohnasgbr/spoken-arabic-digits"
      },
      "dialects": [
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2023,
      "notes": "This dataset contains spoken Arabic digits from 40 speakers from multiple Arab communities and local dialects.",
      "metrics": {
        "downloads": 46,
        "likes": 3,
        "lastModified": "2023-10-16"
      }
    },
    {
      "id": "the-sadid-evaluation-datasets",
      "name": "The SADID Evaluation Datasets",
      "type": "benchmark",
      "country": "INTL",
      "org": "Stanford University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation"
      ],
      "links": {
        "github": "https://github.com/we7el/SADID",
        "hf": "https://huggingface.co/datasets/arbml/SADID",
        "paper": "https://aclanthology.org/2020.coling-main.530.pdf"
      },
      "dialects": [
        "mixed",
        "lev",
        "egy",
        "msa"
      ],
      "size": "29,964 sentences",
      "year": 2020,
      "notes": "Evaluation Datasets for Low-Resource Spoken Language Machine Translation of Arabic Dialects",
      "metrics": {
        "downloads": 46,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "egyptian-dialogue",
      "name": "Egyptian Dialogue",
      "type": "dataset",
      "country": "INTL",
      "org": "fr3on",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fr3on/egyptian-dialogue"
      },
      "dialects": [
        "egy"
      ],
      "notes": "4,322 Egyptian Arabic-English parallel pairs from TV subtitles",
      "metrics": {
        "downloads": 45,
        "likes": 9,
        "lastModified": "2025-12-22"
      }
    },
    {
      "id": "arabic-broad-benchmark-abb",
      "name": "Arabic Broad Benchmark (ABB)",
      "type": "benchmark",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/silma-ai/arabic-broad-benchmark"
      },
      "notes": "Comprehensive evaluation tool for Arabic LLMs",
      "metrics": {
        "downloads": 44,
        "likes": 14,
        "lastModified": "2025-07-21"
      }
    },
    {
      "id": "chatterbox-egyptian-v0",
      "name": "chatterbox-egyptian-v0",
      "type": "tts",
      "country": "EG",
      "org": "oddadmix",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/chatterbox-egyptian-v0"
      },
      "size": "536M",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "egy"
      ],
      "notes": "Chatterbox TTS fine-tuned for Egyptian Arabic.",
      "base_model": [
        "resembleai/chatterbox"
      ],
      "metrics": {
        "downloads": 44,
        "likes": 30,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "dibt-10k-prompts-ranked-arabic",
      "name": "dibt 10k prompts ranked arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I-R",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "prompts",
        "instructions"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I-R/dibt_10k_prompts_ranked_arabic"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Arabic translation of the DIBT 10K ranked prompts with quality ratings, annotator agreement and raw ratings (10,331 rows).",
      "metrics": {
        "downloads": 44,
        "likes": 3,
        "lastModified": "2024-03-07"
      }
    },
    {
      "id": "maoffens",
      "name": "MAOffens",
      "type": "dataset",
      "country": "MA",
      "org": "Mohammed V University",
      "license": "cc-by-nc-2.0",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Randa/MAOffens",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-031-80438-0_2"
      },
      "dialects": [
        "magh"
      ],
      "size": "23,000 sentences",
      "year": 2024,
      "notes": "MAOffens is the first Moroccan Arabic dataset for offensive language detection in social media text.",
      "metrics": {
        "downloads": 44,
        "likes": 1,
        "lastModified": "2025-02-04"
      }
    },
    {
      "id": "aner",
      "name": "ANER",
      "type": "llm",
      "country": "INTL",
      "org": "boda",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/boda/ANER"
      },
      "year": 2023,
      "notes": "This project is made to enrich the Arabic Named Entity Recognition(ANER).",
      "metrics": {
        "downloads": 43,
        "likes": 4,
        "lastModified": "2023-09-18"
      }
    },
    {
      "id": "arabic-100k-reviews",
      "name": "Arabic 100k Reviews",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/abedkhooli/arabic-100k-reviews",
        "hf": "https://huggingface.co/datasets/arbml/arabic_100k_reviews"
      },
      "dialects": [
        "mixed"
      ],
      "size": "99,999 documents",
      "year": 2022,
      "notes": "This dataset is mainly a compilation of several available datasets and a sampling of 100k rows (99999 to be exact).",
      "metrics": {
        "downloads": 43,
        "likes": 1,
        "lastModified": "2024-03-30"
      }
    },
    {
      "id": "arabic-math-sft",
      "name": "Arabic Math SFT",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "instruction-tuning",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-Math-SFT"
      },
      "size": "1K-10K rows",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Multimodal Arabic mathematics dataset for supervised fine-tuning.",
      "metrics": {
        "downloads": 43,
        "likes": 4,
        "lastModified": "2026-04-09"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-ca-poetry",
      "name": "bert base arabic camelbert ca poetry",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-ca-poetry"
      },
      "year": 2021,
      "notes": "For the fine-tuning, we used the APCD dataset.",
      "metrics": {
        "downloads": 43,
        "likes": 5,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "calme-2-2",
      "name": "Calme 2.2",
      "type": "llm",
      "country": "INTL",
      "org": "MaziyarPanahi",
      "license": "tongyi-qianwen",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MaziyarPanahi/calme-2.2-qwen2-72b"
      },
      "base_model": [
        "qwen/qwen2-72b"
      ],
      "size": "72B",
      "notes": "Qwen2-based, strong OALL performance",
      "metrics": {
        "downloads": 43,
        "likes": 5,
        "lastModified": "2024-09-15"
      }
    },
    {
      "id": "classification-arabic-dialects",
      "name": "classification arabic dialects",
      "type": "dataset",
      "country": "INTL",
      "org": "Falah",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Falah/classification_arabic_dialects"
      },
      "dialects": [
        "mixed"
      ],
      "size": "<1K rows",
      "year": 2023,
      "notes": "This dataset contains audio samples of various Arabic dialects for the task of classification and recognition.",
      "metrics": {
        "downloads": 43,
        "likes": 3,
        "lastModified": "2023-07-03"
      }
    },
    {
      "id": "fastconformer-quran-streaming",
      "name": "fastconformer quran streaming",
      "type": "asr",
      "country": "INTL",
      "org": "Muno459",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/Muno459/fastconformer-quran-streaming"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Streaming cache-aware FastConformer ASR for Quran recitation (NeMo, ONNX), trained on EveryAyah, TLOG and Muaalem data; gated.",
      "metrics": {
        "downloads": 43,
        "likes": 14,
        "lastModified": "2026-08-05"
      }
    },
    {
      "id": "lahjamt",
      "name": "LahjaMT",
      "type": "llm",
      "country": "INTL",
      "org": "MohamedAbdallah98",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/MohamedAbdallah98/LahjaMT"
      },
      "year": 2026,
      "notes": "(EG, JO, LB, LY, MA, MR, OM, PS, SA, SD, SY, TN, YE).",
      "base_model": [
        "ubc-nlp/nilechat-3b-base"
      ],
      "metrics": {
        "downloads": 43,
        "likes": 3,
        "lastModified": "2026-09-15"
      }
    },
    {
      "id": "omani-arabic-nmt",
      "name": "OmaniArabicNMT",
      "type": "dataset",
      "country": "INTL",
      "org": "XRI Global",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/xri/OmaniArabicNMT"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "8,000 parallel MSA-Omani Arabic sentences for dialect machine translation.",
      "metrics": {
        "downloads": 43,
        "likes": 2,
        "lastModified": "2025-02-27"
      }
    },
    {
      "id": "tunisian-language-dataset",
      "name": "Tunisian Language Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "AzizBelaweid",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AzizBelaweid/Tunisian_Language_Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "This dataset is a curated compilation of various Tunisian datasets, aimed at gathering as much Tunisian text data as possible in one place.",
      "metrics": {
        "downloads": 43,
        "likes": 7,
        "lastModified": "2024-10-15"
      }
    },
    {
      "id": "ara-nemotron-3-5-asr-streaming-0-6b",
      "name": "Ara nemotron 3.5 asr streaming 0.6b",
      "type": "asr",
      "country": "INTL",
      "org": "Abdelkareem",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Abdelkareem/Ara-nemotron-3.5-asr-streaming-0.6b"
      },
      "dialects": [
        "egy"
      ],
      "size": "0.6B",
      "on_device": true,
      "year": 2026,
      "notes": "This model is an adapter-based fine-tune of nvidia/nemotron-3.5-asr-streaming-0.6b for Egyptian Arabic (arz) speech recognition.",
      "metrics": {
        "downloads": 42,
        "likes": 6,
        "lastModified": "2026-06-05"
      }
    },
    {
      "id": "arabic-reverse-dictionary",
      "name": "arabic reverse dictionary",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/arabic_reverse_dictionary"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "We present a novel Arabic Reverse Dictionary (RD) dataset comprising over 58,000 definitions collected from trusted, open-source books and websites.",
      "metrics": {
        "downloads": 42,
        "likes": 4,
        "lastModified": "2025-11-15"
      }
    },
    {
      "id": "bert-base-arabic-camelbert-msa-did-madar-twitter5",
      "name": "bert base arabic camelbert msa did madar twitter5",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/bert-base-arabic-camelbert-msa-did-madar-twitter5"
      },
      "dialects": [
        "msa",
        "mixed"
      ],
      "year": 2021,
      "notes": "For the fine-tuning, we used the MADAR Twitter-5 dataset, which includes 21 labels.",
      "metrics": {
        "downloads": 42,
        "likes": 3,
        "lastModified": "2021-10-17"
      }
    },
    {
      "id": "pearl",
      "name": "PEARL",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-nd-4.0",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/PEARL"
      },
      "notes": "Multimodal Culturally-Aware Arabic Instruction Dataset",
      "metrics": {
        "downloads": 42,
        "likes": 5,
        "lastModified": "2025-10-27"
      }
    },
    {
      "id": "wasm",
      "name": "WASM",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MagedSaeed/wasm",
        "paper": "https://link.springer.com/article/10.1007/s13369-023-08567-1"
      },
      "dialects": [
        "mixed"
      ],
      "size": "101,099 sentences",
      "year": 2024,
      "notes": "Arabic dataset for twitter hashtags recommendation, generation, and classification",
      "metrics": {
        "downloads": 42,
        "likes": 0,
        "lastModified": "2025-01-05"
      }
    },
    {
      "id": "arabic-poems",
      "name": "Arabic Poems",
      "type": "dataset",
      "country": "INTL",
      "org": "alwalid54321",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/alwalid54321/Arabic_Poems"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Filtered al-Diwan subset of arbml/ashaar: 8,875 Arabic poems with title, meter, verses and poet metadata.",
      "metrics": {
        "downloads": 41,
        "likes": 3,
        "lastModified": "2024-06-16"
      }
    },
    {
      "id": "arabic-sentiment-model",
      "name": "arabic sentiment model",
      "type": "llm",
      "country": "INTL",
      "org": "Walid-Ahmed",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/Walid-Ahmed/arabic-sentiment-model"
      },
      "year": 2024,
      "notes": "This model is trained for Arabic sentiment classification.",
      "base_model": [
        "aubmindlab/bert-base-arabertv2"
      ],
      "metrics": {
        "downloads": 41,
        "likes": 3,
        "lastModified": "2024-09-28"
      }
    },
    {
      "id": "arasenti",
      "name": "AraSenti",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/AraSenti"
      },
      "year": 2017,
      "dialects": [
        "gulf",
        "msa"
      ],
      "notes": "Arabic sentiment tweets corpus covering Saudi dialect and MSA, plus a sentiment lexicon.",
      "metrics": {
        "downloads": 41,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "dvoice",
      "name": "DVoice",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Dvoice"
      },
      "year": 2021,
      "dialects": [
        "magh"
      ],
      "notes": "Dialectal speech corpora for Darija and other African and Arabic languages.",
      "metrics": {
        "downloads": 41,
        "likes": 1,
        "lastModified": "2022-12-03"
      }
    },
    {
      "id": "moroccan-darija-instruct-573k",
      "name": "Moroccan Darija Instruct 573K",
      "type": "dataset",
      "country": "INTL",
      "org": "Lyte",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Lyte/Moroccan-Darija-Instruct-573K"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Synthetic instruction-tuning dataset of 573,175 question-answer pairs written entirely in Moroccan Darija.",
      "metrics": {
        "downloads": 41,
        "likes": 3,
        "lastModified": "2026-03-05"
      }
    },
    {
      "id": "saudi-dialect-test-samples",
      "name": "saudi dialect test samples",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/saudi-dialect-test-samples"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset contains 1280 Saudi dialect utterances across 44 categories, used for testing and evaluating the Omartificial-Intelligence-Space/SA-BERT-V1 model.",
      "metrics": {
        "downloads": 41,
        "likes": 5,
        "lastModified": "2025-05-13"
      }
    },
    {
      "id": "zarra",
      "name": "zarra",
      "type": "embedding",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/zarra"
      },
      "year": 2025,
      "notes": "It is a distilled version of a Sentence Transformer, specifically optimized for the Arabic language.",
      "base_model": [
        "jinaai/jina-embeddings-v3"
      ],
      "metrics": {
        "downloads": 41,
        "likes": 2,
        "lastModified": "2025-06-16"
      }
    },
    {
      "id": "allava-4v-arabic",
      "name": "ALLaVA-4V-Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "qa",
        "chat",
        "ocr",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/ALLaVA-4V-Arabic"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "This is the Arabic version of the ALLaVA-4V data.",
      "metrics": {
        "downloads": 40,
        "likes": 4,
        "lastModified": "2024-04-29"
      }
    },
    {
      "id": "barec-shared-task-2025-doc",
      "name": "BAREC Shared Task 2025 doc",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CAMeL-Lab/BAREC-Shared-Task-2025-doc"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "The dataset is annotated at the sentence level. (also: 1 variants)",
      "metrics": {
        "downloads": 40,
        "likes": 2,
        "lastModified": "2025-06-11"
      }
    },
    {
      "id": "hadith",
      "name": "Hadith",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Hadith"
      },
      "dialects": [
        "classical"
      ],
      "size": "100K–1M rows",
      "year": 2022,
      "notes": "124K Arabic hadith texts.",
      "metrics": {
        "downloads": 40,
        "likes": 3,
        "lastModified": "2022-10-24"
      }
    },
    {
      "id": "nilechat-lhv-egypt",
      "name": "NileChat LHV-Egypt",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/nilechat-lhv-egy"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian dialect dataset for LLM pre-training",
      "metrics": {
        "downloads": 40,
        "likes": 3,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "pangeanic-iraqi-arabic-multidomain-qa",
      "name": "Pangeanic Iraqi Arabic multidomain QA",
      "type": "dataset",
      "country": "INTL",
      "org": "Pangeanic",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Pangeanic/Iraqi-Arabic-multidomain-QA-text"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2026,
      "notes": "Sample of Iraqi Arabic multi-domain question-answer text from Pangeanic.",
      "metrics": {
        "downloads": 40,
        "likes": 1,
        "lastModified": "2026-06-04"
      }
    },
    {
      "id": "pearl-vdr-ar-train-preprocessed",
      "name": "Pearl vdr ar train preprocessed",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "embedding",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Pearl-vdr-ar-train-preprocessed"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic culturally-aligned, VDR-style (query, image, hard-negatives) triplets for training multimodal embedding models with Sentence Transformers.",
      "metrics": {
        "downloads": 40,
        "likes": 3,
        "lastModified": "2026-04-21"
      }
    },
    {
      "id": "quran-multilingual-parallel",
      "name": "quran multilingual parallel",
      "type": "dataset",
      "country": "INTL",
      "org": "freococo",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/freococo/quran_multilingual_parallel"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset presents a clean, structurally-aligned multilingual parallel corpus of the Qur’anic text.",
      "metrics": {
        "downloads": 40,
        "likes": 5,
        "lastModified": "2025-05-30"
      }
    },
    {
      "id": "arabic-satire-dataset",
      "name": "Arabic satire dataset",
      "type": "dataset",
      "country": "SA",
      "org": "University of Jeddah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "retrieval",
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/Noza1234/Arbic-satire-dataset",
        "hf": "https://huggingface.co/datasets/arbml/Arbic-satire-dataset",
        "paper": "https://www.mdpi.com/2076-3417/13/19/10616"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,000 sentences",
      "year": 2023,
      "notes": "500 Arabic news and 500 Arabic satire articles",
      "metrics": {
        "downloads": 39,
        "likes": 1,
        "lastModified": "2024-03-17"
      }
    },
    {
      "id": "arabic-qwen-chat-llamafile",
      "name": "arabic_Qwen_chat_llamafile",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/arabic_Qwen_chat_llamafile"
      },
      "year": 2024,
      "notes": "Arabic chat model fine-tuned with Unsloth, published as Llama-architecture safetensors despite the llamafile name; no base or dataset listed.",
      "metrics": {
        "downloads": 39,
        "likes": 3,
        "lastModified": "2024-05-06"
      }
    },
    {
      "id": "arsl21l",
      "name": "ArSL21L",
      "type": "dataset",
      "country": "AE",
      "org": "UAE University",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/8hrn3bvdvk/1",
        "hf": "https://huggingface.co/datasets/arbml/ArSL21L",
        "paper": "https://ieeexplore.ieee.org/abstract/document/9766497"
      },
      "dialects": [
        "msa"
      ],
      "size": "14,202 tokens",
      "year": 2022,
      "notes": "Annotated Arabic Sign Language Letters Dataset (ArSL21L) consisting of 14202 images of 32 letter signs with various back- grounds collected from 50 people.",
      "metrics": {
        "downloads": 39,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "h4-no-robots",
      "name": "H4 no robots",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "chat",
        "translation",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/H4_no_robots"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "\"No Robots\" is a dataset consisting of 10,000 instructions and demonstrations, created by professional annotators.",
      "metrics": {
        "downloads": 39,
        "likes": 6,
        "lastModified": "2024-02-28"
      }
    },
    {
      "id": "jais-13b-chat-hf",
      "name": "jais 13b chat hf",
      "type": "llm",
      "country": "INTL",
      "org": "derek-thomas",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/derek-thomas/jais-13b-chat-hf"
      },
      "size": "13B",
      "on_device": false,
      "year": 2023,
      "notes": "I made a couple changes, I use LLM.int8() to load this in 8 bits rather than full precision which lowers the GPU VRAM requirements by 3x.",
      "metrics": {
        "downloads": 39,
        "likes": 5,
        "lastModified": "2023-12-13"
      }
    },
    {
      "id": "masd",
      "name": "MASD",
      "type": "dataset",
      "country": "INTL",
      "org": "SaiedAlshahrani",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/MASD"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "Masked Arab States Dataset: 160 masked prompts on capitals, currencies, nationalities and continents of 20 Arab states for probing MLMs.",
      "metrics": {
        "downloads": 39,
        "likes": 3,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "sa-bert-v1",
      "name": "SA-BERT-V1",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/SA-BERT-V1"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi Arabic BERT, fine-tuned MARBERTv2",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 39,
        "likes": 5,
        "lastModified": "2025-11-28"
      }
    },
    {
      "id": "silma-arabic-english-sts-dataset-v1-0",
      "name": "silma arabic english sts dataset v1.0",
      "type": "dataset",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/silma-ai/silma-arabic-english-sts-dataset-v1.0"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The SILMA STS Arabic/English Dataset - v1.0 is a dataset designed for training and evaluating sentence embeddings for Arabic and English tasks.",
      "metrics": {
        "downloads": 39,
        "likes": 3,
        "lastModified": "2024-10-17"
      }
    },
    {
      "id": "4factors-palestinian-levantine-q-a-sample",
      "name": "4FACTORS Palestinian Levantine Q&A Sample",
      "type": "dataset",
      "country": "INTL",
      "org": "factors",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/4factors/arabic-palestinian-levantine-sample"
      },
      "dialects": [
        "lev"
      ],
      "size": "50 sentences",
      "year": 2026,
      "tags": [
        "multilingual"
      ],
      "notes": "Native-written Palestinian Levantine conversational question-answer pairs (50) with English glosses and domain labels; written by a first-language speaker.",
      "metrics": {
        "downloads": 38,
        "likes": 1,
        "lastModified": "2026-07-15"
      }
    },
    {
      "id": "arabic-acoustic-poetry-benchmark",
      "name": "arabic acoustic poetry benchmark",
      "type": "benchmark",
      "country": "SA",
      "org": "KFUPM-JRCAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "poetry",
        "meter",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KFUPM-JRCAI/arabic-acoustic-poetry-benchmark"
      },
      "size": "<1K rows",
      "year": 2022,
      "notes": "268 Arabic poetry recitations with bait text, bahr (meter), reciter and source, for acoustic poetry evaluation.",
      "metrics": {
        "downloads": 38,
        "likes": 3,
        "lastModified": "2022-09-06"
      }
    },
    {
      "id": "arabic-llama3-1-16bit-ft",
      "name": "Arabic llama3.1 16bit FT",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-llama3.1-16bit-FT"
      },
      "year": 2024,
      "notes": "This fine-tuned model is based on the newly released LLaMA 3.1 model and has been specifically trained on the Arabic BigScience xP3 dataset.",
      "metrics": {
        "downloads": 38,
        "likes": 4,
        "lastModified": "2024-08-01"
      }
    },
    {
      "id": "arat5v2-xlsum-arabic-text-summarization",
      "name": "AraT5v2-XLSum-arabic-text-summarization",
      "type": "llm",
      "country": "INTL",
      "org": "omarsabri8756",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/omarsabri8756/AraT5v2-XLSum-arabic-text-summarization"
      },
      "year": 2025,
      "notes": "This model is a fine-tuned version of UBC-NLP/AraT5v2-base-1024 specifically trained for Arabic text summarization using the XLSum dataset.",
      "metrics": {
        "downloads": 38,
        "likes": 2,
        "lastModified": "2025-05-11"
      }
    },
    {
      "id": "darija-english-combined",
      "name": "darija english combined",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/darija-english-combined"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2026,
      "notes": "English-Moroccan Darija-Arabizi parallel sentences merged from several public datasets into one format for translation models.",
      "metrics": {
        "downloads": 38,
        "likes": 2,
        "lastModified": "2026-07-01"
      }
    },
    {
      "id": "jais-13b-8bit",
      "name": "jais 13B 8bit",
      "type": "llm",
      "country": "INTL",
      "org": "Asas AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/asas-ai/jais_13B_8bit"
      },
      "size": "13B",
      "on_device": false,
      "year": 2023,
      "notes": "This is a 13 billion parameter pre-trained bilingual large language model for both Arabic and English.",
      "metrics": {
        "downloads": 38,
        "likes": 9,
        "lastModified": "2023-10-25"
      }
    },
    {
      "id": "lahga-arabic-dialects",
      "name": "Lahga Arabic Dialects Dictionary",
      "type": "dataset",
      "country": "INTL",
      "org": "Lahga",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/lahga/arabic-dialects"
      },
      "year": 2026,
      "dialects": [
        "mixed"
      ],
      "notes": "Open collaborative dictionary of Arabic dialects keyed by MSA meaning.",
      "metrics": {
        "downloads": 38,
        "likes": 0,
        "lastModified": "2026-10-02"
      }
    },
    {
      "id": "whisper-small-for-quran",
      "name": "whisper small for quran",
      "type": "asr",
      "country": "INTL",
      "org": "areaz",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/areaz/whisper-small-for-quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Arabic automatic speech recognition model fine-tuned from openai/whisper-small trained on abdulhamedeid/quran-verses-audio-clips.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 38,
        "likes": 6,
        "lastModified": "2025-01-12"
      }
    },
    {
      "id": "glare",
      "name": "GLARE",
      "type": "dataset",
      "country": "SA",
      "org": "SDAIA",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://zenodo.org/record/6457824",
        "hf": "https://huggingface.co/datasets/Fatima-Gh/GLARE"
      },
      "dialects": [
        "mixed"
      ],
      "size": "76,000,000 sentences",
      "year": 2022,
      "notes": "GLARE an Arabic Apps Reviews dataset collected from Saudi Google PlayStore.",
      "metrics": {
        "downloads": 37,
        "likes": 0,
        "lastModified": "2025-03-14"
      }
    },
    {
      "id": "kani-tts-400m-ar",
      "name": "kani tts 400m ar",
      "type": "tts",
      "country": "INTL",
      "org": "nineninesix",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/nineninesix/kani-tts-400m-ar"
      },
      "size": "400M",
      "on_device": true,
      "year": 2026,
      "notes": "A high-speed, high-fidelity Text-to-Speech model optimized for real-time conversational AI applications.",
      "base_model": [
        "nineninesix/kani-tts-400m-0.3-pt"
      ],
      "metrics": {
        "downloads": 37,
        "likes": 5,
        "lastModified": "2026-02-18"
      }
    },
    {
      "id": "moroccan-arabic-wikipedia-20230101-nobots",
      "name": "Moroccan Arabic Wikipedia 20230101 nobots",
      "type": "dataset",
      "country": "INTL",
      "org": "SaiedAlshahrani",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/Moroccan_Arabic_Wikipedia_20230101_nobots"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "This dataset is created using the Moroccan Arabic Wikipedia articles.",
      "metrics": {
        "downloads": 37,
        "likes": 3,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "sinai-voice-ar-stt",
      "name": "sinai voice ar stt",
      "type": "asr",
      "country": "EG",
      "org": "Bakrianoo",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/bakrianoo/sinai-voice-ar-stt"
      },
      "year": 2022,
      "notes": "This model is a fine-tuned version of facebook/wav2vec2-xls-r-300m on the MOZILLA-FOUNDATION/COMMONVOICE80 - AR dataset.",
      "metrics": {
        "downloads": 37,
        "likes": 13,
        "lastModified": "2022-03-23"
      }
    },
    {
      "id": "stmc",
      "name": "STMC",
      "type": "dataset",
      "country": "SA",
      "org": "Faisal Qarah",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/faisalq/STMC"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi Tweets Mega Corpus (141M+ Saudi dialect tweets)",
      "metrics": {
        "downloads": 37,
        "likes": 0,
        "lastModified": "2024-05-08"
      }
    },
    {
      "id": "tunizi",
      "name": "TUNIZI",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/TUNIZI"
      },
      "year": 2020,
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian Arabizi sentiment dataset in Latin script.",
      "metrics": {
        "downloads": 37,
        "likes": 5,
        "lastModified": "2024-03-23"
      }
    },
    {
      "id": "xlm-r-large-arabic-sent",
      "name": "xlm r large arabic sent",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/xlm-r-large-arabic-sent"
      },
      "year": 2023,
      "notes": "Multilingual sentiment classification (Label0: mixed, Label1: negative, Label2: positive) of Arabic reviews by fine-tuning XLM-Roberta-Large.",
      "metrics": {
        "downloads": 37,
        "likes": 8,
        "lastModified": "2023-02-10"
      }
    },
    {
      "id": "aftd",
      "name": "AFTD",
      "type": "dataset",
      "country": "DZ",
      "org": "University of Biskra",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/zeydferhat/functional_text_dimensions_for_arabic_text_classification",
        "paper": "https://aclanthology.org/2024.arabicnlp-1.29.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,400 documents",
      "year": 2024,
      "notes": "(AFTD Corpus) is introduced as a curated collection of Arabic documents aimed at evaluating text classification methodologies using the Functional Text.",
      "metrics": {
        "downloads": 36,
        "likes": 1,
        "lastModified": "2024-06-27"
      }
    },
    {
      "id": "arabic-preference-data-rlhf",
      "name": "Arabic preference data RLHF",
      "type": "dataset",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preference",
        "rlhf"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Arabic-preference-data-RLHF"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "11.5K Arabic RLHF preference rows with instruction, chosen and rejected responses.",
      "metrics": {
        "downloads": 36,
        "likes": 4,
        "lastModified": "2023-09-21"
      }
    },
    {
      "id": "egyhellaswag",
      "name": "EgyHellaSwag",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "commonsense"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/EgyHellaSwag"
      },
      "year": 2025,
      "dialects": [
        "egy"
      ],
      "notes": "HellaSwag translated into Egyptian Arabic from UBC (Canada).",
      "metrics": {
        "downloads": 36,
        "likes": 2,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "goud-sum",
      "name": "Goud-sum",
      "type": "dataset",
      "country": "INTL",
      "org": "Archipel Cognitive",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "github": "https://github.com/issam9/goud-summarization-dataset",
        "hf": "https://huggingface.co/datasets/Goud/Goud-sum",
        "paper": "https://openreview.net/pdf?id=BMVq5MELb9"
      },
      "dialects": [
        "magh"
      ],
      "size": "158,000 documents",
      "year": 2022,
      "notes": "Goud-sum contains 158k articles and their headlines extracted from Goud.ma news website.",
      "metrics": {
        "downloads": 36,
        "likes": 7,
        "lastModified": "2022-07-04"
      }
    },
    {
      "id": "kalematech-arabic-stt",
      "name": "KalemaTech Arabic STT",
      "type": "asr",
      "country": "INTL",
      "org": "Salama1429",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Salama1429/KalemaTech-Arabic-STT-ASR-based-on-Whisper-Small"
      },
      "notes": "Whisper-small optimized for Arabic STT",
      "metrics": {
        "downloads": 36,
        "likes": 6,
        "lastModified": "2022-12-28"
      }
    },
    {
      "id": "lk-hadith",
      "name": "LK Hadith",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/LK_Hadith"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2022,
      "notes": "34K hadith entries with Arabic and English chapter, section and hadith text (Leeds-King Saud corpus).",
      "metrics": {
        "downloads": 36,
        "likes": 10,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "mediaspeech-ar",
      "name": "MediaSpeech Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/MediaSpeech_ar"
      },
      "size": "10h",
      "year": 2021,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic media speech ASR test set from the MediaSpeech corpus.",
      "metrics": {
        "downloads": 36,
        "likes": 2,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "nilechat-3b-base",
      "name": "NileChat-3B-Base",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/NileChat-3B-Base"
      },
      "base_model": [
        "qwen/qwen2.5-3b"
      ],
      "size": "3.4B",
      "year": 2026,
      "dialects": [
        "egy"
      ],
      "notes": "Base model behind NileChat-3B, Egyptian-Arabic-focused; UBC is in Canada.",
      "metrics": {
        "downloads": 36,
        "likes": 1,
        "lastModified": "2026-05-11"
      }
    },
    {
      "id": "padic",
      "name": "PADIC",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/PADIC"
      },
      "year": 2015,
      "dialects": [
        "mixed"
      ],
      "notes": "Parallel Arabic dialect corpus covering Algerian, Tunisian, Moroccan, Syrian, Palestinian cities and MSA.",
      "metrics": {
        "downloads": 36,
        "likes": 2,
        "lastModified": "2022-10-21"
      }
    },
    {
      "id": "terjamabench",
      "name": "TerjamaBench",
      "type": "benchmark",
      "country": "MA",
      "org": "atlasia",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/TerjamaBench"
      },
      "size": "under 1K pairs",
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Small human-curated English-Darija translation benchmark with cultural nuance.",
      "metrics": {
        "downloads": 36,
        "likes": 17,
        "lastModified": "2025-01-19"
      }
    },
    {
      "id": "whisper-large-v3-turbo-arabic",
      "name": "Whisper Large V3 Turbo (Arabic)",
      "type": "asr",
      "country": "INTL",
      "org": "mboushaba",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mboushaba/whisper-large-v3-turbo-arabic"
      },
      "year": 2024,
      "notes": "This model is a fine-tuned version of openai/whisper-large-v3-turbo on the commonvoice110 dataset.",
      "base_model": [
        "openai/whisper-large-v3-turbo"
      ],
      "metrics": {
        "downloads": 36,
        "likes": 2,
        "lastModified": "2024-10-03"
      }
    },
    {
      "id": "wolof-arabic-parallel-corpus",
      "name": "wolof arabic parallel corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "mbaye930",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mbaye930/wolof-arabic-parallel-corpus"
      },
      "dialects": [
        "msa"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "A publicly available parallel corpus for the Wolof–Arabic language pair, a gold-standard resource containing 1,271 sentence-aligned pairs.",
      "metrics": {
        "downloads": 36,
        "likes": 3,
        "lastModified": "2026-06-11"
      }
    },
    {
      "id": "algerian-arabic-english-translation-50k",
      "name": "algerian arabic english translation 50k",
      "type": "dataset",
      "country": "DZ",
      "org": "Touati Kamel",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/touati-kamel/algerian-arabic-english-translation-50k"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "An open-source parallel corpus of 50,000 Algerian Arabic (Darija) sentences paired with their English translations.",
      "metrics": {
        "downloads": 35,
        "likes": 2,
        "lastModified": "2026-08-28"
      }
    },
    {
      "id": "arabic-dialects-to-msa",
      "name": "Arabic Dialects to MSA",
      "type": "dataset",
      "country": "INTL",
      "org": "PRAli22",
      "license": "afl-3.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/PRAli22/Arabic_dialects_to_MSA"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Parallel dialect-MSA corpus",
      "metrics": {
        "downloads": 35,
        "likes": 10,
        "lastModified": "2024-03-01"
      }
    },
    {
      "id": "muradif",
      "name": "Muradif",
      "type": "benchmark",
      "country": "INTL",
      "org": "U4RASD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/U4RASD/Muradif"
      },
      "dialects": [
        "lev",
        "msa"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Muradif (مُرادِف, \"synonym\") is a synonym-based benchmark that directly assesses embedding quality with no additional fine-tuning.",
      "metrics": {
        "downloads": 35,
        "likes": 5,
        "lastModified": "2026-07-07"
      }
    },
    {
      "id": "orpheus-tts-mediaspeech-ar",
      "name": "Orpheus-TTS-MediaSpeech-AR",
      "type": "tts",
      "country": "INTL",
      "org": "kadirnar",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/kadirnar/Orpheus-TTS-MediaSpeech-AR"
      },
      "year": 2025,
      "notes": "Orpheus TTS model fine-tuned on the MediaSpeech dataset for Arabic speech synthesis.",
      "metrics": {
        "downloads": 35,
        "likes": 6,
        "lastModified": "2025-04-09"
      }
    },
    {
      "id": "qa-arabic",
      "name": "QA Arabic",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "chat",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/QA_Arabic"
      },
      "size": "<1K rows",
      "year": 2023,
      "notes": "This JSON file contains a collection of questions and answers in Arabic.",
      "metrics": {
        "downloads": 35,
        "likes": 8,
        "lastModified": "2023-07-16"
      }
    },
    {
      "id": "rewayatech",
      "name": "Rewayatech",
      "type": "dataset",
      "country": "SA",
      "org": "Imam Mohammad Bin Saud University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "github": "https://github.com/aseelad/Rewayatech-Saudi-Stories/",
        "hf": "https://huggingface.co/datasets/arbml/Rewayatech",
        "paper": "https://www.preprints.org/manuscript/202008.0628/v1"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1,267 documents",
      "year": 2020,
      "notes": "A collection of Arabic stories written in electronic forms between the years of 2003-2015 by online users using anonymized usernames",
      "metrics": {
        "downloads": 35,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "stt-ar-fastconformer-hybrid-large-streaming-pcd-v1-1-mirror",
      "name": "stt ar fastconformer hybrid large streaming pcd v1.1 mirror",
      "type": "asr",
      "country": "INTL",
      "org": "Ahmed Hany",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/dev-ahmedhany/stt_ar_fastconformer_hybrid_large_streaming_pcd_v1.1-mirror"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Streaming-trained Arabic FastConformer-Hybrid (RNNT + CTC heads, 115M params), exported for sherpa-onnx's OnlineRecognizer with cache-aware encoder state.",
      "base_model": [
        "nvidia/stt_ar_fastconformer_hybrid_large_pcd_v1.0"
      ],
      "metrics": {
        "downloads": 35,
        "likes": 3,
        "lastModified": "2026-05-10"
      }
    },
    {
      "id": "anetac",
      "name": "ANETAC",
      "type": "dataset",
      "country": "DZ",
      "org": "USTHB / University of Algiers",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/HadjAmeur/ANETAC-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/ANETAC",
        "paper": "https://arxiv.org/pdf/1907.03110.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "79,924 sentences",
      "year": 2020,
      "notes": "English-Arabic named entity transliteration and classification dataset",
      "metrics": {
        "downloads": 34,
        "likes": 0,
        "lastModified": "2024-03-24"
      }
    },
    {
      "id": "arab-esl",
      "name": "Arab-ESL",
      "type": "dataset",
      "country": "SA",
      "org": "Jazan University",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/ShathaHakami/Arabic-Emoji-Sentiment-Lexicon-Version-1.0/blob/main/Arabic_Emoji_Sentiment_Lexicon_Version_1.0.csv",
        "hf": "https://huggingface.co/datasets/arbml/emoji_sentiment_lexicon",
        "paper": "https://aclanthology.org/2021.wanlp-1.7/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,034 tokens",
      "year": 2021,
      "notes": "This is a context-sensitive Arabic sentiment lexicon of 1034 emoji, extracted from 144,196 tweets in existing Arabic datasets.",
      "metrics": {
        "downloads": 34,
        "likes": 5,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arabic-stories-corpus",
      "name": "Arabic Stories Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stories",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Stories_Corpus"
      },
      "size": "<1K rows",
      "year": 2022,
      "notes": "58 Arabic short stories with title, author, publisher and full text.",
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "arabic-wikipedia-talk-pages",
      "name": "Arabic Wikipedia Talk Pages",
      "type": "dataset",
      "country": "INTL",
      "org": "Language Technologies Institute",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "code-switching"
      ],
      "links": {
        "github": "https://github.com/michaelmilleryoder/wikipedia-codeswitching-data/",
        "hf": "https://huggingface.co/datasets/arbml/wikipedia_talks",
        "paper": "https://aclanthology.org/W17-2911.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,259 sentences",
      "year": 2017,
      "notes": "Comments on Arabic Wikipedia Talk pages that have some code-switching to latin.",
      "metrics": {
        "downloads": 34,
        "likes": 0,
        "lastModified": "2024-03-17"
      }
    },
    {
      "id": "arat5-msa-base",
      "name": "AraT5-msa-base",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-to-text"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/AraT5-msa-base"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "tags": [
        "variants:1"
      ],
      "notes": "AraT5-MSA base checkpoint from the AraT5 paper on text-to-text transformers for Arabic generation.",
      "metrics": {
        "downloads": 34,
        "likes": 7,
        "lastModified": "2023-08-16"
      }
    },
    {
      "id": "brad-1-0",
      "name": "BRAD 1.0",
      "type": "dataset",
      "country": "AE",
      "org": "Unversity of Sharjah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "github": "https://github.com/elnagara/BRAD-Arabic-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/BRAD",
        "paper": "https://ieeexplore.ieee.org/abstract/document/7945800"
      },
      "dialects": [
        "mixed"
      ],
      "size": "156,506 sentences",
      "year": 2016,
      "notes": "The reviews were collected from GoodReads.com website during June/July 2016",
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2022-10-14"
      }
    },
    {
      "id": "dz-sentiment-yt-comments",
      "name": "dz sentiment yt comments",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdou",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdou/dz-sentiment-yt-comments"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "This dataset consists of 50,016 samples of comments extracted from Algerian YouTube channels.",
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2023-11-06"
      }
    },
    {
      "id": "egyptian-arabic-english-v1",
      "name": "Egyptian Arabic English V1",
      "type": "dataset",
      "country": "INTL",
      "org": "Amr-khaled",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Amr-khaled/Egyptian-Arabic_English_V1"
      },
      "dialects": [
        "egy"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "33K Egyptian Arabic-English sentence pairs with source text field.",
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2024-12-09"
      }
    },
    {
      "id": "english-to-darija-2",
      "name": "english to darija 2",
      "type": "llm",
      "country": "INTL",
      "org": "Youssef Chafiqui",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/english-to-darija-2"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Helsinki-NLP model fine-tuned for English-to-Darija translation.",
      "base_model": [
        "helsinki-nlp/opus-mt-tc-big-en-ar"
      ],
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2024-02-12"
      }
    },
    {
      "id": "financial-reasoning-qa-arabic-dataset",
      "name": "Financial Reasoning QA Arabic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Gheras",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Gheras/Financial_Reasoning_QA_Arabic_Dataset"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset is a professionally translated Arabic version of the original English FinQA dataset.",
      "metrics": {
        "downloads": 34,
        "likes": 3,
        "lastModified": "2025-08-27"
      }
    },
    {
      "id": "sdnlp-llama3-1-syrian-lora",
      "name": "sdnlp-llama3.1-syrian-lora",
      "type": "llm",
      "country": "INTL",
      "org": "hasankh (SDNLP)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/hasankh/sdnlp-llama3.1-syrian-lora",
        "paper": "https://aclanthology.org/2026.vardial-1.29/"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "Llama-3.1 LoRA adapter for Syrian Arabic dialect generation (AMIYA 2026).",
      "metrics": {
        "downloads": 34,
        "likes": 0,
        "lastModified": "2026-01-13"
      }
    },
    {
      "id": "arabic-osact5-arabic-hate-speech",
      "name": "Arabic OSACT5 : Arabic Hate Speech",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "other",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "website": "https://codalab.lisn.upsaclay.fr/competitions/2324",
        "hf": "https://huggingface.co/datasets/arbml/osact5_hatespeech",
        "paper": "https://arxiv.org/pdf/2201.06723.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,157 sentences",
      "year": 2022,
      "notes": "Fine-Grained Hate Speech Detection on Arabic Twitter",
      "metrics": {
        "downloads": 33,
        "likes": 0,
        "lastModified": "2024-05-13"
      }
    },
    {
      "id": "arabic-speech-massive",
      "name": "arabic speech massive",
      "type": "dataset",
      "country": "INTL",
      "org": "abdusah",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdusah/arabic_speech_massive"
      },
      "size": "100K–1M rows",
      "year": 2022,
      "notes": "333K-row Arabic speech set with transcriptions, labels, country and dialect columns.",
      "metrics": {
        "downloads": 33,
        "likes": 4,
        "lastModified": "2022-03-18"
      }
    },
    {
      "id": "arabic-moroccan-darija-code-switched",
      "name": "Arabic-Moroccan Darija Code-Switched",
      "type": "dataset",
      "country": "INTL",
      "org": "Institute for Language and Information",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/samihyounes/Moroccan-Codeswitching",
        "paper": "https://aclanthology.org/L16-1658.pdf"
      },
      "dialects": [
        "mixed",
        "magh",
        "msa"
      ],
      "size": "222,881 tokens",
      "year": 2016,
      "notes": "Arabic-Moroccan Darija Code-Switched Corpus from blogs and forums.",
      "metrics": {
        "downloads": 33,
        "likes": 0,
        "lastModified": "2026-03-06"
      }
    },
    {
      "id": "argilla-dpo-mix-7k-arabic",
      "name": "argilla dpo mix 7k arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "preference",
        "dpo"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/argilla-dpo-mix-7k-arabic"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic version of the Argilla DPO mix with 7,500 chosen/rejected response pairs and ratings.",
      "metrics": {
        "downloads": 33,
        "likes": 10,
        "lastModified": "2024-03-18"
      }
    },
    {
      "id": "aya-acegpt-13b-chat-dpo",
      "name": "Aya-AceGPT.13B.Chat-DPO",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Aya-AceGPT.13B.Chat-DPO"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Aya-AceGPT.13B.Chat-DPO is a DPO dataset de",
      "metrics": {
        "downloads": 33,
        "likes": 2,
        "lastModified": "2024-05-16"
      }
    },
    {
      "id": "jameamt-ar-en-350m",
      "name": "JameaMT-ar-en-350M",
      "type": "llm",
      "country": "INTL",
      "org": "ramyibrahim",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/ramyibrahim/JameaMT-ar-en-350M"
      },
      "size": "350M",
      "on_device": true,
      "year": 2026,
      "notes": "Gmail | LinkedIn | GitHub | Hugging Face Author : Ramy I.",
      "base_model": [
        "liquidai/lfm2.5-350m"
      ],
      "metrics": {
        "downloads": 33,
        "likes": 3,
        "lastModified": "2026-06-08"
      }
    },
    {
      "id": "moroccan-arabic-wikipedia-20230101-bots",
      "name": "Moroccan_Arabic_Wikipedia_20230101_bots",
      "type": "dataset",
      "country": "INTL",
      "org": "Clarkson University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/Moroccan_Arabic_Wikipedia_20230101_bots",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.19.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "5,400 documents",
      "year": 2023,
      "notes": "MoroccanArabicWikipedia20230101bots is a dataset created using the Moroccan Arabic Wikipedia articles, including the bot-generated articles, downloaded.",
      "metrics": {
        "downloads": 33,
        "likes": 0,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "moroccanhistory-qa-darija-dataset",
      "name": "MoroccanHistory-QA-Darija-Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "JasperV13",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/JasperV13/MoroccanHistory-QA-Darija-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "This dataset is a translated subset from KBayoud/MoroccanHistory-QA-Dataset.",
      "metrics": {
        "downloads": 33,
        "likes": 4,
        "lastModified": "2024-09-03"
      }
    },
    {
      "id": "multi-tafseer-quran-rag",
      "name": "multi tafseer quran rag",
      "type": "dataset",
      "country": "INTL",
      "org": "omaressam1111",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "embedding",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/omaressam1111/multi-tafseer-quran-rag"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "A structured Arabic dataset of Quranic tafseer collected from eight classical and modern tafseer books.",
      "metrics": {
        "downloads": 33,
        "likes": 7,
        "lastModified": "2026-05-10"
      }
    },
    {
      "id": "namaa-mt-saudi2english",
      "name": "NAMAA MT Saudi2English",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/NAMAA-MT-Saudi2English"
      },
      "dialects": [
        "gulf",
        "mixed"
      ],
      "year": 2025,
      "notes": "It is built on top of mBERT-initialized T5 architecture and aims to improve translation quality for informal and region-specific Arabic commonly used.",
      "metrics": {
        "downloads": 33,
        "likes": 13,
        "lastModified": "2025-11-13"
      }
    },
    {
      "id": "sad",
      "name": "SAD",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Stirling",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/SAD",
        "paper": "https://link.springer.com/content/pdf/10.1007%2F978-3-319-11179-7_29.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6 hours",
      "year": 2014,
      "notes": "The Arabic speech corpus for isolated words contains 9992 utterances of 20 words spoken by 50 native male Arabic speakers.",
      "metrics": {
        "downloads": 33,
        "likes": 0,
        "lastModified": "2022-11-01"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-arabic",
      "name": "wav2vec2 large xlsr arabic",
      "type": "asr",
      "country": "INTL",
      "org": "mohammed",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mohammed/wav2vec2-large-xlsr-arabic"
      },
      "year": 2024,
      "notes": "Fine-tuned facebook/wav2vec2-large-xlsr-53 on Arabic using the train splits of Common Voice and Arabic Speech Corpus.",
      "base_model": [
        "facebook/wav2vec2-large-xlsr-53"
      ],
      "metrics": {
        "downloads": 33,
        "likes": 4,
        "lastModified": "2024-07-22"
      }
    },
    {
      "id": "ara-emotion-naacl2018",
      "name": "ara_emotion_naacl2018",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "emotion"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/ara_emotion_naacl2018",
        "hf": "https://huggingface.co/datasets/arbml/ara_emotion"
      },
      "year": 2018,
      "notes": "This repository provides our datasets for Arabic emotion detection in Twitter",
      "metrics": {
        "downloads": 32,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arab-states-analogy-dataset-asad",
      "name": "Arab States Analogy Dataset (ASAD)",
      "type": "benchmark",
      "country": "INTL",
      "org": "Clarkson University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation",
        "embedding-evaluation"
      ],
      "links": {
        "github": "https://github.com/SaiedAlshahrani/performance-implications/tree/main/Word-Representation-Evals/ASAD",
        "hf": "https://huggingface.co/datasets/SaiedAlshahrani/ASAD",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.19.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,520 sentences",
      "year": 2023,
      "notes": "ASAD is a word analogy dataset created using 20 Arab States with their corresponding capital cities, nationalities, currencies, and on which continents.",
      "metrics": {
        "downloads": 32,
        "likes": 1,
        "lastModified": "2024-01-05"
      }
    },
    {
      "id": "arabicqa-2-1m",
      "name": "ArabicQA_2.1M",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/ArabicQA_2.1M"
      },
      "size": "1M–10M rows",
      "year": 2024,
      "notes": "Our dataset is an amalgamation of several filtered datasets, the total number of rows for all datasets was 4,731,600 which was reduced.",
      "metrics": {
        "downloads": 32,
        "likes": 6,
        "lastModified": "2024-08-04"
      }
    },
    {
      "id": "aracode-7b-full",
      "name": "AraCode-7B-Full",
      "type": "llm",
      "country": "INTL",
      "org": "rahimdzx",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/rahimdzx/AraCode-7B-Full"
      },
      "size": "7B",
      "on_device": false,
      "year": 2026,
      "notes": "AraCode-7B understands, explains, and generates code in Arabic — a capability no existing model provides with such precision.",
      "metrics": {
        "downloads": 32,
        "likes": 8,
        "lastModified": "2026-04-06"
      }
    },
    {
      "id": "bee1reason-arabic-qwen-14b",
      "name": "Bee1reason arabic Qwen 14B",
      "type": "llm",
      "country": "INTL",
      "org": "beetleware",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/beetleware/Bee1reason-arabic-Qwen-14B"
      },
      "base_model": [
        "unsloth/qwen3-14b"
      ],
      "size": "14B",
      "on_device": false,
      "year": 2025,
      "notes": "Qwen3-14B fine-tuned for Arabic logical reasoning on an Arabic reasoning dataset.",
      "metrics": {
        "downloads": 32,
        "likes": 16,
        "lastModified": "2025-05-31"
      }
    },
    {
      "id": "bimedix-ara",
      "name": "BiMediX-Ara",
      "type": "llm",
      "country": "INTL",
      "org": "BiMediX",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "medical",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/BiMediX/BiMediX-Ara"
      },
      "year": 2024,
      "notes": "Bilingual medical Mixtral-8x7B MoE trained on Arabic-only BiMed1.3M-Arabic for MCQA, closed QA and chat.",
      "metrics": {
        "downloads": 32,
        "likes": 6,
        "lastModified": "2024-02-26"
      }
    },
    {
      "id": "medical-reasoning-arabic-dataset",
      "name": "Medical Reasoning Arabic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Gheras",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "medical",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Gheras/Medical_Reasoning_Arabic_Dataset"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "19.7K Arabic medical reasoning rows with question, chain-of-thought and response.",
      "metrics": {
        "downloads": 32,
        "likes": 4,
        "lastModified": "2025-09-09"
      }
    },
    {
      "id": "moroccansocialmedia-multigen",
      "name": "MoroccanSocialMedia-MultiGen",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "['mit']",
      "modality": "text",
      "tasks": [
        "text-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/MoroccanSocialMedia-MultiGen"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "MoroccanSocialMedia-MultiGen: 12,973 pairs of native Darija social-media posts and their synthetic counterparts.",
      "metrics": {
        "downloads": 32,
        "likes": 4,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "pearl-vdr-ar-train-hard-mined",
      "name": "Pearl vdr ar train hard mined",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "embedding",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Pearl-vdr-ar-train-hard-mined"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Arabic culturally-aligned Visual Document Retrieval (VDR) triplets with model-mined hard negatives.",
      "metrics": {
        "downloads": 32,
        "likes": 2,
        "lastModified": "2026-04-22"
      }
    },
    {
      "id": "sambalingo-arabic-chat-70b",
      "name": "SambaLingo-Arabic-Chat-70B",
      "type": "llm",
      "country": "INTL",
      "org": "SambaNova",
      "license": "llama2",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/sambanovasystems/SambaLingo-Arabic-Chat-70B"
      },
      "size": "70B",
      "on_device": false,
      "year": 2024,
      "notes": "SambaLingo-Arabic-Chat-70B is a human aligned chat model trained in Arabic and English.",
      "metrics": {
        "downloads": 32,
        "likes": 3,
        "lastModified": "2024-05-14"
      }
    },
    {
      "id": "speecht5-tts-arabic",
      "name": "speecht5 tts Arabic",
      "type": "tts",
      "country": "INTL",
      "org": "Reyouf",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/Reyouf/speecht5_tts_Arabic"
      },
      "year": 2024,
      "notes": "This model is a fine-tuned version of microsoft/speecht5tts on the Hakawati dataset.",
      "base_model": [
        "microsoft/speecht5_tts"
      ],
      "metrics": {
        "downloads": 32,
        "likes": 3,
        "lastModified": "2024-04-06"
      }
    },
    {
      "id": "whisper-small-quran-lora-everyayah",
      "name": "whisper small quran lora everyayah",
      "type": "asr",
      "country": "INTL",
      "org": "MaddoggProduction",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/MaddoggProduction/whisper-small-quran-lora-everyayah"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "This is a specialized Automatic Speech Recognition (ASR) model for Quranic Recitation with tashkeel or diacritics.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 32,
        "likes": 3,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "whisper-sudanese-dialect-small",
      "name": "Whisper-Sudanese-Dialect-small",
      "type": "asr",
      "country": "SD",
      "org": "Ayman Mansour",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/AymanMansour/Whisper-Sudanese-Dialect-small"
      },
      "year": 2022,
      "on_device": true,
      "dialects": [
        "sudanese"
      ],
      "notes": "Whisper small fine-tuned on Sudanese dialect; Sudan noted.",
      "metrics": {
        "downloads": 32,
        "likes": 2,
        "lastModified": "2022-12-15"
      }
    },
    {
      "id": "evalplus-arabic",
      "name": "3LM Code Arabic (EvalPlus)",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "code"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tiiuae/evalplus-arabic"
      },
      "year": 2025,
      "notes": "Arabic translations of HumanEval+ and MBPP+ from the 3LM benchmark suite.",
      "metrics": {
        "downloads": 31,
        "likes": 1,
        "lastModified": "2026-02-14"
      }
    },
    {
      "id": "al-baka-llama3-8b-experimental",
      "name": "al baka llama3 8b experimental",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/al-baka-llama3-8b-experimental"
      },
      "size": "8B",
      "on_device": false,
      "year": 2024,
      "notes": "Al Baka is an Experimental Fine Tuned Model based on the new released LLAMA3-8B Model on the Stanford Alpaca dataset Arabic version Yasbok/Alpacaarabicinstruct.",
      "metrics": {
        "downloads": 31,
        "likes": 2,
        "lastModified": "2024-04-20"
      }
    },
    {
      "id": "arabglossbert",
      "name": "ArabGlossBERT",
      "type": "llm",
      "country": "PS",
      "org": "SinaLab",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/SinaLab/ArabGlossBERT"
      },
      "year": 2022,
      "notes": "BERT model for Arabic word sense disambiguation via context-gloss pairs.",
      "metrics": {
        "downloads": 31,
        "likes": 0,
        "lastModified": "2026-01-03"
      }
    },
    {
      "id": "l-hsab",
      "name": "L-HSAB",
      "type": "dataset",
      "country": "INTL",
      "org": "iCompass",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/Hala-Mulki/L-HSAB-First-Arabic-Levantine-HateSpeech-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/L_HSAB",
        "paper": "https://aclanthology.org/W19-3512.pdf"
      },
      "dialects": [
        "lev"
      ],
      "size": "5,851 sentences",
      "year": 2019,
      "notes": "Arabic Levantine Hate Speech and Abusive Language Dataset",
      "metrics": {
        "downloads": 31,
        "likes": 0,
        "lastModified": "2022-10-21"
      }
    },
    {
      "id": "moroccan-darija-wikipedia-dataset",
      "name": "moroccan darija wikipedia dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AbderrahmanSkiredj1/moroccan_darija_wikipedia_dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "4,862 Moroccan Darija Wikipedia text rows.",
      "metrics": {
        "downloads": 31,
        "likes": 6,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "quran-embeddings",
      "name": "quran embeddings",
      "type": "dataset",
      "country": "INTL",
      "org": "beardaintweird",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/beardaintweird/quran-embeddings"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "18.7K Quran verses with surah, ayah, text and precomputed embeddings.",
      "metrics": {
        "downloads": 31,
        "likes": 6,
        "lastModified": "2023-01-02"
      }
    },
    {
      "id": "wav2vec2-large-xlsr-moroccan",
      "name": "wav2vec2 large xlsr moroccan",
      "type": "asr",
      "country": "INTL",
      "org": "othrif",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/othrif/wav2vec2-large-xlsr-moroccan"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Fine-tuned facebook/wav2vec2-large-xlsr-53 on MGB5 Moroccan Arabic kindly provided by ELDA and ArabicSpeech.",
      "metrics": {
        "downloads": 31,
        "likes": 4,
        "lastModified": "2025-01-15"
      }
    },
    {
      "id": "whisper-small-yemeni",
      "name": "whisper-small-yemeni",
      "type": "asr",
      "country": "INTL",
      "org": "tahaalselwii",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/tahaalselwii/whisper-small-yemeni"
      },
      "dialects": [
        "yemeni"
      ],
      "year": 2026,
      "notes": "Whisper-small fine-tuned for Yemeni Arabic speech recognition.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 31,
        "likes": 1,
        "lastModified": "2026-07-24"
      }
    },
    {
      "id": "amawal-dataset",
      "name": "amawal dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "prothmane",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/prothmane/amawal-dataset"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "A trilingual Amazigh-French-Arabic lexicon to support AI research for Berber languages.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2025-12-17"
      }
    },
    {
      "id": "aqeedah-rag-dataset",
      "name": "aqeedah rag dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "abdullah-alamodi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdullah-alamodi/aqeedah-rag-dataset"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "A curated Arabic Islamic theology (Aqeedah) dataset with pre-computed FAISS embeddings, designed for advanced Retrieval-Augmented Generation.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2025-10-31"
      }
    },
    {
      "id": "arabic-ecom-data",
      "name": "arabic ecom data",
      "type": "dataset",
      "country": "INTL",
      "org": "Presto AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "retrieval",
        "e-commerce"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/prestoai/arabic-ecom-data"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Arabic e-commerce query-product training data for retrieval and embedding models, in MSA and Libyan dialect.",
      "dialects": [
        "mixed"
      ],
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2026-07-09"
      }
    },
    {
      "id": "arabic-relation-extraction",
      "name": "arabic relation extraction",
      "type": "llm",
      "country": "INTL",
      "org": "ychenNLP",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "relation-extraction"
      ],
      "links": {
        "hf": "https://huggingface.co/ychenNLP/arabic-relation-extraction"
      },
      "year": 2022,
      "notes": "Arabic relation extraction model built on GigaBERTv4 that marks the two entities in a sentence.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2022-07-10"
      }
    },
    {
      "id": "arabic-speech-syllables-recognition-using-wav2vec2",
      "name": "Arabic speech Syllables recognition Using Wav2vec2",
      "type": "asr",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimSalah/Arabic_speech_Syllables_recognition_Using_Wav2vec2"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "This is fine tuned wav2vec2 model to recognize arabic syllables from speech.",
      "metrics": {
        "downloads": 30,
        "likes": 4,
        "lastModified": "2025-01-09"
      }
    },
    {
      "id": "arablegaleval",
      "name": "ArabLegalEval",
      "type": "benchmark",
      "country": "INTL",
      "org": "THIQAH-RD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/THIQAH-RD/ArabLegalEval"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "ArabLegalEval: benchmark for LLMs on Arabic legal tasks, including ArLegalBench, with prompt templates per task.",
      "metrics": {
        "downloads": 30,
        "likes": 4,
        "lastModified": "2025-01-02"
      }
    },
    {
      "id": "atar",
      "name": "ATAR",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "apache-1.0",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/bashartalafha/Arabizi-Transliteration",
        "hf": "https://huggingface.co/datasets/arbml/Arabizi_Transliteration",
        "paper": "http://ijece.iaescore.com/index.php/IJECE/article/view/22767/14781"
      },
      "dialects": [
        "mixed"
      ],
      "size": "2,743 tokens",
      "year": 2021,
      "notes": "The first large-scale 'Arabizi to Arabic script' parallel corpus focusing on the Jordanian dialect and consisting of more than 25k pairs carefully created.",
      "metrics": {
        "downloads": 30,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "atlasocrbench",
      "name": "AtlasOCRBench",
      "type": "benchmark",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/AtlasOCRBench"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Evaluation benchmark for Moroccan Darija OCR.",
      "metrics": {
        "downloads": 30,
        "likes": 2,
        "lastModified": "2025-09-16"
      }
    },
    {
      "id": "fasih-2b",
      "name": "Fasih 2B",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/Fasih-2B"
      },
      "size": "2B",
      "on_device": true,
      "year": 2026,
      "notes": "A 2.09 billion parameter Arabic language model trained from scratch using the nanochat framework.",
      "metrics": {
        "downloads": 30,
        "likes": 9,
        "lastModified": "2026-03-08"
      }
    },
    {
      "id": "islamic-historical-corpus",
      "name": "islamic historical corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "IslamStories",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa",
        "translation",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IslamStories/islamic-historical-corpus"
      },
      "dialects": [
        "classical"
      ],
      "size": "<1K rows",
      "year": 2026,
      "notes": "The public source manifest for the classified, AI-ready Islamic historical knowledge base.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2026-04-19"
      }
    },
    {
      "id": "llama-3-instruct-slerp-arabic",
      "name": "llama-3-instruct-slerp-arabic",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "llama3",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/llama-3-instruct-slerp-arabic"
      },
      "size": "8B",
      "on_device": false,
      "year": 2024,
      "notes": "Mergekit SLERP merge of Meta-Llama-3-8B-Instruct and HeshamHaroon/Egy_llama3 for Arabic chat.",
      "base_model": [
        "meta-llama/meta-llama-3-8b-instruct",
        "heshamharoon/egy_llama3"
      ],
      "metrics": {
        "downloads": 30,
        "likes": 1,
        "lastModified": "2024-06-11"
      }
    },
    {
      "id": "llamalens",
      "name": "LlamaLens",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/LlamaLens"
      },
      "base_model": [
        "meta-llama/llama-3.1-8b-instruct"
      ],
      "year": 2025,
      "notes": "Multilingual news and social-media analysis LLM covering Arabic, from QCRI.",
      "metrics": {
        "downloads": 30,
        "likes": 2,
        "lastModified": "2026-04-23"
      }
    },
    {
      "id": "qwen3-vl-embedding-arabic-vdr",
      "name": "Qwen3-VL-Embedding-2B Arabic VDR",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "embedding",
        "visual-document-retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Qwen3-VL-Embedding-2B-Arabic-VDR"
      },
      "size": "2B",
      "year": 2026,
      "notes": "Arabic visual-document retrieval embedding model with Matryoshka loss.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2026-04-28"
      }
    },
    {
      "id": "shahin-v0-1",
      "name": "Shahin v0.1",
      "type": "llm",
      "country": "INTL",
      "org": "malhajar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/malhajar/Shahin-v0.1"
      },
      "year": 2024,
      "notes": "Arabic LLM tuned for fluent Syrian dialect, covering dialogue generation and cultural analysis.",
      "dialects": [
        "lev"
      ],
      "metrics": {
        "downloads": 30,
        "likes": 6,
        "lastModified": "2024-12-09"
      }
    },
    {
      "id": "tead",
      "name": "TEAD",
      "type": "dataset",
      "country": "TN",
      "org": "National Higher Engineering School of Tunis",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/HSMAabdellaoui/TEAD",
        "hf": "https://huggingface.co/datasets/arbml/TEAD",
        "paper": "https://www.researchgate.net/publication/328105014_Using_Tweets_and_Emojis_to_Build_TEAD_an_Arabic_Dataset_for_Sentiment_Analysis"
      },
      "dialects": [
        "mixed",
        "magh",
        "egy",
        "gulf",
        "iraqi",
        "lev"
      ],
      "size": "6,027,210 sentences",
      "year": 2018,
      "notes": "Dataset for Arabic Sentiment Analysis",
      "metrics": {
        "downloads": 30,
        "likes": 1,
        "lastModified": "2024-03-24"
      }
    },
    {
      "id": "tulu-v2-sft-mixture-role-split-english-darija",
      "name": "tulu v2 sft mixture role split english darija",
      "type": "dataset",
      "country": "INTL",
      "org": "yousef-khoubrane",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/yousef-khoubrane/tulu-v2-sft-mixture-role-split-english-darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset is derived from Tulu-v2-SFT-Mixture-English-Darija.",
      "metrics": {
        "downloads": 30,
        "likes": 3,
        "lastModified": "2025-01-16"
      }
    },
    {
      "id": "arabic-khatt",
      "name": "arabic khatt",
      "type": "dataset",
      "country": "INTL",
      "org": "ahmedheakl",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "vqa",
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ahmedheakl/arabic_khatt"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "1,400 rows pairing images with Arabic questions and answers (Arabic Khatt).",
      "metrics": {
        "downloads": 29,
        "likes": 3,
        "lastModified": "2024-10-29"
      }
    },
    {
      "id": "arabic-literature",
      "name": "Arabic Literature",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Literature"
      },
      "size": "1M–10M rows",
      "year": 2022,
      "notes": "1.6M rows of Arabic literature text for language modelling.",
      "metrics": {
        "downloads": 29,
        "likes": 3,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "arabic-natural-questions",
      "name": "Arabic natural questions",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-natural-questions"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset is an Arabic-translated version of the original Natural Questions dataset by sentence-transformers.",
      "metrics": {
        "downloads": 29,
        "likes": 2,
        "lastModified": "2025-07-02"
      }
    },
    {
      "id": "arasum-corpus",
      "name": "AraSum Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Pázmány Péter Catholic University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "github": "https://github.com/ppke-nlpg/AraSum",
        "hf": "https://huggingface.co/datasets/arbml/AraSum",
        "paper": "https://aclanthology.org/2021.ranlp-1.74/"
      },
      "dialects": [
        "msa"
      ],
      "size": "49,604 sentences",
      "year": 2022,
      "notes": "AraSum corpus is a monolingual corpus for abstractive text summarization for the Arabic language.",
      "metrics": {
        "downloads": 29,
        "likes": 0,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "arqat-aqi-answerable-question-identification-in-arabic-tweets",
      "name": "ArQAT-AQI: Answerable Question Identification in Arabic Tweets",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "answerable-questions-detection"
      ],
      "links": {
        "website": "https://www.dropbox.com/sh/coba3b1nqkyloa8/AAC4Sk5WQvtXZRgH5liBkMiGa?dl=0",
        "hf": "https://huggingface.co/datasets/arbml/AQweets",
        "paper": "https://dl.acm.org/doi/pdf/10.1145/2661829.2661959"
      },
      "dialects": [
        "mixed"
      ],
      "size": "13,252 sentences",
      "year": 2017,
      "notes": "Answerable Question Identification in Arabic Tweets",
      "metrics": {
        "downloads": 29,
        "likes": 0,
        "lastModified": "2024-05-10"
      }
    },
    {
      "id": "escwa",
      "name": "ESCWA",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/escwa",
        "paper": "https://doi.org/10.21437/Interspeech.2021-2231"
      },
      "dialects": [
        "mixed"
      ],
      "size": "2 hours",
      "year": 2021,
      "tags": [
        "multilingual"
      ],
      "notes": "Eight-hour multilingual speech corpus collected from two days of United Nations ESCWA meetings in 2019, focusing on intrasentential Arabic-English.",
      "metrics": {
        "downloads": 29,
        "likes": 0,
        "lastModified": "2025-10-13"
      }
    },
    {
      "id": "misraj-msdd",
      "name": "Misraj MSDD",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/msdd"
      },
      "year": 2025,
      "notes": "Large Arabic multimodal web dataset from Common Crawl, 10M to 100M rows; gated.",
      "metrics": {
        "downloads": 29,
        "likes": 5,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "mistral-7b-v0-1-arabic",
      "name": "Mistral 7B v0.1 arabic",
      "type": "llm",
      "country": "SY",
      "org": "malhajar",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/malhajar/Mistral-7B-v0.1-arabic"
      },
      "base_model": [
        "mistralai/mistral-7b-v0.1"
      ],
      "size": "7B",
      "on_device": false,
      "year": 2024,
      "notes": "malhajar/Mistral-7B-Instruct-v0.2-turkish is a finetuned version of Mistral-7B-v0.1 using SFT Training and Freeze method.",
      "metrics": {
        "downloads": 29,
        "likes": 9,
        "lastModified": "2024-01-19"
      }
    },
    {
      "id": "religious-hate-speech",
      "name": "Religious Hate Speech",
      "type": "dataset",
      "country": "SA",
      "org": "Taibah University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/nuhaalbadi/Arabic_hatespeech",
        "hf": "https://huggingface.co/datasets/arbml/Religious_Hate_Speech",
        "paper": "https://ieeexplore.ieee.org/document/8508247"
      },
      "dialects": [
        "mixed"
      ],
      "size": "6,136 sentences",
      "year": 2018,
      "notes": "Training dataset contains 5,569 examples, while the testing dataset contains 567 examples collected from twittter",
      "metrics": {
        "downloads": 29,
        "likes": 0,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "saudidialect-triplet-21",
      "name": "SaudiDialect-Triplet-21",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/SaudiDialect-Triplet-21"
      },
      "size": "1K-10K rows",
      "year": 2025,
      "dialects": [
        "gulf"
      ],
      "notes": "Saudi dialect triplets for embedding fine-tuning, gated.",
      "metrics": {
        "downloads": 29,
        "likes": 4,
        "lastModified": "2025-11-29"
      }
    },
    {
      "id": "summarization-arabic-english-news",
      "name": "summarization arabic english news",
      "type": "llm",
      "country": "INTL",
      "org": "marefa-nlp",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/marefa-nlp/summarization-arabic-english-news"
      },
      "year": 2021,
      "notes": "Summarizes Arabic and English news stories into short highlights.",
      "metrics": {
        "downloads": 29,
        "likes": 4,
        "lastModified": "2021-07-13"
      }
    },
    {
      "id": "whisper-medium-arabic",
      "name": "whisper medium arabic",
      "type": "asr",
      "country": "INTL",
      "org": "Seyfelislem",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/Seyfelislem/whisper-medium-arabic"
      },
      "year": 2025,
      "notes": "Arabic automatic speech recognition model fine-tuned from openai/whisper-medium.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 29,
        "likes": 7,
        "lastModified": "2025-06-22"
      }
    },
    {
      "id": "ans-corpus-claim-verification",
      "name": "ANS CORPUS:  claim verification",
      "type": "dataset",
      "country": "INTL",
      "org": "Latynt",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "stance-detection",
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/latynt/ans",
        "hf": "https://huggingface.co/datasets/arbml/ANS_stance",
        "paper": "https://arxiv.org/pdf/2005.10410.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,786 sentences",
      "year": 2020,
      "notes": "Corpus comes in two perspectives: a version consisting of 4,547 true and false claims and a version consisting of 3,786 pairs (claim, evidence).",
      "metrics": {
        "downloads": 28,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "arabench",
      "name": "AraBench",
      "type": "benchmark",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/AraBench_test"
      },
      "year": 2020,
      "dialects": [
        "mixed"
      ],
      "notes": "Test sets of the AraBench dialectal Arabic to English machine translation benchmark.",
      "metrics": {
        "downloads": 28,
        "likes": 0,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "arabic-bbc-news",
      "name": "arabic bbc news",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdelkareem",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "news",
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Abdelkareem/arabic-bbc-news"
      },
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "9,380 Arabic BBC News articles with URL, title, summary and full text.",
      "metrics": {
        "downloads": 28,
        "likes": 3,
        "lastModified": "2023-06-21"
      }
    },
    {
      "id": "arabic-english-named-entities-dataset",
      "name": "Arabic-ENglish named entities dataset",
      "type": "dataset",
      "country": "TN",
      "org": "Tunisian university",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "ner"
      ],
      "links": {
        "github": "https://github.com/Hkiri-Emna/Named_Entities_Lexicon_Project",
        "hf": "https://huggingface.co/datasets/arbml/Named_Entities_Lexicon",
        "paper": "http://iajit.org/PDF//vol.%2014,%20no%206/10491.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "48,753 tokens",
      "year": 2017,
      "tags": [
        "multilingual"
      ],
      "notes": "Arabic-ENglish named entities dataset is created using DBpedia Linked datasets and parallel corpus.",
      "metrics": {
        "downloads": 28,
        "likes": 1,
        "lastModified": "2024-07-20"
      }
    },
    {
      "id": "arazn-whisper-small",
      "name": "arazn whisper small",
      "type": "asr",
      "country": "INTL",
      "org": "ahmedheakl",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/ahmedheakl/arazn-whisper-small",
        "paper": "https://arxiv.org/abs/2406.18120"
      },
      "year": 2024,
      "notes": "Whisper-small fine-tuned for code-switched Egyptian Arabic-English speech recognition (ArzEn-LLM).",
      "dialects": [
        "egy"
      ],
      "on_device": true,
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 28,
        "likes": 3,
        "lastModified": "2024-07-01"
      }
    },
    {
      "id": "dart",
      "name": "DART",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://www.dropbox.com/s/jslg6fzxeu47flu/DART.zip?dl=0",
        "hf": "https://huggingface.co/datasets/arbml/DART",
        "paper": "https://aclanthology.org/L18-1579.pdf"
      },
      "dialects": [
        "mixed",
        "egy",
        "gulf",
        "iraqi",
        "lev",
        "magh"
      ],
      "size": "24,280 sentences",
      "year": 2018,
      "notes": "Collection of dialectal Arabic tweets labeled by dialect region (Egyptian, Gulf, Iraqi, Levantine, Maghrebi).",
      "metrics": {
        "downloads": 28,
        "likes": 0,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "fiqh-maliki-talqin",
      "name": "fiqh maliki talqin",
      "type": "dataset",
      "country": "INTL",
      "org": "islamic-datasets",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/islamic-datasets/fiqh-maliki-talqin"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Digitized text of Al-Talqin, a concise classical manual of Maliki jurisprudence, as a structured dataset.",
      "metrics": {
        "downloads": 28,
        "likes": 3,
        "lastModified": "2025-12-31"
      }
    },
    {
      "id": "llmvox",
      "name": "LLMVoX",
      "type": "tts",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/LLMVoX"
      },
      "notes": "30M streaming TTS for any LLM, first Arabic autoregressive TTS",
      "metrics": {
        "downloads": 28,
        "likes": 59,
        "lastModified": "2025-03-16"
      }
    },
    {
      "id": "msac",
      "name": "MSAC",
      "type": "dataset",
      "country": "MA",
      "org": "Ibn Tofail University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/ososs/Arabic-Sentiment-Analysis-corpus",
        "hf": "https://huggingface.co/datasets/arbml/MSAC",
        "paper": "https://dl.acm.org/doi/abs/10.1177/0165551519849516"
      },
      "dialects": [
        "magh"
      ],
      "size": "2,000 sentences",
      "year": 2020,
      "notes": "Rich and publicly available Arabic corpus called Moroccan Sentiment Analysis Corpus (MSAC)",
      "metrics": {
        "downloads": 28,
        "likes": 0,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "qwen2-5-7b-jordanian",
      "name": "Qwen2.5-7B Jordanian",
      "type": "llm",
      "country": "JO",
      "org": "Kareem Bb",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/KareemBb/Qwen2.5-7B-Instruct-Jordanian"
      },
      "base_model": [
        "qwen/qwen2.5-7b-instruct"
      ],
      "dialects": [
        "lev"
      ],
      "notes": "Qwen2.5-7B-Instruct fine-tuned for Jordanian dialect chat.",
      "metrics": {
        "downloads": 28,
        "likes": 2,
        "lastModified": "2026-05-24"
      }
    },
    {
      "id": "translate-en-ar-v1-0-hplt-opus",
      "name": "translate en ar v1.0 hplt opus",
      "type": "llm",
      "country": "INTL",
      "org": "HPLT",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/HPLT/translate-en-ar-v1.0-hplt_opus"
      },
      "year": 2024,
      "notes": "This repository contains the translation model for English-Arabic trained with OPUS and HPLT data.",
      "metrics": {
        "downloads": 28,
        "likes": 3,
        "lastModified": "2024-03-14"
      }
    },
    {
      "id": "whisper-base-arabic",
      "name": "Whisper base Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "YazanSalameh",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/YazanSalameh/Whisper-base-Arabic"
      },
      "year": 2024,
      "notes": "Whisper-base fine-tuned for Arabic ASR on Common Voice 16, MGB-2 and Jordanian audio (WER 34.7).",
      "base_model": [
        "openai/whisper-base"
      ],
      "metrics": {
        "downloads": 28,
        "likes": 4,
        "lastModified": "2024-02-25"
      }
    },
    {
      "id": "arabic-finanical-rag-embedding-dataset",
      "name": "Arabic finanical rag embedding dataset",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "retrieval",
        "finance"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-finanical-rag-embedding-dataset"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "7,000 question-context pairs translated into Arabic from NVIDIA's 2023 SEC filing, for fine-tuning RAG embedding models.",
      "metrics": {
        "downloads": 27,
        "likes": 6,
        "lastModified": "2024-10-09"
      }
    },
    {
      "id": "arabic-news-articles-from-aljazeera-net",
      "name": "Arabic News articles from Aljazeera.net",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "news"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/arhouati/arabic-news-articles-from-aljazeeranet",
        "hf": "https://huggingface.co/datasets/arbml/news_articles_aljazeera"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,870 documents",
      "year": 2020,
      "notes": "This data-set contains 5870 news articles in Arabic language extraced from aljazeera.net website.",
      "metrics": {
        "downloads": 27,
        "likes": 0,
        "lastModified": "2024-03-30"
      }
    },
    {
      "id": "arabic-openai-mmmlu",
      "name": "Arabic Openai MMMLU",
      "type": "benchmark",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic_Openai_MMMLU"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The MMLU is a widely recognized benchmark for assessing general knowledge attained by AI models.",
      "metrics": {
        "downloads": 27,
        "likes": 4,
        "lastModified": "2024-09-23"
      }
    },
    {
      "id": "arabic-quora-duplicates",
      "name": "Arabic Quora Duplicates",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-Quora-Duplicates"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "The Arabic Version of the Quora Question Pairs Dataset 2.",
      "metrics": {
        "downloads": 27,
        "likes": 2,
        "lastModified": "2024-07-03"
      }
    },
    {
      "id": "arabic-raw-text",
      "name": "ARABIC RAW TEXT",
      "type": "dataset",
      "country": "SA",
      "org": "riotu-lab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/riotu-lab/ARABIC-RAW-TEXT"
      },
      "size": "100M–1B rows",
      "year": 2024,
      "notes": "Raw Arabic text corpus combining Aluka, AraWiki, Aya and Islamic Books subsets for pretraining.",
      "metrics": {
        "downloads": 27,
        "likes": 5,
        "lastModified": "2024-08-08"
      }
    },
    {
      "id": "arabic-text-correction",
      "name": "Arabic-Text-Correction",
      "type": "llm",
      "country": "INTL",
      "org": "SuperSl6",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/SuperSl6/Arabic-Text-Correction"
      },
      "notes": "Text Correction - AraT5-based text correction",
      "base_model": [
        "ubc-nlp/arat5-base"
      ],
      "metrics": {
        "downloads": 27,
        "likes": 7,
        "lastModified": "2025-02-03"
      }
    },
    {
      "id": "arabic-llama-math-dataset",
      "name": "Arabic_LLaMA_Math_Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Jr23xd23",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Jr23xd23/Arabic_LLaMA_Math_Dataset"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Instruction: The problem statement or question (text, i",
      "metrics": {
        "downloads": 27,
        "likes": 4,
        "lastModified": "2024-10-07"
      }
    },
    {
      "id": "darija-to-english",
      "name": "darija to english",
      "type": "llm",
      "country": "INTL",
      "org": "centino00",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/centino00/darija-to-english"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Fine-tune of opus-mt-ar-en translating Latin-script Darija to English (BLEU 29.9).",
      "base_model": [
        "helsinki-nlp/opus-mt-ar-en"
      ],
      "metrics": {
        "downloads": 27,
        "likes": 7,
        "lastModified": "2024-03-06"
      }
    },
    {
      "id": "dataset-dyal-darija",
      "name": "dataset dyal darija",
      "type": "dataset",
      "country": "INTL",
      "org": "DRAGOO",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/DRAGOO/dataset_dyal_darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2023,
      "notes": "3.6M lines of Moroccan Darija text for pretraining.",
      "metrics": {
        "downloads": 27,
        "likes": 5,
        "lastModified": "2023-08-25"
      }
    },
    {
      "id": "egytriplets-2m",
      "name": "egytriplets 2m",
      "type": "dataset",
      "country": "INTL",
      "org": "metga97",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/metga97/egytriplets-2m"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "🔨 How It Was Built",
      "metrics": {
        "downloads": 27,
        "likes": 3,
        "lastModified": "2025-06-06"
      }
    },
    {
      "id": "t5-darija-summarization",
      "name": "t5 darija summarization",
      "type": "llm",
      "country": "INTL",
      "org": "Kamel",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/Kamel/t5-darija-summarization"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "T5 summarization model for Moroccan Darija trained on MArSum, 19,806 news articles with summaries.",
      "metrics": {
        "downloads": 27,
        "likes": 8,
        "lastModified": "2024-09-25"
      }
    },
    {
      "id": "twifil",
      "name": "Twifil",
      "type": "dataset",
      "country": "DZ",
      "org": "USTHB / University of Algiers",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "emotion"
      ],
      "links": {
        "github": "https://github.com/kinmokusu/oea_algd",
        "hf": "https://huggingface.co/datasets/arbml/Twifil",
        "paper": "https://aclanthology.org/2020.lrec-1.151.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "14,000 sentences",
      "year": 2020,
      "notes": "An Algerian dialect dataset annotated for both sentiment (9,000 tweets), emotion (about 5,000 tweets) and extra-linguistic information including author.",
      "metrics": {
        "downloads": 27,
        "likes": 1,
        "lastModified": "2024-03-23"
      }
    },
    {
      "id": "aoc-aldi",
      "name": "AOC-ALDi",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Edinburgh",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/AMR-KELEG/ALDi/raw/master/data/AOC-ALDi.tar.gz",
        "hf": "https://huggingface.co/datasets/arbml/AOC_ALDi",
        "paper": "https://aclanthology.org/2023.emnlp-main.655/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "127,835 sentences",
      "year": 2023,
      "notes": "Comments to news articles with a continuous level of dialectness score between 0 and 1.",
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2024-03-23"
      }
    },
    {
      "id": "araeventcoref",
      "name": "AraEventCoref",
      "type": "dataset",
      "country": "INTL",
      "org": "Prince Sattam bin Abdulaziz University",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "coreference-resolution",
        "information-extraction"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.11391166",
        "hf": "https://huggingface.co/datasets/AldawsariNLP/AraEventCoref"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,381 sentences",
      "year": 2025,
      "notes": "The dataset consists of 50 Arabic news articles from the SANAD dataset annotated with Arabic event coreference relation.",
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2025-03-10"
      }
    },
    {
      "id": "aya-aya-23-8b-dpo",
      "name": "Aya Aya.23.8B DPO",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Aya-Aya.23.8B-DPO"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Aya-Aya.23.8B-DPO is a DPO dataset designed to ad",
      "metrics": {
        "downloads": 26,
        "likes": 2,
        "lastModified": "2024-05-29"
      }
    },
    {
      "id": "english-egyptian-arabic-translator",
      "name": "english-egyptian-arabic-translator",
      "type": "llm",
      "country": "INTL",
      "org": "Omar-youssef",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Omar-youssef/english-egyptian-arabic-translator"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Neural machine-translation model from English to Egyptian Arabic.",
      "base_model": [
        "helsinki-nlp/opus-mt-tc-big-en-ar"
      ],
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2026-06-04"
      }
    },
    {
      "id": "jais-7b-chat",
      "name": "jais 7b chat",
      "type": "llm",
      "country": "INTL",
      "org": "erfanvaredi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/erfanvaredi/jais-7b-chat"
      },
      "size": "7B",
      "on_device": false,
      "year": 2024,
      "notes": "This model is the double quantized version of jais-13b-chat by core42.",
      "metrics": {
        "downloads": 26,
        "likes": 6,
        "lastModified": "2024-02-22"
      }
    },
    {
      "id": "kuwaiti-qwen3-v3",
      "name": "Kuwaiti-Qwen3-v3",
      "type": "llm",
      "country": "KW",
      "org": "Community (Kuwait)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/balfaris/kuwaiti-qwen3-v3-GGUF"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "GGUF Qwen3 model fine-tuned for Kuwaiti dialect.",
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2026-06-04"
      }
    },
    {
      "id": "modernarabert",
      "name": "ModernAraBERT",
      "type": "llm",
      "country": "INTL",
      "org": "gizadatateam",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/gizadatateam/ModernAraBERT"
      },
      "year": 2025,
      "notes": "Arabic encoder adapted from ModernBERT-base via continued pretraining on about 9.8GB of Arabic text.",
      "base_model": [
        "answerdotai/modernbert-base"
      ],
      "metrics": {
        "downloads": 26,
        "likes": 5,
        "lastModified": "2025-10-18"
      }
    },
    {
      "id": "organic-sudanese-arabic-dialect-dataset",
      "name": "organic Sudanese Arabic dialect dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ebubekr53",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ebubekr53/organic-sudanese-arabic-dialect-dataset"
      },
      "dialects": [
        "sudanese"
      ],
      "year": 2026,
      "notes": "Sudanese Arabic dialect text dataset.",
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2026-07-21"
      }
    },
    {
      "id": "quran-classical-arabic-english-parallel-texts",
      "name": "Quran Classical Arabic English Parallel texts",
      "type": "dataset",
      "country": "INTL",
      "org": "ImruQays",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ImruQays/Quran-Classical-Arabic-English-Parallel-texts"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "This dataset presents a collection of parallel texts of the Holy Quran in Arabic (Imla'ei & Uthmanic scripts) alongside 17 different English translations.",
      "metrics": {
        "downloads": 26,
        "likes": 6,
        "lastModified": "2023-12-29"
      }
    },
    {
      "id": "tunisian-msa-parallel-corpus-evaluated",
      "name": "tunisian msa parallel corpus evaluated",
      "type": "benchmark",
      "country": "TN",
      "org": "InstaDeep / iCompass",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tunis-ai/tunisian-msa-parallel-corpus-evaluated"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset is a synthetic parallel corpus of Tunisian Arabic (aeb) and Modern Standard Arabic (arb).",
      "metrics": {
        "downloads": 26,
        "likes": 2,
        "lastModified": "2025-09-22"
      }
    },
    {
      "id": "whisper-small-tunisian-arabic",
      "name": "whisper-small-tunisian-arabic",
      "type": "asr",
      "country": "INTL",
      "org": "awaxsama",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/awaxsama/whisper-small-tunisian-arabic"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Whisper small and tiny fine-tuned for Tunisian Arabic speech recognition.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 26,
        "likes": 0,
        "lastModified": "2026-08-06"
      }
    },
    {
      "id": "algerian-darija-dictionary-v1",
      "name": "algerian darija dictionary v1",
      "type": "dataset",
      "country": "INTL",
      "org": "awras",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "dictionary",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/awras/algerian-darija-dictionary-v1"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "Dictionary of 4,636 Algerian Darija words and proverbs with French spelling, definitions and examples; the card warns of quality issues.",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2026-07-26"
      }
    },
    {
      "id": "apcd2",
      "name": "APCD2",
      "type": "dataset",
      "country": "JO",
      "org": "University of Jordan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/Gheith-Abandah/classify-arabic-poetry",
        "hf": "https://huggingface.co/datasets/arbml/APCDv2",
        "paper": "https://www.sciencedirect.com/science/article/pii/S1319157820305784/pdfft?md5=07be922e052bf43933bdb7bea5189718&pid=1-s2.0-S1319157820305784-main.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,831,770 sentences",
      "year": 2020,
      "notes": "1657 k verses of poems and prose to develop neural networks to classify and diacritize Arabic poetry",
      "metrics": {
        "downloads": 25,
        "likes": 0,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "arabic-marbert-poetry-classification",
      "name": "arabic MARBERT poetry classification",
      "type": "llm",
      "country": "INTL",
      "org": "Ammar-alhaj-ali",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/Ammar-alhaj-ali/arabic-MARBERT-poetry-classification"
      },
      "year": 2022,
      "notes": "Arabic MARBERT Poetry Classification Model",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2022-08-16"
      }
    },
    {
      "id": "arat5-darija-to-msa",
      "name": "AraT5_Darija_to_MSA",
      "type": "llm",
      "country": "INTL",
      "org": "saidettousy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/saidettousy/AraT5_Darija_to_MSA"
      },
      "dialects": [
        "magh",
        "msa"
      ],
      "year": 2024,
      "notes": "AraT5-base fine-tuned to translate Moroccan Darija into Modern Standard Arabic.",
      "metrics": {
        "downloads": 25,
        "likes": 4,
        "lastModified": "2024-09-26"
      }
    },
    {
      "id": "aratect",
      "name": "ARATECT",
      "type": "dataset",
      "country": "INTL",
      "org": "CogniSAL",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ai-text-detection"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CogniSAL/ARATECT"
      },
      "dialects": [
        "msa"
      ],
      "size": "4,800 sentences",
      "year": 2025,
      "notes": "The dataset includes Arabic texts written by humans and those generated by various large language models (LLMs).",
      "metrics": {
        "downloads": 25,
        "likes": 0,
        "lastModified": "2025-10-30"
      }
    },
    {
      "id": "darija-bible-aligned",
      "name": "darija bible aligned",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/darija_bible_aligned"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Moroccan Darija Bible readings with aligned audio and text.",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2025-04-20"
      }
    },
    {
      "id": "darija-text-generation",
      "name": "darija text generation",
      "type": "llm",
      "country": "MA",
      "org": "Youssef Chafiqui",
      "license": "bigscience-bloom-rail-1.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/darija-text-generation"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Arabic text generation model fine-tuned from bigscience/bloom-560m.",
      "base_model": [
        "bigscience/bloom-560m"
      ],
      "metrics": {
        "downloads": 25,
        "likes": 2,
        "lastModified": "2024-01-31"
      }
    },
    {
      "id": "iraqi-dialect",
      "name": "Iraqi Dialect",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialects",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Iraqi_Dialect"
      },
      "dialects": [
        "iraqi"
      ],
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "1,672 Iraqi dialect texts with labels.",
      "metrics": {
        "downloads": 25,
        "likes": 2,
        "lastModified": "2022-10-15"
      }
    },
    {
      "id": "nadia",
      "name": "NADiA",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/hhrb7phdyx/1",
        "hf": "https://huggingface.co/datasets/arbml/NADiA"
      },
      "dialects": [
        "msa"
      ],
      "size": "486,646 documents",
      "year": 2019,
      "notes": "NADiA Dataset is the largest, to the best of our knowledge, source for Arabic textual data that can be used in any NLP related task such as text classification.",
      "metrics": {
        "downloads": 25,
        "likes": 0,
        "lastModified": "2022-10-31"
      }
    },
    {
      "id": "neoarabert-msa-synonym-matryoshka-v1",
      "name": "NeoAraBERT-MSA-Synonym-Matryoshka-V1",
      "type": "embedding",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "synonyms"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/NeoAraBERT-MSA-Synonym-Matryoshka-V1"
      },
      "dialects": [
        "msa",
        "classical"
      ],
      "year": 2026,
      "notes": "Diacritics-aware Arabic sentence embeddings fine-tuned from NeoAraBERT-MSA to keep synonym sensitivity in MSA and Classical Arabic.",
      "base_model": [
        "u4rasd/neoarabert_msa"
      ],
      "metrics": {
        "downloads": 25,
        "likes": 2,
        "lastModified": "2026-05-10"
      }
    },
    {
      "id": "organic-iraqi-arabic-dialect-dataset",
      "name": "organic Iraqi Arabic dialect dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ebubekr53",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ebubekr53/organic-iraqi-arabic-dialect-dataset"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2026,
      "notes": "Iraqi Arabic dialect text dataset.",
      "metrics": {
        "downloads": 25,
        "likes": 0,
        "lastModified": "2026-06-23"
      }
    },
    {
      "id": "shahnameh-tajik-corpus",
      "name": "shahnameh tajik corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "ArabovMK",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArabovMK/shahnameh-tajik-corpus"
      },
      "size": "<1K rows",
      "year": 2025,
      "notes": "This dataset contains the full Tajik translation of the Persian epic poem Shahnameh by Abulqosim Firdausi.",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2025-04-26"
      }
    },
    {
      "id": "synthetic-egy-speech-dataset",
      "name": "Synthetic Egy Speech Dataset",
      "type": "dataset",
      "country": "EG",
      "org": "Mohamed Gomaa",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr",
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohamedGomaa30/Synthetic-Egy-Speech-Dataset"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2026,
      "notes": "A curated dataset of 1000 Egyptian Arabic speech samples — the best audio selected across 4 TTS models for each prompt.",
      "metrics": {
        "downloads": 25,
        "likes": 2,
        "lastModified": "2026-05-14"
      }
    },
    {
      "id": "trocr-ar-small",
      "name": "TrOCR-Ar-Small",
      "type": "ocr",
      "country": "INTL",
      "org": "gagan3012",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "evaluation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/gagan3012/TrOCR-Ar-Small"
      },
      "year": 2022,
      "notes": "Arabic image text to text model.",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2022-03-21"
      }
    },
    {
      "id": "trocr-tunisian-arabic",
      "name": "trocr-tunisian-arabic",
      "type": "ocr",
      "country": "INTL",
      "org": "Ghazouaniwala",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/Ghazouaniwala/trocr-tunisian-arabic"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "TrOCR fine-tuned for Tunisian Arabic text recognition.",
      "base_model": [
        "microsoft/trocr-base-handwritten"
      ],
      "metrics": {
        "downloads": 25,
        "likes": 0,
        "lastModified": "2026-09-13"
      }
    },
    {
      "id": "whisper-base-quran",
      "name": "whisper base quran",
      "type": "asr",
      "country": "INTL",
      "org": "raghadOmar",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/raghadOmar/whisper-base-quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "This model is a fine-tuned version of tarteel-ai/whisper-base-ar-quran on the Zolfa Dataset dataset.",
      "base_model": [
        "tarteel-ai/whisper-base-ar-quran"
      ],
      "metrics": {
        "downloads": 25,
        "likes": 4,
        "lastModified": "2024-06-12"
      }
    },
    {
      "id": "whisper-large-v2-arabic-5k-steps",
      "name": "whisper large v2 arabic 5k steps",
      "type": "asr",
      "country": "INTL",
      "org": "clu-ling",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/clu-ling/whisper-large-v2-arabic-5k-steps"
      },
      "year": 2023,
      "notes": "This model is a fine-tuned version of openai/whisper-large-v2 on the Arabic CommonVoice dataset (v11).",
      "metrics": {
        "downloads": 25,
        "likes": 3,
        "lastModified": "2023-03-03"
      }
    },
    {
      "id": "ara-tydi-triplet",
      "name": "Ara-TyDi Triplet",
      "type": "dataset",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/Ara-TyDi-Triplet"
      },
      "size": "100K-1M rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic Mr. TyDi reformatted as query-positive-negative triplets for embedding training.",
      "metrics": {
        "downloads": 24,
        "likes": 2,
        "lastModified": "2024-11-21"
      }
    },
    {
      "id": "arabic-clip-vit-base-patch32",
      "name": "Arabic clip vit base patch32",
      "type": "embedding",
      "country": "INTL",
      "org": "LinaAlhuri",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "clip",
        "image-text"
      ],
      "links": {
        "hf": "https://huggingface.co/LinaAlhuri/Arabic-clip-vit-base-patch32"
      },
      "year": 2023,
      "notes": "Arabic adaptation of OpenAI CLIP (ViT-B/32) relating Arabic text and images.",
      "metrics": {
        "downloads": 24,
        "likes": 3,
        "lastModified": "2023-11-14"
      }
    },
    {
      "id": "arabic-optimized-reasoning-dataset",
      "name": "Arabic Optimized Reasoning Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Jr23xd23",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Jr23xd23/Arabic-Optimized-Reasoning-Dataset"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "The Arabic Optimized Reasoning Dataset helps AI models get better at reasoning in Arabic.",
      "metrics": {
        "downloads": 24,
        "likes": 5,
        "lastModified": "2025-02-25"
      }
    },
    {
      "id": "arabic-satirical-fake-news-dataset",
      "name": "Arabic Satirical Fake News Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Surrey",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/sadanyh/Arabic-Satirical-Fake-News-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/Satirical_Fake_News",
        "paper": "https://arxiv.org/pdf/2011.00452.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "6,895 documents",
      "year": 2020,
      "notes": "This is a dataset for Arabic satirical fake news.",
      "metrics": {
        "downloads": 24,
        "likes": 1,
        "lastModified": "2022-10-15"
      }
    },
    {
      "id": "arabic-web-edu-seed",
      "name": "arabic web edu seed",
      "type": "dataset",
      "country": "INTL",
      "org": "sboughorbel",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/sboughorbel/arabic-web-edu-seed"
      },
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset contains Arabic web content that has been scored for educational value.",
      "metrics": {
        "downloads": 24,
        "likes": 3,
        "lastModified": "2025-10-18"
      }
    },
    {
      "id": "bahraini-arabic-asr",
      "name": "bahraini-arabic ASR",
      "type": "asr",
      "country": "BH",
      "org": "University of Bahrain (ghadeerl)",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/ghadeerl/bahraini-arabic"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Bahraini Arabic transcription model handling Arabic-English code-switching, from a University of Bahrain project.",
      "base_model": [
        "openai/whisper-base"
      ],
      "metrics": {
        "downloads": 24,
        "likes": 1,
        "lastModified": "2026-05-16"
      }
    },
    {
      "id": "barec-10m",
      "name": "BAREC-10M",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "readability",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/CAMeL-Lab/BAREC-10M"
      },
      "size": "10M words",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Expanded 10M-word balanced Arabic readability corpus from NYU Abu Dhabi.",
      "metrics": {
        "downloads": 24,
        "likes": 0,
        "lastModified": "2026-06-08"
      }
    },
    {
      "id": "cleananercorp",
      "name": "CLEANANERCorp",
      "type": "benchmark",
      "country": "SA",
      "org": "King Saud University",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "ner"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/CLEANANERCorp",
        "hf": "https://huggingface.co/datasets/arbml/CLEANANERCorp",
        "paper": "https://aclanthology.org/2024.osact-1.2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "150,000 tokens",
      "year": 2024,
      "notes": "We conducted empirical research to understand the errors in ANERcorp, correct them and propose a cleaner version of the dataset named CLEANANERCorp.",
      "metrics": {
        "downloads": 24,
        "likes": 1,
        "lastModified": "2024-06-23"
      }
    },
    {
      "id": "dallah",
      "name": "Dallah",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP/dallah"
      },
      "notes": "Advanced multimodal LLM for Arabic",
      "metrics": {
        "downloads": 24,
        "likes": 3,
        "lastModified": "2024-11-25"
      }
    },
    {
      "id": "darija-ner",
      "name": "darija ner",
      "type": "llm",
      "country": "INTL",
      "org": "hananour",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/hananour/darija-ner"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "This is the first model for Named Entity Recognition (NER) in the Moroccan dialect (Darija).",
      "metrics": {
        "downloads": 24,
        "likes": 8,
        "lastModified": "2024-02-11"
      }
    },
    {
      "id": "darija-lid",
      "name": "Darija-LID",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Darija-LID"
      },
      "year": 2025,
      "dialects": [
        "magh"
      ],
      "notes": "Language identification dataset for Moroccan Darija versus other Arabic dialects.",
      "metrics": {
        "downloads": 24,
        "likes": 3,
        "lastModified": "2025-05-09"
      }
    },
    {
      "id": "dimi-embedding-v4",
      "name": "DIMI embedding v4",
      "type": "embedding",
      "country": "INTL",
      "org": "AhmedZaky1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sts",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedZaky1/DIMI-embedding-v4"
      },
      "year": 2025,
      "notes": "Sentence embedding model fine-tuned for Arabic-English semantic textual similarity with Matryoshka and CoSENT losses.",
      "base_model": [
        "ahmedzaky1/dimi-embedding-v2"
      ],
      "metrics": {
        "downloads": 24,
        "likes": 4,
        "lastModified": "2025-05-30"
      }
    },
    {
      "id": "disease-ner",
      "name": "Disease NER",
      "type": "dataset",
      "country": "INTL",
      "org": "Jouf University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "https://www.mdpi.com/2306-5729/5/3/60/htm#app1-data-05-00060",
        "hf": "https://huggingface.co/datasets/arbml/Disease_NER",
        "paper": "https://www.mdpi.com/2306-5729/5/3/60"
      },
      "dialects": [
        "msa"
      ],
      "size": "62,506 tokens",
      "year": 2020,
      "notes": "The data consist of 27 Arabic medical articles, totaling around 50,000 words.",
      "metrics": {
        "downloads": 24,
        "likes": 0,
        "lastModified": "2024-07-20"
      }
    },
    {
      "id": "hala-350m",
      "name": "Hala 350M",
      "type": "llm",
      "country": "SA",
      "org": "KAUST",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/hammh0a/Hala-350M"
      },
      "base_model": [
        "liquidai/lfm2-350m"
      ],
      "size": "350M",
      "on_device": true,
      "year": 2025,
      "tags": [
        "variants:2"
      ],
      "notes": "Arabic-centric instruction and translation model from the Hala technical report (KAUST).",
      "metrics": {
        "downloads": 24,
        "likes": 3,
        "lastModified": "2025-09-18"
      }
    },
    {
      "id": "iraqi-arabic-nlp-toolkit-ianlp-dataset",
      "name": "Iraqi Arabic NLP Toolkit (IANLP) dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "hussainhadi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hussainhadi/Iraqi-Arabic-NLP-Toolkit-IANLP"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2026,
      "notes": "Annotated Iraqi Arabic social-media posts for NLP tasks.",
      "metrics": {
        "downloads": 24,
        "likes": 0,
        "lastModified": "2026-08-02"
      }
    },
    {
      "id": "moroccan-darija-wiki-dataset",
      "name": "Moroccan Darija Wiki Dataset",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Moroccan-Darija-Wiki-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Text corpus built from the Moroccan Darija Wikipedia.",
      "metrics": {
        "downloads": 24,
        "likes": 7,
        "lastModified": "2025-03-08"
      }
    },
    {
      "id": "adabtranslate-darija",
      "name": "AdabTranslate-Darija",
      "type": "llm",
      "country": "INTL",
      "org": "itsmeussa",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/itsmeussa/AdabTranslate-Darija"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Darija-to-MSA translator fine-tuned from AraBART (BLEU 46.5).",
      "base_model": [
        "moussakam/arabart"
      ],
      "metrics": {
        "downloads": 23,
        "likes": 8,
        "lastModified": "2024-03-27"
      }
    },
    {
      "id": "anadolu-ocr-corpus",
      "name": "anadolu ocr corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "fatihburakkaragoz",
      "license": "other",
      "modality": "text",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/fatihburakkaragoz/anadolu-ocr-corpus"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "Anadolu OCR Corpus is an OpenCR export of OCR text and document metadata for 52 historical Ottoman Turkish, Turkish, and Arabic-containing PDF sources.",
      "metrics": {
        "downloads": 23,
        "likes": 3,
        "lastModified": "2026-05-12"
      }
    },
    {
      "id": "arabic-news-summarization",
      "name": "arabic news summarization",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/arabic-news-summarization"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "MSA news summarization pairs (article and summary) with translated counterparts.",
      "metrics": {
        "downloads": 23,
        "likes": 2,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "arabic-rc-datasets",
      "name": "Arabic RC datasets",
      "type": "dataset",
      "country": "JO",
      "org": "Princess Sumaya University for Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/MariamBiltawi/Arabic_RC_datasets",
        "hf": "https://huggingface.co/datasets/arbml/Arabic_RC_AQA",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=9300111"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,862 sentences",
      "year": 2020,
      "notes": "Semiautomatically created Arabic reading comprehension benchmarks with question-answer pairs derived from TREC-style passages.",
      "metrics": {
        "downloads": 23,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arabic-deepseek-r1-distill-8b",
      "name": "Arabic-DeepSeek-R1-Distill-8B",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-DeepSeek-R1-Distill-8B"
      },
      "size": "8B",
      "on_device": false,
      "year": 2025,
      "notes": "DeepSeek-R1-Distill-Llama-8B fine-tuned for Arabic reasoning.",
      "base_model": [
        "unsloth/deepseek-r1-distill-llama-8b-unsloth-bnb-4bit"
      ],
      "metrics": {
        "downloads": 23,
        "likes": 4,
        "lastModified": "2025-02-02"
      }
    },
    {
      "id": "arabicconceptualcaptions3m",
      "name": "ArabicConceptualCaptions3M",
      "type": "dataset",
      "country": "INTL",
      "org": "LinaAlhuri",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "translation",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/LinaAlhuri/ArabicConceptualCaptions3M"
      },
      "size": "1M–10M rows",
      "year": 2023,
      "notes": "This dataset consists of conceptual captions translated into Arabic using the Google Translate API.",
      "metrics": {
        "downloads": 23,
        "likes": 3,
        "lastModified": "2023-11-15"
      }
    },
    {
      "id": "ascat",
      "name": "ASCAT",
      "type": "dataset",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/ASCAT-Arabic-Scientific-Translation",
        "paper": "https://arxiv.org/abs/2604.00015"
      },
      "size": "under 1K rows",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic scientific corpus for translation evaluation, English and Arabic abstracts.",
      "metrics": {
        "downloads": 23,
        "likes": 1,
        "lastModified": "2026-04-02"
      }
    },
    {
      "id": "autotweet",
      "name": "AutoTweet",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "gender-id"
      ],
      "links": {
        "website": "https://www.dropbox.com/s/amnv06boef2vn4k/Autotweet-Dataset-v1.0.zip?dl=0",
        "hf": "https://huggingface.co/datasets/arbml/AutoTweet",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-319-28940-3_10"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,503 sentences",
      "year": 2015,
      "notes": "Classification of Arabic tweets into automated or manual.",
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "barka",
      "name": "Barka",
      "type": "llm",
      "country": "INTL",
      "org": "Slim205",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/Slim205/Barka-9b-it-v02"
      },
      "year": 2024,
      "notes": "Arabic instruction-tuned 9B model trained on a high-quality synthetically generated instruction dataset.",
      "base_model": [
        "google/gemma-2-9b-it"
      ],
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2024-10-21"
      }
    },
    {
      "id": "baved",
      "name": "BAVED",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "emotion"
      ],
      "links": {
        "github": "https://github.com/40uf411/Basic-Arabic-Vocal-Emotions-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/BAVED"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,935 tokens",
      "year": 2019,
      "notes": "BAVED: Arabic words spoken at different levels of emotion.",
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2022-11-01"
      }
    },
    {
      "id": "kesc-dataset",
      "name": "KESC dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Zainab984",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "emotion"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Zainab984/CS461-KESC-Anonymous-Dataset"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Anonymised Kuwaiti emotional speech corpus built from short drama clips.",
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2026-10-04"
      }
    },
    {
      "id": "qadi-qcri-arabic-dialects-identification-corpus",
      "name": "QADI (QCRI Arabic Dialects Identification) Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/qadi",
        "hf": "https://huggingface.co/datasets/arbml/QADI"
      },
      "year": 2021,
      "notes": "Country-level Arabic dialect identification dataset of tweets for benchmarking DI.",
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "qari-0-2-2-diacritics-dataset-large",
      "name": "qari 0.2.2 diacritics dataset large",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "diacritics"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/qari-0.2.2-diacritics-dataset-large"
      },
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "Synthetic images of diacritized Arabic text rendered in multiple fonts, sizes and page layouts for OCR training.",
      "metrics": {
        "downloads": 23,
        "likes": 4,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "quran-uthmani",
      "name": "quran uthmani",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/quran_uthmani"
      },
      "dialects": [
        "classical"
      ],
      "size": "1K–10K rows",
      "year": 2022,
      "notes": "6,235 Quran verses in Uthmani script indexed by surah and ayah.",
      "metrics": {
        "downloads": 23,
        "likes": 2,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "seamless-darija-english",
      "name": "Seamless Darija-English",
      "type": "llm",
      "country": "INTL",
      "org": "AnasAber",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/AnasAber/seamless-darija-eng"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "This model is a fine-tuned version of Facebook's Seamless large m4t-v2 model, specifically optimized for translation between Moroccan Arabic (Darija)",
      "base_model": [
        "facebook/seamless-m4t-v2-large"
      ],
      "metrics": {
        "downloads": 23,
        "likes": 0,
        "lastModified": "2024-09-12"
      }
    },
    {
      "id": "whisper-m-quran-lora-dataset-mix",
      "name": "whisper m quran lora dataset mix",
      "type": "asr",
      "country": "INTL",
      "org": "MaddoggProduction",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/MaddoggProduction/whisper-m-quran-lora-dataset-mix"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "This is a specialized Automatic Speech Recognition (ASR) model for Quranic Recitation with tashkeel or diacritics.",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 23,
        "likes": 6,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "1milion-token-egy-songs",
      "name": "1milion token EGY songs",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "lyrics",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/1milion_token_EGY_songs"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "6,554 Egyptian Arabic song lyrics with title and text.",
      "metrics": {
        "downloads": 22,
        "likes": 9,
        "lastModified": "2024-06-03"
      }
    },
    {
      "id": "arabic-document-classification-dataset",
      "name": "Arabic document classification dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Waikato Institute of Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://diab.edublogs.org/dataset-for-arabic-document-classification/",
        "hf": "https://huggingface.co/datasets/arbml/Document_Classification",
        "paper": "https://www.ijcaonline.org/archives/volume101/number7/17701-8680"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,700 documents",
      "year": 2014,
      "notes": "The dataset contains nine major disciplines: Art, Literature, Religion, Politics, Law, Economy, Sport, Health, and Technology.",
      "metrics": {
        "downloads": 22,
        "likes": 0,
        "lastModified": "2022-10-29"
      }
    },
    {
      "id": "arabic-triplet-multi-negatives",
      "name": "Arabic Triplet With Multi-Negatives",
      "type": "dataset",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/Arabic-Triplet-With-Multi-Negatives"
      },
      "size": "10K-100K rows",
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic Mr. TyDi variant with several hard negatives per query for retrieval training.",
      "metrics": {
        "downloads": 22,
        "likes": 2,
        "lastModified": "2024-11-21"
      }
    },
    {
      "id": "darija-to-english-2",
      "name": "darija to english 2",
      "type": "llm",
      "country": "INTL",
      "org": "Youssef Chafiqui",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/darija-to-english-2"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "NLLB-200-distilled-600M fine-tuned for Darija-to-English translation.",
      "base_model": [
        "facebook/nllb-200-distilled-600m"
      ],
      "metrics": {
        "downloads": 22,
        "likes": 3,
        "lastModified": "2024-02-14"
      }
    },
    {
      "id": "emirati-audio-dataset-20h",
      "name": "Emirati Audio Dataset 20H",
      "type": "dataset",
      "country": "INTL",
      "org": "Community (Emirati speech)",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oi-uae/EmiratiAudioDataset-20H"
      },
      "size": "20h",
      "dialects": [
        "gulf"
      ],
      "notes": "About 20 hours of Emirati-dialect speech for ASR and TTS research.",
      "metrics": {
        "downloads": 22,
        "likes": 0,
        "lastModified": "2025-04-24"
      }
    },
    {
      "id": "english-arabic-translations",
      "name": "english arabic translations",
      "type": "dataset",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oddadmix/english-arabic-translations"
      },
      "size": "10K–100K rows",
      "year": 2026,
      "notes": "English-Arabic parallel sentences with source and hash columns.",
      "metrics": {
        "downloads": 22,
        "likes": 3,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "english-egyptian-translation-finance",
      "name": "English Egyptian Translation finance",
      "type": "dataset",
      "country": "INTL",
      "org": "Omar-youssef",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omar-youssef/English-Egyptian-Translation-finance"
      },
      "dialects": [
        "egy"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset contains bilingual text pairs in English and Egyptian Arabic focused on finance and financial topics.",
      "metrics": {
        "downloads": 22,
        "likes": 3,
        "lastModified": "2025-09-25"
      }
    },
    {
      "id": "flodusta",
      "name": "FloDusTA",
      "type": "dataset",
      "country": "SA",
      "org": "Umm Al-Qura University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "event-detection"
      ],
      "links": {
        "github": "https://github.com/BatoolHamawi/FloDusTA",
        "hf": "https://huggingface.co/datasets/arbml/FloDusTA_Dust_Storm",
        "paper": "https://aclanthology.org/2020.lrec-1.174.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "8,998 sentences",
      "year": 2020,
      "notes": "FloDusTA is a dataset of annotated tweets collected for the purpose of developing an event detection system.",
      "metrics": {
        "downloads": 22,
        "likes": 0,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "jev-ar",
      "name": "jev ar",
      "type": "llm",
      "country": "INTL",
      "org": "atmaneayoub",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "intent-routing",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/atmaneayoub/jev-ar"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Gulf Arabic (Emirati, Saudi) code-switching intent-routing classifier for RAG, based on convaiinnovations/laya.",
      "base_model": [
        "convaiinnovations/laya"
      ],
      "metrics": {
        "downloads": 22,
        "likes": 5,
        "lastModified": "2026-10-01"
      }
    },
    {
      "id": "mt5-base-ar",
      "name": "mT5-base_ar",
      "type": "llm",
      "country": "INTL",
      "org": "ArabicNLP",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "seq2seq"
      ],
      "links": {
        "hf": "https://huggingface.co/ArabicNLP/mT5-base_ar"
      },
      "year": 2023,
      "notes": "mT5-base with vocabulary shrunk to Arabic and some English embeddings, reducing the 582M-parameter original.",
      "metrics": {
        "downloads": 22,
        "likes": 6,
        "lastModified": "2023-05-14"
      }
    },
    {
      "id": "s2t-wav2vec2-large-en-ar",
      "name": "s2t wav2vec2 large en ar",
      "type": "asr",
      "country": "INTL",
      "org": "Meta",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/s2t-wav2vec2-large-en-ar"
      },
      "year": 2023,
      "notes": "s2t-wav2vec2-large-en-ar is a Speech to Text Transformer model trained for end-to-end Speech Translation (ST).",
      "metrics": {
        "downloads": 22,
        "likes": 7,
        "lastModified": "2023-01-24"
      }
    },
    {
      "id": "sudanese-mt-benchmark",
      "name": "sudanese-mt-benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "O96a",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/O96a/sudanese-mt-benchmark"
      },
      "dialects": [
        "sudanese"
      ],
      "year": 2026,
      "notes": "Machine-translation benchmark specific to Sudanese Arabic, with a companion dialect benchmark.",
      "metrics": {
        "downloads": 22,
        "likes": 0,
        "lastModified": "2026-04-04"
      }
    },
    {
      "id": "vits-ar-sa-huba",
      "name": "vits ar sa huba",
      "type": "tts",
      "country": "INTL",
      "org": "wasmdashai",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/wasmdashai/vits-ar-sa-huba"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2024,
      "tags": [
        "variants:1"
      ],
      "notes": "An advanced text-to-speech (TTS) system specifically designed for the Saudi dialect.",
      "metrics": {
        "downloads": 22,
        "likes": 26,
        "lastModified": "2024-09-15"
      }
    },
    {
      "id": "whisper-small-quran-lora-dataset-mix",
      "name": "whisper small quran lora dataset mix",
      "type": "asr",
      "country": "INTL",
      "org": "MaddoggProduction",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/MaddoggProduction/whisper-small-quran-lora-dataset-mix"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "This is a specialized Automatic Speech Recognition (ASR) model for Quranic Recitation with tashkeel or diacritics.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 22,
        "likes": 4,
        "lastModified": "2026-03-23"
      }
    },
    {
      "id": "zipformer-p-arabic-v2",
      "name": "zipformer p arabic v2",
      "type": "asr",
      "country": "INTL",
      "org": "Muno459",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Muno459/zipformer_p-arabic-v2"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Zipformer ASR model for Arabic (v2); gated, free non-commercial use only.",
      "metrics": {
        "downloads": 22,
        "likes": 7,
        "lastModified": "2026-08-06"
      }
    },
    {
      "id": "arabic-ala-lc-romanization",
      "name": "Arabic ALA LC  Romanization",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/Arabic_ALA-LC_Romanization",
        "hf": "https://huggingface.co/datasets/arbml/ALA_LC_Romanization",
        "paper": "https://aclanthology.org/2021.wanlp-1.23.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "107,439 sentences",
      "year": 2021,
      "notes": "Parallel Arabic and Romanized bibliographic entries",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2022-10-22"
      }
    },
    {
      "id": "arabic-reasoning-dataset-logic",
      "name": "arabic reasoning dataset logic",
      "type": "dataset",
      "country": "INTL",
      "org": "beetleware",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/beetleware/arabic-reasoning-dataset-logic"
      },
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "This dataset comprises a series of logical reasoning tasks designed to evaluate and train artificial intelligence models on understanding.",
      "metrics": {
        "downloads": 21,
        "likes": 14,
        "lastModified": "2025-05-21"
      }
    },
    {
      "id": "arabic-vicuna-80",
      "name": "Arabic Vicuna 80",
      "type": "benchmark",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/FreedomIntelligence/Arabic-Vicuna-80"
      },
      "size": "<1K rows",
      "year": 2023,
      "notes": "Arabic version of the Vicuna-80 benchmark: 80 queries with category labels.",
      "metrics": {
        "downloads": 21,
        "likes": 3,
        "lastModified": "2023-09-21"
      }
    },
    {
      "id": "arabic-clue-instruct",
      "name": "Arabic-Clue-Instruct",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Siena",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-generation"
      ],
      "links": {
        "github": "https://github.com/KamyarZeinalipour/Arabic-Text-to-Crosswords",
        "hf": "https://huggingface.co/datasets/Kamyar-zeinalipour/Arabic-Clue-Instruct",
        "paper": "https://aclanthology.org/2025.loreslm-1.36/"
      },
      "dialects": [
        "msa"
      ],
      "size": "14,497 documents",
      "year": 2024,
      "notes": "Dataset for Arabic educational crosswords",
      "metrics": {
        "downloads": 21,
        "likes": 1,
        "lastModified": "2025-01-06"
      }
    },
    {
      "id": "arafa",
      "name": "ARAFA",
      "type": "dataset",
      "country": "LB",
      "org": "American University of Beirut",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/chriskhalil/ARAFA",
        "hf": "https://huggingface.co/datasets/ChristopheKhalil/ARAFA",
        "paper": "https://doi.org/10.21203/rs.3.rs-7335564/v1"
      },
      "dialects": [
        "msa"
      ],
      "size": "181,976 sentences",
      "year": 2025,
      "notes": "Large-scale Arabic fact-checking dataset generated by LLMs from Wikipedia.",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2026-09-23"
      }
    },
    {
      "id": "arapoems",
      "name": "AraPoems",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/faisalq/AraPoems"
      },
      "year": 2024,
      "notes": "Arabic poetry corpus used to train AraPoemBERT.",
      "metrics": {
        "downloads": 21,
        "likes": 1,
        "lastModified": "2024-03-19"
      }
    },
    {
      "id": "arasencorpus",
      "name": "AraSenCorpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Engineering and Technology",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/yemen2016/AraSenCorpus",
        "hf": "https://huggingface.co/datasets/arbml/AraSenCorpus",
        "paper": "https://www.mdpi.com/2076-3417/11/5/2434"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4,500,000 sentences",
      "year": 2021,
      "notes": "Contains 4.5 million tweets and covers both modern standard Arabic and some of the Arabic dialects",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2022-10-30"
      }
    },
    {
      "id": "asr-hassaniya-whisper-medium-v1",
      "name": "ASR-hassaniya-whisper-medium-v1",
      "type": "asr",
      "country": "INTL",
      "org": "DesertMindAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/DesertMindAI/ASR-hassaniya-whisper-medium-v1"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2025,
      "notes": "Whisper-medium speech recognition model for Hassaniya Arabic from DesertMindAI.",
      "base_model": [
        "ychafiqui/whisper-medium-darija"
      ],
      "metrics": {
        "downloads": 21,
        "likes": 2,
        "lastModified": "2025-09-18"
      }
    },
    {
      "id": "aya-command-r-dpo",
      "name": "Aya Command.R DPO",
      "type": "dataset",
      "country": "INTL",
      "org": "2A2I",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/2A2I/Aya-Command.R-DPO"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Aya-Command.R-DPO is a DPO dataset designed to ad",
      "metrics": {
        "downloads": 21,
        "likes": 9,
        "lastModified": "2024-05-16"
      }
    },
    {
      "id": "aydid-public",
      "name": "AYDID public",
      "type": "dataset",
      "country": "INTL",
      "org": "Mansoor Saleh",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mansoorSaleh/AYDID-public"
      },
      "size": "<1K rows",
      "year": 2026,
      "notes": "Public sample and test set of AYDID, the first speech corpus for Yemeni Arabic at sub-dialect level, for ASR and dialect ID.",
      "dialects": [
        "yemeni"
      ],
      "metrics": {
        "downloads": 21,
        "likes": 2,
        "lastModified": "2026-06-02"
      }
    },
    {
      "id": "callhome-egyptian-arabic-speech-translation-corpus",
      "name": "CALLHOME: Egyptian Arabic Speech Translation Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Google Brain",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/noisychannel/ARZ_callhome_corpus",
        "hf": "https://huggingface.co/datasets/arbml/CALLHOME",
        "paper": "https://www.cis.upenn.edu/~ccb/publications/callhome-egyptian-arabic-speech-translations.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "39,213 sentences",
      "year": 2014,
      "tags": [
        "multilingual"
      ],
      "notes": "Three-way parallel dataset of Egyptian Arabic Speech, transcriptions and English translations",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "darija-sentiment-analysis",
      "name": "darija sentiment analysis",
      "type": "llm",
      "country": "MA",
      "org": "Youssef Chafiqui",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "sentiment",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/ychafiqui/darija_sentiment_analysis"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Darija Sentiment Analysis model fine-tuned on DarijaBERT on Online scraped Comments written in Darija.",
      "metrics": {
        "downloads": 21,
        "likes": 4,
        "lastModified": "2024-03-09"
      }
    },
    {
      "id": "dawqas-a-dataset-for-arabic-why-question-answering-system",
      "name": "DAWQAS: A Dataset for Arabic Why Question Answering System",
      "type": "dataset",
      "country": "AE",
      "org": "Emirates College of Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/masun/DAWQAS",
        "hf": "https://huggingface.co/datasets/arbml/DAWQAS",
        "paper": "https://www.sciencedirect.com/science/article/pii/S1877050918321690"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,205 sentences",
      "year": 2018,
      "notes": "DAWQAS contains 3,205 Arabic why-question answer pairs scraped from websites, with rhetorical relation annotations for why-QA research.",
      "metrics": {
        "downloads": 21,
        "likes": 1,
        "lastModified": "2022-10-21"
      }
    },
    {
      "id": "fanar-sadiq-bench",
      "name": "Fanar-Sadiq Classifier Datasets",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "classification",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/Fanar-Sadiq-Bench"
      },
      "size": "1K-10K rows",
      "year": 2026,
      "dialects": [
        "msa"
      ],
      "notes": "Bilingual query-classification datasets released with the Fanar-Sadiq Islamic multi-agent system.",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2026-06-25"
      }
    },
    {
      "id": "moroccanwikipedia-qa",
      "name": "MoroccanWikipedia-QA",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/MoroccanWikipedia-QA"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "MoroccanWikipedia-QA: question-answering dataset derived from the Moroccan Wikipedia dump for Darija QA.",
      "metrics": {
        "downloads": 21,
        "likes": 6,
        "lastModified": "2024-09-27"
      }
    },
    {
      "id": "nada",
      "name": "NADA",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://www.researchgate.net/publication/326060650_NADA_A_New_Arabic_Dataset",
        "hf": "https://huggingface.co/datasets/arbml/NADA",
        "paper": "https://thesai.org/Downloads/Volume9No9/Paper_28-NADA_New_Arabic_Dataset_for_Text_Classification.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "13,066 documents",
      "year": 2018,
      "notes": "NADA corpus is collected from two existing corpora, which are Diab Dataset DAA corpus and OSAC corpus.",
      "metrics": {
        "downloads": 21,
        "likes": 0,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "tachelhiyt-darija",
      "name": "tachelhiyt darija",
      "type": "dataset",
      "country": "INTL",
      "org": "NoureddineMOR",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NoureddineMOR/tachelhiyt-darija",
        "paper": "https://aclanthology.org/2025.icnlsp-1.37.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Parallel speech corpus for Moroccan Darija and Tachelhiyt, two underrepresented languages (ICNLSP 2025).",
      "metrics": {
        "downloads": 21,
        "likes": 4,
        "lastModified": "2025-11-18"
      }
    },
    {
      "id": "terjman-large",
      "name": "Terjman-Large",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Terjman-Large-v2.0"
      },
      "size": "239M",
      "dialects": [
        "magh"
      ],
      "notes": "English to Moroccan Darija translation model from atlasia.",
      "base_model": [
        "atlasia/terjman-large-v1.2"
      ],
      "metrics": {
        "downloads": 21,
        "likes": 1,
        "lastModified": "2025-03-11"
      }
    },
    {
      "id": "alpaca-arabic-cleaned",
      "name": "alpaca arabic cleaned",
      "type": "dataset",
      "country": "INTL",
      "org": "saillab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/saillab/alpaca-arabic-cleaned"
      },
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "This repository contains the dataset used for the TaCo paper.",
      "metrics": {
        "downloads": 20,
        "likes": 3,
        "lastModified": "2024-09-20"
      }
    },
    {
      "id": "alukah-arabic",
      "name": "Alukah Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "ImruQays",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ImruQays/Alukah-Arabic"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "This dataset is a comprehensive collection of articles sourced from the Alukah website, a renowned platform offering extensive content primarily in Arabic.",
      "metrics": {
        "downloads": 20,
        "likes": 4,
        "lastModified": "2024-03-22"
      }
    },
    {
      "id": "antcorpus",
      "name": "ANTCORPUS",
      "type": "dataset",
      "country": "TN",
      "org": "National Engineering School of Sousse",
      "license": "other",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "github": "https://github.com/antcorpus/antcorpus.data",
        "hf": "https://huggingface.co/datasets/arbml/antcorpus",
        "paper": "https://ieeexplore.ieee.org/abstract/document/8308275/authors#authors"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,005 documents",
      "year": 2017,
      "notes": "ANT Corpus, which is collected from RSS Feeds.",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "arabic-image-captioning-100m",
      "name": "Arabic-Image-Captioning_100M",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/Arabic-Image-Captioning_100M"
      },
      "notes": "100 million Arabic image captions",
      "metrics": {
        "downloads": 20,
        "likes": 4,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "aragemma2b-instruct",
      "name": "araGemma2B-instruct",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/araGemma2B-instruct"
      },
      "year": 2024,
      "notes": "Gemma 2B fine-tuned for Arabic instruction following; the card gives few details.",
      "size": "2B",
      "on_device": true,
      "metrics": {
        "downloads": 20,
        "likes": 3,
        "lastModified": "2024-02-28"
      }
    },
    {
      "id": "at-odtsa",
      "name": "AT-ODTSA",
      "type": "dataset",
      "country": "INTL",
      "org": "Fatih Sultan Mehmet Vakif University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/sabudalfa/AT-ODTSA",
        "hf": "https://huggingface.co/datasets/arbml/AT_ODSTA",
        "paper": "https://www.researchgate.net/publication/359171347_AT-ODTSA_a_Dataset_of_Arabic_Tweets_for_Open_Domain_Targeted_Sentiment_Analysis"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,000 sentences",
      "year": 2022,
      "notes": "A dataset of Arabic Tweets for Open-Domain Targeted Sentiment Analysis, which includes Arabic tweets along with labels that specify targets (topics)",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "bert2bert",
      "name": "bert2bert",
      "type": "llm",
      "country": "INTL",
      "org": "malmarjeh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/malmarjeh/bert2bert"
      },
      "dialects": [
        "msa"
      ],
      "year": 2023,
      "notes": "BERT2BERT abstractive summarizer initialized from AraBERT, fine-tuned on 84,764 paragraph-summary pairs.",
      "metrics": {
        "downloads": 20,
        "likes": 2,
        "lastModified": "2023-07-01"
      }
    },
    {
      "id": "dacs",
      "name": "DACS",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "github": "https://github.com/qcri/Arabic_speech_code_switching",
        "hf": "https://huggingface.co/datasets/QCRI/DACS",
        "paper": "https://www.isca-archive.org/interspeech_2020/chowdhury20c_interspeech.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "2 hours",
      "year": 2020,
      "notes": "First spoken Egyptian Arabic intra-utterance code-switching corpus.",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2025-10-13"
      }
    },
    {
      "id": "darijabanking",
      "name": "DarijaBanking",
      "type": "dataset",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "intent-detection",
        "banking"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AbderrahmanSkiredj1/DarijaBanking"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "DarijaBanking: 6,504 banking-domain Darija queries with intent label and language field.",
      "metrics": {
        "downloads": 20,
        "likes": 3,
        "lastModified": "2024-09-07"
      }
    },
    {
      "id": "doda-sentences-darija-english",
      "name": "DoDA_sentences_darija_english",
      "type": "dataset",
      "country": "INTL",
      "org": "AnasAber",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AnasAber/DoDA_sentences_darija_english"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Dataset of english/moroccan arabic translations, from DoDA dataset.",
      "metrics": {
        "downloads": 20,
        "likes": 4,
        "lastModified": "2024-09-06"
      }
    },
    {
      "id": "fanar-2-oryx-ig",
      "name": "Fanar-2-Oryx-IG",
      "type": "llm",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "text-to-image"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI/Fanar-2-Oryx-IG",
        "paper": "https://arxiv.org/abs/2603.16397"
      },
      "year": 2026,
      "notes": "Culturally aligned Arabic-English image generation model fine-tuned from FLUX.1-schnell.",
      "base_model": [
        "black-forest-labs/flux.1-schnell"
      ],
      "metrics": {
        "downloads": 20,
        "likes": 3,
        "lastModified": "2026-03-25"
      }
    },
    {
      "id": "iraqi-arabic-msa-english-parallel",
      "name": "iraqi-arabic-msa-english-parallel",
      "type": "dataset",
      "country": "INTL",
      "org": "SaifSilverHand",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SaifSilverHand/iraqi-arabic-msa-english-parallel"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2026,
      "notes": "Parallel corpus aligning Iraqi Arabic, MSA and English sentences.",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2026-07-30"
      }
    },
    {
      "id": "jais",
      "name": "Jais",
      "type": "llm",
      "country": "AE",
      "org": "Inception AI, Cerebras",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/inceptionai/jais-30b-v3"
      },
      "base_model": [
        "from-scratch"
      ],
      "size": "13B, 30B",
      "dialects": [
        "msa"
      ],
      "notes": "Arabic-centric, bilingual, instruction-tuned",
      "metrics": {
        "downloads": 20,
        "likes": 13,
        "lastModified": "2024-09-11"
      }
    },
    {
      "id": "ksucca",
      "name": "KSUCCA",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/KSUCCA"
      },
      "year": 2013,
      "dialects": [
        "classical"
      ],
      "notes": "King Saud University Corpus of Classical Arabic, about 50M words of classical texts.",
      "metrics": {
        "downloads": 20,
        "likes": 1,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "moroccan-darija-domain-classifier-dataset",
      "name": "moroccan darija domain classifier dataset",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/moroccan_darija_domain_classifier_dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2025,
      "notes": "This dataset is designed for text classification in Moroccan Darija, a dialect spoken in Morocco.",
      "metrics": {
        "downloads": 20,
        "likes": 3,
        "lastModified": "2025-02-14"
      }
    },
    {
      "id": "orca-benchmark",
      "name": "ORCA",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/orca",
        "website": "https://orca.dlnlp.ai/",
        "github": "https://github.com/UBC-NLP/orca"
      },
      "size": "60 datasets",
      "year": 2022,
      "dialects": [
        "mixed"
      ],
      "notes": "Unified Arabic NLU leaderboard of 60 datasets across seven task clusters; the ARLUE successor.",
      "metrics": {
        "downloads": 20,
        "likes": 10,
        "lastModified": "2023-11-22"
      }
    },
    {
      "id": "ptcc",
      "name": "PTCC",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AMR-KELEG/PTCC"
      },
      "dialects": [
        "msa",
        "magh"
      ],
      "size": "149 sentences",
      "year": 2024,
      "notes": "Parallel corpus of 149 articles of the 2014 Tunisian Constitution in Modern Standard Arabic and Tunisian Arabic.",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2024-01-02"
      }
    },
    {
      "id": "qatar-verdicts-arabic",
      "name": "Qatar-Verdicts-Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "ymoslem",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ymoslem/Qatar-Verdicts-Arabic"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "notes": "Arabic court verdicts from Qatar.",
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2026-09-16"
      }
    },
    {
      "id": "seasmed-fine-tuned-on-common-voice-17-arabic-samehelalfi",
      "name": "Seasmed Fine Tuned on Common Voice 17 Arabic",
      "type": "tts",
      "country": "INTL",
      "org": "samehelalfi",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/samehelalfi/Seasmed-Fine-Tuned-on-Common-Voice-17-Arabic"
      },
      "year": 2025,
      "notes": "This model is a fine-tuned version of sesame/csm-1b on the Arabic subset of Common Voice 17.0 dataset.",
      "base_model": [
        "sesame/csm-1b"
      ],
      "metrics": {
        "downloads": 20,
        "likes": 4,
        "lastModified": "2025-07-07"
      }
    },
    {
      "id": "sudanese-dialect-tweet",
      "name": "Sudanese Dialect Tweet",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialects",
        "sentiment"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Sudanese_Dialect_Tweet"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "2,119 Sudanese dialect tweets with three annotators' labels and mode label.",
      "metrics": {
        "downloads": 20,
        "likes": 2,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "tam-omani-adapter-tam-omani-v3",
      "name": "Tam Omani adapter (tam-omani-v3)",
      "type": "llm",
      "country": "INTL",
      "org": "shaaarplegs",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/shaaarplegs/tam-omani-v3"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "LoRA/QLoRA adapter for Tam, an Omani-Arabic assistant.",
      "base_model": [
        "qwen/qwen3-14b"
      ],
      "metrics": {
        "downloads": 20,
        "likes": 0,
        "lastModified": "2026-06-19"
      }
    },
    {
      "id": "alkhalilcorpustopic",
      "name": "AlkhalilCorpusTopic",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohammed First University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/oujda-nlp-team/TopicClassifiedTexts",
        "paper": "https://aclanthology.org/2026.abjadnlp-1.27.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "766 documents",
      "year": 2026,
      "notes": "Thematic Modern Standard Arabic collected form the web with 7 different topics.",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2025-07-13"
      }
    },
    {
      "id": "arabic-keyphrase-dataset",
      "name": "Arabic Keyphrase dataset",
      "type": "dataset",
      "country": "SA",
      "org": "Saudi Aramco",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "keyphrase-extraction"
      ],
      "links": {
        "github": "https://github.com/logmani/ArabicDataset",
        "hf": "https://huggingface.co/datasets/arbml/Keyphrase_Extraction",
        "paper": "https://airccj.org/CSCP/vol7/csit76321.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "400 documents",
      "year": 2017,
      "notes": "A dataset in Arabic language for automatic keyphrase extraction algorithms",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2022-11-02"
      }
    },
    {
      "id": "arabic-news-tweets",
      "name": "Arabic News Tweets",
      "type": "dataset",
      "country": "SA",
      "org": "Umm Al-Qura University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/9dxgbgx86k/3",
        "hf": "https://huggingface.co/datasets/arbml/Arabic_News_Tweets"
      },
      "dialects": [
        "msa"
      ],
      "size": "89,179 sentences",
      "year": 2021,
      "notes": "This dataset is a relatively great size collection of Arabic news tweets that were collected from an official and verified users in Twitter.",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "arabic-question-generation",
      "name": "Arabic question generation",
      "type": "llm",
      "country": "INTL",
      "org": "MIIB-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "question-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/MIIB-NLP/Arabic-question-generation"
      },
      "year": 2022,
      "notes": "AraT5-base fine-tuned for Arabic question generation from a passage and an answer.",
      "metrics": {
        "downloads": 19,
        "likes": 7,
        "lastModified": "2022-10-09"
      }
    },
    {
      "id": "arabic-qwq-32b-preview",
      "name": "Arabic QWQ 32B Preview",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "math"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-QWQ-32B-Preview"
      },
      "size": "32B",
      "on_device": false,
      "year": 2024,
      "notes": "Arabic fine-tune of QwQ-32B-Preview focused on Arabic reasoning, mainly math.",
      "base_model": [
        "qwen/qwq-32b-preview"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 4,
        "lastModified": "2024-12-03"
      }
    },
    {
      "id": "arabic-summarization",
      "name": "arabic summarization",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "summarization",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/arabic-summarization"
      },
      "year": 2026,
      "notes": "هذا المشروع يقدّم نموذج تلخيص نصوص باللغة العربية مبني على النموذج الأساسي LiquidAI/LFM2-350M، وتمت إعادة تدريبه.",
      "base_model": [
        "liquidai/lfm2-350m"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 2,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "arabml-darija-english-parallel-dataset",
      "name": "arabml darija english parallel dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "darija"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AbderrahmanSkiredj1/arabml_darija_english_parallel_dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "7,314 Moroccan Darija-English sentence pairs with Darija in Latin and Arabic letters.",
      "metrics": {
        "downloads": 19,
        "likes": 3,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "aramodernbert-topic-classifier",
      "name": "AraModernBert-Topic-Classifier",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/AraModernBert-Topic-Classifier"
      },
      "year": 2025,
      "notes": "This is an Experimental Arabic version of ModernBERT-base, trained ONLY on Topic Classification Task using the base model of original modernbert.",
      "base_model": [
        "answerdotai/modernbert-base"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 5,
        "lastModified": "2025-01-11"
      }
    },
    {
      "id": "arc-wmi",
      "name": "ARC-WMI",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/ARC-WMI",
        "hf": "https://huggingface.co/datasets/arbml/ARC_WMI",
        "paper": "http://lrec-conf.org/workshops/lrec2018/W30/pdf/9_W30.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4,476 sentences",
      "year": 2018,
      "notes": "4476 sentences with over 61k words, extracted from 94 sources of Arabic written medicine information",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2022-10-07"
      }
    },
    {
      "id": "arcovidvac",
      "name": "ArCovidVac",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stance-detection"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/ArCovidVac.zip",
        "hf": "https://huggingface.co/datasets/arbml/ArCovidVac",
        "paper": "https://arxiv.org/pdf/2201.06496.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2022,
      "notes": "The largest manually annotated Arabic tweet dataset, ArCovidVac, for the COVID-19 vaccination campaign, covering many countries in the Arab region",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "caamt",
      "name": "CAAMT",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Toledo",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Senju2/context-aware-arabic-to-english-model-with-register",
        "paper": "https://arxiv.org/pdf/2604.06456v1.pdf"
      },
      "dialects": [
        "mixed",
        "magh",
        "egy",
        "gulf",
        "lev",
        "iraqi"
      ],
      "size": "6,400 sentences",
      "year": 2026,
      "notes": "57.6k parallel Arabic dialect MT dataset.",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2026-04-07"
      }
    },
    {
      "id": "egyptian-arabic-eot",
      "name": "Egyptian Arabic End-of-Turn Detector",
      "type": "llm",
      "country": "EG",
      "org": "Waqf AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "turn-detection"
      ],
      "links": {
        "hf": "https://huggingface.co/Waqf-AI/egyptian-arabic-eot-qwen35-0.8b"
      },
      "size": "0.8B",
      "year": 2026,
      "on_device": true,
      "dialects": [
        "egy"
      ],
      "notes": "0.8B Qwen3.5 classifier for end-of-turn detection in Egyptian Arabic voice agents.",
      "base_model": [
        "qwen/qwen3.5-0.8b"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2026-09-13"
      }
    },
    {
      "id": "egyptian-arabic-translator-llama-3-8b",
      "name": "Egyptian Arabic Translator Llama 3 8B",
      "type": "llm",
      "country": "INTL",
      "org": "ahmedsamirio",
      "license": "llama3",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/ahmedsamirio/Egyptian-Arabic-Translator-Llama-3-8B"
      },
      "dialects": [
        "egy"
      ],
      "size": "8B",
      "on_device": false,
      "year": 2024,
      "notes": "See axolotl config axolotl version: 0.4.1",
      "base_model": [
        "meta-llama/meta-llama-3-8b"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 8,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "hassaniya-punctuation-restoration",
      "name": "hassaniya-punctuation-restoration",
      "type": "tool",
      "country": "INTL",
      "org": "Emin009",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "hf": "https://huggingface.co/Emin009/hassaniya-punctuation-restoration"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "Open-source punctuation-restoration model for Hassaniya Arabic.",
      "metrics": {
        "downloads": 19,
        "likes": 2,
        "lastModified": "2026-01-15"
      }
    },
    {
      "id": "merged-arabic-corpus-of-isolated-words",
      "name": "Merged Arabic Corpus of Isolated Words",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-1.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "isolated-words"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/mohamedanwarvic/merged-arabic-corpus-of-isolated-words",
        "hf": "https://huggingface.co/datasets/arbml/Merged_Arabic_Corpus_of_Isolated_Words"
      },
      "dialects": [
        "msa"
      ],
      "size": "9,992 tokens",
      "year": 2019,
      "notes": "Voice recordings of 50 native Arabic speakers saying 20 words about 10 times each.",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "nilechat-arabizi-mor",
      "name": "nilechat arabizi mor",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/nilechat-arabizi-mor"
      },
      "size": "1M–10M rows",
      "year": 2025,
      "notes": "Arabizi-Morocco: Moroccan Arabic text converted from Arabic script to Arabizi for language-model training.",
      "dialects": [
        "magh"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 5,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "oman-legal-corpus",
      "name": "oman-legal-corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "AlAdawi",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AlAdawi/oman-legal-corpus"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "notes": "Structured Arabic corpus of Omani legislation issued since 1972.",
      "metrics": {
        "downloads": 19,
        "likes": 0,
        "lastModified": "2026-09-25"
      }
    },
    {
      "id": "tarc",
      "name": "TArC",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Rome Sapienza",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/eligugliotta/tarc",
        "hf": "https://huggingface.co/datasets/arbml/TArC",
        "paper": "https://aclanthology.org/2020.lrec-1.770.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4,790 sentences",
      "year": 2020,
      "notes": "Flexible and multi-purpose open corpus in order to be a useful support for different types of analyses: computational and linguistics, as well as for NLP.",
      "metrics": {
        "downloads": 19,
        "likes": 1,
        "lastModified": "2024-03-24"
      }
    },
    {
      "id": "thatiar",
      "name": "ThatiAR",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/QCRI/ThatiAR"
      },
      "year": 2024,
      "dialects": [
        "msa"
      ],
      "notes": "Subjectivity detection dataset of Arabic news sentences with experimental resources.",
      "metrics": {
        "downloads": 19,
        "likes": 1,
        "lastModified": "2024-10-21"
      }
    },
    {
      "id": "tts-transformer-ar-cv7",
      "name": "tts transformer ar cv7",
      "type": "tts",
      "country": "INTL",
      "org": "Meta",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/facebook/tts_transformer-ar-cv7"
      },
      "year": 2022,
      "notes": "Fairseq S^2 Transformer single-speaker Arabic text-to-speech model.",
      "metrics": {
        "downloads": 19,
        "likes": 8,
        "lastModified": "2022-01-28"
      }
    },
    {
      "id": "whisper-small-full-finetune",
      "name": "whisper small full finetune",
      "type": "asr",
      "country": "INTL",
      "org": "mosama",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/mosama/whisper_small_full_finetune"
      },
      "year": 2025,
      "notes": "Full finetuning in float32 bit of whisper small on SADA 2022 Datset.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 3,
        "lastModified": "2025-06-18"
      }
    },
    {
      "id": "whisper-small-codeswitching-arabicenglish",
      "name": "whisper-small-codeswitching-ArabicEnglish",
      "type": "asr",
      "country": "INTL",
      "org": "azeem23",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/azeem23/whisper-small-codeswitching-ArabicEnglish"
      },
      "year": 2025,
      "notes": "Whisper-small fine-tuned for Arabic-English code-switching speech recognition.",
      "on_device": true,
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 19,
        "likes": 3,
        "lastModified": "2025-05-01"
      }
    },
    {
      "id": "zalmati",
      "name": "Zalmati",
      "type": "llm",
      "country": "INTL",
      "org": "PetraAI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "translation",
        "sentiment",
        "qa",
        "summarization",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/PetraAI/Zalmati"
      },
      "year": 2024,
      "notes": "Zalmati is a powerful multilingual language model trained on the massive and diverse PetraAI dataset.",
      "metrics": {
        "downloads": 19,
        "likes": 3,
        "lastModified": "2024-04-05"
      }
    },
    {
      "id": "adawat",
      "name": "Adawat",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "catalog"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/adawat"
      },
      "size": "<1K rows",
      "year": 2022,
      "notes": "Hugging Face dataset aggregating a catalog of Arabic NLP tools and resources.",
      "metrics": {
        "downloads": 18,
        "likes": 3,
        "lastModified": "2022-11-16"
      }
    },
    {
      "id": "alriyadh-newspaper-covid-dataset",
      "name": "AlRiyadh-Newspaper-Covid-Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "corpus",
        "covid"
      ],
      "links": {
        "github": "https://github.com/alioh/AlRiyadh-Newspaper-Covid-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/AlRiyadh_Newspaper_Covid"
      },
      "dialects": [
        "msa"
      ],
      "size": "24,084 documents",
      "year": 2021,
      "notes": "It is a dataset of Arabic newspapers articles addressing COVID-19 related events.",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2022-10-14"
      }
    },
    {
      "id": "arabic-dev2",
      "name": "Arabic DEv2",
      "type": "dataset",
      "country": "INTL",
      "org": "Bloomberg L.P.",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "entity-retrieval"
      ],
      "links": {
        "website": "https://zenodo.org/record/4560653#.YqSGWXZBxD9",
        "hf": "https://huggingface.co/datasets/arbml/ArabicDEv2",
        "paper": "https://aclanthology.org/2021.wanlp-1.24.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "139 sentences",
      "year": 2021,
      "notes": "Arabic translation of a 139-query subset of DBpedia-Entity v2 into Modern Standard Arabic for entity retrieval.",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "arabic-flood-twitter-dataset",
      "name": "Arabic Flood Twitter Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "Taibah University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "crisis-detection"
      ],
      "links": {
        "github": "https://github.com/alaa-a-a/Arabic-Twitter-Corpus-for-Flood-Detection",
        "hf": "https://huggingface.co/datasets/arbml/twitter_flood_detection",
        "paper": "https://aclanthology.org/W19-5609.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4,037 sentences",
      "year": 2019,
      "notes": "It includes 4,037 human-labelled Arabic Twitter messages for four high-risk flood events that occurred in 2018",
      "metrics": {
        "downloads": 18,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arabic-nli-pair-score",
      "name": "Arabic NLi Pair Score",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nli",
        "embedding-training"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-NLi-Pair-Score"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "Arabic translation of SNLI and MultiNLI (pair-score subset: sentence pairs with scores) for embedding training.",
      "metrics": {
        "downloads": 18,
        "likes": 3,
        "lastModified": "2024-08-02"
      }
    },
    {
      "id": "arabic-punctuation-dataset",
      "name": "Arabic Punctuation Dataset",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "punctuation-detection"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/2pkxckwgs3/1",
        "hf": "https://huggingface.co/datasets/arbml/CBT",
        "paper": "https://www.sciencedirect.com/science/article/pii/S2352340924000908"
      },
      "dialects": [
        "msa"
      ],
      "size": "12,183,000 sentences",
      "year": 2024,
      "notes": "This is a curated dataset, specifically designed to facilitate the study of punctuation.",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2024-05-04"
      }
    },
    {
      "id": "arat5-dialects-translation",
      "name": "arat5-dialects-translation",
      "type": "llm",
      "country": "INTL",
      "org": "PRAli22",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "task-specific"
      ],
      "links": {
        "hf": "https://huggingface.co/PRAli22/arat5-arabic-dialects-translation"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Dialect→MSA - AraT5 dialect translation",
      "metrics": {
        "downloads": 18,
        "likes": 3,
        "lastModified": "2024-03-01"
      }
    },
    {
      "id": "areej",
      "name": "AREEj",
      "type": "dataset",
      "country": "INTL",
      "org": "Arab Center for Research and Policy Studies",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "relation-extraction"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/dru-ac/ArSRED",
        "paper": "https://aclanthology.org/2024.arabicnlp-1.6.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "500,000 sentences",
      "year": 2024,
      "notes": "This dataset was made by adding evidence annotations to the Arabic subset of SREDFM.",
      "metrics": {
        "downloads": 18,
        "likes": 1,
        "lastModified": "2025-03-26"
      }
    },
    {
      "id": "ceap",
      "name": "CEAP",
      "type": "dataset",
      "country": "INTL",
      "org": "University in Krakow",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/ceap-bp/",
        "hf": "https://huggingface.co/datasets/arbml/CEAP",
        "paper": "https://www.ejournals.eu/pliki/art/20227/"
      },
      "dialects": [
        "classical"
      ],
      "size": "50 documents",
      "year": 2021,
      "notes": "50 text files of classical Arabic poetry from the 6th and 7th centuries, derived from KSUCAC and another corpus.",
      "metrics": {
        "downloads": 18,
        "likes": 1,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "darijastory",
      "name": "DarijaStory",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-generation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI-Paris/DarijaStory"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "DarijaStory: 4,392 long Moroccan Darija stories scraped from a story website, for story completion.",
      "metrics": {
        "downloads": 18,
        "likes": 2,
        "lastModified": "2024-11-13"
      }
    },
    {
      "id": "egyptial-gulf-twitter-dataset",
      "name": "Egyptial Gulf Twitter Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialects",
        "twitter"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/Egyptial_Gulf_Twitter_Dataset"
      },
      "dialects": [
        "egy",
        "gulf"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Egyptian and Gulf Arabic tweets with raw tweet metadata (1K-10K rows).",
      "metrics": {
        "downloads": 18,
        "likes": 2,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "egyptian-arabic-text-summarization-dataset",
      "name": "Egyptian Arabic Text Summarization Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omar-youssef/Egyptian-text-summarization"
      },
      "dialects": [
        "egy"
      ],
      "size": "3,689 sentences",
      "year": 2025,
      "notes": "Text-summary pairs in Egyptian Arabic for training and evaluating summarization models.",
      "metrics": {
        "downloads": 18,
        "likes": 1,
        "lastModified": "2025-09-28"
      }
    },
    {
      "id": "emirati-14b-v3",
      "name": "Emirati-14B-v3",
      "type": "llm",
      "country": "INTL",
      "org": "Airev AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/airev-ai/emirati-14b-v3"
      },
      "size": "14B",
      "dialects": [
        "gulf"
      ],
      "notes": "14B chat model fine-tuned for Emirati dialect and UAE context.",
      "metrics": {
        "downloads": 18,
        "likes": 1,
        "lastModified": "2024-10-13"
      }
    },
    {
      "id": "idat",
      "name": "IDAT",
      "type": "dataset",
      "country": "INTL",
      "org": "PRHLT Research Center",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "sarcasm"
      ],
      "links": {
        "github": "https://github.com/bilalghanem/multilingual_irony",
        "hf": "https://huggingface.co/datasets/arbml/multilingual_irony",
        "paper": "http://ceur-ws.org/Vol-2517/T4-1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,030 sentences",
      "year": 2019,
      "notes": "Written in Modern Standard Arabic but also in different Arabic language varieties including Egypt, Gulf, Levantine and Maghrebi dialects",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "masrawy-english-arabic-translator",
      "name": "masrawy english arabic translator",
      "type": "llm",
      "country": "EG",
      "org": "oddadmix",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/oddadmix/masrawy-english-arabic-translator"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Arabic translation model fine-tuned from Helsinki-NLP/opus-mt-en-ar trained on oddadmix/english-arabic-translations.",
      "base_model": [
        "helsinki-nlp/opus-mt-en-ar"
      ],
      "metrics": {
        "downloads": 18,
        "likes": 2,
        "lastModified": "2026-04-17"
      }
    },
    {
      "id": "mcwc",
      "name": "MCWC",
      "type": "dataset",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/drelhaj/MCWC"
      },
      "year": 2024,
      "notes": "Multilingual constitutions from 191 nations aligned for legal NLP and MT.",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2025-11-27"
      }
    },
    {
      "id": "misraj-mudd",
      "name": "Misraj MUDD",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/mudd",
        "paper": "https://arxiv.org/abs/2505.17894"
      },
      "year": 2025,
      "notes": "Arabic pretraining text translated from SlimPajama-627B; gated.",
      "metrics": {
        "downloads": 18,
        "likes": 5,
        "lastModified": "2026-09-17"
      }
    },
    {
      "id": "sudanese-flores",
      "name": "Sudanese-Flores",
      "type": "benchmark",
      "country": "INTL",
      "org": "Mila-Quebec AI Institute",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/McGill-NLP/Sudanese-Flores",
        "paper": "https://aclanthology.org/2026.africanlp-main.25.pdf"
      },
      "dialects": [
        "sudanese",
        "msa"
      ],
      "size": "2,009 sentences",
      "year": 2026,
      "notes": "An extension of the popular Flores+ machine translation (MT) benchmark to the Sudanese Arabic dialect.",
      "metrics": {
        "downloads": 18,
        "likes": 0,
        "lastModified": "2026-04-15"
      }
    },
    {
      "id": "ar-reranking-eval",
      "name": "Ar Reranking Eval",
      "type": "benchmark",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/NAMAA-Space/Ar-Reranking-Eval"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "This dataset, containing 468 rows, is curated for evaluating reranking and retrieval models in Arabic.",
      "metrics": {
        "downloads": 17,
        "likes": 2,
        "lastModified": "2024-11-01"
      }
    },
    {
      "id": "arabic-semantic-highlighter",
      "name": "arabic semantic highlighter",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "other",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/arabic-semantic-highlighter"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "A sentence-level semantic highlighting model for Arabic text, designed for RAG (Retrieval-Augmented Generation) systems.",
      "base_model": [
        "baai/bge-reranker-base"
      ],
      "metrics": {
        "downloads": 17,
        "likes": 3,
        "lastModified": "2026-01-13"
      }
    },
    {
      "id": "arabic-sentiment-lexicons",
      "name": "Arabic Sentiment Lexicons",
      "type": "dataset",
      "country": "INTL",
      "org": "National Research Council Canada",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://saifmohammad.com/WebPages/ArabicSA.html",
        "hf": "https://huggingface.co/datasets/arbml/Sentiment_Lexicons",
        "paper": "https://aclanthology.org/L16-1006.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "176,364 tokens",
      "year": 2016,
      "notes": "By using distant supervision techniques on Arabic tweets, and by translating English sentiment lexicons into Arabic using a freely available statistical.",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2022-10-14"
      }
    },
    {
      "id": "darija-asr-bench",
      "name": "Darija ASR Bench",
      "type": "benchmark",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Darija_ASR_Bench"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2025,
      "notes": "Benchmark for Moroccan Darija speech recognition from the Atlasia Darija lab.",
      "metrics": {
        "downloads": 17,
        "likes": 2,
        "lastModified": "2025-07-18"
      }
    },
    {
      "id": "darija-vlm-gqa-dataset",
      "name": "Darija VLM GQA Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "KBayoud",
      "license": "apache-2.0",
      "modality": "multimodal",
      "tasks": [
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KBayoud/Darija-VLM-GQA-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "Original dataset : vikhyatk/gqa-val",
      "metrics": {
        "downloads": 17,
        "likes": 3,
        "lastModified": "2025-05-03"
      }
    },
    {
      "id": "dataset-for-arabic-classification",
      "name": "DataSet for Arabic Classification",
      "type": "dataset",
      "country": "INTL",
      "org": "Sultan Moulay Slimane University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/v524p5dhpj/2",
        "hf": "https://huggingface.co/datasets/arbml/DataSet_Arabic_Classification"
      },
      "dialects": [
        "msa"
      ],
      "size": "111,700 documents",
      "year": 2018,
      "notes": "DataSet for Arabic text classification.",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2022-10-31"
      }
    },
    {
      "id": "helsinki-translation-english-moroccan-arabic",
      "name": "Helsinki translation English Moroccan Arabic",
      "type": "llm",
      "country": "INTL",
      "org": "lachkarsalim",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/lachkarsalim/Helsinki-translation-English_Moroccan-Arabic"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Helsinki-en-ar Transformer fine-tuned to translate English into Darija (Moroccan Arabic).",
      "metrics": {
        "downloads": 17,
        "likes": 9,
        "lastModified": "2024-03-07"
      }
    },
    {
      "id": "magpie-saqr-najdi",
      "name": "Magpie Saqr Najdi",
      "type": "tts",
      "country": "INTL",
      "org": "mabahboh",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/mabahboh/Magpie-Saqr-Najdi"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Najdi Saudi Arabic TTS fine-tuned from NVIDIA Magpie multilingual 357M on the SADA 2022 speakers dataset.",
      "base_model": [
        "nvidia/magpie_tts_multilingual_357m"
      ],
      "metrics": {
        "downloads": 17,
        "likes": 2,
        "lastModified": "2026-09-29"
      }
    },
    {
      "id": "omar-al-saleh-manuscripts-full",
      "name": "omar al saleh manuscripts full",
      "type": "dataset",
      "country": "INTL",
      "org": "U4RASD",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "htr",
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/U4RASD/omar-al-saleh-manuscripts-full"
      },
      "size": "<1K rows",
      "year": 2026,
      "notes": "Scanned Omar Al-Saleh memoir manuscripts (1951-1965) with expert DOCX transcriptions, from the NAKBA NLP 2026 shared task.",
      "metrics": {
        "downloads": 17,
        "likes": 3,
        "lastModified": "2026-04-08"
      }
    },
    {
      "id": "pearl-lite",
      "name": "PEARL-LITE",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-nd-4.0",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/PEARL-LITE",
        "paper": "https://doi.org/10.18653/v1/2025.findings-emnlp.1254"
      },
      "dialects": [
        "msa"
      ],
      "size": "893 images",
      "year": 2025,
      "notes": "PEARL-LITE is a lightweight, smaller subset of the main PEARL benchmark.",
      "metrics": {
        "downloads": 17,
        "likes": 1,
        "lastModified": "2025-10-27"
      }
    },
    {
      "id": "senzi",
      "name": "SenZi",
      "type": "dataset",
      "country": "INTL",
      "org": "Knowledge Media Institute",
      "license": "other",
      "modality": "text",
      "tasks": [
        "sentiment",
        "transliteration"
      ],
      "links": {
        "website": "https://tahatobaili.github.io/project-rbz/",
        "hf": "https://huggingface.co/datasets/arbml/SenZi",
        "paper": "https://aclanthology.org/R19-1138.pdf"
      },
      "dialects": [
        "lev"
      ],
      "size": "24,600 tokens",
      "year": 2019,
      "notes": "By translating, annotating, and transliterating other resources to have an initial set of 2K sentiment words.",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "smolkalam",
      "name": "SmolKalam",
      "type": "dataset",
      "country": "SA",
      "org": "KAUST",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/SultanR/smolkalam",
        "paper": "https://arxiv.org/pdf/2511.18411.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,790,478 sentences",
      "year": 2025,
      "notes": "Large-scale Arabic SFT dataset with reasoning and tool-calling.",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2025-12-01"
      }
    },
    {
      "id": "tudicoi",
      "name": "TuDiCoI",
      "type": "dataset",
      "country": "TN",
      "org": "University of Sfax",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/MarwaGraja/TuDiCOI",
        "hf": "https://huggingface.co/datasets/arbml/TuDiCoI",
        "paper": "https://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.452.7847&rep=rep1&type=pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "127 conversations",
      "year": 2010,
      "notes": "The corpus consists of 434 1465 staff utterances and 1615 client utterances",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "wdc",
      "name": "WDC",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Regensburg",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/Maha-J-Althobaiti/Arabic_NER_Wiki-Corpus",
        "hf": "https://huggingface.co/datasets/arbml/WDC",
        "paper": "https://aclanthology.org/E14-3012.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,000,000 tokens",
      "year": 2014,
      "notes": "Wikipedia-derived Arabic NER corpus of about 6 million tokens across different genres.",
      "metrics": {
        "downloads": 17,
        "likes": 0,
        "lastModified": "2022-10-25"
      }
    },
    {
      "id": "whisper-small-cv-ar",
      "name": "whisper small cv ar",
      "type": "asr",
      "country": "SA",
      "org": "ARBML",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/arbml/whisper-small-cv-ar"
      },
      "year": 2024,
      "notes": "This model is a fine-tuned version of openai/whisper-small on the Common Voice 11.0 dataset.",
      "metrics": {
        "downloads": 17,
        "likes": 5,
        "lastModified": "2024-08-31"
      }
    },
    {
      "id": "algerian-stt-dialect",
      "name": "algerian-STT-dialect",
      "type": "dataset",
      "country": "INTL",
      "org": "hananeek2",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/hananeek2/algerian-STT-dialect"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Algerian dialect speech-to-text corpus of naturalistic speech from Algerian TV programs, social content and interviews.",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2026-07-08"
      }
    },
    {
      "id": "arabgend",
      "name": "ArabGend",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "gender-id",
        "gender-bias-detection"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/ArabGend.zip",
        "hf": "https://huggingface.co/datasets/asas-ai/ArabGend",
        "paper": "https://aclanthology.org/2022.wnut-1.14.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "167,000 tokens",
      "year": 2022,
      "notes": "Dataset of 167K Arabic Twitter accounts labeled for gender and location.",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2024-05-05"
      }
    },
    {
      "id": "arabic-cohere-include-base-44-mmlu-style",
      "name": "Arabic Cohere include base 44 mmlu style",
      "type": "benchmark",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-Cohere-include-base-44-mmlu-style"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "INCLUDE is a comprehensive knowledge- and reasoning-centric benchmark spanning 44 languages that evaluates multilingual LLMs.",
      "metrics": {
        "downloads": 16,
        "likes": 3,
        "lastModified": "2024-12-08"
      }
    },
    {
      "id": "arabic-llama3-1-lora-ft",
      "name": "Arabic llama3.1 lora FT",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Arabic-llama3.1-lora-FT"
      },
      "year": 2024,
      "notes": "This fine-tuned model is based on the newly released LLaMA 3.1 model and has been specifically trained on the Arabic BigScience xP3 dataset.",
      "metrics": {
        "downloads": 16,
        "likes": 11,
        "lastModified": "2024-07-27"
      }
    },
    {
      "id": "arabic-sahm",
      "name": "Arabic Sahm",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "safety",
        "red-teaming"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Arabic-Sahm"
      },
      "year": 2026,
      "notes": "Qwen3-4B fine-tune for Arabic red-teaming, adversarial prompts and safety evaluation.",
      "base_model": [
        "qwen/qwen3-4b"
      ],
      "metrics": {
        "downloads": 16,
        "likes": 7,
        "lastModified": "2026-03-15"
      }
    },
    {
      "id": "arabic-senti-lexicon",
      "name": "Arabic senti-lexicon",
      "type": "dataset",
      "country": "INTL",
      "org": "Universiti Kebangsaan Malaysia",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pos",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/almoslmi/masc",
        "hf": "https://huggingface.co/datasets/arbml/Senti_Lexicon",
        "paper": "https://journals.sagepub.com/doi/pdf/10.1177/0165551516683908"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,880 tokens",
      "year": 2018,
      "notes": "A list of 3880 positive and negative synsets annotated with their part of speech, polarity scores, dialects synsets and inflected forms",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2024-07-20"
      }
    },
    {
      "id": "calyou",
      "name": "CALYOU",
      "type": "dataset",
      "country": "INTL",
      "org": "Ecole Supérieure d’Informatique",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/abidikarima/CALYOU",
        "hf": "https://huggingface.co/datasets/arbml/CAYLOU",
        "paper": "https://hal.archives-ouvertes.fr/hal-01531591/document"
      },
      "dialects": [
        "magh"
      ],
      "size": "5,190 sentences",
      "year": 2017,
      "notes": "A Comparable Spoken Algerian Corpus Harvested from YouTube",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2022-10-21"
      }
    },
    {
      "id": "ctab-corpus-of-tunisian-arabizi",
      "name": "CTAB: Corpus of Tunisian Arabizi",
      "type": "dataset",
      "country": "TN",
      "org": "University of Sfax",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://zenodo.org/record/4781769#.YqSPY3ZBxD9",
        "hf": "https://huggingface.co/datasets/arbml/CTAB"
      },
      "dialects": [
        "magh"
      ],
      "size": "5,702 sentences",
      "year": 2021,
      "notes": "This dataset has been created between 2017 and 2021 to provide a textual resource that can be used to study the behaviors of Tunisian people in writing.",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "dialex",
      "name": "dialex",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "embedding"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/dialex",
        "hf": "https://huggingface.co/datasets/arbml/dialex"
      },
      "year": 2021,
      "notes": "DiaLex - A Benchmark for Evaluating Multidialectal Arabic Word Embeddings",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "khaleej-2004",
      "name": "Khaleej-2004",
      "type": "dataset",
      "country": "INTL",
      "org": "INRIA-LORIA",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/arabiccorpus/files/",
        "hf": "https://huggingface.co/datasets/arbml/khaleej_2004",
        "paper": "https://hal.inria.fr/inria-00000448/document"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,690 documents",
      "year": 2004,
      "notes": "Extracted from the daily Arabic news paper Akhbar al Khaleej, it includes 5120 news articles corresponding to 2,855,069 words covering four topics sport.",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "narabizi-corpus",
      "name": "NArabizi corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of the Basque Country UPV/EHU",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos",
        "parsing",
        "translation",
        "sentiment",
        "transliteration",
        "classification"
      ],
      "links": {
        "github": "https://github.com/SamiaTouileb/Narabizi",
        "hf": "https://huggingface.co/datasets/arbml/NArabizi",
        "paper": "https://arxiv.org/pdf/2105.07400.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "1,500 sentences",
      "year": 2021,
      "notes": "Extension of NArabizi treebank by adding to annotations.",
      "metrics": {
        "downloads": 16,
        "likes": 0,
        "lastModified": "2022-10-15"
      }
    },
    {
      "id": "span-marker-xlm-roberta-base-ar",
      "name": "span marker xlm roberta base ar",
      "type": "llm",
      "country": "INTL",
      "org": "iahlt",
      "license": "other",
      "modality": "text",
      "tasks": [
        "encoder",
        "ner"
      ],
      "links": {
        "hf": "https://huggingface.co/iahlt/span-marker-xlm-roberta-base-ar"
      },
      "year": 2023,
      "notes": "This is a SpanMarker model trained on the wikiann dataset that can be used for Named Entity Recognition.",
      "base_model": [
        "xlm-roberta-base",
        "facebookai/xlm-roberta-base"
      ],
      "metrics": {
        "downloads": 16,
        "likes": 4,
        "lastModified": "2023-11-24"
      }
    },
    {
      "id": "spark-tts-arabic",
      "name": "Spark TTS Arabic",
      "type": "tts",
      "country": "INTL",
      "org": "MrEzzat",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/MrEzzat/Spark_TTS_Arabic"
      },
      "year": 2025,
      "notes": "Spark-TTS is an advanced text-to-speech system that uses the power of large language models (LLM) for highly accurate and natural-sounding voice synthesis.",
      "base_model": [
        "sparkaudio/spark-tts-0.5b"
      ],
      "metrics": {
        "downloads": 16,
        "likes": 12,
        "lastModified": "2025-09-23"
      }
    },
    {
      "id": "thaqalayn-classical-arabic-english-parallel-texts",
      "name": "Thaqalayn Classical Arabic English Parallel texts",
      "type": "dataset",
      "country": "INTL",
      "org": "ImruQays",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "hadith"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ImruQays/Thaqalayn-Classical-Arabic-English-Parallel-texts"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "This dataset represents a comprehensive collection of parallel Arabic-English texts from the Thaqalayn Hadith Library.",
      "metrics": {
        "downloads": 16,
        "likes": 8,
        "lastModified": "2024-03-22"
      }
    },
    {
      "id": "vibevoice-egy",
      "name": "VibeVoice-Egy",
      "type": "tts",
      "country": "INTL",
      "org": "MAdel121",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "diacritization"
      ],
      "links": {
        "hf": "https://huggingface.co/MAdel121/VibeVoice-Egy"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Multi-speaker Egyptian Arabic fine-tune built on VibeVoice-Large (7B).",
      "base_model": [
        "aoi-ot/vibevoice-large"
      ],
      "metrics": {
        "downloads": 16,
        "likes": 2,
        "lastModified": "2026-03-02"
      }
    },
    {
      "id": "a7-ta",
      "name": "A7'ta",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "grammatical-analysis",
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/A-Monolingual-Arabic-Parallel-Corpus-",
        "hf": "https://huggingface.co/datasets/arbml/A7ta",
        "paper": "https://www.sciencedirect.com/science/article/pii/S2352340918315397"
      },
      "dialects": [
        "msa"
      ],
      "size": "300 documents",
      "year": 2019,
      "notes": "The data contains 300 documents, 445 erroneous sentences and their error-free counterparts, and a total of 3,532 words.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-02-26"
      }
    },
    {
      "id": "aladdinbench",
      "name": "AladdinBench",
      "type": "benchmark",
      "country": "INTL",
      "org": "palmaoui",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/palmaoui/AladdinBench",
        "paper": "https://arxiv.org/abs/2502.20973"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "notes": "Real-world Lebanese Arabizi messages for testing how well LLMs understand Arabizi.",
      "metrics": {
        "downloads": 15,
        "likes": 5,
        "lastModified": "2025-04-21"
      }
    },
    {
      "id": "aqad-arabic-question-answer-dataset",
      "name": "AQAD: Arabic Question-Answer dataset",
      "type": "dataset",
      "country": "EG",
      "org": "Alexandria University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/adelmeleka/AQAD",
        "hf": "https://huggingface.co/datasets/arbml/AQAD",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=9316526"
      },
      "dialects": [
        "msa"
      ],
      "size": "17,911 sentences",
      "year": 2020,
      "notes": "AQAD 17,000+ Arabic Questions & Answers dataset",
      "metrics": {
        "downloads": 15,
        "likes": 4,
        "lastModified": "2022-10-14"
      }
    },
    {
      "id": "arabic-egyptian-comparable-wikipedia-corpus",
      "name": "Arabic - Egyptian comparable Wikipedia corpus",
      "type": "dataset",
      "country": "PS",
      "org": "Islamic University of Gaza",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "pretraining"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/mksaad/arb-egy-cmp-corpus",
        "hf": "https://huggingface.co/datasets/arbml/Comparable_Wikipedia",
        "paper": "https://ieeexplore.ieee.org/document/8038320"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,197 documents",
      "year": 2017,
      "notes": "The dataset is composed of a set of text documents in both Arabic (Modern Standard) and Egyptian dialect aligned at document level. comparable documents.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2022-11-02"
      }
    },
    {
      "id": "arabic-egypt-english-world-facts",
      "name": "arabic egypt english world facts",
      "type": "dataset",
      "country": "INTL",
      "org": "miscovery",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/miscovery/arabic_egypt_english_world_facts"
      },
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "The World Facts General Knowledge Dataset (v2.0) is a high-quality, human-reviewed Q&A resource by Miscovery.",
      "metrics": {
        "downloads": 15,
        "likes": 13,
        "lastModified": "2025-04-26"
      }
    },
    {
      "id": "arabic-tweets-about-infectious-diseases",
      "name": "Arabic tweets about infectious diseases",
      "type": "dataset",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "http://www.research.lancs.ac.uk/portal/en/datasets/arabic-tweets-about-infectious-diseases(68b307a8-510b-42f6-93af-f0cc7d9a75a1).html",
        "hf": "https://huggingface.co/datasets/arbml/tweets_infectious_diseases"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,266 sentences",
      "year": 2019,
      "notes": "A dataset of 1266 tweets by two Arabic native speakers into five types of sources: academic, media, government, health professional, and public.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "arabic-with-ranked-hard-negatives",
      "name": "Arabic With Ranked Hard Negatives",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/Arabic-With-Ranked-Hard-Negatives"
      },
      "dialects": [
        "lev"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "The Arabic Hard Negative Dataset is derived from the Arabic subset of the Mr.",
      "metrics": {
        "downloads": 15,
        "likes": 3,
        "lastModified": "2024-11-25"
      }
    },
    {
      "id": "arabicsense",
      "name": "ArabicSense",
      "type": "benchmark",
      "country": "INTL",
      "org": "University of Luxembourg",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "evaluation",
        "commonsense"
      ],
      "links": {
        "github": "https://github.com/EL-Amrany/Arabic-Commonsense-Reasoning",
        "hf": "https://huggingface.co/datasets/Kamyar-zeinalipour/ArabicSense",
        "paper": "https://aclanthology.org/2025.wacl-1.1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,650 sentences",
      "year": 2025,
      "notes": "A synthetic commonsense reasoning benchmark in Arabic covering validation and explanation tasks.",
      "metrics": {
        "downloads": 15,
        "likes": 1,
        "lastModified": "2024-12-14"
      }
    },
    {
      "id": "aradata",
      "name": "araData",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/malek-hedhli/araData",
        "hf": "https://huggingface.co/datasets/arbml/arData"
      },
      "dialects": [
        "mixed",
        "gulf",
        "egy",
        "lev",
        "magh",
        "iraqi"
      ],
      "size": "12,231 sentences",
      "year": 2022,
      "notes": "AraData: clean, balanced sentence dataset covering 7 Arabic dialects.",
      "metrics": {
        "downloads": 15,
        "likes": 3,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "arbanking77",
      "name": "ArBanking77",
      "type": "llm",
      "country": "PS",
      "org": "SinaLab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/SinaLab/ArBanking77"
      },
      "dialects": [
        "gulf",
        "lev",
        "magh",
        "msa",
        "mixed"
      ],
      "year": 2024,
      "notes": "ArBanking77: Intent Detection Neural Model and a New Dataset in Modern and Dialectical Arabic ArBanking77 is an MSA and Dialectal Arabic Corpus.",
      "metrics": {
        "downloads": 15,
        "likes": 5,
        "lastModified": "2024-12-17"
      }
    },
    {
      "id": "comparable-wikipedia-coprus",
      "name": "Comparable Wikipedia Coprus",
      "type": "dataset",
      "country": "PS",
      "org": "Islamic University of Gaza",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/motazsaad/comparableWikiCoprus",
        "hf": "https://huggingface.co/datasets/arbml/comparable_arabizi",
        "paper": "https://ieeexplore.ieee.org/document/8038320"
      },
      "dialects": [
        "mixed",
        "msa",
        "egy"
      ],
      "size": "10,197 documents",
      "year": 2017,
      "notes": "Comparable Wikipedia Corpus (aligned documents) Corpus extracts from 20-01-2017 Wikipedia dumps",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-03-23"
      }
    },
    {
      "id": "elecmorocco2016",
      "name": "ElecMorocco2016",
      "type": "dataset",
      "country": "MA",
      "org": "Mohammed V University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/sentiprojects/ElecMorocco2016",
        "hf": "https://huggingface.co/datasets/arbml/ElecMorocco",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-319-66854-3_20"
      },
      "dialects": [
        "magh"
      ],
      "size": "10,000 sentences",
      "year": 2016,
      "notes": "A sentiment analysis dataset containing 10254 Arabic facebook comments about the Moroccan elections of 2016.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "fanar-math-r1-grpo",
      "name": "Fanar Math R1 GRPO",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "math",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Fanar-Math-R1-GRPO"
      },
      "base_model": [
        "qcri/fanar-1-9b-instruct"
      ],
      "year": 2025,
      "notes": "Fanar-1-9B-Instruct fine-tuned with GRPO for Arabic math reasoning.",
      "size": "9B",
      "metrics": {
        "downloads": 15,
        "likes": 3,
        "lastModified": "2025-06-17"
      }
    },
    {
      "id": "financesecondtrial",
      "name": "financesecondtrial",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "finance",
        "synthetic-data"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/omar-emad/financesecondtrial"
      },
      "dialects": [
        "msa"
      ],
      "size": "30 sentences",
      "year": 2025,
      "notes": "Synthetic finance dataset created with Llama 3.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "hassaniya-french-speech-translation",
      "name": "hassaniya-french-speech-translation",
      "type": "asr",
      "country": "INTL",
      "org": "Mamadou-Aw",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech-translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Mamadou-Aw/hassaniya-french-speech-translation"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "Speech-translation model from spoken Hassaniya Arabic to French.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2026-02-03"
      }
    },
    {
      "id": "kind",
      "name": "KIND",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "translation",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KIND-Dataset/KIND",
        "paper": "https://aclanthology.org/2024.eacl-srw.3.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "55,484 sentences",
      "year": 2024,
      "notes": "The KIND dataset consists of fine-grained Arabic dialect data collected through a social collaboration approach, emphasizing community involvement.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-03-19"
      }
    },
    {
      "id": "llama-2-7b-chat-ar",
      "name": "llama-2-7b-chat-ar",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/llama-2-7b-chat-ar"
      },
      "base_model": [
        "nousresearch/llama-2-7b-chat-hf"
      ],
      "size": "7B",
      "on_device": false,
      "year": 2023,
      "notes": "PEFT adapter on NousResearch/Llama-2-7b-chat-hf trained on HeshamHaroon/oasst1-ar-threads, Arabic OpenAssistant threads.",
      "metrics": {
        "downloads": 15,
        "likes": 4,
        "lastModified": "2024-01-18"
      }
    },
    {
      "id": "marsum-moroccan-articles-summarisation",
      "name": "MArSUM: Moroccan Articles Summarisation",
      "type": "dataset",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "github": "https://github.com/KamelGaanoun/MoroccanSummarization",
        "hf": "https://huggingface.co/datasets/arbml/MArSum",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-031-06458-6_13"
      },
      "dialects": [
        "magh"
      ],
      "size": "20,000 sentences",
      "year": 2022,
      "notes": "MArSUM is the first open corpus destinated for Moroccan dialect text summarization.",
      "metrics": {
        "downloads": 15,
        "likes": 1,
        "lastModified": "2022-10-14"
      }
    },
    {
      "id": "mpold-multi-platforms-offensive-language-dataset",
      "name": "MPOLD: Multi Platforms Offensive Language Dataset",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "github": "https://github.com/shammur/Arabic-Offensive-Multi-Platform-SocialMedia-Comment-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/MPOLD",
        "paper": "https://aclanthology.org/2020.lrec-1.761.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "400 documents",
      "year": 2020,
      "notes": "Arabic Offensive Comments dataset from Multiple Social Media Platforms",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "palmx-ic",
      "name": "PalmX-IC",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "other",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/palmx_2025_subtask2_islamic",
        "paper": "https://doi.org/10.18653/v1/2025.arabicnlp-sharedtasks.107"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,900 sentences",
      "year": 2025,
      "notes": "PalmX-IC assesses a model’s knowledge of Islamic culture—rituals, Qurʾān verses, Ḥadīth, historic events, jurisprudence, and religious holidays—core.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2025-10-07"
      }
    },
    {
      "id": "snad",
      "name": "SNAD",
      "type": "dataset",
      "country": "SA",
      "org": "Prince Sultan University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "news"
      ],
      "links": {
        "website": "https://drive.google.com/file/d/1uwD56jaVIbsQQWVqqyL08TgjuTraYFJC/view",
        "hf": "https://huggingface.co/datasets/arbml/SNAD",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-030-55180-3_47"
      },
      "dialects": [
        "msa"
      ],
      "size": "45,000 documents",
      "year": 2020,
      "notes": "Arabic news dataset scraped from two of the most popular Saudi Arabian news sources.",
      "metrics": {
        "downloads": 15,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "talaa",
      "name": "TALAA",
      "type": "dataset",
      "country": "DZ",
      "org": "Université des Sciences et de la Technologie Houari Boumediene",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification",
        "news"
      ],
      "links": {
        "github": "https://github.com/saidziani/Arabic-News-Article-Classification",
        "hf": "https://huggingface.co/datasets/arbml/TALAA",
        "paper": "https://www.scitepress.org/Papers/2015/53521/53521.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "57,827 documents",
      "year": 2015,
      "notes": "TALAA, a general categorized Arabic corpus built from daily newspaper websites at USTHB.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2022-10-24"
      }
    },
    {
      "id": "whisper-small-fleurs-ar-heshamharoon",
      "name": "whisper-small-with-google-fleurs-ar",
      "type": "asr",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/whisper-small-with-google-fleurs-ar"
      },
      "size": "small",
      "year": 2024,
      "notes": "Whisper small fine-tuned on the Google FLEURS Arabic split; the card gives no further details.",
      "metrics": {
        "downloads": 15,
        "likes": 0,
        "lastModified": "2024-02-08"
      }
    },
    {
      "id": "akec",
      "name": "AKEC",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Udine",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "keyphrase-extraction"
      ],
      "links": {
        "github": "https://github.com/ailab-uniud/akec",
        "hf": "https://huggingface.co/datasets/arbml/AKEC",
        "paper": "https://ieeexplore.ieee.org/document/7875927"
      },
      "dialects": [
        "msa"
      ],
      "size": "160 documents",
      "year": 2016,
      "notes": "The corpus consists in 160 arabic documents and their keyphrases.",
      "metrics": {
        "downloads": 14,
        "likes": 0,
        "lastModified": "2022-11-02"
      }
    },
    {
      "id": "aoc",
      "name": "AOC",
      "type": "dataset",
      "country": "INTL",
      "org": "Johns Hopkins University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/sjeblee/AOC",
        "hf": "https://huggingface.co/datasets/arbml/annotated_aoc",
        "paper": "https://aclanthology.org/P11-2007.pdf"
      },
      "dialects": [
        "mixed",
        "msa"
      ],
      "size": "108,173 sentences",
      "year": 2011,
      "notes": "A 52M-word monolingual dataset rich in dialectal content",
      "metrics": {
        "downloads": 14,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "arabic-cultural-mcq-dataset",
      "name": "Arabic Cultural MCQ Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Raniahossam33/Arabic_cultural_dataset"
      },
      "dialects": [
        "mixed",
        "egy",
        "gulf",
        "magh",
        "lev"
      ],
      "size": "12,072 sentences",
      "year": 2025,
      "notes": "This dataset contains culturally-aware multiple-choice questions (MCQs) in various Arabic dialects, designed to test and evaluate language models'.",
      "metrics": {
        "downloads": 14,
        "likes": 2,
        "lastModified": "2025-09-08"
      }
    },
    {
      "id": "arabic-prompts-mini-175",
      "name": "Arabic prompts Mini 175",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Arabic_prompts_Mini_175"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "Small collection of Arabic prompts across literature, science, technology and culture for NLP experiments.",
      "metrics": {
        "downloads": 14,
        "likes": 8,
        "lastModified": "2024-07-24"
      }
    },
    {
      "id": "arabicquoraduplicates-stsb-alue-holyquran-aranli-900k-anchor-positive",
      "name": "ArabicQuoraDuplicates_stsb_Alue_holyquran_aranli_900k_anchor_positive_negative",
      "type": "dataset",
      "country": "INTL",
      "org": "AbderrahmanSkiredj1",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "triplets"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/AbderrahmanSkiredj1/ArabicQuoraDuplicates_stsb_Alue_holyquran_aranli_900k_anchor_positive_negative"
      },
      "size": "100K–1M rows",
      "year": 2024,
      "notes": "873K Arabic embedding triplets (anchor, positive, negative) mixing Quora duplicates, STSB, ALUE, Quran and AraNLI sources.",
      "metrics": {
        "downloads": 14,
        "likes": 3,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "diraya-3b-instruct-ar",
      "name": "Diraya 3B Instruct Ar",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/Diraya-3B-Instruct-Ar"
      },
      "base_model": [
        "unsloth/qwen2.5-3b-instruct-unsloth-bnb-4bit"
      ],
      "size": "3B",
      "on_device": true,
      "year": 2025,
      "notes": "3B Arabic instruction model from the DIRA (Diraya Arabic Reasoning AI) collection.",
      "metrics": {
        "downloads": 14,
        "likes": 3,
        "lastModified": "2025-03-15"
      }
    },
    {
      "id": "emirati-fastpitch-bilingual-v1-0",
      "name": "emirati fastpitch bilingual v1.0",
      "type": "tts",
      "country": "AE",
      "org": "Vadim Belsky",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/emirati-fastpitch-bilingual-v1.0"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Designed for natural prosody, dialectal Emirati pronunciation, and seamless AR/EN mixing in real-world speech.",
      "base_model": [
        "vadimbelsky/emirati-fastpitch-bilingual-v1.0"
      ],
      "metrics": {
        "downloads": 14,
        "likes": 20,
        "lastModified": "2026-01-04"
      }
    },
    {
      "id": "mbart-large-cc25-ar-en",
      "name": "mbart large cc25 ar en",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/mbart-large-cc25-ar-en"
      },
      "year": 2020,
      "notes": "This is mbart-large-cc25, finetuned on a subset of the OPUS corpus for aren.",
      "metrics": {
        "downloads": 14,
        "likes": 4,
        "lastModified": "2020-12-11"
      }
    },
    {
      "id": "mbart-large-cc25-en-ar",
      "name": "mbart large cc25 en ar",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/mbart-large-cc25-en-ar"
      },
      "year": 2020,
      "notes": "This is mbart-large-cc25, finetuned on a subset of the UN corpus for enar.",
      "metrics": {
        "downloads": 14,
        "likes": 3,
        "lastModified": "2020-12-11"
      }
    },
    {
      "id": "morrbert",
      "name": "MorrBERT",
      "type": "llm",
      "country": "INTL",
      "org": "otmangi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/otmangi/MorrBERT"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "MorrBERT is a Transformer-based Language Model designed specifically for the Moroccan Dialect.",
      "metrics": {
        "downloads": 14,
        "likes": 1,
        "lastModified": "2023-07-01"
      }
    },
    {
      "id": "osac",
      "name": "OSAC",
      "type": "dataset",
      "country": "PS",
      "org": "Islamic University of Gaza",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/ar-text-mining/files/Arabic-Corpora/",
        "hf": "https://huggingface.co/datasets/arbml/OSAC_CNN",
        "paper": "http://site.iugaza.edu.ps/wp-content/uploads/mksaad-OSAC-OpenSourceArabicCorpora-EECS10-rev9(1).pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "22,429 documents",
      "year": 2010,
      "notes": "Collecting the largest free accessible Arabic corpus, OSAC, which contains about 18M words and about 0.5M district keywords.",
      "metrics": {
        "downloads": 14,
        "likes": 1,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "poetry2023",
      "name": "poetry2023",
      "type": "llm",
      "country": "INTL",
      "org": "akhooli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "poetry"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/poetry2023"
      },
      "year": 2023,
      "notes": "Fine-tuned model of Arabic poetry dataset based on aragpt2-medium.",
      "metrics": {
        "downloads": 14,
        "likes": 3,
        "lastModified": "2023-03-20"
      }
    },
    {
      "id": "sudanese-dialect-tweets-about-telecommunication-companies",
      "name": "Sudanese Dialect tweets about telecommunication companies",
      "type": "dataset",
      "country": "SD",
      "org": "University of Khartoum",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://docs.google.com/spreadsheets/d/13fIV8oHss-QRBKN-2h5LYF1i_1O9qH1R/edit?usp=sharing&ouid=101796411348671465142&rtpof=true&sd=true",
        "hf": "https://huggingface.co/datasets/arbml/Sudanese_Dialect_Tweet_Tele",
        "paper": "https://ieeexplore.ieee.org/document/8515862"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "4,712 sentences",
      "year": 2018,
      "notes": "Sentiment Analysis dataset written in Sudanese Arabic Dialect",
      "metrics": {
        "downloads": 14,
        "likes": 1,
        "lastModified": "2024-07-18"
      }
    },
    {
      "id": "anad-arabic-natural-audio-dataset",
      "name": "ANAD: Arabic Natural Audio Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "emotion"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/xm232yxf7t/1",
        "hf": "https://huggingface.co/datasets/arbml/ANAD"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,384 hours",
      "year": 2018,
      "notes": "Eight videos of live calls between an anchor and a human outside the studio were downloaded from online Arabic talk shows.",
      "metrics": {
        "downloads": 13,
        "likes": 0,
        "lastModified": "2022-10-31"
      }
    },
    {
      "id": "arabart-zaebuc-gec-ged-13",
      "name": "arabart zaebuc gec ged 13",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "gec"
      ],
      "links": {
        "hf": "https://huggingface.co/CAMeL-Lab/arabart-zaebuc-gec-ged-13"
      },
      "year": 2024,
      "notes": "MSA grammatical error correction model, AraBART fine-tuned with morphology features on QALB and ZAEBUC.",
      "dialects": [
        "msa"
      ],
      "metrics": {
        "downloads": 13,
        "likes": 2,
        "lastModified": "2024-01-09"
      }
    },
    {
      "id": "arabic-named-entities",
      "name": "Arabic Named Entities",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Groningen",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/arabicnes/",
        "hf": "https://huggingface.co/datasets/arbml/Arabic_Named_Entities",
        "paper": "http://doras.dcu.ie/15979/1/An_automatically_built_Named_Entity_lexicon_for_Arabic.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "45,000 tokens",
      "year": 2010,
      "notes": "Extracted approximately 45,000 Arabic NEs and built the largest, most mature and well-structured Arabic NE lexical resource to date",
      "metrics": {
        "downloads": 13,
        "likes": 0,
        "lastModified": "2022-11-05"
      }
    },
    {
      "id": "arabichar-v3",
      "name": "arabichar v3",
      "type": "ocr",
      "country": "INTL",
      "org": "asyafalni",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "handwriting",
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/asyafalni/arabichar-v3"
      },
      "year": 2023,
      "notes": "Custom CNN classifying handwritten Arabic (Hijaiyah) characters, 97.6% accuracy on its evaluation set.",
      "metrics": {
        "downloads": 13,
        "likes": 4,
        "lastModified": "2023-10-04"
      }
    },
    {
      "id": "belebele-arabic",
      "name": "belebele arabic",
      "type": "benchmark",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "reading-comprehension",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/belebele_arabic"
      },
      "size": "1K–10K rows",
      "year": 2023,
      "notes": "Arabic subset of Belebele, a 5,400-question multiple-choice reading comprehension benchmark with FLORES passages.",
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2023-09-05"
      }
    },
    {
      "id": "chatterbox-multilingual-finetuned-arabic",
      "name": "chatterbox multilingual finetuned arabic",
      "type": "tts",
      "country": "INTL",
      "org": "juliardi",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/juliardi/chatterbox-multilingual-finetuned-arabic"
      },
      "year": 2026,
      "notes": "This is an Arabic-focused fine-tuned version of ResembleAI/chatterbox multilingual model using LoRA (Low-Rank Adaptation).",
      "base_model": [
        "resembleai/chatterbox"
      ],
      "metrics": {
        "downloads": 13,
        "likes": 5,
        "lastModified": "2026-01-16"
      }
    },
    {
      "id": "hamsa-v0-1-beta",
      "name": "hamsa v0.1 beta",
      "type": "asr",
      "country": "INTL",
      "org": "nadsoft",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/nadsoft/hamsa-v0.1-beta"
      },
      "year": 2023,
      "notes": "Hamsa (همسة) represents a sophisticated advancement in the realm of Arabic speech recognition.",
      "metrics": {
        "downloads": 13,
        "likes": 6,
        "lastModified": "2023-11-20"
      }
    },
    {
      "id": "hearbert",
      "name": "HeArBERT",
      "type": "llm",
      "country": "INTL",
      "org": "aviadrom",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder",
        "embedding",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aviadrom/HeArBERT"
      },
      "year": 2024,
      "notes": "A bilingual BERT for Arabic and Hebrew, pretrained on the respective parts of the OSCAR corpus.",
      "metrics": {
        "downloads": 13,
        "likes": 4,
        "lastModified": "2024-03-11"
      }
    },
    {
      "id": "jordanian-to-fusha-model",
      "name": "jordanian-to-fusha-model",
      "type": "llm",
      "country": "INTL",
      "org": "faresalawneh",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/faresalawneh/jordanian-to-fusha-model"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "Seq2seq model that translates Jordanian Arabic into Modern Standard Arabic.",
      "base_model": [
        "facebook/nllb-200-distilled-600m"
      ],
      "metrics": {
        "downloads": 13,
        "likes": 0,
        "lastModified": "2026-03-21"
      }
    },
    {
      "id": "kalamdz",
      "name": "KalamDZ",
      "type": "dataset",
      "country": "DZ",
      "org": "Université Amar Telidji Laghouat",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/LIM-MoDos/KalamDZ",
        "hf": "https://huggingface.co/datasets/arbml/KalamDZ",
        "paper": "https://aclanthology.org/W17-1317.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "104 hours",
      "year": 2017,
      "notes": "8 major Algerian Arabic sub-dialects with 4881 speakers and more than 104.4 hours segmented in utterances of at least 6 s",
      "metrics": {
        "downloads": 13,
        "likes": 0,
        "lastModified": "2022-11-01"
      }
    },
    {
      "id": "lhv-morocco",
      "name": "LHV-Morocco",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/nilechat-lhv-mor",
        "paper": "https://aclanthology.org/2025.emnlp-main.556.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "761,041 documents",
      "year": 2025,
      "notes": "LHV-Morocco is a substantial dataset specifically developed to foster the creation and improvement of language models for the Moroccan Arabic dialect.",
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2025-11-11"
      }
    },
    {
      "id": "pearl-x",
      "name": "PEARL X",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-nd-4.0",
      "modality": "multimodal",
      "tasks": [
        "culture",
        "vqa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/UBC-NLP/PEARL-X"
      },
      "size": "<1K rows",
      "year": 2025,
      "notes": "PEARL-X: benchmark of single- and multi-image QA on shared cultural concepts that differ across Arab contexts.",
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "viobert-v3",
      "name": "vioBERT-v3",
      "type": "llm",
      "country": "INTL",
      "org": "Vionex-digital",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder",
        "medical"
      ],
      "links": {
        "hf": "https://huggingface.co/Vionex-digital/vioBERT-v3"
      },
      "year": 2026,
      "notes": "Arabic medical BERT: MARBERTv2 continued on about 1.12M medical documents, cutting medical perplexity 82.7%.",
      "base_model": [
        "ubc-nlp/marbertv2"
      ],
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2026-04-23"
      }
    },
    {
      "id": "whisper-base-ar-quran-ft-hijaiyah-2",
      "name": "whisper base ar quran ft hijaiyah 2",
      "type": "asr",
      "country": "INTL",
      "org": "ojisetyawan",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/ojisetyawan/whisper-base-ar-quran-ft-hijaiyah-2"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Arabic audio classification model fine-tuned from tarteel-ai/whisper-base-ar-quran trained on Reinjin/Pelafalan_Huruf_Hijaiyah.",
      "base_model": [
        "tarteel-ai/whisper-base-ar-quran"
      ],
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2024-12-02"
      }
    },
    {
      "id": "whisper-medium-arabic-suite-ii",
      "name": "whisper medium arabic suite II",
      "type": "asr",
      "country": "INTL",
      "org": "Seyfelislem",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/Seyfelislem/whisper-medium-arabic-suite-II"
      },
      "year": 2023,
      "notes": "Arabic automatic speech recognition model trained on mozilla-foundation/common_voice_11_0.",
      "metrics": {
        "downloads": 13,
        "likes": 3,
        "lastModified": "2023-04-30"
      }
    },
    {
      "id": "xtts-omani-qlora-merged",
      "name": "xtts-omani-qlora-merged",
      "type": "tts",
      "country": "INTL",
      "org": "AhmedHamedElheity",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedHamedElheity/xtts-omani-qlora-merged"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "notes": "XTTS fine-tuned with QLoRA (merged weights) for Omani Arabic speech synthesis.",
      "metrics": {
        "downloads": 13,
        "likes": 0,
        "lastModified": "2025-09-29"
      }
    },
    {
      "id": "adab",
      "name": "ADAB",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/IWAN/adab-arabic-politeness",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2026/pdf/2026.lrec2026-1.244.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2026,
      "notes": "Large-scale Arabic dataset for automated politeness benchmarking with 10K texts across multiple domains and dialects.",
      "metrics": {
        "downloads": 12,
        "likes": 1,
        "lastModified": "2026-06-27"
      }
    },
    {
      "id": "aqmar",
      "name": "AQMAR",
      "type": "dataset",
      "country": "INTL",
      "org": "Carnegie Mellon University",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "https://www.cs.cmu.edu/~ark/ArabicNER/",
        "hf": "https://huggingface.co/datasets/arbml/AQMAR",
        "paper": "https://aclanthology.org/E12-1017.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "74,000 tokens",
      "year": 2012,
      "notes": "This is a 74,000-token corpus of 28 Arabic Wikipedia articles hand-annotated for named entities.",
      "metrics": {
        "downloads": 12,
        "likes": 0,
        "lastModified": "2024-07-20"
      }
    },
    {
      "id": "arabic-reasoning-qa",
      "name": "ARabic Reasoning QA",
      "type": "dataset",
      "country": "INTL",
      "org": "MohammedNasser",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "reasoning",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MohammedNasser/ARabic_Reasoning_QA"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Arabic reasoning questions at beginner, intermediate and advanced levels for training and evaluating reasoning.",
      "metrics": {
        "downloads": 12,
        "likes": 7,
        "lastModified": "2024-09-07"
      }
    },
    {
      "id": "arastyletransfer-21",
      "name": "AraStyleTransfer-21",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "style-transfer"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/AraStyleTransfer-21"
      },
      "year": 2025,
      "notes": "Style-transfer model rewriting text in the style of 21 Arabic authors; first place at AraGenEval 2025.",
      "base_model": [
        "ubc-nlp/arat5v2-base-1024"
      ],
      "metrics": {
        "downloads": 12,
        "likes": 3,
        "lastModified": "2025-11-26"
      }
    },
    {
      "id": "awesome-arabic-chatgpt-prompts",
      "name": "Awesome Arabic Chatgpt Prompts",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Omartificial-Intelligence-Space/awesome_chatgpt_prompts_ar"
      },
      "dialects": [
        "msa"
      ],
      "size": "201 sentences",
      "year": 2025,
      "notes": "Contains a collection of Arabic prompts designed for use with AI language models (such as ChatGPT).",
      "metrics": {
        "downloads": 12,
        "likes": 1,
        "lastModified": "2025-10-05"
      }
    },
    {
      "id": "gpt-oss-math-ar",
      "name": "gpt oss math ar",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "math",
        "reasoning"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/gpt-oss-math-ar"
      },
      "year": 2025,
      "notes": "gpt-oss-20B LoRA fine-tune for step-by-step Arabic math solving on GSM8K-style problems.",
      "size": "20B",
      "base_model": [
        "unsloth/gpt-oss-20b-unsloth-bnb-4bit"
      ],
      "metrics": {
        "downloads": 12,
        "likes": 3,
        "lastModified": "2025-08-09"
      }
    },
    {
      "id": "lebanon-uprising-arabic-tweets",
      "name": "Lebanon Uprising Arabic Tweets",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "tweets",
        "social-media"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/abedkhooli/lebanon-uprising-october-2019-tweets",
        "hf": "https://huggingface.co/datasets/arbml/Lebanon_Uprising_Arabic_Tweets"
      },
      "dialects": [
        "mixed"
      ],
      "size": "100,000 sentences",
      "year": 2019,
      "tags": [
        "multilingual"
      ],
      "notes": "Tweets under the Arabic hashtag #لبنان_ينتفض about the October 2019 Lebanon uprising.",
      "metrics": {
        "downloads": 12,
        "likes": 0,
        "lastModified": "2024-05-10"
      }
    },
    {
      "id": "medical-corpus",
      "name": "Medical Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "CRSTDLA",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "disease-identification"
      ],
      "links": {
        "github": "https://github.com/licvol/Arabic-Spoken-Language-Understanding/tree/master/MedicalCorpus",
        "hf": "https://huggingface.co/datasets/arbml/MedicalCorpus",
        "paper": "https://aclanthology.org/W19-7407.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "152 sentences",
      "year": 2019,
      "notes": "Corpus from a medical care forum known as Doctissimo",
      "metrics": {
        "downloads": 12,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "nsurl-2019-shared-task-8",
      "name": "NSURL-2019 Shared Task 8",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "sts"
      ],
      "links": {
        "website": "https://ai.mawdoo3.com/nsurl-2019-task8",
        "hf": "https://huggingface.co/datasets/arbml/nsurl_2019_task8_train",
        "paper": "https://aclanthology.org/2019.nsurl-1.1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "15,712 sentences",
      "year": 2019,
      "notes": "This dataset is composed of 12000 question pairs labelled with 1 for semantically similar questions and 0 for semantically different",
      "metrics": {
        "downloads": 12,
        "likes": 0,
        "lastModified": "2024-04-26"
      }
    },
    {
      "id": "oasst-arabic",
      "name": "oasst arabic",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/oasst-arabic"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "Arabic translation of the OASST conversation dataset with message, parent, role and review fields.",
      "metrics": {
        "downloads": 12,
        "likes": 2,
        "lastModified": "2023-12-28"
      }
    },
    {
      "id": "twt15da-lists",
      "name": "Twt15DA_Lists",
      "type": "dataset",
      "country": "INTL",
      "org": "Taif University",
      "license": "cc-by-nc-nd-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/Maha-J-Althobaiti/Twt15DA_Lists",
        "hf": "https://huggingface.co/datasets/arbml/Twt15DA_Lists",
        "paper": "https://web.archive.org/web/20210813220628id_/https://www.cambridge.org/core/services/aop-cambridge-core/content/view/2DE64B777EF0277557AFA90E2BB75B62/S135132492100019Xa.pdf/div-class-title-creation-of-annotated-country-level-dialectal-arabic-resources-an-unsupervised-approach-div.pdf"
      },
      "dialects": [
        "mixed",
        "yemeni",
        "gulf",
        "iraqi",
        "lev",
        "egy"
      ],
      "size": "312,829 sentences",
      "year": 2021,
      "notes": "The annotated dialectal Arabic corpus (Twt15DA) is collected from Twitter and consists of 311,785 tweets containing 3,858,459 words in total.",
      "metrics": {
        "downloads": 12,
        "likes": 0,
        "lastModified": "2024-07-14"
      }
    },
    {
      "id": "arabic-business-corpora",
      "name": "Arabic business corpora",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos",
        "classification"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/arabic-business-copora/",
        "hf": "https://huggingface.co/datasets/arbml/buisness_corpora"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,200 documents",
      "year": 2016,
      "notes": "The main corpora contains 1200 articles.",
      "metrics": {
        "downloads": 11,
        "likes": 0,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "arcorona-analyzing-arabic-tweets-in-the-early-days-of-coronavirus-covi",
      "name": "ArCorona: Analyzing Arabic Tweets in the Early Days of Coronavirus (COVID-19) Pandemic",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "covid",
        "rumors"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/ArCorona.tsv",
        "hf": "https://huggingface.co/datasets/arbml/ArCorona",
        "paper": "https://www.aclweb.org/anthology/2021.louhi-1.1/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "8,000 sentences",
      "year": 2020,
      "notes": "Arabic tweets from the early COVID-19 pandemic, collected to study rumors and misinformation about the virus and bad cures.",
      "metrics": {
        "downloads": 11,
        "likes": 0,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "blip-arabic-flickr-8k",
      "name": "blip-Arabic-flickr-8k",
      "type": "llm",
      "country": "INTL",
      "org": "omarsabri8756",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "image-captioning"
      ],
      "links": {
        "hf": "https://huggingface.co/omarsabri8756/blip-Arabic-flickr-8k"
      },
      "notes": "BLIP fine-tuned for Arabic image captioning",
      "metrics": {
        "downloads": 11,
        "likes": 1,
        "lastModified": "2025-05-10"
      }
    },
    {
      "id": "dvoice-darija",
      "name": "dvoice darija",
      "type": "asr",
      "country": "INTL",
      "org": "aioxlabs",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/aioxlabs/dvoice-darija"
      },
      "dialects": [
        "magh"
      ],
      "year": 2022,
      "notes": "This repository provides all the necessary tools to perform automatic speech recognition from an end-to-end system pretrained.",
      "metrics": {
        "downloads": 11,
        "likes": 8,
        "lastModified": "2022-05-28"
      }
    },
    {
      "id": "egy-instruct",
      "name": "Egy_instruct",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Egy_instruct"
      },
      "dialects": [
        "egy"
      ],
      "year": 2024,
      "notes": "Egyptian Arabic instruction dataset (10K-100K rows) in a single parquet file; the card gives no further details.",
      "metrics": {
        "downloads": 11,
        "likes": 0,
        "lastModified": "2024-06-03"
      }
    },
    {
      "id": "egyptian-english-parallel",
      "name": "Egyptian English parallel",
      "type": "dataset",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/HeshamHaroon/Egyptian_English_parallel"
      },
      "dialects": [
        "egy"
      ],
      "size": "<1K rows",
      "year": 2023,
      "notes": "892 Egyptian Arabic-English sentence pairs.",
      "metrics": {
        "downloads": 11,
        "likes": 9,
        "lastModified": "2023-08-03"
      }
    },
    {
      "id": "financetripletsecond",
      "name": "FinanceTripletSecond",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "finance",
        "synthetic-data"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/omar-emad/FinanceTripletSecond"
      },
      "dialects": [
        "msa"
      ],
      "size": "30 sentences",
      "year": 2025,
      "notes": "Finance triplet dataset generated with a Llama 3 synthetic data pipeline.",
      "metrics": {
        "downloads": 11,
        "likes": 0,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "fw-darija",
      "name": "fw darija",
      "type": "dataset",
      "country": "INTL",
      "org": "sawalni-ai",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/sawalni-ai/fw-darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Darija subset of FineWeb 2, re-classified and cleaned with the Gherbal language identifier.",
      "metrics": {
        "downloads": 11,
        "likes": 10,
        "lastModified": "2024-12-08"
      }
    },
    {
      "id": "hadiths-dataset",
      "name": "hadiths dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "cibfaye",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/cibfaye/hadiths_dataset"
      },
      "dialects": [
        "classical"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "37K hadith rows with book, reference, narrator and English and Arabic text.",
      "metrics": {
        "downloads": 11,
        "likes": 5,
        "lastModified": "2024-05-20"
      }
    },
    {
      "id": "paad",
      "name": "PAAD",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Technology",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/spvbf5bgjs/2",
        "hf": "https://huggingface.co/datasets/arbml/PAAD",
        "paper": "https://ijci.uoitc.edu.iq/index.php/ijci/article/view/246/174"
      },
      "dialects": [
        "msa"
      ],
      "size": "206 documents",
      "year": 2020,
      "notes": "The dataset is 206 articles distributed into three categories as (Reform, Conservative and Revolutionary) that we offer to the research community on Arabic.",
      "metrics": {
        "downloads": 11,
        "likes": 0,
        "lastModified": "2024-07-19"
      }
    },
    {
      "id": "saudi-tts",
      "name": "saudi tts",
      "type": "tts",
      "country": "SA",
      "org": "Ahmed Eladl",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedEladl/saudi-tts"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "notes": "A high-quality Text-to-Speech model for Saudi Arabic dialect.",
      "metrics": {
        "downloads": 11,
        "likes": 21,
        "lastModified": "2025-12-15"
      }
    },
    {
      "id": "uae-laws",
      "name": "uae laws",
      "type": "dataset",
      "country": "INTL",
      "org": "obadabaq",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "legal",
        "qa"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/obadabaq/uae-laws"
      },
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Collection of UAE laws and regulations (business, finance, family, justice, labour, residency) for QA and text generation.",
      "metrics": {
        "downloads": 11,
        "likes": 4,
        "lastModified": "2024-07-24"
      }
    },
    {
      "id": "whisper-large-arabic-cv-11",
      "name": "whisper large arabic cv 11",
      "type": "asr",
      "country": "INTL",
      "org": "KalamTech",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/KalamTech/whisper-large-arabic-cv-11"
      },
      "year": 2024,
      "notes": "نموذج كلام للتعرف على الصوت، هذا النموذج يتميز بدقة عالية في التعرف على الصوت باللغة العربية.",
      "base_model": [
        "openai/whisper-large"
      ],
      "metrics": {
        "downloads": 11,
        "likes": 4,
        "lastModified": "2024-07-30"
      }
    },
    {
      "id": "bahraini-arabic-whisper",
      "name": "Whisper-base Bahraini",
      "type": "asr",
      "country": "BH",
      "org": "Community (Bahrain)",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Fatimaa75/whisper-base-bahraini"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Whisper base fine-tuned on Bahraini-dialect speech.",
      "base_model": [
        "openai/whisper-base"
      ],
      "metrics": {
        "downloads": 11,
        "likes": 1,
        "lastModified": "2026-05-03"
      }
    },
    {
      "id": "al-atlas-llm-0-5b",
      "name": "Al Atlas LLM 0.5B",
      "type": "llm",
      "country": "MA",
      "org": "BounharAbdelaziz",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/BounharAbdelaziz/Al-Atlas-LLM-0.5B"
      },
      "base_model": [
        "qwen/qwen2.5-0.5b"
      ],
      "dialects": [
        "magh"
      ],
      "size": "0.5B",
      "on_device": true,
      "year": 2025,
      "tags": [
        "variants:1"
      ],
      "notes": "Qwen2.5-0.5B continued on the AL-Atlas Moroccan Darija pretraining dataset.",
      "metrics": {
        "downloads": 10,
        "likes": 3,
        "lastModified": "2025-03-05"
      }
    },
    {
      "id": "arabic-colbertv2-711k-norm",
      "name": "arabic colbertv2 711k norm",
      "type": "embedding",
      "country": "INTL",
      "org": "akhooli",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/akhooli/arabic-colbertv2-711k-norm"
      },
      "year": 2024,
      "notes": "ColBERT v2 Arabic retrieval model trained on 711K queries with Latin-word queries removed; only partly trained (22K steps).",
      "metrics": {
        "downloads": 10,
        "likes": 4,
        "lastModified": "2024-08-10"
      }
    },
    {
      "id": "arabic-dataset-for-capt",
      "name": "Arabic-Dataset-for-CAPT",
      "type": "dataset",
      "country": "INTL",
      "org": "Khaled Necibi & Halima Bahi",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/bhalima/Arabic-Dataset-for-CAPT",
        "hf": "https://huggingface.co/datasets/arbml/CAPT",
        "paper": "https://link.springer.com/article/10.1007/s10772-014-9248-2"
      },
      "dialects": [
        "msa"
      ],
      "size": "143 sentences",
      "year": 2015,
      "notes": "The dataset includes both “correct” and “wrong” non-artificial pronunciations.",
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2022-10-31"
      }
    },
    {
      "id": "arsenl",
      "name": "ArSenL",
      "type": "dataset",
      "country": "LB",
      "org": "American University of Beirut",
      "license": "other",
      "modality": "text",
      "tasks": [
        "pos",
        "sentiment"
      ],
      "links": {
        "website": "http://oma-project.com/ArSenL/download_intro",
        "hf": "https://huggingface.co/datasets/arbml/ArSenL",
        "paper": "https://aclanthology.org/W14-3623.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "28,760 tokens",
      "year": 2014,
      "notes": "Large scale Standard Arabic sentiment lexicon (ArSenL) using a combination of existing resources: ESWN, Arabic WordNet, and the Standard Arabic.",
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "darija-lid-benchmark-v1",
      "name": "darija lid benchmark v1",
      "type": "benchmark",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/darija-lid-benchmark-v1"
      },
      "dialects": [
        "magh"
      ],
      "size": "1M–10M rows",
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Language-identification benchmark for telling Moroccan Darija apart from other languages and Arabic varieties.",
      "metrics": {
        "downloads": 10,
        "likes": 2,
        "lastModified": "2026-06-13"
      }
    },
    {
      "id": "dataset-for-evaluating-root-extraction",
      "name": "Dataset for evaluating root extraction",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "root-extraction",
        "stemming"
      ],
      "links": {
        "github": "https://github.com/arabic-digital-humanities/root-extraction-validation-data",
        "hf": "https://huggingface.co/datasets/arbml/root_extraction_vaidation"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,962 tokens",
      "year": 2018,
      "notes": "Data for evaluating the roots extracted by Arabic stemmers and morphological analyzers.",
      "metrics": {
        "downloads": 10,
        "likes": 1,
        "lastModified": "2022-10-23"
      }
    },
    {
      "id": "flair-arabic-dialects-codeswitch-egy-lev",
      "name": "flair-arabic-dialects-codeswitch-egy-lev",
      "type": "tool",
      "country": "INTL",
      "org": "megantosh",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pos",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/megantosh/flair-arabic-dialects-codeswitch-egy-lev"
      },
      "dialects": [
        "egy",
        "lev"
      ],
      "year": 2022,
      "notes": "Flair and fastText part-of-speech tagger for Egyptian and Levantine code-switched Arabic.",
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2022-03-09"
      }
    },
    {
      "id": "moroccan-lyrics",
      "name": "moroccan lyrics",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/moroccan_lyrics"
      },
      "dialects": [
        "magh"
      ],
      "size": "<1K rows",
      "year": 2024,
      "notes": "A compilation of morocccan songs lyrics both in arabic and latin scripts (arabizi).",
      "metrics": {
        "downloads": 10,
        "likes": 2,
        "lastModified": "2024-11-03"
      }
    },
    {
      "id": "osman-readability-metric",
      "name": "OSMAN readability metric",
      "type": "tool",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "github": "https://github.com/drelhaj/OsmanReadability",
        "hf": "https://huggingface.co/datasets/arbml/Osman_Un_Corpus"
      },
      "year": 2016,
      "notes": "Arabic readability metric analogous to Flesch, with parallel Arabic-English corpus.",
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2022-11-02"
      }
    },
    {
      "id": "shamibert",
      "name": "ShamiBERT",
      "type": "llm",
      "country": "INTL",
      "org": "mabahboh",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/mabahboh/ShamiBERT"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "BERT encoder model for Levantine (Shami) Arabic.",
      "base_model": [
        "aubmindlab/bert-base-arabertv02-twitter"
      ],
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2026-03-14"
      }
    },
    {
      "id": "tunisian-english-parallel-pairs",
      "name": "tunisian english parallel pairs",
      "type": "dataset",
      "country": "INTL",
      "org": "KKKarim711",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/KKKarim711/tunisian-english-parallel-pairs"
      },
      "dialects": [
        "magh"
      ],
      "size": "100K–1M rows",
      "year": 2026,
      "notes": "This dataset contains synthetic parallel sentence pairs in Tunisian Arabic (Darija) and English.",
      "metrics": {
        "downloads": 10,
        "likes": 5,
        "lastModified": "2026-07-14"
      }
    },
    {
      "id": "ufal-parallel-corpus-of-north-levantine-1-0",
      "name": "UFAL Parallel Corpus of North Levantine 1.0",
      "type": "dataset",
      "country": "INTL",
      "org": "Charles University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "dialect-id"
      ],
      "links": {
        "website": "https://lindat.mff.cuni.cz/repository/xmlui/handle/11234/1-5033",
        "hf": "https://huggingface.co/datasets/arbml/UFAL",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.34/"
      },
      "dialects": [
        "lev"
      ],
      "size": "120,600 sentences",
      "year": 2023,
      "tags": [
        "multilingual"
      ],
      "notes": "120,600 multiparallel sentences in English, French, German, Greek, Spanish, and Standard Arabic selected from the OpenSubtitles2018 corpus [1] and manually.",
      "metrics": {
        "downloads": 10,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "aghlat",
      "name": "Aghlat",
      "type": "dataset",
      "country": "DZ",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/linuxscout/aghlat",
        "hf": "https://huggingface.co/datasets/arbml/aghlat"
      },
      "dialects": [
        "msa"
      ],
      "size": "331 tokens",
      "year": 2019,
      "notes": "Arabic misspelling corpus for spell-checking, from the linuxscout project.",
      "metrics": {
        "downloads": 9,
        "likes": 1,
        "lastModified": "2024-03-24"
      }
    },
    {
      "id": "arsarcasmoji",
      "name": "ArSarcasMoji",
      "type": "dataset",
      "country": "SA",
      "org": "Jazan University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "sarcasm"
      ],
      "links": {
        "github": "https://github.com/ShathaHakami/ArSarcasMoji-Dataset",
        "hf": "https://huggingface.co/datasets/arbml/ArSarcasMoji",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.18.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "24,630 sentences",
      "year": 2023,
      "notes": "A dataset of 24,630 emoji-augmented Arabic texts, with 17.5% that are ironic (either humorous or sarcastic).",
      "metrics": {
        "downloads": 9,
        "likes": 0,
        "lastModified": "2024-03-17"
      }
    },
    {
      "id": "askfm",
      "name": "ASKFM",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/Omarito2412/ASKFM",
        "hf": "https://huggingface.co/datasets/arbml/ASKFM"
      },
      "dialects": [
        "mixed"
      ],
      "size": "98,000 sentences",
      "year": 2017,
      "notes": "Askfm: 98k questions and answers written by different authors on Ask.fm.",
      "metrics": {
        "downloads": 9,
        "likes": 1,
        "lastModified": "2022-11-03"
      }
    },
    {
      "id": "ayatec",
      "name": "AyaTEC",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "http://qufaculty.qu.edu.qa/telsayed/datasets/",
        "hf": "https://huggingface.co/datasets/arbml/AyaTEC",
        "paper": "https://dl.acm.org/doi/pdf/10.1145/3400396"
      },
      "dialects": [
        "classical"
      ],
      "size": "207 sentences",
      "year": 2020,
      "notes": "QA on the Holy Qur’an Dataset",
      "metrics": {
        "downloads": 9,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "conformer-ctc-arabic-asr",
      "name": "Conformer CTC Arabic ASR",
      "type": "asr",
      "country": "INTL",
      "org": "MostafaAhmed98",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MostafaAhmed98/Conformer-CTC-Arabic-ASR"
      },
      "year": 2024,
      "notes": "NeMo Conformer-CTC (small) Arabic ASR trained on Common Voice 17.0.",
      "on_device": true,
      "metrics": {
        "downloads": 9,
        "likes": 4,
        "lastModified": "2024-09-18"
      }
    },
    {
      "id": "coronavirus",
      "name": "Coronavirus",
      "type": "dataset",
      "country": "SA",
      "org": "Imam Mohammad Bin Saud University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "github": "https://github.com/aseelad/Coronavirus-Public-Arabic-Twitter-Data-Set/",
        "hf": "https://huggingface.co/datasets/arbml/CoronaVirus",
        "paper": "https://www.preprints.org/manuscript/202004.0263/v1"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,800,856 sentences",
      "year": 2020,
      "notes": "Contains data collected from December 1st 2019 until April 11th 2020",
      "metrics": {
        "downloads": 9,
        "likes": 0,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "hijja",
      "name": "Hijja",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/israksu/Hijja2",
        "hf": "https://huggingface.co/datasets/arbml/Hijja2",
        "paper": "https://link.springer.com/content/pdf/10.1007/s00521-020-05070-8.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "47,434 images",
      "year": 2021,
      "notes": "Arabic handwritten letters dataset written by children aged 7–12",
      "metrics": {
        "downloads": 9,
        "likes": 2,
        "lastModified": "2024-06-21"
      }
    },
    {
      "id": "moroccandarija-llama-3-1-8b",
      "name": "MoroccanDarija-Llama-3.1-8B",
      "type": "llm",
      "country": "INTL",
      "org": "Anassk",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat",
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/Anassk/MoroccanDarija-Llama-3.1-8B"
      },
      "dialects": [
        "magh"
      ],
      "size": "8B",
      "on_device": false,
      "year": 2024,
      "notes": "This model translates Moroccan Darija (Moroccan Arabic) to English.",
      "metrics": {
        "downloads": 9,
        "likes": 4,
        "lastModified": "2024-09-29"
      }
    },
    {
      "id": "morroberta",
      "name": "MorRoBERTa",
      "type": "llm",
      "country": "INTL",
      "org": "otmangi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/otmangi/MorRoBERTa"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "MorRoBERTa is a Transformer-based Language Model designed specifically for the Moroccan Dialect.",
      "metrics": {
        "downloads": 9,
        "likes": 0,
        "lastModified": "2023-06-12"
      }
    },
    {
      "id": "qa-finetuned-arabiangpt-01b",
      "name": "QA_FineTuned_ArabianGpt-01B",
      "type": "llm",
      "country": "INTL",
      "org": "gp-tar4",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "qa",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/gp-tar4/QA_FineTuned_ArabianGpt-01B"
      },
      "size": "01B",
      "on_device": true,
      "year": 2024,
      "notes": "This model is a fine-tuned version of riotu-lab/ArabianGPT-01B on an arcd dataset(Arabic dataset).",
      "base_model": [
        "riotu-lab/arabiangpt-01b"
      ],
      "metrics": {
        "downloads": 9,
        "likes": 3,
        "lastModified": "2024-04-29"
      }
    },
    {
      "id": "waw-corpus",
      "name": "WAW Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/wawcorpus/",
        "hf": "https://huggingface.co/datasets/arbml/WAW"
      },
      "year": 2015,
      "notes": "Bilingual Arabic-English translation and interpretation corpus from WISE 2013, ARC14 and WISH conference recordings.",
      "metrics": {
        "downloads": 9,
        "likes": 1,
        "lastModified": "2024-03-31"
      }
    },
    {
      "id": "4factors-modern-standard-arabic-msa-sample",
      "name": "4FACTORS Modern Standard Arabic (MSA) Sample",
      "type": "dataset",
      "country": "INTL",
      "org": "FACTORS GmbH",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/4factors/arabic-msa-sample",
        "paper": "https://www.4factors.ch/arabic-nlp-training-data/"
      },
      "dialects": [
        "msa"
      ],
      "size": "50 sentences",
      "year": 2026,
      "notes": "Native-written, human-verified Modern Standard Arabic: 50 formal sentences (news/official-statement register) with English glosses and domain labels.",
      "metrics": {
        "downloads": 8,
        "likes": 0,
        "lastModified": "2026-07-15"
      }
    },
    {
      "id": "acqad",
      "name": "ACQAD",
      "type": "dataset",
      "country": "DZ",
      "org": "EMP Algeria",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/abdellahhamouda/acqad-dataset",
        "hf": "https://huggingface.co/datasets/arbml/acqad_multihop",
        "paper": "https://hal.science/hal-03992129v1/preview/ICCSAITCS_2022_paper_9032.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "118,841 sentences",
      "year": 2022,
      "notes": "Contains more than 118k questions, covering both comparison and multi-hop types.",
      "metrics": {
        "downloads": 8,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "algerian-dialect-translation",
      "name": "algerian-dialect-translation",
      "type": "llm",
      "country": "INTL",
      "org": "zakigll",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/zakigll/algerian-dialect-translation"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Machine-translation model for the Algerian Arabic dialect.",
      "base_model": [
        "google-t5/t5-small"
      ],
      "metrics": {
        "downloads": 8,
        "likes": 0,
        "lastModified": "2024-03-26"
      }
    },
    {
      "id": "annotated-al-jazeera-dialectal-speech-corpus",
      "name": "Annotated Al Jazeera Dialectal Speech Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/aljazeeraspeechcorpus/",
        "hf": "https://huggingface.co/datasets/arbml/aljazeera_dialectal_speech"
      },
      "dialects": [
        "msa",
        "egy",
        "lev",
        "magh",
        "gulf"
      ],
      "year": 2015,
      "notes": "57 hours of Al Jazeera speech with dialect labels for MSA, Egyptian, Levantine, North African and Gulf.",
      "metrics": {
        "downloads": 8,
        "likes": 2,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "arabic-named-entity-gazetteer",
      "name": "Arabic Named Entity Gazetteer",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Birmingham",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "ner",
        "retrieval"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/arabic-named-entity-gazetteer/",
        "hf": "https://huggingface.co/datasets/arbml/WikiFANEGazet",
        "paper": "https://aclanthology.org/I13-1045/"
      },
      "dialects": [
        "msa"
      ],
      "size": "68,355 tokens",
      "year": 2013,
      "notes": "A gazetteer of entities curated from Wikipedia.",
      "metrics": {
        "downloads": 8,
        "likes": 0,
        "lastModified": "2024-04-07"
      }
    },
    {
      "id": "arpis",
      "name": "ArPiS",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "retrieval"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Haneen84/Arabic_news/tree/main",
        "paper": "https://ieeexplore.ieee.org/document/10085561"
      },
      "dialects": [
        "classical"
      ],
      "size": "2,000 documents",
      "year": 2022,
      "notes": "More than 2k articles about Hajj",
      "metrics": {
        "downloads": 8,
        "likes": 0,
        "lastModified": "2023-09-23"
      }
    },
    {
      "id": "basira-omni-30b-v0-1",
      "name": "basira omni 30b v0.1",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "other",
      "modality": "multimodal",
      "tasks": [
        "chat",
        "qa",
        "vision"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/basira-omni-30b-v0.1"
      },
      "size": "30B",
      "on_device": false,
      "year": 2026,
      "notes": "A first open-source attempt at an Arabic multimodal large language model.",
      "base_model": [
        "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning",
        "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-bf16"
      ],
      "metrics": {
        "downloads": 8,
        "likes": 4,
        "lastModified": "2026-05-05"
      }
    },
    {
      "id": "darija-translation",
      "name": "darija translation",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "cc",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/darija-translation"
      },
      "dialects": [
        "magh"
      ],
      "size": "10K–100K rows",
      "year": 2024,
      "notes": "Moroccan Darija translation pairs with English and French.",
      "metrics": {
        "downloads": 8,
        "likes": 17,
        "lastModified": "2024-04-25"
      }
    },
    {
      "id": "dibt-prompt-translation-for-arabic",
      "name": "dibt prompt translation for arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "data-is-better-together",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "prompts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/data-is-better-together/dibt-prompt-translation-for-arabic"
      },
      "size": "<1K rows",
      "year": 2024,
      "notes": "501 Arabic prompt translations with quality ratings from the Data Is Better Together translation effort.",
      "metrics": {
        "downloads": 8,
        "likes": 3,
        "lastModified": "2024-03-21"
      }
    },
    {
      "id": "falcon-7b-qlora-alpaca-arabic",
      "name": "falcon-7b-QLoRA-alpaca-arabic",
      "type": "llm",
      "country": "INTL",
      "org": "alielfilali01",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/alielfilali01/falcon-7b-QLoRA-alpaca-arabic"
      },
      "size": "7B",
      "on_device": false,
      "year": 2023,
      "notes": "This repo contains a low-rank adapter for Falcon-7b fit on the Stanford Alpaca dataset Arabic version Yasbok/Alpacaarabicinstruct.",
      "metrics": {
        "downloads": 8,
        "likes": 8,
        "lastModified": "2023-06-10"
      }
    },
    {
      "id": "quran-whisper-tiny-v1",
      "name": "quran whisper tiny v1",
      "type": "asr",
      "country": "INTL",
      "org": "cherifkhalifah",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation",
        "quran"
      ],
      "links": {
        "hf": "https://huggingface.co/cherifkhalifah/quran-whisper-tiny-v1"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "This model is a fine-tuned version of openai/whisper-small on the audiofolder dataset.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 8,
        "likes": 4,
        "lastModified": "2024-09-04"
      }
    },
    {
      "id": "adpbc",
      "name": "ADPBC",
      "type": "dataset",
      "country": "EG",
      "org": "Menoufia University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos",
        "parsing",
        "classification"
      ],
      "links": {
        "github": "https://github.com/salsama/Arabic-Information-Extraction-Corpus",
        "hf": "https://huggingface.co/datasets/arbml/ADPBC",
        "paper": "http://www.mecs-press.org/ijitcs/ijitcs-v13-n1/IJITCS-V13-N1-4.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "16 documents",
      "year": 2021,
      "notes": "This corpus contains the words and their dependency relation produced by performing some steps",
      "metrics": {
        "downloads": 7,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "aqqac",
      "name": "AQQAC",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Leeds",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://archive.researchdata.leeds.ac.uk/464/1/AAQQAC.XML",
        "hf": "https://huggingface.co/datasets/arbml/AQQAC",
        "paper": "https://archive.researchdata.leeds.ac.uk/464/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,224 sentences",
      "year": 2018,
      "notes": "Each question and answer is annotated with the question ID, question word (particles), chapter number, verse number, question topic, question type, Al-Quran.",
      "metrics": {
        "downloads": 7,
        "likes": 0,
        "lastModified": "2024-05-10"
      }
    },
    {
      "id": "arabiccorpus2b",
      "name": "ArabicCorpus2B",
      "type": "dataset",
      "country": "INTL",
      "org": "tarekeldeeb",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/tarekeldeeb/ArabicCorpus2B"
      },
      "notes": "1.9B word Arabic corpus",
      "metrics": {
        "downloads": 7,
        "likes": 2,
        "lastModified": "2022-12-14"
      }
    },
    {
      "id": "arsyra-dialect-datasets",
      "name": "ArSyra dialect datasets",
      "type": "dataset",
      "country": "INTL",
      "org": "ArSyra",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "nlp",
        "dialects"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArSyra/arsyra-complete"
      },
      "dialects": [
        "iraqi",
        "lev",
        "sudanese",
        "yemeni",
        "gulf"
      ],
      "year": 2026,
      "notes": "Gated family of Arabic dialect datasets: Iraqi, Levantine, Sudanese, Yemeni, Gulf, plus social-media and tech subsets.",
      "metrics": {
        "downloads": 7,
        "likes": 1,
        "lastModified": "2026-03-05"
      }
    },
    {
      "id": "atlaset-audio",
      "name": "Atlaset-audio",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/abdeljalilELmajjodi/Atlaset-audio"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Moroccan Darija speech data curated for MoulSot ASR.",
      "metrics": {
        "downloads": 7,
        "likes": 0,
        "lastModified": "2025-08-02"
      }
    },
    {
      "id": "okapi-arabic",
      "name": "okapi arabic",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/okapi_arabic"
      },
      "size": "10K–100K rows",
      "year": 2023,
      "notes": "64.7K Okapi Arabic instruction, input and output rows.",
      "metrics": {
        "downloads": 7,
        "likes": 4,
        "lastModified": "2023-08-15"
      }
    },
    {
      "id": "q8berta-v2",
      "name": "Q8BERTa-v2",
      "type": "llm",
      "country": "INTL",
      "org": "Kalmundi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/Kalmundi/Q8BERTa-v2"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "notes": "Optimised RoBERTa encoder for Kuwaiti dialect, version 2 of Q8BERTa.",
      "metrics": {
        "downloads": 7,
        "likes": 0,
        "lastModified": "2025-01-16"
      }
    },
    {
      "id": "spark-tts-arabic-complete",
      "name": "Spark TTS Arabic Complete",
      "type": "tts",
      "country": "INTL",
      "org": "azeddinShr",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/azeddinShr/Spark-TTS-Arabic-Complete"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Fine-tuned version of SparkAudio/Spark-TTS-0.5B specialized for Arabic text-to-speech synthesis.",
      "base_model": [
        "sparkaudio/spark-tts-0.5b"
      ],
      "metrics": {
        "downloads": 7,
        "likes": 4,
        "lastModified": "2025-12-16"
      }
    },
    {
      "id": "synthetic-data-english-darija",
      "name": "Synthetic Data English Darija",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/Synthetic-Data-English-Darija"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2024,
      "notes": "Synthetic English–Moroccan Darija parallel sentences for machine translation.",
      "metrics": {
        "downloads": 7,
        "likes": 2,
        "lastModified": "2024-11-03"
      }
    },
    {
      "id": "transliteration-moroccan-darija",
      "name": "Transliteration Moroccan Darija",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Transliteration-Moroccan-Darija"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Moroccan Darija transliteration model trained on the atlasia ATAM dataset; access gated.",
      "metrics": {
        "downloads": 7,
        "likes": 4,
        "lastModified": "2024-04-30"
      }
    },
    {
      "id": "anti-social-behaviour-in-online-communication",
      "name": "Anti-Social Behaviour in Online Communication",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Limerick",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "website": "https://onedrive.live.com/?authkey=!ACDXj_ZNcZPqzy0&id=6EF6951FBF8217F9!105&cid=6EF6951FBF8217F9",
        "hf": "https://huggingface.co/datasets/arbml/offensive_language_arabic",
        "paper": "https://ulir.ul.ie/bitstream/handle/10344/9946/Alakrot_2019_Detection.pdf?sequence=2"
      },
      "dialects": [
        "mixed"
      ],
      "size": "15,050 sentences",
      "year": 2018,
      "notes": "A corpus of 15,050 labelled YouTube comments in Arabic",
      "metrics": {
        "downloads": 6,
        "likes": 0,
        "lastModified": "2024-04-14"
      }
    },
    {
      "id": "arabic-textual-entailment-dataset",
      "name": "Arabic Textual Entailment Dataset",
      "type": "dataset",
      "country": "IQ",
      "org": "University of Basrah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/arbml/ArabicTE",
        "paper": "https://arxiv.org/pdf/1402.0578"
      },
      "dialects": [
        "msa"
      ],
      "size": "600 sentences",
      "year": 2013,
      "notes": "This dataset contains 600 Arabic premise:hypothesis pairs.",
      "metrics": {
        "downloads": 6,
        "likes": 0,
        "lastModified": "2024-07-15"
      }
    },
    {
      "id": "bahraini-dialect-llm",
      "name": "Bahraini Dialect LLM",
      "type": "llm",
      "country": "BH",
      "org": "Community (Bahrain)",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Hishambarakat/Bahraini_Dialect_LLM"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Chat model fine-tuned for the Bahraini dialect.",
      "base_model": [
        "humain-ai/allam-7b-instruct-preview"
      ],
      "metrics": {
        "downloads": 6,
        "likes": 1,
        "lastModified": "2026-02-23"
      }
    },
    {
      "id": "levantine-dialects",
      "name": "levantine dialects",
      "type": "dataset",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/atlasia/levantine_dialects"
      },
      "dialects": [
        "lev"
      ],
      "size": "10K–100K rows",
      "year": 2025,
      "notes": "Levantine Arabic dialect text collection released by the Atlasia lab.",
      "metrics": {
        "downloads": 6,
        "likes": 2,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "llama2-qlora-finetuned-arabic",
      "name": "llama2-qlora-finetunined-Arabic",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/llama2-qlora-finetunined-Arabic"
      },
      "year": 2023,
      "notes": "QLoRA PEFT adapter for Llama 2 fine-tuned on Arabic; the card lists no base model or dataset.",
      "metrics": {
        "downloads": 6,
        "likes": 8,
        "lastModified": "2023-07-21"
      }
    },
    {
      "id": "byt5-darija-emphatic",
      "name": "byt5 darija emphatic",
      "type": "llm",
      "country": "INTL",
      "org": "anasskabil",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "hf": "https://huggingface.co/anasskabil/byt5-darija-emphatic"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "ByT5-base fine-tuned to predict emphatic consonants in Moroccan Darija Arabizi-to-Arabic transliteration.",
      "base_model": [
        "anasskabil/byt5-darija-emphatic"
      ],
      "metrics": {
        "downloads": 5,
        "likes": 3,
        "lastModified": "2026-05-11"
      }
    },
    {
      "id": "covid-fakes",
      "name": "COVID-FAKES",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Victoria",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/mohaddad/COVID-FAKES",
        "hf": "https://huggingface.co/datasets/arbml/COVID_FAES_ar",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-030-57796-4_25"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,263,000 sentences",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "Bilingual (Arabic/English) COVID-19 Twitter dataset for misleading information detection",
      "metrics": {
        "downloads": 5,
        "likes": 0,
        "lastModified": "2022-10-26"
      }
    },
    {
      "id": "hassaniya-gpt2-talk",
      "name": "hassaniya-gpt2-talk",
      "type": "llm",
      "country": "INTL",
      "org": "ahmed200512",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/ahmed200512/hassaniya-gpt2-talk"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2026,
      "notes": "GPT-2 fine-tuned for conversation in Hassaniya Arabic.",
      "base_model": [
        "openai-community/gpt2"
      ],
      "metrics": {
        "downloads": 5,
        "likes": 0,
        "lastModified": "2026-05-26"
      }
    },
    {
      "id": "arabicsenamticsimilaritydataset",
      "name": "ArabicSenamticSimilarityDataset",
      "type": "dataset",
      "country": "PS",
      "org": "Birzeit University",
      "license": "openrail",
      "modality": "text",
      "tasks": [
        "sts"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ArimaKn/ArabicSemanticSimilairtyDataset"
      },
      "dialects": [
        "msa"
      ],
      "size": "778 sentences",
      "year": 2023,
      "notes": "This is a simple dataset for the semantic similarity task with 3 different judges",
      "metrics": {
        "downloads": 4,
        "likes": 1,
        "lastModified": "2023-09-04"
      }
    },
    {
      "id": "darija-therapy-qa",
      "name": "darija therapy qa",
      "type": "dataset",
      "country": "INTL",
      "org": "ayatallah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "therapy"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/ayatallah/darija-therapy-qa"
      },
      "dialects": [
        "magh"
      ],
      "size": "1K–10K rows",
      "year": 2025,
      "notes": "6,181 Darija therapy-style context and response pairs.",
      "metrics": {
        "downloads": 4,
        "likes": 4,
        "lastModified": "2025-12-27"
      }
    },
    {
      "id": "darija-translator",
      "name": "darija translator",
      "type": "llm",
      "country": "INTL",
      "org": "Dhiadev-tn",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/Dhiadev-tn/darija-translator"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "A Nano-Transformer for Tunisian Darija-to-English translation, built from scratch.",
      "metrics": {
        "downloads": 4,
        "likes": 4,
        "lastModified": "2026-07-24"
      }
    },
    {
      "id": "modernbert-arabic",
      "name": "ModernBERT-Arabic",
      "type": "embedding",
      "country": "INTL",
      "org": "BounharAbdelaziz",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/BounharAbdelaziz/ModernBERT-Arabic-Embeddings"
      },
      "notes": "ModernBERT-based sentence embeddings",
      "base_model": [
        "answerdotai/modernbert-base"
      ],
      "metrics": {
        "downloads": 3,
        "likes": 4,
        "lastModified": "2024-12-29"
      }
    },
    {
      "id": "whisper-with-augmentation-small-arabic-with-diacritics",
      "name": "whisper with augmentation small arabic with diacritics",
      "type": "asr",
      "country": "INTL",
      "org": "mohmdsh",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "diacritization",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/mohmdsh/whisper-with-augmentation-small-arabic_with-diacritics"
      },
      "year": 2025,
      "notes": "Arabic automatic speech recognition model fine-tuned from openai/whisper-small.",
      "base_model": [
        "openai/whisper-small"
      ],
      "metrics": {
        "downloads": 3,
        "likes": 3,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "ar-stablelm-2-base",
      "name": "ar stablelm 2 base",
      "type": "llm",
      "country": "INTL",
      "org": "Stability AI",
      "license": "['other']",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/stabilityai/ar-stablelm-2-base"
      },
      "year": 2024,
      "notes": "Arabic-English StableLM 2 base model trained with CulturaX data.",
      "metrics": {
        "downloads": 2,
        "likes": 7,
        "lastModified": "2024-12-06"
      }
    },
    {
      "id": "terjman-nano-v1",
      "name": "Terjman Nano v1",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Terjman-Nano-v1"
      },
      "year": 2024,
      "tags": [
        "variants:3"
      ],
      "notes": "Terjman-Nano: English-Darija translation model fine-tuned from opus-mt-en-ar on atlasia/darija_english; access gated.",
      "dialects": [
        "magh"
      ],
      "base_model": [
        "helsinki-nlp/opus-mt-en-ar"
      ],
      "metrics": {
        "downloads": 2,
        "likes": 4,
        "lastModified": "2024-05-19"
      }
    },
    {
      "id": "terjman-supreme-v1",
      "name": "Terjman Supreme v1",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Terjman-Supreme-v1"
      },
      "year": 2024,
      "notes": "Terjman-Supreme: English-Darija translation model fine-tuned from NLLB-200 3.3B; access gated.",
      "dialects": [
        "magh"
      ],
      "base_model": [
        "facebook/nllb-200-3.3b"
      ],
      "metrics": {
        "downloads": 1,
        "likes": 8,
        "lastModified": "2024-06-02"
      }
    },
    {
      "id": "0-10-in-arabic-speech-voice",
      "name": "0 _ 10 in Arabic speech voice",
      "type": "dataset",
      "country": "INTL",
      "org": "maalielhaj",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "digits"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/maalielhaj/0-10-in-arabic-speech-voice"
      },
      "notes": "Recordings of spoken Arabic digits 0-9 with background noise samples in WAV format.",
      "metrics": null
    },
    {
      "id": "2-5m-rows-egyptian-datasets-collection",
      "name": "2.5M Rows Egyptian Datasets Collection",
      "type": "dataset",
      "country": "INTL",
      "org": "Mostafanofal453",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/Mostafanofal453/2.5-Million-Rows-Egyptian-Datasets-Collection"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Collection of over 2M rows of Egyptian-dialect text assembled for a master's project.",
      "metrics": null
    },
    {
      "id": "2006-conll-shared-task-arabic-czech",
      "name": "2006 CoNLL Shared Task - Arabic & Czech",
      "type": "dataset",
      "country": "INTL",
      "org": "Charles University",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "parsing"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2015T12"
      },
      "dialects": [
        "msa"
      ],
      "size": "113,500 tokens",
      "year": 2006,
      "tags": [
        "multilingual"
      ],
      "notes": "2006 CoNLL Shared Task - Arabic & Czech consists of Arabic and Czech dependency treebanks used as part of the CoNLL 2006 shared task on multi-lingual.",
      "metrics": null
    },
    {
      "id": "2007-conll-shared-task-arabic-english",
      "name": "2007 CoNLL Shared Task - Arabic & English",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "parsing"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2018T08"
      },
      "dialects": [
        "msa"
      ],
      "size": "116,800 tokens",
      "year": 2018,
      "tags": [
        "multilingual"
      ],
      "notes": "The source data in the treebanks in this release consists principally of various texts (e.g., textbooks, news, literature) annotated in dependency format.",
      "metrics": null
    },
    {
      "id": "2008-2010-nist-metrics-for-machine-translation-metricsmatr-gale-evalua",
      "name": "2008/2010 NIST Metrics for Machine Translation (MetricsMaTr) GALE Evaluation Set",
      "type": "benchmark",
      "country": "INTL",
      "org": "NIST",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2011T05"
      },
      "dialects": [
        "msa"
      ],
      "size": "149 documents",
      "year": 2011,
      "tags": [
        "multilingual"
      ],
      "notes": "NIST MetricsMaTr GALE evaluation set with Arabic-English and Chinese-English translations and human adequacy assessments.",
      "metrics": null
    },
    {
      "id": "3arab-tts-500m-v2",
      "name": "3arab TTS 500M v2",
      "type": "tts",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/3arab-TTS-500M-v2"
      },
      "size": "500M",
      "on_device": true,
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Arabic text to speech model fine-tuned from sherif1313/3arab-TTS-500M-v1.",
      "base_model": [
        "sherif1313/3arab-tts-500m-v1"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2026-06-12"
      }
    },
    {
      "id": "3arab-tts-500m-v2-voicedesign",
      "name": "3arab-TTS-500M-v2-VoiceDesign",
      "type": "tts",
      "country": "INTL",
      "org": "sherif1313",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/sherif1313/3arab-TTS-500M-v2-VoiceDesign"
      },
      "size": "500M",
      "on_device": true,
      "year": 2026,
      "tags": [
        "variants:1"
      ],
      "notes": "Arabic text to speech model fine-tuned from sherif1313/3arab-TTS-500M-v2.",
      "base_model": [
        "sherif1313/3arab-tts-500m-v2"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2026-06-22"
      }
    },
    {
      "id": "400k-egyptian-arabic-lines",
      "name": "400K Egyptian Arabic Lines",
      "type": "dataset",
      "country": "INTL",
      "org": "fadisarwat",
      "license": "gpl-3.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "asr"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/fadisarwat/egyptian-arabic-lines"
      },
      "dialects": [
        "egy"
      ],
      "year": 2024,
      "notes": "Egyptian Arabic indexed audio files for text-to-speech or speech-to-text models.",
      "metrics": null
    },
    {
      "id": "a-multidialectal-parallel-arabic-corpus-llm",
      "name": "a Multidialectal Parallel Arabic Corpus - LLM",
      "type": "dataset",
      "country": "INTL",
      "org": "Khalid Almeman",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "dialects"
      ],
      "links": {
        "website": "https://zenodo.org/records/17776617"
      },
      "year": 2025,
      "notes": "Multidialectal parallel Arabic corpus for LLM work by Khalid Almeman, hosted on Zenodo.",
      "metrics": null
    },
    {
      "id": "acegpt",
      "name": "AceGPT",
      "type": "llm",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/FreedomIntelligence/AceGPT"
      },
      "base_model": [
        "meta-llama/llama-2-7b-hf"
      ],
      "size": "7B",
      "notes": "Top performance, culturally aligned",
      "metrics": null
    },
    {
      "id": "acegpt-v2",
      "name": "AceGPT-v2",
      "type": "llm",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "alignment",
        "chat"
      ],
      "links": {
        "github": "https://github.com/FreedomIntelligence/AceGPT-v2"
      },
      "notes": "Arabic LLMs (8B, 32B, 70B) with native alignment applied at pre-training, available on Hugging Face.",
      "metrics": null
    },
    {
      "id": "adabner",
      "name": "AdabNer",
      "type": "dataset",
      "country": "INTL",
      "org": "Sorbonne Université",
      "license": "other",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.19468385"
      },
      "dialects": [
        "msa"
      ],
      "size": "876,000 tokens",
      "year": 2026,
      "notes": "First large-scale nested NER dataset for MSA literary texts.",
      "metrics": null
    },
    {
      "id": "adcc",
      "name": "ADCC",
      "type": "dataset",
      "country": "SA",
      "org": "Kingdom of Saudi Arabia Ministry of Education Al-Imam Muhammad Ibn Saud Islamic ",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "classification",
        "retrieval",
        "nli",
        "tts"
      ],
      "links": {
        "website": "https://adccorpus.wixsite.com/site/single-post/2016/11/30/what-is-adcc",
        "paper": "https://ieeexplore.ieee.org/document/7899133"
      },
      "dialects": [
        "msa"
      ],
      "size": "4,000,000 tokens",
      "year": 2017,
      "notes": "Arabic Daily Comunication Corpus (ADCC) is daily conversations written text in Modern Standard Arabic which was collected from different resources.",
      "metrics": null
    },
    {
      "id": "adult-content-detection-on-arabic-twitter-analysis-and-experiments",
      "name": "Adult Content Detection on Arabic Twitter: Analysis and Experiments",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/AdultContentDetection.zip",
        "paper": "https://aclanthology.org/2021.wanlp-1.14.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "50,000 sentences",
      "year": 2020,
      "notes": "Adult Content Detection on Arabic Twitter",
      "metrics": null
    },
    {
      "id": "agentic-ai-design-patterns",
      "name": "Agentic-AI-Design-Patterns",
      "type": "tool",
      "country": "INTL",
      "org": "Muhannad-Khaled",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/Muhannad-Khaled/Agentic-AI-Design-Patterns"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "The 21 agentic AI design patterns from Antonio Gulli's book, explained in Egyptian Arabic — with framework-free, runnable Python examples.",
      "metrics": null
    },
    {
      "id": "ahawp",
      "name": "AHAWP",
      "type": "dataset",
      "country": "INTL",
      "org": "Prince Mohammad Bin Fahd University",
      "license": "cc-by-nc-3.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/2h76672znt/2",
        "paper": "https://doi.org/10.1016/j.dib.2022.107947"
      },
      "dialects": [
        "msa"
      ],
      "size": "61,584 images",
      "year": 2022,
      "notes": "Handwritten Arabic alphabets, words and paragraphs from 82 writers for OCR and writer identification.",
      "metrics": null
    },
    {
      "id": "ahd-arabic-healthcare-dataset",
      "name": "AHD: Arabic Healthcare Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "medical",
        "generation"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/mgj29ndgrk"
      },
      "year": 2024,
      "notes": "Large Arabic healthcare text dataset introduced to improve Arabic natural language generation in medicine.",
      "metrics": null
    },
    {
      "id": "ahdq-arabic-handwriting-dataset-from-quran",
      "name": "AHDQ: (Arabic Handwriting Dataset from Quran)",
      "type": "dataset",
      "country": "INTL",
      "org": "abdoukhadrembacke",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "quran"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/abdoukhadrembacke/ahdq-arabic-handwriting-dataset-from-quran"
      },
      "year": 2024,
      "notes": "Handwritten Quranic Texts in Six Styles for Arabic OCR and Machine Learning",
      "metrics": null
    },
    {
      "id": "ai-quran-video-composer",
      "name": "Ai-Quran-Video-Composer",
      "type": "tool",
      "country": "INTL",
      "org": "WalidZein",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/WalidZein/Ai-Quran-Video-Composer"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "What if you can generate a a video with cinematic backgrounds behind the majestic verses of the Quran.",
      "metrics": null
    },
    {
      "id": "ai-rtl-resolver",
      "name": "ai-rtl-resolver",
      "type": "tool",
      "country": "INTL",
      "org": "miladniroee",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/miladniroee/ai-rtl-resolver",
        "website": "https://miladniroee.github.io/ai-rtl-resolver/"
      },
      "year": 2026,
      "notes": "Ai Chatbot RTL Resolver",
      "metrics": null
    },
    {
      "id": "aida-lab-psu",
      "name": "AIDA Lab (PSU)",
      "type": "org",
      "country": "SA",
      "org": "Prince Sultan University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://aidalabpsu.com/"
      },
      "notes": "AI and Data Analytics lab at Prince Sultan University, Riyadh.",
      "metrics": null
    },
    {
      "id": "aim-lab-nu-q",
      "name": "AIM Lab (NU-Q)",
      "type": "org",
      "country": "QA",
      "org": "Northwestern University in Qatar",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.qatar.northwestern.edu/research/aim-lab/"
      },
      "notes": "AI and media research lab at Northwestern University in Qatar.",
      "metrics": null
    },
    {
      "id": "ain",
      "name": "AIN",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "github": "https://github.com/mbzuai-oryx/AIN"
      },
      "size": "8B",
      "notes": "Arabic-centric Large Multimodal Model",
      "metrics": null
    },
    {
      "id": "ain-shams-university",
      "name": "Ain Shams University",
      "type": "org",
      "country": "EG",
      "org": "Ain Shams University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://cis.asu.edu.eg/"
      },
      "notes": "Arabic NLP, sentiment analysis, NER research",
      "metrics": null
    },
    {
      "id": "airabic",
      "name": "AIRABIC",
      "type": "benchmark",
      "country": "INTL",
      "org": "University of Bridgeport",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "ai-text-detection"
      ],
      "links": {
        "github": "https://github.com/Hamed1Hamed/AIRABIC",
        "paper": "https://ieeexplore.ieee.org/abstract/document/10459781"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,000 documents",
      "year": 2023,
      "notes": "Arabic benchmark dataset for evaluating AI-generated text detectors, containing 500 human-written and 500 AI-generated passages with and without diacritics.",
      "metrics": null
    },
    {
      "id": "aitd",
      "name": "AITD",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "github": "https://github.com/shammur/Arabic_news_text_classification_datasets",
        "paper": "https://aclanthology.org/2020.wanlp-1.21.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "115,692 sentences",
      "year": 2020,
      "notes": "A weakly annotated dataset for the most common tweet category.",
      "metrics": null
    },
    {
      "id": "ajdir-corpora",
      "name": "Ajdir Corpora",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "corpus"
      ],
      "links": {
        "website": "http://aracorpus.e3rab.com/argistestsrv.nmsu.edu/AraCorpus/"
      },
      "dialects": [
        "msa"
      ],
      "size": "28 documents",
      "year": 2010,
      "notes": "This is a raw text from Arabic daily newspapers collected over a year between 2004 and 2005.",
      "metrics": null
    },
    {
      "id": "al-kindi-ai",
      "name": "Al Kindi AI",
      "type": "tool",
      "country": "AE",
      "org": "Alpha to Digits",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "social-listening"
      ],
      "links": {
        "website": "https://arabic.ai/al-kindi-ai-built-by-alpha-to-digits-on-the-worlds-top-ranked-arabic-llm"
      },
      "year": 2026,
      "notes": "Arabic social listening intelligence product built on the Arabic.AI LLM.",
      "metrics": null
    },
    {
      "id": "al-atlas-moroccan-darija-pretraining",
      "name": "Al-Atlas-Moroccan-Darija-Pretraining",
      "type": "dataset",
      "country": "INTL",
      "org": "atlasia-ma",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "github": "https://github.com/atlasia-ma/Al-Atlas-Moroccan-Darija-Pretraining"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Training and data analysis code for Al-Atlas dataset and models",
      "metrics": null
    },
    {
      "id": "al-khatma",
      "name": "AL-Khatma",
      "type": "tool",
      "country": "INTL",
      "org": "oaokm",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/oaokm/AL-Khatma"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "A library Specialized About Islamic",
      "metrics": null
    },
    {
      "id": "al-munir",
      "name": "Al-Munir",
      "type": "tool",
      "country": "INTL",
      "org": "zedsalim",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/zedsalim/Al-Munir",
        "website": "https://almunir.netlify.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "المنير: للاستماع وقراءة القرآن الكريم مع التفسير لتسهيل الحفظ والفهم والمراجعة",
      "metrics": null
    },
    {
      "id": "al-quran-v3",
      "name": "al_quran_v3",
      "type": "tool",
      "country": "INTL",
      "org": "IsmailHosenIsmailJames",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/IsmailHosenIsmailJames/al_quran_v3"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A Flutter application for reading the Holy Quran, tracking prayer times, and managing Islamic practices.",
      "metrics": null
    },
    {
      "id": "albitaqat-quran",
      "name": "albitaqat_quran",
      "type": "tool",
      "country": "INTL",
      "org": "rn0x",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/rn0x/albitaqat_quran",
        "website": "http://albitaqat-quran.i8x.net/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Complete data for 114 Quran surahs (Al-Bitaqat).",
      "metrics": null
    },
    {
      "id": "alc-arabic-learner-corpus",
      "name": "ALC: Arabic Learner Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Leeds",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2015S10",
        "paper": "https://eprints.whiterose.ac.uk/75470/22/AtwellVer2.13.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "1 hours",
      "year": 2013,
      "notes": "Comprises a collection of texts written by learners of Arabic in Saudi Arabia",
      "metrics": null
    },
    {
      "id": "alcd",
      "name": "ALCD",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "dbcl-1.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/sohaasiri/arabic-lega-case-dataset",
        "paper": "https://doi.org/10.1016/j.dib.2025.112429"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,170 documents",
      "year": 2026,
      "notes": "Arabic legal case dataset for NLP collected from a government body in Saudi Arabia",
      "metrics": null
    },
    {
      "id": "alfanous",
      "name": "Alfanous",
      "type": "tool",
      "country": "INTL",
      "org": "Alfanous team",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "search",
        "quran"
      ],
      "links": {
        "github": "https://github.com/Alfanous-team/alfanous"
      },
      "year": 2012,
      "notes": "Arabic search engine API for the Quran with simple and advanced queries.",
      "metrics": null
    },
    {
      "id": "algeria-69-wilayas",
      "name": "algeria_69_wilayas",
      "type": "tool",
      "country": "INTL",
      "org": "Mohamed-gp",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "vision"
      ],
      "links": {
        "github": "https://github.com/Mohamed-gp/algeria_69_wilayas",
        "website": "https://mohamed-gp.github.io/algeria_69_wilayas/"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "🇩🇿 All 69 Algerian wilayas + 1,541 communes as free JSON — official numbering, Arabic names, coordinates, dairas.",
      "metrics": null
    },
    {
      "id": "alghafa",
      "name": "AlGhafa",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "website": "https://gitlab.com/tiiuae/alghafa",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.21.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "33,236 tokens",
      "year": 2023,
      "notes": "Evaluation benchmark for Arabic LLMs.",
      "metrics": null
    },
    {
      "id": "alhd",
      "name": "ALHD",
      "type": "dataset",
      "country": "INTL",
      "org": "Queen Mary University of London",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "ai-text-detection"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.17249602"
      },
      "dialects": [
        "mixed"
      ],
      "size": "405,456 documents",
      "year": 2025,
      "notes": "A large-scale Arabic dataset for detecting human vs LLM-generated texts, with 405k samples across 3 genres (news, social media, reviews) and MSA + multiple.",
      "metrics": null
    },
    {
      "id": "alhd-a-large-scale-and-multigenre-benchmark-dataset-for-arabic-llm-gen-unknown",
      "name": "ALHD: A Large-Scale and Multigenre Benchmark Dataset for Arabic LLM-Generated Text Detection",
      "type": "benchmark",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "llm-detection"
      ],
      "links": {
        "website": "https://zenodo.org/records/17249602"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "ALHD: multigenre dataset for distinguishing LLM-generated from human-written Arabic text.",
      "metrics": null
    },
    {
      "id": "aljazeera-deleted-comments",
      "name": "Aljazeera Deleted Comments",
      "type": "dataset",
      "country": "INTL",
      "org": "The University of Edinburgh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "website": "https://alt.qcri.org/people/hmubarak/public_html/offensive/",
        "paper": "https://aclanthology.org/W17-3008.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "33,100 sentences",
      "year": 2017,
      "notes": "Aljazeera Deleted Comments is a dataset of 32K deleted comments from Aljazeera.net, a popular Arabic news channel, moderates all the comments that appear.",
      "metrics": null
    },
    {
      "id": "alkhalil-morpho-sys",
      "name": "Alkhalil Morpho Sys",
      "type": "tool",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "parsing"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/alkhalil/"
      },
      "notes": "A morphosyntactic parser for Arabic words that can process both vocalized and non-vocalized texts",
      "metrics": null
    },
    {
      "id": "all-words-in-all-languages",
      "name": "all-words-in-all-languages",
      "type": "tool",
      "country": "INTL",
      "org": "eymenefealtun",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/eymenefealtun/all-words-in-all-languages"
      },
      "year": 2026,
      "notes": "This repository contains all the words from every language that exists in the universe.",
      "metrics": null
    },
    {
      "id": "allam-34b",
      "name": "ALLaM 34B",
      "type": "llm",
      "country": "SA",
      "org": "HUMAIN (Saudi)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://www.humain.com/en/news/humain-chat-launch"
      },
      "size": "34B",
      "notes": "Most advanced Arabic LLM, 8PB training data, powers HUMAIN Chat",
      "metrics": null
    },
    {
      "id": "allam-based-retrieval-augmented-generation-arabic-conversational-alzhe",
      "name": "ALLAM based Retrieval Augmented Generation Arabic Conversational Alzheimer Assistance",
      "type": "tool",
      "country": "INTL",
      "org": "ShahadAljohani",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "rag",
        "healthcare"
      ],
      "links": {
        "hf": "https://huggingface.co/ShahadAljohani/ALLAM-based-Retrieval-Augmented-Generation-Arabic-Conversational-Alzheimer-Assistance"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "ALLAM-RAG: Saudi Arabic conversational RAG system combining retrieval with the ALLaM LLM to support Alzheimer's patients and caregivers.",
      "metrics": {
        "downloads": 0,
        "likes": 5,
        "lastModified": "2026-07-31"
      }
    },
    {
      "id": "allam-2",
      "name": "ALLaM-2",
      "type": "llm",
      "country": "SA",
      "org": "SDAIA & IBM",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/ALLaM-AI"
      },
      "size": "7B-70B",
      "notes": "500B+ Arabic tokens, largest Arabic training set",
      "metrics": null
    },
    {
      "id": "aloochat",
      "name": "AlooChat",
      "type": "org",
      "country": "OM",
      "org": "AlooChat",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://aloochat.ai"
      },
      "notes": "AI customer support chat agents for WhatsApp, Instagram, web and email, found via an Arabic conversational AI search in Oman.",
      "metrics": null
    },
    {
      "id": "alpaca-instruction-fine-tune-arabic",
      "name": "Alpaca instruction fine tune Arabic",
      "type": "llm",
      "country": "INTL",
      "org": "Yasbok",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/Yasbok/Alpaca_instruction_fine_tune_Arabic"
      },
      "year": 2023,
      "notes": "LoRA adapter on LLaMA-7B for Arabic Alpaca-style instruction following.",
      "size": "7B",
      "metrics": {
        "downloads": 0,
        "likes": 11,
        "lastModified": "2023-03-23"
      }
    },
    {
      "id": "alphamwe-arabic",
      "name": "AlphaMWE-Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Tours",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "multiword-expression-identification"
      ],
      "links": {
        "github": "https://github.com/aaronlifenghan/AlphaMWE",
        "paper": "https://aclanthology.org/2023.ranlp-1.50.pdf"
      },
      "dialects": [
        "mixed",
        "msa",
        "magh",
        "egy"
      ],
      "size": "7,250 tokens",
      "year": 2023,
      "notes": "Arabic parallel corpus with MWE annotations",
      "metrics": null
    },
    {
      "id": "alpinejs-i18n",
      "name": "alpinejs-i18n",
      "type": "tool",
      "country": "INTL",
      "org": "rehhouari",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/rehhouari/alpinejs-i18n",
        "website": "https://alpinejs-i18n-example.vercel.app/"
      },
      "year": 2026,
      "notes": "Easy i18n (Internationalization) for Alpine.js!",
      "metrics": null
    },
    {
      "id": "alqari",
      "name": "AlQari",
      "type": "org",
      "country": "SA",
      "org": "AlQari",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://alqari.sa"
      },
      "notes": "Arabic-first document intelligence turning Arabic and English documents into structured, validated data.",
      "metrics": null
    },
    {
      "id": "alquranalkareem",
      "name": "alquranalkareem",
      "type": "tool",
      "country": "INTL",
      "org": "alheekmahlib",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/alheekmahlib/alquranalkareem",
        "website": "https://alhikmah.vexaltech.dev/download/quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "التطبيق الأمثل لقراءة القرآن الكريم",
      "metrics": null
    },
    {
      "id": "alr-arabic-laptop-reviews-dataset",
      "name": "ALR: Arabic Laptop Reviews dataset",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/bashartalafha/Arabic-Laptop-Reviews-ALR-Dataset",
        "paper": "https://www.researchgate.net/publication/329557366_Aspect-Based_Sentiment_Analysis_of_Arabic_Laptop_Reviews"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,753 sentences",
      "year": 2017,
      "notes": "Arabic Laptops Reviews (ALR) dataset focuses on laptops reviews written in Arabic",
      "metrics": null
    },
    {
      "id": "alsanaa-emirati-dataset",
      "name": "alsanaa-emirati-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Maha AlBlooki",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/MahaAlBlooki/alsanaa-emirati-dataset"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Traditional Emirati Arabic speech dataset for ASR.",
      "metrics": null
    },
    {
      "id": "alue",
      "name": "ALUE",
      "type": "benchmark",
      "country": "JO",
      "org": "Mawdoo3",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "website": "https://www.alue.org/"
      },
      "notes": "Arabic Language Understanding Evaluation benchmark",
      "metrics": null
    },
    {
      "id": "alyahmor",
      "name": "Alyahmor",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/linuxscout/alyahmor"
      },
      "year": 2019,
      "notes": "Arabic flexional morphology generator; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "amd-saer",
      "name": "Amd’SaEr",
      "type": "dataset",
      "country": "DZ",
      "org": "University of Amar Telidji Laghouat",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sentiment",
        "emotion"
      ],
      "links": {
        "github": "https://github.com/belgats/Arabic-Multimodal-Dataset",
        "paper": "https://doi.org/10.1145/3774880"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4 hours",
      "year": 2025,
      "notes": "Arabic multimodal dataset for sentiment analysis and emotion recognition with 1037 samples across audio, text, and visual modalities",
      "metrics": null
    },
    {
      "id": "amfds",
      "name": "AMFDS",
      "type": "dataset",
      "country": "INTL",
      "org": "King Abdullah II School for Information Technology",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://drive.google.com/drive/folders/1mRefmN4Yzy60Uh7z3B6cllyyOXaxQrgg?usp=sharing",
        "paper": "https://arxiv.org/pdf/2009.01987v1"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,160,000 images",
      "year": 2020,
      "notes": "Custom Arabic OCR dataset with 18 fonts from Wikipedia.",
      "metrics": null
    },
    {
      "id": "ami",
      "name": "AMI",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "github": "https://github.com/Allarwa/AMI-Dataset",
        "paper": "https://doi.org/10.1111/exsy.70128"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,500 sentences",
      "year": 2025,
      "notes": "An Arabic Twitter-based mental-illness focused sentiment dataset automatically annotated via transfer learning.",
      "metrics": null
    },
    {
      "id": "amiri-font",
      "name": "Amiri",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "typography"
      ],
      "links": {
        "github": "https://github.com/aliftype/amiri"
      },
      "year": 2015,
      "notes": "Classical Naskh body-text typeface for Arabic, open under OFL.",
      "metrics": null
    },
    {
      "id": "an-arabic-dataset-for-disease-named-entity-recognition-with-multi-anno",
      "name": "An Arabic Dataset for Disease Named Entity Recognition with Multi-Annotation Schemes",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-1.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "https://zenodo.org/records/3926432"
      },
      "year": 2020,
      "notes": "Manually annotated Arabic corpus of over 60 thousand words for disease named entity recognition.",
      "metrics": null
    },
    {
      "id": "anacd-arabic-news-article-classification-dataset",
      "name": "ANACD-Arabic-News-Article-Classification-Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/w8njshybth"
      },
      "year": 2025,
      "notes": "Arabic news article classification dataset introduced with a joint SVM and word-embedding model paper.",
      "metrics": null
    },
    {
      "id": "anees",
      "name": "Anees",
      "type": "tool",
      "country": "INTL",
      "org": "aashrafh",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/aashrafh/Anees"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "notes": "Multi-turn open-domain Arabic chatbot with a wide set of features.",
      "metrics": null
    },
    {
      "id": "anercorp",
      "name": "ANERcorp",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/anercorp/",
        "paper": "https://aclanthology.org/2020.lrec-1.868.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "316 documents",
      "year": 2020,
      "notes": "Arabic named-entity recognition corpus, collected from different resources and hosted by the CAMeL Lab.",
      "metrics": null
    },
    {
      "id": "anerd-a-large-scale-arabic-corpus-for-named-entity-recognition-and-dis",
      "name": "ANERD: A Large-Scale Arabic Corpus for Named Entity Recognition and Disambiguation",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "ner",
        "entity-disambiguation"
      ],
      "links": {
        "website": "https://zenodo.org/records/21909063"
      },
      "year": 2026,
      "notes": "ANERD v1.0.0: large-scale Arabic corpus for named entity recognition and named entity disambiguation.",
      "metrics": null
    },
    {
      "id": "annotated-corpus-of-arabic-tweets-for-hate-speech-analysis",
      "name": "Annotated Corpus of Arabic Tweets for Hate Speech Analysis",
      "type": "dataset",
      "country": "QA",
      "org": "Northwestern University in Qatar",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "offensive-language"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.14669917"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2025,
      "notes": "Multilabel Arabic hate speech tweet dataset.",
      "metrics": null
    },
    {
      "id": "annotated-shami-corpus",
      "name": "Annotated Shami Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Charles University in Prague",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pos",
        "morphology"
      ],
      "links": {
        "github": "https://github.com/christios/annotated-shami-corpus",
        "paper": "https://dspace.cuni.cz/handle/20.500.11956/147949"
      },
      "dialects": [
        "lev"
      ],
      "size": "10,000 tokens",
      "year": 2021,
      "notes": "Subsection of the Lebanese portion of the Shami Corpus annotated for spelling standardization (CODA), morphological segmentation and tagging.",
      "metrics": null
    },
    {
      "id": "annotated-tweet-corpus-in-arabizi-french-and-english",
      "name": "Annotated tweet corpus in Arabizi, French and English",
      "type": "dataset",
      "country": "INTL",
      "org": "ELRA",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "classification",
        "sentiment"
      ],
      "links": {
        "website": "https://catalogue.elra.info/en-us/repository/browse/ELRA-W0323/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "134,041 sentences",
      "year": 2022,
      "tags": [
        "multilingual"
      ],
      "notes": "In total, 17,103 sequences were annotated from 585,163 tweets (196,374 in English, 254,748 in French and 134,041 in Arabizi), including the themes “Others”.",
      "metrics": null
    },
    {
      "id": "ansari-skill",
      "name": "ansari-skill",
      "type": "tool",
      "country": "INTL",
      "org": "ansari-project",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/ansari-project/ansari-skill"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Islamic Knowledge Agent Skill — answers questions about Islam using authentic sources.",
      "metrics": null
    },
    {
      "id": "antigravity-rtl",
      "name": "antigravity-rtl",
      "type": "tool",
      "country": "INTL",
      "org": "mmnaderi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mmnaderi/antigravity-rtl",
        "website": "https://www.npmjs.com/package/antigravity-rtl"
      },
      "year": 2026,
      "notes": "Smart RTL (Right-to-Left) UI patcher for Antigravity & Antigravity IDE with Persian, Arabic, and Hebrew typography support.",
      "metrics": null
    },
    {
      "id": "aoc-id",
      "name": "aoc_id",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/aoc_id"
      },
      "year": 2018,
      "notes": "Arabic Dialect Identification on AOC data.",
      "metrics": null
    },
    {
      "id": "apd2020",
      "name": "APD2020",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/mcj6vkg6zw/1",
        "paper": "https://ieeexplore.ieee.org/document/9316520"
      },
      "dialects": [
        "classical"
      ],
      "size": "86,061 documents",
      "year": 2020,
      "notes": "Updated Classical Arabic poetry dataset (APD2020) covering five eras scraped from Adab.com, cleaned, made publicly available via Kaggle",
      "metrics": null
    },
    {
      "id": "apgc-v1-0-arabic-parallel-gender-corpus-v1-0",
      "name": "APGC v1.0: Arabic Parallel Gender Corpus v1.0",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "other",
      "modality": "text",
      "tasks": [
        "gender-id"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/arabic-parallel-gender-corpus/",
        "paper": "https://aclanthology.org/W19-3822v2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "12,000 sentences",
      "year": 2019,
      "tags": [
        "multilingual"
      ],
      "notes": "A corpus designed to support research on gender bias in natural language processing applications working on Arabic",
      "metrics": null
    },
    {
      "id": "api",
      "name": "api",
      "type": "tool",
      "country": "INTL",
      "org": "sunnah-com",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/sunnah-com/api",
        "website": "https://sunnah.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "API for sunnah.com",
      "metrics": null
    },
    {
      "id": "api-js",
      "name": "api-js",
      "type": "tool",
      "country": "INTL",
      "org": "Quran Foundation",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/quran/api-js",
        "website": "https://api-docs.quran.foundation/docs/sdk/javascript"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran Foundation's Official JS SDK https://npmjs.com/package/@quranjs/api",
      "metrics": null
    },
    {
      "id": "applied-innovation-center",
      "name": "Applied Innovation Center (AIC)",
      "type": "org",
      "country": "EG",
      "org": "Applied Innovation Center (MCIT Egypt)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://aic.gov.eg"
      },
      "notes": "Egyptian MCIT AI center; builds the Karnak LLM family and publishes models on Hugging Face (Applied-Innovation-Center).",
      "metrics": null
    },
    {
      "id": "aqmar-arabic-tagger",
      "name": "AQMAR Arabic Tagger",
      "type": "tool",
      "country": "INTL",
      "org": "CMU AQMAR",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "pos",
        "ner"
      ],
      "links": {
        "github": "https://github.com/nschneid/arabic-tagger"
      },
      "year": 2012,
      "notes": "Sequence tagger with cost-augmented structured perceptron for Arabic.",
      "metrics": null
    },
    {
      "id": "ar-apt",
      "name": "Ar-APT",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ai-text-detection"
      ],
      "links": {
        "github": "https://github.com/Saleh-Almohaimeed/Ar-APT",
        "paper": "https://arxiv.org/pdf/2511.16690v2"
      },
      "dialects": [
        "msa"
      ],
      "size": "16,400 documents",
      "year": 2025,
      "notes": "Datasets for AI-generated and AI-polished Arabic text detection.",
      "metrics": null
    },
    {
      "id": "ar-asag",
      "name": "AR-ASAG",
      "type": "dataset",
      "country": "INTL",
      "org": "Bouira University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "answer-grading"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/dj95jh332j/1",
        "paper": "https://aclanthology.org/2020.lrec-1.321.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,133 sentences",
      "year": 2020,
      "notes": "Reported evaluations relate to answers submitted for three different exams submitted to three classes of students.",
      "metrics": null
    },
    {
      "id": "ar-dad",
      "name": "Ar-DAD",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "speaker-id"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/3kndp5vs6b/3",
        "paper": "https://doi.org/10.1016/j.dib.2020.106503"
      },
      "dialects": [
        "mixed"
      ],
      "size": "15,810 sentences",
      "year": 2020,
      "notes": "A large Arabic audio dataset of 15,810 clips from 30 Quran reciters plus 397 imitation clips from 12 skilled imitators, organized by chapter and verse.",
      "metrics": null
    },
    {
      "id": "ar-embeddings",
      "name": "ar-embeddings",
      "type": "embedding",
      "country": "INTL",
      "org": "iamaziz",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/iamaziz/ar-embeddings",
        "website": "https://huggingface.co/azizalto/arabic-news-embeddings"
      },
      "year": 2024,
      "notes": "Sentiment Analysis for Arabic Text (tweets, reviews, and standard Arabic) using word2vec",
      "metrics": null
    },
    {
      "id": "ar-php",
      "name": "ar-php",
      "type": "tool",
      "country": "INTL",
      "org": "khaled-alshamaa",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/khaled-alshamaa/ar-php"
      },
      "year": 2026,
      "notes": "Set of functionalities enable Arabic website developers to serve professional search, present and process Arabic content in PHP",
      "metrics": null
    },
    {
      "id": "ar-sparc",
      "name": "ar-sparc",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Central Florida",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-to-sql"
      ],
      "links": {
        "website": "https://drive.google.com/drive/folders/1hwla8aSZ8wFhLZJHNiPFSkJvavypl-In",
        "paper": "https://arxiv.org/pdf/2511.20677v1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,225 sentences",
      "year": 2025,
      "notes": "First Arabic cross-domain text-to-SQL dataset",
      "metrics": null
    },
    {
      "id": "ara-timebank",
      "name": "ARA-TimeBank",
      "type": "dataset",
      "country": "TN",
      "org": "Tunisian university",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "event-detection"
      ],
      "links": {
        "github": "https://github.com/nafaa5/Arabic-event-timex-gazetteers-",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-030-63007-2_51"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,000 sentences",
      "year": 2020,
      "notes": "Enriched Arabic corpus, called “ARA-TimeBank”, for events, temporal expressions and temporal relations based on the new Arabic TimeML.",
      "metrics": null
    },
    {
      "id": "ara150-multidialect-tts-dataset",
      "name": "Ara150 MultiDialect TTS Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "g2p"
      ],
      "links": {
        "website": "https://zenodo.org/records/20452142"
      },
      "size": "20,606 sentences",
      "year": 2026,
      "notes": "Ara-150: about 152 hours of multi-dialect Arabic speech for text-to-speech research with an automated pipeline.",
      "metrics": null
    },
    {
      "id": "ara-dangspeech",
      "name": "ara_dangspeech",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dangerous-speech"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/ara_dangspeech"
      },
      "year": 2020,
      "notes": "Repository for Arabic dangerous speech detection from UBC-NLP.",
      "metrics": null
    },
    {
      "id": "arab-acquis",
      "name": "Arab-Acquis",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/arabacquis/",
        "paper": "https://aclanthology.org/E17-2038.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "12,000 sentences",
      "year": 2017,
      "tags": [
        "multilingual"
      ],
      "notes": "Consists of over 12,000 sentences from the JRCAcquis (Acquis Communautaire) corpus",
      "metrics": null
    },
    {
      "id": "arab-andalusian-music-corpus",
      "name": "Arab-Andalusian music corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Padova",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://zenodo.org/record/1291776#.YqTFeHZBxD9",
        "paper": "https://zenodo.org/records/1257388#.Wyeb9J9fjCI"
      },
      "dialects": [
        "mixed"
      ],
      "size": "125 hours",
      "year": 2018,
      "notes": "164 concert recordings (overall playable time more than 125 hours)",
      "metrics": null
    },
    {
      "id": "arab-enterprise-kit-lite",
      "name": "Arab-Enterprise-Kit-Lite",
      "type": "tool",
      "country": "INTL",
      "org": "MahmoudFarouk2479",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MahmoudFarouk2479/Arab-Enterprise-Kit-Lite"
      },
      "year": 2026,
      "notes": "A production-ready, open-source starter kit for building enterprise-grade SaaS applications.",
      "metrics": null
    },
    {
      "id": "arababytalk-egy",
      "name": "ARABABYTALK-EGY",
      "type": "dataset",
      "country": "INTL",
      "org": "Stony Brook University",
      "license": "cc-by-nc-sa-3.0",
      "modality": "speech",
      "tasks": [
        "morphology",
        "asr"
      ],
      "links": {
        "website": "http://arababytalk.camel-lab.com/",
        "paper": "https://aclanthology.org/2026.eacl-long.102.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "26,000 tokens",
      "year": 2026,
      "notes": "AraBabyTalk-EGY is an enriched release of the Egyptian Arabic CHILDES corpus that opens the child-adult interactions genre to modern Arabic NLP research.",
      "metrics": null
    },
    {
      "id": "arabagentskills",
      "name": "ArabAgentSkills",
      "type": "agent-skill",
      "country": "INTL",
      "org": "ArabAgentSkills",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/ArabAgentSkills/Skills"
      },
      "year": 2026,
      "notes": "Public source-backed Arab-world agent skills covering APIs, marketing, sales, support and localization.",
      "metrics": null
    },
    {
      "id": "arabceleb",
      "name": "ArabCeleb",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Milano-Bicocca",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/CeLuigi/ArabCeleb",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-031-08421-8_23"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,930 sentences",
      "year": 2021,
      "notes": "ArabCeleb is an audio dataset collected in the wild that specifically focuses on arabic language.",
      "metrics": null
    },
    {
      "id": "arabert",
      "name": "AraBERT",
      "type": "llm",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "github": "https://github.com/aub-mind/arabert"
      },
      "notes": "First BERT for Arabic, multiple versions",
      "metrics": null
    },
    {
      "id": "arabfactcheck-an-arabic-fake-news-dataset-with-fine-tuned-language-mod",
      "name": "ArabFactCheck: An Arabic Fake News Dataset with Fine-Tuned Language Models for Misinformation Detection",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "fake-news"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/ArabFactCheck_An_Arabic_Fake_News_Dataset_with_Fine-Tuned_Language_Models_for_Misinformation_Detection/33929857"
      },
      "notes": "ArabFactCheck: curated multi-domain Arabic fake news dataset with 7,691 news instances.",
      "metrics": null
    },
    {
      "id": "arabgec",
      "name": "ArabGEC",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "grammar-correction"
      ],
      "links": {
        "website": "https://drive.google.com/file/d/10BJakBI7BOqldxYDJTGVozKTUldJvPDe/view?usp=drive_link",
        "paper": "https://arxiv.org/pdf/2502.05312"
      },
      "dialects": [
        "msa"
      ],
      "size": "30,219,310 sentences",
      "year": 2025,
      "notes": "Synthetic dataset for Arabic grammatical error correction",
      "metrics": null
    },
    {
      "id": "arabguard",
      "name": "ArabGuard",
      "type": "agent-skill",
      "country": "INTL",
      "org": "d12o6aa",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "safety"
      ],
      "links": {
        "github": "https://github.com/d12o6aa/arabguard"
      },
      "year": 2026,
      "notes": "Python SDK protecting LLMs and chatbots from prompt injection in Arabic text.",
      "metrics": null
    },
    {
      "id": "arabic-analogy",
      "name": "Arabic analogy",
      "type": "benchmark",
      "country": "INTL",
      "org": "University of West Bohemia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "embedding-evaluation"
      ],
      "links": {
        "website": "http://computersystemsartists.net/arabic-morphological-analogies.zip",
        "paper": "https://ieeexplore.ieee.org/document/8374386"
      },
      "dialects": [
        "msa"
      ],
      "size": "240 tokens",
      "year": 2018,
      "notes": "The corpus consists of pairs of Arabic words that explore different morphological constructs and relationships.",
      "metrics": null
    },
    {
      "id": "arabic-audio-text-dataset-repository",
      "name": "Arabic Audio Text Dataset Repository",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/Arabic_Audio_Text_Dataset_Repository/30121789"
      },
      "year": 2025,
      "notes": "IDEAL 2025 dataset for evaluating how well commercial speech-to-text tools transcribe Arabic speech.",
      "metrics": null
    },
    {
      "id": "arabic-broad-leaderboard-abl",
      "name": "Arabic Broad Leaderboard (ABL)",
      "type": "benchmark",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/silma-ai/Arabic-LLM-Broad-Leaderboard"
      },
      "notes": "NextGen evaluation for Arabic LLMs by SILMA AI",
      "metrics": null
    },
    {
      "id": "arabic-chainbank",
      "name": "Arabic ChainBank",
      "type": "dataset",
      "country": "INTL",
      "org": "New York University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/ArabicChainBank",
        "paper": "https://arxiv.org/pdf/2410.20463v2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "23,333 tokens",
      "year": 2025,
      "notes": "Graph network for Arabic derivational morphology.",
      "metrics": null
    },
    {
      "id": "arabic-coco",
      "name": "Arabic COCO",
      "type": "dataset",
      "country": "INTL",
      "org": "canesee-project",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/canesee-project/Arabic-COCO"
      },
      "notes": "MS COCO captions in Arabic.",
      "metrics": null
    },
    {
      "id": "arabic-content-studio",
      "name": "Arabic Content Studio",
      "type": "agent-skill",
      "country": "INTL",
      "org": "smeseik-ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/smeseik-ai/arabic-content-studio"
      },
      "year": 2026,
      "notes": "Claude Skills for Arabic content teams covering production, culture, SEO and brand voice.",
      "metrics": null
    },
    {
      "id": "arabic-conversation-and-monologue-speech-dataset",
      "name": "Arabic Conversation and Monologue speech dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "nexdatafrank",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/nexdatafrank/arabic-real-world-speech-dataset"
      },
      "notes": "NexData Saudi Arabic real-world casual conversation and monologue speech (interviews, variety shows, live).",
      "dialects": [
        "gulf"
      ],
      "metrics": null
    },
    {
      "id": "arabic-corpus",
      "name": "Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "corpus"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/newarabiccorpus/"
      },
      "notes": "Collection of more than 460 Arabic books for language engineering applications.",
      "metrics": null
    },
    {
      "id": "arabic-customers-reviews-in-the-fashion-retail-industry",
      "name": "Arabic Customers Reviews in the Fashion retail industry",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "reviews"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/vh85c2bcp7"
      },
      "year": 2026,
      "notes": "10,000 Arabic customer reviews in the fashion retail domain.",
      "metrics": null
    },
    {
      "id": "arabic-dataset-for-llm-safeguard-evaluation-mbzuai",
      "name": "Arabic Dataset for LLM Safeguard Evaluation",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "safety"
      ],
      "links": {
        "github": "https://github.com/mbzuai-nlp/Arabic_safety_evaluation",
        "paper": "https://aclanthology.org/2025.naacl-long.285.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,799 sentences",
      "year": 2025,
      "notes": "A comprehensive dataset of 5,799 Arabic questions for evaluating Large Language Model safety.",
      "metrics": null
    },
    {
      "id": "arabic-deep-learning-ocr",
      "name": "Arabic Deep Learning OCR",
      "type": "ocr",
      "country": "JO",
      "org": "Mohammed Fasha",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/msfasha/Arabic-Deep-Learning-OCR"
      },
      "year": 2020,
      "notes": "Deep-learning OCR experiments for Arabic text.",
      "metrics": null
    },
    {
      "id": "swshon-arabic-dialect-id",
      "name": "Arabic dialect ID (17 countries)",
      "type": "tool",
      "country": "INTL",
      "org": "swshon",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/swshon/arabic-dialect-identification"
      },
      "year": 2019,
      "notes": "Country-level spoken Arabic dialect identification across 17 Arab countries.",
      "metrics": null
    },
    {
      "id": "arabic-dict-mcp",
      "name": "Arabic Dict MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "arnizamani",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/arnizamani/arabic-dict-mcp"
      },
      "year": 2026,
      "notes": "MCP server for grounded Arabic dictionary lookup wrapping arramooz (MSA verbs and nouns).",
      "metrics": null
    },
    {
      "id": "arabic-fact-checking-and-stance-detection-corpus",
      "name": "Arabic Fact-Checking and Stance Detection Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking",
        "stance-detection"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/arabic-fact-checking-and-stance-detection-corpus/"
      },
      "year": 2018,
      "notes": "422 claims about the Syrian war and Middle East politics labelled for factuality, with stance-annotated documents (NAACL 2018).",
      "metrics": null
    },
    {
      "id": "camel-arabic-gec",
      "name": "Arabic GEC (CAMeL Lab)",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "grammatical-error-correction"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/arabic-gec"
      },
      "year": 2022,
      "notes": "Code, models and data for Arabic grammatical error detection and correction.",
      "metrics": null
    },
    {
      "id": "arabic-gigaword-fifth-edition",
      "name": "Arabic Gigaword Fifth Edition",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2011T11"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,346,167 documents",
      "year": 2011,
      "tags": [
        "variants:4"
      ],
      "notes": "It is a comprehensive archive of newswire text data that has been acquired from Arabic news sources by LDC at the University of Pennsylvania.",
      "metrics": null
    },
    {
      "id": "ahr-ocr-handwriting",
      "name": "Arabic Handwriting Recognition E2E",
      "type": "ocr",
      "country": "INTL",
      "org": "AHR-OCR2024",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "handwriting"
      ],
      "links": {
        "github": "https://github.com/AHR-OCR2024/Arabic-Handwriting-Recognition"
      },
      "year": 2024,
      "notes": "End-to-end Arabic handwritten text OCR with an extraction application.",
      "metrics": null
    },
    {
      "id": "arabic-handwritten-digits-dataset",
      "name": "Arabic Handwritten Digits Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "digits"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/Arabic_Handwritten_Digits_Dataset/12236948"
      },
      "notes": "Arabic handwritten digits dataset for handwritten digit recognition.",
      "metrics": null
    },
    {
      "id": "arabic-hate-speech-dataset-2023",
      "name": "Arabic Hate Speech Dataset 2023",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/mcnzzpgrdj"
      },
      "year": 2023,
      "notes": "Jordanian Hate Speech Corpus (JHSC) of annotated hate tweets in four classes.",
      "dialects": [
        "lev"
      ],
      "metrics": null
    },
    {
      "id": "arabic-humor",
      "name": "Arabic Humor",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "humor-detection"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/Arabic-Humor",
        "paper": "https://aclanthology.org/2022.icnlsp-1.25.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,039 sentences",
      "year": 2022,
      "notes": "Arabic humor detection dataset with 10039 tweets in dialectal and MSA Arabic annotated as Humor vs Non-Humor.",
      "metrics": null
    },
    {
      "id": "arabic-infectious-disease-ontology",
      "name": "Arabic Infectious Disease Ontology",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ontology"
      ],
      "links": {
        "website": "http://www.research.lancs.ac.uk/portal/en/datasets/arabic-infectious-disease-ontology(39dbef60-ae9b-4405-99c8-35a41e95a3e0).html",
        "paper": "https://eprints.lancs.ac.uk/id/eprint/142307/1/LREC_2020_Paper_Developing_an_Arabic_Infectious_Ontology_.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "215 sentences",
      "year": 2020,
      "notes": "First Arabic ontology for the infectious disease domain, with 11 classes, 21 object properties and 215 individual concepts.",
      "metrics": null
    },
    {
      "id": "arabic-ke",
      "name": "Arabic KE",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "knowledge-editing"
      ],
      "links": {
        "github": "https://github.com/baselmousi/arabic-knowledge-editing",
        "paper": "http://arxiv.org/pdf/2507.09629v2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,140 sentences",
      "year": 2025,
      "notes": "Arabic translations of ZsRE and Counterfact benchmarks for knowledge editing analysis.",
      "metrics": null
    },
    {
      "id": "arabic-language-web-pages-dataset",
      "name": "Arabic language Web pages dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "web-pages",
        "language-id"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/Arabic_Sample/4588702"
      },
      "year": 2014,
      "notes": "7,976 URIs of Arabic-language web pages from the Arabic DMOZ, Raddadi and Star28 directories.",
      "metrics": null
    },
    {
      "id": "arabic-llm-guard",
      "name": "arabic llm guard",
      "type": "llm",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/arabic-llm-guard"
      },
      "year": 2026,
      "notes": "This model is designed to assess Arabic prompt-response pairs and generate a safety judgment such as whether the response is safe or belongs.",
      "base_model": [
        "unsloth/granite-4.0-h-small-base"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2026-03-13"
      }
    },
    {
      "id": "arabic-mental-health-dataset",
      "name": "Arabic Mental Health Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "medical",
        "mental-health"
      ],
      "links": {
        "website": "https://zenodo.org/records/20568736"
      },
      "year": 2026,
      "notes": "Mental-state descriptions collected via a questionnaire website from Arabic speakers.",
      "metrics": null
    },
    {
      "id": "arabic-natural-audio-dataset",
      "name": "Arabic Natural Audio Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "emotion"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/xm232yxf7t"
      },
      "year": 2018,
      "notes": "ANAD: Arabic Natural Audio Dataset for recognizing happy, angry and surprised emotions.",
      "metrics": null
    },
    {
      "id": "arabic-news-and-public-opinion-dataset-from-youtube",
      "name": "Arabic news and public opinion dataset from YouTube",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "news",
        "comments"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/3mnjw5hjkh"
      },
      "year": 2025,
      "notes": "19.4M public comments and replies on 70,000 news videos from 20 Arabic news YouTube channels.",
      "metrics": null
    },
    {
      "id": "arabic-nli-semantic-similarity",
      "name": "Arabic NLI & Semantic Similarity",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nli",
        "similarity"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/Omartificial-Intelligence-Space/arabic-nli-and-semantic-similarity-datasets"
      },
      "notes": "Collection of Arabic versions of the SNLI and MultiNLI natural language inference datasets.",
      "metrics": null
    },
    {
      "id": "arabic-nlp",
      "name": "Arabic NLP",
      "type": "tool",
      "country": "INTL",
      "org": "SemanticFrontiers",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/SemanticFrontiers/ArabicNLP"
      },
      "notes": "Collection of Arabic NLP and text processing scripts and utilities.",
      "metrics": null
    },
    {
      "id": "josa-arabic-ocr-studygroup",
      "name": "Arabic OCR Study Group",
      "type": "tool",
      "country": "JO",
      "org": "Jordan Open Source Association",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "handwriting"
      ],
      "links": {
        "github": "https://github.com/jordanopensource/arabic-ocr-studygroup"
      },
      "year": 2017,
      "notes": "Arabic handwriting dataset and starter code for a deep learning study group.",
      "metrics": null
    },
    {
      "id": "arabic-paraphrased-dataset",
      "name": "Arabic Paraphrased Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "summarization",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/Arabic-Paraphrased-Dataset",
        "paper": "https://doi.org/10.1016/j.dib.2024.111004"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,010 sentences",
      "year": 2024,
      "notes": "A synthetic Arabic parallel paraphrase dataset generated via back-translation to advance NLP applications like MT, summarization and sentiment analysis.",
      "metrics": null
    },
    {
      "id": "arabic-paraphrasing-corpus-for-nlp-applications",
      "name": "Arabic Paraphrasing Corpus for NLP Applications",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "paraphrase"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/x8wj68wrfj"
      },
      "year": 2025,
      "notes": "143 short Arabic texts (100 paraphrased, 43 not) for paraphrasing and text similarity detection.",
      "metrics": null
    },
    {
      "id": "ara-pronunciation-tool",
      "name": "Arabic pronunciation tool",
      "type": "tool",
      "country": "PS",
      "org": "Motaz Saad",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "g2p"
      ],
      "links": {
        "github": "https://github.com/motazsaad/ara-pronunciation-tool"
      },
      "year": 2017,
      "notes": "Converts diacritised Arabic text to phoneme sequences; Palestine.",
      "metrics": null
    },
    {
      "id": "arabic-rootfinder",
      "name": "Arabic RootFinder",
      "type": "tool",
      "country": "INTL",
      "org": "tb0yd",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/tb0yd/rootfinder"
      },
      "notes": "Arabic root finder with neural nets!",
      "metrics": null
    },
    {
      "id": "arabic-scholar-mcp-server",
      "name": "Arabic Scholar MCP Server",
      "type": "agent-skill",
      "country": "INTL",
      "org": "EngDawood",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/EngDawood/arabic-scholar-mcp-server"
      },
      "year": 2026,
      "notes": "MCP server for searching Arabic academic research, articles and dissertations across sources.",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis-dataset-ss2030-dataset",
      "name": "Arabic Sentiment Analysis Dataset SS2030 Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "snalyami3",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/snalyami3/arabic-sentiment-analysis-dataset-ss2030-dataset"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2020,
      "notes": "Sentiment Analysis of Social Events in Arabic Saudi Dialect",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-datasets",
      "name": "Arabic Sentiment Datasets",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/6w9g62xc67"
      },
      "year": 2025,
      "notes": "Dataset designed for sentiment analysis in the Arabic language.",
      "metrics": null
    },
    {
      "id": "arabic-speech-mispronunciation-detection-dataset",
      "name": "Arabic Speech Mispronunciation Detection Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "abdelrhmansalah22",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "pronunciation"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/abdelrhmansalah22/arabic-speech-mispronunciation-detection-dataset"
      },
      "dialects": [
        "egy"
      ],
      "size": "100 words",
      "year": 2024,
      "notes": "Dataset of Arabic speech in Egyptian dialect",
      "metrics": null
    },
    {
      "id": "arabic-spoken-dialects-regional-archive-sara",
      "name": "Arabic Spoken Dialects Regional Archive (SARA)",
      "type": "dataset",
      "country": "INTL",
      "org": "murtadhayaseen",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id",
        "asr"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/murtadhayaseen/arabic-spoken-regional-archive-sara"
      },
      "dialects": [
        "egy",
        "gulf"
      ],
      "year": 2022,
      "notes": "SARA: spoken dialects archive including Egyptian and Arabian Peninsula dialect recordings, on Kaggle.",
      "metrics": null
    },
    {
      "id": "arabic-tacotron-tts",
      "name": "Arabic Tacotron TTS",
      "type": "tts",
      "country": "INTL",
      "org": "yoosif0",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/yoosif0/arabic-tacotron-tts"
      },
      "notes": "End-to-end Arabic TTS based on Tacotron",
      "metrics": null
    },
    {
      "id": "arabic-treebank-broadcast-news-v1-0",
      "name": "Arabic Treebank - Broadcast News v1.0",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2012T07"
      },
      "dialects": [
        "msa"
      ],
      "size": "120 documents",
      "year": 2012,
      "notes": "Arabic Treebank Broadcast News v1.0 with syntactic annotations of Arabic TV broadcast news transcripts from 17 channels.",
      "metrics": null
    },
    {
      "id": "arabic-treebank-weblog",
      "name": "Arabic Treebank - Weblog",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "pos"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2016T02",
        "paper": "https://catalog.ldc.upenn.edu/docs/LDC2016T02/KulickBiesMaamouri-LREC2010.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "243,117 tokens",
      "year": 2016,
      "notes": "Arabic Treebank Weblog with syntactic annotations of Arabic weblog text collected from various web sources.",
      "metrics": null
    },
    {
      "id": "arabic-treebank-part-2-v-3-1",
      "name": "Arabic Treebank: Part 2 v 3.1",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2011T09"
      },
      "dialects": [
        "msa"
      ],
      "size": "501 documents",
      "year": 2011,
      "tags": [
        "variants:9"
      ],
      "notes": "ATB2 v 3.1 contains a total of 144,199 source tokens before clitics are split, and 169,319 tree tokens after clitics are separated for the treebank annotation.",
      "metrics": null
    },
    {
      "id": "asrajeh-arabic-tts",
      "name": "Arabic TTS (Al-Natiq)",
      "type": "tts",
      "country": "SA",
      "org": "Asrajeh",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/asrajeh/arabic-tts"
      },
      "year": 2019,
      "notes": "Arabic text-to-speech system (Al-Natiq) in Java.",
      "metrics": null
    },
    {
      "id": "arabic-tts-arena",
      "name": "Arabic TTS Arena",
      "type": "tool",
      "country": "INTL",
      "org": "Navid Gen AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "tts"
      ],
      "links": {
        "github": "https://github.com/Navid-Gen-AI/arabic-tts-arena"
      },
      "year": 2026,
      "notes": "Modal backend for a blind-comparison arena ranking Arabic TTS systems.",
      "metrics": null
    },
    {
      "id": "arabic-verbnet",
      "name": "Arabic VerbNet",
      "type": "dataset",
      "country": "INTL",
      "org": "JaouadMousser",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "lexicon"
      ],
      "links": {
        "github": "https://github.com/JaouadMousser/Arabic-Verbnet"
      },
      "year": 2005,
      "notes": "Arabic Verbnet is a lage scale verb lexicon that classifies verbs in Arabic using syntactic alternations inspired by the work of Kipper Schuler (2005)",
      "metrics": null
    },
    {
      "id": "arabic-video-subtitles-skill",
      "name": "Arabic Video Subtitles Skill",
      "type": "agent-skill",
      "country": "INTL",
      "org": "EngDawood",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/EngDawood/arabic-video-subtitles-skill"
      },
      "year": 2026,
      "notes": "Skill with Netflix-style Arabic subtitling guidelines and an SRT/VTT checker.",
      "metrics": null
    },
    {
      "id": "arabic-wiki-data-dump-2018",
      "name": "Arabic Wiki data Dump 2018",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://www.kaggle.com/datasets/abedkhooli/arabic-wiki-data-dump-2018"
      },
      "dialects": [
        "msa"
      ],
      "year": 2018,
      "notes": "Arabic Wikipedia articles dump from 2018 on Kaggle, intended for training Word2Vec embeddings.",
      "metrics": null
    },
    {
      "id": "arabic-word-production",
      "name": "Arabic Word Production Skill",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Bannovich",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/Bannovich/arabic-word-production"
      },
      "year": 2026,
      "notes": "Deterministic agent skill and plugin for Arabic-first and bilingual Word DOCX production.",
      "metrics": null
    },
    {
      "id": "arabic-agent-eval",
      "name": "arabic-agent-eval",
      "type": "benchmark",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "agent"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/arabic-agent-eval"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "notes": "Arabic function-calling and agent evaluation benchmark, distributed as a pip package.",
      "metrics": null
    },
    {
      "id": "arabic-bert",
      "name": "Arabic-BERT",
      "type": "llm",
      "country": "INTL",
      "org": "alisafaya",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "github": "https://github.com/alisafaya/Arabic-BERT"
      },
      "year": 2020,
      "notes": "Arabic edition of BERT pretrained language models",
      "metrics": null
    },
    {
      "id": "arabic-bidi-engineering",
      "name": "arabic-bidi-engineering",
      "type": "agent-skill",
      "country": "INTL",
      "org": "MosaabGalmod",
      "license": "mit",
      "modality": "none",
      "tasks": [
        "skill",
        "rtl"
      ],
      "links": {
        "github": "https://github.com/MosaabGalmod/arabic-bidi-engineering"
      },
      "notes": "Agent skill for correct Arabic RTL/BiDi in chat, terminal and documents",
      "metrics": null
    },
    {
      "id": "arabic-conjugator",
      "name": "Arabic-Conjugator",
      "type": "tool",
      "country": "INTL",
      "org": "awillborn",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/awillborn/Arabic-Conjugator"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Conjugates MSA verbs given three root letters, verb form, tense, and pronoun",
      "metrics": null
    },
    {
      "id": "arabic-darija-nlp-resources",
      "name": "Arabic-Darija-NLP-Resources",
      "type": "tool",
      "country": "INTL",
      "org": "MoroccoAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MoroccoAI/Arabic-Darija-NLP-Resources"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "A curated collection of resources and repositories for Natural Language Processing (NLP) tasks specific to Darija, the Moroccan Arabic dialect.",
      "metrics": null
    },
    {
      "id": "arabic-dialect-identification",
      "name": "arabic-dialect-identification",
      "type": "tool",
      "country": "INTL",
      "org": "Lafifi-24",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/Lafifi-24/arabic-dialect-identification"
      },
      "dialects": [
        "mixed"
      ],
      "year": 2023,
      "notes": "Fine-tune BERT models to classify Arabic text by different dialects.",
      "metrics": null
    },
    {
      "id": "arabic-dialects-voice-recognition",
      "name": "Arabic-Dialects-Voice-Recognition",
      "type": "dataset",
      "country": "INTL",
      "org": "MaherSaleem",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id",
        "speech"
      ],
      "links": {
        "github": "https://github.com/MaherSaleem/Arabic-Dialects-Voice-Recognition"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Arabic and Palestinian speech sound datasets with dialect voice-recognition code.",
      "metrics": null
    },
    {
      "id": "arabic-empathetic-chatbot",
      "name": "Arabic-Empathetic-Chatbot",
      "type": "dataset",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/aub-mind/Arabic-Empathetic-Chatbot"
      },
      "year": 2025,
      "notes": "Seq2Seq-based open domain empathetic conversational model for Arabic: Dataset & Model",
      "metrics": null
    },
    {
      "id": "arabic-f5-tts-v2",
      "name": "Arabic-F5-TTS-v2",
      "type": "tts",
      "country": "INTL",
      "org": "IbrahimSalah",
      "license": "fair-noncommercial-research-license",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/IbrahimSalah/Arabic-F5-TTS-v2"
      },
      "notes": "Improved F5-TTS Arabic fine-tune",
      "base_model": [
        "swivid/f5-tts"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 38,
        "lastModified": "2025-11-13"
      }
    },
    {
      "id": "arabic-hebrew-ted-talks-parallel-corpus",
      "name": "Arabic-Hebrew TED Talks Parallel Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Fondazione Bruno Kessler (FBK)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "retrieval",
        "nli"
      ],
      "links": {
        "github": "https://github.com/ajinkyakulkarni14/TED-Multilingual-Parallel-Corpus",
        "paper": "https://arxiv.org/pdf/1610.00572.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "225,000 sentences",
      "year": 2016,
      "tags": [
        "multilingual"
      ],
      "notes": "This dataset consists of 2023 TED talks with aligned Arabic and Hebrew subtitles.",
      "metrics": null
    },
    {
      "id": "arabic-llm-benchmarks-repository",
      "name": "Arabic-LLM-Benchmarks Repository",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/tiiuae/Arabic-LLM-Benchmarks"
      },
      "notes": "List of Arabic Benchmarks for Arabic LLMs.",
      "metrics": null
    },
    {
      "id": "arabic-local-gpt",
      "name": "Arabic-Local-GPT",
      "type": "tool",
      "country": "INTL",
      "org": "minar09",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "github": "https://github.com/minar09/arabic-local-gpt"
      },
      "notes": "Chat locally and securely in Arabic with your local Arabic documents using open LLM/GPT models",
      "metrics": null
    },
    {
      "id": "hatmimoha-arabic-ner",
      "name": "arabic-ner (hatmimoha)",
      "type": "tool",
      "country": "MA",
      "org": "Mohamed Hatmi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/hatmimoha/arabic-ner"
      },
      "year": 2020,
      "notes": "Named entity recognition system for Arabic.",
      "metrics": null
    },
    {
      "id": "arabic-nlp-toolkit-nidalwatfa",
      "name": "arabic-nlp-toolkit (nidalwatfa)",
      "type": "tool",
      "country": "INTL",
      "org": "Nidal Watfa",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/nidalwatfa/arabic-nlp-toolkit"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Simple open-source Python toolkit for Arabic text processing.",
      "metrics": null
    },
    {
      "id": "arabic-ocr-king-saud-university",
      "name": "Arabic-OCR",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "other",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/Kareem-Emad/arabic-ocr",
        "paper": "https://doi.org/10.1016/j.jksuci.2018.07.003"
      },
      "dialects": [
        "mixed"
      ],
      "size": "83 images",
      "year": 2018,
      "notes": "This dataset contains multi-font Arabic printed text images used for text segmentation and OCR tasks.",
      "metrics": null
    },
    {
      "id": "arabic-ocr",
      "name": "Arabic-OCR",
      "type": "ocr",
      "country": "INTL",
      "org": "ssraza21",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/ssraza21/Arabic-OCR"
      },
      "year": 2024,
      "notes": "Streamlit app that converts PDFs of Arabic text images to searchable text (PDF and Word output) using Tesseract.",
      "metrics": null
    },
    {
      "id": "hussein-arabic-ocr",
      "name": "Arabic-OCR (HusseinYoussef)",
      "type": "ocr",
      "country": "EG",
      "org": "Hussein Youssef",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/HusseinYoussef/Arabic-OCR"
      },
      "year": 2019,
      "notes": "OCR pipeline converting images of typed Arabic text into machine-encoded text.",
      "metrics": null
    },
    {
      "id": "arabic-ocr-dataset",
      "name": "Arabic-OCR-Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "mssqpi",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/mssqpi/Arabic-OCR-Dataset"
      },
      "notes": "1M+ Arabic OCR samples",
      "metrics": null
    },
    {
      "id": "arabic-pii-py",
      "name": "arabic-pii-py",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Aajil Labs",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "pii"
      ],
      "links": {
        "github": "https://github.com/Aajil-Labs/arabic-pii-py"
      },
      "year": 2026,
      "notes": "Local-first reversible PII tokenization for Arabic and Gulf data before it reaches an LLM.",
      "metrics": null
    },
    {
      "id": "arabic-poem-emotion",
      "name": "Arabic-Poem-Emotion",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "emotion"
      ],
      "links": {
        "github": "https://github.com/SakibShahriar95/Arabic-Poem-Emotion",
        "paper": "https://www.mdpi.com/2073-431X/12/5/89"
      },
      "dialects": [
        "mixed"
      ],
      "size": "9,000 sentences",
      "year": 2021,
      "notes": "A dataset containing over 9000 Arabic poems labeled by three emotion classes.",
      "metrics": null
    },
    {
      "id": "arabic-resources",
      "name": "Arabic-Resources",
      "type": "tool",
      "country": "INTL",
      "org": "NNLP-IL",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/NNLP-IL/Arabic-Resources"
      },
      "year": 2025,
      "notes": "A comprehensive list of Arabic NLP resources.",
      "metrics": null
    },
    {
      "id": "arabic-rtl-fixer-ai-skill",
      "name": "arabic-rtl-fixer-ai-skill",
      "type": "tool",
      "country": "INTL",
      "org": "ibadwi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ibadwi/arabic-rtl-fixer-ai-skill"
      },
      "year": 2026,
      "notes": "AI skill for fixing Arabic RTL, BiDi, and mixed Arabic-English content in documents, Word (DOCX), PowerPoint (PPTX), and other formatted files.",
      "metrics": null
    },
    {
      "id": "arabic-sentiment-analysis",
      "name": "arabic-sentiment-analysis",
      "type": "tool",
      "country": "PS",
      "org": "Motaz Saad",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/motazsaad/arabic-sentiment-analysis"
      },
      "year": 2025,
      "notes": "Sentiment Analysis in Arabic tweets",
      "metrics": null
    },
    {
      "id": "arabic-services",
      "name": "Arabic-Services",
      "type": "tool",
      "country": "INTL",
      "org": "Seen-Arabic",
      "license": "agpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/Seen-Arabic/Arabic-Services",
        "website": "https://seen-arabic.github.io/Arabic-Services"
      },
      "year": 2025,
      "notes": "بعض الخدمات البرمجية على نصوص اللغة العربية",
      "metrics": null
    },
    {
      "id": "arabic-services-javascript",
      "name": "Arabic-Services-JavaScript",
      "type": "tool",
      "country": "INTL",
      "org": "Seen-Arabic",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Seen-Arabic/Arabic-Services-JavaScript",
        "website": "https://www.npmjs.com/package/arabic-services"
      },
      "year": 2025,
      "notes": "A versatile library offering utility functions for processing and transforming Arabic text.",
      "metrics": null
    },
    {
      "id": "arabic-speech-recognition-by-machine-learning-and-feature-extraction",
      "name": "Arabic-Speech-Recognition-by-Machine-learning-and-feature-extraction",
      "type": "asr",
      "country": "INTL",
      "org": "AlinaBaber",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "github": "https://github.com/AlinaBaber/Arabic-Speech-Recognition-by-Machine-learning-and-feature-extraction"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "This project implements an Arabic Speech Recognition system using an ensemble voting classifier.",
      "metrics": null
    },
    {
      "id": "arabic-stop-words",
      "name": "arabic-stop-words",
      "type": "tool",
      "country": "INTL",
      "org": "mohataher",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mohataher/arabic-stop-words"
      },
      "notes": "Largest list of Arabic stop words",
      "metrics": null
    },
    {
      "id": "arabic-text-diacritization",
      "name": "arabic-text-diacritization",
      "type": "dataset",
      "country": "INTL",
      "org": "AliOsm",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/AliOsm/arabic-text-diacritization"
      },
      "notes": "Benchmark dataset with systems comparison",
      "metrics": null
    },
    {
      "id": "arabic-twitter-corpus-ajgt",
      "name": "Arabic-twitter-corpus-AJGT",
      "type": "dataset",
      "country": "INTL",
      "org": "komari6",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/komari6/Arabic-twitter-corpus-AJGT"
      },
      "dialects": [
        "lev",
        "msa"
      ],
      "year": 2023,
      "notes": "introduces an Arabic Jordanian General Tweets (AJGT) Corpus consisted of 1,800 tweets annotated as positive and negative.",
      "metrics": null
    },
    {
      "id": "arabic-vocalization",
      "name": "arabic-vocalization",
      "type": "tool",
      "country": "INTL",
      "org": "nipponjo",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/nipponjo/arabic-vocalization"
      },
      "year": 2023,
      "notes": "Shakkala and Shakkelha diacritization models ported to PyTorch and ONNX.",
      "metrics": null
    },
    {
      "id": "arabic-ai-tarjama",
      "name": "Arabic.AI (Tarjama)",
      "type": "org",
      "country": "AE",
      "org": "Arabic.AI (Tarjama)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.tarjama.com/"
      },
      "notes": "Arabic-first autonomous AI - Pronoia Arabic LLM, Agentic AI platform",
      "metrics": null
    },
    {
      "id": "arabic-ai-llm-x-llm-s",
      "name": "Arabic.AI LLM-X / LLM-S",
      "type": "llm",
      "country": "AE",
      "org": "Arabic.AI (Tarjama)",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "enterprise"
      ],
      "links": {
        "website": "https://arabic.ai/llm"
      },
      "year": 2025,
      "notes": "Dual Arabic LLMs (LLM-X, LLM-S) for sovereign on-prem enterprise deployment, validated on seven benchmarks.",
      "metrics": null
    },
    {
      "id": "arabic-ai-suite",
      "name": "Arabic.AI Suite",
      "type": "tool",
      "country": "AE",
      "org": "Arabic.AI (Tarjama)",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "translation",
        "ocr",
        "speech"
      ],
      "links": {
        "website": "https://arabic.ai"
      },
      "year": 2026,
      "notes": "Assistants, Translate, OCR and Speech (22 dialects) on one Arabic-first LLM, deployable in VPC or air-gapped.",
      "metrics": null
    },
    {
      "id": "arabic-asr-and-di",
      "name": "arabic_asr_and_di",
      "type": "asr",
      "country": "INTL",
      "org": "ai-zahran",
      "license": "gpl-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/ai-zahran/arabic_asr_and_di"
      },
      "year": 2026,
      "notes": "Arabic speech recognition and dialect identification (Red Hen Lab - GSoC 2018)",
      "metrics": null
    },
    {
      "id": "arabic-diacritization-pytorch",
      "name": "Arabic_Diacritization",
      "type": "tool",
      "country": "INTL",
      "org": "almodhfer",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/almodhfer/Arabic_Diacritization"
      },
      "year": 2021,
      "notes": "Several deep learning models for restoring Arabic diacritics in PyTorch.",
      "metrics": null
    },
    {
      "id": "arabic-mistral-7b",
      "name": "Arabic_mistral_7b",
      "type": "llm",
      "country": "EG",
      "org": "Hesham Haroon",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat",
        "instruction-tuning"
      ],
      "links": {
        "hf": "https://huggingface.co/HeshamHaroon/Arabic_mistral_7b"
      },
      "size": "7B",
      "on_device": false,
      "year": 2024,
      "notes": "Arabic Mistral 7B adapter trained with Unsloth; tagged ar, text-generation. The card lists no base model or dataset.",
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2024-05-06"
      }
    },
    {
      "id": "arabic-ocr-from-pdf",
      "name": "Arabic_OCR_From_PDF",
      "type": "ocr",
      "country": "INTL",
      "org": "zaakki-ahamed",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/zaakki-ahamed/Arabic_OCR_From_PDF"
      },
      "year": 2023,
      "notes": "Perform Optical Character Recognition (OCR) on a scanned PDF file containing Arabic text and output a searchable PDF",
      "metrics": null
    },
    {
      "id": "arabic-vocalizer",
      "name": "arabic_vocalizer",
      "type": "tool",
      "country": "INTL",
      "org": "nipponjo",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/nipponjo/arabic_vocalizer"
      },
      "notes": "Deep-learning diacritization (ONNX format)",
      "metrics": null
    },
    {
      "id": "arabica",
      "name": "Arabica",
      "type": "tool",
      "country": "INTL",
      "org": "PetrKorab",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/PetrKorab/Arabica"
      },
      "year": 2025,
      "notes": "Python package for text mining of time-series data",
      "metrics": null
    },
    {
      "id": "arabicaqa",
      "name": "ArabicaQA",
      "type": "dataset",
      "country": "INTL",
      "org": "DataScienceUIBK",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/DataScienceUIBK/ArabicaQA"
      },
      "notes": "Large-scale Arabic Question Answering",
      "metrics": null
    },
    {
      "id": "arabiccompetitiveprogramming",
      "name": "ArabicCompetitiveProgramming",
      "type": "tool",
      "country": "INTL",
      "org": "mostafa-saad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mostafa-saad/ArabicCompetitiveProgramming",
        "website": "https://www.youtube.com/user/nobody123497"
      },
      "year": 2023,
      "notes": "The repository contains the ENGLISH description files attached to the video series in my ARABIC algorithms channel.",
      "metrics": null
    },
    {
      "id": "arabiccr",
      "name": "ArabiCCR",
      "type": "dataset",
      "country": "SA",
      "org": "University of Ha’il",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "legal"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/np538c95yy/2",
        "paper": "https://doi.org/10.1016/j.dib.2026.112844"
      },
      "dialects": [
        "msa"
      ],
      "size": "12,806 documents",
      "year": 2026,
      "notes": "Commercial Arabic court rulings dataset from the official Saudi Ministry of Justice.",
      "metrics": null
    },
    {
      "id": "arabichatespeechdataset",
      "name": "ArabicHateSpeechDataset",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Regina",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "offensive-language",
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/sbalsefri/ArabicHateSpeechDataset",
        "paper": "https://www.sciencedirect.com/science/article/abs/pii/S2468696420300379?fr=RR-2&ref=pdf_download&rr=8c11aa96ed17794c"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,361 sentences",
      "year": 2020,
      "notes": "The dataset contains 5361 Arabic tweets annotated with six categories: clean, offensive, and hateful speech.",
      "metrics": null
    },
    {
      "id": "arabicner-lucas",
      "name": "ArabicNER",
      "type": "tool",
      "country": "INTL",
      "org": "Liyuan Liu",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/LiyuanLucasLiu/ArabicNER"
      },
      "year": 2019,
      "notes": "Arabic named entity recognition system with strong reported performance.",
      "metrics": null
    },
    {
      "id": "arabicnlptoolslist",
      "name": "arabicnlptoolslist",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/linuxscout/arabicnlptoolslist"
      },
      "year": 2026,
      "notes": "Arabic NLP tools List inventory",
      "metrics": null
    },
    {
      "id": "arabicprocess",
      "name": "arabicprocess",
      "type": "tool",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "website": "https://pypi.org/project/arabicprocess/"
      },
      "notes": "Python library for Arabic preprocessing",
      "metrics": null
    },
    {
      "id": "arabicstopwords",
      "name": "ArabicStopWords (linuxscout)",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "github": "https://github.com/linuxscout/arabicstopwords"
      },
      "year": 2016,
      "notes": "Classified Arabic stop word list; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "arabicsurvey",
      "name": "ArabicSurvey",
      "type": "tool",
      "country": "INTL",
      "org": "iwan-rg",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "survey"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/ArabicSurvey"
      },
      "year": 2026,
      "notes": "مستودع الأوراق المسحية في معالجة اللغة العربية (أسبر) A Repository for survey and review papers in Arabic Natural Language processing (ANLP).",
      "metrics": null
    },
    {
      "id": "mtg-arabic-transliterator",
      "name": "ArabicTransliterator",
      "type": "tool",
      "country": "INTL",
      "org": "Music Technology Group",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/MTG/ArabicTransliterator"
      },
      "year": 2014,
      "notes": "Romanizes Arabic text following the ALA-LC scheme.",
      "metrics": null
    },
    {
      "id": "arabictts",
      "name": "ArabicTTS",
      "type": "tts",
      "country": "INTL",
      "org": "karrarkazuya",
      "license": "gpl-3.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/karrarkazuya/ArabicTTS"
      },
      "year": 2023,
      "notes": "ArabicTTS (TextToSpeech) Android library with a sample",
      "metrics": null
    },
    {
      "id": "arabicweb16",
      "name": "ArabicWeb16",
      "type": "dataset",
      "country": "INTL",
      "org": "TOBB University of Economics and Technology",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://sites.google.com/view/arabicweb16/",
        "paper": "https://www.ischool.utexas.edu/~ml/papers/sigir16-arabicweb.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "150,211,934 documents",
      "year": 2016,
      "notes": "Public Web crawl of 150,211,934 Arabic Web pages with high coverage of dialectal Arabic as well as Modern Standard Arabic (MSA)",
      "metrics": null
    },
    {
      "id": "arabimaak-mcp",
      "name": "ArabiMaak MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "deepdiver4ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/deepdiver4ai/arabimaak-mcp"
      },
      "year": 2026,
      "notes": "Gulf Arabic language MCP server for word, phrase and cultural-context lookup.",
      "metrics": null
    },
    {
      "id": "arabot",
      "name": "Arabot",
      "type": "org",
      "country": "AE",
      "org": "Arabot",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://arabot.io/"
      },
      "notes": "Conversational AI for Arabic - Arabic NLP chatbot engine",
      "metrics": null
    },
    {
      "id": "arabscribe",
      "name": "ArabScribe",
      "type": "dataset",
      "country": "INTL",
      "org": "NYU",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "lexicon"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/arabscribe/",
        "paper": "https://aclanthology.org/W17-1315.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,234 tokens",
      "year": 2017,
      "tags": [
        "multilingual"
      ],
      "notes": "The ArabScribe dataset contains 10,000 transcriptions of Arabic words with both Roman and Arabic keyboards based on audio impressions of native.",
      "metrics": null
    },
    {
      "id": "arabsign",
      "name": "ArabSign",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://hamzah-luqman.github.io/ArabSign/",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?arnumber=10042720"
      },
      "dialects": [
        "msa"
      ],
      "size": "10 hours",
      "year": 2023,
      "tags": [
        "multilingual"
      ],
      "notes": "A continuous Arabic Sign Language dataset with 9,335 video samples from 6 signers across 3 modalities (color, depth, skeleton).",
      "metrics": null
    },
    {
      "id": "arabycia",
      "name": "Arabycia",
      "type": "tool",
      "country": "INTL",
      "org": "mohabmes",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "diacritization",
        "pos",
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/mohabmes/Arabycia"
      },
      "notes": "Arabic NLP tool used to perform Text Search, POS tagging, Translation, auto-diacritization, etc..",
      "metrics": null
    },
    {
      "id": "araclean",
      "name": "araclean",
      "type": "tool",
      "country": "INTL",
      "org": "MhdMartini",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/MhdMartini/araclean"
      },
      "notes": "Offset-preserving Arabic text normalization & cleaning for NLP — map cleaned spans back to the original.",
      "metrics": null
    },
    {
      "id": "aracon",
      "name": "AraCon",
      "type": "tool",
      "country": "INTL",
      "org": "JaouadMousser",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/JaouadMousser/Aracon"
      },
      "notes": "ARACON is a verb conjugator for Arabic implemented as part of a morphological Analyser and generator.",
      "metrics": null
    },
    {
      "id": "aracust",
      "name": "AraCust",
      "type": "dataset",
      "country": "INTL",
      "org": "Durham University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://peerj.com/articles/cs-510/#supplemental-information",
        "paper": "https://peerj.com/articles/cs-510/#MainContent"
      },
      "dialects": [
        "gulf"
      ],
      "size": "20,000 sentences",
      "year": 2021,
      "notes": "Saudi Telecom Tweets corpus for sentiment analysis",
      "metrics": null
    },
    {
      "id": "arad-mohamaddarvishi",
      "name": "Arad",
      "type": "tool",
      "country": "INTL",
      "org": "MohamadDarvishi",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MohamadDarvishi/Arad",
        "website": "https://mohamaddarvishi.ir/Arad/"
      },
      "year": 2026,
      "notes": "یک فونت انگلیسی-عربی (فارسی).",
      "metrics": null
    },
    {
      "id": "arad-mdarvishi5124",
      "name": "Arad (MDarvishi5124)",
      "type": "tool",
      "country": "INTL",
      "org": "MDarvishi5124",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MDarvishi5124/Arad",
        "website": "https://mdarvishi5124.github.io/Arad/"
      },
      "year": 2024,
      "notes": "یک فونت انگلیسی-عربی (فارسی).",
      "metrics": null
    },
    {
      "id": "aradhati",
      "name": "AraDhati+",
      "type": "dataset",
      "country": "INTL",
      "org": "Université de Ghardaia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/Attia14/AraDhati",
        "paper": "http://arxiv.org/pdf/2508.19966v1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "77,916 sentences",
      "year": 2025,
      "notes": "Dataset for Arabic subjectivity classification combined by levering existing Arabic datasets.",
      "metrics": null
    },
    {
      "id": "aradice",
      "name": "AraDiCE",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2409.11404"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Benchmarks for dialectal and cultural capabilities of LLMs",
      "metrics": null
    },
    {
      "id": "arafacts",
      "name": "AraFacts",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.26/",
        "website": "https://gitlab.com/bigirqu/AraFacts/"
      },
      "year": 2021,
      "venue": "WANLP 2021",
      "notes": "We introduce AraFacts, the first large Arabic dataset of naturally occurring claims collected from 5 Arabic fact-checking websites, e.g., Fatabyyano.",
      "metrics": null
    },
    {
      "id": "arafinnlp-shared-task",
      "name": "AraFinNLP shared task",
      "type": "benchmark",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "translation"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/arbanking77/arafinnlp"
      },
      "year": 2024,
      "notes": "Arabic Financial NLP shared task: multi-dialect intent detection and cross-dialect translation in banking.",
      "metrics": null
    },
    {
      "id": "arallama",
      "name": "AraLLaMA",
      "type": "llm",
      "country": "INTL",
      "org": "Bashar Talafha",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/bashar-talafha/AraLLaMA"
      },
      "size": "7B",
      "notes": "LLaMA2 pre-trained on Arabic data",
      "metrics": null
    },
    {
      "id": "aramco-digital",
      "name": "Aramco Digital",
      "type": "org",
      "country": "SA",
      "org": "Aramco Digital",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.aramcodigital.com"
      },
      "notes": "Aramco digital arm, behind Metabrain generative AI assistant and AI partnerships.",
      "metrics": null
    },
    {
      "id": "aramed",
      "name": "AraMed",
      "type": "dataset",
      "country": "SA",
      "org": "King Khalid University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/Waadtss/AraMed",
        "paper": "https://aclanthology.org/2024.osact-1.6.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "137,293 sentences",
      "year": 2024,
      "notes": "Large-scale Arabic Medical Question Answering dataset.",
      "metrics": null
    },
    {
      "id": "aranet",
      "name": "AraNet",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "emotion",
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/AraNet"
      },
      "year": 2020,
      "notes": "AraNet: deep learning toolkit for Arabic social media (age, gender, dialect, emotion, irony, sentiment)",
      "metrics": null
    },
    {
      "id": "aranizer",
      "name": "AraNizer",
      "type": "tool",
      "country": "SA",
      "org": "riotu-lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "tokenization"
      ],
      "links": {
        "github": "https://github.com/riotu-lab/aranizer"
      },
      "year": 2023,
      "notes": "Custom SentencePiece and BPE tokenizers tailored for Arabic.",
      "metrics": null
    },
    {
      "id": "aranjiyyacorpus-a-span-annotated-arabic-news-dataset-of-anglicised-sty",
      "name": "AranjiyyaCorpus: A span-annotated Arabic news dataset of anglicised style, calques, and borrowings across six domains",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "span-annotation",
        "style"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/rh8gf885hz"
      },
      "year": 2026,
      "notes": "AranjiyyaCorpus: span-annotated Arabic news dataset of anglicised style, Arabic prose with imported foreign constructions.",
      "metrics": null
    },
    {
      "id": "arap-tweet-corpus",
      "name": "Arap-Tweet Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "dialect-id",
        "author-profiling",
        "authorship-attribution"
      ],
      "links": {
        "website": "https://arap.qatar.cmu.edu/",
        "paper": "https://arxiv.org/pdf/1808.07674"
      },
      "dialects": [
        "mixed"
      ],
      "size": "2,400,000 sentences",
      "year": 2018,
      "notes": "Arap-Tweet is a large-scale, multi-dialectal Arabic Twitter corpus containing 2.4 million tweets from 11 regions across 16 countries in the Arab world.",
      "metrics": null
    },
    {
      "id": "arareq",
      "name": "AraREQ",
      "type": "dataset",
      "country": "PS",
      "org": "Birzeit University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "requirements",
        "conflict-detection"
      ],
      "links": {
        "website": "https://docs.google.com/forms/d/e/1FAIpQLSc5_BS12CQ4FHWBn_TC7xTOReB3NtXcnFnPUfHIswDu9-hQ8w/viewform",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2026/pdf/2026.lrec2026-1.553.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "448 documents",
      "year": 2026,
      "notes": "A large-scale Arabic dataset for requirement-level conflict detection and resolution.",
      "metrics": null
    },
    {
      "id": "arasafe",
      "name": "AraSafe",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "safety"
      ],
      "links": {
        "website": "https://www.fanar.qa/en/publications",
        "github": "https://github.com/qcri/AraSafe-benchmark"
      },
      "year": 2025,
      "notes": "Benchmark for safety in Arabic LLMs.",
      "metrics": null
    },
    {
      "id": "araspell",
      "name": "AraSpell",
      "type": "tool",
      "country": "INTL",
      "org": "Msalhab96",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/msalhab96/AraSpell"
      },
      "year": 2022,
      "notes": "Framework for Arabic spelling correction with several seq2seq architectures.",
      "metrics": null
    },
    {
      "id": "araspider",
      "name": "ARASPIDER",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "text-to-sql"
      ],
      "links": {
        "github": "https://github.com/ahmedheakl/AraSpider",
        "paper": "https://arxiv.org/pdf/2402.07448"
      },
      "dialects": [
        "msa"
      ],
      "size": "10,181 tokens",
      "year": 2024,
      "notes": "AraSpider is a translated version of the Spider dataset, which is commonly used for semantic parsing and text-to-SQL generation.",
      "metrics": null
    },
    {
      "id": "arastories",
      "name": "AraStories",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "story-generation"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/arastories",
        "paper": "https://arxiv.org/pdf/2407.07551v1.pdf"
      },
      "year": 2024,
      "notes": "Code and data for Arabic automatic story generation with large language models (ArabicNLP 2024).",
      "metrics": null
    },
    {
      "id": "aravec",
      "name": "aravec",
      "type": "embedding",
      "country": "EG",
      "org": "Bakrianoo",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "pretraining"
      ],
      "links": {
        "github": "https://github.com/bakrianoo/aravec"
      },
      "year": 2021,
      "notes": "AraVec is a pre-trained distributed word representation (word embedding) open source project which aims to provide the Arabic NLP research community.",
      "metrics": null
    },
    {
      "id": "arbml",
      "name": "ARBML",
      "type": "org",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/ARBML"
      },
      "notes": "Democratizing Arabic NLP - masader, klaam, tkseem",
      "metrics": null
    },
    {
      "id": "arbml-collection",
      "name": "ARBML",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ARBML/ARBML"
      },
      "year": 2019,
      "notes": "Collection of Arabic NLP and computer vision projects with demo notebooks.",
      "metrics": null
    },
    {
      "id": "arbqa",
      "name": "ArbQA",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Petra",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/msfasha/Research-Resources/tree/main/ArabicLegalLLM",
        "paper": "https://arxiv.org/pdf/2601.17364v1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,000 sentences",
      "year": 2026,
      "notes": "The generated dataset comprised around 6,000 QA pairs, and the context was used to create a question and answer based on each legal article",
      "metrics": null
    },
    {
      "id": "arcee-meraj",
      "name": "Arcee-Meraj",
      "type": "llm",
      "country": "INTL",
      "org": "Arcee AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/arcee-ai/Arcee-Meraj"
      },
      "size": "72B",
      "notes": "Enterprise Arabic LLM based on Qwen2-72B",
      "metrics": null
    },
    {
      "id": "arcee-meraj-mini",
      "name": "Arcee-Meraj-Mini",
      "type": "llm",
      "country": "INTL",
      "org": "Arcee AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/arcee-ai/Arcee-Meraj-Mini"
      },
      "size": "7B",
      "notes": "Top OALL among 7B models",
      "metrics": null
    },
    {
      "id": "arcybc",
      "name": "ArCybC",
      "type": "dataset",
      "country": "JO",
      "org": "University of Jordan",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/z2dfgrzx47/1",
        "paper": "https://www.researchgate.net/publication/360238061_The_design_construction_and_evaluation_of_annotated_Arabic_cyberbullying_corpus"
      },
      "dialects": [
        "mixed"
      ],
      "size": "4,505 sentences",
      "year": 2022,
      "notes": "Multi-dialect Arabic cyberbullying corpus of 4,505 annotated tweets spanning gaming, sports, news, and celebrity topics.",
      "metrics": null
    },
    {
      "id": "areeg-arabic-inner-speech-eeg-dataset",
      "name": "ArEEG: Arabic Inner Speech EEG Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "eslam101ahmed",
      "license": "gpl-3.0",
      "modality": "multimodal",
      "tasks": [
        "eeg",
        "inner-speech"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/eslam101ahmed/arabic-eeg-sessions"
      },
      "year": 2025,
      "notes": "Kaggle dataset of EEG recordings for Arabic inner speech.",
      "metrics": null
    },
    {
      "id": "aref-ruqaa",
      "name": "aref-ruqaa",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aliftype/aref-ruqaa"
      },
      "year": 2026,
      "notes": "Aref Ruqaa (رقعة عارف) is a Ruqaa typeface",
      "metrics": null
    },
    {
      "id": "arich-and-balanced-phonetics-corpus-for-modern-standard-arabic-asr-sys",
      "name": "Arich and balanced phonetics corpus for modern standard Arabic ASR systems",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "asr",
        "phonetics"
      ],
      "links": {
        "website": "https://zenodo.org/records/19838862"
      },
      "dialects": [
        "msa"
      ],
      "year": 2026,
      "notes": "Balanced Modern Standard Arabic phonetic corpus following Zipf's law, for ASR system development.",
      "metrics": null
    },
    {
      "id": "armi-arabic-misogynistic-dataset",
      "name": "ArMI: Arabic Misogynistic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ORSAM Center for Middle Eastern Studies",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "gender-bias-detection"
      ],
      "links": {
        "github": "https://github.com/bilalghanem/armi",
        "paper": "http://ceur-ws.org/Vol-3159/T5-1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "9,833 sentences",
      "year": 2022,
      "notes": "Arabic multidialectal dataset for misogynistic language",
      "metrics": null
    },
    {
      "id": "arnli",
      "name": "ArNLI",
      "type": "dataset",
      "country": "INTL",
      "org": "Arab International University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "github": "https://github.com/Khloud-AL/ArNLI",
        "paper": "https://arxiv.org/pdf/2209.13953v1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,366 sentences",
      "year": 2022,
      "notes": "Arabic NLI dataset for natural language inference collected by translating and re-annotating English and Arabic datasets.",
      "metrics": null
    },
    {
      "id": "arparallel",
      "name": "arparallel",
      "type": "dataset",
      "country": "INTL",
      "org": "Carnegie Mellon University",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "translation",
        "summarization",
        "pretraining",
        "retrieval"
      ],
      "links": {
        "website": "http://www.cs.cmu.edu/~fraisi/arabic/arparallel/",
        "paper": "https://pdf.sciencedirectassets.com/280203/1-s2.0-S1877050918X00192/1-s2.0-S1877050918321914/main.pdf?X-Amz-Security-Token=IQoJb3JpZ2luX2VjEC0aCXVzLWVhc3QtMSJIMEYCIQCbQvxVlZ9GIUmIOvgPUAl1BqNBeWsStsWYH5rg%2BT1HqgIhAPrEmo9yzsahnXBTJzuhZbCPg2kpfd12iRAEKBMToduAKrMFCFYQBRoMMDU5MDAzNTQ2ODY1Igwm4ZFMRlungcpl2U4qkAU8nYrBMd7oiFvmaJTOPWzC3L2uw1kahG0Yrkula18wW0zx%2FIUkV3V6LItDPYC31iLDYwbIEC5g1PrzAcKScy1EWnaWYOidb4N71kI8BzxhNvwOoyPAXkJMhBf%2FDOCL8Q%2FV2VXT%2FuBW9y1zF5o%2F1F49DHcLumSE9Lm7suYDvArrmn7FQ9tfjEsTa7NbKbgNMsvWYWCJae9cUYPEeAVWRVLyoC0R9q0YCzRTz%2F8PdZtLq5TmIxeoNjNbPPJMErPeU8rFn7nyhYoWY0RdJPduL2OYU3y%2Fg%2Fnwhvjp6ken6Z0NqDnl4YWq5QuqC78O199%2BlNEBugyARzjt2KAu7Tl0mu1YTnaXpri6prtemy1fL5pn5GPmcjtAbhHntSL1lOCtLafdjQsC8G1D3JuKr%2F%2B9%2FhxvoZN4jEWCNfKVnyriSO%2FrFvhNqx7F%2BZlHogHPraPbPMK%2BXhSufvAscyIBdzD3OL%2B%2FQky2gWz8iHvXl6RySxLhAXLwrqkNquO6nynCjvQ5BsWH%2FH2iAnGNkwNT7%2B1NCwhCA4CqSVA9H72kau4v0w2wgO%2FCLbV4cYFseWi6cNPCiPiaJwq%2FjzjHQ385N2NzUKuFxNg2bM5EfIXijGHJvzD5hA2AFmG74B4up232ReZWLPneXFcbKeTS71u0FaLZXxCzpaV%2BZc6oagDe%2Bhp57ExBsoSnzUud9%2BkZcNamxKzpA2Gr0PL3ag3%2FhXNIn0lYh0wcnMJNArcnfRhCKjHceb4X%2BVLcFFT5fhqxUK36gs5gebTNkcXVhUlC6%2B49p8%2Bk0aLZMueDnxZ%2B9vZkI9g439epOJiW%2Br6IlbatFcs8Fy0Cec%2FTWHm96DDUEMdmjsKxfq%2FVgVj5paFSSU8F38us6vabCzCN%2FPm2BjqwAb3UZR4cdrow6VtHEg4fvhpw7oxSngzeKNzydVoQcSASHKrOB5AAEqNmFxN4Ywr8Lx9mkIXaVvMIInd2tHv5SOfhKAw2cmYZZ1XzWvOh8QJC%2F2gN1v4qYBsDckmV5XUYZeTovzZhMn2Cp8nLI9f786fqdh%2FMajAXuK%2Fex6cDBcc2Gd5bKeTdlNPn4o4jzEZ8CJvTEi9i3SJqXJqoBuLt%2Bp38j6Yu5fMqycsvNX8UVLkE&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Date=20240909T054833Z&X-Amz-SignedHeaders=host&X-Amz-Expires=300&X-Amz-Credential=ASIAQ3PHCVTYUUT2AQTS%2F20240909%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Signature=414eec0d0f0d5a45ad2d256adea15fe60e1bc0a192948f11eb3b0d71c4b30273&hash=8b112bc1cce1c6665a5739e0ab392dc42397c9c6ae1b9d15b8c5173fd382656d&host=68042c943591013ac2b2430a89b270f6af2c76d8dfd086a07176afe7c76c2c61&pii=S1877050918321914&tid=spdf-eb173916-2ec2-4fd7-8ad8-818dcda8cb5d&sid=0873a5bd51e572450a0af2f40180fda8c32fgxrqb&type=client&tsoh=d3d3LnNjaWVuY2VkaXJlY3QuY29t&ua=1f055a03575004560b&rr=8c04e3d71ecb6edb&cc=qa"
      },
      "dialects": [
        "msa"
      ],
      "size": "100,000 sentences",
      "year": 2018,
      "notes": "The first monolingual parallel corpus of Arabic generated automatically from translating a bilingual English-French corpus.",
      "metrics": null
    },
    {
      "id": "arpatent",
      "name": "ArPatent",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/Arabic-Patents",
        "paper": "https://aclanthology.org/2022.wanlp-1.26.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "9,765 documents",
      "year": 2022,
      "notes": "The first public Arabic patent dataset for classification.",
      "metrics": null
    },
    {
      "id": "arpc-a-corpus-for-paraphrase-identification-in-arabic-text",
      "name": "ARPC: A Corpus for Paraphrase Identification in Arabic Text",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "paraphrase-identification"
      ],
      "links": {
        "website": "https://ieee-dataport.org/documents/arpc-corpus-paraphrase-identification-arabic-text#files"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,331 sentences",
      "year": 2019,
      "notes": "ArPC: Arabic paraphrase identification corpus of 1,331 sentence pairs with binary scores.",
      "metrics": null
    },
    {
      "id": "arpd-the-academic-arabic-research-papers-dataset-corpus",
      "name": "ARPD: The Academic Arabic Research Papers Dataset (corpus).",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "academic-papers"
      ],
      "links": {
        "website": "https://zenodo.org/records/15547781"
      },
      "year": 2025,
      "notes": "ARPD: corpus of academic Arabic research papers for NLP models and text analysis.",
      "metrics": null
    },
    {
      "id": "arpfn",
      "name": "ArPFN",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://gitlab.com/bigirqu/ArPFN",
        "paper": "https://aclanthology.org/2022.osact-1.2.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,546 tokens",
      "year": 2022,
      "notes": "1546-user Arabic Twitter dataset for fake news.",
      "metrics": null
    },
    {
      "id": "arpm-corpus-punctuations-corpus-for-arabic",
      "name": "ArPM corpus: Punctuations corpus for Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "punctuation"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/jnz483dypx"
      },
      "year": 2024,
      "notes": "ArPM: Arabic corpus for punctuation prediction covering period, comma, question mark, exclamation mark and colon.",
      "metrics": null
    },
    {
      "id": "arpot",
      "name": "ArPoT",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "parsing"
      ],
      "links": {
        "github": "https://github.com/ArPoT-KSU/ArPoT_v1.0",
        "paper": "https://aclanthology.org/2021.depling-1.1.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "35,459 tokens",
      "year": 2021,
      "notes": "First syntactic treebank for Classical Arabic poetry collected from adab and aldiwan.",
      "metrics": null
    },
    {
      "id": "arramooz",
      "name": "Arramooz",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/linuxscout/arramooz"
      },
      "year": 2016,
      "notes": "Arabic dictionary for morphological analysis; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "arsen",
      "name": "ArSen",
      "type": "benchmark",
      "country": "INTL",
      "org": "Huaibei Normal University",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/123fangyang/ArSen",
        "paper": "https://aclanthology.org/2024.conll-1.39.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,193 sentences",
      "year": 2024,
      "notes": "Arabic COVID-19 sentiment analysis benchmark.",
      "metrics": null
    },
    {
      "id": "arsl-31-pilot-arabic-sign-language-arsl-dataset",
      "name": "ArSL-31-Pilot: Arabic Sign Language (ArSL) Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://zenodo.org/records/18363162"
      },
      "year": 2026,
      "notes": "ArSL-31-Pilot: right-hand landmark coordinates extracted with MediaPipe from Arabic Sign Language videos.",
      "metrics": null
    },
    {
      "id": "arsyra-medical-arabic-healthcare-dialect-data",
      "name": "ArSyra Medical Arabic — Healthcare Dialect Data",
      "type": "dataset",
      "country": "INTL",
      "org": "aqlomate",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "medical"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/aqlomate/arsyra-medical"
      },
      "notes": "Arabic medical dialect data covering healthcare terminology, symptoms, medical conversations, and clinical expressions in regional Arabic dialects.",
      "metrics": null
    },
    {
      "id": "arsyra-sudanese-arabic-dataset",
      "name": "ArSyra Sudanese Arabic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "aqlomate",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "dialect-corpus"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/aqlomate/arsyra-sudanese"
      },
      "dialects": [
        "sudanese"
      ],
      "notes": "Kaggle dataset of Sudanese Arabic text covering linguistic categories, part of the ArSyra series.",
      "metrics": null
    },
    {
      "id": "artest",
      "name": "ArTest",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "website": "https://www.dropbox.com/s/openq7fgt3kd6jg/Artest-Test-Collection.zip?dl=0",
        "paper": "https://dl.acm.org/doi/pdf/10.1145/3397271.3401223"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,529 sentences",
      "year": 2020,
      "notes": "ArTest: Arabic test collection for retrieval built on top of the ArabicWeb'16 web collection.",
      "metrics": null
    },
    {
      "id": "artst-data",
      "name": "ArTST",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "tts"
      ],
      "links": {
        "github": "https://github.com/mbzuai-nlp/ArTST"
      },
      "year": 2023,
      "dialects": [
        "msa"
      ],
      "notes": "Arabic text and speech transformer repository with data recipes for ASR and TTS.",
      "metrics": null
    },
    {
      "id": "artstv2",
      "name": "ArTSTv2",
      "type": "asr",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech-pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI/ArTSTv2"
      },
      "year": 2025,
      "notes": "ArTST v2 checkpoints: dialect-pretrained base plus models fine-tuned for ASR (MGB2) and other Arabic speech tasks.",
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2025-09-10"
      }
    },
    {
      "id": "arwi",
      "name": "ARWI",
      "type": "tool",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "writing-assistant"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.11814",
        "github": "https://github.com/mbzuai-nlp/arabic-aes-bea25"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "Writing assistant helping Arabic language learners improve MSA writing with feedback.",
      "metrics": null
    },
    {
      "id": "arwordvec",
      "name": "ArWordVec",
      "type": "embedding",
      "country": "INTL",
      "org": "mmdoha200",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "github": "https://github.com/mmdoha200/ArWordVec"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "year": 2020,
      "notes": "Word-embedding models trained on Arabic tweets.",
      "metrics": null
    },
    {
      "id": "arysl",
      "name": "ArYSL",
      "type": "dataset",
      "country": "YE",
      "org": "Yemeni researchers",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "paper": "https://pmc.ncbi.nlm.nih.gov/articles/PMC12433467/"
      },
      "year": 2025,
      "notes": "First large-scale open dataset of Arabic Yemeni Sign Language for accessible technology.",
      "metrics": null
    },
    {
      "id": "arzen",
      "name": "ArzEn",
      "type": "dataset",
      "country": "EG",
      "org": "German University in Cairo",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://sites.google.com/view/arzen-corpus/resources?authuser=0",
        "paper": "https://aclanthology.org/2020.lrec-1.523.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "12 hours",
      "year": 2020,
      "tags": [
        "multilingual"
      ],
      "notes": "12-hour Egyptian Arabic-English code-switched spontaneous speech corpus with transcriptions and speaker metadata.",
      "metrics": null
    },
    {
      "id": "asad-arabic-social-media-analytics-and-understanding",
      "name": "ASAD (Arabic Social media Analytics and unDerstanding)",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "emotion"
      ],
      "links": {
        "website": "https://alt.qcri.org/demos"
      },
      "notes": "Toolkit providing user profiling, sentiment, emotion and inappropriate-content detection for Arabic social media.",
      "metrics": null
    },
    {
      "id": "asad-cleaned",
      "name": "ASAD Cleaned",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "speech-act-classification"
      ],
      "links": {
        "github": "https://github.com/alshehrikhadejaa/ASAD",
        "paper": "https://arxiv.org/pdf/2401.17373v1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "22,352 sentences",
      "year": 2024,
      "notes": "Annotated subset of ASAD covering six speech act categories in cleaned Arabic social media text.",
      "metrics": null
    },
    {
      "id": "asas-ai",
      "name": "ASAS AI",
      "type": "org",
      "country": "INTL",
      "org": "ASAS AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://asas.ai"
      },
      "notes": "Saudi AI company offering Arabic models, automation and in-Kingdom infrastructure.",
      "metrics": null
    },
    {
      "id": "asayar",
      "name": "ASAYAR",
      "type": "dataset",
      "country": "MA",
      "org": "Sidi Mohamed Ben Abdellah University",
      "license": "proprietary",
      "modality": "vision",
      "tasks": [
        "scene-text-detection"
      ],
      "links": {
        "website": "https://vcar.github.io/ASAYAR/",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=9233923"
      },
      "dialects": [
        "magh"
      ],
      "size": "1,800 images",
      "year": 2020,
      "notes": "The ASAYAR dataset is the first public dataset for Arabic and Latin scene text detection in highway traffic panels.",
      "metrics": null
    },
    {
      "id": "asbab-al-nuzul-dataset",
      "name": "asbab-al-nuzul-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "mostafaahmed97",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/mostafaahmed97/asbab-al-nuzul-dataset"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A structured dataset of authenticated Asbāb al-Nuzūl (occasions of revelation) for Qur’anic verses, available in plaintext, JSON.",
      "metrics": null
    },
    {
      "id": "ashaar",
      "name": "Ashaar",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry",
        "generation"
      ],
      "links": {
        "github": "https://github.com/ARBML/Ashaar"
      },
      "year": 2020,
      "notes": "Arabic poetry analysis and generation toolkit.",
      "metrics": null
    },
    {
      "id": "ask-hadith",
      "name": "ask-hadith",
      "type": "tool",
      "country": "INTL",
      "org": "Ananto30",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "github": "https://github.com/Ananto30/ask-hadith",
        "website": "https://askhadith.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "🔎 A Hadith search engine",
      "metrics": null
    },
    {
      "id": "assem-arabic-stemmer",
      "name": "Assem Arabic Light Stemmer",
      "type": "tool",
      "country": "DZ",
      "org": "Assem Chelli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stemming"
      ],
      "links": {
        "github": "https://github.com/assem-ch/arabicstemmer"
      },
      "year": 2016,
      "notes": "Snowball-based Arabic light stemming algorithm with multiple language ports; Algeria.",
      "metrics": null
    },
    {
      "id": "astra-tech-botim",
      "name": "Astra Tech (Botim)",
      "type": "org",
      "country": "AE",
      "org": "Astra Tech (Botim)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://astratech.ae"
      },
      "notes": "Abu Dhabi tech group behind Botim, building AI assistant products.",
      "metrics": null
    },
    {
      "id": "aswat",
      "name": "Aswat",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.10/",
        "github": "https://github.com/AswatDataset/AswatDataset"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "Furthermore, we introduce Aswat dataset, which covers multiple genres and features speakers with vocal variety.",
      "metrics": null
    },
    {
      "id": "atlas-chat",
      "name": "Atlas-Chat",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/MBZUAI-Paris/atlas-chat"
      },
      "base_model": [
        "google/gemma-2-2b-it",
        "google/gemma-2-9b-it",
        "google/gemma-2-27b-it"
      ],
      "size": "2B-27B",
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan Darija dialect",
      "metrics": null
    },
    {
      "id": "atlasia",
      "name": "AtlasIA",
      "type": "org",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.atlasia.ma/"
      },
      "notes": "Behind AL Atlas Moroccan Darija models",
      "metrics": null
    },
    {
      "id": "atlasocr",
      "name": "AtlasOCR",
      "type": "ocr",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/AtlasOCR"
      },
      "dialects": [
        "magh"
      ],
      "notes": "First Darija/Moroccan Arabic OCR, Qwen2.5-VL-3B based",
      "metrics": {
        "downloads": 0,
        "likes": 8,
        "lastModified": "2025-09-16"
      }
    },
    {
      "id": "attica",
      "name": "ATTICA",
      "type": "dataset",
      "country": "MA",
      "org": "Sidi Mohamed Ben Abdellah University",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "scene-text-detection"
      ],
      "links": {
        "github": "https://github.com/kkawtar/ATTICA",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?arnumber=9466101"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,215 images",
      "year": 2021,
      "notes": "Arabic text-based traffic panels detection dataset with 1215 images.",
      "metrics": null
    },
    {
      "id": "aub-mind-lab",
      "name": "AUB MIND Lab",
      "type": "org",
      "country": "LB",
      "org": "AUB MIND Lab",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/aub-mind"
      },
      "notes": "Foundational Arabic NLP models - AraBERT, AraGPT2, AraELECTRA",
      "metrics": null
    },
    {
      "id": "audar-diarization-v1",
      "name": "Audar Diarization V1",
      "type": "asr",
      "country": "INTL",
      "org": "Audar AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/audarai/Audar-Diarization-V1"
      },
      "year": 2026,
      "notes": "Real-time streaming speaker diarization — up to 8 speakers, state of the art on 8 corpora.",
      "metrics": {
        "downloads": 0,
        "likes": 7,
        "lastModified": "2026-07-15"
      }
    },
    {
      "id": "audar-asr-v1",
      "name": "Audar-ASR-V1",
      "type": "asr",
      "country": "INTL",
      "org": "Audar AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/AudarAI/Audar-ASR-V1"
      },
      "year": 2026,
      "notes": "Arabic-first generative speech recognition in Flash and Turbo variants.",
      "metrics": null
    },
    {
      "id": "audar-tts-v1",
      "name": "Audar-TTS-V1",
      "type": "tts",
      "country": "INTL",
      "org": "Audar AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts",
        "voice-cloning"
      ],
      "links": {
        "github": "https://github.com/AudarAI/Audar-TTS-V1"
      },
      "year": 2026,
      "notes": "Arabic-first expressive zero-shot speech synthesis, Flash and Turbo variants.",
      "metrics": null
    },
    {
      "id": "aured",
      "name": "AuRED",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.3/",
        "github": "https://github.com/Fatima-Haouari/AuRED"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We introduce the new task of rumor verification using evidence that are exclusively captured from authorities, i.e., entities holding the right.",
      "metrics": null
    },
    {
      "id": "austr",
      "name": "AuSTR",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stance-detection"
      ],
      "links": {
        "github": "https://github.com/Fatima-Haouari/AuSTR",
        "paper": "https://arxiv.org/pdf/2301.05863v1.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "409 sentences",
      "year": 2023,
      "notes": "Stance dataset for authorities in Arabic tweets.",
      "metrics": null
    },
    {
      "id": "avanemo",
      "name": "AVANEmo",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "cc-by-nc-sa-4.0",
      "modality": "multimodal",
      "tasks": [
        "emotion"
      ],
      "links": {
        "website": "https://www.kaggle.com/suso172/arabic-natural-audio-dataset/home",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=8972836"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,000 videos",
      "year": 2019,
      "notes": "First audio-visual Arabic emotional dataset with 3000 clips covering six emotions from natural YouTube videos",
      "metrics": null
    },
    {
      "id": "aweasome-yemeni-open-source",
      "name": "aweasome-yemeni-open-source",
      "type": "tool",
      "country": "INTL",
      "org": "yementechcollective",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/yementechcollective/aweasome-yemeni-open-source",
        "website": "https://yementc.org"
      },
      "dialects": [
        "yemeni"
      ],
      "year": 2026,
      "notes": "A curated directory of Hundreds of open-source projects by Yemeni developers —PHP Laravel, Flutter, Python, AI, payments, and Arabic/RTL tooling.",
      "metrics": null
    },
    {
      "id": "awesome-arabic",
      "name": "awesome-arabic",
      "type": "tool",
      "country": "INTL",
      "org": "01walid",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/01walid/awesome-arabic"
      },
      "year": 2026,
      "notes": "A curated list of awesome projects and dev/design resources for supporting Arabic computational needs.",
      "metrics": null
    },
    {
      "id": "awesome-arabic-claude-skills",
      "name": "awesome-arabic-claude-skills",
      "type": "agent-skill",
      "country": "INTL",
      "org": "EngDawood",
      "license": "mit",
      "modality": "none",
      "tasks": [
        "skill",
        "curated-list"
      ],
      "links": {
        "github": "https://github.com/EngDawood/awesome-arabic-claude-skills"
      },
      "notes": "Curated library of open-source Arabic skills for Claude Code and agents",
      "metrics": null
    },
    {
      "id": "awesome-arabic-nlp",
      "name": "awesome-arabic-nlp",
      "type": "tool",
      "country": "INTL",
      "org": "Curated-Awesome-Lists",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/Curated-Awesome-Lists/awesome-arabic-nlp"
      },
      "year": 2023,
      "notes": "Dive into the world of Arabic NLP with this extensive collection of resources, tools, datasets, and best practices tailored for the Arabic language.",
      "metrics": null
    },
    {
      "id": "awesome-arabic-speakers",
      "name": "awesome-arabic-speakers",
      "type": "tool",
      "country": "INTL",
      "org": "sahaba-ai",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/sahaba-ai/awesome-arabic-speakers",
        "website": "https://awesome-arabic-speakers.dev"
      },
      "year": 2026,
      "notes": "A curated and collaborative list of awesome Arabic speaker's contributions in tech regardless of their ethnicity, nationality, or location.",
      "metrics": null
    },
    {
      "id": "awesome-darija-arabic-nlp-resources",
      "name": "Awesome-Darija-Arabic-NLP-Resources",
      "type": "tool",
      "country": "MA",
      "org": "UM6P-EMINES",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "catalogue",
        "darija"
      ],
      "links": {
        "github": "https://github.com/UM6P-EMINES/Awesome-Darija-Arabic-NLP-Resources"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Curated awesome-list of NLP resources for Moroccan Darija, covering text and speech datasets.",
      "metrics": null
    },
    {
      "id": "awn-v3",
      "name": "AWN V3",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Trento",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sts",
        "word-sense-disambiguation"
      ],
      "links": {
        "github": "https://github.com/HadiPTUK/AWN3.0",
        "paper": "https://aclanthology.org/2024.osact-1.9.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "9,576 tokens",
      "year": 2024,
      "notes": "9576 synsets that was enhanced from AWN V1.",
      "metrics": null
    },
    {
      "id": "awzan",
      "name": "Awzan",
      "type": "benchmark",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Awzan"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Awzān is a detailed dataset capturing all metrical patterns (taf'ilat) of classical Arabic poetic meters, addressing gaps in existing research.",
      "metrics": null
    },
    {
      "id": "aya-expanse",
      "name": "Aya-Expanse",
      "type": "llm",
      "country": "INTL",
      "org": "Cohere",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/CohereForAI/c4ai-aya-expanse-66f573116fef65271be752e9"
      },
      "size": "8B-32B",
      "notes": "State-of-the-art multilingual with strong Arabic",
      "metrics": null
    },
    {
      "id": "ayah",
      "name": "ayah",
      "type": "tool",
      "country": "INTL",
      "org": "nawafalqari",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/nawafalqari/ayah"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "API مفتوح المصدر للأذكار والقرآن والأحاديث",
      "metrics": null
    },
    {
      "id": "ayaspell",
      "name": "AyaSpell",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/linuxscout/ayaspell"
      },
      "year": 2016,
      "notes": "Arabic dictionary for the Hunspell spell checker; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "azan-mcp",
      "name": "Azan MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Ahmed Eltaher",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/ahmedeltaher/azan-mcp"
      },
      "year": 2026,
      "notes": "Lightweight MCP library for Islamic prayer times and Qibla for AI agents; not Arabic-language specific.",
      "metrics": null
    },
    {
      "id": "baca-quran-id",
      "name": "baca-quran.id",
      "type": "tool",
      "country": "INTL",
      "org": "mazipan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/mazipan/baca-quran.id",
        "website": "https://www.baca-quran.id/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "📖 Read Qur'an from Your Web Browser.",
      "metrics": null
    },
    {
      "id": "bahraincorpus",
      "name": "BahrainCorpus",
      "type": "dataset",
      "country": "BH",
      "org": "University of Bahrain",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "morphology",
        "asr"
      ],
      "links": {
        "website": "http://www.bahraincorpus.com",
        "paper": "https://aclanthology.org/2022.lrec-1.251.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "620,301 tokens",
      "year": 2022,
      "notes": "A 620K-word multi-genre corpus of Bahraini Arabic with written texts and spoken transcripts, automatically morphologically annotated.",
      "metrics": null
    },
    {
      "id": "baladi",
      "name": "Baladi",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2205.09692"
      },
      "year": 2022,
      "dialects": [
        "lev"
      ],
      "notes": "Lebanese Arabic corpus (~9.6K tokens) annotated as an extension of Curras (Lebanon/Palestine).",
      "metrics": null
    },
    {
      "id": "baladi-lebanese-dialect-corpus",
      "name": "Baladi Lebanese Dialect Corpus",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/resources/"
      },
      "dialects": [
        "lev"
      ],
      "year": 2022,
      "notes": "Annotated Lebanese dialect corpus with morphological annotations.",
      "metrics": null
    },
    {
      "id": "balsam",
      "name": "BALSAM",
      "type": "benchmark",
      "country": "SA",
      "org": "King Salman Global Academy for Arabic Language",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "website": "https://benchmarks.ksaa.gov.sa/"
      },
      "notes": "Benchmark of Arabic Language AI Systems and Models",
      "metrics": null
    },
    {
      "id": "bangla-quran",
      "name": "bangla-quran",
      "type": "tool",
      "country": "INTL",
      "org": "imranpollob",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/imranpollob/bangla-quran",
        "website": "https://banglaquran.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Read the Quran with Arabic text, Bangla translation, tafsir, audio recitation, bookmarks, and search.",
      "metrics": null
    },
    {
      "id": "baseer",
      "name": "Baseer",
      "type": "ocr",
      "country": "SA",
      "org": "Misraj",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/Misraj/Baseer-Qwen2.5-VL-3B-Instruct"
      },
      "notes": "Arabic document-to-markdown, Qwen2.5-VL-3B based (Misraj AI), WER 0.25",
      "metrics": null
    },
    {
      "id": "bayan",
      "name": "Bayan",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Bayan"
      },
      "year": 2024,
      "notes": "Bayan is a fully annotated treebank designed for the syntactic analysis of Arabic poetry.",
      "metrics": null
    },
    {
      "id": "bayanat",
      "name": "Bayanat",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "dataset-exploration"
      ],
      "links": {
        "github": "https://github.com/ARBML/bayanat"
      },
      "year": 2020,
      "notes": "Explore the contents of Arabic text datasets.",
      "metrics": null
    },
    {
      "id": "be-arabic-9k",
      "name": "BE-Arabic-9K",
      "type": "benchmark",
      "country": "EG",
      "org": "Electronics Research Institute Cairo",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "ocr"
      ],
      "links": {
        "github": "https://github.com/wdqin/BE-Arabic-9K",
        "paper": "https://doi.org/10.1007/s10032-021-00382-4"
      },
      "dialects": [
        "mixed"
      ],
      "size": "9,000 images",
      "year": 2021,
      "notes": "A large-scale benchmark dataset of over 9000 high-quality scanned images from Arabic books for document layout analysis and text extraction.",
      "metrics": null
    },
    {
      "id": "bedaya",
      "name": "Bedaya",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Haifa",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/muhammad-majadly/Bedaya-dataset",
        "paper": "https://aclanthology.org/2021.wanlp-1.12.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "20,500 tokens",
      "year": 2021,
      "notes": "Historical Arabic NER dataset based on Ibn Kathir book.",
      "metrics": null
    },
    {
      "id": "behdadfont",
      "name": "BehdadFont",
      "type": "tool",
      "country": "INTL",
      "org": "font-store",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/font-store/BehdadFont",
        "website": "http://libre.font-store.ir/BehdadFont/"
      },
      "year": 2021,
      "notes": "Farbod: Persian/Arabic Open Source Font - بهداد: فونت فارسی با مجوز آزاد",
      "metrics": null
    },
    {
      "id": "bel-masry",
      "name": "Bel Masry",
      "type": "tool",
      "country": "EG",
      "org": "Applied Innovation Center (MCIT)",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "ocr",
        "asr",
        "translation"
      ],
      "links": {
        "website": "https://egyptinnovate.com/en/news/bel-masry-launches-egypts-first-free-ai-platform-for-arabic-reading-transcription-and-translation"
      },
      "dialects": [
        "egy",
        "msa"
      ],
      "year": 2026,
      "notes": "Free government AI platform for reading, transcribing and translating Egyptian colloquial and MSA, launched Aug 2026.",
      "metrics": null
    },
    {
      "id": "belebele-eqa",
      "name": "BELEBELE-EQA",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/mbzuai-nlp/MultiChoice2ExtractiveQA",
        "paper": "https://arxiv.org/pdf/2404.17342v2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "415 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "Parallel extractive QA dataset for English and Arabic derived from BELEBELE.",
      "metrics": null
    },
    {
      "id": "biasfignews",
      "name": "BiasFigNews",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "propaganda"
      ],
      "links": {
        "github": "https://github.com/SinaLab/BiasFignews"
      },
      "year": 2024,
      "notes": "Bias and propaganda detection corpus of social media posts (FIGNEWS shared task).",
      "metrics": null
    },
    {
      "id": "bodmaghdataset",
      "name": "BoDmaghDataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ImadSaddik",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "github": "https://github.com/ImadSaddik/BoDmaghDataset"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "BoDmagh dataset is a Supervised Fine-Tuning (SFT) dataset for the Darija language",
      "metrics": null
    },
    {
      "id": "bohor",
      "name": "Bohor",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Bohor"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Bohor is an advanced system for extracting and analyzing metrical patterns (taf'ilat) in classical Arabic poetry.",
      "metrics": null
    },
    {
      "id": "bolt-egyptian-arabic-treebank-sms-chat",
      "name": "BOLT Egyptian Arabic Treebank - SMS/Chat",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "pos",
        "morphology"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2021T17"
      },
      "dialects": [
        "egy"
      ],
      "size": "349,414 tokens",
      "year": 2021,
      "tags": [
        "variants:2"
      ],
      "notes": "BOLT Egyptian Arabic Treebank of SMS and chat text with syntactic annotations for parsing and dialectal NLP. (also: 2 other LDC releases)",
      "metrics": null
    },
    {
      "id": "bootstrap-3-arabic",
      "name": "bootstrap-3-arabic",
      "type": "tool",
      "country": "INTL",
      "org": "zeroxme",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/zeroxme/bootstrap-3-arabic"
      },
      "year": 2015,
      "notes": "bootstrap 3 arabic",
      "metrics": null
    },
    {
      "id": "bootstrap-v4-rtl",
      "name": "bootstrap-v4-rtl",
      "type": "tool",
      "country": "INTL",
      "org": "MahdiMajidzadeh",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MahdiMajidzadeh/bootstrap-v4-rtl",
        "website": "http://mahdimajidzadeh.github.io/bootstrap-v4-rtl/"
      },
      "year": 2026,
      "notes": "RTL edition of bootstrap v4 for rtl languages like Farsi and Arabic",
      "metrics": null
    },
    {
      "id": "buckwalter-python",
      "name": "Buckwalter transliteration script",
      "type": "tool",
      "country": "INTL",
      "org": "Kenton Murray",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/KentonMurray/Buckwalter"
      },
      "year": 2014,
      "notes": "Small Python script for Buckwalter transliteration of Arabic.",
      "metrics": null
    },
    {
      "id": "burhan",
      "name": "Burhan",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Burhan"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "📖 Burhan: The QR-Rhetoric Computational Semantic Dataset for Classical Arabic",
      "metrics": null
    },
    {
      "id": "cafe-code-switching-speech-dataset",
      "name": "CAFE code-switching speech dataset",
      "type": "dataset",
      "country": "DZ",
      "org": "HBKU and Algerian universities",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "website": "https://elmi.hbku.edu.qa/en/publications/cafe-spontaneous-code-switching-speech-dataset-in-algerian-dialec",
        "paper": "https://pubmed.ncbi.nlm.nih.gov/41311731"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Spontaneous code-switching speech in Algerian dialect mixed with French and English.",
      "metrics": null
    },
    {
      "id": "cai-corpus",
      "name": "CAI-Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Leeds",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/alaabouomar/Optimizing-Arabic-Dialect-Translation-for-Children-s-Literature-Using-Neural-Models.git",
        "paper": "https://aclanthology.org/2025.wacl-1.11.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "130 documents",
      "year": 2025,
      "notes": "Parallel corpus of 130 children’s stories translated from Modern Standard Arabic to Egyptian (Cairo) dialect.",
      "metrics": null
    },
    {
      "id": "calfa",
      "name": "Calfa",
      "type": "org",
      "country": "INTL",
      "org": "Calfa",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://calfa.fr"
      },
      "notes": "OCR service extracting printed and handwritten text from scanned documents, including Arabic script.",
      "metrics": null
    },
    {
      "id": "calima-glf",
      "name": "CALIMA-GLF",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "cc-by-sa-3.0",
      "modality": "text",
      "tasks": [
        "pos",
        "morphology"
      ],
      "links": {
        "github": "https://github.com/unimorph/afb",
        "paper": "https://aclanthology.org/W17-1305.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "2,648 tokens",
      "year": 2017,
      "notes": "The dataset is part of the CALIMAGLF morphological analyzer, focusing on Gulf Arabic verbs.",
      "metrics": null
    },
    {
      "id": "callhome-egyptian-arabic-speech",
      "name": "CALLHOME Egyptian Arabic Speech",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC97S45"
      },
      "dialects": [
        "egy"
      ],
      "size": "120 sentences",
      "year": 2002,
      "tags": [
        "variants:1"
      ],
      "notes": "All calls, which lasted up to 30 minutes, originated in North America and were placed to locations overseas (typically Egypt). (also: 1 other LDC releases)",
      "metrics": null
    },
    {
      "id": "callhome-egyptian-arabic-transcripts",
      "name": "CALLHOME Egyptian Arabic Transcripts",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC97T19"
      },
      "dialects": [
        "egy"
      ],
      "size": "120 sentences",
      "year": 2002,
      "tags": [
        "variants:1"
      ],
      "notes": "The transcripts are timestamped by speaker turn for alignment with the speech signal and are provided in standard orthography. (also: 1 other LDC releases)",
      "metrics": null
    },
    {
      "id": "calliar",
      "name": "Calliar",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "github": "https://github.com/ARBML/Calliar"
      },
      "notes": "Online Arabic calligraphy (2500 samples)",
      "metrics": null
    },
    {
      "id": "camel-lab-nyu-abu-dhabi",
      "name": "CAMeL Lab (NYU Abu Dhabi)",
      "type": "org",
      "country": "AE",
      "org": "CAMeL Lab (NYU Abu Dhabi)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://camel-lab.com/"
      },
      "notes": "CAMeLBERT, camel_tools, morphological analysis",
      "metrics": null
    },
    {
      "id": "camel-lab-nyuad",
      "name": "CAMeL Lab, NYUAD",
      "type": "org",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab"
      },
      "notes": "Arabic NLP tools and models - CAMeLBERT, camel_tools",
      "metrics": null
    },
    {
      "id": "camel-treebank",
      "name": "CAMeL Treebank",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "treebank"
      ],
      "links": {
        "website": "http://treebank.camel-lab.com/"
      },
      "dialects": [
        "msa",
        "classical"
      ],
      "year": 2022,
      "notes": "Open-source dependency treebank of Modern Standard and Classical Arabic.",
      "metrics": null
    },
    {
      "id": "camel-tools",
      "name": "camel_tools",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/camel_tools"
      },
      "notes": "Suite of Arabic NLP tools (morphology, POS, NER, etc.)",
      "metrics": null
    },
    {
      "id": "camelbert",
      "name": "CAMeLBERT",
      "type": "llm",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/CAMeLBERT"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "MSA, Dialectal, and Classical Arabic",
      "metrics": null
    },
    {
      "id": "camelira",
      "name": "Camelira",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://camelira.abudhabi.nyu.edu/"
      },
      "year": 2022,
      "notes": "An Arabic multi-dialect morphological disambiguator (web interface).",
      "metrics": null
    },
    {
      "id": "camel-morph",
      "name": "CamelMorph",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/camel_morph"
      },
      "year": 2022,
      "notes": "Builds large open-source morphological models for Arabic and its dialects.",
      "metrics": null
    },
    {
      "id": "camel-parser",
      "name": "CamelParser2.0",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "parsing"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/camel_parser"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Arabic dependency parser built on CAMeL Tools morphology for MSA and dialects.",
      "metrics": null
    },
    {
      "id": "canercorpus",
      "name": "CANERCorpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Universiti Kebangsaan Malaysia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/RamziSalah/Classical-Arabic-Named-Entity-Recognition-Corpus",
        "hf": "https://huggingface.co/datasets/caner",
        "paper": "https://ieeexplore.ieee.org/document/8464820/authors#authors"
      },
      "dialects": [
        "msa"
      ],
      "size": "72,108 tokens",
      "year": 2018,
      "notes": "It is freely available and manual annotation by human experts, containing more than 7,000 Hadiths",
      "metrics": null
    },
    {
      "id": "caraner",
      "name": "CAraNER",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.1/",
        "github": "https://github.com/kacst-ncdaai/caraner"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "First, we introduce Wassem, a web-based annotation platform for Arabic NLP applications.",
      "metrics": null
    },
    {
      "id": "casablanca",
      "name": "Casablanca",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2410.04527"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Multidialectal Arabic speech dataset (NADI 2025)",
      "metrics": null
    },
    {
      "id": "cass",
      "name": "CASS",
      "type": "dataset",
      "country": "LB",
      "org": "American University of Beirut",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sts"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.21726079"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,048 sentences",
      "year": 2026,
      "notes": "A comprehensive Arabic semantic similarity dataset.",
      "metrics": null
    },
    {
      "id": "catt",
      "name": "CATT",
      "type": "tool",
      "country": "INTL",
      "org": "Abjad AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.03236"
      },
      "notes": "Character-based Tashkeel Transformer, SOTA results",
      "metrics": null
    },
    {
      "id": "catt-official",
      "name": "CATT (official)",
      "type": "tool",
      "country": "INTL",
      "org": "Abjad AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/abjadai/catt"
      },
      "year": 2024,
      "notes": "Official implementation of the CATT Arabic diacritization models.",
      "metrics": null
    },
    {
      "id": "cf-arb",
      "name": "CF-Arb",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Birmingham",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/ShathaHakami/Context-Free-Arabic-Emoji-Sentiment-Lexicon",
        "paper": "https://aclanthology.org/2022.osact-1.6.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,069 tokens",
      "year": 2022,
      "notes": "Context-free Arabic emoji sentiment lexicon.",
      "metrics": null
    },
    {
      "id": "checkthat-ar",
      "name": "CheckThat-AR",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "website": "https://gitlab.com/bigirqu/checkthat-ar/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "7,500 sentences",
      "year": 2020,
      "notes": "We make freely accessible ANETAC1 our English-Arabic named entity transliteration and classification dataset that we built from freely available parallel.",
      "metrics": null
    },
    {
      "id": "child-language-corpus-of-jordanian-arabic",
      "name": "Child Language Corpus of Jordanian Arabic",
      "type": "dataset",
      "country": "JO",
      "org": "University of Jordan",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "website": "https://sites.ju.edu.jo/en/childcorpus/home.aspx"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "notes": "Child-language corpus of Jordanian Arabic of about 500K words.",
      "metrics": null
    },
    {
      "id": "childes-egyptian-arabic-salama-corpus",
      "name": "CHILDES Egyptian Arabic Salama Corpus",
      "type": "dataset",
      "country": "EG",
      "org": "Alexandria University",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://childes.talkbank.org/access/Other/Arabic/Salama.html",
        "paper": "https://www.academia.edu/44521353/Building_a_spoken_Arabic_corpus_for_Egyptian_children_Data_collection_and_transcription"
      },
      "dialects": [
        "egy"
      ],
      "size": "40,513 sentences",
      "year": 2015,
      "notes": "Participants The Egyptian Arabic corpus includes data from ten children.",
      "metrics": null
    },
    {
      "id": "claude-adhkar",
      "name": "Claude Adhkar",
      "type": "agent-skill",
      "country": "INTL",
      "org": "nosseralaa7-rgb",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/nosseralaa7-rgb/claude-adhkar"
      },
      "year": 2026,
      "notes": "Shows Arabic-script adhkar in the Claude Code spinner with a terminal font that renders Arabic.",
      "metrics": null
    },
    {
      "id": "claude-arabic-writing",
      "name": "Claude Arabic Writing",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Ahmed Dabak",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/ahmeddabak/claude-arabic-writing"
      },
      "year": 2026,
      "notes": "Claude Agent Skill for natural, grammatically correct Arabic writing and translation.",
      "metrics": null
    },
    {
      "id": "claude-code-rtl-extension",
      "name": "Claude Code RTL Extension",
      "type": "agent-skill",
      "country": "INTL",
      "org": "yechielby",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/yechielby/claude-code-rtl-extension"
      },
      "year": 2026,
      "notes": "VS Code and Cursor extension adding RTL support for Hebrew and Arabic in Claude Code.",
      "metrics": null
    },
    {
      "id": "claude-desktop-rtl-patch-mac",
      "name": "Claude Desktop RTL Patch (macOS)",
      "type": "agent-skill",
      "country": "INTL",
      "org": "toboly",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/toboly/claude-desktop-rtl-patch-mac"
      },
      "year": 2026,
      "notes": "Adds auto-detected Hebrew and Arabic RTL support to Claude Desktop on macOS.",
      "metrics": null
    },
    {
      "id": "claude-desktop-rtl-mac",
      "name": "claude-desktop-rtl-mac",
      "type": "tool",
      "country": "INTL",
      "org": "soguy",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/soguy/claude-desktop-rtl-mac"
      },
      "year": 2026,
      "notes": "Automatic RTL (Hebrew/Arabic) text support for Claude Desktop on macOS",
      "metrics": null
    },
    {
      "id": "claude-desktop-rtl-patch",
      "name": "claude-desktop-rtl-patch",
      "type": "tool",
      "country": "INTL",
      "org": "shraga100",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/shraga100/claude-desktop-rtl-patch",
        "website": "https://github.com/shraga100/claude-desktop-rtl-patch"
      },
      "year": 2026,
      "notes": "CSS patch for Claude Desktop windows version to enable RTL (right-to-left) support for Hebrew and Arabic",
      "metrics": null
    },
    {
      "id": "climategpt",
      "name": "ClimateGPT",
      "type": "tool",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/mbzuai-oryx/ClimateGPT"
      },
      "year": 2024,
      "notes": "[EMNLP'23] ClimateGPT: a specialized LLM for conversations related to Climate Change and Sustainability topics in both English and Arabic languages.",
      "metrics": null
    },
    {
      "id": "clusterlab-ai",
      "name": "Clusterlab AI",
      "type": "org",
      "country": "INTL",
      "org": "ClusterlabAi",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "hf": "https://huggingface.co/ClusterlabAi"
      },
      "notes": "Behind 101 Billion Arabic Words Dataset and InstAr-500k",
      "metrics": null
    },
    {
      "id": "code-switched-tunisian-speech-recognition",
      "name": "Code Switched Tunisian Speech Recognition",
      "type": "asr",
      "country": "INTL",
      "org": "SalahZa",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "hf": "https://huggingface.co/SalahZa/Code_Switched_Tunisian_Speech_Recognition"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "End-to-end ASR for code-switched Tunisian Arabic with English and French; 29.47% WER on TunSwitch CS.",
      "metrics": {
        "downloads": 0,
        "likes": 5,
        "lastModified": "2023-09-25"
      }
    },
    {
      "id": "cohere",
      "name": "Cohere",
      "type": "org",
      "country": "INTL",
      "org": "Cohere",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://cohere.com/"
      },
      "notes": "Multilingual LLMs - Command R Arabic, RAG optimization",
      "metrics": null
    },
    {
      "id": "cohere-transcribe-arabic",
      "name": "Cohere Transcribe batch inference",
      "type": "asr",
      "country": "INTL",
      "org": "AliOsm",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/AliOsm/cohere-transcribe"
      },
      "year": 2026,
      "notes": "Production batch inference wrapper for Cohere's Arabic and English ASR model.",
      "metrics": null
    },
    {
      "id": "colorful-quran",
      "name": "colorful-quran",
      "type": "tool",
      "country": "INTL",
      "org": "kodepandai",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kodepandai/colorful-quran",
        "website": "https://colorful-quran.pages.dev/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Holy Qur'an with colorful tajweed anotation",
      "metrics": null
    },
    {
      "id": "conflictfactar-a-dataset-for-arabic-multi-conflict-fake-and-real-news",
      "name": "ConflictFactAR: A Dataset For Arabic Multi-Conflict Fake and Real News for Misinformation Research",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/ConflictFactAR_A_Dataset_For_Arabic_Multi-Conflict_Fake_and_Real_News_for_Misinformation_Research/33295215"
      },
      "notes": "ConflictFactAR: Arabic fake and real news dataset on regional conflicts.",
      "metrics": null
    },
    {
      "id": "convertedin",
      "name": "Convertedin",
      "type": "org",
      "country": "EG",
      "org": "Convertedin",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.converted.in/"
      },
      "notes": "AI marketing automation - Arabic/English e-commerce personalization, $3M funded",
      "metrics": null
    },
    {
      "id": "core42",
      "name": "Core42",
      "type": "org",
      "country": "AE",
      "org": "Core42",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.core42.ai"
      },
      "notes": "G42 sovereign cloud and AI company; Compass AI platform and Jais deployment.",
      "metrics": null
    },
    {
      "id": "corenlp-arabic",
      "name": "corenlp arabic",
      "type": "tool",
      "country": "INTL",
      "org": "Stanford NLP",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "nlp-pipeline"
      ],
      "links": {
        "hf": "https://huggingface.co/stanfordnlp/corenlp-arabic"
      },
      "year": 2026,
      "notes": "Arabic models for Stanford CoreNLP, a Java NLP toolkit (tagging, parsing, NER).",
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2026-09-27"
      }
    },
    {
      "id": "cross-dialectal-named-entity-recognition-in-arabic",
      "name": "Cross-Dialectal Named Entity Recognition in Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "dialect"
      ],
      "links": {
        "paper": "https://aclanthology.org/2023.arabicnlp-1.12/",
        "github": "https://github.com/niamaelkhbir/Arabic-Cross-Dialectal-NER"
      },
      "year": 2023,
      "venue": "ArabicNLP 2023",
      "notes": "Study of NER transferability between Arabic dialects, releasing four manually annotated dialectal NER datasets.",
      "metrics": null
    },
    {
      "id": "crowd-analyzer",
      "name": "Crowd Analyzer",
      "type": "org",
      "country": "EG",
      "org": "Crowd Analyzer",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://crowdanalyzer.com/"
      },
      "notes": "Arabic social media monitoring - Arabic NLP analytics, sentiment analysis, media monitoring",
      "metrics": null
    },
    {
      "id": "curras",
      "name": "Curras",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/curras/"
      },
      "year": 2014,
      "dialects": [
        "lev"
      ],
      "notes": "Palestinian Arabic morphologically annotated corpus (Palestine), basis of the Levantine corpus.",
      "metrics": null
    },
    {
      "id": "daict",
      "name": "DAICT",
      "type": "dataset",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "sarcasm"
      ],
      "links": {
        "paper": "http://www.lrec-conf.org/proceedings/lrec2020/pdf/2020.lrec-1.768.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "5,588 sentences",
      "year": 2020,
      "notes": "The dataset includes 5,588 tweets -- written in both MSA and dialectual Arabic -- manually annotated by two professional linguistics from HBKU",
      "metrics": null
    },
    {
      "id": "dar-al-raqmana",
      "name": "Dar Al-Raqmana",
      "type": "org",
      "country": "SA",
      "org": "Dar Al-Raqmana",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://daralraqmana.com"
      },
      "notes": "Digitization company offering Arabic text recognition (OCR), digital repositories and morphological search.",
      "metrics": null
    },
    {
      "id": "dares",
      "name": "DARES",
      "type": "dataset",
      "country": "INTL",
      "org": "Lancaster University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "github": "https://github.com/ArabicNLP-UK/DARES-Readability-Dataset"
      },
      "year": 2024,
      "notes": "Dataset for Arabic readability estimation of Saudi school materials.",
      "metrics": null
    },
    {
      "id": "darija-chatbot-arena",
      "name": "Darija Chatbot Arena",
      "type": "benchmark",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/atlasia/darija-chatbot-arena"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Arena platform comparing LLM responses to Darija prompts with Elo ratings.",
      "metrics": null
    },
    {
      "id": "darija-gpt",
      "name": "Darija-GPT",
      "type": "tool",
      "country": "INTL",
      "org": "Kirouane-Ayoub",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "text-generation",
        "darija"
      ],
      "links": {
        "github": "https://github.com/Kirouane-Ayoub/Darija-GPT"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Trains a GPT-2 model from scratch to generate Moroccan Darija text.",
      "metrics": null
    },
    {
      "id": "darija-sentiment-analysis-tatimohammed",
      "name": "Darija-Sentiment-Analysis",
      "type": "tool",
      "country": "INTL",
      "org": "tatimohammed",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "darija"
      ],
      "links": {
        "github": "https://github.com/tatimohammed/Darija-Sentiment-Analysis"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "Notebook and web demo fine-tuning a model for Moroccan Darija sentiment analysis of customer feedback.",
      "metrics": null
    },
    {
      "id": "darija-translator-dhiadev-tn",
      "name": "darija-translator",
      "type": "tool",
      "country": "INTL",
      "org": "Dhiadev-tn",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/Dhiadev-tn/darija-translator"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "A from scratch open-source Tunisian Darija-to-English NLP pipeline.",
      "metrics": null
    },
    {
      "id": "darijat-tts",
      "name": "Darijat TTS",
      "type": "org",
      "country": "INTL",
      "org": "Darijat",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://tts.darijat.com/"
      },
      "notes": "Arabic dialect TTS service claiming 23 dialects.",
      "metrics": null
    },
    {
      "id": "darijatokenizers",
      "name": "DarijaTokenizers",
      "type": "tool",
      "country": "INTL",
      "org": "ImadSaddik",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ImadSaddik/DarijaTokenizers"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Free to use tokenizers trained on the Darija language.",
      "metrics": null
    },
    {
      "id": "darijavoice-dysarthria-a-moroccan-arabic-dysarthric-speech-corpus",
      "name": "DarijaVoice-Dysarthria: A Moroccan Arabic Dysarthric Speech Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "dysarthria"
      ],
      "links": {
        "website": "https://zenodo.org/records/18991743"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "First dysarthric speech corpus for Moroccan Arabic (Darija), with 1,071 scripted utterances.",
      "metrics": null
    },
    {
      "id": "dataset-arabic-speech-mispronunciation-detection",
      "name": "Dataset_Arabic_speech_mispronunciation_detection",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc0-1.0",
      "modality": "speech",
      "tasks": [
        "mispronunciation"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/x54dg53rmr"
      },
      "year": 2021,
      "notes": "Egyptian Arabic speech mispronunciation dataset of 100 frequent words pronounced by 100 children aged 2-8.",
      "dialects": [
        "egy"
      ],
      "metrics": null
    },
    {
      "id": "deast",
      "name": "DEAST",
      "type": "dataset",
      "country": "EG",
      "org": "Benha University",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://zenodo.org/records/17754563",
        "paper": "https://www.sciencedirect.com/science/article/pii/S2352340925010947"
      },
      "dialects": [
        "msa"
      ],
      "size": "33,000 sentences",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Parallel English-Arabic corpus of 33k thesis titles in 13 scientific domains for machine translation.",
      "metrics": null
    },
    {
      "id": "deep-diacritization",
      "name": "Deep Diacritization",
      "type": "tool",
      "country": "INTL",
      "org": "BKHMSI",
      "license": "agpl-3.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/BKHMSI/deep-diacritization"
      },
      "year": 2020,
      "notes": "Official repository of the Deep Diacritization paper.",
      "metrics": null
    },
    {
      "id": "defarabicqa",
      "name": "DefArabicQA",
      "type": "dataset",
      "country": "INTL",
      "org": "ANLP Research Group- MIRACL Laboratory",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/triguiomar/ADQA",
        "paper": "http://personales.upv.es/prosso/resources/TriguiEtAl_LREC10.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,000 sentences",
      "year": 2010,
      "notes": "2000 snippets returned by Google search engine and Wikipedia Arabic version and a set of 50 organization definition questions",
      "metrics": null
    },
    {
      "id": "dengjen-tashkeel",
      "name": "dengjen-tashkeel",
      "type": "tool",
      "country": "INTL",
      "org": "ZirekHQ",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/ZirekHQ/dengjen-tashkeel",
        "website": "https://mush42.github.io/libtashkeel/"
      },
      "year": 2026,
      "notes": "Add Arabic diacritics (tashkeel/harakat) using Rust/Python/C++/WASM and NLP models",
      "metrics": null
    },
    {
      "id": "derja-ninja-scraper",
      "name": "derja_ninja_scraper",
      "type": "tool",
      "country": "INTL",
      "org": "ArmelVidali",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech"
      ],
      "links": {
        "github": "https://github.com/ArmelVidali/derja_ninja_scraper"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Get Tunisian translation, audio and sample sentence for the most common 20.000 english word",
      "metrics": null
    },
    {
      "id": "dial2msa-verified",
      "name": "Dial2MSA-Verified",
      "type": "dataset",
      "country": "INTL",
      "org": "The University of Manchester",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/khered20/Dial2MSA-Verified",
        "paper": "https://aclanthology.org/2025.wacl-1.6/"
      },
      "dialects": [
        "mixed",
        "egy",
        "gulf",
        "lev",
        "magh"
      ],
      "size": "31,887 sentences",
      "year": 2024,
      "notes": "Parallel tweets in 4 dialects with verified MSA translations for neural MT evaluation",
      "metrics": null
    },
    {
      "id": "dialect-classification-using-acoustic-and-linguistic-features-in-arabi",
      "name": "Dialect classification using acoustic and linguistic features in Arabic speech",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://zenodo.org/records/7455768"
      },
      "year": 2023,
      "notes": "Arabic speech dialect classification dataset using acoustic and linguistic features.",
      "metrics": null
    },
    {
      "id": "dialectal-ai",
      "name": "Dialectal AI",
      "type": "org",
      "country": "SD",
      "org": "Dialectal AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://dialectal.me"
      },
      "notes": "Speech AI platform specialised in Sudanese Arabic for customer-service automation.",
      "metrics": null
    },
    {
      "id": "dialectal-arabic-datasets",
      "name": "Dialectal Arabic Datasets",
      "type": "dataset",
      "country": "INTL",
      "org": "mahmoudreda55",
      "license": "other",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/mahmoudreda55/dialectal-arabic-datasets"
      },
      "dialects": [
        "egy",
        "lev",
        "gulf",
        "magh"
      ],
      "year": 2023,
      "notes": "Kaggle dataset for four dialect groups: Egyptian, Levantine, Gulf and Maghrebi.",
      "metrics": null
    },
    {
      "id": "dialectal-arabic-tools",
      "name": "Dialectal Arabic Tools",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "segmentation"
      ],
      "links": {
        "github": "https://github.com/qcri/dialectal_arabic_tools"
      },
      "dialects": [
        "egy",
        "lev",
        "gulf",
        "magh"
      ],
      "notes": "QCRI segmentation tools for dialectal Arabic (Egyptian, Levantine, Gulf, Maghrebi).",
      "metrics": null
    },
    {
      "id": "dialogue-arabic-dialects",
      "name": "dialogue-arabic-dialects",
      "type": "dataset",
      "country": "INTL",
      "org": "tareknaous",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/tareknaous/dialogue-arabic-dialects"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Levantine, Egyptian, Gulf dialect dialogues",
      "metrics": null
    },
    {
      "id": "diaset",
      "name": "DiaSet",
      "type": "dataset",
      "country": "INTL",
      "org": "Reichman University",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "ner"
      ],
      "links": {
        "website": "https://idc-dsi.github.io/DiaCorpus/",
        "paper": "https://aclanthology.org/2024.lrec-main.436.pdf"
      },
      "dialects": [
        "lev"
      ],
      "size": "644,800 tokens",
      "year": 2024,
      "notes": "Annotated Arabic conversational dataset for SA and NER in Palestinian dialect",
      "metrics": null
    },
    {
      "id": "digital-quran-docs",
      "name": "digital-quran-docs",
      "type": "tool",
      "country": "INTL",
      "org": "quranacademy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/quranacademy/digital-quran-docs"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Documentation for Quran Academy data distribution project: Digital Quran - https://quranacademy.gitbook.io/digital-quran/",
      "metrics": null
    },
    {
      "id": "dimi-arabic-ocr",
      "name": "DIMI-Arabic-OCR",
      "type": "ocr",
      "country": "INTL",
      "org": "AhmedZaky1",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/AhmedZaky1/DIMI-Arabic-OCR"
      },
      "notes": "Printed Arabic with diacritics OCR (Qwen2-VL based)",
      "base_model": [
        "unsloth/qwen2.5-vl-7b-instruct-bnb-4bit"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 4,
        "lastModified": "2025-10-08"
      }
    },
    {
      "id": "diwan",
      "name": "Diwan",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Diwan"
      },
      "year": 2026,
      "notes": "Diwan is the largest Arabic poetry dataset, containing nearly 500,000 poems and over 15 million verses.",
      "metrics": null
    },
    {
      "id": "doda",
      "name": "DODa Darija Open Dataset",
      "type": "dataset",
      "country": "MA",
      "org": "Darija Open Dataset",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/darija-open-dataset/dataset"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Open Moroccan Darija dictionary and parallel dataset, one of the first Darija resources.",
      "metrics": null
    },
    {
      "id": "doo",
      "name": "DOO",
      "type": "org",
      "country": "INTL",
      "org": "DOO",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://doo.ooo"
      },
      "notes": "AI customer-experience platform automating customer interactions across channels, listed as an Arabic AI customer service platform.",
      "metrics": null
    },
    {
      "id": "dorar-hadith-api",
      "name": "Dorar Hadith API",
      "type": "tool",
      "country": "EG",
      "org": "Ahmed El Tabarani",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "hadith",
        "api"
      ],
      "links": {
        "github": "https://github.com/AhmedElTabarani/dorar-hadith-api"
      },
      "year": 2022,
      "notes": "Intermediary API over dorar.net hadith search and grading.",
      "metrics": null
    },
    {
      "id": "dorar-hadith-mcp",
      "name": "Dorar Hadith MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "ibnsaleem29",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/ibnsaleem29/dorar-hadith-mcp"
      },
      "year": 2026,
      "notes": "Claude extension and MCP server for Hadith research including isnad and takhrij via Dorar.",
      "metrics": null
    },
    {
      "id": "doxci",
      "name": "Doxci",
      "type": "org",
      "country": "INTL",
      "org": "Doxci",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://doxci.ai"
      },
      "notes": "AI document processing platform for enterprise workflows, listed for Arabic document processing.",
      "metrics": null
    },
    {
      "id": "dxwand",
      "name": "DXwand",
      "type": "org",
      "country": "EG",
      "org": "DXwand",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://dxwand.com/"
      },
      "notes": "Generative AI for Arabic business - ORXTRA platform, Arabic dialect chatbots, 20+ LLM support",
      "metrics": null
    },
    {
      "id": "dzdc12",
      "name": "DZDC12",
      "type": "dataset",
      "country": "DZ",
      "org": "Université",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "pretraining",
        "dialect-id",
        "retrieval",
        "offensive-language",
        "gender-id"
      ],
      "links": {
        "github": "https://github.com/xprogramer/DZDC12",
        "paper": "https://link.springer.com/article/10.1007/s10579-019-09454-8"
      },
      "dialects": [
        "magh"
      ],
      "size": "2,400 sentences",
      "year": 2020,
      "notes": "DZDC12 is a multi-purpose parallel corpus crawled from facebook",
      "metrics": null
    },
    {
      "id": "dzner",
      "name": "DzNER",
      "type": "dataset",
      "country": "INTL",
      "org": "Leibniz-Institute for the Social Sciences Cologne",
      "license": "cc-by-nc-nd-4.0",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/Dahouabdelhalim/NER-model-on-the-DzNER-corpus",
        "paper": "https://www.sciencedirect.com/science/article/pii/S294971912300002X"
      },
      "dialects": [
        "magh"
      ],
      "size": "220,000 tokens",
      "year": 2023,
      "notes": "The DzNER dataset is designed for Named Entity Recognition (NER) in the Algerian dialect, a significantly low-resource language in NLP research.",
      "metrics": null
    },
    {
      "id": "dzsentia",
      "name": "DzSentiA",
      "type": "tool",
      "country": "INTL",
      "org": "adelabdelli",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/adelabdelli/DzSentiA"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Sentiment-analysis code and data covering Algerian dialect and MSA text.",
      "metrics": null
    },
    {
      "id": "easyocr",
      "name": "EasyOCR",
      "type": "tool",
      "country": "INTL",
      "org": "JaidedAI",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/JaidedAI/EasyOCR"
      },
      "notes": "Ready-to-use OCR with Arabic support (80+ languages)",
      "metrics": null
    },
    {
      "id": "edgad",
      "name": "EDGAD",
      "type": "dataset",
      "country": "EG",
      "org": "Cairo University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "gender-id"
      ],
      "links": {
        "github": "https://github.com/shery91/Egyptian-Dialect-Gender-Annotated-Dataset",
        "paper": "https://www.sciencedirect.com/science/article/pii/S1110866518302044"
      },
      "dialects": [
        "egy"
      ],
      "size": "140,000 sentences",
      "year": 2019,
      "notes": "Egyptian Dialect Gender Annotated Dataset (EDGAD) obtained from Twitter as well as a proposed text classification solution for the Gender Identification.",
      "metrics": null
    },
    {
      "id": "eg-drugs",
      "name": "eg-drugs",
      "type": "dataset",
      "country": "INTL",
      "org": "mahmoudfalous",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/mahmoudfalous/eg-drugs"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "26K+ Egyptian drugs dataset with active ingredients, FDA mapping, Arabic drug information, prices, barcodes, and medical warning flags.",
      "metrics": null
    },
    {
      "id": "egy-names",
      "name": "egy-names",
      "type": "tool",
      "country": "INTL",
      "org": "AbdullahAfifyKhalil",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/AbdullahAfifyKhalil/egy-names",
        "website": "https://afify.co/egy-names"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Production-grade Egyptian onomastic intelligence: generate, translate, split, vocalize, and analyze names offline.",
      "metrics": null
    },
    {
      "id": "egypt-geo-data",
      "name": "egypt-geo-data",
      "type": "dataset",
      "country": "INTL",
      "org": "useswype",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/useswype/egypt-geo-data",
        "website": "https://www.swypex.com"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "JSON dataset of all Egyptian cities and governorates with English and Arabic names",
      "metrics": null
    },
    {
      "id": "egypt-nlp",
      "name": "egypt-nlp",
      "type": "tool",
      "country": "INTL",
      "org": "Amr Eleraqi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aeleraqi/egypt-nlp"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Python toolkit for processing Egyptian Arabic dialect text.",
      "metrics": null
    },
    {
      "id": "egyptian-asr-diarization",
      "name": "Egyptian Arabic ASR and Diarization",
      "type": "asr",
      "country": "EG",
      "org": "Speech Squad",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "diarization"
      ],
      "links": {
        "github": "https://github.com/yousefkotp/Egyptian-Arabic-ASR-and-Diarization"
      },
      "year": 2024,
      "notes": "MTC-AIC 2 competition submission for Egyptian Arabic ASR and speaker diarization.",
      "metrics": null
    },
    {
      "id": "egyptian-arabic-conversational-speech-corpus-asr-egarbcsc",
      "name": "Egyptian Arabic Conversational Speech Corpus (ASR-EgArbCSC)",
      "type": "dataset",
      "country": "INTL",
      "org": "MagicHub (Magic Data)",
      "license": "MagicHub license",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://magichub.com/datasets/egyptian-arabic-conversational-speech-corpus/"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic conversational speech corpus of spontaneous themed conversations at 16 kHz; free with registration.",
      "metrics": null
    },
    {
      "id": "egyptian-arabic-tts-chatterbox",
      "name": "egyptian arabic tts chatterbox",
      "type": "tts",
      "country": "INTL",
      "org": "AliAbdallah",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "chat",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/AliAbdallah/egyptian-arabic-tts-chatterbox"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "First open-source Egyptian Arabic TTS model based on Chatterbox Multilingual TTS.",
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2026-02-05"
      }
    },
    {
      "id": "egyptian-arabic-english-parallel-corpus",
      "name": "Egyptian Arabic-English Parallel Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "mohamedabdalkader",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/mohamedabdalkader/egyptian-arabic-english-parallel-corpus"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "420,000 bilingual samples · 1,800 topics · 1M+ sentences · Egyptian Arabic (arz)",
      "metrics": null
    },
    {
      "id": "egyptian-drug-database",
      "name": "egyptian-drug-database",
      "type": "dataset",
      "country": "INTL",
      "org": "karem505",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/karem505/egyptian-drug-database"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Egyptian drug database (May 2026) — 24,868 medicines with Arabic + English trade names, scientific composition, manufacturer, drug class, route, and EGP price.",
      "metrics": null
    },
    {
      "id": "ehsan",
      "name": "EHSAN",
      "type": "dataset",
      "country": "INTL",
      "org": "Newcastle University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "website": "https://zenodo.org/records/15418860",
        "paper": "https://arxiv.org/pdf/2508.02574"
      },
      "dialects": [
        "gulf"
      ],
      "size": "6,000 sentences",
      "year": 2025,
      "notes": "Arabic aspect-based sentiment dataset for healthcare reviews from Saudi hospitals.",
      "metrics": null
    },
    {
      "id": "elixir-fm",
      "name": "ElixirFM",
      "type": "tool",
      "country": "INTL",
      "org": "Otakar Smrz",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/otakar-smrz/elixir-fm"
      },
      "year": 2016,
      "notes": "Functional Arabic morphology system in Haskell.",
      "metrics": null
    },
    {
      "id": "elm",
      "name": "Elm",
      "type": "org",
      "country": "SA",
      "org": "Elm",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://elm.sa/en/"
      },
      "notes": "Digital transformation, gov AI (PIF-backed) - Nuha Arabic LLM, legal AI assistant, gov platform",
      "metrics": null
    },
    {
      "id": "elves",
      "name": "Elves",
      "type": "org",
      "country": "EG",
      "org": "Elves",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.elves.com/"
      },
      "notes": "Conversational commerce - Arabic AI-assisted concierge, human-in-the-loop ML",
      "metrics": null
    },
    {
      "id": "emg-arabic-sign-language",
      "name": "EMG Arabic Sign Language",
      "type": "dataset",
      "country": "TN",
      "org": "University of Tunis",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/ft9bhdgybs/2",
        "paper": "https://www.sciencedirect.com/science/article/pii/S2352340923008375"
      },
      "dialects": [
        "msa"
      ],
      "size": "18,716 images",
      "year": 2023,
      "notes": "Electromyography dataset of 18,716 gesture files capturing 28 Arabic alphabet handshapes and 10 digits via the Myo armband.",
      "metrics": null
    },
    {
      "id": "emirati-dialect-web-corpus",
      "name": "Emirati Dialect Web Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.wacl-1.7/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "notes": "Web-based corpus compilation of the Emirati Arabic dialect (WACL-4, 2025).",
      "metrics": null
    },
    {
      "id": "emirati-vits-male-1-0",
      "name": "emirati vits male 1.0",
      "type": "tts",
      "country": "INTL",
      "org": "Vadim Belsky",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/vadimbelsky/emirati-vits-male-1.0"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Bilingual Emirati Arabic and English VITS male TTS, trained on 70 hours of audio for call-center use.",
      "metrics": {
        "downloads": 0,
        "likes": 7,
        "lastModified": "2026-02-19"
      }
    },
    {
      "id": "emohopespeech",
      "name": "EmoHopeSpeech",
      "type": "dataset",
      "country": "QA",
      "org": "Northwestern University in Qatar",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "emotion"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.14669301"
      },
      "dialects": [
        "mixed"
      ],
      "size": "27,456 sentences",
      "year": 2025,
      "tags": [
        "multilingual"
      ],
      "notes": "Bilingual dataset of Arabic and English social media posts annotated for emotions and hope speech.",
      "metrics": null
    },
    {
      "id": "estedad",
      "name": "Estedad",
      "type": "tool",
      "country": "INTL",
      "org": "aminabedi68",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aminabedi68/Estedad",
        "website": "https://aminabedi68.github.io/Estedad/"
      },
      "year": 2026,
      "notes": "Sans Serif Arabic-Latin text typeface",
      "metrics": null
    },
    {
      "id": "estedad-mad",
      "name": "Estedad-Mad",
      "type": "tool",
      "country": "INTL",
      "org": "MDarvishi5124",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MDarvishi5124/Estedad-Mad",
        "website": "https://mdarvishi5124.github.io/Estedad-Mad"
      },
      "year": 2024,
      "notes": "یک فونت انگلیسی-عربی (فارسی).",
      "metrics": null
    },
    {
      "id": "evetar",
      "name": "EveTAR",
      "type": "dataset",
      "country": "INTL",
      "org": "TOBB University of Economics and Technology",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "event-detection",
        "summarization"
      ],
      "links": {
        "website": "https://sites.google.com/view/bigir/datasets?authuser=0#h.p_dB9cxP-26Xnc",
        "paper": "https://link.springer.com/article/10.1007/s10791-017-9325-7"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,550,000 sentences",
      "year": 2017,
      "notes": "A crawl of 355M Arabic tweets and covers 50 significant events",
      "metrics": null
    },
    {
      "id": "falcon-h1-arabic",
      "name": "Falcon-H1-Arabic",
      "type": "llm",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/tiiuae/falcon-h1-6819f2795bc4d0b25a2567e3"
      },
      "size": "3B-34B",
      "notes": "Hybrid Mamba-Transformer, 128K-256K context",
      "metrics": null
    },
    {
      "id": "fanar-mcp-server",
      "name": "Fanar MCP Server",
      "type": "agent-skill",
      "country": "INTL",
      "org": "danijeun",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/danijeun/fanar-mcp-server"
      },
      "year": 2026,
      "notes": "MCP server exposing Fanar API tools such as Islamic RAG and image generation.",
      "metrics": null
    },
    {
      "id": "fanar-star",
      "name": "Fanar Star",
      "type": "llm",
      "country": "QA",
      "org": "QCRI (Qatar)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://www.fanar.qa/en"
      },
      "size": "7B",
      "notes": "Trained from scratch on 1T Arabic/English tokens",
      "metrics": null
    },
    {
      "id": "farasa",
      "name": "Farasa",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "website": "https://farasa.qcri.org/"
      },
      "notes": "Fast and accurate Arabic text processing toolkit",
      "metrics": null
    },
    {
      "id": "farasa-segmentor",
      "name": "Farasa Segmentor",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "segmentation"
      ],
      "links": {
        "github": "https://github.com/qcri/FarasaSegmenter"
      },
      "notes": "Farasa package for segmenting and tokenizing Arabic text.",
      "metrics": null
    },
    {
      "id": "farasapy",
      "name": "farasapy",
      "type": "tool",
      "country": "SA",
      "org": "Maged Saeed",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "segmentation",
        "pos",
        "ner"
      ],
      "links": {
        "github": "https://github.com/MagedSaeed/farasapy"
      },
      "year": 2020,
      "notes": "Python wrapper around the Farasa Arabic NLP toolkit.",
      "metrics": null
    },
    {
      "id": "fareh",
      "name": "fareh",
      "type": "dataset",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/linuxscout/fareh"
      },
      "year": 2025,
      "notes": "Fareh: Arabic rules database for grammar and style checking فارح: لغتنا الجميلة",
      "metrics": null
    },
    {
      "id": "farspeech-2-0",
      "name": "FarSpeech 2.0",
      "type": "asr",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "translation",
        "dialect-id"
      ],
      "links": {
        "website": "https://alt.qcri.org/demos"
      },
      "notes": "System combining QCRI Arabic speech recognition, NLP, machine translation and dialect identification.",
      "metrics": null
    },
    {
      "id": "fassarli-ai",
      "name": "Fassarli-Ai",
      "type": "tool",
      "country": "INTL",
      "org": "elaaabidi04",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat",
        "embedding"
      ],
      "links": {
        "github": "https://github.com/elaaabidi04/Fassarli-Ai"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Fassarli (فسرلي) — A multilingual RAG chatbot with native Tunisian Darija support.",
      "metrics": null
    },
    {
      "id": "fassila",
      "name": "FASSILA",
      "type": "dataset",
      "country": "DZ",
      "org": "Ahmed Draia University",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "fact-checking",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/amincoding/FASSILA",
        "paper": "https://journal.iberamia.org/index.php/intartif/article/view/1151/204"
      },
      "dialects": [
        "magh"
      ],
      "size": "10,087 sentences",
      "year": 2023,
      "notes": "This corpus comprises 10,087 sentences, encompassing over 19,497 unique words in Algerian Dialect, and addresses the significant lack of linguistic.",
      "metrics": null
    },
    {
      "id": "fawry-camp-data-spring",
      "name": "fawry-camp-data-spring",
      "type": "tool",
      "country": "INTL",
      "org": "tawfik-s",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/tawfik-s/fawry-camp-data-spring"
      },
      "year": 2026,
      "notes": "fawry java && angular camp data and road map",
      "metrics": null
    },
    {
      "id": "festival-arabic-voices-docker",
      "name": "Festival Arabic TTS Docker",
      "type": "tts",
      "country": "SY",
      "org": "Nawar Halabi",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/nawarhalabi/festival-tts-arabic-voices-docker"
      },
      "year": 2020,
      "notes": "Docker image packaging a light full Arabic speech synthesis system on Festival; Syria.",
      "metrics": null
    },
    {
      "id": "festival-tts-arabic-voices",
      "name": "Festival Arabic voices",
      "type": "tts",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/linuxscout/festival-tts-arabic-voices"
      },
      "year": 2020,
      "notes": "Arabic voices for the Festival TTS engine; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "find-quran-verse",
      "name": "Find_Quran_Verse",
      "type": "asr",
      "country": "INTL",
      "org": "ameerssb",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "github": "https://github.com/ameerssb/Find_Quran_Verse",
        "website": "https://quran.pythonanywhere.com/"
      },
      "dialects": [
        "lev",
        "classical"
      ],
      "year": 2023,
      "notes": "The Quran Verse Finder is a voice-based Quran search application is a user-friendly and innovative tool designed to assist individuals.",
      "metrics": null
    },
    {
      "id": "fine-tashkeel",
      "name": "Fine-Tashkeel",
      "type": "tool",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "website": "https://www.researchgate.net/publication/372616004"
      },
      "notes": "Fine-tuned ByT5, 40% WER reduction",
      "metrics": null
    },
    {
      "id": "fineweb2-arabic",
      "name": "FineWeb2 Arabic",
      "type": "dataset",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/Omartificial-Intelligence-Space/huggingface-fineweb2-arabic-dataset-portions"
      },
      "notes": "Collection of curated Arabic portions of the FineWeb2 web corpus.",
      "metrics": null
    },
    {
      "id": "frahidi",
      "name": "Frahidi",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Frahidi"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Frahidi is a comprehensive system for performing prosodic analysis of classical Arabic poetry.",
      "metrics": null
    },
    {
      "id": "franco-arabic-transliterator",
      "name": "Franco-Arabic Transliterator",
      "type": "tool",
      "country": "EG",
      "org": "Amr Keleg",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "transliteration"
      ],
      "links": {
        "github": "https://github.com/AMR-KELEG/Franco-Arabic-Transliterator"
      },
      "year": 2019,
      "notes": "Rule-based converter from romanized (Franco) Arabic to Arabic script.",
      "metrics": null
    },
    {
      "id": "freedomintelligence",
      "name": "FreedomIntelligence",
      "type": "org",
      "country": "INTL",
      "org": "FreedomIntelligence",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/FreedomIntelligence"
      },
      "notes": "Arabic LLMs and alignment - AceGPT, Arabic cultural datasets",
      "metrics": null
    },
    {
      "id": "from-arabic-sentiment-analysis-to-sarcasm-detection",
      "name": "From Arabic Sentiment Analysis to Sarcasm Detection",
      "type": "dataset",
      "country": "INTL",
      "org": "The University of Edinburgh",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "sarcasm"
      ],
      "links": {
        "paper": "https://aclanthology.org/2020.osact-1.5/",
        "github": "https://github.com/iabufarha/ArSarcasm"
      },
      "year": 2020,
      "venue": "OSACT 2020",
      "notes": "We present ArSarcasm, an Arabic sarcasm detection dataset, which was created through the reannotation of available Arabic sentiment analysis datasets.",
      "metrics": null
    },
    {
      "id": "future-look-itc-flitc",
      "name": "Future Look ITC (FLITC)",
      "type": "org",
      "country": "SA",
      "org": "Future Look ITC (FLITC)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://flitc.ai/"
      },
      "notes": "Arabic-native AI solutions, venture studio - LABEAH, Smart Hire, Rayee Media, Nabadat",
      "metrics": null
    },
    {
      "id": "fuzzyarabicdict",
      "name": "FuzzyArabicDict",
      "type": "tool",
      "country": "INTL",
      "org": "michelleful",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "transliteration",
        "lexicon"
      ],
      "links": {
        "github": "https://github.com/michelleful/FuzzyArabicDict"
      },
      "notes": "Dictionary app that allows you to look up Arabic words in transliteration",
      "metrics": null
    },
    {
      "id": "g42",
      "name": "G42",
      "type": "org",
      "country": "AE",
      "org": "G42",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.g42.ai/"
      },
      "notes": "AI holding company, Arabic LLMs - Jais LLM, enterprise AI solutions",
      "metrics": null
    },
    {
      "id": "g42-inception-ai",
      "name": "G42 / Inception AI",
      "type": "org",
      "country": "AE",
      "org": "G42 / Inception AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.g42.ai/"
      },
      "notes": "Arabic-centric LLMs - Jais LLM family",
      "metrics": null
    },
    {
      "id": "gale-arabic-english-parallel-aligned-treebank-broadcast-news-part-2",
      "name": "GALE Arabic-English Parallel Aligned Treebank -- Broadcast News Part 2",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2014T03"
      },
      "dialects": [
        "msa"
      ],
      "size": "141,058 tokens",
      "year": 2014,
      "tags": [
        "multilingual",
        "variants:2"
      ],
      "notes": "The source data consists of Arabic broadcast news programming collected by LDC in 2007 and 2008 from Al Arabiya, Abu Dhabi TV, Al Baghdadya TV, Al Fayha.",
      "metrics": null
    },
    {
      "id": "gale-arabic-english-word-alignment-training-part-3-web",
      "name": "GALE Arabic-English Word Alignment Training Part 3 -- Web",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "retrieval",
        "translation"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2014T14"
      },
      "dialects": [
        "msa"
      ],
      "size": "7,332 sentences",
      "year": 2014,
      "tags": [
        "multilingual",
        "variants:4"
      ],
      "notes": "GALE Arabic-English word-aligned Arabic web text (Part 3) for machine translation research. (also: 4 other LDC releases)",
      "metrics": null
    },
    {
      "id": "gale-phase-1-distillation-training",
      "name": "GALE Phase 1 Distillation Training",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "retrieval"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2007T20"
      },
      "dialects": [
        "msa"
      ],
      "size": "93 sentences",
      "year": 2007,
      "tags": [
        "multilingual"
      ],
      "notes": "The annotation task involves responding to a series of user queries.",
      "metrics": null
    },
    {
      "id": "gale-phase-4-arabic-broadcast-news-speech",
      "name": "GALE Phase 4 Arabic Broadcast News Speech",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2018S05"
      },
      "dialects": [
        "msa"
      ],
      "size": "37 hours",
      "year": 2018,
      "tags": [
        "variants:9"
      ],
      "notes": "The recordings in this release feature news broadcasts focusing principally on current events from the following sources: Abu Dhabi TV, a television station.",
      "metrics": null
    },
    {
      "id": "gale-phase-4-arabic-broadcast-news-transcripts",
      "name": "GALE Phase 4 Arabic Broadcast News Transcripts",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2018T14"
      },
      "dialects": [
        "msa"
      ],
      "size": "204,735 tokens",
      "year": 2018,
      "tags": [
        "variants:9"
      ],
      "notes": "The transcripts were created with the LDC tool XTrans, which supports manual transcription and annotation of audio recordings. (also: 9 other LDC releases)",
      "metrics": null
    },
    {
      "id": "gale-phase-4-arabic-weblog-parallel-sentences",
      "name": "GALE Phase 4 Arabic Weblog Parallel Sentences",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2016T14"
      },
      "dialects": [
        "msa"
      ],
      "size": "68,346 tokens",
      "year": 2016,
      "tags": [
        "variants:17"
      ],
      "notes": "GALE Phase 4 Arabic Weblog Parallel Sentences includes 1,067 source-translation document pairs, comprising 68,346 words (Arabic source) of translated data.",
      "metrics": null
    },
    {
      "id": "garabic",
      "name": "garabic",
      "type": "tool",
      "country": "INTL",
      "org": "AbdullahDiaa",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/AbdullahDiaa/garabic",
        "website": "https://pkg.go.dev/github.com/AbdullahDiaa/garabic"
      },
      "year": 2026,
      "notes": "🛠 📦 A set of functions for Arabic text processing in golang",
      "metrics": null
    },
    {
      "id": "gatmath-and-gatlc",
      "name": "GATmath and GATLc",
      "type": "benchmark",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "website": "https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0329129"
      },
      "notes": "Benchmarks from Saudi GAT exams",
      "metrics": null
    },
    {
      "id": "gazelle",
      "name": "Gazelle",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "hf": "https://huggingface.co/papers/2410.18163"
      },
      "notes": "Arabic writing assistance dataset",
      "metrics": null
    },
    {
      "id": "gemma-3",
      "name": "Gemma 3",
      "type": "llm",
      "country": "INTL",
      "org": "Google",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/google/gemma-3-release-67c6c6f89c4f76621268bb6d"
      },
      "size": "1B-27B",
      "notes": "Multimodal capabilities",
      "metrics": null
    },
    {
      "id": "gemmar",
      "name": "GemmAr",
      "type": "llm",
      "country": "INTL",
      "org": "ClusterlabAi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.02147"
      },
      "size": "7B",
      "notes": "Gemma Arabic instruction-tuned on InstAr-500k",
      "metrics": null
    },
    {
      "id": "glare-google-apps-arabic-reviews-dataset-unknown",
      "name": "GLARE: Google Apps Arabic Reviews Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "reviews"
      ],
      "links": {
        "website": "https://zenodo.org/records/6457824"
      },
      "year": 2022,
      "notes": "GLARE: Arabic app reviews dataset collected from the Saudi Google Play Store.",
      "metrics": null
    },
    {
      "id": "goarabic",
      "name": "goarabic",
      "type": "tool",
      "country": "INTL",
      "org": "01walid",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/01walid/goarabic"
      },
      "year": 2023,
      "notes": "A Go Lang package for dealing with Arabic text.",
      "metrics": null
    },
    {
      "id": "goud",
      "name": "Goud",
      "type": "org",
      "country": "INTL",
      "org": "Goud",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "summarization"
      ],
      "links": {
        "hf": "https://huggingface.co/Goud"
      },
      "notes": "Hugging Face organization publishing text summarization models trained on the Goud-sum dataset.",
      "metrics": null
    },
    {
      "id": "gpdf",
      "name": "Gpdf",
      "type": "tool",
      "country": "INTL",
      "org": "omaralalwi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/omaralalwi/Gpdf"
      },
      "year": 2026,
      "notes": "Gpdf: PHP PDF Generator - HTML to PDF Converter for Laravel & PHP with native Arabic/RTL support, with built-in 17 fonts, and S3 storage.",
      "metrics": null
    },
    {
      "id": "graduation-project-advisor",
      "name": "graduation-project-advisor",
      "type": "tool",
      "country": "EG",
      "org": "h9-tec",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/h9-tec/graduation-project-advisor"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "Arabic-first LLM-backed advisor that helps Egyptian CS students pick a production-grade graduation project.",
      "metrics": null
    },
    {
      "id": "growth-marketing-os",
      "name": "growth-marketing-os",
      "type": "tool",
      "country": "INTL",
      "org": "growthack88",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/growthack88/growth-marketing-os",
        "website": "https://mahmoudomar.com"
      },
      "year": 2026,
      "notes": "Growth Marketing OS | Mahmoud Omar — open-source AI marketing prompts, Claude skills, agents & growth playbooks (EN + AR)",
      "metrics": null
    },
    {
      "id": "gumar",
      "name": "Gumar",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "other",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/gumar/?page=download&lang=en",
        "paper": "https://aclanthology.org/L16-1679.pdf"
      },
      "dialects": [
        "mixed",
        "gulf",
        "msa"
      ],
      "size": "1,235 documents",
      "year": 2016,
      "notes": "A large-scale corpus of Gulf Arabic consisting of 110 million words from 1,200 forum novels",
      "metrics": null
    },
    {
      "id": "gumar-corpus",
      "name": "Gumar Corpus",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "dialect-id"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/gumar/"
      },
      "year": 2018,
      "dialects": [
        "gulf"
      ],
      "notes": "Large Gulf Arabic corpus of about 110M words from social-media novels, annotated by dialect.",
      "metrics": null
    },
    {
      "id": "haad",
      "name": "HAAD",
      "type": "dataset",
      "country": "JO",
      "org": "Jordan University of Science and Technology",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/msmadi/HAAD",
        "paper": "https://ieeexplore.ieee.org/document/7300895"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,389 sentences",
      "year": 2015,
      "notes": "HAAD provides 2,389 human-annotated Arabic book reviews with aspect-level sentiment labels for aspect-based sentiment analysis.",
      "metrics": null
    },
    {
      "id": "habibi-tts",
      "name": "Habibi-TTS",
      "type": "tts",
      "country": "INTL",
      "org": "SWivid",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts",
        "dialect-tts"
      ],
      "links": {
        "github": "https://github.com/SWivid/Habibi-TTS"
      },
      "year": 2026,
      "notes": "Open-source unified-dialectal Arabic speech synthesis code built on F5-TTS.",
      "metrics": null
    },
    {
      "id": "hadith-omarshafie",
      "name": "hadith",
      "type": "tool",
      "country": "INTL",
      "org": "OmarShafie",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "vision",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/OmarShafie/hadith",
        "website": "https://dev.omarshafie.com/hadith/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "a search engine which provides Visual analysis of Hadith Isnad tree",
      "metrics": null
    },
    {
      "id": "hadith-halimbahae",
      "name": "Hadith (halimbahae)",
      "type": "dataset",
      "country": "INTL",
      "org": "halimbahae",
      "license": "other",
      "modality": "text",
      "tasks": [
        "diacritization",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/halimbahae/Hadith"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "A comprehensive open Hadith Library project featuring full databases of 9 renowned books, including Sahih al-Bukhari and Sahih Muslim.",
      "metrics": null
    },
    {
      "id": "hadith-api-fawaz",
      "name": "Hadith API (fawazahmed0)",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "unlicense",
      "modality": "text",
      "tasks": [
        "hadith",
        "api"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/hadith-api"
      },
      "year": 2022,
      "notes": "Free hadith API with multiple languages and grading.",
      "metrics": null
    },
    {
      "id": "hadith-mcp-ovehbe",
      "name": "Hadith MCP (ovehbe)",
      "type": "agent-skill",
      "country": "INTL",
      "org": "ovehbe",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/ovehbe/hadith-mcp"
      },
      "year": 2026,
      "notes": "MCP server for searchable, citation-safe hadith text.",
      "metrics": null
    },
    {
      "id": "hadith-data-sets",
      "name": "Hadith-Data-Sets",
      "type": "tool",
      "country": "INTL",
      "org": "abdelrahmaan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "github": "https://github.com/abdelrahmaan/Hadith-Data-Sets"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "All Hadith With Tashkil and Without Tashkel from the Nine Books that are 62,169 Hadith.",
      "metrics": null
    },
    {
      "id": "hadith-islamware",
      "name": "hadith-islamware",
      "type": "dataset",
      "country": "INTL",
      "org": "ceefour",
      "license": "unlicense",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "github": "https://github.com/ceefour/hadith-islamware",
        "website": "https://www.islamware.com/app/downloads"
      },
      "dialects": [
        "classical"
      ],
      "year": 2014,
      "notes": "Hadith database from Islam Ware",
      "metrics": null
    },
    {
      "id": "hadith-json",
      "name": "hadith-json",
      "type": "dataset",
      "country": "INTL",
      "org": "AhmedBaset",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith"
      ],
      "links": {
        "github": "https://github.com/AhmedBaset/hadith-json"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Database of Prophet hadiths include 50,884 hadiths from 17 book, among of them the nine books",
      "metrics": null
    },
    {
      "id": "hakim",
      "name": "Hakim",
      "type": "org",
      "country": "INTL",
      "org": "Hakim",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://tryhakim.ai/en/blog/khaleeji-arabic-voice-ai-explained"
      },
      "notes": "Voice AI company focused on Khaleeji Arabic.",
      "metrics": null
    },
    {
      "id": "halwasa",
      "name": "Halwasa",
      "type": "dataset",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ai-text-detection"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/ArabicLLMsHallucination.zip",
        "paper": "https://aclanthology.org/2024.lrec-main.705.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "10,000 sentences",
      "year": 2024,
      "notes": "First Arabic dataset with 10K LLM-generated sentences annotated for factuality.",
      "metrics": null
    },
    {
      "id": "hbku",
      "name": "Hamad Bin Khalifa University",
      "type": "org",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.hbku.edu.qa"
      },
      "notes": "Qatar university hosting QCRI; Arabic NLP, speech and Fanar research.",
      "metrics": null
    },
    {
      "id": "hamsa",
      "name": "Hamsa",
      "type": "org",
      "country": "AE",
      "org": "Hamsa",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://tryhamsa.com"
      },
      "notes": "Arabic dialect speech recognition and voice AI platform.",
      "metrics": null
    },
    {
      "id": "handjet",
      "name": "handjet",
      "type": "tool",
      "country": "INTL",
      "org": "rosettatype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/rosettatype/handjet",
        "website": "http://rosettatype.com/Handjet"
      },
      "year": 2024,
      "notes": "Handjet: an element-based variable font",
      "metrics": null
    },
    {
      "id": "haqa",
      "name": "HAQA",
      "type": "dataset",
      "country": "SA",
      "org": "King Abdulaziz University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/scsaln/HAQA-and-QUQA",
        "paper": "https://aclanthology.org/2023.ranlp-1.10.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,598 sentences",
      "year": 2023,
      "notes": "The Arabic HAQA dataset of Hadith sharif answers that contains 1598 records and 1359 questions.",
      "metrics": null
    },
    {
      "id": "hareef",
      "name": "hareef",
      "type": "tool",
      "country": "INTL",
      "org": "Musharraf Omer",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/mush42/hareef"
      },
      "year": 2025,
      "notes": "state-of-the-art models for diacritics restoration for Arabic language",
      "metrics": null
    },
    {
      "id": "harness",
      "name": "HArnESS",
      "type": "asr",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "dialect-id",
        "emotion"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2604.14186"
      },
      "year": 2026,
      "notes": "Family of lightweight self-supervised Arabic speech models distilled from a bilingual teacher for ASR, DID and SER.",
      "metrics": null
    },
    {
      "id": "hassaniya-dataset-mendeley",
      "name": "HASSANIYA Dataset (Mendeley)",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/m2swkr2bhx"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Mauritanian Hassaniya dialect dataset published on Mendeley Data.",
      "metrics": null
    },
    {
      "id": "hate-speech-detection-with-adhar-a-multi-dialectal-hate-speech-corpus",
      "name": "Hate speech detection with ADHAR: a multi-dialectal hate speech corpus in Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/Data_Sheet_1_Hate_speech_detection_with_ADHAR_a_multi-dialectal_hate_speech_corpus_in_Arabic_pdf/25931464"
      },
      "year": 2024,
      "notes": "Supplementary data sheet of ADHAR, a multi-dialectal Arabic hate speech corpus.",
      "metrics": null
    },
    {
      "id": "hatespeecharabic",
      "name": "HateSpeechArabic",
      "type": "dataset",
      "country": "QA",
      "org": "Northwestern University in Qatar",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "offensive-language"
      ],
      "links": {
        "website": "https://zenodo.org/records/14669917",
        "paper": "https://aclanthology.org/2025.ranlp-1.163/"
      },
      "dialects": [
        "mixed"
      ],
      "size": "10,000 sentences",
      "year": 2025,
      "notes": "A multilabel Arabic hate speech corpus of 10,000 tweets annotated for offensive content and hate targets.",
      "metrics": null
    },
    {
      "id": "hazawi",
      "name": "Hazawi+",
      "type": "dataset",
      "country": "KW",
      "org": "Kuwaiti researchers",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "storytelling"
      ],
      "links": {
        "paper": "https://dl.acm.org/doi/full/10.1145/3800688"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Structured corpus of Kuwaiti Arabic stories.",
      "metrics": null
    },
    {
      "id": "hazen-ai",
      "name": "Hazen.ai",
      "type": "org",
      "country": "SA",
      "org": "Hazen.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.hazen.ai/"
      },
      "notes": "AI traffic safety & computer vision - Deep learning road safety, seatbelt/phone detection",
      "metrics": null
    },
    {
      "id": "hellothere",
      "name": "HelloThere",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "other",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/hellothere-corpus/",
        "paper": "https://aclanthology.org/2024.sigdial-1.12.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "317 documents",
      "year": 2024,
      "notes": "Dialogue corpus with 317 multi-turn conversations between users and Time-Offset Interaction Application avatars.",
      "metrics": null
    },
    {
      "id": "helsinki-nlp",
      "name": "Helsinki-NLP",
      "type": "org",
      "country": "INTL",
      "org": "Helsinki-NLP",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/Helsinki-NLP"
      },
      "notes": "Machine translation models - OPUS-MT Arabic translation models",
      "metrics": null
    },
    {
      "id": "hierarchical-dataset-of-ancient-arabic-manuscripts-haam",
      "name": "Hierarchical Dataset of Ancient Arabic Manuscripts (HAAM)",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "segmentation",
        "ocr"
      ],
      "links": {
        "website": "https://zenodo.org/records/21627021"
      },
      "year": 2026,
      "notes": "HAAM: ancient Arabic manuscripts annotated for segmentation into lines, words, pseudo-words and characters.",
      "metrics": null
    },
    {
      "id": "hosn",
      "name": "Hosn",
      "type": "org",
      "country": "OM",
      "org": "Hosn",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://hosn.om"
      },
      "notes": "Arabic-first on-premise AI for Omani institutions, with an air-gapped option built on open models.",
      "metrics": null
    },
    {
      "id": "htr-model-arabic-handwritten-recognition-model-trained-on-the-muharaf",
      "name": "HTR Model - Arabic Handwritten Recognition Model Trained on the Muharaf Corpus",
      "type": "ocr",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "ocr",
        "handwriting"
      ],
      "links": {
        "website": "https://zenodo.org/records/14295489"
      },
      "year": 2024,
      "notes": "Kraken handwriting recognition model for Arabic trained on the Muharaf manuscripts dataset.",
      "metrics": null
    },
    {
      "id": "hudhud-ai",
      "name": "Hudhud AI",
      "type": "org",
      "country": "SA",
      "org": "Hudhud AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://hudhud.ai/"
      },
      "notes": "Arabic conversational AI (no-code SaaS) - Saudi-accent chatbots, Arabic-first customer engagement",
      "metrics": null
    },
    {
      "id": "humain",
      "name": "HUMAIN",
      "type": "org",
      "country": "SA",
      "org": "HUMAIN",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.humain.com/"
      },
      "notes": "PIF-backed full-stack AI company - ALLaM 34B, HUMAIN Chat, 8PB Arabic training data",
      "metrics": null
    },
    {
      "id": "humain-saudi-pif",
      "name": "HUMAIN (Saudi PIF)",
      "type": "org",
      "country": "SA",
      "org": "HUMAIN (Saudi PIF)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.humain.com/"
      },
      "notes": "Full-stack AI company - ALLaM 34B, HUMAIN Chat, AI factories",
      "metrics": null
    },
    {
      "id": "humain-chat",
      "name": "HUMAIN Chat",
      "type": "tool",
      "country": "SA",
      "org": "HUMAIN",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://www.humain.ai/en/news/humain-chat-launch/"
      },
      "year": 2025,
      "notes": "Arabic-first conversational AI app powered by ALLAM 34B, launched in Saudi Arabia on web, iOS and Android.",
      "metrics": null
    },
    {
      "id": "humain-create",
      "name": "HUMAIN Create",
      "type": "tool",
      "country": "SA",
      "org": "HUMAIN",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "marketing"
      ],
      "links": {
        "website": "https://www.humain.ai/en/create"
      },
      "year": 2026,
      "notes": "AI-native marketing operating system producing campaigns in Arabic and English.",
      "metrics": null
    },
    {
      "id": "humain-one",
      "name": "HUMAIN ONE",
      "type": "tool",
      "country": "SA",
      "org": "HUMAIN",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "agents"
      ],
      "links": {
        "website": "https://www.humain.ai/en/humain-os"
      },
      "year": 2025,
      "notes": "AI-agent operating system connecting enterprise systems and automating tasks from one interface (Arabic/English).",
      "metrics": null
    },
    {
      "id": "hurmoz",
      "name": "Hurmoz",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/hurmoz"
      },
      "year": 2026,
      "notes": "Collection of 63 Arabic skills for the Hermes Agent framework.",
      "metrics": null
    },
    {
      "id": "i18n",
      "name": "i18n",
      "type": "tool",
      "country": "INTL",
      "org": "softvenue",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/softvenue/i18n",
        "website": "https://i18n.softvenue.net"
      },
      "year": 2026,
      "notes": "internationalize projects to Arabic",
      "metrics": null
    },
    {
      "id": "ia2d-iraqi-arabic-dialect-dataset",
      "name": "IA2D Iraqi Arabic Dialect Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "ebady",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/ebady/Iraqi-Arabic-Dialect-Dataset"
      },
      "dialects": [
        "iraqi"
      ],
      "notes": "Annotated Iraqi Arabic dialect dataset (IA2D).",
      "metrics": null
    },
    {
      "id": "icompass",
      "name": "iCompass",
      "type": "org",
      "country": "TN",
      "org": "iCompass",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.instadeep.com/2021/03/instadeep-and-icompass-announce-tunbert-the-first-ai-based-tunisian-dialect-system/"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian AI company; co-built TunBERT and a Tunisian dialect speech stack with InstaDeep.",
      "metrics": null
    },
    {
      "id": "identifying-code-switching-in-arabizi",
      "name": "Identifying Code-switching in Arabizi",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Haifa",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "transliteration",
        "code-switching"
      ],
      "links": {
        "paper": "https://aclanthology.org/2022.wanlp-1.18/",
        "github": "https://github.com/HaifaCLG/Arabizi"
      },
      "year": 2022,
      "venue": "WANLP 2022",
      "notes": "We describe a corpus of social media posts that include utterances in Arabizi, a Roman-script rendering of Arabic, mixed with other languages, notably.",
      "metrics": null
    },
    {
      "id": "ilm",
      "name": "ilm",
      "type": "tool",
      "country": "INTL",
      "org": "arriqaaq",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/arriqaaq/ilm",
        "website": "https://arriqaaq.github.io/ilm/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A semantic search platform for Islamic scholarship — Quran with tafsir, 34K+ hadiths with narrator chains, and interactive isnad graphs.",
      "metrics": null
    },
    {
      "id": "imageeval-2025",
      "name": "ImageEval 2025",
      "type": "benchmark",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.52/",
        "website": "https://sina.birzeit.edu/image_eval2025/"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "We present ImageEval 2025, the first shared task dedicated to Arabic image captioning.",
      "metrics": null
    },
    {
      "id": "imam-university",
      "name": "Imam Mohammad ibn Saud Islamic University",
      "type": "org",
      "country": "SA",
      "org": "Imam Mohammad ibn Saud Islamic University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://imamu.edu.sa"
      },
      "notes": "Riyadh university with Arabic NLP and computational linguistics research.",
      "metrics": null
    },
    {
      "id": "inception-ai",
      "name": "Inception AI",
      "type": "org",
      "country": "AE",
      "org": "Inception AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.g42.ai/"
      },
      "notes": "Arabic-centric foundation models - Jais model family (with Cerebras)",
      "metrics": null
    },
    {
      "id": "inference-free-splade-distilbert-base-arabic-cased-nq",
      "name": "inference free splade distilbert base Arabic cased nq",
      "type": "llm",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space/inference-free-splade-distilbert-base-Arabic-cased-nq"
      },
      "year": 2025,
      "notes": "This is a Asymmetric Inference-free SPLADE Sparse Encoder model finetuned from distilbert/distilbert-base-multilingual-cased.",
      "base_model": [
        "distilbert/distilbert-base-multilingual-cased"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2025-07-02"
      }
    },
    {
      "id": "injaz-solutions",
      "name": "INJAZ Solutions",
      "type": "org",
      "country": "EG",
      "org": "INJAZ Solutions",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://injaz.com.eg"
      },
      "notes": "Egyptian company offering Arabic AI solutions; surfaced by an Arabic OCR company search.",
      "metrics": null
    },
    {
      "id": "instadeep",
      "name": "InstaDeep",
      "type": "org",
      "country": "TN",
      "org": "InstaDeep",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.instadeep.com"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Tunis-founded AI company; co-released TunBERT, the first Tunisian dialect language model.",
      "metrics": null
    },
    {
      "id": "intella",
      "name": "Intella",
      "type": "org",
      "country": "EG",
      "org": "Intella",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://intella.ai/"
      },
      "notes": "Arabic speech AI intelligence - Arabic STT across 25+ dialects (95.7% accuracy), Ziila digital human",
      "metrics": null
    },
    {
      "id": "international-corpus-of-arabic",
      "name": "International Corpus of Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "Bibliotheca Alexandrina",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "http://www.bibalex.org/ica/ar/",
        "paper": "http://www.bibalex.org/isis/UploadedFiles/Publications/Building%20an%20Intl%20corpus%20of%20arabic.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "100,000,000 tokens",
      "year": 2007,
      "notes": "There are two points of view about the need for corpora.",
      "metrics": null
    },
    {
      "id": "invizo-ocr",
      "name": "Invizo-OCR",
      "type": "ocr",
      "country": "INTL",
      "org": "Hedrax",
      "license": "other",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/Hedrax/Invizo-OCR",
        "website": "https://arxiv.org/abs/2502.05277"
      },
      "year": 2025,
      "notes": "This project offers an end-to-end Arabic OCR solution for handwritten and printed text on template-based documents.",
      "metrics": null
    },
    {
      "id": "ipa-transcriptions-of-the-arabic-speech-corpus-asc",
      "name": "IPA Transcriptions of the Arabic Speech Corpus (ASC)",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "phonetics",
        "tts"
      ],
      "links": {
        "website": "https://zenodo.org/records/17111978"
      },
      "dialects": [
        "lev"
      ],
      "year": 2025,
      "notes": "IPA transcriptions of the Arabic Speech Corpus, a 4-hour South Levantine (Damascus) recording set.",
      "metrics": null
    },
    {
      "id": "iqra-ai",
      "name": "IqraAI",
      "type": "asr",
      "country": "INTL",
      "org": "Abdirahman Nomad",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran-recitation"
      ],
      "links": {
        "github": "https://github.com/AbdirahmanNomad/IqraAI"
      },
      "year": 2026,
      "notes": "Quran speech recognition with verse matching and a recitation practice mode.",
      "metrics": null
    },
    {
      "id": "iqro-json",
      "name": "iqro-json",
      "type": "tool",
      "country": "INTL",
      "org": "dyazincahya",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/dyazincahya/iqro-json",
        "website": "https://dyazincahya.github.io/iqro-json/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "(Proyek ini masih berjalan) - Buku untuk belajar mengaji Al-quran, berisi data Iqro 1 sampai 6 dalam format json.",
      "metrics": null
    },
    {
      "id": "iraq-speech-labs",
      "name": "Iraq Speech Labs",
      "type": "org",
      "country": "IQ",
      "org": "Iraq SpeechLabs",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://iraqspeechlaps.shop/"
      },
      "notes": "Iraqi data company building custom Arabic speech, TTS and LLM datasets; claims 12K speech hours across 18 dialects.",
      "metrics": null
    },
    {
      "id": "iraqi-arabic-conversational-telephone-speech",
      "name": "Iraqi Arabic Conversational Telephone Speech",
      "type": "dataset",
      "country": "INTL",
      "org": "LDC",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2006S45"
      },
      "dialects": [
        "iraqi"
      ],
      "size": "50 hours",
      "year": 2006,
      "notes": "Iraqi Arabic Conversational Telephone Speech was developed by Appen Pty Ltd, Sydney, Australia and contains roughly 3000 mins of speech from Iraqi Arabic.",
      "metrics": null
    },
    {
      "id": "iraqi-dialect-tts-corpus",
      "name": "Iraqi Dialect TTS Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "hayderkharrufa",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/hayderkharrufa/iraqi-dialect-tts-corpus"
      },
      "dialects": [
        "iraqi"
      ],
      "notes": "Custom-recorded Iraqi dialect audio with transcripts for text-to-speech training.",
      "metrics": null
    },
    {
      "id": "iraqi-dialect-llm",
      "name": "iraqi_dialect_llm",
      "type": "llm",
      "country": "INTL",
      "org": "EzioDevio",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/EzioDevio/iraqi_dialect_llm"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2024,
      "notes": "Language model fine-tuned on Iraqi Arabic dialect text.",
      "metrics": {
        "downloads": 0,
        "likes": 0,
        "lastModified": "2024-11-05"
      }
    },
    {
      "id": "isharah",
      "name": "Isharah",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://snalyami.github.io/Isharah_CSLR/",
        "paper": "https://arxiv.org/pdf/2506.03615.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "30,000 videos",
      "year": 2025,
      "notes": "Large-scale multi-scene Saudi Sign Language dataset for continuous sign language recognition.",
      "metrics": null
    },
    {
      "id": "islam-companion-web-api",
      "name": "Islam-Companion-Web-Api",
      "type": "tool",
      "country": "INTL",
      "org": "pakjiddat",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "translation",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/pakjiddat/Islam-Companion-Web-Api",
        "website": "https://pakjiddat.netlify.app/posts/islam-companion-web-api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "A RESTFul API for accessing Holy Quran and Hadith data",
      "metrics": null
    },
    {
      "id": "islam-js",
      "name": "islam.js",
      "type": "tool",
      "country": "INTL",
      "org": "dev-ahmadbilal",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/dev-ahmadbilal/islam.js",
        "website": "https://www.npmjs.com/package/islam.js"
      },
      "dialects": [
        "classical",
        "mixed"
      ],
      "year": 2026,
      "notes": "A comprehensive Typescript package offering Quranic text with multiple dialects.",
      "metrics": null
    },
    {
      "id": "islamic-knowledge-skill",
      "name": "Islamic Knowledge Skill",
      "type": "agent-skill",
      "country": "INTL",
      "org": "shadysalman",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/shadysalman/islamic-knowledge-skill"
      },
      "year": 2026,
      "notes": "RAG-powered Claude skill over the Quran (6,236 ayat) and Sahih Bukhari.",
      "metrics": null
    },
    {
      "id": "islamic-api-s",
      "name": "Islamic-API-S",
      "type": "tool",
      "country": "INTL",
      "org": "alfa155518",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/alfa155518/Islamic-API-S"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Islamic APIS collection Contains [Quran-Datasets, Hadith-Datasets, Azkar-Datasets, Quran-APIs, Prayer-Times-API, Quraan voice ]",
      "metrics": null
    },
    {
      "id": "islamic-data-repository",
      "name": "islamic-data-repository",
      "type": "tool",
      "country": "INTL",
      "org": "6km",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/6km/islamic-data-repository",
        "website": "https://islamic-data.vercel.app/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "مستودع البيانات الإسلامية - قائمة بالموارد التي قد تفيد المبرمجين في تطوير تطبيقات ومواقع الويب الإسلامية.",
      "metrics": null
    },
    {
      "id": "islamiceval-2025",
      "name": "IslamicEval 2025",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hallucination",
        "islamic"
      ],
      "links": {
        "website": "https://www.fanar.qa/en/publications"
      },
      "year": 2025,
      "notes": "First shared task on capturing LLM hallucination in Islamic content.",
      "metrics": null
    },
    {
      "id": "islamicquizapi",
      "name": "IslamicQuizAPI",
      "type": "tool",
      "country": "INTL",
      "org": "rn0x",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/rn0x/IslamicQuizAPI",
        "website": "https://islamicquiz.i8x.net/api/questions/random?count=5"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "قاعدة بيانات وواجهة برمجة تطبيقات API لتقديم أسئلة لتقيم مستواك في العلوم الشرعية وتطوير حصيلتك العلمية في مجالات متنوعة، جميع الإجابات مصدرها موقع الدرر.",
      "metrics": null
    },
    {
      "id": "isnad",
      "name": "isnad",
      "type": "benchmark",
      "country": "INTL",
      "org": "alizahidraja",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/alizahidraja/isnad",
        "website": "https://alizahidraja.com/isnad"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Grade every agent, scraper and model in a claim's chain — provenance, trust scoring and audit evidence for LLM pipelines",
      "metrics": null
    },
    {
      "id": "itida-mcit",
      "name": "ITIDA / MCIT",
      "type": "org",
      "country": "EG",
      "org": "ITIDA / MCIT",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://itida.gov.eg/"
      },
      "notes": "National AI authority, sovereign models - Karnak LLM, SIA AI tutor, AcQua NLP, BelMasry, Torgoman",
      "metrics": null
    },
    {
      "id": "itida-mcit-egypt",
      "name": "ITIDA / MCIT (Egypt)",
      "type": "org",
      "country": "EG",
      "org": "ITIDA / MCIT (Egypt)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://itida.gov.eg/"
      },
      "notes": "Egypt's national AI, sovereign models - Karnak LLM, BelMasry, Torgoman, SIA, AcQua",
      "metrics": null
    },
    {
      "id": "itqan",
      "name": "Itqan",
      "type": "tool",
      "country": "INTL",
      "org": "R3GENESI5",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/R3GENESI5/Itqan",
        "website": "https://r3genesi5.github.io/Itqan/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "The first computational Quran-Hadith concordance (96.3% root coverage, 1.5M links) and the largest open-source narrator database.",
      "metrics": null
    },
    {
      "id": "jcca-jordan-comprehensive-contemporary-arabic-corpus",
      "name": "JCCA: Jordan Comprehensive Contemporary Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4616/"
      },
      "dialects": [
        "lev"
      ],
      "year": 2019,
      "notes": "Jordan Comprehensive Contemporary Arabic corpus: construction and annotation, presented at WANLP 2019.",
      "metrics": null
    },
    {
      "id": "jhsc",
      "name": "JHSC",
      "type": "dataset",
      "country": "JO",
      "org": "Princess Sumaya University for Technology",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "offensive-language",
        "sentiment"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/mcnzzpgrdj/2",
        "paper": "https://doi.org/10.3389/frai.2024.1345445"
      },
      "dialects": [
        "lev"
      ],
      "size": "403,688 sentences",
      "year": 2023,
      "notes": "A multi-class Arabic hate-speech corpus of Jordanian-dialect tweets labeled for hate speech presence.",
      "metrics": null
    },
    {
      "id": "joda",
      "name": "JODA",
      "type": "dataset",
      "country": "JO",
      "org": "Gheith Abandah (Univ. of Jordan)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/Gheith-Abandah/JODA"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Jordanian dialect dataset paired with Jordanian-to-MSA translations.",
      "metrics": null
    },
    {
      "id": "joda-a-dataset-of-jordanian-dialect-and-erroneous-modern-arabic-senten",
      "name": "JODA - A Dataset of Jordanian Dialect and Erroneous Modern Arabic Sentences coupled with Proper MSA and Full Diacritics",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "error-correction",
        "dialects"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/ffrskd27f4"
      },
      "dialects": [
        "lev",
        "msa"
      ],
      "year": 2025,
      "notes": "JODA: Jordanian dialect and erroneous Modern Standard Arabic sentences for dialect processing and error correction.",
      "metrics": null
    },
    {
      "id": "jordanian-arabic-tts",
      "name": "jordanian-arabic-tts",
      "type": "tts",
      "country": "INTL",
      "org": "Abdelkareem",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/Abdelkareem/jordanian-arabic-tts"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Hugging Face Space demo of Jordanian Arabic text-to-speech.",
      "metrics": null
    },
    {
      "id": "jslingua",
      "name": "JsLingua",
      "type": "tool",
      "country": "DZ",
      "org": "Abdelkrime Aries",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/kariminf/jslingua"
      },
      "year": 2016,
      "notes": "JavaScript libraries to process Arabic and other languages; Algeria.",
      "metrics": null
    },
    {
      "id": "kacst",
      "name": "KACST",
      "type": "dataset",
      "country": "SA",
      "org": "KACST",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "website": "http://www.kacstac.org.sa",
        "paper": "https://link.springer.com/article/10.1007/s10579-014-9284-1"
      },
      "dialects": [
        "mixed"
      ],
      "size": "7,000,000 tokens",
      "year": 2015,
      "notes": "The KACST Arabic corpus comprises more than 700 million words from the pre-Islamic era to the present day (a period covering more than 1,500 years)",
      "metrics": null
    },
    {
      "id": "kaiflematha",
      "name": "KaifLematha",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "reading-comprehension"
      ],
      "links": {
        "github": "https://github.com/esulaiman/Arabic-WikiReading-and-KaifLematha-datasets",
        "paper": "https://link.springer.com/content/pdf/10.1007/s10579-022-09577-5.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "100,000 documents",
      "year": 2022,
      "notes": "Arabic WikiReading and KaifLematha: large-scale Arabic reading comprehension datasets with over 100K instances.",
      "metrics": null
    },
    {
      "id": "kaldi-arabic",
      "name": "kaldi-arabic",
      "type": "asr",
      "country": "SA",
      "org": "Asrajeh",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/asrajeh/kaldi-arabic"
      },
      "year": 2021,
      "notes": "HMM-based Arabic ASR recipe built on Kaldi.",
      "metrics": null
    },
    {
      "id": "kalimat-multipurpose-arabic-corpus",
      "name": "KALIMAT Multipurpose Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization",
        "classification"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/kalimat/"
      },
      "notes": "Multipurpose corpus of 20,291 Arabic articles collected from the Omani newspaper Alwatan.",
      "metrics": null
    },
    {
      "id": "karem-arabic-presentation",
      "name": "karem-arabic-presentation",
      "type": "agent-skill",
      "country": "INTL",
      "org": "karem505",
      "license": "mit",
      "modality": "none",
      "tasks": [
        "skill",
        "presentation"
      ],
      "links": {
        "github": "https://github.com/karem505/karem-arabic-presentation"
      },
      "notes": "Claude Code skill for Arabic/English bilingual RTL HTML presentations",
      "metrics": null
    },
    {
      "id": "karnak",
      "name": "Karnak",
      "type": "llm",
      "country": "EG",
      "org": "ITIDA (Egypt)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://itida.gov.eg/English/PressReleases/Pages/egypt-national-ai-karnak-llm-launch-Ai-Everything-MEA-2026.aspx"
      },
      "size": "30B-70B",
      "notes": "Egypt's national sovereign LLM, top Arabic in its class",
      "metrics": null
    },
    {
      "id": "karsl",
      "name": "KArSL",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://hamzah-luqman.github.io/KArSL/",
        "paper": "https://dl.acm.org/doi/pdf/10.1145/3423420"
      },
      "dialects": [
        "msa"
      ],
      "size": "75,300 videos",
      "year": 2021,
      "notes": "Arabic Sign Language database with 502 signs.",
      "metrics": null
    },
    {
      "id": "kaust",
      "name": "KAUST",
      "type": "org",
      "country": "SA",
      "org": "KAUST",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://cemse.kaust.edu.sa/"
      },
      "notes": "AI research, Arabic NLP - Center of Excellence in Generative AI",
      "metrics": null
    },
    {
      "id": "kaust-cemse",
      "name": "KAUST (CEMSE)",
      "type": "org",
      "country": "SA",
      "org": "KAUST (CEMSE)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://cemse.kaust.edu.sa/"
      },
      "notes": "Generative AI center, Arabic NLP research, sentiment analysis",
      "metrics": null
    },
    {
      "id": "kawarith",
      "name": "Kawarith",
      "type": "dataset",
      "country": "SA",
      "org": "University of Birmingham / Taibah University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "crisis",
        "twitter"
      ],
      "links": {
        "paper": "https://aclanthology.org/2021.wanlp-1.5/",
        "github": "https://github.com/alaa-a-a/kawarith"
      },
      "year": 2021,
      "venue": "WANLP 2021",
      "notes": "Multi-dialect Arabic Twitter corpus of over a million tweets collected during 22 crisis events.",
      "dialects": [
        "mixed"
      ],
      "metrics": null
    },
    {
      "id": "kawn",
      "name": "Kawn",
      "type": "llm",
      "country": "SA",
      "org": "Misraj AI (Saudi)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://misraj.ai/"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Arabic-first ecosystem with 15-dialect support",
      "metrics": null
    },
    {
      "id": "kawn-embed-islamic",
      "name": "Kawn Embed Islamic",
      "type": "embedding",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "retrieval",
        "quran",
        "hadith"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/kawn-embed-islamic"
      },
      "year": 2026,
      "notes": "Arabic embeddings optimised for retrieval over Quran, Hadith, Fiqh and Fatwa collections.",
      "metrics": null
    },
    {
      "id": "kawn-embed-light",
      "name": "Kawn Embed Light",
      "type": "embedding",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "retrieval"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/kawn-embed-light"
      },
      "year": 2026,
      "notes": "Lightweight (139M) general-purpose Arabic embeddings for search, retrieval and RAG.",
      "metrics": null
    },
    {
      "id": "kawn-embed-medical",
      "name": "Kawn Embed Medical",
      "type": "embedding",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "medical"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/kawn-embed-medical"
      },
      "year": 2026,
      "notes": "Domain-specific Arabic embeddings for healthcare applications.",
      "metrics": null
    },
    {
      "id": "kazma",
      "name": "Kazma",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Mubder",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/Mubder/kazma"
      },
      "year": 2026,
      "notes": "Self-hosted agent platform, bilingual English and Arabic by design.",
      "metrics": null
    },
    {
      "id": "kfupm-jrcai",
      "name": "KFUPM-JRCAI",
      "type": "org",
      "country": "SA",
      "org": "KFUPM-JRCAI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/KFUPM-JRCAI"
      },
      "notes": "Joint SDAIA-KFUPM AI research - Arabic AI text detection datasets",
      "metrics": null
    },
    {
      "id": "khalifa-university",
      "name": "Khalifa University",
      "type": "org",
      "country": "AE",
      "org": "Khalifa University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.ku.ac.ae"
      },
      "notes": "Abu Dhabi university with NLP, speech and AI research.",
      "metrics": null
    },
    {
      "id": "khawas",
      "name": "Khawas",
      "type": "dataset",
      "country": "SA",
      "org": "KACST",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/kacst-acptool/",
        "paper": "https://ieeexplore.ieee.org/abstract/document/6646005"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,910 documents",
      "year": 2013,
      "notes": "A corpus containing more than two million words and a corpora processing tool that is specifically designed for Arabic",
      "metrics": null
    },
    {
      "id": "khoja-stemmer-cli",
      "name": "Khoja Stemmer CLI",
      "type": "tool",
      "country": "PS",
      "org": "Motaz Saad",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "stemming"
      ],
      "links": {
        "github": "https://github.com/motazsaad/khoja-stemmer-command-line"
      },
      "year": 2015,
      "notes": "Command line version of the Khoja Arabic rooting stemmer; Palestine.",
      "metrics": null
    },
    {
      "id": "kitab-bench",
      "name": "KITAB-Bench",
      "type": "benchmark",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/MBZUAI/KITAB-Bench"
      },
      "notes": "Arabic OCR benchmark: 8,809 samples, 9 domains, 36 sub-domains (MBZUAI)",
      "metrics": null
    },
    {
      "id": "kitab-font",
      "name": "kitab-font",
      "type": "tool",
      "country": "INTL",
      "org": "nuqayah",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/nuqayah/kitab-font"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "A classical Arabic Naskh font",
      "metrics": null
    },
    {
      "id": "kivun-terminal-wsl",
      "name": "Kivun Terminal WSL",
      "type": "agent-skill",
      "country": "INTL",
      "org": "noambrand",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/noambrand/kivun-terminal-wsl"
      },
      "year": 2026,
      "notes": "Claude Code in WSL with Hebrew, Arabic and Persian rendered correctly.",
      "metrics": null
    },
    {
      "id": "klaam",
      "name": "Klaam",
      "type": "asr",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/ARBML/klaam"
      },
      "dialects": [
        "msa",
        "egy"
      ],
      "notes": "Arabic ASR/TTS/classification library (MSA + Egyptian)",
      "metrics": null
    },
    {
      "id": "kngine",
      "name": "Kngine",
      "type": "org",
      "country": "EG",
      "org": "Kngine",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://kngine.com/"
      },
      "notes": "Semantic search & NLP - Arabic semantic search, data mining, knowledge engine",
      "metrics": null
    },
    {
      "id": "ku-corpus-of-arabic-recordings-urban-jordanian",
      "name": "KU Corpus of Arabic Recordings (Urban Jordanian)",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Kansas",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech"
      ],
      "links": {
        "website": "https://kuppl.ku.edu/corpus-arabic-recordings"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Downloadable recordings of urban Jordanian Arabic speakers from the University of Kansas.",
      "metrics": null
    },
    {
      "id": "kuwain",
      "name": "Kuwain",
      "type": "llm",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.15120"
      },
      "size": "1.5B",
      "notes": "Arabic SLM via language injection, 70% cost reduction",
      "on_device": true,
      "metrics": null
    },
    {
      "id": "kuwait-university-data-ai-lab",
      "name": "Kuwait University Data & AI Lab",
      "type": "org",
      "country": "KW",
      "org": "Kuwait University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.ku.edu.kw/centers/datalabku"
      },
      "notes": "Kuwait University data and AI lab with Arabic computational-linguistics work.",
      "metrics": null
    },
    {
      "id": "l2-ksu-native-and-non-native-arabic-speech",
      "name": "L2-KSU Native and Non-Native Arabic Speech",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "LDC",
      "modality": "speech",
      "tasks": [
        "speech",
        "pronunciation"
      ],
      "links": {
        "website": "https://catalog.ldc.upenn.edu/LDC2024S11"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "notes": "Native and non-native Arabic speech for pronunciation research (LDC2024S11).",
      "metrics": null
    },
    {
      "id": "labeah-ai",
      "name": "Labeah AI",
      "type": "org",
      "country": "SA",
      "org": "Labeah AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://labeah.ai"
      },
      "notes": "Saudi conversational AI platform for Arabic chatbots, callbots and voice agents for government and enterprise.",
      "metrics": null
    },
    {
      "id": "labess-chat",
      "name": "Labess Chat",
      "type": "llm",
      "country": "INTL",
      "org": "Linagora",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Linagora/Labess-chat-7b"
      },
      "size": "7B",
      "dialects": [
        "magh"
      ],
      "notes": "Tunisian Arabic, based on Jais architecture",
      "metrics": null
    },
    {
      "id": "lahajati",
      "name": "Lahajati",
      "type": "org",
      "country": "EG",
      "org": "Lahajati",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://lahajati.ai"
      },
      "notes": "Voice AI platform: Arabic TTS in 192+ dialects, speech-to-text, voice profiling and audio isolation.",
      "metrics": null
    },
    {
      "id": "lahgtna",
      "name": "Lahgtna",
      "type": "tts",
      "country": "INTL",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "tts",
        "dialect"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/oddadmix/lahgtna-chatterbox-demo"
      },
      "year": 2026,
      "notes": "Arabic dialect TTS model supporting Egyptian, Saudi, Moroccan and Iraqi dialects with full diacritics support.",
      "dialects": [
        "egy",
        "gulf",
        "magh",
        "iraqi"
      ],
      "metrics": null
    },
    {
      "id": "lahjawi",
      "name": "Lahjawi",
      "type": "llm",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/lahjawi"
      },
      "dialects": [
        "msa"
      ],
      "year": 2025,
      "notes": "Family of Arabic dialect translation models: dialect-to-dialect and dialect-to-MSA.",
      "metrics": null
    },
    {
      "id": "laravel-arabic-files",
      "name": "laravel-arabic-files",
      "type": "tool",
      "country": "INTL",
      "org": "awssat",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/awssat/laravel-arabic-files"
      },
      "year": 2024,
      "notes": "🇸🇦 Arabic Translations/Config for Laravel 📿",
      "metrics": null
    },
    {
      "id": "large-arabic-sentiment-analysis-resouces",
      "name": "large-arabic-sentiment-analysis-resouces",
      "type": "tool",
      "country": "INTL",
      "org": "hadyelsahar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/hadyelsahar/large-arabic-sentiment-analysis-resouces"
      },
      "year": 2018,
      "notes": "Large Arabic Resources For Sentiment Analysis",
      "metrics": null
    },
    {
      "id": "latis-off-line-handwriting-and-signature-database",
      "name": "LATIS OFF LINE HANDWRITING and SIGNATURE DATABASE",
      "type": "dataset",
      "country": "INTL",
      "org": "anouarbenkhalifa",
      "license": "cc-by-nc-nd-4.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/anouarbenkhalifa/latis-off-line-handwriting-and-signature-database"
      },
      "year": 2026,
      "notes": "Off-line signature and off-line arabic handwriting",
      "metrics": null
    },
    {
      "id": "layla-witheeb-jordanian-arabic-acoustic-dataset",
      "name": "Layla Witheeb: Jordanian Arabic Acoustic Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "phonetics"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/v9n7g7ns49"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "Phonetic data from 109 Jordanian undergraduate students: Jordanian Arabic acoustic dataset.",
      "metrics": null
    },
    {
      "id": "lfm2-5-1-2b-instruct-saudi-dialect",
      "name": "LFM2.5 1.2B Instruct Saudi Dialect",
      "type": "llm",
      "country": "INTL",
      "org": "AyoubChLin",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/AyoubChLin/LFM2.5-1.2B-Instruct-Saudi-Dialect"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1.2B",
      "on_device": true,
      "year": 2026,
      "notes": "LiquidAI LFM2.5-1.2B-Instruct fine-tuned on saudi-dialect-conversations for Saudi dialect chat.",
      "base_model": [
        "liquidai/lfm2.5-1.2b-instruct"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 11,
        "lastModified": "2026-02-19"
      }
    },
    {
      "id": "libigpt",
      "name": "LibiGPT",
      "type": "llm",
      "country": "LY",
      "org": "Smart Co for Technology Projects and AI",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://www.middleeastainews.com/p/first-libyan-national-large-language"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Libya's first national LLM (beta): LibiGPT-Base 7B and LibiGPT-Instruct 13B with a Libyan-dialect public chat app.",
      "metrics": null
    },
    {
      "id": "libraqm",
      "name": "libraqm",
      "type": "tool",
      "country": "OM",
      "org": "HOST-Oman",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "rtl-rendering"
      ],
      "links": {
        "github": "https://github.com/HOST-Oman/libraqm"
      },
      "notes": "Complex text-layout library (bidi and shaping) for Arabic-script rendering, maintained by HOST-Oman.",
      "metrics": null
    },
    {
      "id": "libtashkeel",
      "name": "libtashkeel",
      "type": "tool",
      "country": "INTL",
      "org": "Musharraf Omer",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/mush42/libtashkeel"
      },
      "year": 2023,
      "notes": "Arabic diacritization in Rust with Python, C++ and WASM bindings.",
      "metrics": null
    },
    {
      "id": "libya-telecom-companies-corpus",
      "name": "Libya Telecom companies corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Mansour Essgaer",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/Mansour-Essgaer/Libya-Telecom-companies-corpus"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Libyan dialect sentiment corpus of comments about telecom companies.",
      "metrics": null
    },
    {
      "id": "libyan-restaurants-sentiment-benchmark",
      "name": "Libyan Restaurants sentiment benchmark",
      "type": "dataset",
      "country": "INTL",
      "org": "Mansour Essgaer",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/Mansour-Essgaer/Libyan-Resturant"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Sentiment-analysis benchmark of Libyan-dialect restaurant reviews.",
      "metrics": null
    },
    {
      "id": "libyan-voice-ai",
      "name": "Libyan Voice AI",
      "type": "org",
      "country": "LY",
      "org": "Libyan Voice AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://libyan-voice.ly"
      },
      "notes": "Speech recognition models trained on Libyan Arabic, with Arabic summaries and English translation.",
      "metrics": null
    },
    {
      "id": "lighton-ai",
      "name": "LightOn AI",
      "type": "org",
      "country": "INTL",
      "org": "LightOn AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/lightonai"
      },
      "notes": "Arabic web data - ArabicWeb24 corpus (39B+ tokens)",
      "metrics": null
    },
    {
      "id": "lince-msa-da-lid-code-switching",
      "name": "LinCE - MSA-DA  (LID - Code Switching )",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Houston",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "code-switching"
      ],
      "links": {
        "website": "https://ritual.uh.edu/lince/datasets",
        "paper": "https://aclanthology.org/W16-5805.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "11,241 sentences",
      "year": 2016,
      "notes": "Twitter data and 9 entity types to establish a new dataset for code-switched NER benchmarks.",
      "metrics": null
    },
    {
      "id": "linto-asr-ar-tn-0-1",
      "name": "linto asr ar tn 0.1",
      "type": "asr",
      "country": "INTL",
      "org": "LINAGORA",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/linagora/linto-asr-ar-tn-0.1"
      },
      "year": 2025,
      "notes": "Arabic automatic speech recognition model.",
      "metrics": {
        "downloads": 0,
        "likes": 16,
        "lastModified": "2025-04-11"
      }
    },
    {
      "id": "lisan",
      "name": "Lisan",
      "type": "dataset",
      "country": "PS",
      "org": "Birzeit University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "dialect-id",
        "tokenization",
        "pos",
        "morphology"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/currasat/about-en.html",
        "paper": "https://doi.org/10.1109/AICCSA59173.2023.10479250"
      },
      "dialects": [
        "mixed",
        "iraqi",
        "magh",
        "sudanese",
        "yemeni"
      ],
      "size": "1,205,000 tokens",
      "year": 2023,
      "notes": "A morphologically-annotated Yemeni, Sudanese, Iraqi, and Libyan Arabic dialects Lisan corpora.",
      "metrics": null
    },
    {
      "id": "lisan-dialect-corpora",
      "name": "Lisan Dialect Corpora",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/resources/"
      },
      "dialects": [
        "yemeni",
        "iraqi",
        "magh",
        "sudanese"
      ],
      "year": 2023,
      "notes": "Yemeni, Iraqi, Libyan and Sudanese Arabic dialect corpora with morphological annotations.",
      "metrics": null
    },
    {
      "id": "lk-hadith-corpus",
      "name": "LK-Hadith-Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "ShathaTm",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith",
        "translation"
      ],
      "links": {
        "github": "https://github.com/ShathaTm/LK-Hadith-Corpus"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "LK Hadith Corpus: bilingual English-Arabic parallel corpus of Islamic Hadith from Leeds University and King Saud University.",
      "metrics": null
    },
    {
      "id": "llama-2-13b-chat-arabic-lora",
      "name": "Llama 2 13b chat arabic lora",
      "type": "llm",
      "country": "INTL",
      "org": "Icebear-AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Icebear-AI/Llama-2-13b-chat-arabic-lora"
      },
      "size": "13B",
      "on_device": false,
      "year": 2023,
      "tags": [
        "variants:1"
      ],
      "notes": "Experimental LoRA adapter fine-tuning Llama-2-13b-chat-hf for Arabic chat, by IceBear-AI.",
      "metrics": {
        "downloads": 0,
        "likes": 7,
        "lastModified": "2023-08-13"
      }
    },
    {
      "id": "llamar",
      "name": "LlamAr",
      "type": "llm",
      "country": "INTL",
      "org": "ClusterlabAi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.02147"
      },
      "size": "8B",
      "notes": "LLaMA 3 Arabic instruction-tuned on InstAr-500k",
      "metrics": null
    },
    {
      "id": "llm-arabic-instruct-gen",
      "name": "llm-arabic-instruct-gen",
      "type": "tool",
      "country": "EG",
      "org": "oddadmix",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning",
        "data-generation"
      ],
      "links": {
        "github": "https://github.com/Oddadmix/llm-arabic-instruct-gen"
      },
      "year": 2025,
      "notes": "Tool that generates instruction datasets from PDF or TXT files or Hugging Face datasets using LLMs.",
      "metrics": null
    },
    {
      "id": "lmaana-2-4",
      "name": "lmaana 2.4",
      "type": "asr",
      "country": "INTL",
      "org": "Lmaana",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/Lmaana/lmaana-2.4"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Moroccan Darija CTC speech recognition model built with fairseq2 on Omnilingual ASR; access gated.",
      "metrics": {
        "downloads": 0,
        "likes": 4,
        "lastModified": "2026-09-24"
      }
    },
    {
      "id": "lstarab100words",
      "name": "lstArab100words",
      "type": "asr",
      "country": "INTL",
      "org": "Megamind22",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "vision"
      ],
      "links": {
        "github": "https://github.com/Megamind22/lstArab100words"
      },
      "year": 2023,
      "notes": "Deep Visual Speech Recognition in arabic words",
      "metrics": null
    },
    {
      "id": "lucene-arabic-analyzer",
      "name": "Lucene Arabic Analyzer",
      "type": "tool",
      "country": "INTL",
      "org": "Msarhan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "search",
        "stemming"
      ],
      "links": {
        "github": "https://github.com/msarhan/lucene-arabic-analyzer"
      },
      "year": 2016,
      "notes": "Apache Lucene analyzer for Arabic with a root-based stemmer.",
      "metrics": null
    },
    {
      "id": "lucidya",
      "name": "Lucidya",
      "type": "org",
      "country": "SA",
      "org": "Lucidya",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.lucidya.com/"
      },
      "notes": "AI customer experience analytics - Arabic social listening, sentiment analysis",
      "metrics": null
    },
    {
      "id": "ly-abusive-language",
      "name": "LY-Abusive-Language",
      "type": "dataset",
      "country": "INTL",
      "org": "Mansour Essgaer",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/Mansour-Essgaer/LY-Abusive-Language"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Libyan dialect abusive-language dataset.",
      "metrics": null
    },
    {
      "id": "maad-multi-label-arabic-articles-dataset",
      "name": "MAAD : Multi-Label Arabic Articles Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification",
        "multi-label"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/hbfc9j8hj8"
      },
      "year": 2025,
      "notes": "MAAD: multi-label Arabic news articles for classification, text generation and summarization.",
      "metrics": null
    },
    {
      "id": "mac",
      "name": "MAC",
      "type": "dataset",
      "country": "INTL",
      "org": "Univ. Littoral Cote d’Opale Calais",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/LeMGarouani/MAC",
        "paper": "https://hal.science/hal-03670346v1"
      },
      "dialects": [
        "mixed",
        "msa",
        "magh"
      ],
      "size": "18,000 sentences",
      "year": 2024,
      "notes": "An open and free Moroccan Arabic corpus of 18000 sentiment-annotated tweets plus a 30k-word lexicon.",
      "metrics": null
    },
    {
      "id": "mac-ar-layout-for-win",
      "name": "Mac-Ar-Layout-for-Win",
      "type": "tool",
      "country": "INTL",
      "org": "Bishoy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/Bishoy/Mac-Ar-Layout-for-Win"
      },
      "year": 2025,
      "notes": "An arabic keyboard layout for use on windows users using apple keyboards which differs from its PC counterparts",
      "metrics": null
    },
    {
      "id": "machine-learning-models",
      "name": "machine-learning-models",
      "type": "tool",
      "country": "INTL",
      "org": "RiadKatby",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/RiadKatby/machine-learning-models"
      },
      "year": 2026,
      "notes": "An attempt to spread knowledge about machine learning and artificial intelligence to everyone using the Arabic language.",
      "metrics": null
    },
    {
      "id": "mada",
      "name": "mada",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aliftype/mada"
      },
      "year": 2026,
      "notes": "Mada (مدى) is a geometric, low-contrast Arabic typeface",
      "metrics": null
    },
    {
      "id": "madamira",
      "name": "MADAMIRA",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "website": "https://nyuad.nyu.edu/en/research/faculty-labs-and-projects/computational-approaches-to-modeling-language-lab/research/morphological-analysis-of-arabic.html"
      },
      "notes": "Morphological analysis, diacritization, POS tagging",
      "metrics": null
    },
    {
      "id": "madar-lexicon",
      "name": "MADAR Lexicon",
      "type": "dataset",
      "country": "QA",
      "org": "Carnegie Mellon University in Qatar",
      "license": "other",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "transliteration"
      ],
      "links": {
        "website": "https://docs.google.com/forms/d/e/1FAIpQLSe2LHYmHsxdkHPYHgcZDz25dTNbnygPkmClIaLd_fwud-XnTQ/viewform",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2018/pdf/351.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "47,000 tokens",
      "year": 2022,
      "notes": "The MADAR Lexicon is a collection of 1,042 concepts expressed in 25 city dialects totaling 47K entries.",
      "metrics": null
    },
    {
      "id": "madar-corpus",
      "name": "MADAR Parallel Corpus",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "translation"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/madar-parallel-corpus/"
      },
      "size": "25 cities",
      "year": 2018,
      "dialects": [
        "mixed"
      ],
      "notes": "Parallel sentences in 25 Arab city dialects plus MSA, English and French; basis of MADAR tasks.",
      "metrics": null
    },
    {
      "id": "madar-turk",
      "name": "MADAR-Turk",
      "type": "dataset",
      "country": "INTL",
      "org": "Sakarya University",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "http://resources.camel-lab.com/",
        "paper": "https://aclanthology.org/2023.mtsummit-research.22.pdf"
      },
      "dialects": [
        "lev"
      ],
      "size": "2,000 sentences",
      "year": 2023,
      "tags": [
        "multilingual"
      ],
      "notes": "First dataset for Syrian Arabic-Turkish machine translation sourced from MADAR.",
      "metrics": null
    },
    {
      "id": "maghreb-hof-a-large-scale-annotated-corpus-for-hate-and-offensive-lang",
      "name": "MAGHREB-HOF: A Large-Scale Annotated Corpus for Hate and Offensive Language Detection in Maghrebi Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "hate-speech",
        "offensive-language"
      ],
      "links": {
        "website": "https://zenodo.org/records/20597387"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "MAGHREB-HOF: annotated corpus for hate and offensive language detection in Maghrebi Arabic social media.",
      "metrics": null
    },
    {
      "id": "maglic",
      "name": "MAGLIC",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://www.isca-archive.org/odyssey_2024/jones24_odyssey.pdf"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Maghrebi Language Identification Corpus covering Libyan, Tunisian, Algerian and Moroccan speech (Odyssey 2024).",
      "metrics": null
    },
    {
      "id": "maha",
      "name": "Maha",
      "type": "tool",
      "country": "INTL",
      "org": "TRoboto",
      "license": "bsd-3-clause",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/TRoboto/Maha"
      },
      "notes": "Text processing library for Arabic text",
      "metrics": null
    },
    {
      "id": "majaz",
      "name": "Majaz",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Majaz"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Annotated corpus and toolkit for pragmatic analysis of similes and metaphors in Qur'anic Arabic.",
      "metrics": null
    },
    {
      "id": "mana",
      "name": "Mana",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Mana"
      },
      "year": 2025,
      "notes": "Mana: Thematic Corpus for Arabic Poetry with Multi-label Annotation",
      "metrics": null
    },
    {
      "id": "manara-3b",
      "name": "Manara-3B",
      "type": "llm",
      "country": "INTL",
      "org": "AgenThink AI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/agenthinkai/manara"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "year": 2026,
      "notes": "3B bilingual small language model for Kuwaiti banking, distilled from a Jais teacher for on-premise use.",
      "metrics": null
    },
    {
      "id": "manazir-ocr",
      "name": "Manazir OCR",
      "type": "ocr",
      "country": "EG",
      "org": "h9-tec",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/h9-tec/Manazir-OCR"
      },
      "year": 2025,
      "notes": "Arabic-first multi-model OCR pipeline for high-quality text extraction.",
      "metrics": null
    },
    {
      "id": "manhuw",
      "name": "manhuw",
      "type": "tool",
      "country": "INTL",
      "org": "musa11971",
      "license": "gpl-3.0",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/musa11971/manhuw"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Recognizing and identifying Quran reciters from audio recordings.",
      "metrics": null
    },
    {
      "id": "maqasid",
      "name": "Maqasid",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Maqasid"
      },
      "year": 2025,
      "notes": "Maqāṣid is a deep learning framework for multi-label thematic classification of Arabic poetry.",
      "metrics": null
    },
    {
      "id": "maqsam",
      "name": "Maqsam",
      "type": "org",
      "country": "JO",
      "org": "Maqsam",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://maqsam.com/"
      },
      "notes": "Arabic speech AI, call center AI - Arabic dialect STT, AI voice bots, surpasses Google/Microsoft",
      "metrics": null
    },
    {
      "id": "marasta",
      "name": "MARASTA",
      "type": "dataset",
      "country": "QA",
      "org": "Carnegie Mellon University in Qatar",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "stance-detection"
      ],
      "links": {
        "website": "https://zenodo.org/records/18536173",
        "paper": "https://aclanthology.org/2024.lrec-main.964.pdf"
      },
      "dialects": [
        "mixed",
        "egy",
        "magh",
        "gulf",
        "lev"
      ],
      "size": "4,657 sentences",
      "year": 2024,
      "notes": "A multi-dialectal Arabic cross-domain stance corpus",
      "metrics": null
    },
    {
      "id": "marbert",
      "name": "MARBERT",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/marbert"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Focused on Dialectal Arabic and MSA",
      "metrics": null
    },
    {
      "id": "markdown-arabic",
      "name": "markdown-arabic",
      "type": "tool",
      "country": "INTL",
      "org": "ahmadajmi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ahmadajmi/markdown-arabic"
      },
      "year": 2017,
      "notes": "Write Markdown in Arabic",
      "metrics": null
    },
    {
      "id": "marsa-multi-domain-arabic-resources-for-sentiment-analysis",
      "name": "MARSA: Multi-Domain Arabic Resources for Sentiment Analysis",
      "type": "dataset",
      "country": "SA",
      "org": "Imam Mohammad Ibn Saud Islamic University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=9576756"
      },
      "dialects": [
        "gulf"
      ],
      "size": "61,353 sentences",
      "year": 2021,
      "notes": "MARSA—the largest sentiment annotated corpus for Dialectal Arabic (DA) in the Gulf region, which consists of 61,353 manually labeled tweets that contain.",
      "metrics": null
    },
    {
      "id": "marsl",
      "name": "mArSL",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://hamzah-luqman.github.io/marsl/",
        "paper": "https://www.mdpi.com/2079-9292/10/14/1739/pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,748 videos",
      "year": 2021,
      "notes": "A multi-modality Arabic Sign Language dataset with manual and non-manual gestures.",
      "metrics": null
    },
    {
      "id": "masader",
      "name": "masader",
      "type": "dataset",
      "country": "SA",
      "org": "ARBML",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/ARBML/masader"
      },
      "notes": "Largest public catalogue of Arabic NLP datasets (600+)",
      "metrics": null
    },
    {
      "id": "masaq",
      "name": "MASAQ",
      "type": "dataset",
      "country": "JO",
      "org": "University of Jordan",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/9yvrzxktmr/2",
        "paper": "https://aclanthology.org/2025.clrel-1.7.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "131,930 tokens",
      "year": 2025,
      "notes": "The Morphologically-Annotated and Syntactically-Annotated Quran (MASAQ) dataset presents significant potential applications across domains.",
      "metrics": null
    },
    {
      "id": "masrkit",
      "name": "MasrKit",
      "type": "agent-skill",
      "country": "EG",
      "org": "asasemahmed",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/asasemahmed/MasrKit"
      },
      "year": 2026,
      "notes": "Open skills for coding agents to build products that feel Egyptian.",
      "metrics": null
    },
    {
      "id": "math-arabic-llama-3-2-3b-instruct",
      "name": "Math Arabic Llama 3.2 3B Instruct",
      "type": "llm",
      "country": "INTL",
      "org": "Jr23xd23",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "math",
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/Jr23xd23/Math_Arabic_Llama-3.2-3B-Instruct"
      },
      "base_model": [
        "meta-llama/llama-3.2-3b-instruct"
      ],
      "size": "3B",
      "on_device": true,
      "year": 2024,
      "notes": "Llama-3.2-3B-Instruct fine-tuned to solve math problems in Arabic.",
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2024-10-11"
      }
    },
    {
      "id": "mawdoo3",
      "name": "Mawdoo3",
      "type": "org",
      "country": "JO",
      "org": "Mawdoo3",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://mawdoo3.com/"
      },
      "notes": "Arabic AI & content, NLP toolkit - Arabic LLMs, largest Arabic website, Saudi expansion",
      "metrics": null
    },
    {
      "id": "ma-aks",
      "name": "MA’AKS",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/sabudalfa/ArabicTextSentimentSwap",
        "paper": "https://www.researchgate.net/publication/394575152_MA%27AKS_manually-curated_parallel_dataset_for_Arabic_text_sentiment_swap"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,000 sentences",
      "year": 2025,
      "notes": "A novel Arabic parallel dataset for sentiment style transfer.",
      "metrics": null
    },
    {
      "id": "mbzuai",
      "name": "MBZUAI",
      "type": "org",
      "country": "AE",
      "org": "MBZUAI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI"
      },
      "notes": "Multimodal and speech models - AIN, ArTST, ClArTTS",
      "metrics": null
    },
    {
      "id": "mbzuai-paris",
      "name": "MBZUAI-Paris",
      "type": "org",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/MBZUAI-Paris"
      },
      "notes": "Institute of Foundation Models behind Atlas-Chat and Nile-Chat",
      "metrics": null
    },
    {
      "id": "mcp-quran-ai-skills",
      "name": "mcp.quran.ai Skills and Docs",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Quran Foundation",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server",
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/quran/mcp.quran.ai"
      },
      "year": 2026,
      "notes": "Public skills, documentation and static assets for the hosted mcp.quran.ai Quran server.",
      "metrics": null
    },
    {
      "id": "mder-ma",
      "name": "MDER-MA",
      "type": "benchmark",
      "country": "MA",
      "org": "Sidi Mohamed Ben Abdellah University",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "evaluation",
        "emotion",
        "asr",
        "gender-id",
        "multimodal-evaluation"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/yzsw3ff6rn/1",
        "paper": "https://doi.org/10.1016/j.dib.2025.112005"
      },
      "dialects": [
        "magh"
      ],
      "size": "1,322 images",
      "year": 2025,
      "notes": "A multimodal Emotion Recognition Dataset for Moroccan Arabic, contain 5288 data items that express one of the four emotions: Happy, Sad, Angry, and Neutral.",
      "metrics": null
    },
    {
      "id": "mder-ma-multimodal-emotion-recognition-dataset-for-the-moroccan-arabic",
      "name": "MDER-MA: Multimodal Emotion Recognition Dataset for the Moroccan Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "emotion"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/yzsw3ff6rn"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "MDER-MA: 5,288 Moroccan Arabic items labeled happy, sad, angry or neutral across audio, text and spectrograms.",
      "metrics": null
    },
    {
      "id": "medqa-ma",
      "name": "MedQA-MA",
      "type": "dataset",
      "country": "MA",
      "org": "Sidi Mohamed Ben Abdellah University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/v6gs7nsy9z/1",
        "paper": "https://doi.org/10.1016/j.dib.2026.112537"
      },
      "dialects": [
        "magh"
      ],
      "size": "108,943 sentences",
      "year": 2025,
      "notes": "This dataset constitutes the first large-scale collection of medical question–answer pairs in Moroccan Arabic,",
      "metrics": null
    },
    {
      "id": "mentalqa",
      "name": "MentalQA",
      "type": "dataset",
      "country": "SA",
      "org": "Umm Al-Qura University",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/hasanhuz/MentalQA",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=10600466"
      },
      "dialects": [
        "mixed"
      ],
      "size": "1,000 sentences",
      "year": 2024,
      "notes": "Annotated Arabic corpus for questions and answers on mental healthcare.",
      "metrics": null
    },
    {
      "id": "meshakkelaty-ai",
      "name": "Meshakkelaty.ai",
      "type": "tool",
      "country": "INTL",
      "org": "Omar-Al-Sharif",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Omar-Al-Sharif/Meshakkelaty.ai"
      },
      "year": 2025,
      "notes": "A neural and statistical engine for accurately adding diacritics (Tashkeel) to Arabic text.",
      "metrics": null
    },
    {
      "id": "arabic-asr-challenge-2016",
      "name": "MGB-2 Arabic ASR Challenge 2016",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/qcri/ArabicASRChallenge2016"
      },
      "year": 2015,
      "notes": "Resources and recipes for the 2016 multi-genre broadcast Arabic ASR challenge.",
      "metrics": null
    },
    {
      "id": "microdialects",
      "name": "microdialects",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/microdialects"
      },
      "year": 2020,
      "notes": "Dataset and models for micro-dialect identification in diglossic and code-switched settings (EMNLP 2020), available on registration.",
      "metrics": null
    },
    {
      "id": "mikhak",
      "name": "Mikhak",
      "type": "ocr",
      "country": "INTL",
      "org": "aminabedi68",
      "license": "ofl-1.1",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/aminabedi68/Mikhak",
        "website": "https://aminabedi68.github.io/Mikhak/"
      },
      "year": 2025,
      "notes": "simple mono-width Arabic-Latin semi handwriting typeface",
      "metrics": null
    },
    {
      "id": "mirathqa",
      "name": "MirathQA",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/7jhycpbdpw/4",
        "paper": "https://doi.org/10.1016/j.dib.2026.112589"
      },
      "dialects": [
        "msa"
      ],
      "size": "1,394 sentences",
      "year": 2026,
      "notes": "A dataset for evaluating LLMs on Hanbali Islamic inheritance reasoning tasks.",
      "metrics": null
    },
    {
      "id": "miraya",
      "name": "Miraya",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "other",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Miraya"
      },
      "year": 2026,
      "notes": "A Critically Validated Computational Corpus for Arabic Poetry and Feminist Analysis",
      "metrics": null
    },
    {
      "id": "mishkal",
      "name": "Mishkal",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/linuxscout/mishkal"
      },
      "notes": "Rule-based diacritizer with dictionary lookups",
      "metrics": null
    },
    {
      "id": "mishkat",
      "name": "Mishkat",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Mishkat"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Mishkat is an AI-driven multilingual Quran project that delivers audio translations and explanations of Quranic verses.",
      "metrics": null
    },
    {
      "id": "misraj-ai",
      "name": "Misraj AI",
      "type": "org",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://misraj.ai/"
      },
      "notes": "Arabic-first AI ecosystem - Kawn LLM, Baseer OCR, Mutarjim, Workforces, SeamlessAPI",
      "metrics": null
    },
    {
      "id": "misraj-islamic-qa-classifier",
      "name": "Misraj Islamic QA Classifier",
      "type": "tool",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "legal"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/islamic-qa-classifier"
      },
      "year": 2026,
      "notes": "Hierarchical classifier routing Fatwa and Fiqh questions to legal categories.",
      "metrics": null
    },
    {
      "id": "misraj-islamic-quotation-extractor",
      "name": "Misraj Islamic Quotation Extractor",
      "type": "tool",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/islamic-quotation-extractor"
      },
      "year": 2026,
      "notes": "Information extraction identifying verbatim Quran, Hadith and book quotations.",
      "metrics": null
    },
    {
      "id": "misraj-open-arabic-corpus-35b",
      "name": "Misraj Open Arabic Corpus (35B)",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://misraj.ai/en/research/datasets/open-arabic-corpus-35b"
      },
      "year": 2025,
      "notes": "35B+ curated Arabic tokens released publicly for pretraining research.",
      "metrics": null
    },
    {
      "id": "misraj-dococr-benchmark",
      "name": "Misraj-DocOCR Benchmark",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/datasets/Misraj/Misraj-DocOCR-Benchmark"
      },
      "notes": "Arabic document OCR evaluation benchmark by Misraj AI",
      "metrics": null
    },
    {
      "id": "mistral-ai",
      "name": "Mistral AI",
      "type": "org",
      "country": "INTL",
      "org": "Mistral AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://mistral.ai/"
      },
      "notes": "Multilingual LLMs - Mistral Saba (Arabic-optimized)",
      "metrics": null
    },
    {
      "id": "mistral-saba",
      "name": "Mistral Saba",
      "type": "llm",
      "country": "INTL",
      "org": "Mistral",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://mistral.ai/news/mistral-saba"
      },
      "size": "24B",
      "notes": "Commercial API",
      "metrics": null
    },
    {
      "id": "mistral-arabic-ocr-test",
      "name": "Mistral-Arabic-OCR-test",
      "type": "ocr",
      "country": "INTL",
      "org": "Pythonation",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "vision"
      ],
      "links": {
        "github": "https://github.com/Pythonation/Mistral-Arabic-OCR-test"
      },
      "year": 2025,
      "notes": "A powerful Python toolkit using Mistral AI's OCR to accurately convert Arabic PDFs and images to text and editable documents.",
      "metrics": null
    },
    {
      "id": "mixat",
      "name": "Mixat",
      "type": "dataset",
      "country": "AE",
      "org": "MBZUAI",
      "license": "cc-by-nc-sa-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "github": "https://github.com/mbzuai-nlp/mixat",
        "paper": "https://aclanthology.org/2024.sigul-1.26.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "15 hours",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "15-hour Emirati Arabic-English code-switched speech corpus derived from two public podcasts featuring native Emirati speakers, with manual transcriptions.",
      "metrics": null
    },
    {
      "id": "mkhlab",
      "name": "mkhlab",
      "type": "tool",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/mkhlab"
      },
      "year": 2026,
      "notes": "🦅 مخلب — Arabic-first OpenClaw plugin. 14 Arabic AI skills, dialect-aware, culturally sensitive.",
      "metrics": null
    },
    {
      "id": "mmac",
      "name": "MMAC",
      "type": "dataset",
      "country": "INTL",
      "org": "Misr International University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "lexicon",
        "ocr"
      ],
      "links": {
        "website": "http://www.ashrafraouf.com/mmac",
        "paper": "https://link.springer.com/content/pdf/10.1007/s10032-010-0128-2.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,000,000 tokens",
      "year": 2010,
      "notes": "The multi-modal Arabic corpus contains 6 million Arabic words selected from various sources covering old Arabic, religious texts, traditional language.",
      "metrics": null
    },
    {
      "id": "moarlex",
      "name": "MoArLex",
      "type": "dataset",
      "country": "EG",
      "org": "Nile University",
      "license": "cc-by-3.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/Mohabyoussef09/MoArLex",
        "paper": "https://pdf.sciencedirectassets.com/280203/1-s2.0-S1877050918X00192/1-s2.0-S1877050918321665/main.pdf?X-Amz-Security-Token=IQoJb3JpZ2luX2VjEC4aCXVzLWVhc3QtMSJGMEQCIAW9EYFtY1ONOGkDXMEJfB01H4LyM9JA%2FFhtMdJO61crAiAJdDKIosWmXuPsCznRaE1uKw%2B0UqVt9QlTFOoPsTVGmyqzBQhXEAUaDDA1OTAwMzU0Njg2NSIMEjyc7r8vKtFBEOkZKpAFmGb%2BxNOfuy7pBi5%2FVxPfSs8PLHT5RlRToOL%2BOUM6oPGjIMoWyIbXrOQSu2MTIx2EgIfkyH%2FSGzCw%2FDMLMn7a0ruVmuSqXDbvQJo7igBktf0SnmVqNYC%2F5uznmg0wyzDjW5SIkSlHcl%2FOgU34bPVRzPxQeEBxKLX%2B5u0sXCONIyI8u1mz4%2FxT7RxKSqwwAUEv6RSAfV8UoyVBMtnOPmqnKjBaFNwB5S%2FIdrsB0Rf7pLOFmCocTOrmeyJYIMotRYGR1W7BQV5GwRsS%2BOO%2FobnXxOa7zxygOpLnL9%2Fk7uI0Svk03sJX347PPbDjPVyd1dHfiKJFc2%2B9BZvb4Gm7HNBtPW0EfxpkOiV4SHx8qUUkJtE8AgiLImkQQmcBgftYgr26F2b2Ve7AecOi9Dbp%2BMkJVYXl0hTyz7AT9n9bgZ1CLmoOz0pavqwGLYXm0JPZPEcjDnLfAMLYfGEDb%2Fz54tSteCbQch2Ga6Gg50YKZVnqnSklSIdYia4d%2Fz2HVioywI7SfZdUb6BrSYp9DVd8M7vYd4O3TioG9KiSfnwABA8WAM52cxtegb9jzKLuGEdmnnm9yjBkNyD9TLgMXNINRPM8Ic3%2B%2BdnhNiJg9raqhrEJ67RU1C6S74noQ6%2FpTblwGDh9FIUNZl1hluU%2F11DbtIfup0fooaimrng4wx0Vm3AUHJ5eScokeQbT7OCMcAcePCPVhe3KS7vCx1kUOvcUCwRuJZgJGKl6aRhyZdsG8Ylf491LrNPxmq2kTGRHf8vSKiBMtYBsElyRj7HhtRNUvVp%2BUvfdlYr7qKG27viZcvJi%2BxWxAvTFXxmCv5%2BEB2N4nJOacmIXKFcxLBYNFGBa0O1xW%2FHh2N5ThAU7oOOOzp5ezOcwkJv6tgY6sgHVfvgavguNAfVkorFJ1Vwz5db4Jo5%2BC6M%2B%2FjqYJ2srjmFlh3SoZM%2FUSA7wnfzvUMM6Cq7VLGGlmrHlLnXC1n8AdBNzVrpNi0exnD4po4DenlUhePy6Dgc5bfx5jB2bFccXRzw%2BO2rxoeBlp6jZPKJMCYdr74cAlZc%2BbdRz793x6V6Shg42Jp2QWqyVD2NfbQFDvj1MJsw5N2G2v39UwmJefk%2FDIn819MMj81oljvzuEW1V&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Date=20240909T055657Z&X-Amz-SignedHeaders=host&X-Amz-Expires=300&X-Amz-Credential=ASIAQ3PHCVTYVEXIFTOC%2F20240909%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Signature=bf7f76eaecad1ede8a46d8fb961b4439796762a1d125a40cfbb1c86fc9aeab6a&hash=a7d938a12e8726cb86d9966bc8c420919e9e4d1c59cf86c64117b307fa969f9c&host=68042c943591013ac2b2430a89b270f6af2c76d8dfd086a07176afe7c76c2c61&pii=S1877050918321665&tid=spdf-931269d2-6477-4ab8-9540-fafa769f5139&sid=0873a5bd51e572450a0af2f40180fda8c32fgxrqb&type=client&tsoh=d3d3LnNjaWVuY2VkaXJlY3QuY29t&ua=1f055a03575007555d&rr=8c04f02018cff0fc&cc=qa"
      },
      "dialects": [
        "mixed"
      ],
      "size": "36,775 sentences",
      "year": 2018,
      "notes": "MoArLex is a large-scale Arabic sentiment lexicon developed by expanding an existing seed lexicon, NileULex, using word embeddings from AraVec.",
      "metrics": null
    },
    {
      "id": "modern-standard-arabic-pronunciation-dictionary",
      "name": "Modern Standard Arabic Pronunciation Dictionary",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "asr",
        "pronunciation",
        "lexicon"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/msa-dictionary"
      },
      "dialects": [
        "msa"
      ],
      "year": 2014,
      "notes": "Pronunciation dictionary for Modern Standard Arabic ASR, used with the Kaldi GALE recipe.",
      "metrics": null
    },
    {
      "id": "moknah",
      "name": "Moknah",
      "type": "org",
      "country": "INTL",
      "org": "Moknah",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://moknah.io"
      },
      "notes": "Arabic TTS with 60+ voices across Gulf, Egyptian, Levantine and MSA, plus diacritics support, dubbing and an API.",
      "metrics": null
    },
    {
      "id": "monta-ai",
      "name": "Monta AI",
      "type": "org",
      "country": "EG",
      "org": "Monta AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://monta-ai.com/"
      },
      "notes": "Enterprise AI solutions - LLM & RAG-based Arabic business automation",
      "metrics": null
    },
    {
      "id": "moroccan-arabic-plurals-corpus",
      "name": "Moroccan Arabic Plurals Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://zenodo.org/records/14642330"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "1,166 singular-plural noun pairs in Moroccan Arabic (Darija), derived from DODa.",
      "metrics": null
    },
    {
      "id": "moroccan-darija-offensive-language-detection-dataset",
      "name": "Moroccan Darija Offensive Language Detection Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "a-ibrahimi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/a-ibrahimi/Moroccan-Darija-Offensive-Language-Detection-Dataset"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan Darija offensive-language detection dataset (MDOLDD).",
      "metrics": null
    },
    {
      "id": "moroccan-license-plates-ocr-dataset",
      "name": "Moroccan License Plates OCR Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2104.08244"
      },
      "year": 2021,
      "notes": "Labeled open dataset of Moroccan license plates mixing Arabic and Latin characters, for OCR.",
      "metrics": null
    },
    {
      "id": "moroccan-news-dataset-and-source-code-for-arabic-and-french-topic-dete",
      "name": "Moroccan News Dataset and Source Code for Arabic and French Topic Detection",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "topic-detection",
        "news"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/2496km8mw6"
      },
      "year": 2026,
      "notes": "News articles from 37 Moroccan press sources (21 Arabic, 16 French) collected via RSS for topic detection.",
      "metrics": null
    },
    {
      "id": "moroccan-darija-datasets",
      "name": "moroccan-darija-datasets",
      "type": "tool",
      "country": "INTL",
      "org": "nainiayoub",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "catalogue",
        "darija"
      ],
      "links": {
        "github": "https://github.com/nainiayoub/moroccan-darija-datasets",
        "website": "https://nainiayoub.github.io/nlp-arabic-glossary/"
      },
      "dialects": [
        "magh"
      ],
      "year": 2024,
      "notes": "Catalogue of 13 Moroccan Darija datasets grouped by name, source, region and size.",
      "metrics": null
    },
    {
      "id": "moroccan-nlp-zenodo",
      "name": "moroccan_nlp (Zenodo)",
      "type": "tool",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "code-switching",
        "darija"
      ],
      "links": {
        "website": "https://zenodo.org/records/21154423"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Linguistic resources and models for Moroccan Darija and Arabic: code-switching detection, Darija LM, DarijaBERT baseline classifier.",
      "metrics": null
    },
    {
      "id": "morocco-darija-open-source-ai-tools-ministry-of-digital-transition-x-m",
      "name": "Morocco Darija open-source AI tools (Ministry of Digital Transition x Mistral AI)",
      "type": "tool",
      "country": "MA",
      "org": "Morocco Ministry of Digital Transition",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id",
        "asr"
      ],
      "links": {
        "website": "https://iafrica.com/morocco-releases-first-open-source-darija-ai-tools-from-mistral-partnership/"
      },
      "year": 2026,
      "notes": "Government open-source Darija tools built with Mistral AI: a dialect identification model and a speech recognition model.",
      "metrics": null
    },
    {
      "id": "morocco-darija-sentence-embedding-v0-2",
      "name": "Morocco Darija Sentence Embedding v0.2",
      "type": "embedding",
      "country": "INTL",
      "org": "BounharAbdelaziz",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "sts"
      ],
      "links": {
        "hf": "https://huggingface.co/BounharAbdelaziz/Morocco-Darija-Sentence-Embedding-v0.2"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Moroccan Darija sentence embedding model fine-tuned from XLM-RoBERTa-Morocco on 637K pairs with Matryoshka and CoSENT losses.",
      "base_model": [
        "bounharabdelaziz/xlm-roberta-morocco"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 2,
        "lastModified": "2025-02-20"
      }
    },
    {
      "id": "morocco-darija-word-embedding",
      "name": "Morocco Darija Word Embedding",
      "type": "embedding",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/Morocco-Darija-Word-Embedding"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "Moroccan Darija word embeddings trained on the AL-Atlas Darija pretraining dataset; access gated.",
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2025-02-02"
      }
    },
    {
      "id": "morphems-without-borders",
      "name": "morphems_without_borders",
      "type": "dataset",
      "country": "SA",
      "org": "SDAIA",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/YaraAlakeel/morphems_without_borders",
        "paper": "https://arxiv.org/pdf/2603.15773v1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "130 documents",
      "year": 2026,
      "notes": "A datasets to control evaluation of Arabic derivational morphology across multiple tasks",
      "metrics": null
    },
    {
      "id": "mosl",
      "name": "MoSL",
      "type": "dataset",
      "country": "MA",
      "org": "Ibno Zohr University Agadir",
      "license": "cc-by-4.0",
      "modality": "multimodal",
      "tasks": [
        "sign-language"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/23phgyt3mt/1",
        "paper": "https://doi.org/10.1016/j.dib.2025.112395"
      },
      "dialects": [
        "magh"
      ],
      "size": "2,199 videos",
      "year": 2026,
      "notes": "The MoSL (Moroccan Sign Language) dataset consists of 2,199 videos representing isolated Moroccan Sign Language (MoSL) signs",
      "metrics": null
    },
    {
      "id": "mozn",
      "name": "Mozn",
      "type": "org",
      "country": "SA",
      "org": "Mozn",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.mozn.ai/"
      },
      "notes": "Enterprise AI, Arabic NLU - OSOS Arabic NLU platform, FOCAL compliance suite",
      "metrics": null
    },
    {
      "id": "msdd-misraj-structured-data-dump",
      "name": "MSDD - Misraj Structured Data Dump",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://misraj.ai/en/research/datasets/msdd"
      },
      "year": 2025,
      "notes": "Large-scale Arabic multimodal dataset built with the Wasm pipeline from Common Crawl.",
      "metrics": null
    },
    {
      "id": "mteb-arabic-leaderboard",
      "name": "MTEB Arabic Leaderboard",
      "type": "benchmark",
      "country": "INTL",
      "org": "MTEB",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/mteb/leaderboard"
      },
      "notes": "Massive Text Embedding Benchmark for Arabic",
      "metrics": null
    },
    {
      "id": "mu-een-maeen",
      "name": "Mu'een / Maeen",
      "type": "llm",
      "country": "OM",
      "org": "Oman MTCIT",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "chat",
        "summarization"
      ],
      "links": {
        "website": "https://initiatives.weforum.org/connected-future-initiative/case-study-details/the-omani-language-model-%E2%80%9Cmaeen%E2%80%9D/aJYTG000000058H4AQ"
      },
      "year": 2025,
      "notes": "Oman's national shared AI platform and LLM for government drafting, summarisation and Arabic content.",
      "metrics": null
    },
    {
      "id": "mubeen",
      "name": "mubeen",
      "type": "llm",
      "country": "SA",
      "org": "MASARAT",
      "license": "other",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/MASARAT-SA/mubeen"
      },
      "year": 2025,
      "notes": "Specialized Arabic LLM from MASARAT SA (Saudi Arabia) for Arabic linguistic and heritage tasks, free in beta.",
      "metrics": {
        "downloads": 0,
        "likes": 4,
        "lastModified": "2025-08-05"
      }
    },
    {
      "id": "mudd-misraj-unstructured-data-dump",
      "name": "MUDD - Misraj Unstructured Data Dump",
      "type": "dataset",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://misraj.ai/en/research/datasets/mudd"
      },
      "year": 2025,
      "notes": "Large-scale Arabic plain-text pretraining dataset translated from SlimPajama-627B.",
      "metrics": null
    },
    {
      "id": "muharaf",
      "name": "Muharaf",
      "type": "dataset",
      "country": "INTL",
      "org": "Muharaf Project",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/aamijar/muharaf"
      },
      "year": 2024,
      "dialects": [
        "classical"
      ],
      "notes": "Handwritten Arabic manuscript line images with transcripts for handwriting recognition.",
      "metrics": null
    },
    {
      "id": "mukhtasar",
      "name": "mukhtasar",
      "type": "tool",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/mukhtasar",
        "website": "https://moshe-ship.github.io/mukhtasar/"
      },
      "year": 2026,
      "notes": "مختصر — Arabic text summarizer CLI.",
      "metrics": null
    },
    {
      "id": "mulhem",
      "name": "Mulhem",
      "type": "llm",
      "country": "SA",
      "org": "SDAIA",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://sdaia.gov.sa/"
      },
      "notes": "Open-source Arabic-first LLM from Saudi Arabia",
      "metrics": null
    },
    {
      "id": "multi-dialect-multi-genre-informal-written-arabic-corpus",
      "name": "Multi-Dialect Multi-Genre Informal Written Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "JHU / UPenn (Cotterell, Callison-Burch)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://www.cis.upenn.edu/~ccb/publications/arabic-dialect-corpus-2.pdf"
      },
      "dialects": [
        "egy",
        "gulf",
        "iraqi",
        "lev",
        "magh"
      ],
      "year": 2014,
      "notes": "Multi-dialect, multi-genre corpus of informal written Arabic (LREC 2014).",
      "metrics": null
    },
    {
      "id": "multilevel-diacritizer",
      "name": "Multilevel Diacritizer",
      "type": "tool",
      "country": "DZ",
      "org": "Hamza5",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Hamza5/multilevel-diacritizer"
      },
      "year": 2019,
      "notes": "Extensible deep-learning Arabic diacritizer restoring several diacritic levels; Algeria.",
      "metrics": null
    },
    {
      "id": "multilingual-nlp-for-islamic-theology",
      "name": "Multilingual-NLP-for-Islamic-Theology",
      "type": "embedding",
      "country": "INTL",
      "org": "mobassir94",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/mobassir94/Multilingual-NLP-for-Islamic-Theology"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Cross Lingual Language models for making search engines for Holy Quran and Sahih Hadiths",
      "metrics": null
    },
    {
      "id": "multilingual-quran-hadith-islamic-content-database-api-hub",
      "name": "multilingual-quran-hadith-islamic-content-database-api-hub",
      "type": "tool",
      "country": "INTL",
      "org": "IslamHouse-API",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "api",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/IslamHouse-API/multilingual-quran-hadith-islamic-content-database-api-hub",
        "website": "https://islamhouse.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "IslamHouse API hub serving multilingual Quran, Hadith, books, articles, fatwas and verified translations from IslamHouse, QuranEnc and HadeethEnc.",
      "metrics": null
    },
    {
      "id": "munazarat-1-0",
      "name": "Munazarat 1.0",
      "type": "dataset",
      "country": "OM",
      "org": "Sultan Qaboos University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "asr",
        "dialogue"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.osact-1.3/",
        "github": "https://github.com/moh72y/Munazarat1.0/"
      },
      "dialects": [
        "msa"
      ],
      "year": 2024,
      "venue": "OSACT 2024",
      "notes": "This paper introduces the Corpus of Arabic Competitive Debates (Munazarat).",
      "metrics": null
    },
    {
      "id": "munsit",
      "name": "Munsit",
      "type": "org",
      "country": "AE",
      "org": "Munsit",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://munsit.com"
      },
      "notes": "Arabic speech recognition platform covering 25+ dialects, built and hosted in the UAE.",
      "metrics": null
    },
    {
      "id": "murabaa",
      "name": "Murabaa",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohammadia School of Engineers",
      "license": "cc-by-nc-nd-4.0",
      "modality": "text",
      "tasks": [
        "lemmatization",
        "stemming",
        "morphology"
      ],
      "links": {
        "github": "https://github.com/alelm-lab/Murabaa/",
        "paper": "https://aclanthology.org/2026.abjadnlp-1.47.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "717,380 tokens",
      "year": 2026,
      "notes": "Comprehensive platform for Arabic morphology resources that combines eight dictionaries",
      "metrics": null
    },
    {
      "id": "mushaf-layout",
      "name": "mushaf-layout",
      "type": "dataset",
      "country": "INTL",
      "org": "zonetecde",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/zonetecde/mushaf-layout"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "604-page Madani Mushaf dataset — JSON lines/words + QPC glyphs to render Hafs 'an 'Asim in the browser.",
      "metrics": null
    },
    {
      "id": "mushafalmadinahvector",
      "name": "MushafAlMadinahVector",
      "type": "dataset",
      "country": "SA",
      "org": "MushafAlMadinahVector",
      "license": "other",
      "modality": "vision",
      "tasks": [
        "quran",
        "vector-graphics"
      ],
      "links": {
        "github": "https://github.com/MushafAlMadinahVector/MushafAlMadinahVector"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Official vector edition of Mushaf Al-Madinah (Hafs) from King Fahd Glorious Quran Printing Complex, with full Uthmanic script and diacritics.",
      "metrics": null
    },
    {
      "id": "muslim-api",
      "name": "muslim-api",
      "type": "tool",
      "country": "INTL",
      "org": "otangid",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/otangid/muslim-api",
        "website": "https://muslim-api-frontend.vercel.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Rest api al-quran",
      "metrics": null
    },
    {
      "id": "mutarjim",
      "name": "Mutarjim",
      "type": "llm",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2505.17894"
      },
      "size": "1.5B",
      "notes": "Arabic-English translation, rivals GPT-4o mini",
      "on_device": true,
      "metrics": null
    },
    {
      "id": "nabarati",
      "name": "Nabarati",
      "type": "org",
      "country": "INTL",
      "org": "Nabarati",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://nabarati.ai"
      },
      "notes": "Arabic AI voice-over platform with 1,000+ voices and dialects, plus music generation and dubbing.",
      "metrics": null
    },
    {
      "id": "nabr",
      "name": "Nabr",
      "type": "asr",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran",
        "pronunciation"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/nabr"
      },
      "year": 2026,
      "notes": "Acoustic model for Quranic recitation: ASR and tajweed-aware mispronunciation detection.",
      "metrics": null
    },
    {
      "id": "nabra-syrian",
      "name": "Nabra",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2310.17315"
      },
      "year": 2023,
      "dialects": [
        "lev"
      ],
      "notes": "Syrian Arabic dialect corpus (~60K words) with morphological annotations (Syria/Palestine).",
      "metrics": null
    },
    {
      "id": "nabra-syrian-arabic-dialect-corpora",
      "name": "Nabra Syrian Arabic Dialect Corpora",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/resources/"
      },
      "year": 2023,
      "notes": "Syrian Arabic dialect corpora with morphological annotations.",
      "metrics": null
    },
    {
      "id": "nadi",
      "name": "nadi",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/nadi"
      },
      "year": 2021,
      "notes": "Nuanced Arabic Dialect Identification Shared Tasks (NADI) 2020 and 2021",
      "metrics": null
    },
    {
      "id": "nadi-2023-subtask-2",
      "name": "NADI 2023 Subtask 2",
      "type": "dataset",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://codalab.lisn.upsaclay.fr/competitions/14643",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.62.pdf"
      },
      "dialects": [
        "mixed",
        "egy",
        "gulf",
        "lev"
      ],
      "size": "2,400 sentences",
      "year": 2023,
      "notes": "Country-level Arabic dialect tweets from 18 countries for dialect identification; parallel Egyptian/Emirati/Jordanian/Palestinian→MSA sentences for MT.",
      "metrics": null
    },
    {
      "id": "nadi-2024",
      "name": "NADI 2024",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.04910"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "NADI 2024 - Fifth Nuanced Arabic Dialect Identification",
      "metrics": null
    },
    {
      "id": "nadi-2025",
      "name": "NADI 2025",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://nadi.dlnlp.ai/2025/"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "NADI 2025 - Multidialectal Arabic Speech Processing (8-way dialect + ASR)",
      "metrics": null
    },
    {
      "id": "nadi-shared-tasks",
      "name": "NADI Shared Tasks",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://nadi.dlnlp.ai/"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "NADI Shared Tasks - Ongoing series of Arabic DID shared tasks",
      "metrics": null
    },
    {
      "id": "nafis-normalized-arabic-fragments-for-inestimable-stemming",
      "name": "NAFIS: Normalized Arabic Fragments for Inestimable Stemming",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "stemming"
      ],
      "links": {
        "website": "https://catalog.elra.info/en-us/repository/browse/ELRA-W0127/"
      },
      "dialects": [
        "msa"
      ],
      "size": "154 tokens",
      "year": 2018,
      "notes": "ELRA corpus of 37 Arabic sentences (154 tokens) for evaluating stemming.",
      "metrics": null
    },
    {
      "id": "nahw",
      "name": "Nahw",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "evaluation",
        "qa"
      ],
      "links": {
        "github": "https://github.com/qcri/nahw-arabic-grammar-benchmark/",
        "paper": "https://aclanthology.org/2026.eacl-long.296.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "5,100 sentences",
      "year": 2026,
      "notes": "Nahw is a comprehensive benchmark for evaluating Arabic grammar understanding in large language models.",
      "metrics": null
    },
    {
      "id": "nahw-benchmark",
      "name": "Nahw benchmark",
      "type": "benchmark",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "grammar"
      ],
      "links": {
        "website": "https://www.fanar.qa/en/publications"
      },
      "year": 2026,
      "notes": "Comprehensive benchmark of Arabic grammar understanding, error detection, correction and explanation.",
      "metrics": null
    },
    {
      "id": "najdi-arabic-corpus",
      "name": "Najdi Arabic Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "King Saud University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "paper": "https://doi.org/10.1007/s10579-024-09749-5",
        "website": "http://faculty.ksu.edu.sa/en/ralhedayani/publication/411127"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2024,
      "notes": "Corpus for the underrepresented Najdi dialect, published in Language Resources and Evaluation (2024).",
      "metrics": null
    },
    {
      "id": "namaa-space",
      "name": "NAMAA-Space",
      "type": "org",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space"
      },
      "notes": "Arabic NLP models & dialect hub - Qari-OCR, EgypTalk-ASR, Masrawy translator, GLiNER Arabic",
      "metrics": null
    },
    {
      "id": "nanovate",
      "name": "Nanovate",
      "type": "org",
      "country": "EG",
      "org": "Nanovate",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.linkedin.com/company/nanovateai"
      },
      "year": 2025,
      "notes": "Cairo startup building Arabic AI products; raised a $1M pre-seed round in 2025.",
      "metrics": null
    },
    {
      "id": "narabizi-treebank",
      "name": "NArabizi treebank",
      "type": "dataset",
      "country": "INTL",
      "org": "Inria",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pos",
        "parsing",
        "translation"
      ],
      "links": {
        "website": "https://parsiti.github.io/NArabizi/",
        "paper": "https://aclanthology.org/2020.acl-main.107.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "1,500 sentences",
      "year": 2020,
      "notes": "Fully annotated in morpho-syntax and Universal Dependency syntax, with full translation at both the word and the sentence levels",
      "metrics": null
    },
    {
      "id": "nateq",
      "name": "Nateq",
      "type": "org",
      "country": "SA",
      "org": "Nateq",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://nateq.io"
      },
      "notes": "AI customer-service inbox for GCC businesses that answers in Gulf Arabic and English across WhatsApp, chat and calls.",
      "metrics": null
    },
    {
      "id": "native-tall-muttasiq-dot-com",
      "name": "NATIVE_TALL_muttasiq-dot-com",
      "type": "tool",
      "country": "INTL",
      "org": "GoodM4ven",
      "license": "other",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/GoodM4ven/NATIVE_TALL_muttasiq-dot-com",
        "website": "https://muttasiq.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "منصة تطبيقات تعين على الإسلام والالتزام باتساق ويسر بإذن الله",
      "metrics": null
    },
    {
      "id": "navid-ai",
      "name": "Navid AI",
      "type": "org",
      "country": "SA",
      "org": "Navid-AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "hf": "https://huggingface.co/Navid-AI"
      },
      "notes": "Saudi Arabia based group behind the Yehia-7B Arabic LLM, published on Hugging Face.",
      "metrics": null
    },
    {
      "id": "negation-and-speculation-in-arabic-review-nsar",
      "name": "Negation and Speculation in Arabic Review (NSAR)",
      "type": "dataset",
      "country": "EG",
      "org": "Ain Shams University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "retrieval",
        "classification"
      ],
      "links": {
        "github": "https://github.com/amahany/NSAR",
        "paper": "https://github.com/amahany/NSAR/blob/main/NSAR_Paper.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "3,011 sentences",
      "year": 2022,
      "notes": "The Negation and Speculation Arabic Review (NSAR) corpus consists of 3K randomly selected review sentences from three well-known and benchmarked Arabic.",
      "metrics": null
    },
    {
      "id": "nerdz",
      "name": "NERDz",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Bergen",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "github": "https://github.com/SamiaTouileb/NERDz",
        "paper": "https://aclanthology.org/2022.aacl-short.13.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "19,258 tokens",
      "year": 2022,
      "notes": "A preliminary annotated NER dataset for Algerian in Latin (NArabizi), Arabic, and code-switched scripts built atop the NArabizi treebank extension.",
      "metrics": null
    },
    {
      "id": "newstent",
      "name": "NewsTent",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "summarization"
      ],
      "links": {
        "paper": "https://openreview.net/pdf?id=Sbf9j9WcAkk"
      },
      "dialects": [
        "msa"
      ],
      "size": "8,443,484 documents",
      "year": 2021,
      "notes": "NewsTent: 8.4M articles with summaries from 22 newspapers of 19 Arab countries, 1999-2019.",
      "metrics": null
    },
    {
      "id": "nexdata-egyptian-call-center-telephony-speech",
      "name": "Nexdata Egyptian call-center telephony speech",
      "type": "dataset",
      "country": "INTL",
      "org": "Nexdata",
      "license": "commercial",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://data.nexdata.ai/products/nexdata-arabic-egypt-unscripted-call-center-telephony-spee-nexdata"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Commercial corpus of unscripted Egyptian Arabic call-center telephony speech.",
      "metrics": null
    },
    {
      "id": "nexdata-saudi-arabic-speech-849h",
      "name": "Nexdata Saudi Arabic speech (849h)",
      "type": "dataset",
      "country": "INTL",
      "org": "Nexdata",
      "license": "commercial",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/Nexdata-AI/849-Hours-Saudi-Arabic-Spontaneous-Speech-Data",
        "website": "https://www.nexdata.ai/datasets/speechrecog/1150"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Commercial 849-hour Saudi Arabic spontaneous speech dataset.",
      "metrics": null
    },
    {
      "id": "nexdata-uae-arabic-speech",
      "name": "Nexdata UAE Arabic speech",
      "type": "dataset",
      "country": "INTL",
      "org": "Nexdata",
      "license": "commercial",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/Nexdata-AI/1503-Hours-UAE-Arabic-Speech-Dataset"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Commercial UAE Arabic speech datasets of 1,503 and 749 hours.",
      "metrics": null
    },
    {
      "id": "nile-university",
      "name": "Nile University",
      "type": "org",
      "country": "EG",
      "org": "Nile University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://nu.edu.eg/"
      },
      "notes": "AI research, M.Sc. in AI co-designed with MIT/IBM",
      "metrics": null
    },
    {
      "id": "nile-chat",
      "name": "Nile-Chat",
      "type": "llm",
      "country": "AE",
      "org": "MBZUAI-Paris Lab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/MBZUAI-Paris/nile-chat"
      },
      "base_model": [
        "google/gemma-3-4b-pt",
        "google/gemma-3-12b-pt"
      ],
      "size": "4B-12B",
      "dialects": [
        "egy"
      ],
      "notes": "Egyptian Arabic and Arabizi scripts",
      "metrics": null
    },
    {
      "id": "niyyah",
      "name": "NIYYAH",
      "type": "benchmark",
      "country": "INTL",
      "org": "niyyah-research",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "classification"
      ],
      "links": {
        "github": "https://github.com/niyyah-research/niyyah"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Human-validated Saudi Arabic intent benchmark with 10,500 utterances.",
      "metrics": null
    },
    {
      "id": "nli",
      "name": "NLI",
      "type": "dataset",
      "country": "INTL",
      "org": "Fraunhofer IAIS",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nli"
      ],
      "links": {
        "github": "https://github.com/fraunhofer-iais/arabic_nlp/",
        "paper": "https://arxiv.org/pdf/2307.14666v1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "14,758 sentences",
      "year": 2023,
      "notes": "Dataset for NLI and Contradiction Detection in Arabic collected from three datasets.",
      "metrics": null
    },
    {
      "id": "nlp-arabic",
      "name": "nlp arabic",
      "type": "tool",
      "country": "INTL",
      "org": "othmanela",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "stopwords",
        "text-cleaning"
      ],
      "links": {
        "github": "https://github.com/othmanela/nlp_arabic"
      },
      "notes": "Ruby gem of Arabic NLP tools, including text cleaning with a hand-validated stop list built from over 900 articles.",
      "metrics": null
    },
    {
      "id": "nmatheg",
      "name": "nmatheg",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "training"
      ],
      "links": {
        "github": "https://github.com/ARBML/nmatheg"
      },
      "year": 2021,
      "notes": "Simple strategy for training and fine-tuning NLP models for Arabic.",
      "metrics": null
    },
    {
      "id": "nnlp-il",
      "name": "NNLP-IL",
      "type": "tool",
      "country": "INTL",
      "org": "NNLP-IL",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/NNLP-IL/NNLP-IL"
      },
      "year": 2022,
      "notes": "A national initiative for the creation of infrastructure, research and development of advanced capabilities for the advancement of the field of NLP.",
      "metrics": null
    },
    {
      "id": "noor",
      "name": "NOOR",
      "type": "llm",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://noor.tii.ae/"
      },
      "size": "10B",
      "notes": "World's largest Arabic NLP model at launch, GPT-3 architecture",
      "metrics": null
    },
    {
      "id": "noor-sharaye",
      "name": "Noor-Sharaye",
      "type": "dataset",
      "country": "INTL",
      "org": "Iran University of Science & Technology",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "morphology",
        "stemming",
        "pos",
        "lemmatization"
      ],
      "links": {
        "website": "https://zenodo.org/records/21481222",
        "paper": "https://reference-global.com/download/article/10.5334/johd.572.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "205,000 tokens",
      "year": 2026,
      "notes": "The dataset is a morphologically annotated Classical Arabic corpus containing word forms, stems, lemmas, roots, part-of-speech tags, and affix information.",
      "metrics": null
    },
    {
      "id": "notah",
      "name": "Notah",
      "type": "org",
      "country": "SA",
      "org": "Notah",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://notah.ai"
      },
      "notes": "Turns meetings, voice notes and Arabic-English conversations into transcripts, summaries and tasks.",
      "metrics": null
    },
    {
      "id": "nourvoice",
      "name": "NourVoice",
      "type": "org",
      "country": "SA",
      "org": "NourVoice",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://nourvoice.com"
      },
      "notes": "Arabic text-to-speech web tool with multiple voices and dialects, exporting MP3 voice-overs.",
      "metrics": null
    },
    {
      "id": "ntcc",
      "name": "NTCC",
      "type": "dataset",
      "country": "PS",
      "org": "Palestine Technical University-Kadoorie",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "github": "https://github.com/PsArNLP/Nakba",
        "paper": "https://aclanthology.org/2025.nakbanlp-1.6.pdf"
      },
      "dialects": [
        "lev"
      ],
      "size": "470 sentences",
      "year": 2025,
      "notes": "An annotated Arabic corpus of 470 sentences from Nakba short stories for topic classification.",
      "metrics": null
    },
    {
      "id": "nuha",
      "name": "Nuha",
      "type": "llm",
      "country": "SA",
      "org": "Elm (Saudi)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://elm.sa/en/about-us/why-elm/case-studies/Pages/Nuha-Bridging-Technology-and-Arabic-Culture.aspx"
      },
      "notes": "Multi-modal Arabic-first LLM for gov services, dialect-aware",
      "metrics": null
    },
    {
      "id": "oca-opinion-corpus-for-arabic",
      "name": "OCA: Opinion corpus for Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "niversity of Jaén",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "http://150.214.174.171:8059/investigacion/recursos/oca-corpus",
        "paper": "https://onlinelibrary.wiley.com/doi/full/10.1002/asi.21598"
      },
      "dialects": [
        "mixed"
      ],
      "size": "500 sentences",
      "year": 2011,
      "notes": "The corpus contains 500 movie reviews collected from different web pages and blogs in Arabic, 250 of them considered as positive reviews, and the other 250.",
      "metrics": null
    },
    {
      "id": "ocr-rs",
      "name": "ocr-rs",
      "type": "ocr",
      "country": "INTL",
      "org": "zibo-chen",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/zibo-chen/ocr-rs",
        "website": "https://crates.io/crates/ocr-rs"
      },
      "year": 2026,
      "notes": "高性能OCR识别库，支持上百种语言，提供命令行、图形界面及C API多种调用方式，使用便捷高效。 High-performance OCR library powered by PaddleOCR v4/v5/v6 with MNN backend.",
      "metrics": null
    },
    {
      "id": "ocrsmith",
      "name": "OCRSmith",
      "type": "ocr",
      "country": "MA",
      "org": "atlasia",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/atlasia-ma/OCRSmith"
      },
      "year": 2025,
      "notes": "Toolkit for synthetic OCR data generation used to train AtlasOCR.",
      "metrics": null
    },
    {
      "id": "octopus",
      "name": "Octopus",
      "type": "llm",
      "country": "INTL",
      "org": "Elyadata",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/ArabicSpeech/Octopus"
      },
      "year": 2025,
      "notes": "Bilingual Arabic-English audio LLM family for ASR, speech translation and Arabic dialect identification.",
      "base_model": [
        "deepseek-ai/deepseek-r1-distill-qwen-1.5b",
        "meta-llama/llama-3.2-1b"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 5,
        "lastModified": "2025-11-08"
      }
    },
    {
      "id": "octopus-ubc-nlp",
      "name": "octopus",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "generation"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/octopus"
      },
      "year": 2023,
      "notes": "Octopus is a neural machine generation toolkit for Arabic Natural Lnagauge Generation (NLG)",
      "metrics": null
    },
    {
      "id": "oman-gpt",
      "name": "Oman GPT",
      "type": "llm",
      "country": "OM",
      "org": "Oman MTCIT",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "website": "https://www.mtcit.gov.om/sectors?sector=artificial_intelligence"
      },
      "year": 2025,
      "notes": "Omani large language model implemented by MTCIT alongside Oman AI Studio.",
      "metrics": null
    },
    {
      "id": "oman-speech",
      "name": "OMAN-SPEECH",
      "type": "dataset",
      "country": "INTL",
      "org": "Various",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://aclanthology.org/2026.abjadnlp-1.31/"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "Sociolinguistically stratified, multi-layer annotated speech corpus of Omani Arabic dialects.",
      "metrics": null
    },
    {
      "id": "omani-parallel-corpus",
      "name": "Omani Parallel Corpus",
      "type": "dataset",
      "country": "OM",
      "org": "Sultan Qaboos University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/khoula-k/OmaniArabicTranslation",
        "paper": "https://aclanthology.org/2023.arabicnlp-1.24.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "2,595 sentences",
      "year": 2023,
      "notes": "Parallel Omani Arabic-English MT dataset",
      "metrics": null
    },
    {
      "id": "omartificial-intelligence-space",
      "name": "Omartificial-Intelligence-Space",
      "type": "org",
      "country": "SA",
      "org": "Omartificial-Intelligence-Space",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/Omartificial-Intelligence-Space"
      },
      "notes": "Arabic embedding models - GATE, Matryoshka embeddings",
      "metrics": null
    },
    {
      "id": "omcca",
      "name": "omcca",
      "type": "dataset",
      "country": "INTL",
      "org": "Isra University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id",
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/AhmedObaidi/omcca",
        "paper": "http://www.iaeng.org/publication/WCECS2016/WCECS2016_pp470-475.pdf"
      },
      "dialects": [
        "mixed",
        "gulf",
        "lev"
      ],
      "size": "28,576 sentences",
      "year": 2016,
      "notes": "Opinion Mining Corpus for Colloquial Variety of Arabic language",
      "metrics": null
    },
    {
      "id": "ontologyrag-q",
      "name": "OntologyRAG-Q",
      "type": "dataset",
      "country": "SA",
      "org": "KFUPM",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "rag"
      ],
      "links": {
        "github": "https://github.com/sazani/OntologyRAG-Q",
        "paper": "https://aclanthology.org/2025.emnlp-main.784.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "4,199 sentences",
      "year": 2025,
      "notes": "An annotated Tafsir ontology, a dataset of approximately 4,200 question-answer pairs, and a collection of 15 structured Tafsir books available in two formats",
      "metrics": null
    },
    {
      "id": "open-arabic-llm-leaderboard",
      "name": "Open Arabic LLM Leaderboard",
      "type": "benchmark",
      "country": "INTL",
      "org": "OALL",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/OALL/Open-Arabic-LLM-Leaderboard"
      },
      "notes": "Evaluation of Arabic LLMs across multiple benchmarks",
      "metrics": null
    },
    {
      "id": "open-hadith-data",
      "name": "Open Hadith Data",
      "type": "tool",
      "country": "INTL",
      "org": "mhashim6",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hadith",
        "resources"
      ],
      "links": {
        "github": "https://github.com/mhashim6/Open-Hadith-Data"
      },
      "year": 2017,
      "notes": "Open hadith databases covering nine books including the six canonical collections.",
      "metrics": null
    },
    {
      "id": "open-universal-arabic-asr-leaderboard",
      "name": "Open Universal Arabic ASR Leaderboard",
      "type": "benchmark",
      "country": "SA",
      "org": "Elm Research Center",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/elmresearchcenter/open_universal_arabic_asr_leaderboard"
      },
      "dialects": [
        "mixed"
      ],
      "notes": "Multi-dialectal Arabic speech recognition benchmark",
      "metrics": null
    },
    {
      "id": "open-source-arabic-tts-benchmark",
      "name": "Open-Source Arabic TTS Benchmark",
      "type": "dataset",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/silma-ai/opensource-arabic-tts-benchmark"
      },
      "notes": "SILMA AI's auditory assessment benchmark",
      "metrics": null
    },
    {
      "id": "openclaw-desktop",
      "name": "openclaw-desktop",
      "type": "tool",
      "country": "INTL",
      "org": "rshodoskar-star",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "github": "https://github.com/rshodoskar-star/openclaw-desktop",
        "website": "https://github.com/rshodoskar-star/openclaw-desktop/releases"
      },
      "year": 2026,
      "notes": "🖥️ A native desktop client for OpenClaw premium UI experience without the browser.",
      "metrics": null
    },
    {
      "id": "openhikmah-web",
      "name": "openhikmah-web",
      "type": "tool",
      "country": "INTL",
      "org": "OpenHikmah",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/OpenHikmah/openhikmah-web",
        "website": "https://openhikmah.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "An AI-powered Quran knowledge graph — place verses on a canvas and discover thematic, linguistic, and theological connections.",
      "metrics": null
    },
    {
      "id": "openiti-corpus",
      "name": "OpenITI Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "KITAB Project",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining",
        "ocr"
      ],
      "links": {
        "github": "https://github.com/OpenITI/RELEASE",
        "website": "https://kitab-project.org/"
      },
      "year": 2019,
      "dialects": [
        "classical"
      ],
      "notes": "Open Islamicate Texts Initiative corpus of premodern Arabic texts, billion-word scale.",
      "metrics": null
    },
    {
      "id": "openiti-proc",
      "name": "OpenITI-proc",
      "type": "dataset",
      "country": "INTL",
      "org": "MIT Computer Science and Artificial Intelligence Laboratory",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://zenodo.org/record/2535593#.YWh7FS8RozU",
        "paper": "https://arxiv.org/pdf/1809.03891.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,500,000,000 tokens",
      "year": 2019,
      "notes": "A linguistically annotated version of the OpenITI corpus, with annotations for lemmas, POS tags, parse trees, and morphological segmentation",
      "metrics": null
    },
    {
      "id": "osian",
      "name": "OSIAN",
      "type": "dataset",
      "country": "INTL",
      "org": "Mohamed First University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "paper": "https://aclanthology.org/W19-4619.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "3,500,000 documents",
      "year": 2019,
      "notes": "The corpus data was collected from international Arabic news websites,",
      "metrics": null
    },
    {
      "id": "osman-arabic-text-readability",
      "name": "Osman Arabic Text Readability",
      "type": "tool",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "readability"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/osmanreadability/"
      },
      "notes": "An open-source tool for measuring Arabic text readability, allowing users to calculate readability for Arabic text with or without diacritics",
      "metrics": null
    },
    {
      "id": "othman",
      "name": "othman",
      "type": "tool",
      "country": "INTL",
      "org": "ojuba-org",
      "license": "other",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/ojuba-org/othman"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Quran browser and search engine",
      "metrics": null
    },
    {
      "id": "paddleocr",
      "name": "PaddleOCR",
      "type": "tool",
      "country": "INTL",
      "org": "PaddlePaddle",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/PaddlePaddle/PaddleOCR"
      },
      "notes": "High-performance multilingual OCR with Arabic support",
      "metrics": null
    },
    {
      "id": "palmx-2025",
      "name": "palmx_2025",
      "type": "benchmark",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/palmx_2025"
      },
      "year": 2025,
      "notes": "This repository contains the evaluation code and data for the PalmX 2025 Shared Task on Benchmarking LLMs for Arabic and Islamic Culture.",
      "metrics": null
    },
    {
      "id": "pan-arabic-intrinsic-plagiarism-detection-shared-task-corpus",
      "name": "PAN Arabic Intrinsic Plagiarism Detection Shared Task Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "¹MISC Lab. Constantine",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "plagiarism-detection"
      ],
      "links": {
        "website": "https://zenodo.org/record/6609196#.YqTYvNrMLIV",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-642-40802-1_6"
      },
      "dialects": [
        "msa"
      ],
      "size": "2,048 documents",
      "year": 2015,
      "notes": "Each part of the corpus (training and test) consists mainly of 2 datasets: textual files and XML files.",
      "metrics": null
    },
    {
      "id": "peach",
      "name": "PEACH",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/5k6yrrhng7/3",
        "paper": "https://www.euppublishing.com/doi/10.3366/cor.2024.0320"
      },
      "dialects": [
        "msa"
      ],
      "size": "51,671 sentences",
      "year": 2024,
      "tags": [
        "multilingual"
      ],
      "notes": "A sentence-aligned parallel English–Arabic corpus of healthcare texts encompassing patient information leaflets and educational materials.",
      "metrics": null
    },
    {
      "id": "peach-a-sentence-aligned-parallel-english-arabic-corpus-for-healthcare-unknown",
      "name": "PEACH: A Sentence-Aligned Parallel English-Arabic Corpus for Healthcare",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "medical"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/5k6yrrhng7"
      },
      "year": 2024,
      "notes": "PEACH: sentence-aligned parallel English-Arabic healthcare corpus.",
      "metrics": null
    },
    {
      "id": "phonbank-arabic-kuwaiti-corpus",
      "name": "PhonBank Arabic Kuwaiti Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Shaikh Salem Al-Ali Centre",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "child-speech",
        "phonology"
      ],
      "links": {
        "website": "https://phon.talkbank.org/access/Other/Arabic/Kuwaiti.html",
        "paper": "https://core.ac.uk/download/pdf/153779285.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "35 hours",
      "year": 2015,
      "notes": "Speech data from 70 Kuwaiti children sampled from the general Kuwaiti population, hosted on PhonBank.",
      "metrics": null
    },
    {
      "id": "pipeline-diacritizer",
      "name": "Pipeline Diacritizer",
      "type": "tool",
      "country": "DZ",
      "org": "Hamza5",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Hamza5/Pipeline-diacritizer"
      },
      "year": 2020,
      "notes": "Multi-level Arabic diacritics restoration tool; Algeria.",
      "metrics": null
    },
    {
      "id": "piper-tts",
      "name": "Piper TTS",
      "type": "tts",
      "country": "INTL",
      "org": "Rhasspy",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/rhasspy/piper"
      },
      "on_device": true,
      "notes": "Fast local neural TTS, Arabic voices available",
      "metrics": null
    },
    {
      "id": "placequran",
      "name": "placequran",
      "type": "tool",
      "country": "INTL",
      "org": "faizshukri",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/faizshukri/placequran",
        "website": "http://placequran.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Quran API that return image based on url",
      "metrics": null
    },
    {
      "id": "political-arabic-article-dataset",
      "name": "Political Arabic Article Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/spvbf5bgjs"
      },
      "year": 2020,
      "notes": "PAAD: political Arabic articles from newspapers, blogs and social networks for text classification.",
      "metrics": null
    },
    {
      "id": "pragmatapro",
      "name": "pragmatapro",
      "type": "tool",
      "country": "INTL",
      "org": "fabrizioschiavi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/fabrizioschiavi/pragmatapro",
        "website": "https://fsd.it/shop/fonts/pragmatapro/"
      },
      "year": 2026,
      "notes": "PragmataPro font is designed to help pros to work better",
      "metrics": null
    },
    {
      "id": "prepocressor",
      "name": "PrepOCRessor",
      "type": "ocr",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://alt.qcri.org/tools/prepocressor"
      },
      "notes": "Tool for preprocessing document images before Arabic OCR, pipeline of image operations.",
      "metrics": null
    },
    {
      "id": "presight-ai",
      "name": "Presight AI",
      "type": "org",
      "country": "AE",
      "org": "Presight AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.presight.ai"
      },
      "notes": "G42 big-data analytics firm; government AI products including Arabic NLP.",
      "metrics": null
    },
    {
      "id": "prince-sultan-university-riotu-lab",
      "name": "Prince Sultan University (RIOTU Lab)",
      "type": "org",
      "country": "SA",
      "org": "Prince Sultan University (RIOTU Lab)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/riotu-lab"
      },
      "notes": "ArabianGPT, Arabic IoT/robotics AI",
      "metrics": null
    },
    {
      "id": "prince-sultan-university-riotu",
      "name": "Prince Sultan University (RIOTU)",
      "type": "org",
      "country": "SA",
      "org": "Prince Sultan University (RIOTU)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/riotu-lab"
      },
      "notes": "Arabic language models - ArabianGPT, Arabic IoT AI",
      "metrics": null
    },
    {
      "id": "printed-arabic-base-model-trained-on-the-openiti-corpus",
      "name": "Printed Arabic Base Model Trained on the OpenITI Corpus",
      "type": "ocr",
      "country": "INTL",
      "org": "unknown",
      "license": "cc0-1.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://zenodo.org/records/7050296"
      },
      "year": 2022,
      "notes": "Kraken text recognition model for printed Arabic trained on the OpenITI corpus.",
      "metrics": null
    },
    {
      "id": "process-arabic-text",
      "name": "process-arabic-text",
      "type": "tool",
      "country": "PS",
      "org": "Motaz Saad",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "github": "https://github.com/motazsaad/process-arabic-text"
      },
      "year": 2017,
      "notes": "Pre-processes Arabic text by removing diacritics, punctuation and repeated characters; Palestine.",
      "metrics": null
    },
    {
      "id": "pyarabic",
      "name": "PyArabic",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/linuxscout/pyarabic"
      },
      "notes": "Python package for Arabic text manipulation",
      "metrics": null
    },
    {
      "id": "pyfarasa",
      "name": "pyFarasa",
      "type": "tool",
      "country": "INTL",
      "org": "OpenITI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/OpenITI/pyFarasa"
      },
      "dialects": [
        "msa"
      ],
      "notes": "Python wrapper for the Farasa Arabic segmenter and POS tagger.",
      "metrics": null
    },
    {
      "id": "pyquran",
      "name": "PyQuran",
      "type": "tool",
      "country": "INTL",
      "org": "hci-lab",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/hci-lab/PyQuran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "PyQuran: The Python package for Quranic Analysis https://hci-lab.github.io/PyQuran-Private",
      "metrics": null
    },
    {
      "id": "python-arabic-reshaper",
      "name": "python-arabic-reshaper",
      "type": "tool",
      "country": "INTL",
      "org": "Abdullah Diab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "rtl-rendering"
      ],
      "links": {
        "github": "https://github.com/mpcabd/python-arabic-reshaper"
      },
      "year": 2012,
      "notes": "Reshapes Arabic text so it renders correctly in apps without Arabic shaping support.",
      "metrics": null
    },
    {
      "id": "python-bidi",
      "name": "python-bidi",
      "type": "tool",
      "country": "INTL",
      "org": "Meir Kriheli",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "rtl-rendering"
      ],
      "links": {
        "github": "https://github.com/MeirKriheli/python-bidi"
      },
      "year": 2010,
      "notes": "Pure-Python Unicode bidirectional algorithm, commonly paired with arabic-reshaper.",
      "metrics": null
    },
    {
      "id": "qac-qatari-arabic-corpus",
      "name": "QAC: Qatari Arabic Corpus",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "http://www.isle.illinois.edu/dialect/QAC/index.html",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2014/pdf/430_Paper.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "18 hours",
      "year": 2014,
      "notes": "A wide-band speech corpus has been collected and transcribed from several Qatari TV series and talk-show programs.",
      "metrics": null
    },
    {
      "id": "qaes",
      "name": "QAES",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "essay-scoring"
      ],
      "links": {
        "paper": "https://aclanthology.org/2024.arabicnlp-1.28/",
        "website": "https://www.kaggle.com/c/asap-aes"
      },
      "year": 2024,
      "venue": "ArabicNLP 2024",
      "notes": "We introduce QAES, the first publicly available trait-specific annotations for Arabic AES, built on the Qatari Corpus of Argumentative Writing (QCAW).",
      "metrics": null
    },
    {
      "id": "qahiri",
      "name": "qahiri",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aliftype/qahiri",
        "website": "https://aliftype.com/qahiri/"
      },
      "year": 2026,
      "notes": "Qahiri (قاهري) is a manuscript Kufic typeface",
      "metrics": null
    },
    {
      "id": "qalam",
      "name": "Qalam",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "multimodal"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2407.13559"
      },
      "notes": "Arabic OCR/HWR multimodal LLM, 0.80% WER",
      "metrics": null
    },
    {
      "id": "qalam-pdf",
      "name": "Qalam",
      "type": "tool",
      "country": "SA",
      "org": "Misraj AI",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "ocr",
        "pdf-extraction"
      ],
      "links": {
        "github": "https://github.com/misraj-ai/qalam"
      },
      "year": 2026,
      "notes": "Arabic PDF text extraction that preserves logical word order and avoids corrupting corpora.",
      "metrics": null
    },
    {
      "id": "qalb-mt",
      "name": "QALB-MT",
      "type": "dataset",
      "country": "QA",
      "org": "Carnegie Mellon University in Qatar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "http://nlp.qatar.cmu.edu/qalb/",
        "paper": "https://aclanthology.org/L16-1295.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "100,000 tokens",
      "year": 2016,
      "notes": "A 100K-word human post-edited English-to-Arabic machine-translation corpus created from Wikinews articles.",
      "metrics": null
    },
    {
      "id": "qalsadi",
      "name": "Qalsadi",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/linuxscout/qalsadi"
      },
      "notes": "Arabic morphological analyzer and lemmatizer",
      "metrics": null
    },
    {
      "id": "qari-ocr",
      "name": "QARI-OCR",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/NAMAA-Space/qari-ocr-a-high-accuracy-model-for-arabic-optical-character"
      },
      "notes": "Arabic OCR models by NAMAA-Space built on Qwen2 VL 2B and fine-tuned on an Arabic OCR dataset, released as versions v0.1 to v0.4.",
      "metrics": null
    },
    {
      "id": "qari-ocr-v0-2",
      "name": "QARI-OCR v0.2",
      "type": "ocr",
      "country": "SA",
      "org": "NAMAA-Space",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "hf": "https://huggingface.co/NAMAA-Space/Qari-OCR-0.2-VL-2B-Instruct"
      },
      "notes": "Updated Qwen2-VL 2B for Arabic OCR; WER 0.160, CER 0.061",
      "metrics": null
    },
    {
      "id": "qatar-center-for-artificial-intelligence",
      "name": "Qatar Center for Artificial Intelligence",
      "type": "org",
      "country": "QA",
      "org": "Qatar Center for Artificial Intelligence",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://qcai.qcri.org"
      },
      "notes": "QCRI centre advancing AI research from foundations and systems to precision health and AI for social good.",
      "metrics": null
    },
    {
      "id": "qatar-university",
      "name": "Qatar University",
      "type": "org",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.qu.edu.qa"
      },
      "notes": "Qatar University NLP group; bigIR lab released ArCOV-19 and Arabic QA resources.",
      "metrics": null
    },
    {
      "id": "qatari-heritage-corpus",
      "name": "Qatari heritage corpus",
      "type": "dataset",
      "country": "QA",
      "org": "Hamad Bin Khalifa University",
      "license": "cdla-permissive-1.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "website": "https://data.world/saraalmulla/qatari-heritage-expressions",
        "paper": "https://aclanthology.org/2020.osact-1.4.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "1,000 sentences",
      "year": 2020,
      "notes": "Qatari heritage expressions dataset with translations",
      "metrics": null
    },
    {
      "id": "qatip",
      "name": "QATIP",
      "type": "ocr",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/"
      },
      "notes": "Continuous text recognition that works best for entire pages of historic documents with challenging script.",
      "metrics": null
    },
    {
      "id": "qats",
      "name": "QATS",
      "type": "asr",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://alt.qcri.org/demos"
      },
      "notes": "Improved Arabic transcription with AI (QCRI Arabic speech transcription system demo).",
      "metrics": null
    },
    {
      "id": "qawafi",
      "name": "qawafi",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ARBML/qawafi"
      },
      "notes": "Arabic poetry analysis",
      "metrics": null
    },
    {
      "id": "qcri",
      "name": "QCRI",
      "type": "org",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/QCRI"
      },
      "notes": "Arabic LLMs, text processing - Fanar LLMs, AraDiCE, Farasa",
      "metrics": null
    },
    {
      "id": "qcri-arabic-normalizer",
      "name": "QCRI Arabic Normalizer",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "spell-checking"
      ],
      "links": {
        "website": "https://alt.qcri.org/tools/arabic-normalizer"
      },
      "notes": "Script normalising Arabic punctuation, digits, diacritics and spelling for consistent MT evaluation.",
      "metrics": null
    },
    {
      "id": "qcri-arabic-pos-tagger-library-qatara",
      "name": "QCRI Arabic POS Tagger Library (QATARA)",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization",
        "ner",
        "pos"
      ],
      "links": {
        "website": "https://alt.qcri.org/tools/apost"
      },
      "year": 2014,
      "notes": "CRF-based Arabic tokenizer, POS tagger, NER, gender/number tagger and diacritizer library.",
      "metrics": null
    },
    {
      "id": "qcri-dialectal-arabic-resources",
      "name": "QCRI Dialectal Arabic Resources",
      "type": "dataset",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "pos"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/da_resources/"
      },
      "year": 2017,
      "notes": "Dialectal Arabic segmentation and POS-tagging datasets compiled at QCRI for research.",
      "metrics": null
    },
    {
      "id": "qcri-kaldi-gale-arabic-recipe",
      "name": "QCRI Kaldi GALE Arabic Recipe",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://alt.qcri.org/resources/gale-recipe"
      },
      "year": 2014,
      "notes": "Files for building Arabic ASR using the GALE LDC database and the Kaldi toolkit.",
      "metrics": null
    },
    {
      "id": "qf-api-docs",
      "name": "qf-api-docs",
      "type": "tool",
      "country": "INTL",
      "org": "Quran Foundation",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/quran/qf-api-docs",
        "website": "https://api-docs.quran.foundation/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran Foundation API Documentation Portal using Docusaurus",
      "metrics": null
    },
    {
      "id": "qimma-leaderboard",
      "name": "qimma-leaderboard",
      "type": "benchmark",
      "country": "AE",
      "org": "TII (UAE)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/tiiuae/QIMMA-leaderboard"
      },
      "notes": "QIMMA ⛰ قِمّة A quality-first Arabic LLM leaderboard that validates benchmarks before evaluating models. 📖 Blog Post • 🏆 Leaderboard • 📄 Paper ⛰️ QIMMA.",
      "metrics": null
    },
    {
      "id": "qirtaas-js",
      "name": "qirtaas-js",
      "type": "tool",
      "country": "INTL",
      "org": "alrimaal",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/alrimaal/qirtaas-js",
        "website": "https://qirtaas.io/developers"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Qirtaas SDK — embeddable rich-text editor for Islamic writing (Quran, hadith, Arabic RTL). @qirtaas/core, @qirtaas/vue, @qirtaas/react.",
      "metrics": null
    },
    {
      "id": "qnl-arabic-ocr-corpus-v-2",
      "name": "QNL Arabic OCR Corpus v.2",
      "type": "dataset",
      "country": "QA",
      "org": "Qatar National Library",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://manara.qnl.qa/articles/dataset/Arabic_OCR_Corpus_2_894_items_from_QNL_Collection_/26984785",
        "paper": "https://doi.org/10.57945/manara.26984785"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "year": 2024,
      "notes": "2,894 digitised items from the Qatar National Library collection for Arabic OCR, about 1 GB.",
      "metrics": null
    },
    {
      "id": "quran-qzaidi",
      "name": "quran",
      "type": "tool",
      "country": "INTL",
      "org": "qzaidi",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/qzaidi/quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "node,websql and javascript API for Holy quran",
      "metrics": null
    },
    {
      "id": "quran-fawazahmed0",
      "name": "quran (fawazahmed0)",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/quran",
        "website": "https://fawazahmed0.github.io/quran/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Read Quran in 90+ Languages",
      "metrics": null
    },
    {
      "id": "quran-quran-bundle",
      "name": "quran (Quran-Bundle)",
      "type": "tool",
      "country": "INTL",
      "org": "Quran-Bundle",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/Quran-Bundle/quran"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Simplifying The Holy Quran Typesetting in XeLaTeX/LuaLaTeX.",
      "metrics": null
    },
    {
      "id": "quran-hadith-search",
      "name": "Quran and Hadith Search",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "search",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/quran-hadith-search"
      },
      "year": 2022,
      "notes": "Search engine over Quran and hadith texts.",
      "metrics": null
    },
    {
      "id": "quran-mcp-server-djalal",
      "name": "Quran MCP Server (djalal)",
      "type": "agent-skill",
      "country": "INTL",
      "org": "djalal",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/djalal/quran-mcp-server"
      },
      "year": 2026,
      "notes": "MCP server wrapping the Quran.com API for verse search, translation and tafsir.",
      "metrics": null
    },
    {
      "id": "quran-search-engine-mcp",
      "name": "Quran Search Engine MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "adelpro",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/adelpro/quran-search-engine-mcp"
      },
      "year": 2026,
      "notes": "MCP server giving AI clients fast, citation-grounded Quran search to reduce hallucinated verses.",
      "metrics": null
    },
    {
      "id": "quran-ai-transcribing",
      "name": "quran-ai-transcribing",
      "type": "tool",
      "country": "INTL",
      "org": "sayedmahmoud266",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/sayedmahmoud266/quran-ai-transcribing"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Quran AI transcribing with accurate Ayah, Surah Matching with audio timestamps",
      "metrics": null
    },
    {
      "id": "quran-align",
      "name": "quran-align",
      "type": "asr",
      "country": "INTL",
      "org": "cpfair",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "github": "https://github.com/cpfair/quran-align"
      },
      "dialects": [
        "classical"
      ],
      "year": 2017,
      "notes": "Word-accurate timestamps for Qur'anic audio.",
      "metrics": null
    },
    {
      "id": "quran-api",
      "name": "quran-api",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "unlicense",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/quran-api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Free Quran API Service with 90+ different languages and 400+ translations",
      "metrics": null
    },
    {
      "id": "quran-api-saikothasan",
      "name": "quran-api (saikothasan)",
      "type": "tool",
      "country": "INTL",
      "org": "saikothasan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/saikothasan/quran-api",
        "website": "https://alquran-api.pages.dev/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Al-Quran API - Multilingual Quran API with Translations",
      "metrics": null
    },
    {
      "id": "quran-api-the-quran-project",
      "name": "Quran-API (The-Quran-Project)",
      "type": "tool",
      "country": "INTL",
      "org": "The-Quran-Project",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/The-Quran-Project/Quran-API",
        "website": "https://quranapi.pages.dev"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "An API for the Holy Quran with no rate limit.",
      "metrics": null
    },
    {
      "id": "quran-api-with-php-codeigniter",
      "name": "quran-api-with-php-codeigniter",
      "type": "tool",
      "country": "INTL",
      "org": "dyazincahya",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/dyazincahya/quran-api-with-php-codeigniter",
        "website": "https://demo.kang-cahya.web.id/quran-api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Holy Quran API | 6236 verses, 114 surah, 30 Juz |",
      "metrics": null
    },
    {
      "id": "quran-audio-api",
      "name": "Quran-audio-api",
      "type": "tool",
      "country": "INTL",
      "org": "aamirbhat382",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/aamirbhat382/Quran-audio-api",
        "website": "https://aamirbhat382.github.io/Quran-audio-api/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Restful Api fetch Quran Audio and Arabic text",
      "metrics": null
    },
    {
      "id": "quran-data-rn0x",
      "name": "Quran-Data",
      "type": "dataset",
      "country": "INTL",
      "org": "rn0x",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/rn0x/Quran-Data"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Quran data in JSON, CSV and SQLite covering surahs, verses and audio, with an API for access.",
      "metrics": null
    },
    {
      "id": "quran-data-aliftype",
      "name": "quran-data (aliftype)",
      "type": "dataset",
      "country": "INTL",
      "org": "Aliftype",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/aliftype/quran-data"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Unicode-encoded Quran text in plain files with tools to convert it to other formats; not formally verified by Mushaf review committees.",
      "metrics": null
    },
    {
      "id": "quran-database",
      "name": "quran-database",
      "type": "dataset",
      "country": "INTL",
      "org": "gaitco",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/gaitco/quran-database",
        "website": "https://gaitco.com/open-source/quran-database"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "A complete, structured Quran database — verses, translations and metadata, free for any project.",
      "metrics": null
    },
    {
      "id": "quran-database-abdallah-mekky",
      "name": "Quran-Database (Abdallah-Mekky)",
      "type": "dataset",
      "country": "INTL",
      "org": "Abdallah-Mekky",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/Abdallah-Mekky/Quran-Database"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "This repository contains a large database of the Quran, with many features.",
      "metrics": null
    },
    {
      "id": "quran-dataset",
      "name": "quran-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "nafiskabbo",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/nafiskabbo/quran-dataset"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "An open-source Quran dataset with tools to export verses, translations, transliteration, audio, and metadata into JSON, CSV, or custom formats.",
      "metrics": null
    },
    {
      "id": "quran-db-aqeelshamz",
      "name": "quran-db",
      "type": "tool",
      "country": "INTL",
      "org": "aqeelshamz",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/aqeelshamz/quran-db"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "JavaScript library for Quran text, translation, audio URLs, and details of pages, juz, surah, ayah, place of revelation etc.",
      "metrics": null
    },
    {
      "id": "quran-flutter-tajweed",
      "name": "Quran-flutter-tajweed",
      "type": "tool",
      "country": "INTL",
      "org": "rovshan-b",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/rovshan-b/Quran-flutter-tajweed"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Quran Tajweed rules highlighter for flutter",
      "metrics": null
    },
    {
      "id": "quran-fonts-hafs-uthmanic-colored-by-tajweed-ruls",
      "name": "Quran-Fonts-HAFS-Uthmanic-Colored-By-Tajweed-Ruls",
      "type": "tool",
      "country": "INTL",
      "org": "AbuYusof",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/AbuYusof/Quran-Fonts-HAFS-Uthmanic-Colored-By-Tajweed-Ruls"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "KFGQPC HAFS Uthmanic Script Bold 13 .",
      "metrics": null
    },
    {
      "id": "quran-image-generator",
      "name": "quran-image-generator",
      "type": "tool",
      "country": "INTL",
      "org": "ZeyadAbbas",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/ZeyadAbbas/quran-image-generator"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Generate endless fully customizable designs of Quran verses to post online",
      "metrics": null
    },
    {
      "id": "quran-image-with-coordinates-generator",
      "name": "quran-image-with-coordinates-generator",
      "type": "dataset",
      "country": "INTL",
      "org": "fai9al7dad",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/fai9al7dad/quran-image-with-coordinates-generator"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "takes a QCF font and return coordinated word by word database with images",
      "metrics": null
    },
    {
      "id": "quran-images-api",
      "name": "quran-images-api",
      "type": "tool",
      "country": "INTL",
      "org": "BetimShala",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/BetimShala/quran-images-api",
        "website": "http://quran-images-api.herokuapp.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "This API is built to fetch Holy Quran pages.",
      "metrics": null
    },
    {
      "id": "quran-json-wpdynamo",
      "name": "quran-json",
      "type": "dataset",
      "country": "INTL",
      "org": "wpdynamo",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "translation"
      ],
      "links": {
        "github": "https://github.com/wpdynamo/quran-json",
        "website": "https://quran-json-sigma.vercel.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran in Arabic Uthmani script with English (Sahih International) translation as JSON, with Juz and Madani page indexes from Quran.com API.",
      "metrics": null
    },
    {
      "id": "quran-json",
      "name": "quran-json (risan)",
      "type": "tool",
      "country": "INTL",
      "org": "risan",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/risan/quran-json",
        "website": "https://quran-json.risanb.com/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran text and translations in JSON format.",
      "metrics": null
    },
    {
      "id": "quran-mcp",
      "name": "quran-mcp",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Quran.com",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp",
        "retrieval"
      ],
      "links": {
        "github": "https://github.com/quran/quran-mcp"
      },
      "notes": "MCP server giving AI assistants grounded access to Quran text",
      "metrics": null
    },
    {
      "id": "quran-meta",
      "name": "quran-meta",
      "type": "tool",
      "country": "INTL",
      "org": "quran-center",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/quran-center/quran-meta",
        "website": "https://quran-center.github.io/quran-meta/docs"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran Meta Data",
      "metrics": null
    },
    {
      "id": "quran-muaalem",
      "name": "quran-muaalem",
      "type": "tool",
      "country": "INTL",
      "org": "obadx",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/obadx/quran-muaalem",
        "website": "https://obadx.github.io/quran-muaalem/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "AI Teacher for the Holy Quran",
      "metrics": null
    },
    {
      "id": "quran-neural-chunker",
      "name": "quran-neural-chunker",
      "type": "tool",
      "country": "INTL",
      "org": "kaisdukes",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kaisdukes/quran-neural-chunker"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "A data preprocessor for the Quranic Treebank using neural networks.",
      "metrics": null
    },
    {
      "id": "quran-neural-parser",
      "name": "quran-neural-parser",
      "type": "tool",
      "country": "INTL",
      "org": "kaisdukes",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kaisdukes/quran-neural-parser"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "A neural parser for the Quranic Treebank.",
      "metrics": null
    },
    {
      "id": "islamandai-quran-nlp",
      "name": "QURAN-NLP (islamAndAi)",
      "type": "tool",
      "country": "INTL",
      "org": "islamAndAi",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith",
        "nlp-resources"
      ],
      "links": {
        "github": "https://github.com/islamAndAi/QURAN-NLP"
      },
      "year": 2022,
      "notes": "Quran, hadith, tafsir and corpus linguistics resources prepared for NLP.",
      "metrics": null
    },
    {
      "id": "quran-pages-images",
      "name": "quran-pages-images",
      "type": "tool",
      "country": "INTL",
      "org": "QuranHub",
      "license": "unlicense",
      "modality": "text",
      "tasks": [
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/QuranHub/quran-pages-images",
        "website": "https://www.quranhub.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Images of the Quran pages for Hafs, Warsh & Tajweed",
      "metrics": null
    },
    {
      "id": "quran-qcf4",
      "name": "quran-qcf4",
      "type": "dataset",
      "country": "INTL",
      "org": "MohamadHajjRabee",
      "license": "other",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/MohamadHajjRabee/quran-qcf4",
        "website": "https://mohamadhajjrabee.github.io/quran-qcf4/demo.html"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Developer-friendly Quran database using QCF v4 fonts (Hafs, Madinah Mushaf 1441 AH) — page JSON, verse index, font map, TTF & WOFF2.",
      "metrics": null
    },
    {
      "id": "quran-roots",
      "name": "quran-roots",
      "type": "tool",
      "country": "INTL",
      "org": "AbstractThinker0",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/AbstractThinker0/quran-roots"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "list of Quranic roots and their derivatives in JSON format",
      "metrics": null
    },
    {
      "id": "quran-search-engine",
      "name": "quran-search-engine",
      "type": "tool",
      "country": "INTL",
      "org": "adelpro",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/adelpro/quran-search-engine",
        "website": "https://quran.us.kg/search"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quran search engine with semantic understanding, fast indexing, and developer-friendly architecture.",
      "metrics": null
    },
    {
      "id": "quran-svg",
      "name": "quran-svg",
      "type": "tool",
      "country": "INTL",
      "org": "quran-ws",
      "license": "other",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/quran-ws/quran-svg",
        "website": "https://quranpedia.net"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "SVG Mushaf pages with mapped ayah positions for displaying and interacting with printed Quran pages in apps and websites.",
      "metrics": null
    },
    {
      "id": "quran-svm-parser",
      "name": "quran-svm-parser",
      "type": "dataset",
      "country": "INTL",
      "org": "kaisdukes",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kaisdukes/quran-svm-parser"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "The original Quranic Arabic Corpus parser using SVM-based machine learning, from Dukes & Habash's 2011 paper.",
      "metrics": null
    },
    {
      "id": "quran-tajweed",
      "name": "quran-tajweed",
      "type": "tool",
      "country": "INTL",
      "org": "cpfair",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/cpfair/quran-tajweed"
      },
      "dialects": [
        "classical"
      ],
      "year": 2021,
      "notes": "Tajweed annotation for the Qur'an",
      "metrics": null
    },
    {
      "id": "quran-text",
      "name": "Quran-text",
      "type": "tool",
      "country": "INTL",
      "org": "naveed-ahmad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/naveed-ahmad/Quran-text"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Word by word and Ayah by Ayah text of Quran in indopak script",
      "metrics": null
    },
    {
      "id": "quran-transcript",
      "name": "quran-transcript",
      "type": "tool",
      "country": "INTL",
      "org": "obadx",
      "license": "other",
      "modality": "text",
      "tasks": [
        "quran",
        "phonetic-transcription"
      ],
      "links": {
        "github": "https://github.com/obadx/quran-transcript"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Python library giving the phonetic transcription of the Quran that captures tajweed rules and letter characteristics.",
      "metrics": null
    },
    {
      "id": "quran-translation",
      "name": "Quran-Translation",
      "type": "tool",
      "country": "INTL",
      "org": "kudanai",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/kudanai/Quran-Translation"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Maintaining the Dhivehi Quran Translation",
      "metrics": null
    },
    {
      "id": "quran-validator",
      "name": "quran-validator",
      "type": "tool",
      "country": "INTL",
      "org": "Yazin Sai",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/yazinsai/quran-validator",
        "website": "https://quranvalidator.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Validate and verify Quranic verses in LLM-generated text with high accuracy",
      "metrics": null
    },
    {
      "id": "quran-validator-php",
      "name": "quran-validator-php",
      "type": "tool",
      "country": "INTL",
      "org": "WatheqAlshowaiter",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/WatheqAlshowaiter/quran-validator-php"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Validate and verify Quranic verses in LLM-generated text with high accuracy.",
      "metrics": null
    },
    {
      "id": "quran-verse-detection",
      "name": "quran-verse-detection",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "unlicense",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/quran-verse-detection",
        "website": "https://fawazahmed0.github.io/quran-verse-detection/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "A Simple Program, which takes quranic verse as input and outputs the chapter & verse No",
      "metrics": null
    },
    {
      "id": "quran-videos",
      "name": "quran-videos",
      "type": "tool",
      "country": "INTL",
      "org": "fawazahmed0",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/fawazahmed0/quran-videos",
        "website": "https://www.youtube.com/user/JavaDB9/playlists"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Publish Quran Recitation Videos at Youtube in every language daily",
      "metrics": null
    },
    {
      "id": "quran-com",
      "name": "Quran.com",
      "type": "tool",
      "country": "INTL",
      "org": "dreygur",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/dreygur/Quran.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "This is a python wraper for quran.com v3 api.",
      "metrics": null
    },
    {
      "id": "quran-dataset-bzekeria",
      "name": "quran_dataset (bzekeria)",
      "type": "dataset",
      "country": "INTL",
      "org": "bzekeria",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/bzekeria/quran_dataset"
      },
      "dialects": [
        "classical"
      ],
      "year": 2022,
      "notes": "The Holy Quran (Islam) Dataset",
      "metrics": null
    },
    {
      "id": "quran-db",
      "name": "quran_db",
      "type": "dataset",
      "country": "INTL",
      "org": "faisalill",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/faisalill/quran_db"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "A curated collection of nuanced quran translations in json format.",
      "metrics": null
    },
    {
      "id": "quran-hadith-datasets",
      "name": "Quran_Hadith_Datasets",
      "type": "dataset",
      "country": "INTL",
      "org": "ShathaTm",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/ShathaTm/Quran_Hadith_Datasets"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "QQ and QH datasets from the paper challenging Transformer models with Classical Arabic Quran and Hadith text.",
      "metrics": null
    },
    {
      "id": "quran-mutashabihat-data",
      "name": "Quran_Mutashabihat_Data",
      "type": "dataset",
      "country": "INTL",
      "org": "Waqar144",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/Waqar144/Quran_Mutashabihat_Data"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Dataset for quran mutashabihat (similar ayats)",
      "metrics": null
    },
    {
      "id": "qurana-backend",
      "name": "qurana-backend",
      "type": "tool",
      "country": "INTL",
      "org": "Qur-ana",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/Qur-ana/qurana-backend",
        "website": "https://api-qurana.digitalkode.com/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Qur'ana adalah sebuah aplikasi Al-Quran yang dibangun menggunakan framework Laravel.",
      "metrics": null
    },
    {
      "id": "qurananalysis",
      "name": "qurananalysis",
      "type": "tool",
      "country": "INTL",
      "org": "karimouda",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa",
        "quran"
      ],
      "links": {
        "github": "https://github.com/karimouda/qurananalysis",
        "website": "http://www.qurananalysis.com/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2017,
      "notes": "Smart Search, Exploration, Analysis and Question Answering System for the Quran",
      "metrics": null
    },
    {
      "id": "quranapi",
      "name": "QuranApi (fcat97)",
      "type": "tool",
      "country": "INTL",
      "org": "fcat97",
      "license": "lgpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/fcat97/QuranApi"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "A small API to add Quran with Tajweed Color in Android App",
      "metrics": null
    },
    {
      "id": "qurancaption",
      "name": "QuranCaption",
      "type": "tool",
      "country": "INTL",
      "org": "zonetecde",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/zonetecde/QuranCaption",
        "website": "https://qurancaption.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Transform Quranic recitations into stunning captioned videos with professional quality subtitles and translations in 40+ languages.",
      "metrics": null
    },
    {
      "id": "qurangpt",
      "name": "QuranGPT",
      "type": "embedding",
      "country": "INTL",
      "org": "hazemabdelkawy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/hazemabdelkawy/QuranGPT",
        "website": "https://hazemabdelkawy.github.io/QuranGPT/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Quran GPT is a project that leverages the power of the GPT-4 language model to generate meaningful embeddings for Quran verses.",
      "metrics": null
    },
    {
      "id": "quranhub",
      "name": "quranhub",
      "type": "tool",
      "country": "SA",
      "org": "Misraj AI",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/misraj-ai/quranhub",
        "website": "https://qurani.ai/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "QuranHub API - A comprehensive REST API providing access to the Holy Quran with multiple editions and languages.",
      "metrics": null
    },
    {
      "id": "qurani",
      "name": "qurani",
      "type": "tool",
      "country": "INTL",
      "org": "nacerbaaziz",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/nacerbaaziz/qurani",
        "website": "https://ng-space.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "برنامج قرآني بواجهة بسيطة وبميزات خرافية مع قواعد بيانات كبيرة للقرآن الكريم وتفسيره",
      "metrics": null
    },
    {
      "id": "quranic",
      "name": "Quranic",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Quranic"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Dataset for Quran Text Processing.",
      "metrics": null
    },
    {
      "id": "quranic-arabic-corpus",
      "name": "Quranic Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "other",
      "modality": "text",
      "tasks": [
        "morphology",
        "quran"
      ],
      "links": {
        "website": "https://corpus.quran.com/download/"
      },
      "dialects": [
        "classical"
      ],
      "size": "128,218 tokens",
      "year": 2017,
      "notes": "Download page of the Quranic Arabic Corpus morphological annotation of the Quran.",
      "metrics": null
    },
    {
      "id": "quranic-universal-library",
      "name": "Quranic Universal Library",
      "type": "tool",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "mit",
      "modality": "multimodal",
      "tasks": [
        "quran",
        "resources"
      ],
      "links": {
        "github": "https://github.com/TarteelAI/quranic-universal-library"
      },
      "year": 2024,
      "notes": "Tarteel collection of Quran text, audio, translation and tafsir resources.",
      "metrics": null
    },
    {
      "id": "quranic-arabic-recognition-dl",
      "name": "quranic-arabic-recognition-dl",
      "type": "ocr",
      "country": "INTL",
      "org": "affaan-m",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr",
        "quran"
      ],
      "links": {
        "github": "https://github.com/affaan-m/quranic-arabic-recognition-dl"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Enhancing Handwritten Quranic Arabic Recognition through Deep Learning",
      "metrics": null
    },
    {
      "id": "quranic-corpus",
      "name": "quranic-corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "kaisdukes",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kaisdukes/quranic-corpus",
        "website": "https://qurancorpus.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "The Quranic Arabic Corpus, an invaluable linguistic resource, is due for a revamp.",
      "metrics": null
    },
    {
      "id": "quranic-corpus-api",
      "name": "quranic-corpus-api",
      "type": "dataset",
      "country": "INTL",
      "org": "kaisdukes",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/kaisdukes/quranic-corpus-api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "Backend server API for the Quranic Arabic Corpus",
      "metrics": null
    },
    {
      "id": "quranic-phonemizer",
      "name": "quranic-phonemizer",
      "type": "tool",
      "country": "INTL",
      "org": "QUD-Technologies",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/QUD-Technologies/quranic-phonemizer",
        "website": "https://phonemizer.qud.dev"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Quranic grapheme to phoneme (G2P) converter and tajweed annotator in Hafs and Warsh",
      "metrics": null
    },
    {
      "id": "quranic-search-v2",
      "name": "quranic-search-v2",
      "type": "embedding",
      "country": "INTL",
      "org": "ahr9n",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran"
      ],
      "links": {
        "github": "https://github.com/ahr9n/quranic-search-v2"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Quranic Lexical/Semantic Search",
      "metrics": null
    },
    {
      "id": "quranic-universal-audio",
      "name": "quranic-universal-audio",
      "type": "asr",
      "country": "INTL",
      "org": "QUD-Technologies",
      "license": "other",
      "modality": "speech",
      "tasks": [
        "asr",
        "vision",
        "quran"
      ],
      "links": {
        "github": "https://github.com/QUD-Technologies/quranic-universal-audio",
        "website": "https://audio.qud.dev"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Unified audio and timing (word + letter) for Quran apps, developers, and researchers.",
      "metrics": null
    },
    {
      "id": "quranic-verse-recognition",
      "name": "Quranic-Verse-Recognition",
      "type": "tool",
      "country": "INTL",
      "org": "Abdelrahman47-code",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "quran",
        "asr"
      ],
      "links": {
        "github": "https://github.com/Abdelrahman47-code/Quranic-Verse-Recognition"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Tool that recognizes Quranic verses from recorded audio using AI speech models.",
      "metrics": null
    },
    {
      "id": "quranicmorphology",
      "name": "QuranicMorphology",
      "type": "dataset",
      "country": "INTL",
      "org": "Abbas1997",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/Abbas1997/QuranicMorphology",
        "website": "https://play.google.com/store/apps/details?id=com.quranvocabgrammar"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "The Holy Quran with grammatical annotations, based on corpus.quran.com data.",
      "metrics": null
    },
    {
      "id": "quranjson",
      "name": "quranjson",
      "type": "tool",
      "country": "INTL",
      "org": "semarketir",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "translation",
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/semarketir/quranjson",
        "website": "https://semarketir.github.io/quranjson-web"
      },
      "dialects": [
        "classical"
      ],
      "year": 2021,
      "notes": "Quran JSON 6236 verses, 114 surah, 30 Juz",
      "metrics": null
    },
    {
      "id": "qurank-kareem",
      "name": "qurank-kareem",
      "type": "tool",
      "country": "INTL",
      "org": "Mohamed20a",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/Mohamed20a/qurank-kareem",
        "website": "https://qurank-kareem.netlify.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "Programming Ongoing charity for all dead Muslims.",
      "metrics": null
    },
    {
      "id": "quranmetaphor",
      "name": "QuranMetaphor",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/QuranMetaphor"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Multi-task framework and code for Qur'anic metaphor analysis (type, origin, functional context).",
      "metrics": null
    },
    {
      "id": "qurantree-jl",
      "name": "QuranTree.jl",
      "type": "dataset",
      "country": "INTL",
      "org": "alstat",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/alstat/QuranTree.jl",
        "website": "https://alstat.github.io/QuranTree.jl/dev/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "A Julia package for working with the Quranic Arabic Corpus.",
      "metrics": null
    },
    {
      "id": "quranvideomaker",
      "name": "QuranVideoMaker",
      "type": "tool",
      "country": "INTL",
      "org": "QuranVideoMaker",
      "license": "other",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/QuranVideoMaker/QuranVideoMaker"
      },
      "dialects": [
        "classical"
      ],
      "year": 2025,
      "notes": "A video editor with the emphasis on creating Quran translation videos.",
      "metrics": null
    },
    {
      "id": "qursci-onto",
      "name": "QurSci-Onto",
      "type": "dataset",
      "country": "INTL",
      "org": "Government Post Graduate College Mansehra",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "retrieval",
        "qa",
        "classification",
        "rag"
      ],
      "links": {
        "github": "https://github.com/Ebad-urRehman/QurSci-Onto",
        "paper": "https://aclanthology.org/2026.abjadnlp-1.22.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "194 sentences",
      "year": 2026,
      "notes": "A hierarchical ontology and dataset for scientific exegesis (Tafsir Ilmi) in the Quran.",
      "metrics": null
    },
    {
      "id": "qutrub",
      "name": "Qutrub",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/linuxscout/qutrub"
      },
      "year": 2015,
      "notes": "Arabic verb conjugator; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "qutuf",
      "name": "Qutuf",
      "type": "tool",
      "country": "INTL",
      "org": "Qutuf",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology",
        "pos"
      ],
      "links": {
        "github": "https://github.com/Qutuf/Qutuf"
      },
      "year": 2017,
      "notes": "Arabic morphological analyzer and part-of-speech tagger.",
      "metrics": null
    },
    {
      "id": "qwen-3",
      "name": "Qwen 3",
      "type": "llm",
      "country": "INTL",
      "org": "Alibaba",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/collections/Qwen/qwen3-67dd247413f0e2e4f653967f"
      },
      "size": "0.6B-235B",
      "notes": "Multilingual with Arabic support",
      "metrics": null
    },
    {
      "id": "qwen3-asr-1-7b-jordanian-dialect-arabic",
      "name": "Qwen3 ASR 1.7B Jordanian Dialect Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "sarapd",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/sarapd/Qwen3-ASR-1.7B_Jordanian_Dialect_Arabic"
      },
      "dialects": [
        "lev"
      ],
      "size": "1.7B",
      "on_device": false,
      "year": 2026,
      "notes": "A full fine-tune of Qwen/Qwen3-ASR-1.7B on 10.8 hours of Jordanian Arabic speech. set.",
      "base_model": [
        "qwen/qwen3-asr-1.7b"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 4,
        "lastModified": "2026-08-30"
      }
    },
    {
      "id": "qwencleo-asr-mohammedaly22",
      "name": "QwenCleo-ASR (MohammedAly22)",
      "type": "asr",
      "country": "EG",
      "org": "Mohammed Aly",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/MohammedAly22/QwenCleo-ASR"
      },
      "dialects": [
        "egy"
      ],
      "year": 2026,
      "notes": "QwenCleo-ASR is the currently State-Of-The-Art (SOTA) open-source model for Egyptian Arabic & code-switching speech recognition Built on Qwen3-ASR-1.7B.",
      "metrics": null
    },
    {
      "id": "rababa",
      "name": "rababa",
      "type": "tool",
      "country": "INTL",
      "org": "interscript",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/interscript/rababa"
      },
      "year": 2026,
      "notes": "Archived origin of secryst/secryst-train/arabic — full history merged there.",
      "metrics": null
    },
    {
      "id": "raqq",
      "name": "raqq",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "agpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aliftype/raqq",
        "website": "https://aliftype.com/raqq"
      },
      "year": 2026,
      "notes": "Raqq (رَقّ) is a manuscript Kufic typeface",
      "metrics": null
    },
    {
      "id": "react-quran",
      "name": "react-quran",
      "type": "tool",
      "country": "INTL",
      "org": "6km",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/6km/react-quran",
        "website": "https://react-quran1.vercel.app/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Easily add Quran viewer to your react applications!",
      "metrics": null
    },
    {
      "id": "real-time-quran-recitation-tracker-system",
      "name": "Real-Time-Quran-recitation-tracker-System",
      "type": "asr",
      "country": "INTL",
      "org": "yayaiu6",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran"
      ],
      "links": {
        "github": "https://github.com/yayaiu6/Real-Time-Quran-recitation-tracker-System"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "An open-source AI system enabling real time Quran recitation tracking, word-level alignment, error detection, and adaptive feedback.",
      "metrics": null
    },
    {
      "id": "reem-kufi",
      "name": "reem-kufi",
      "type": "tool",
      "country": "INTL",
      "org": "Aliftype",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/aliftype/reem-kufi"
      },
      "year": 2026,
      "notes": "Reem Kufi (كوفي ريم) is a modern kufic typeface",
      "metrics": null
    },
    {
      "id": "rsac",
      "name": "RSAC",
      "type": "dataset",
      "country": "EG",
      "org": "Minia University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "classification"
      ],
      "links": {
        "github": "https://github.com/asooft/Sentiment-Analysis-Hotel-Reviews-Dataset",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=9047822"
      },
      "dialects": [
        "mixed"
      ],
      "size": "8,425 sentences",
      "year": 2020,
      "notes": "This dataset contains 6318 hotel reviews collected from the Booking.com website.",
      "metrics": null
    },
    {
      "id": "rtl-for-vs-code-agents",
      "name": "rtl-for-vs-code-agents",
      "type": "tool",
      "country": "INTL",
      "org": "GuyRonnen",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/GuyRonnen/rtl-for-vs-code-agents"
      },
      "year": 2026,
      "notes": "Native-like RTL support for VS Code AI Agents (i.e.",
      "metrics": null
    },
    {
      "id": "rtl-skill",
      "name": "rtl-skill",
      "type": "agent-skill",
      "country": "INTL",
      "org": "mhamedmohammed92-arch",
      "license": "mit",
      "modality": "none",
      "tasks": [
        "skill",
        "rtl"
      ],
      "links": {
        "github": "https://github.com/mhamedmohammed92-arch/rtl-skill"
      },
      "notes": "Teaches coding agents to build correct RTL UIs with a checker",
      "metrics": null
    },
    {
      "id": "rtlify",
      "name": "RTLify",
      "type": "agent-skill",
      "country": "INTL",
      "org": "idanlevi1",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "agent-skill"
      ],
      "links": {
        "github": "https://github.com/idanlevi1/rtlify"
      },
      "year": 2026,
      "notes": "RTL rules for Claude Code, Cursor, Copilot and other coding agents; covers Arabic and Hebrew.",
      "metrics": null
    },
    {
      "id": "rtlterminal",
      "name": "RtlTerminal",
      "type": "tool",
      "country": "INTL",
      "org": "mirbehnam",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mirbehnam/RtlTerminal",
        "website": "https://mirbehnam.github.io/RtlTerminal/"
      },
      "year": 2026,
      "notes": "Open-source Windows terminal emulator with Persian, Arabic and RTL support for CMD, PowerShell, WSL, ANSI colors, links and custom fonts.",
      "metrics": null
    },
    {
      "id": "rtltmpro",
      "name": "RTLTMPro",
      "type": "tool",
      "country": "INTL",
      "org": "Pnarimani",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "rtl-rendering"
      ],
      "links": {
        "github": "https://github.com/pnarimani/RTLTMPro"
      },
      "year": 2018,
      "notes": "Right-to-left Text Mesh Pro plugin for Unity supporting Arabic and Persian.",
      "metrics": null
    },
    {
      "id": "ruqia-library",
      "name": "Ruqia",
      "type": "tool",
      "country": "SA",
      "org": "Ruqyai",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "github": "https://github.com/Ruqyai/Ruqia-Library"
      },
      "year": 2022,
      "notes": "Python library to process, prepare and clean Arabic text.",
      "metrics": null
    },
    {
      "id": "saal-ai",
      "name": "Saal.ai",
      "type": "org",
      "country": "AE",
      "org": "Saal.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://saal.ai/"
      },
      "notes": "Cognitive AI solutions - Arabic NLP, speech, generative AI",
      "metrics": null
    },
    {
      "id": "sadeed",
      "name": "Sadeed",
      "type": "tool",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2504.21635"
      },
      "notes": "Small language model for diacritization",
      "metrics": null
    },
    {
      "id": "sadeed-sadeeddiac-25",
      "name": "Sadeed / SadeedDiac-25",
      "type": "benchmark",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/misraj-ai/Sadeed"
      },
      "dialects": [
        "classical",
        "msa"
      ],
      "notes": "Sadeed: Advancing Arabic Diacritization This repository contains evaluation code and benchmark data for Sadeed, a state-of-the-art model for Arabic text.",
      "metrics": null
    },
    {
      "id": "sadslyc",
      "name": "SADSLyC",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Leeds",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/SalwaAlahmari/SADSLyC_Corpus",
        "paper": "https://aclanthology.org/2025.wacl-1.4.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "31,358 sentences",
      "year": 2025,
      "notes": "First Saudi song-lyrics corpus covering five Saudi dialects (Najdi, Hijazi, Shamali, Janoubi, Shargawi).",
      "metrics": null
    },
    {
      "id": "sakhr-software",
      "name": "Sakhr Software",
      "type": "org",
      "country": "EG",
      "org": "Sakhr Software",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://sakhr.com"
      },
      "notes": "Pioneer of Arabic language technology and NLP solutions, with 28+ years of R&D.",
      "metrics": null
    },
    {
      "id": "salma-ai-mawdoo3-ai",
      "name": "Salma.ai (Mawdoo3 AI)",
      "type": "tool",
      "country": "JO",
      "org": "Mawdoo3",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "assistant"
      ],
      "links": {
        "website": "https://ai.mawdoo3.com"
      },
      "notes": "Mawdoo3 Arabic AI assistant and automation platform for Arabic data-to-decision workflows.",
      "metrics": null
    },
    {
      "id": "sambanova-systems",
      "name": "SambaNova Systems",
      "type": "org",
      "country": "INTL",
      "org": "SambaNova Systems",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://sambanova.ai/"
      },
      "notes": "Arabic language adaptation - SambaLingo Arabic models",
      "metrics": null
    },
    {
      "id": "samer",
      "name": "SAMER",
      "type": "dataset",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "readability",
        "lexicon"
      ],
      "links": {
        "website": "http://samer.camel-lab.com/"
      },
      "year": 2023,
      "notes": "Simplification of Arabic Masterpieces for Extensive Reading: readability lexicon and corpus.",
      "metrics": null
    },
    {
      "id": "samer-readability-lexicon",
      "name": "SAMER readability lexicon",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "other",
      "modality": "text",
      "tasks": [
        "readability"
      ],
      "links": {
        "website": "https://camel.abudhabi.nyu.edu/samer-readability-lexicon/",
        "paper": "https://aclanthology.org/2020.lrec-1.373.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "26,000 tokens",
      "year": 2020,
      "notes": "The SAMER readability lexicon is a large-scale 26,000-lemma leveled readability lexicon for Modern Standard Arabic.",
      "metrics": null
    },
    {
      "id": "samt",
      "name": "samt",
      "type": "tts",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/samt",
        "website": "https://moshe-ship.github.io/samt/"
      },
      "year": 2026,
      "notes": "Arabic Audio/TTS Quality Checker — 14 checks, quality score, zero dependencies",
      "metrics": null
    },
    {
      "id": "saqr",
      "name": "Saqr",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "offensive-language",
        "stance-detection"
      ],
      "links": {
        "website": "https://alt.qcri.org/demos"
      },
      "notes": "Social media analysis platform with stance detection, dialect and offensive-language analysis of Arabic tweets.",
      "metrics": null
    },
    {
      "id": "sarf",
      "name": "Sarf",
      "type": "tool",
      "country": "INTL",
      "org": "Alsaydi",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "github": "https://github.com/alsaydi/sarf"
      },
      "year": 2018,
      "notes": "Arabic morphology system.",
      "metrics": null
    },
    {
      "id": "satour",
      "name": "SAtour",
      "type": "dataset",
      "country": "SA",
      "org": "King AbdulAziz University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/sbasabain/SAtour-A-dataset-with-Zero-shot-automatic-labelling-for-Arabic-short-text-Sentiment-Analysis",
        "paper": "https://doi.org/10.1111/exsy.70030"
      },
      "dialects": [
        "mixed"
      ],
      "size": "2,293 sentences",
      "year": 2025,
      "notes": "A corpus of Arabic short tweets in the tourism domain annotated for sentiment analysis in positive, negative, and neutral classes.",
      "metrics": null
    },
    {
      "id": "saudi-aswat-corpus",
      "name": "Saudi Aswat Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "King Salman Global Academy for Arabic Language",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://falak.ksaa.gov.sa/corpora",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2026/pdf/2026.lrec2026-1.124.pdf"
      },
      "dialects": [
        "gulf"
      ],
      "size": "2,500 hours",
      "year": 2026,
      "notes": "Corpus of spontaneous Saudi Arabic speech that covers multiple regions in Saudi",
      "metrics": null
    },
    {
      "id": "saudi-dialect-corpus-sdc",
      "name": "Saudi Dialect Corpus (SDC)",
      "type": "dataset",
      "country": "INTL",
      "org": "Taghreed Tarmom",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/TaghreedT/SDC"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "210K-word Saudi dialect corpus built for dialect identification.",
      "metrics": null
    },
    {
      "id": "saudi-dialect-irony-dataset",
      "name": "Saudi Dialect Irony Dataset",
      "type": "dataset",
      "country": "SA",
      "org": "iWAN research group (KSU)",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "github": "https://github.com/iwan-rg/Saudi-Dialect-Irony-Dataset"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Sa`7r: Saudi dialect irony dataset from the iWAN research group at King Saud University.",
      "metrics": null
    },
    {
      "id": "saudi-novel-corpus",
      "name": "Saudi Novel Corpus",
      "type": "dataset",
      "country": "SA",
      "org": "Islamic University of Madinah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner"
      ],
      "links": {
        "website": "http://sncorpus.com/home",
        "paper": "https://www.mdpi.com/2076-3417/12/13/6648/"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,134,074 tokens",
      "year": 2022,
      "notes": "Tagged corpus of 53 Saudi novels (1930–2019) for corpus-stylistic research.",
      "metrics": null
    },
    {
      "id": "saudi-riyal-font",
      "name": "Saudi Riyal Font",
      "type": "tool",
      "country": "SA",
      "org": "Emran Alhaddad",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "typography"
      ],
      "links": {
        "github": "https://github.com/emran-alhaddad/Saudi-Riyal-Font"
      },
      "year": 2025,
      "notes": "Open-source font for the Saudi Riyal currency symbol.",
      "metrics": null
    },
    {
      "id": "saudi-dialect-allam",
      "name": "Saudi-Dialect-ALLaM",
      "type": "tool",
      "country": "INTL",
      "org": "HasanBGit",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "github": "https://github.com/HasanBGit/Saudi-Dialect-ALLaM"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2025,
      "notes": "Code for LoRA fine-tuning of ALLaM on Saudi dialect text.",
      "metrics": null
    },
    {
      "id": "saultc-genres",
      "name": "SauLTC genres.",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "pos"
      ],
      "links": {
        "website": "https://figshare.com/articles/dataset/SauLTC_genres_/27287135"
      },
      "notes": "SauLTC: Saudi learner English-Arabic parallel translation corpus with part-of-speech tagging.",
      "metrics": null
    },
    {
      "id": "sautna",
      "name": "Sautna",
      "type": "org",
      "country": "SD",
      "org": "Sautna",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://sautna.com"
      },
      "notes": "Sudanese Arabic text-to-speech with the Zol and Zola voices and a community studio.",
      "metrics": null
    },
    {
      "id": "sawt",
      "name": "Sawt",
      "type": "org",
      "country": "INTL",
      "org": "Sawt",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://trysawt.com/tools/text-to-speech-iraqi"
      },
      "notes": "Arabic dialect text-to-speech platform with Iraqi and other dialect voices.",
      "metrics": null
    },
    {
      "id": "sawt-najd",
      "name": "Sawt Najd",
      "type": "tts",
      "country": "SA",
      "org": "Misraj AI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "dialogue"
      ],
      "links": {
        "website": "https://misraj.ai/en/models/sawt-najd"
      },
      "dialects": [
        "gulf"
      ],
      "year": 2026,
      "notes": "High-quality Najdi Arabic text-to-speech for conversational AI.",
      "metrics": null
    },
    {
      "id": "sawtarabi",
      "name": "SawtArabi",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "website": "https://www.isca-archive.org/interspeech_2025/lodagala25_interspeech.pdf"
      },
      "notes": "First Arabic dialectal + code-switching TTS benchmark",
      "metrics": null
    },
    {
      "id": "sawtify",
      "name": "Sawtify",
      "type": "org",
      "country": "TN",
      "org": "Sawtify",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://tribetechie.com/blog/sawtify-ai-arabic-speech-recognition-tunisia-north"
      },
      "year": 2026,
      "notes": "Tunisian startup building speech recognition for Tunisian Arabic.",
      "metrics": null
    },
    {
      "id": "sawtik",
      "name": "Sawtik",
      "type": "org",
      "country": "INTL",
      "org": "Sawtik",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://sawtik.ai/"
      },
      "notes": "Khaleeji Arabic text-to-speech service with an API.",
      "metrics": null
    },
    {
      "id": "scanread-ai",
      "name": "ScanRead.ai",
      "type": "org",
      "country": "INTL",
      "org": "ScanRead.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://scanread.ai"
      },
      "notes": "Free OCR web tool extracting text from photos, screenshots and PDFs, listed for Arabic OCR.",
      "metrics": null
    },
    {
      "id": "sdaia",
      "name": "SDAIA",
      "type": "org",
      "country": "SA",
      "org": "SDAIA",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://sdaia.gov.sa/"
      },
      "notes": "Sovereign AI, national data authority - ALLaM model, SADA dataset, NCAI, BALSAM benchmark",
      "metrics": null
    },
    {
      "id": "sdaia-saudi-data-ai-authority",
      "name": "SDAIA (Saudi Data & AI Authority)",
      "type": "org",
      "country": "SA",
      "org": "SDAIA (Saudi Data & AI Authority)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://sdaia.gov.sa/"
      },
      "notes": "Sovereign Arabic LLM, national AI strategy - ALLaM model, SADA dataset, BALSAM benchmark",
      "metrics": null
    },
    {
      "id": "sdaia-kfupm-joint-research-center-jrcai",
      "name": "SDAIA-KFUPM Joint Research Center (JRCAI)",
      "type": "org",
      "country": "SA",
      "org": "SDAIA-KFUPM Joint Research Center (JRCAI)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/KFUPM-JRCAI"
      },
      "notes": "Arabic AI text detection, Arabic NLP datasets",
      "metrics": null
    },
    {
      "id": "sdc-shami-dialect-corpus",
      "name": "SDC (Shami Dialect Corpus)",
      "type": "dataset",
      "country": "INTL",
      "org": "GU-CLASP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect",
        "tweets"
      ],
      "links": {
        "github": "https://github.com/GU-CLASP/shami-corpus"
      },
      "dialects": [
        "lev"
      ],
      "notes": "Shami Dialect Corpus repository with code to collect Levantine Arabic tweets via the Twitter API.",
      "metrics": null
    },
    {
      "id": "seacorpus",
      "name": "SEACorpus",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Illinois Urbana-Champaign",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "pretraining",
        "morphology"
      ],
      "links": {
        "github": "https://github.com/maimm2/SaidiCorpus2025",
        "paper": "https://aclanthology.org/2025.nlp4dh-1.26.pdf"
      },
      "dialects": [
        "egy"
      ],
      "size": "413 documents",
      "year": 2025,
      "notes": "First open-source literary corpus of Sa'idi Egyptian Arabic (SEA), containing 4 million words from novels and poetry.",
      "metrics": null
    },
    {
      "id": "search-surah-in-quran",
      "name": "search-surah-in-quran",
      "type": "tool",
      "country": "INTL",
      "org": "d3sc",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/d3sc/search-surah-in-quran",
        "website": "https://search-surah.d3sc.my.id"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "i made an interface web of the search for surah in the quran with API.",
      "metrics": null
    },
    {
      "id": "sentiment-dataset-of-algerian-dialect",
      "name": "sentiment dataset of Algerian dialect",
      "type": "dataset",
      "country": "INTL",
      "org": "Research & Development & Innovation Direction",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/alg-dialect-sentiment-dataset/",
        "paper": "https://www.scitepress.org/Papers/2019/83539/83539.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "11,760 sentences",
      "year": 2019,
      "notes": "Annotated comments/posts in Algerian Arabizi from Facebook pages, collected via APIs/Facepager & Google Forms.",
      "metrics": null
    },
    {
      "id": "sfaia-lid-arabic-dialect-identifier",
      "name": "SfaIA-LID-Arabic-Dialect-Identifier",
      "type": "llm",
      "country": "MA",
      "org": "BounharAbdelaziz",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-identification"
      ],
      "links": {
        "hf": "https://huggingface.co/BounharAbdelaziz/SfaIA-LID-Arabic-Dialect-Identifier"
      },
      "year": 2025,
      "notes": "fastText Arabic dialect identifier trained on the No-Arabic-Dialect-Left-Behind dataset.",
      "metrics": {
        "downloads": 0,
        "likes": 3,
        "lastModified": "2025-01-10"
      }
    },
    {
      "id": "sh-arabiciraqiaccent",
      "name": "SH_ArabicIraqiAccent",
      "type": "dataset",
      "country": "IQ",
      "org": "Middle Technical University",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/SaraEsHassan/SH_ArabicIraqiAccent",
        "paper": "https://jeng.utq.edu.iq/index.php/main/article/view/742"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2025,
      "notes": "Speech corpus of Iraqi Arabic recorded in Baghdad's northern suburbs.",
      "metrics": null
    },
    {
      "id": "shahin",
      "name": "Shahin",
      "type": "llm",
      "country": "SY",
      "org": "malhajar",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "chat"
      ],
      "links": {
        "hf": "https://huggingface.co/malhajar/Shahin-v0.1-14B"
      },
      "size": "14B",
      "dialects": [
        "lev"
      ],
      "notes": "Syrian Arabic dialect, Qwen2.5-based",
      "metrics": null
    },
    {
      "id": "shakkala",
      "name": "Shakkala",
      "type": "tool",
      "country": "INTL",
      "org": "AliOsm",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/AliOsm/shakkelha"
      },
      "notes": "Neural vocalization using bidirectional LSTM",
      "metrics": null
    },
    {
      "id": "shakkala-project",
      "name": "Shakkala Project مشروع شكّالة",
      "type": "tool",
      "country": "INTL",
      "org": "Barqawiz",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts",
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Barqawiz/Shakkala"
      },
      "notes": "The model can also be used in other applications such as improving search results.",
      "metrics": null
    },
    {
      "id": "shamela",
      "name": "Shamela",
      "type": "dataset",
      "country": "INTL",
      "org": "MIT Computer Science and Artificial Intelligence Laboratory",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "pretraining"
      ],
      "links": {
        "website": "https://github.com/OpenArabic/",
        "paper": "https://arxiv.org/pdf/1612.08989.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "6,100 documents",
      "year": 2016,
      "notes": "A large-scale, historical corpus of Arabic of about 1 billion words from diverse periods of time",
      "metrics": null
    },
    {
      "id": "shamela-diacritics-corpus",
      "name": "Shamela Diacritics Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Independent",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "website": "https://archive.org/details/shamela-diacritics-corpus"
      },
      "dialects": [
        "classical"
      ],
      "size": "1,305 documents",
      "year": 2023,
      "notes": "An Arabic diacriticized corpus using data from the old Maktaba Shamela website",
      "metrics": null
    },
    {
      "id": "si2m-lab",
      "name": "SI2M Lab",
      "type": "org",
      "country": "MA",
      "org": "SI2M Lab (INSEA)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/SI2M-Lab"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Moroccan NLP lab behind DarijaBERT and its Arabizi variants.",
      "metrics": null
    },
    {
      "id": "silma-ai",
      "name": "SILMA AI",
      "type": "org",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://silma.ai/"
      },
      "notes": "US-based company founded by Egyptians; SILMA LLMs, SILMA TTS (open v1, commercial v2), Arabic Broad Benchmark.",
      "metrics": null
    },
    {
      "id": "silma-arabic-tts-benchmark",
      "name": "SILMA Arabic TTS Benchmark",
      "type": "benchmark",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/silma-ai/arabic-tts-benchmark"
      },
      "year": 2026,
      "notes": "Side-by-side comparison tool for Arabic speech synthesis models.",
      "metrics": null
    },
    {
      "id": "silma-tts-v2",
      "name": "SILMA TTS v2",
      "type": "tts",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "tts",
        "voice-cloning"
      ],
      "links": {
        "website": "https://silma.ai/arabic-text-to-speech",
        "github": "https://github.com/SILMA-AI/livekit-plugins-silma"
      },
      "year": 2026,
      "notes": "Commercial Arabic/English TTS API from SILMA AI with LiveKit and Pipecat plugins; successor to the open v1.",
      "metrics": null
    },
    {
      "id": "silma-tts-v2-saudi-najdi",
      "name": "SILMA TTS v2 (Saudi Najdi)",
      "type": "tts",
      "country": "INTL",
      "org": "SILMA AI",
      "license": "proprietary",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "website": "https://silma.ai/saudi-tts-model"
      },
      "dialects": [
        "gulf",
        "msa"
      ],
      "year": 2026,
      "notes": "Commercial low-latency (280ms TTFT) streaming TTS for MSA and Saudi Najdi, on-prem/VPC.",
      "metrics": null
    },
    {
      "id": "sima",
      "name": "Sima",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Sima"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Code and expert-annotated dataset for multi-label classification of Qur'anic similes with Arabic transformer models.",
      "metrics": null
    },
    {
      "id": "sinalab",
      "name": "SinaLab",
      "type": "org",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/SinaLab"
      },
      "notes": "SinaTools, Wojood NER corpus",
      "metrics": null
    },
    {
      "id": "sinalab-synonyms-tool",
      "name": "SinaLab Synonyms tool",
      "type": "tool",
      "country": "PS",
      "org": "SinaLab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "synonyms"
      ],
      "links": {
        "github": "https://github.com/SinaLab/Synonyms"
      },
      "year": 2023,
      "notes": "Tool for extending and evaluating Arabic synonyms.",
      "metrics": null
    },
    {
      "id": "sinalab-birzeit-university",
      "name": "SinaLab, Birzeit University",
      "type": "org",
      "country": "PS",
      "org": "SinaLab, Birzeit University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/SinaLab"
      },
      "notes": "Arabic NLP tools and datasets - SinaTools, Wojood NER",
      "metrics": null
    },
    {
      "id": "sinatools",
      "name": "SinaTools",
      "type": "tool",
      "country": "PS",
      "org": "SinaLab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/SinaLab/sinatools"
      },
      "notes": "Open source toolkit by SinaLab (Python APIs, CLI)",
      "metrics": null
    },
    {
      "id": "site",
      "name": "site",
      "type": "tool",
      "country": "INTL",
      "org": "GlobalQuran",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/GlobalQuran/site",
        "website": "http://GlobalQuran.com"
      },
      "dialects": [
        "classical"
      ],
      "year": 2021,
      "notes": "Complete Quran Site Code developed with GlobalQuran api in javascript. you can use it anywhere, on desktop or just upload on any site with your own layouts.",
      "metrics": null
    },
    {
      "id": "smart-glasses-for-blind-people",
      "name": "Smart-glasses-for-blind-people-",
      "type": "ocr",
      "country": "INTL",
      "org": "AliKhedr2",
      "license": "unknown",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/AliKhedr2/Smart-glasses-for-blind-people-"
      },
      "year": 2022,
      "notes": "Eye in AI is a smart project that can make bill detection, object detection, face detection, and OCR in Arabic and English",
      "metrics": null
    },
    {
      "id": "smartly-ai",
      "name": "Smartly.ai",
      "type": "org",
      "country": "MA",
      "org": "Smartly.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://smartly.ai"
      },
      "notes": "Moroccan company building AI agents and conversational experiences for customer service, finance and document workflows.",
      "metrics": null
    },
    {
      "id": "smia-moroccan-society-of-ai",
      "name": "SMIA (Moroccan Society of AI)",
      "type": "org",
      "country": "MA",
      "org": "SMIA",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://smia.ma/en/"
      },
      "notes": "Moroccan Society of Artificial Intelligence, uniting researchers, universities and startups.",
      "metrics": null
    },
    {
      "id": "sohateful",
      "name": "SoHateful",
      "type": "dataset",
      "country": "INTL",
      "org": "Wajdi Zaghouani",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/rafiulbiswas/hatespeech-detection",
        "paper": "https://aclanthology.org/2024.lrec-main.1308.pdf"
      },
      "dialects": [
        "mixed"
      ],
      "size": "15,965 sentences",
      "year": 2024,
      "notes": "70,000 Arabic tweets, from which 15,965 tweets were selected and annotated, to identify hate speech patterns and train classification models",
      "metrics": null
    },
    {
      "id": "soqal",
      "name": "SOQAL",
      "type": "tool",
      "country": "LB",
      "org": "Hussein Mozannar",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "question-answering"
      ],
      "links": {
        "github": "https://github.com/husseinmozannar/SOQAL"
      },
      "year": 2019,
      "notes": "Arabic open-domain question answering with neural reading comprehension.",
      "metrics": null
    },
    {
      "id": "sparta-benchmark",
      "name": "SPARTA benchmark",
      "type": "benchmark",
      "country": "JO",
      "org": "Mawdoo3",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "dialect-id"
      ],
      "links": {
        "github": "https://github.com/mawdoo3/sparta-benchmark"
      },
      "dialects": [
        "gulf",
        "lev",
        "egy"
      ],
      "notes": "Speech Profiling for ARabic TAlk: benchmark for dialect, gender and age profiling of Arabic speakers.",
      "metrics": null
    },
    {
      "id": "speecht5-darija",
      "name": "speecht5-darija",
      "type": "tts",
      "country": "INTL",
      "org": "HAMMALE",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/spaces/HAMMALE/speecht5-darija"
      },
      "dialects": [
        "magh"
      ],
      "notes": "SpeechT5-based Moroccan Darija text-to-speech demo Space.",
      "metrics": null
    },
    {
      "id": "spiral",
      "name": "SPIRAL",
      "type": "dataset",
      "country": "DZ",
      "org": "Ahmed Draia University",
      "license": "cc-by-sa-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "morphology",
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/Dahouabdelhalim/SPIRAL",
        "paper": "https://link.springer.com/article/10.1007/s42979-022-01499-x"
      },
      "dialects": [
        "msa"
      ],
      "size": "248,441,892 tokens",
      "year": 2022,
      "notes": "SPIRAL is a corpus dedicated to the detection and correction of spelling errors in MSA Arabic texts.",
      "metrics": null
    },
    {
      "id": "spoken-adi17-demo",
      "name": "Spoken ADI17 demo",
      "type": "tool",
      "country": "QA",
      "org": "QCRI",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "website": "https://alt.qcri.org/demos"
      },
      "notes": "Spoken Arabic dialect identification from YouTube speech into one of 17 Arab countries.",
      "metrics": null
    },
    {
      "id": "stc-group",
      "name": "stc Group",
      "type": "org",
      "country": "SA",
      "org": "stc Group",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.stc.com.sa"
      },
      "notes": "Saudi telecom group with AI subsidiaries and Arabic chatbot products.",
      "metrics": null
    },
    {
      "id": "styler",
      "name": "Styler",
      "type": "tool",
      "country": "INTL",
      "org": "TidyFactor",
      "license": "other",
      "modality": "text",
      "tasks": [
        "vision"
      ],
      "links": {
        "github": "https://github.com/TidyFactor/Styler",
        "website": "https://tidyfactor.com"
      },
      "year": 2026,
      "notes": "Production-Stage Visual Design & In-Codebase UI Engineering Suite",
      "metrics": null
    },
    {
      "id": "styletts2-libritts-arabic",
      "name": "StyleTTS2-LibriTTS-arabic",
      "type": "tts",
      "country": "INTL",
      "org": "fadi77",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/fadi77/StyleTTS2-LibriTTS-arabic"
      },
      "year": 2025,
      "notes": "This is an Arabic text-to-speech model based on StyleTTS2 architecture, specifically adapted for Arabic language synthesis.",
      "metrics": {
        "downloads": 0,
        "likes": 7,
        "lastModified": "2025-04-19"
      }
    },
    {
      "id": "sudanese-arabic-ai-dataset",
      "name": "Sudanese Arabic AI Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Anwar Dafa-Alla group",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "instruction-tuning"
      ],
      "links": {
        "github": "https://github.com/AnwarCS/Sudanese-Arabic-AI-Dataset"
      },
      "dialects": [
        "sudanese"
      ],
      "notes": "Collaborative Sudanese Arabic dataset for fine-tuning LLMs.",
      "metrics": null
    },
    {
      "id": "sudanese-dialect-speech-dataset",
      "name": "Sudanese dialect speech dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "dialect-speech"
      ],
      "links": {
        "website": "https://zenodo.org/records/6721683"
      },
      "dialects": [
        "sudanese"
      ],
      "year": 2022,
      "notes": "Sudanese dialect speech collected from YouTube videos, mainly Khartoum Arabic.",
      "metrics": null
    },
    {
      "id": "sudanese-arabic-llm",
      "name": "Sudanese-Arabic-LLM",
      "type": "dataset",
      "country": "INTL",
      "org": "AnwarCS",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/AnwarCS/Sudanese-Arabic-LLM",
        "website": "https://github.com/AnwarCS/Sudanese-Arabic-LLM#readme"
      },
      "dialects": [
        "sudanese"
      ],
      "year": 2025,
      "notes": "Building a Sudanese Arabic dataset and fine-tuning LLMs to improve representation of this dialect.",
      "metrics": null
    },
    {
      "id": "sudannese-arabic-telcom-sentiment-classification-pre-processed",
      "name": "Sudannese Arabic Telcom Sentiment Classification Pre Processed",
      "type": "dataset",
      "country": "SD",
      "org": "University of Khartoum",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://docs.google.com/spreadsheets/d/13fIV8oHss-QRBKN-2h5LYF1i_1O9qH1R/edit?usp=sharing&ouid=113975694262803649646&rtpof=true&sd=true",
        "paper": "https://ieeexplore.ieee.org/document/8515862"
      },
      "dialects": [
        "sudanese"
      ],
      "size": "5,349 sentences",
      "year": 2018,
      "notes": "It is pre processed dataset from Twitter about Telecom companies in Sudan, it labelled by 3 different labels from different age, gender and background",
      "metrics": null
    },
    {
      "id": "sudaverse",
      "name": "Sudaverse",
      "type": "org",
      "country": "INTL",
      "org": "Sudaverse",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "github": "https://github.com/sudaverse"
      },
      "notes": "Open-source ecosystem for Sudanese dialect NLP.",
      "metrics": null
    },
    {
      "id": "sultan-qaboos-university",
      "name": "Sultan Qaboos University",
      "type": "org",
      "country": "OM",
      "org": "Sultan Qaboos University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.squ.edu.om"
      },
      "notes": "Omani university with Arabic NLP and AI research.",
      "metrics": null
    },
    {
      "id": "sunnahgpt",
      "name": "SunnahGPT",
      "type": "embedding",
      "country": "INTL",
      "org": "hazemabdelkawy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran",
        "hadith"
      ],
      "links": {
        "github": "https://github.com/hazemabdelkawy/SunnahGPT",
        "website": "https://hazemabdelkawy.github.io/SunnahGPT/"
      },
      "dialects": [
        "classical"
      ],
      "year": 2023,
      "notes": "SunnahGPT is a natural language processing (NLP) project aimed at scraping hadith data from the popular website sunnah.com.",
      "metrics": null
    },
    {
      "id": "swan",
      "name": "Swan",
      "type": "embedding",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding"
      ],
      "links": {
        "paper": "https://arxiv.org/abs/2411.01192"
      },
      "notes": "Dialect-aware, cross-lingual",
      "metrics": null
    },
    {
      "id": "synapse-analytics",
      "name": "Synapse Analytics",
      "type": "org",
      "country": "EG",
      "org": "Synapse Analytics",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://synapseanalytics.com/"
      },
      "notes": "AI for financial inclusion - ML-powered credit scoring, Arabic data analytics",
      "metrics": null
    },
    {
      "id": "t-hsab",
      "name": "T-HSAB",
      "type": "dataset",
      "country": "TN",
      "org": "Tunisian university",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "hate-speech"
      ],
      "links": {
        "github": "https://github.com/Hala-Mulki/T-HSAB-A-Tunisian-Hate-Speech-and-Abusive-Dataset",
        "paper": "https://link.springer.com/chapter/10.1007/978-3-030-32959-4_18"
      },
      "dialects": [
        "magh"
      ],
      "size": "3,834 sentences",
      "year": 2019,
      "notes": "Tunisian Arabic hate speech and abusive language corpus built from Facebook and YouTube comments.",
      "metrics": null
    },
    {
      "id": "tacotron2-arabic",
      "name": "Tacotron2-Arabic",
      "type": "tts",
      "country": "INTL",
      "org": "youssefsharief",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/youssefsharief/arabic-tacotron-tts"
      },
      "notes": "End to end Arabic TTS system based on tacotron.",
      "metrics": null
    },
    {
      "id": "tadabur-fherran",
      "name": "tadabur",
      "type": "dataset",
      "country": "INTL",
      "org": "fherran",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/fherran/tadabur"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Tadabur is a large-scale, high-diversity Qur'anic speech dataset designed to advance Arabic speech research.",
      "metrics": null
    },
    {
      "id": "tafilat",
      "name": "Tafilat",
      "type": "dataset",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "poetry"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Tafilat"
      },
      "year": 2024,
      "notes": "A complete dataset offering all possible patterns for Arabic poetic meters, meticulously curated from classical prosody sources.",
      "metrics": null
    },
    {
      "id": "tafseer-api",
      "name": "tafseer_api",
      "type": "tool",
      "country": "INTL",
      "org": "Quran-Tafseer",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "quran"
      ],
      "links": {
        "github": "https://github.com/Quran-Tafseer/tafseer_api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "Quran Tafseer REST APIs and Quran Text",
      "metrics": null
    },
    {
      "id": "tafsir-dataset",
      "name": "Tafsir Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Goethe University Frankfurt",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "ner",
        "classification"
      ],
      "links": {
        "website": "https://aiwg.de/beta_version_open_tafsir_database",
        "paper": "https://aclanthology.org/2022.coling-1.330.pdf"
      },
      "dialects": [
        "classical"
      ],
      "size": "51,704 sentences",
      "year": 2022,
      "notes": "A multi-task dataset for Named Entity Recognition and Topic Modeling on Classical Arabic Quranic exegesis (Tafsir Al-Tabari).",
      "metrics": null
    },
    {
      "id": "tafsir-mcp",
      "name": "Tafsir MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Tafsir Center",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/tafsircenter/tafsir-mcp"
      },
      "year": 2026,
      "notes": "MCP server for the Quran with 5 classical tafsirs and word-level linguistic data.",
      "metrics": null
    },
    {
      "id": "tafsir-api",
      "name": "tafsir_api",
      "type": "tool",
      "country": "INTL",
      "org": "spa5k",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "translation",
        "quran"
      ],
      "links": {
        "github": "https://github.com/spa5k/tafsir_api",
        "website": "https://github.com/spa5k/tafsir_api"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Free Tafsir API Service with different languages and translations",
      "metrics": null
    },
    {
      "id": "tafsir-semantic-search",
      "name": "tafsir_semantic_search",
      "type": "tool",
      "country": "INTL",
      "org": "misbahsy",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "embedding",
        "quran"
      ],
      "links": {
        "github": "https://github.com/misbahsy/tafsir_semantic_search",
        "website": "https://tafsir-semantic-search.vercel.app"
      },
      "dialects": [
        "classical"
      ],
      "year": 2024,
      "notes": "This repo is for semantic search app to search over Quran tafsir books",
      "metrics": null
    },
    {
      "id": "tajmeeaton",
      "name": "tajmeeaton",
      "type": "tool",
      "country": "INTL",
      "org": "mobadarah",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mobadarah/tajmeeaton",
        "website": "https://mobadarah.github.io/tajmeeaton-web/"
      },
      "year": 2024,
      "notes": "تجميعة من المشاريع، وخصوصا مفتوحة المصدر",
      "metrics": null
    },
    {
      "id": "takalam",
      "name": "Takalam",
      "type": "org",
      "country": "INTL",
      "org": "Takalam",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://takalam.net/en/"
      },
      "notes": "Arabic text-to-speech platform offering Egyptian and Saudi dialect voices.",
      "metrics": null
    },
    {
      "id": "tanqeeh",
      "name": "Tanqeeh",
      "type": "tool",
      "country": "INTL",
      "org": "Noor Bayan",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "preprocessing"
      ],
      "links": {
        "github": "https://github.com/NoorBayan/Tanqeeh"
      },
      "year": 2025,
      "notes": "Python library for preprocessing and cleaning Arabic text.",
      "metrics": null
    },
    {
      "id": "taqeem-2025",
      "name": "TAQEEM 2025",
      "type": "benchmark",
      "country": "QA",
      "org": "Qatar University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "evaluation",
        "shared-task"
      ],
      "links": {
        "paper": "https://aclanthology.org/2025.arabicnlp-sharedtasks.134/",
        "website": "https://gitlab.com/bigirqu/taqeem2025"
      },
      "year": 2025,
      "venue": "ArabicNLP 2025",
      "notes": "This paper presents an overview of the task, outlines the approaches employed, and discusses the results of the participating teams.",
      "metrics": null
    },
    {
      "id": "taric-slu",
      "name": "TARIC-SLU",
      "type": "dataset",
      "country": "INTL",
      "org": "Avignon University",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://demo-lia.univ-avignon.fr/taric-dataset/",
        "paper": "https://aclanthology.org/2024.lrec-main.1357.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "6 hours",
      "year": 2024,
      "notes": "Tunisian railway-transport conversations annotated with dialogue acts and slots for SLU benchmarking.",
      "metrics": null
    },
    {
      "id": "tarjama-amt",
      "name": "Tarjama AMT",
      "type": "tool",
      "country": "AE",
      "org": "Arabic.AI (Tarjama)",
      "license": "proprietary",
      "modality": "text",
      "tasks": [
        "translation",
        "legal",
        "medical"
      ],
      "links": {
        "website": "https://tarjama.com/solutions/technology/ai-machine-translation"
      },
      "year": 2023,
      "notes": "Arabic machine translation engines customised for legal, medical and financial domains with post-editing.",
      "metrics": null
    },
    {
      "id": "tarmeez",
      "name": "Tarmeez",
      "type": "tool",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "https://sourceforge.net/projects/tarmeez"
      },
      "notes": "Binary data format for etymological Arabic system",
      "metrics": null
    },
    {
      "id": "tarteel-ml",
      "name": "tarteel-ml",
      "type": "tool",
      "country": "INTL",
      "org": "Tarteel AI",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran-recitation"
      ],
      "links": {
        "github": "https://github.com/TarteelAI/tarteel-ml"
      },
      "year": 2018,
      "notes": "Pre-processing and training scripts for the Tarteel Quran recitation dataset.",
      "metrics": null
    },
    {
      "id": "tashaphyne",
      "name": "Tashaphyne",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "stemming"
      ],
      "links": {
        "github": "https://github.com/linuxscout/tashaphyne"
      },
      "year": 2017,
      "notes": "Arabic light stemmer and segmentor; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "tashkeela",
      "name": "Tashkeela",
      "type": "dataset",
      "country": "INTL",
      "org": "Anwarvic",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/Anwarvic/Arabic-Tashkeela-Model"
      },
      "notes": "Arabic diacritization corpus",
      "metrics": null
    },
    {
      "id": "tashkeela-arabic-diacritization-corpus",
      "name": "Tashkeela: Arabic diacritization corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "linuxscout",
      "license": "gpl-2.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "website": "https://kaggle.com/datasets/linuxscout/tashkeela"
      },
      "size": "867,913 words",
      "year": 2017,
      "notes": "A corpus for Arabic diacritizer development",
      "metrics": null
    },
    {
      "id": "tashkil",
      "name": "tashkil",
      "type": "tool",
      "country": "INTL",
      "org": "LGUG2Z",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/LGUG2Z/tashkil"
      },
      "year": 2022,
      "notes": "A lightweight Rust library for removing Arabic diacritics",
      "metrics": null
    },
    {
      "id": "tawasul-egy-stt",
      "name": "tawasul egy stt",
      "type": "asr",
      "country": "INTL",
      "org": "TawasulAI",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "hf": "https://huggingface.co/TawasulAI/tawasul-egy-stt"
      },
      "dialects": [
        "egy"
      ],
      "year": 2025,
      "notes": "Arabic automatic speech recognition model.",
      "metrics": {
        "downloads": 0,
        "likes": 6,
        "lastModified": "2025-06-09"
      }
    },
    {
      "id": "teachyourselfcs-ar",
      "name": "TeachYourselfCS-AR",
      "type": "tool",
      "country": "INTL",
      "org": "ounissi-zakaria",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/ounissi-zakaria/TeachYourselfCS-AR"
      },
      "year": 2022,
      "notes": "An Arabic translation of TeachYourselfCS | ترجمة عربية TeachYourselfCS",
      "metrics": null
    },
    {
      "id": "teammates-ai",
      "name": "Teammates.ai",
      "type": "org",
      "country": "AE",
      "org": "Teammates.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://teammates.ai"
      },
      "notes": "AI agent platform (Raya, Adam, Sara) resolving tickets and booking meetings, with Arabic agents for 20+ dialects.",
      "metrics": null
    },
    {
      "id": "technology-innovation-institute-tii",
      "name": "Technology Innovation Institute (TII)",
      "type": "org",
      "country": "AE",
      "org": "Technology Innovation Institute (TII)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.tii.ae/"
      },
      "notes": "Open-source LLMs, research - Falcon LLM family",
      "metrics": null
    },
    {
      "id": "telewizard-ramsa",
      "name": "TeleWizard Ramsa",
      "type": "org",
      "country": "AE",
      "org": "TeleWizard Ramsa",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://telewizard.ai"
      },
      "notes": "AI call-centre platform with AI phone agents and omnichannel messaging, listed for Emirati Arabic AI voice.",
      "metrics": null
    },
    {
      "id": "tesseract-ocr",
      "name": "Tesseract OCR",
      "type": "tool",
      "country": "INTL",
      "org": "Tesseract OCR",
      "license": "apache-2.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "github": "https://github.com/tesseract-ocr/tesseract"
      },
      "notes": "Open-source OCR engine with Arabic language packs",
      "metrics": null
    },
    {
      "id": "textblob-ar",
      "name": "textblob-ar",
      "type": "tool",
      "country": "INTL",
      "org": "adhaamehab",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "sentiment",
        "pos"
      ],
      "links": {
        "github": "https://github.com/adhaamehab/textblob-ar"
      },
      "notes": "Arabic language support extension for the TextBlob NLP library.",
      "metrics": null
    },
    {
      "id": "tibyan",
      "name": "Tibyan",
      "type": "dataset",
      "country": "SA",
      "org": "King AbdulAziz University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "grammar-correction"
      ],
      "links": {
        "website": "https://zenodo.org/records/14623621",
        "paper": "https://doi.org/10.7717/peerj-cs.2724"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,191 sentences",
      "year": 2025,
      "notes": "A balanced and comprehensive Arabic grammatical error correction corpus created using ChatGPT consisting of 600K tokens.",
      "metrics": null
    },
    {
      "id": "tii-technology-innovation-institute",
      "name": "TII (Technology Innovation Institute)",
      "type": "org",
      "country": "AE",
      "org": "TII (Technology Innovation Institute)",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.tii.ae/"
      },
      "notes": "Arabic LLM benchmarks, Falcon - Open Arabic LLM Leaderboard, Falcon LLM",
      "metrics": null
    },
    {
      "id": "tilawa",
      "name": "Tilawa",
      "type": "asr",
      "country": "INTL",
      "org": "Yazin Sai",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "quran-recitation"
      ],
      "links": {
        "github": "https://github.com/yazinsai/tilawa"
      },
      "year": 2026,
      "notes": "Offline Quran verse recognition that identifies surah and ayah from audio.",
      "metrics": null
    },
    {
      "id": "tinyoctopus",
      "name": "TinyOctopus",
      "type": "llm",
      "country": "INTL",
      "org": "SaraAlthubaiti",
      "license": "unknown",
      "modality": "multimodal",
      "tasks": [
        "speech"
      ],
      "links": {
        "hf": "https://huggingface.co/SaraAlthubaiti/TinyOctopus"
      },
      "year": 2025,
      "notes": "TinyOctopus maintaining the architectural princip",
      "base_model": [
        "deepseek-ai/deepseek-r1-distill-qwen-1.5b"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 10,
        "lastModified": "2025-03-05"
      }
    },
    {
      "id": "tkseem",
      "name": "tkseem",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ARBML/tkseem"
      },
      "notes": "Arabic Tokenization",
      "metrics": null
    },
    {
      "id": "tnkeeh",
      "name": "tnkeeh",
      "type": "tool",
      "country": "SA",
      "org": "ARBML",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/ARBML/tnkeeh"
      },
      "notes": "Arabic text cleaning, normalization, preprocessing",
      "metrics": null
    },
    {
      "id": "toia",
      "name": "TOIA",
      "type": "tool",
      "country": "AE",
      "org": "CAMeL Lab, NYUAD",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialogue"
      ],
      "links": {
        "website": "https://nyuad.nyu.edu/en/research/faculty-labs-and-projects/computational-approaches-to-modeling-language-lab/research.html"
      },
      "notes": "Time-offset interaction application: bilingual Arabic-English conversational avatar built from pre-recorded videos.",
      "metrics": null
    },
    {
      "id": "toumai",
      "name": "ToumAI",
      "type": "org",
      "country": "MA",
      "org": "ToumAI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://toum.ai"
      },
      "dialects": [
        "magh",
        "mixed"
      ],
      "notes": "Moroccan multilingual voice-agent company focused on dialect-aware voice AI.",
      "metrics": null
    },
    {
      "id": "troll-detection",
      "name": "Troll Detection",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "troll-detection"
      ],
      "links": {
        "website": "https://www.dropbox.com/s/hqab7kp2zyex01h/Trolls%20Dataset.zip?dl=0"
      },
      "dialects": [
        "mixed"
      ],
      "size": "128 sentences",
      "year": 2020,
      "notes": "Arabic tweets labeled for troll detection, distributed as a Dropbox zip.",
      "metrics": null
    },
    {
      "id": "tts-arabic-flutter",
      "name": "tts-arabic Flutter app",
      "type": "tts",
      "country": "INTL",
      "org": "nipponjo",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/nipponjo/tts-arabic-flutter"
      },
      "year": 2024,
      "notes": "Flutter demo app for offline ONNX-based Arabic speech synthesis.",
      "metrics": null
    },
    {
      "id": "tts-arabic-pytorch",
      "name": "tts-arabic-pytorch",
      "type": "tts",
      "country": "INTL",
      "org": "nipponjo",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/nipponjo/tts-arabic-pytorch"
      },
      "on_device": true,
      "notes": "Tacotron2 + FastPitch + HiFi-GAN for Arabic",
      "metrics": null
    },
    {
      "id": "tts-arabic-onnx",
      "name": "tts_arabic (ONNX)",
      "type": "tts",
      "country": "INTL",
      "org": "nipponjo",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/nipponjo/tts_arabic"
      },
      "on_device": true,
      "notes": "FastPitch + Mixer-TTS in ONNX for offline Arabic TTS",
      "metrics": null
    },
    {
      "id": "tunartts",
      "name": "TunArTTS",
      "type": "dataset",
      "country": "INTL",
      "org": "ELYADATA",
      "license": "cc-by-nc-4.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/elyadata/TunArTTS",
        "paper": "https://aclanthology.org/2024.lrec-main.1467.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "3 hours",
      "year": 2024,
      "notes": "First speech corpus for Tunisian Arabic TTS with 3+ hours of male speaker audio",
      "metrics": null
    },
    {
      "id": "tunico",
      "name": "TUNICO",
      "type": "dataset",
      "country": "INTL",
      "org": "Austrian Academy of Sciences (ACDH)",
      "license": "cc0-1.0",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "website": "https://www.oeaw.ac.at/acdh/research/linguistics/research/project-archive/tunico"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Linguistic corpus of the Tunis dialect from the Austrian Academy of Sciences, released under CC0.",
      "metrics": null
    },
    {
      "id": "tunisian-arabic-corpus",
      "name": "Tunisian Arabic Corpus",
      "type": "dataset",
      "country": "INTL",
      "org": "Edinburgh University",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "morphology"
      ],
      "links": {
        "website": "http://www.tunisiya.org/",
        "paper": "https://www.academia.edu/28966672/Tunisian_Arabic_Corpus_Creating_a_written_corpus_of_an_unwritten_language"
      },
      "dialects": [
        "magh"
      ],
      "size": "2,874 documents",
      "year": 2010,
      "notes": "There are currently 2,874 texts in the corpus, comprising 1,088,614 words.",
      "metrics": null
    },
    {
      "id": "tunisian-arabizi-dataset",
      "name": "Tunisian Arabizi dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "iCompass",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "sentiment"
      ],
      "links": {
        "website": "https://zenodo.org/record/4275240",
        "paper": "https://aclanthology.org/2021.wanlp-1.25.pdf"
      },
      "dialects": [
        "magh"
      ],
      "size": "100,000 sentences",
      "year": 2021,
      "notes": "A large Tunisian Arabizi dialectal sentiment analysis dataset containing 100k comments",
      "metrics": null
    },
    {
      "id": "tunisian-automatic-speech-recognition",
      "name": "Tunisian Automatic Speech Recognition",
      "type": "asr",
      "country": "INTL",
      "org": "SalahZa",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/SalahZa/Tunisian_Automatic_Speech_Recognition"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "ASR model for Tunisian Arabic dialect, aimed at underrepresented speech communities.",
      "metrics": {
        "downloads": 0,
        "likes": 14,
        "lastModified": "2023-09-25"
      }
    },
    {
      "id": "openslr-arabic-46",
      "name": "Tunisian MSA Speech (OpenSLR 46)",
      "type": "dataset",
      "country": "INTL",
      "org": "OpenSLR",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://www.openslr.org/46/"
      },
      "year": 2018,
      "dialects": [
        "msa"
      ],
      "notes": "OpenSLR Tunisian Modern Standard Arabic speech corpus (Tunisia).",
      "metrics": null
    },
    {
      "id": "tunisian-arabic-ai-dataset",
      "name": "tunisian-arabic-ai-dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "bahaeddinmselmi",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/bahaeddinmselmi/tunisian-arabic-ai-dataset"
      },
      "dialects": [
        "magh"
      ],
      "notes": "Open-source Tunisian Arabic dataset for AI training.",
      "metrics": null
    },
    {
      "id": "tunisian-derja-nlp-resources",
      "name": "Tunisian-Derja-NLP-Resources",
      "type": "tool",
      "country": "INTL",
      "org": "jjlalli",
      "license": "other",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/jjlalli/Tunisian-Derja-NLP-Resources"
      },
      "dialects": [
        "magh"
      ],
      "year": 2026,
      "notes": "Open, maintained inventory of NLP resources for Tunisian Arabic (aeb).",
      "metrics": null
    },
    {
      "id": "tunswitch-code-switched-tunisian-arabic-speech-dataset",
      "name": "TunSwitch: Code-Switched Tunisian Arabic Speech Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "asr",
        "code-switching"
      ],
      "links": {
        "website": "https://zenodo.org/records/8370566"
      },
      "dialects": [
        "magh"
      ],
      "year": 2023,
      "notes": "Code-switched Tunisian Arabic speech data used to develop and test a Tunisian ASR model.",
      "metrics": null
    },
    {
      "id": "turath-mcp",
      "name": "turath-mcp",
      "type": "agent-skill",
      "country": "INTL",
      "org": "opin22",
      "license": "mit",
      "modality": "none",
      "tasks": [
        "mcp",
        "retrieval"
      ],
      "links": {
        "github": "https://github.com/opin22/turath-mcp"
      },
      "notes": "MCP server for turath.io: classical Arabic and Islamic books",
      "metrics": null
    },
    {
      "id": "turjuman",
      "name": "TURJUMAN",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "translation"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/turjuman"
      },
      "year": 2022,
      "notes": "Neural toolkit translating from 20 languages into Modern Standard Arabic.",
      "metrics": null
    },
    {
      "id": "twitter-arabic-image-spam",
      "name": "Twitter Arabic Image Spam",
      "type": "dataset",
      "country": "INTL",
      "org": "University of York",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "spam-detection"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/gfc32vndz8/2",
        "paper": "https://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=8892954"
      },
      "dialects": [
        "mixed"
      ],
      "size": "300 images",
      "year": 2019,
      "notes": "Images with Arabic embedded text collected from Twitter for spam detection research",
      "metrics": null
    },
    {
      "id": "tzamun-voicehub",
      "name": "Tzamun VoiceHub",
      "type": "org",
      "country": "SA",
      "org": "Tzamun VoiceHub",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://tzamun.sa"
      },
      "notes": "Saudi enterprise software and AI company with 11+ products, including conversational AI and voice automation.",
      "metrics": null
    },
    {
      "id": "uae-arabic-speech-recognition-corpus-mobile",
      "name": "UAE Arabic Speech Recognition Corpus (Mobile)",
      "type": "dataset",
      "country": "INTL",
      "org": "ELRA",
      "license": "ELRA",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "website": "https://catalogue.elra.info/en-us/repository/browse/ELRA-S0228_130/"
      },
      "dialects": [
        "gulf"
      ],
      "notes": "Licensed UAE Arabic ASR corpus recorded on mobile; a desktop version is sold via DataOcean.",
      "metrics": null
    },
    {
      "id": "ubc-nlp",
      "name": "UBC-NLP",
      "type": "org",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "hf": "https://huggingface.co/UBC-NLP"
      },
      "notes": "Dialectal Arabic, multimodal models - MARBERT, AraT5, NileChat, PEARL, Dallah",
      "metrics": null
    },
    {
      "id": "udistilwhisper",
      "name": "uDistilWhisper",
      "type": "asr",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/uDistilWhisper"
      },
      "year": 2025,
      "notes": "UDistil-Whisper: Label-Free Data Filtering for Knowledge Distillation in Low-Data Regimes ( NAACL'2025 )",
      "metrics": null
    },
    {
      "id": "unifonic",
      "name": "Unifonic",
      "type": "org",
      "country": "SA",
      "org": "Unifonic",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.unifonic.com/"
      },
      "notes": "Conversational AI platform - Arabic-first CX Intelligence, AI chatbots",
      "metrics": null
    },
    {
      "id": "uae-university",
      "name": "United Arab Emirates University",
      "type": "org",
      "country": "AE",
      "org": "United Arab Emirates University",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.uaeu.ac.ae"
      },
      "notes": "Al Ain university with Arabic NLP and AI research.",
      "metrics": null
    },
    {
      "id": "university-of-bahrain",
      "name": "University of Bahrain",
      "type": "org",
      "country": "BH",
      "org": "University of Bahrain",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "research"
      ],
      "links": {
        "website": "https://www.uob.edu.bh"
      },
      "notes": "Bahrain national university with Arabic NLP and AI research.",
      "metrics": null
    },
    {
      "id": "unixel",
      "name": "Unixel",
      "type": "tool",
      "country": "INTL",
      "org": "MDarvishi5124",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/MDarvishi5124/Unixel",
        "website": "https://mdarvishi5124.github.io/Unixel/"
      },
      "year": 2024,
      "notes": "An English-Arabic pixel font.",
      "metrics": null
    },
    {
      "id": "unlabelled-arabic-speech-dataset",
      "name": "Unlabelled Arabic Speech Dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "unknown",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "speech-corpus"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/yhy76b2c7b"
      },
      "year": 2023,
      "notes": "Unlabelled Arabic speech dataset (8.84 GB) for speech analysis such as speaker identification.",
      "metrics": null
    },
    {
      "id": "uralicnlp",
      "name": "uralicNLP",
      "type": "tool",
      "country": "INTL",
      "org": "mikahama",
      "license": "apache-2.0",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/mikahama/uralicNLP",
        "website": "http://uralicnlp.com/"
      },
      "year": 2026,
      "notes": "An NLP library for Uralic languages such as Finnish, Skolt Sami, Moksha and so on.",
      "metrics": null
    },
    {
      "id": "utep-corpus-of-iraqi-arabic",
      "name": "UTEP Corpus of Iraqi Arabic",
      "type": "dataset",
      "country": "INTL",
      "org": "University of Texas at El Paso",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "paper": "https://scholarworks.utep.edu/cgi/viewcontent.cgi?article=1128&context=cs_techrep"
      },
      "dialects": [
        "iraqi"
      ],
      "year": 2006,
      "notes": "Conversational Iraqi Arabic corpus described in a 2006 UTEP technical report.",
      "metrics": null
    },
    {
      "id": "vazirmatn",
      "name": "vazirmatn",
      "type": "tool",
      "country": "INTL",
      "org": "rastikerdar",
      "license": "ofl-1.1",
      "modality": "text",
      "tasks": [
        "nlp-toolkit"
      ],
      "links": {
        "github": "https://github.com/rastikerdar/vazirmatn",
        "website": "https://rastikerdar.github.io/vazirmatn/"
      },
      "year": 2023,
      "notes": "Vazirmatn is a Persian/Arabic font.",
      "metrics": null
    },
    {
      "id": "vibevoice-arabic-z",
      "name": "vibevoice arabic Z",
      "type": "tts",
      "country": "INTL",
      "org": "ABDALLALSWAITI",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "hf": "https://huggingface.co/ABDALLALSWAITI/vibevoice-arabic-Z"
      },
      "year": 2025,
      "notes": "This is a LoRA (Low-Rank Adaptation) fine-tuned model for Arabic text-to-speech, based on aoi-ot/VibeVoice-Large.",
      "base_model": [
        "aoi-ot/vibevoice-large"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 6,
        "lastModified": "2025-09-30"
      }
    },
    {
      "id": "video-caption-mcp",
      "name": "Video Caption MCP",
      "type": "agent-skill",
      "country": "INTL",
      "org": "EngDawood",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/EngDawood/video-caption"
      },
      "year": 2026,
      "notes": "MCP server and skill that burns Arabic or translated captions into videos.",
      "metrics": null
    },
    {
      "id": "violetv2",
      "name": "VioletV2",
      "type": "llm",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "mit",
      "modality": "vision",
      "tasks": [
        "image-captioning"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/VioletV2"
      },
      "year": 2024,
      "notes": "Violet v2, an Arabic image-to-text (captioning) model.",
      "metrics": null
    },
    {
      "id": "vml-moc",
      "name": "VML-MOC",
      "type": "dataset",
      "country": "INTL",
      "org": "Ben-Gurion University",
      "license": "cc-by-4.0",
      "modality": "vision",
      "tasks": [
        "ocr"
      ],
      "links": {
        "website": "https://zenodo.org/records/3559101",
        "paper": "https://doi.org/10.1109/ICDARW.2019.50109"
      },
      "dialects": [
        "mixed"
      ],
      "size": "30 images",
      "year": 2019,
      "notes": "Handwritten document images with multiply oriented and curved text lines for text line segmentation.",
      "metrics": null
    },
    {
      "id": "voice213",
      "name": "Voice213",
      "type": "org",
      "country": "DZ",
      "org": "Voice213",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://www.voice213.com/"
      },
      "notes": "Algerian AI voice studio offering TTS and voice cloning in Algerian and other Arabic dialects.",
      "metrics": null
    },
    {
      "id": "voicetut-tts",
      "name": "VoiceTut-TTS",
      "type": "tts",
      "country": "EG",
      "org": "Mohammed Aly",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "tts"
      ],
      "links": {
        "github": "https://github.com/MohammedAly22/VoiceTuT-TTS"
      },
      "year": 2026,
      "notes": "Egyptian-Arabic text-to-speech fine-tuned from OmniVoice.",
      "metrics": null
    },
    {
      "id": "w3c-alreq",
      "name": "W3C Arabic Layout Requirements",
      "type": "tool",
      "country": "INTL",
      "org": "W3C",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "rtl-rendering",
        "documentation"
      ],
      "links": {
        "github": "https://github.com/w3c/alreq"
      },
      "year": 2015,
      "notes": "Documents gaps and requirements for Arabic-script layout on the web and in digital publishing.",
      "metrics": null
    },
    {
      "id": "wanlp2020-arabic-fake-news-detection",
      "name": "wanlp2020_arabic_fake_news_detection",
      "type": "tool",
      "country": "INTL",
      "org": "UBC-NLP",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "fact-checking"
      ],
      "links": {
        "github": "https://github.com/UBC-NLP/wanlp2020_arabic_fake_news_detection"
      },
      "year": 2020,
      "notes": "Machine Generation and Detection of Arabic Manipulated and Fake News",
      "metrics": null
    },
    {
      "id": "warsh-quran-audio",
      "name": "warsh-quran-audio",
      "type": "dataset",
      "country": "INTL",
      "org": "Yousr-Allah-Allouani",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "speech",
        "quran"
      ],
      "links": {
        "github": "https://github.com/Yousr-Allah-Allouani/warsh-quran-audio"
      },
      "dialects": [
        "classical"
      ],
      "year": 2026,
      "notes": "Open research: Warsh Quranic recitation tracking — dataset, alignments, live tracker",
      "metrics": null
    },
    {
      "id": "wasl-mcp",
      "name": "Wasl",
      "type": "agent-skill",
      "country": "INTL",
      "org": "Moshe-ship",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "mcp-server"
      ],
      "links": {
        "github": "https://github.com/Moshe-ship/wasl"
      },
      "year": 2026,
      "notes": "Arabic MCP server bundling prayer times, Quran, Hadith, translation and dialect tools.",
      "metrics": null
    },
    {
      "id": "whisper-arabic-dialects",
      "name": "Whisper Arabic dialects",
      "type": "asr",
      "country": "INTL",
      "org": "Ahmed Hany",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "github": "https://github.com/dev-ahmedhany/whisper-arabic-dialects"
      },
      "year": 2026,
      "notes": "Whisper fine-tuning code for Arabic dialects.",
      "metrics": null
    },
    {
      "id": "whisper-egyptian-arabic",
      "name": "Whisper Egyptian Arabic",
      "type": "asr",
      "country": "INTL",
      "org": "MAdel121",
      "license": "apache-2.0",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/MAdel121/whisper-medium-egy"
      },
      "dialects": [
        "egy"
      ],
      "notes": "Whisper-medium fine-tuned on 72h Egyptian speech",
      "base_model": [
        "openai/whisper-medium"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 6,
        "lastModified": "2025-05-21"
      }
    },
    {
      "id": "whisper-large-v2-jordanian-arabic-ft",
      "name": "whisper-large-v2-jordanian-arabic-ft",
      "type": "asr",
      "country": "INTL",
      "org": "xtz999",
      "license": "unknown",
      "modality": "speech",
      "tasks": [
        "asr"
      ],
      "links": {
        "hf": "https://huggingface.co/xtz999/whisper-large-v2-jordanian-arabic-ft"
      },
      "dialects": [
        "lev"
      ],
      "year": 2026,
      "notes": "Whisper large-v2 fine-tuned for Jordanian Arabic speech recognition.",
      "metrics": {
        "downloads": 0,
        "likes": 0,
        "lastModified": "2026-05-14"
      }
    },
    {
      "id": "whisper-small-openai-finetuned-on-arabic-language",
      "name": "Whisper_small_openai_finetuned_on_arabic_language",
      "type": "asr",
      "country": "INTL",
      "org": "Huzaifa-X",
      "license": "mit",
      "modality": "speech",
      "tasks": [
        "asr",
        "evaluation"
      ],
      "links": {
        "github": "https://github.com/Huzaifa-X/Whisper_small_openai_finetuned_on_arabic_language"
      },
      "year": 2024,
      "notes": "Arabic Speech Recognition with Whisper: Fine-tune the Whisper model from OpenAI for Arabic speech recognition tasks.",
      "metrics": null
    },
    {
      "id": "widebot-ai",
      "name": "WideBot AI",
      "type": "org",
      "country": "EG",
      "org": "WideBot AI",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://widebot.ai/"
      },
      "dialects": [
        "egy",
        "mixed"
      ],
      "notes": "Arabic-first conversational AI - AQL Arabic LLM, chatbots, voicebots, AI agents",
      "metrics": null
    },
    {
      "id": "wihard",
      "name": "WiHArD",
      "type": "dataset",
      "country": "INTL",
      "org": "University Centre -SALHI Ahmed",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "classification"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/kdkryh5rs2",
        "paper": "https://ieeexplore.ieee.org/document/10783418"
      },
      "dialects": [
        "msa"
      ],
      "size": "6,027 documents",
      "year": 2024,
      "notes": "The first hierarchical Arabic classification dataset derived from Wikipedia with up-to-date domain-annotated articles",
      "metrics": null
    },
    {
      "id": "wikinews",
      "name": "Wikinews",
      "type": "dataset",
      "country": "INTL",
      "org": "kdarwish",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "qa"
      ],
      "links": {
        "github": "https://github.com/kdarwish/Farasa"
      },
      "notes": "Farasa This repo contains the WikiNews corpus that was annotated by the Arabic Language Techonolgies team at the Qatar Computing Research Institute.",
      "metrics": null
    },
    {
      "id": "wikinewsmaxdiacs",
      "name": "WikiNewsMaxDiacs",
      "type": "dataset",
      "country": "AE",
      "org": "University of Sharjah",
      "license": "cc-by-nc-sa-4.0",
      "modality": "text",
      "tasks": [
        "diacritization"
      ],
      "links": {
        "github": "https://github.com/CAMeL-Lab/wild_diacritics",
        "paper": "https://arxiv.org/pdf/2406.05760v1.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "16,215 tokens",
      "year": 2024,
      "notes": "A new version of WikiNews that was manually extended for maximal diacritization",
      "metrics": null
    },
    {
      "id": "wittify-ai",
      "name": "Wittify.ai",
      "type": "org",
      "country": "SA",
      "org": "Wittify.ai",
      "license": "unknown",
      "modality": "none",
      "tasks": [
        "industry"
      ],
      "links": {
        "website": "https://wittify.ai/"
      },
      "notes": "Conversational AI for Arabic - Interactive Arabic AI agents",
      "metrics": null
    },
    {
      "id": "wojood",
      "name": "Wojood",
      "type": "dataset",
      "country": "PS",
      "org": "SinaLab",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/SinaLab/ArabicNER"
      },
      "notes": "Nested NER corpus (550K tokens)",
      "metrics": null
    },
    {
      "id": "wojoodhadath",
      "name": "WojoodHadath",
      "type": "dataset",
      "country": "PS",
      "org": "Birzeit University",
      "license": "cc-by-4.0",
      "modality": "text",
      "tasks": [
        "pretraining",
        "tokenization",
        "ner",
        "nli"
      ],
      "links": {
        "website": "https://sina.birzeit.edu/relations/",
        "paper": "https://doi.org/10.18653/v1/2024.arabicnlp-1.26"
      },
      "dialects": [
        "mixed"
      ],
      "size": "550,000 tokens",
      "year": 2024,
      "notes": "Extends the Wojood dataset by incorporating relations into Wojood's nested structure.",
      "metrics": null
    },
    {
      "id": "wsd",
      "name": "WSD",
      "type": "dataset",
      "country": "INTL",
      "org": "Zayed University",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "translation",
        "sentiment",
        "word-sense-disambiguation",
        "spam-detection"
      ],
      "links": {
        "website": "https://data.mendeley.com/datasets/pmdbs9tby8/1",
        "paper": "https://pdf.sciencedirectassets.com/311593/1-s2.0-S2352340924X00049/1-s2.0-S2352340924005584/main.pdf?X-Amz-Security-Token=IQoJb3JpZ2luX2VjEEYaCXVzLWVhc3QtMSJHMEUCICyC47GOQc9rLG43raI5rqQ0ZIKcmy0aY5HV7%2BtawE9AAiEAu8Pqw0sxjMstvisjLSOl1hloKtKd12xsZctWRKBjshAqsgUIbxAFGgwwNTkwMDM1NDY4NjUiDBv72KIPY2vlMA70wSqPBVqKd9bfxEDEjF0mw337Ir7hLqqWyfrDSahUdhtprlQHWapI3eNO3ShCK6MGhMVvb0ldLVsj6gy0L20tQ%2BtyWneqKrTVNtF6SQnod6c2z4m5ec7Mph8mHXjjZfp%2BTh0YsajjOuCJ37TjwwszAoiiLFGOH3S52MkBO3z5eRp3jXbMRVoTN%2F7yslstJO1yTXCPt3N6dK0FEfnxGRFB85Bmov%2FsKX86jIf0H%2FAutZ2dx7S0M8rRbzek8X%2B%2FpVHnEPX8TYbaUemNDp3d1QkJtR3iWDj0jOQmUsHUeR8D1vs52rpnxqOun1sHQx6l7hrSCdi7SVX61MllgSpQvMSEKZ1%2BvovE5IuopSPKrGfhTfKTuAXsNEbnguLkjuIoOEh0krzBi3r%2F6np5BIZig0O4L4Rq4d0NQv99xkIGNengxYfLElTI3n%2Blo4qZ7edbzLOwRUhuu%2F68lEkzXdWq9wuR2Wh3QDGl6bmFgJO7DfPEi8eRnoHcLN6TaAhwi7ewukEyBFHUy4yaXvNWs15KFt3tWjrZBzFrnM3QYN5vXckL%2FoU6PYd6U9D5h3FsfmlDczewmhQaSBzHKkCkQkue4yNC77ltxYrQ00BY6HLpgSkHLJejHQg%2F%2FfYs8eX11hGkZcXqVHCrX73VHShauLD0H9wE%2BHneGG9SNVh9ZAjxlRfd%2BbMOSBe%2BcFqMlv%2FRPJew1UPcRh%2BcUf6dv0YLIZEw3mXpOk%2FAaimQEsPpkwrDDr4uUITxE%2FVvwjrcY1x%2BmqqxLlvTNiWjCdKZbRM%2FLjBEhqMgcvG6IJS%2Fll6XkjTJ2BfeOkhd4r53Qs1sINsZB450DHRV9SuJCV%2Be5obOsw3GeZeB8I7RGd01V%2FPa5cUrnc8dMrl9gQww5LD%2FtgY6sQGEoDpPPGTo%2Bkmx2mz03L35xMZe0Mdp7Y27QOHulY%2FWeHW0SGiOQIcJB3Va1cvp%2Fn9hTIZwyEX45utfTqAOSVJUQ0RqvsFqu8QwlgM7OxEWVLtSMn6z%2FuuyCPcYWMrWtFyEBzyZ3%2Bpqvf0S2LKJPFzWABxofP%2BpA5bEDDCwzjYaR%2Fy7jIh7ZAImIJwXYNYketf9CEP0YU7WHBQmtPuzyRRd%2FPa7W7ondBhLh0OStDwQLFA%3D&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Date=20240910T060944Z&X-Amz-SignedHeaders=host&X-Amz-Expires=300&X-Amz-Credential=ASIAQ3PHCVTY5IKQZ3FD%2F20240910%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Signature=a36ec18d4a9140dc350b0d4a0eb4440e3ea8e4563a3357aefa2569cafa23550b&hash=99dd9adee29cfbbe7252e6030a56fc2b84fa1c3ad3b8598e30f0085a389b2c6c&host=68042c943591013ac2b2430a89b270f6af2c76d8dfd086a07176afe7c76c2c61&pii=S2352340924005584&tid=spdf-a4150106-d0f3-4789-84a7-075ca1d600ad&sid=0873a5bd51e572450a0af2f40180fda8c32fgxrqb&type=client&tsoh=d3d3LnNjaWVuY2VkaXJlY3QuY29t&ua=1f055a03570055555c&rr=8c0d403c08499e5a&cc=qa"
      },
      "dialects": [
        "msa"
      ],
      "size": "3,670 sentences",
      "year": 2024,
      "notes": "A dataset for Arabic Word Sense Disambiguation (WSD) consisting of 3670 labeled examples of 100 polysemous Arabic words.",
      "metrics": null
    },
    {
      "id": "xlm-roberta-morocco",
      "name": "XLM-RoBERTa-Morocco",
      "type": "llm",
      "country": "MA",
      "org": "atlasia",
      "license": "mit",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "hf": "https://huggingface.co/atlasia/XLM-RoBERTa-Morocco"
      },
      "dialects": [
        "magh"
      ],
      "year": 2025,
      "notes": "XLM-RoBERTa-large adapted with masked-LM training for Moroccan Darija on atlasia/Atlaset; access gated.",
      "base_model": [
        "facebookai/xlm-roberta-large"
      ],
      "metrics": {
        "downloads": 0,
        "likes": 4,
        "lastModified": "2025-03-06"
      }
    },
    {
      "id": "yaraspell",
      "name": "YaraSpell",
      "type": "tool",
      "country": "DZ",
      "org": "linuxscout",
      "license": "gpl-3.0",
      "modality": "text",
      "tasks": [
        "spell-checking"
      ],
      "links": {
        "github": "https://github.com/linuxscout/yaraspell"
      },
      "year": 2015,
      "notes": "Simplified Arabic spell checker; maintainer is Algerian.",
      "metrics": null
    },
    {
      "id": "yemeni-proverbs-dataset",
      "name": "Yemeni Proverbs dataset",
      "type": "dataset",
      "country": "INTL",
      "org": "Nasser Thmer",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "nlp"
      ],
      "links": {
        "github": "https://github.com/NasserThmer/Yemeni-Proverbs-dataset-code"
      },
      "dialects": [
        "yemeni"
      ],
      "notes": "Dataset and code accompanying the Yemeni Proverbs dataset paper.",
      "metrics": null
    },
    {
      "id": "ymgd",
      "name": "YMGD",
      "type": "benchmark",
      "country": "INTL",
      "org": "IBB University",
      "license": "cc-by-4.0",
      "modality": "speech",
      "tasks": [
        "evaluation",
        "dialect-id"
      ],
      "links": {
        "paper": "https://doi.org/10.5281/zenodo.19543208"
      },
      "dialects": [
        "yemeni"
      ],
      "size": "1,150 sentences",
      "year": 2026,
      "notes": "A benchmark for Arabic Yemeni musical traditions with five genres.",
      "metrics": null
    },
    {
      "id": "youdacc",
      "name": "YouDACC",
      "type": "dataset",
      "country": "QA",
      "org": "CMU-Q",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "dialect-id"
      ],
      "links": {
        "paper": "https://aclanthology.org/L14-1456/"
      },
      "dialects": [
        "egy",
        "gulf",
        "iraqi",
        "lev",
        "magh"
      ],
      "year": 2014,
      "notes": "YouTube Dialectal Arabic Commentary Corpus of user comments across five dialect groups (LREC 2014).",
      "metrics": null
    },
    {
      "id": "zaebuc",
      "name": "ZAEBUC",
      "type": "dataset",
      "country": "AE",
      "org": "NYU Abu Dhabi",
      "license": "cc-by-nc-4.0",
      "modality": "text",
      "tasks": [
        "essays",
        "annotation"
      ],
      "links": {
        "website": "https://sites.google.com/view/zaebuc/home",
        "paper": "http://www.lrec-conf.org/proceedings/lrec2022/pdf/2022.lrec-1.9.pdf"
      },
      "dialects": [
        "msa"
      ],
      "size": "33,300 tokens",
      "year": 2022,
      "tags": [
        "multilingual"
      ],
      "notes": "The corpus is an annotated Arabic-English bilingual writer corpus comprising short essays by first-year university students at Zayed University.",
      "metrics": null
    },
    {
      "id": "zol-roberta",
      "name": "Zol-RoBERTa",
      "type": "llm",
      "country": "INTL",
      "org": "Duaa Alshareef",
      "license": "unknown",
      "modality": "text",
      "tasks": [
        "encoder"
      ],
      "links": {
        "github": "https://github.com/DuaaAlshareef/Sudanese-Arabic-Dialect-Encoding"
      },
      "dialects": [
        "sudanese"
      ],
      "notes": "RoBERTa language model for Sudanese Arabic.",
      "metrics": null
    }
  ],
  "wanted": [
    {
      "id": "open-kuwaiti-qatari-tts",
      "title": "Open Arabic TTS with a Kuwaiti or Qatari voice",
      "why": "No openly licensed text-to-speech model with a Kuwaiti or Qatari voice is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=tts&country=KW,QA&license=open"
    },
    {
      "id": "moroccan-ocr-dataset",
      "title": "Open Moroccan Arabic OCR dataset",
      "why": "The only Moroccan OCR dataset listed has no clear license, so no openly licensed one is available.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#q=ocr&type=dataset&country=MA&license=open"
    },
    {
      "id": "sudanese-asr",
      "title": "Sudanese Arabic speech dataset for ASR",
      "why": "A Sudanese ASR model exists, but no Sudanese speech dataset to train or test one is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#q=asr&type=dataset&country=SD"
    },
    {
      "id": "mauritania-anything",
      "title": "Anything from Mauritania",
      "why": "Mauritania has no listed model, dataset, or tool of any kind.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=llm,asr,tts,ocr,embedding,dataset,tool,benchmark&country=MR"
    },
    {
      "id": "yemeni-dataset",
      "title": "Open Yemeni speech dataset",
      "why": "No openly licensed Yemeni dataset for speech recognition or synthesis is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=YE&license=open"
    },
    {
      "id": "sudanese-yemeni-iraqi-embedding",
      "title": "Embedding model for Sudanese, Yemeni or Iraqi dialects",
      "why": "No Arabic embedding model targets the Sudanese, Yemeni, or Iraqi dialects.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=embedding&dialect=sudanese,yemeni,iraqi"
    },
    {
      "id": "iraqi-tts",
      "title": "Open-licensed Iraqi Arabic TTS",
      "why": "The only TTS covering Iraqi has no clear license, so no openly licensed one is available.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=tts&dialect=iraqi&license=open"
    },
    {
      "id": "libyan-corpus",
      "title": "Libyan Arabic corpus",
      "why": "No dataset from or about Libya is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=LY"
    },
    {
      "id": "open-syrian-dialect-dataset",
      "title": "Open Syrian-dialect dataset",
      "why": "No openly licensed dataset of Syrian Levantine Arabic is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=SY&dialect=lev&license=open"
    },
    {
      "id": "palestinian-asr",
      "title": "Palestinian Arabic speech recognition",
      "why": "No speech recognition model from Palestine is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=asr&country=PS"
    },
    {
      "id": "omani-dialect-dataset",
      "title": "Open Omani dialect dataset",
      "why": "Omani datasets are listed, but none has a clear open license.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=OM&license=open"
    },
    {
      "id": "kuwaiti-dataset",
      "title": "Open Kuwaiti Arabic dataset",
      "why": "The only Kuwaiti dataset listed has no clear license.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=KW&license=open"
    },
    {
      "id": "bahraini-dataset",
      "title": "Open Bahraini Arabic dataset",
      "why": "The only Bahraini dataset listed is non-commercial, so no openly licensed one is available.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=dataset&country=BH&license=open"
    },
    {
      "id": "horn-comoros-resource",
      "title": "Resources from the Horn of Africa and Comoros",
      "why": "Nothing from Comoros, Djibouti, or Somalia is listed.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#country=KM,DJ,SO"
    },
    {
      "id": "handwriting-ocr-benchmark",
      "title": "Open Arabic handwriting benchmark",
      "why": "No openly licensed benchmark covers Arabic handwriting recognition.",
      "status": "open",
      "filled_on": null,
      "by": [],
      "recent": false,
      "hash": "#type=benchmark&license=open"
    }
  ],
  "lineage": {
    "roots": [
      {
        "id": "other",
        "label": "Other",
        "label_ar": "أخرى",
        "count": 99
      },
      {
        "id": "qwen",
        "label": "Qwen",
        "label_ar": "كوين",
        "count": 65
      },
      {
        "id": "bert",
        "label": "BERT",
        "label_ar": "بيرت",
        "count": 58
      },
      {
        "id": "whisper",
        "label": "Whisper",
        "label_ar": "ويسبر",
        "count": 56
      },
      {
        "id": "gemma",
        "label": "Gemma",
        "label_ar": "جيما",
        "count": 24
      },
      {
        "id": "from-scratch",
        "label": "From scratch",
        "label_ar": "من الصفر",
        "count": 19
      },
      {
        "id": "llama",
        "label": "Llama",
        "label_ar": "لاما",
        "count": 15
      },
      {
        "id": "cohere",
        "label": "Cohere",
        "label_ar": "Cohere",
        "count": 9
      },
      {
        "id": "t5",
        "label": "T5",
        "label_ar": "تي 5",
        "count": 9
      },
      {
        "id": "xtts",
        "label": "XTTS",
        "label_ar": "XTTS",
        "count": 7
      },
      {
        "id": "bge",
        "label": "BGE",
        "label_ar": "BGE",
        "count": 4
      },
      {
        "id": "f5",
        "label": "F5-TTS",
        "label_ar": "F5-TTS",
        "count": 4
      },
      {
        "id": "xlsr",
        "label": "XLS-R",
        "label_ar": "إكس إل إس-آر",
        "count": 4
      },
      {
        "id": "nllb",
        "label": "NLLB",
        "label_ar": "NLLB",
        "count": 3
      },
      {
        "id": "seamless",
        "label": "SeamlessM4T",
        "label_ar": "SeamlessM4T",
        "count": 2
      },
      {
        "id": "wav2vec",
        "label": "wav2vec",
        "label_ar": "واف تو فيك",
        "count": 2
      },
      {
        "id": "bloom",
        "label": "BLOOM",
        "label_ar": "بلوم",
        "count": 1
      },
      {
        "id": "e5",
        "label": "E5",
        "label_ar": "E5",
        "count": 1
      },
      {
        "id": "falcon",
        "label": "Falcon",
        "label_ar": "فالكون",
        "count": 1
      },
      {
        "id": "mistral",
        "label": "Mistral",
        "label_ar": "ميسترال",
        "count": 1
      },
      {
        "id": "mms",
        "label": "MMS",
        "label_ar": "إم إم إس",
        "count": 1
      }
    ],
    "edges": [
      [
        "3arab-tts-500m-v2",
        "3arab-tts-500m-v2-voicedesign"
      ],
      [
        "50m-2048-emhotob",
        "50m-darija-english-v1"
      ],
      [
        "50m-2048-emhotob",
        "nawah-50m-rag-support-2k"
      ],
      [
        "50m-2048-emhotob",
        "nawah-math-reasoning"
      ],
      [
        "ahmedzaky1/dimi-embedding-v2",
        "dimi-embedding-v4"
      ],
      [
        "alibaba-nlp/gte-multilingual-reranker-base",
        "mizan-rerank-v2"
      ],
      [
        "allam-7b-instruct-preview",
        "bahraini-dialect-llm"
      ],
      [
        "allam-7b-instruct-preview",
        "mawrooth-allam-7b-lora"
      ],
      [
        "allam-7b-instruct-preview",
        "yehia"
      ],
      [
        "allam-7b-instruct-preview",
        "yehia-7b-preview"
      ],
      [
        "answerdotai/modernbert-base",
        "aramodernbert-topic-classifier"
      ],
      [
        "answerdotai/modernbert-base",
        "modernarabert"
      ],
      [
        "answerdotai/modernbert-base",
        "modernbert-arabic"
      ],
      [
        "aoi-ot/vibevoice-large",
        "vibevoice-arabic-z"
      ],
      [
        "aoi-ot/vibevoice-large",
        "vibevoice-egy"
      ],
      [
        "arabart",
        "adabtranslate-darija"
      ],
      [
        "arabert-all-nli-triplet-matryoshka",
        "gate-arabert-v0"
      ],
      [
        "arabertv02",
        "arabert-all-nli-triplet-matryoshka"
      ],
      [
        "arabertv02",
        "arabic-colbert-100k"
      ],
      [
        "arabertv02",
        "arabic-retrieval-v1-0"
      ],
      [
        "arabertv02",
        "arabic-sbert-100k"
      ],
      [
        "arabertv02",
        "arabic-sts-matryoshka-v2"
      ],
      [
        "arabertv02",
        "arabic-triplet-matryoshka-v2"
      ],
      [
        "arabertv02",
        "ararest-arabic-restaurant-reviews-sentiment-analysis"
      ],
      [
        "arabertv02",
        "camel-readability-arabertv02"
      ],
      [
        "arabertv02",
        "dimi-embedding"
      ],
      [
        "arabertv02",
        "silma-embedding-matryoshka-v0-1"
      ],
      [
        "arabiangpt",
        "qa-finetuned-arabiangpt-01b"
      ],
      [
        "arabic-english-bge-m3",
        "muffakir-embedding-v2"
      ],
      [
        "arabic-english-handwritten-ocr-v3",
        "ketaba-ocr-lora"
      ],
      [
        "arabic-f5-tts-v2",
        "f5tts-algerian-darja"
      ],
      [
        "arabic-f5-tts-v2",
        "hadra-tts-f5"
      ],
      [
        "arabic-triplet-matryoshka-v2",
        "arabic-reranker"
      ],
      [
        "arabic-triplet-matryoshka-v2",
        "gate-arabert-v1"
      ],
      [
        "arabic-triplet-matryoshka-v2",
        "namaa-reranker"
      ],
      [
        "aragpt2",
        "fanar-2-diwan"
      ],
      [
        "aramodernbert-base-v1-0",
        "aramodernbert-base-sts"
      ],
      [
        "arat5",
        "arabic-text-correction"
      ],
      [
        "arat5v2-base-1024",
        "arastyletransfer-21"
      ],
      [
        "arat5v2-base-1024",
        "masrawy-bilingual-v1"
      ],
      [
        "arat5v2-base-1024",
        "saudispell-arat5"
      ],
      [
        "arat5v2-base-1024",
        "shami-mt"
      ],
      [
        "atlasia/terjman-large-v1.2",
        "terjman-large"
      ],
      [
        "baai/bge-m3",
        "arabic-english-bge-m3"
      ],
      [
        "baai/bge-m3",
        "bge-m3-law"
      ],
      [
        "baai/bge-reranker-base",
        "arabic-semantic-highlighter"
      ],
      [
        "baidu/unlimited-ocr",
        "unlimited-ocr-quran-uthmani"
      ],
      [
        "bert-base-arabertv02-twitter",
        "shamibert"
      ],
      [
        "bert-base-arabertv2",
        "arabic-sentiment-model"
      ],
      [
        "bert-base-arabic-camelbert-da-sentiment",
        "bert-base-arabic-hate-speech"
      ],
      [
        "bert-mini-arabic",
        "dialect-router-v0-1"
      ],
      [
        "bigscience/bloom-560m",
        "darija-text-generation"
      ],
      [
        "black-forest-labs/flux.1-schnell",
        "fanar-2-oryx-ig"
      ],
      [
        "bounharabdelaziz/modernbert-morocco",
        "modernbert-morocco-sentence-embeddings-v0-2-bs-32-lr-2e-05-ep-2-wp-0-0"
      ],
      [
        "bounharabdelaziz/xlm-roberta-morocco",
        "morocco-darija-sentence-embedding-v0-2"
      ],
      [
        "cardiffnlp/twitter-xlm-roberta-base-sentiment",
        "sentimentareng"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-jordanian-dialect"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-speech-tashkeel-2b"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-transcribe-arabic-07-2026-dialectal"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-transcribe-arabic-07-2026-dialectal-v2"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-transcribe-arabic-07-2026-int4"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "cohere-transcribe-arabic-cpu"
      ],
      [
        "cohere-transcribe-arabic-07-2026",
        "namaa-saudi-asr-v1"
      ],
      [
        "coherelabs/c4ai-command-r7b-12-2024",
        "command-r7b-arabic"
      ],
      [
        "coherelabs/cohere-transcribe-03-2026",
        "cohere-transcribe-arabic-07-2026"
      ],
      [
        "convaiinnovations/laya",
        "jev-ar"
      ],
      [
        "convaiinnovations/laya-multilingual",
        "laya-ara-rag"
      ],
      [
        "deepseek-ai/deepseek-r1-distill-qwen-1.5b",
        "octopus"
      ],
      [
        "deepseek-ai/deepseek-r1-distill-qwen-1.5b",
        "tinyoctopus"
      ],
      [
        "distilbert/distilbert-base-multilingual-cased",
        "inference-free-splade-distilbert-base-arabic-cased-nq"
      ],
      [
        "dziribert",
        "dz-emobert"
      ],
      [
        "eurobert/eurobert-210m",
        "araeurobert-210m"
      ],
      [
        "eurobert/eurobert-610m",
        "araeurobert-610m"
      ],
      [
        "facebook/hubert-base-ls960",
        "hubert-arabic-spoken-dialect-classifier"
      ],
      [
        "facebook/mms-300m",
        "mms-300m-arabic-dialect-identifier"
      ],
      [
        "facebook/nllb-200-3.3b",
        "terjman-supreme-v1"
      ],
      [
        "facebook/nllb-200-distilled-600m",
        "darija-to-english-2"
      ],
      [
        "facebook/nllb-200-distilled-600m",
        "jordanian-to-fusha-model"
      ],
      [
        "facebook/nougat-base",
        "arabic-base-nougat"
      ],
      [
        "facebook/nougat-small",
        "arabic-small-nougat"
      ],
      [
        "facebook/w2v-bert-2.0",
        "muaalem-model-v3-2"
      ],
      [
        "facebook/w2v-bert-2.0",
        "recitation-segmenter-v2"
      ],
      [
        "facebook/wav2vec2-base",
        "wav2vec2-base-word-by-word-quran-asr"
      ],
      [
        "facebook/wav2vec2-base",
        "wav2vec2-quran-phonetics"
      ],
      [
        "facebook/wav2vec2-large-xlsr-53",
        "wav2vec2-arabic-phoneme-asr"
      ],
      [
        "facebook/wav2vec2-large-xlsr-53",
        "wav2vec2-large-xlsr-arabic"
      ],
      [
        "facebook/wav2vec2-large-xlsr-53",
        "wav2vec2-large-xlsr-moroccan-darija"
      ],
      [
        "facebookai/xlm-roberta-base",
        "span-marker-xlm-roberta-base-ar"
      ],
      [
        "facebookai/xlm-roberta-large",
        "arabic-english-sts-matryoshka-v2-0"
      ],
      [
        "facebookai/xlm-roberta-large",
        "arabic-sts-matryoshka"
      ],
      [
        "facebookai/xlm-roberta-large",
        "xlm-roberta-morocco"
      ],
      [
        "fanar-1-9b",
        "fanar-math-r1-grpo"
      ],
      [
        "fibonacciai/fibonacci-1-en-8b-chat.p1_5",
        "fibonacci-2-14b"
      ],
      [
        "from-scratch",
        "allam"
      ],
      [
        "from-scratch",
        "allam-7b-instruct-preview"
      ],
      [
        "from-scratch",
        "aragpt2"
      ],
      [
        "from-scratch",
        "aragpt2-base"
      ],
      [
        "from-scratch",
        "aragpt2-large"
      ],
      [
        "from-scratch",
        "aragpt2-medium"
      ],
      [
        "from-scratch",
        "jais"
      ],
      [
        "from-scratch",
        "jais-13b"
      ],
      [
        "from-scratch",
        "jais-2-70b-chat"
      ],
      [
        "from-scratch",
        "jais-2-8b-chat"
      ],
      [
        "from-scratch",
        "jais-family-30b-8k"
      ],
      [
        "from-scratch",
        "jais-family-590m"
      ],
      [
        "gate-arabert-v1",
        "gate-reranker-v1"
      ],
      [
        "google-bert/bert-base-multilingual-cased",
        "arabic-base-all-nli-stsb-quora"
      ],
      [
        "google-t5/t5-small",
        "algerian-dialect-translation"
      ],
      [
        "google/embeddinggemma-300m",
        "aragemma-embedding-300m"
      ],
      [
        "google/gemma-2-27b-it",
        "atlas-chat"
      ],
      [
        "google/gemma-2-27b-it",
        "atlas-chat-27b"
      ],
      [
        "google/gemma-2-2b-it",
        "atlas-chat"
      ],
      [
        "google/gemma-2-2b-it",
        "atlas-chat-2b"
      ],
      [
        "google/gemma-2-9b",
        "fanar-1-9b"
      ],
      [
        "google/gemma-2-9b-it",
        "atlas-chat"
      ],
      [
        "google/gemma-2-9b-it",
        "atlas-chat-9b"
      ],
      [
        "google/gemma-2-9b-it",
        "barka"
      ],
      [
        "google/gemma-2b-it",
        "fine-tuning-gemma-2b-it-for-arabic"
      ],
      [
        "google/gemma-3-12b-pt",
        "nile-chat"
      ],
      [
        "google/gemma-3-12b-pt",
        "nile-chat-12b"
      ],
      [
        "google/gemma-3-1b-pt",
        "arabic-gec-v1"
      ],
      [
        "google/gemma-3-270m",
        "aisa-ar-functioncall-think"
      ],
      [
        "google/gemma-3-27b-it",
        "gemmaroc-27b-it"
      ],
      [
        "google/gemma-3-27b-pt",
        "fanar-2-27b-instruct"
      ],
      [
        "google/gemma-3-4b-it",
        "arabic-legal-documents-ocr-1-0"
      ],
      [
        "google/gemma-3-4b-pt",
        "nile-chat"
      ],
      [
        "google/gemma-3-4b-pt",
        "nile-chat-4b"
      ],
      [
        "google/gemma-4-12b-it",
        "gemma-iraqi-finetune-v2"
      ],
      [
        "google/gemma-4-e2b-it",
        "gemma-4-e2b-arabic-english-vision"
      ],
      [
        "google/gemma-4-e2b-it",
        "yemeni-arabic-assistant"
      ],
      [
        "google/gemma-4-e4b-it",
        "gemma4-e4b-claims-comparison"
      ],
      [
        "google/medgemma-4b-it",
        "dr-ai-v2"
      ],
      [
        "heshamharoon/egy_llama3",
        "llama-3-instruct-slerp-arabic"
      ],
      [
        "hexgrad/kokoro-82m",
        "nabra-82m-v0-1"
      ],
      [
        "inception42/jais-adapted-13b",
        "jais-adapted"
      ],
      [
        "inception42/jais-family-13b",
        "jais-family-chat"
      ],
      [
        "intfloat/multilingual-e5-base",
        "algerianme5"
      ],
      [
        "jais-13b",
        "jais-13b-chat"
      ],
      [
        "jais-family-590m",
        "jais-family-590m-chat"
      ],
      [
        "jhu-clsp/mmbert-base",
        "mmbert-base-arabic-nli"
      ],
      [
        "jinaai/jina-embeddings-v3",
        "zarra"
      ],
      [
        "k2-fsa/omnivoice",
        "darija-omnivoice-kore-v1"
      ],
      [
        "k2-fsa/omnivoice",
        "lahgtna-omnivoice-v2"
      ],
      [
        "liquidai/lfm2-350m",
        "arabic-summarization"
      ],
      [
        "liquidai/lfm2-350m",
        "english-moroccan-darija-v1"
      ],
      [
        "liquidai/lfm2-350m",
        "hala-350m"
      ],
      [
        "liquidai/lfm2-700m",
        "tashkeel-700m"
      ],
      [
        "liquidai/lfm2.5-1.2b-instruct",
        "lfm2-5-1-2b-instruct-saudi-dialect"
      ],
      [
        "liquidai/lfm2.5-350m",
        "jameamt-ar-en-350m"
      ],
      [
        "livekit/turn-detector",
        "livekit-turn-detector-arabic"
      ],
      [
        "marbertv2",
        "marbert-all-nli-triplet-matryoshka"
      ],
      [
        "marbertv2",
        "marbertv2-arabic-written-dialect-classifier"
      ],
      [
        "marbertv2",
        "marbertv2-finetuned-egyptian-hate-speech-detection"
      ],
      [
        "marbertv2",
        "masribert-v4"
      ],
      [
        "marbertv2",
        "sa-bert-v1"
      ],
      [
        "marbertv2",
        "viobert-v3"
      ],
      [
        "meta-llama/llama-2-13b-hf",
        "acegpt-13b"
      ],
      [
        "meta-llama/llama-2-7b-hf",
        "acegpt"
      ],
      [
        "meta-llama/llama-3.1-70b",
        "llama-3-3"
      ],
      [
        "meta-llama/llama-3.1-8b-instruct",
        "llamalens"
      ],
      [
        "meta-llama/llama-3.2-1b",
        "octopus"
      ],
      [
        "meta-llama/llama-3.2-3b-instruct",
        "math-arabic-llama-3-2-3b-instruct"
      ],
      [
        "meta-llama/llama-prompt-guard-2-86m",
        "ara-prompt-guard-v0"
      ],
      [
        "meta-llama/llama-prompt-guard-2-86m",
        "ara-prompt-guard-v1"
      ],
      [
        "meta-llama/meta-llama-3-8b",
        "egyptian-arabic-translator-llama-3-8b"
      ],
      [
        "meta-llama/meta-llama-3-8b-instruct",
        "arabic-orpo-llama3-8b"
      ],
      [
        "meta-llama/meta-llama-3-8b-instruct",
        "llama-3-instruct-slerp-arabic"
      ],
      [
        "microsoft/harrier-oss-v1-0.6b",
        "harrier-arabic-matryoshka-0-6b"
      ],
      [
        "microsoft/speecht5_tts",
        "speecht5-tts-arabic"
      ],
      [
        "microsoft/trocr-base-handwritten",
        "trocr-tunisian-arabic"
      ],
      [
        "mistralai/mistral-7b-v0.1",
        "mistral-7b-v0-1-arabic"
      ],
      [
        "mohamed2811/muffakir_embedding",
        "badr-embedding-v0"
      ],
      [
        "muaalem-model-v3-2",
        "hifzguide-muaalem-mini"
      ],
      [
        "multilingual-chatterbox",
        "chatterbox-egyptian-v0"
      ],
      [
        "multilingual-chatterbox",
        "chatterbox-multilingual-finetuned-arabic"
      ],
      [
        "multilingual-chatterbox",
        "lahgtna-chatterbox-v1"
      ],
      [
        "multilingual-chatterbox",
        "namaa-egyptian-tts"
      ],
      [
        "multilingual-chatterbox",
        "namaa-saudi-tts"
      ],
      [
        "muno459/zipformer_p-arabic",
        "zipformer-p-quran"
      ],
      [
        "nabra-82m-v0-1",
        "kemetone"
      ],
      [
        "nilechat-3b-base",
        "lahjamt"
      ],
      [
        "nineninesix/kani-tts-400m-0.3-pt",
        "kani-tts-400m-ar"
      ],
      [
        "nousresearch/llama-2-7b-chat-hf",
        "llama-2-7b-chat-ar"
      ],
      [
        "nvidia-fastconformer-arabic-diacritics",
        "egyptalk-asr-v2"
      ],
      [
        "nvidia-fastconformer-arabic-diacritics",
        "fastconformer-quran"
      ],
      [
        "nvidia-fastconformer-arabic-diacritics",
        "stt-ar-fastconformer-hybrid-large-streaming-pcd-v1-1-mirror"
      ],
      [
        "nvidia/magpie_tts_multilingual_357m",
        "magpie-saqr-najdi"
      ],
      [
        "nvidia/magpie_tts_multilingual_357m",
        "magpie-tts-saudi-arabic"
      ],
      [
        "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning",
        "basira-omni-30b-v0-1"
      ],
      [
        "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-bf16",
        "basira-omni-30b-v0-1"
      ],
      [
        "nvidia/nemotron-3.5-asr-streaming-0.6b",
        "nemotron-3-5-asr-streaming-0-6b-jordanian"
      ],
      [
        "nvidia/nemotron-3.5-asr-streaming-0.6b",
        "nemotron-asr-arabic-dialectal-v2"
      ],
      [
        "oddadmix/emhotob-25m",
        "emhotob-25m-english-msa-v1"
      ],
      [
        "oddadmix/nawah-50m-rag-chat-8k",
        "nawah-50m-rag-chat"
      ],
      [
        "oddadmix/nawah-bert-6m-v2",
        "nawah-dialect-bert-6m"
      ],
      [
        "oddadmix/nawah-bert-6m-v2",
        "nawah-router-bert-6m-bilingual"
      ],
      [
        "oddadmix/whisper-large-v3-tunisian-codeswitch-asr-v2",
        "whisperv3-tunisian-codeswitch"
      ],
      [
        "omartificial-intelligence-space/sa-sts-embeddings-0.2b",
        "sa-retrieval-embeddings-0-2b"
      ],
      [
        "omarxadel/wav2vec2-large-xlsr-53-arabic-egyptian",
        "egyptian-arabic-wav2vec2-xlsr-53"
      ],
      [
        "openai-community/gpt2",
        "hassaniya-gpt2-talk"
      ],
      [
        "openai-whisper-large-v3",
        "arabic-morocco-speech-to-text"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-arabic-dialects-v5"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-libyan"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-ar"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-arabic-byne"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-arabic-dialectal-v2"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-egyptian-arabic"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-tarteel"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-large-v3-turbo"
      ],
      [
        "openai-whisper-large-v3",
        "whisper-largev3-medical"
      ],
      [
        "openai-whisper-large-v3",
        "whisperlevantine"
      ],
      [
        "openai/whisper-base",
        "bahraini-arabic-asr"
      ],
      [
        "openai/whisper-base",
        "bahraini-arabic-whisper"
      ],
      [
        "openai/whisper-base",
        "faster-whisper-base-ar-quran"
      ],
      [
        "openai/whisper-base",
        "whisper-base-arabic"
      ],
      [
        "openai/whisper-large",
        "whisper-large-arabic-cv-11"
      ],
      [
        "openai/whisper-large-v2",
        "whisper-yemeni"
      ],
      [
        "openai/whisper-medium",
        "hadra-asr-whisper-medium"
      ],
      [
        "openai/whisper-medium",
        "tarbiyah-ai-whisper-medium-merged"
      ],
      [
        "openai/whisper-medium",
        "whisper-algerian-darja-medium"
      ],
      [
        "openai/whisper-medium",
        "whisper-egyptian-arabic"
      ],
      [
        "openai/whisper-medium",
        "whisper-m-quran-lora-dataset-mix"
      ],
      [
        "openai/whisper-medium",
        "whisper-medium-arabic"
      ],
      [
        "openai/whisper-medium",
        "whisper-medium-darija"
      ],
      [
        "openai/whisper-medium",
        "whisper-medium-finetuned-sada-asr"
      ],
      [
        "openai/whisper-small",
        "arazn-whisper-small"
      ],
      [
        "openai/whisper-small",
        "quran-whisper-tiny-v1"
      ],
      [
        "openai/whisper-small",
        "stt-arabic-whisper-finetuned-diactires"
      ],
      [
        "openai/whisper-small",
        "tadabur-whisper-small"
      ],
      [
        "openai/whisper-small",
        "voho-saudi-stt-small"
      ],
      [
        "openai/whisper-small",
        "whisper-arabic-small"
      ],
      [
        "openai/whisper-small",
        "whisper-small-arabic-dialectal-v2"
      ],
      [
        "openai/whisper-small",
        "whisper-small-codeswitching-arabicenglish"
      ],
      [
        "openai/whisper-small",
        "whisper-small-darija"
      ],
      [
        "openai/whisper-small",
        "whisper-small-egyptian-codeswitch"
      ],
      [
        "openai/whisper-small",
        "whisper-small-for-quran"
      ],
      [
        "openai/whisper-small",
        "whisper-small-full-finetune"
      ],
      [
        "openai/whisper-small",
        "whisper-small-libyan"
      ],
      [
        "openai/whisper-small",
        "whisper-small-quran-lora-dataset-mix"
      ],
      [
        "openai/whisper-small",
        "whisper-small-quran-lora-everyayah"
      ],
      [
        "openai/whisper-small",
        "whisper-small-tunisian-arabic"
      ],
      [
        "openai/whisper-small",
        "whisper-small-yemeni"
      ],
      [
        "openai/whisper-small",
        "whisper-with-augmentation-small-arabic-with-diacritics"
      ],
      [
        "openbmb/voxcpm2",
        "fasee7-najdi-small"
      ],
      [
        "opus-mt-ar-en",
        "darija-to-english"
      ],
      [
        "opus-mt-en-ar",
        "masrawy-english-arabic-translator"
      ],
      [
        "opus-mt-en-ar",
        "terjman-nano-v1"
      ],
      [
        "opus-mt-tc-big-en-ar",
        "english-egyptian-arabic-translator"
      ],
      [
        "opus-mt-tc-big-en-ar",
        "english-to-darija-2"
      ],
      [
        "opus-mt-tc-big-en-ar",
        "masrawy-translator"
      ],
      [
        "outeai/outetts-0.2-500m",
        "darijatts-v0-1-500m"
      ],
      [
        "qwen/qwen-vl",
        "mubsir-qwen-2b-vl"
      ],
      [
        "qwen/qwen1.5-32b",
        "acegpt-v2-32b-chat"
      ],
      [
        "qwen/qwen2-72b",
        "calme-2-2"
      ],
      [
        "qwen/qwen2.5-0.5b",
        "al-atlas-0-5b"
      ],
      [
        "qwen/qwen2.5-0.5b",
        "al-atlas-llm-0-5b"
      ],
      [
        "qwen/qwen2.5-0.5b",
        "rightnow-arabic-0-5b-turbo"
      ],
      [
        "qwen/qwen2.5-1.5b-instruct",
        "masrygpt-chat-1-5b"
      ],
      [
        "qwen/qwen2.5-1.5b-instruct",
        "python-assistant"
      ],
      [
        "qwen/qwen2.5-1.5b-instruct",
        "qwen2-5-1-5b-amiya-palestinian"
      ],
      [
        "qwen/qwen2.5-3b",
        "nilechat-3b"
      ],
      [
        "qwen/qwen2.5-3b",
        "nilechat-3b-base"
      ],
      [
        "qwen/qwen2.5-7b-instruct",
        "fikr-7b-reasoning"
      ],
      [
        "qwen/qwen2.5-7b-instruct",
        "meraj-mini"
      ],
      [
        "qwen/qwen2.5-7b-instruct",
        "probel-mtl"
      ],
      [
        "qwen/qwen2.5-7b-instruct",
        "qwen2-5-7b-jordanian"
      ],
      [
        "qwen/qwen2.5-coder-1.5b",
        "ayncoding-qwen2-5-coder-1-5b"
      ],
      [
        "qwen/qwen2.5-vl-3b-instruct",
        "arabic-english-handwritten-ocr-v3"
      ],
      [
        "qwen/qwen2.5-vl-3b-instruct",
        "arabic-handwritten-ocr-4bit-qwen2-5-vl-3b-v2"
      ],
      [
        "qwen/qwen2.5-vl-7b-instruct",
        "fanar-2-oryx-ivu"
      ],
      [
        "qwen/qwen3-0.6b",
        "voho-saudi-speak-0-6b"
      ],
      [
        "qwen/qwen3-14b",
        "tam-omani-adapter-tam-omani-v3"
      ],
      [
        "qwen/qwen3-235b-a22b-instruct-2507",
        "vivo-c-v1"
      ],
      [
        "qwen/qwen3-30b-a3b-instruct-2507",
        "karnak-40b-v1-0"
      ],
      [
        "qwen/qwen3-30b-a3b-instruct-2507",
        "karnak-llm"
      ],
      [
        "qwen/qwen3-4b",
        "arabic-sahm"
      ],
      [
        "qwen/qwen3-4b",
        "kallamni-4b-v1"
      ],
      [
        "qwen/qwen3-4b",
        "qwen3-4b-oman-qlora"
      ],
      [
        "qwen/qwen3-4b-base",
        "qwen3-4b-algerian-darja"
      ],
      [
        "qwen/qwen3-4b-instruct-2507",
        "karnak-6b"
      ],
      [
        "qwen/qwen3-4b-instruct-2507",
        "voho-saudi-chat-4b"
      ],
      [
        "qwen/qwen3-8b",
        "ayncoding-qwen3-8b-slim"
      ],
      [
        "qwen/qwen3-8b",
        "esprit-derja-qwen3-8b"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "lemura-arabic-asr-qwen3"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "moulsot-v0-3"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "qwen3-asr-1-7b-jordanian-dialect-arabic"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "qwen3-asr-arabic-ksa"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "qwen3-asr-arabic-uae"
      ],
      [
        "qwen/qwen3-asr-1.7b",
        "qwencleo-asr"
      ],
      [
        "qwen/qwen3-embedding-0.6b",
        "qwen3-embedding-0-6b-arabic-ecom"
      ],
      [
        "qwen/qwen3-embedding-0.6b",
        "semantic-ar-qwen-embed-0-6b"
      ],
      [
        "qwen/qwen3-tts-12hz-1.7b-base",
        "egy-arabic-qwen3-tts-12hz-1-7b-base"
      ],
      [
        "qwen/qwen3-tts-12hz-1.7b-base",
        "qwen3-5-tts-emirati"
      ],
      [
        "qwen/qwen3-tts-12hz-1.7b-base",
        "qwen3-tts-ksa"
      ],
      [
        "qwen/qwen3-vl-2b-instruct",
        "qwen3-vl-2b-persian-arabic-ocr-v1-0"
      ],
      [
        "qwen/qwen3-vl-4b-instruct",
        "qari-ocr-0-4-0-vl-4b-instruct"
      ],
      [
        "qwen/qwen3-vl-8b-instruct",
        "mtrini-svl-1-1-merged"
      ],
      [
        "qwen/qwen3.5-0.8b",
        "arabic-qwen3-5-ocr-v4"
      ],
      [
        "qwen/qwen3.5-0.8b",
        "egyptian-arabic-eot"
      ],
      [
        "qwen/qwq-32b-preview",
        "arabic-qwq-32b-preview"
      ],
      [
        "qwen2.5vl",
        "baseer-nakba"
      ],
      [
        "qwen3.5",
        "asl-4b-v1"
      ],
      [
        "seamlessm4t-v2",
        "seamless-darija-english"
      ],
      [
        "sentence-transformers/labse",
        "arabic-labse-matryoshka"
      ],
      [
        "sentence-transformers/paraphrase-multilingual-minilm-l12-v2",
        "arabic-minilm-l12-v2-all-nli-triplet"
      ],
      [
        "sentence-transformers/paraphrase-multilingual-mpnet-base-v2",
        "arabic-all-nli-triplet-matryoshka"
      ],
      [
        "sesame/csm-1b",
        "seasmed-fine-tuned-on-common-voice-17-arabic-samehelalfi"
      ],
      [
        "sherif1313/3arab-tts-500m-v1",
        "3arab-tts-500m-v2"
      ],
      [
        "sherif1313/3arablm-4b-fiqh-v1",
        "3arablm-4b-islamic-v2"
      ],
      [
        "silma-ai/silma-embeddding-matryoshka-0.1",
        "silma-embedding-sts-v0-1"
      ],
      [
        "silma-embedding-matryoshka-v0-1",
        "silma-embedding-sts-v0-1"
      ],
      [
        "silma-tts",
        "saudi-tts-v4"
      ],
      [
        "sparkaudio/spark-tts-0.5b",
        "arabic-tts-spark"
      ],
      [
        "sparkaudio/spark-tts-0.5b",
        "spark-tts-arabic"
      ],
      [
        "sparkaudio/spark-tts-0.5b",
        "spark-tts-arabic-complete"
      ],
      [
        "swivid/f5-tts",
        "arabic-f5-tts-v2"
      ],
      [
        "swivid/f5-tts",
        "f5-tts-arabic"
      ],
      [
        "swivid/habibi-tts",
        "f5-tts-egyptian-arabic"
      ],
      [
        "swivid/habibi-tts",
        "habibi-tts-doda-darija"
      ],
      [
        "swivid/habibi-tts",
        "namaa-saudi-tts-v2"
      ],
      [
        "tiiuae/falcon3-7b-base",
        "falcon-arabic"
      ],
      [
        "tomaarsen/mpnet-base-all-nli-triplet",
        "arabic-mpnet-base-all-nli-triplet"
      ],
      [
        "u4rasd/neoarabert_msa",
        "neoarabert-msa-synonym-matryoshka-v1"
      ],
      [
        "unsloth/deepseek-r1-distill-llama-8b-unsloth-bnb-4bit",
        "arabic-deepseek-r1-distill-8b"
      ],
      [
        "unsloth/gemma-3n-e4b-it",
        "masriswitch-gemma3n"
      ],
      [
        "unsloth/gemma-3n-e4b-it",
        "shako-iraqi-4b"
      ],
      [
        "unsloth/gemma-4-e2b-it",
        "yemeni-arabic-assistant"
      ],
      [
        "unsloth/gpt-oss-20b-unsloth-bnb-4bit",
        "gpt-oss-math-ar"
      ],
      [
        "unsloth/granite-4.0-h-small-base",
        "arabic-llm-guard"
      ],
      [
        "unsloth/llama-3-8b-bnb-4bit",
        "arabic-llama3"
      ],
      [
        "unsloth/llama-3.2-1b-instruct",
        "vynis-0-1-2b-instant"
      ],
      [
        "unsloth/meta-llama-3.1-8b-instruct-bnb-4bit",
        "lebanese-llama-3-1-8b"
      ],
      [
        "unsloth/muse-glimmer-30b-unsloth-bnb-4bit",
        "khattvision-muse-glimmer-30b-lora"
      ],
      [
        "unsloth/qwen2-vl-2b-instruct-unsloth-bnb-4bit",
        "qari-ocr-0-1-vl-2b-instruct"
      ],
      [
        "unsloth/qwen2-vl-2b-instruct-unsloth-bnb-4bit",
        "qari-ocr-0-2-2-1-vl-2b-instruct"
      ],
      [
        "unsloth/qwen2.5-3b-instruct-unsloth-bnb-4bit",
        "diraya-3b-instruct-ar"
      ],
      [
        "unsloth/qwen2.5-7b-instruct",
        "qwen2-5-7b-instruct-arabic-yt-merged"
      ],
      [
        "unsloth/qwen2.5-7b-instruct-bnb-4bit",
        "qween7-5-arabic-story-teller-2"
      ],
      [
        "unsloth/qwen2.5-vl-7b-instruct-bnb-4bit",
        "arabic-ocr-qwen2-5-vl-7b-vision"
      ],
      [
        "unsloth/qwen2.5-vl-7b-instruct-bnb-4bit",
        "dimi-arabic-ocr"
      ],
      [
        "unsloth/qwen3-14b",
        "bee1reason-arabic-qwen-14b"
      ],
      [
        "unsloth/qwen3.5-0.8b",
        "katib-qwen3-5-0-8b-0-1"
      ],
      [
        "unsloth/qwen3.5-9b",
        "qwen3-5-9b-saudi-dialect"
      ],
      [
        "urchade/gliner_multi-v2.1",
        "gliner-arabic"
      ],
      [
        "wasmdashai/vits-ar-sa-a",
        "lahja-sa-ahmad-v1"
      ],
      [
        "wasmdashai/vits-ar-sa-huba-v2",
        "lahja-sa-huba-v1"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-l-v3-turbo-quran-lora-dataset-mix"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-large-v3-turbo-ar-quran"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-large-v3-turbo-arabic"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-large-v3-turbo-arabic-dialectal"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-large-v3-turbo-arabic-dialectal-v2"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-large-v3-turbo-darija"
      ],
      [
        "whisper-large-v3-turbo",
        "whisper-turbo-egyptian-codeswitch"
      ],
      [
        "whisper-medium-darija",
        "asr-hassaniya-whisper-medium-v1"
      ],
      [
        "whisper-quran",
        "faster-whisper-base-ar-quran"
      ],
      [
        "whisper-quran",
        "whisper-base-ar-quran-ft-hijaiyah-2"
      ],
      [
        "whisper-quran",
        "whisper-base-quran"
      ],
      [
        "xlm-roberta-base",
        "span-marker-xlm-roberta-base-ar"
      ],
      [
        "xtts-v2",
        "arabic-tts-xtts-v2"
      ],
      [
        "xtts-v2",
        "egtts-v0-1"
      ],
      [
        "xtts-v2",
        "fasih-tts-v1"
      ],
      [
        "xtts-v2",
        "leva-tts"
      ],
      [
        "xtts-v2",
        "niletts-xtts"
      ],
      [
        "xtts-v2",
        "tunisian-tts"
      ],
      [
        "yatharths/miratts",
        "sofelia-tts"
      ],
      [
        "zai-org/glm-ocr",
        "arabic-glm-ocr-v1"
      ],
      [
        "zai-org/glm-ocr",
        "arabic-glm-ocr-v2"
      ]
    ],
    "root_of": {
      "3arab-tts-500m-v2": "other",
      "3arab-tts-500m-v2-voicedesign": "other",
      "3arablm-4b-islamic-v2": "other",
      "50m-2048-emhotob": "other",
      "50m-darija-english-v1": "other",
      "acegpt": "llama",
      "acegpt-13b": "llama",
      "acegpt-v2-32b-chat": "qwen",
      "adabtranslate-darija": "other",
      "aisa-ar-functioncall-think": "gemma",
      "al-atlas-0-5b": "qwen",
      "al-atlas-llm-0-5b": "qwen",
      "algerian-dialect-translation": "t5",
      "algerianme5": "e5",
      "allam": "from-scratch",
      "allam-7b-instruct-preview": "from-scratch",
      "ara-prompt-guard-v0": "llama",
      "ara-prompt-guard-v1": "llama",
      "arabart": "other",
      "arabert-all-nli-triplet-matryoshka": "bert",
      "arabertv02": "bert",
      "arabiangpt": "other",
      "arabic-all-nli-triplet-matryoshka": "other",
      "arabic-base-all-nli-stsb-quora": "bert",
      "arabic-base-nougat": "other",
      "arabic-colbert-100k": "bert",
      "arabic-deepseek-r1-distill-8b": "llama",
      "arabic-english-bge-m3": "bge",
      "arabic-english-handwritten-ocr-v3": "qwen",
      "arabic-english-sts-matryoshka-v2-0": "bert",
      "arabic-f5-tts-v2": "f5",
      "arabic-gec-v1": "gemma",
      "arabic-glm-ocr-v1": "other",
      "arabic-glm-ocr-v2": "other",
      "arabic-handwritten-ocr-4bit-qwen2-5-vl-3b-v2": "qwen",
      "arabic-labse-matryoshka": "other",
      "arabic-legal-documents-ocr-1-0": "gemma",
      "arabic-llama3": "llama",
      "arabic-llm-guard": "other",
      "arabic-minilm-l12-v2-all-nli-triplet": "other",
      "arabic-morocco-speech-to-text": "whisper",
      "arabic-mpnet-base-all-nli-triplet": "other",
      "arabic-ocr-qwen2-5-vl-7b-vision": "qwen",
      "arabic-orpo-llama3-8b": "llama",
      "arabic-qwen3-5-ocr-v4": "qwen",
      "arabic-qwq-32b-preview": "qwen",
      "arabic-reranker": "bert",
      "arabic-retrieval-v1-0": "bert",
      "arabic-sahm": "qwen",
      "arabic-sbert-100k": "bert",
      "arabic-semantic-highlighter": "bge",
      "arabic-sentiment-model": "bert",
      "arabic-small-nougat": "other",
      "arabic-sts-matryoshka": "bert",
      "arabic-sts-matryoshka-v2": "bert",
      "arabic-summarization": "other",
      "arabic-text-correction": "t5",
      "arabic-triplet-matryoshka-v2": "bert",
      "arabic-tts-spark": "other",
      "arabic-tts-xtts-v2": "xtts",
      "araeurobert-210m": "bert",
      "araeurobert-610m": "bert",
      "aragemma-embedding-300m": "gemma",
      "aragpt2": "from-scratch",
      "aragpt2-base": "from-scratch",
      "aragpt2-large": "from-scratch",
      "aragpt2-medium": "from-scratch",
      "aramodernbert-base-sts": "bert",
      "aramodernbert-base-v1-0": "bert",
      "aramodernbert-topic-classifier": "bert",
      "ararest-arabic-restaurant-reviews-sentiment-analysis": "bert",
      "arastyletransfer-21": "t5",
      "arat5": "t5",
      "arat5v2-base-1024": "t5",
      "arazn-whisper-small": "whisper",
      "asl-4b-v1": "qwen",
      "asr-hassaniya-whisper-medium-v1": "whisper",
      "atlas-chat": "gemma",
      "atlas-chat-27b": "gemma",
      "atlas-chat-2b": "gemma",
      "atlas-chat-9b": "gemma",
      "ayncoding-qwen2-5-coder-1-5b": "qwen",
      "ayncoding-qwen3-8b-slim": "qwen",
      "badr-embedding-v0": "other",
      "bahraini-arabic-asr": "whisper",
      "bahraini-arabic-whisper": "whisper",
      "bahraini-dialect-llm": "from-scratch",
      "barka": "gemma",
      "baseer-nakba": "qwen",
      "basira-omni-30b-v0-1": "other",
      "bee1reason-arabic-qwen-14b": "qwen",
      "bert-base-arabertv02-twitter": "bert",
      "bert-base-arabertv2": "bert",
      "bert-base-arabic-camelbert-da-sentiment": "bert",
      "bert-base-arabic-hate-speech": "bert",
      "bert-mini-arabic": "bert",
      "bge-m3-law": "bge",
      "byt5-darija-emphatic": "t5",
      "calme-2-2": "qwen",
      "camel-readability-arabertv02": "bert",
      "chatterbox-egyptian-v0": "other",
      "chatterbox-multilingual-finetuned-arabic": "other",
      "cohere-jordanian-dialect": "cohere",
      "cohere-speech-tashkeel-2b": "cohere",
      "cohere-transcribe-arabic-07-2026": "cohere",
      "cohere-transcribe-arabic-07-2026-dialectal": "cohere",
      "cohere-transcribe-arabic-07-2026-dialectal-v2": "cohere",
      "cohere-transcribe-arabic-07-2026-int4": "cohere",
      "cohere-transcribe-arabic-cpu": "cohere",
      "command-r7b-arabic": "cohere",
      "darija-omnivoice-kore-v1": "other",
      "darija-text-generation": "bloom",
      "darija-to-english": "other",
      "darija-to-english-2": "nllb",
      "darijatts-v0-1-500m": "other",
      "dialect-router-v0-1": "bert",
      "dimi-arabic-ocr": "qwen",
      "dimi-embedding": "bert",
      "dimi-embedding-v4": "other",
      "diraya-3b-instruct-ar": "qwen",
      "dr-ai-v2": "gemma",
      "dz-emobert": "bert",
      "dziribert": "bert",
      "egtts-v0-1": "xtts",
      "egy-arabic-qwen3-tts-12hz-1-7b-base": "qwen",
      "egyptalk-asr-v2": "other",
      "egyptian-arabic-eot": "qwen",
      "egyptian-arabic-translator-llama-3-8b": "llama",
      "egyptian-arabic-wav2vec2-xlsr-53": "xlsr",
      "emhotob-25m-english-msa-v1": "other",
      "emirati-fastpitch-bilingual-v1-0": "other",
      "english-egyptian-arabic-translator": "other",
      "english-moroccan-darija-v1": "other",
      "english-to-darija-2": "other",
      "esprit-derja-qwen3-8b": "qwen",
      "f5-tts-arabic": "f5",
      "f5-tts-egyptian-arabic": "other",
      "f5tts-algerian-darja": "f5",
      "falcon-arabic": "falcon",
      "fanar-1-9b": "gemma",
      "fanar-2-27b-instruct": "gemma",
      "fanar-2-diwan": "from-scratch",
      "fanar-2-oryx-ig": "other",
      "fanar-2-oryx-ivu": "qwen",
      "fanar-math-r1-grpo": "gemma",
      "fasee7-najdi-small": "other",
      "fasih-tts-v1": "xtts",
      "fastconformer-quran": "other",
      "faster-whisper-base-ar-quran": "whisper",
      "fibonacci-2-14b": "other",
      "fikr-7b-reasoning": "qwen",
      "fine-tuning-gemma-2b-it-for-arabic": "gemma",
      "gate-arabert-v0": "bert",
      "gate-arabert-v1": "bert",
      "gate-reranker-v1": "bert",
      "gemma-4-e2b-arabic-english-vision": "gemma",
      "gemma-iraqi-finetune-v2": "gemma",
      "gemma4-e4b-claims-comparison": "gemma",
      "gemmaroc-27b-it": "gemma",
      "gliner-arabic": "other",
      "gpt-oss-math-ar": "other",
      "habibi-tts-doda-darija": "other",
      "hadra-asr-whisper-medium": "whisper",
      "hadra-tts-f5": "f5",
      "hala-350m": "other",
      "harrier-arabic-matryoshka-0-6b": "other",
      "hassaniya-gpt2-talk": "other",
      "hifzguide-muaalem-mini": "bert",
      "hubert-arabic-spoken-dialect-classifier": "bert",
      "inference-free-splade-distilbert-base-arabic-cased-nq": "bert",
      "jais": "from-scratch",
      "jais-13b": "from-scratch",
      "jais-13b-chat": "from-scratch",
      "jais-2-70b-chat": "from-scratch",
      "jais-2-8b-chat": "from-scratch",
      "jais-adapted": "other",
      "jais-family-30b-8k": "from-scratch",
      "jais-family-590m": "from-scratch",
      "jais-family-590m-chat": "from-scratch",
      "jais-family-chat": "other",
      "jameamt-ar-en-350m": "other",
      "jev-ar": "other",
      "jordanian-to-fusha-model": "nllb",
      "kallamni-4b-v1": "qwen",
      "kani-tts-400m-ar": "other",
      "karnak-40b-v1-0": "qwen",
      "karnak-6b": "qwen",
      "karnak-llm": "qwen",
      "katib-qwen3-5-0-8b-0-1": "qwen",
      "kemetone": "other",
      "ketaba-ocr-lora": "qwen",
      "khattvision-muse-glimmer-30b-lora": "other",
      "lahgtna-chatterbox-v1": "other",
      "lahgtna-omnivoice-v2": "other",
      "lahja-sa-ahmad-v1": "other",
      "lahja-sa-huba-v1": "other",
      "lahjamt": "qwen",
      "laya-ara-rag": "other",
      "lebanese-llama-3-1-8b": "llama",
      "lemura-arabic-asr-qwen3": "qwen",
      "leva-tts": "xtts",
      "lfm2-5-1-2b-instruct-saudi-dialect": "other",
      "livekit-turn-detector-arabic": "other",
      "llama-2-7b-chat-ar": "llama",
      "llama-3-3": "llama",
      "llama-3-instruct-slerp-arabic": "llama",
      "llamalens": "llama",
      "magpie-saqr-najdi": "other",
      "magpie-tts-saudi-arabic": "other",
      "marbert-all-nli-triplet-matryoshka": "bert",
      "marbertv2": "bert",
      "marbertv2-arabic-written-dialect-classifier": "bert",
      "marbertv2-finetuned-egyptian-hate-speech-detection": "bert",
      "masrawy-bilingual-v1": "t5",
      "masrawy-english-arabic-translator": "other",
      "masrawy-translator": "other",
      "masribert-v4": "bert",
      "masriswitch-gemma3n": "gemma",
      "masrygpt-chat-1-5b": "qwen",
      "math-arabic-llama-3-2-3b-instruct": "llama",
      "mawrooth-allam-7b-lora": "from-scratch",
      "meraj-mini": "qwen",
      "mistral-7b-v0-1-arabic": "mistral",
      "mizan-rerank-v2": "other",
      "mmbert-base-arabic-nli": "bert",
      "mms-300m-arabic-dialect-identifier": "mms",
      "modernarabert": "bert",
      "modernbert-arabic": "bert",
      "modernbert-morocco-sentence-embeddings-v0-2-bs-32-lr-2e-05-ep-2-wp-0-0": "bert",
      "morocco-darija-sentence-embedding-v0-2": "bert",
      "moulsot-v0-3": "qwen",
      "mtrini-svl-1-1-merged": "qwen",
      "muaalem-model-v3-2": "bert",
      "mubsir-qwen-2b-vl": "qwen",
      "muffakir-embedding-v2": "bge",
      "multilingual-chatterbox": "other",
      "nabra-82m-v0-1": "other",
      "namaa-egyptian-tts": "other",
      "namaa-reranker": "bert",
      "namaa-saudi-asr-v1": "cohere",
      "namaa-saudi-tts": "other",
      "namaa-saudi-tts-v2": "other",
      "nawah-50m-rag-chat": "other",
      "nawah-50m-rag-support-2k": "other",
      "nawah-dialect-bert-6m": "bert",
      "nawah-math-reasoning": "other",
      "nawah-router-bert-6m-bilingual": "bert",
      "nemotron-3-5-asr-streaming-0-6b-jordanian": "other",
      "nemotron-asr-arabic-dialectal-v2": "other",
      "neoarabert": "bert",
      "neoarabert-msa-synonym-matryoshka-v1": "bert",
      "nile-chat": "gemma",
      "nile-chat-12b": "gemma",
      "nile-chat-4b": "gemma",
      "nilechat-3b": "qwen",
      "nilechat-3b-base": "qwen",
      "niletts-xtts": "xtts",
      "noormontai": "other",
      "nvidia-fastconformer-arabic-diacritics": "other",
      "octopus": "qwen",
      "openai-whisper-large-v3": "whisper",
      "opus-mt-ar-en": "other",
      "opus-mt-en-ar": "other",
      "opus-mt-tc-big-en-ar": "other",
      "probel-mtl": "qwen",
      "python-assistant": "qwen",
      "qa-finetuned-arabiangpt-01b": "other",
      "qari-ocr-0-1-vl-2b-instruct": "qwen",
      "qari-ocr-0-2-2-1-vl-2b-instruct": "qwen",
      "qari-ocr-0-4-0-vl-4b-instruct": "qwen",
      "quran-whisper-tiny-v1": "whisper",
      "qween7-5-arabic-story-teller-2": "qwen",
      "qwen2-5-1-5b-amiya-palestinian": "qwen",
      "qwen2-5-7b-instruct-arabic-yt-merged": "qwen",
      "qwen2-5-7b-jordanian": "qwen",
      "qwen3-4b-algerian-darja": "qwen",
      "qwen3-4b-oman-qlora": "qwen",
      "qwen3-5-9b-saudi-dialect": "qwen",
      "qwen3-5-tts-emirati": "qwen",
      "qwen3-asr-1-7b-jordanian-dialect-arabic": "qwen",
      "qwen3-asr-arabic-ksa": "qwen",
      "qwen3-asr-arabic-uae": "qwen",
      "qwen3-embedding-0-6b-arabic-ecom": "qwen",
      "qwen3-tts-ksa": "qwen",
      "qwen3-vl-2b-persian-arabic-ocr-v1-0": "qwen",
      "qwencleo-asr": "qwen",
      "recitation-segmenter-v2": "bert",
      "rightnow-arabic-0-5b-turbo": "qwen",
      "sa-bert-v1": "bert",
      "sa-retrieval-embeddings-0-2b": "other",
      "saudi-tts-v4": "other",
      "saudispell-arat5": "t5",
      "seamless-darija-english": "seamless",
      "seamlessm4t-v2": "seamless",
      "seasmed-fine-tuned-on-common-voice-17-arabic-samehelalfi": "other",
      "semantic-ar-qwen-embed-0-6b": "qwen",
      "sentimentareng": "bert",
      "shako-iraqi-4b": "gemma",
      "shami-mt": "t5",
      "shamibert": "bert",
      "silma-embedding-matryoshka-v0-1": "bert",
      "silma-embedding-sts-v0-1": "other",
      "silma-tts": "other",
      "sofelia-tts": "other",
      "span-marker-xlm-roberta-base-ar": "bert",
      "spark-tts-arabic": "other",
      "spark-tts-arabic-complete": "other",
      "spark-tts-normazlied-masri-mega": "other",
      "speecht5-tts-arabic": "other",
      "stt-ar-fastconformer-hybrid-large-streaming-pcd-v1-1-mirror": "other",
      "stt-arabic-whisper-finetuned-diactires": "whisper",
      "tadabur-whisper-small": "whisper",
      "tam-omani-adapter-tam-omani-v3": "qwen",
      "tarbiyah-ai-whisper-medium-merged": "whisper",
      "tashkeel-700m": "other",
      "terjman-large": "other",
      "terjman-nano-v1": "other",
      "terjman-supreme-v1": "nllb",
      "tinyoctopus": "qwen",
      "trocr-tunisian-arabic": "other",
      "tunisian-tts": "xtts",
      "unlimited-ocr-quran-uthmani": "other",
      "vibevoice-arabic-z": "other",
      "vibevoice-egy": "other",
      "viobert-v3": "bert",
      "vivo-c-v1": "qwen",
      "voho-saudi-chat-4b": "qwen",
      "voho-saudi-speak-0-6b": "qwen",
      "voho-saudi-stt-small": "whisper",
      "vynis-0-1-2b-instant": "llama",
      "wav2vec2-arabic-phoneme-asr": "xlsr",
      "wav2vec2-base-word-by-word-quran-asr": "wav2vec",
      "wav2vec2-large-xlsr-arabic": "xlsr",
      "wav2vec2-large-xlsr-moroccan-darija": "xlsr",
      "wav2vec2-quran-phonetics": "wav2vec",
      "whisper-algerian-darja-medium": "whisper",
      "whisper-arabic-small": "whisper",
      "whisper-base-ar-quran-ft-hijaiyah-2": "whisper",
      "whisper-base-arabic": "whisper",
      "whisper-base-quran": "whisper",
      "whisper-egyptian-arabic": "whisper",
      "whisper-l-v3-turbo-quran-lora-dataset-mix": "whisper",
      "whisper-large-arabic-cv-11": "whisper",
      "whisper-large-arabic-dialects-v5": "whisper",
      "whisper-large-libyan": "whisper",
      "whisper-large-v3-ar": "whisper",
      "whisper-large-v3-arabic-byne": "whisper",
      "whisper-large-v3-arabic-dialectal-v2": "whisper",
      "whisper-large-v3-egyptian-arabic": "whisper",
      "whisper-large-v3-tarteel": "whisper",
      "whisper-large-v3-turbo": "whisper",
      "whisper-large-v3-turbo-ar-quran": "whisper",
      "whisper-large-v3-turbo-arabic": "whisper",
      "whisper-large-v3-turbo-arabic-dialectal": "whisper",
      "whisper-large-v3-turbo-arabic-dialectal-v2": "whisper",
      "whisper-large-v3-turbo-darija": "whisper",
      "whisper-largev3-medical": "whisper",
      "whisper-m-quran-lora-dataset-mix": "whisper",
      "whisper-medium-arabic": "whisper",
      "whisper-medium-darija": "whisper",
      "whisper-medium-finetuned-sada-asr": "whisper",
      "whisper-quran": "whisper",
      "whisper-small-arabic-dialectal-v2": "whisper",
      "whisper-small-codeswitching-arabicenglish": "whisper",
      "whisper-small-darija": "whisper",
      "whisper-small-egyptian-codeswitch": "whisper",
      "whisper-small-for-quran": "whisper",
      "whisper-small-full-finetune": "whisper",
      "whisper-small-libyan": "whisper",
      "whisper-small-quran-lora-dataset-mix": "whisper",
      "whisper-small-quran-lora-everyayah": "whisper",
      "whisper-small-tunisian-arabic": "whisper",
      "whisper-small-yemeni": "whisper",
      "whisper-turbo-egyptian-codeswitch": "whisper",
      "whisper-with-augmentation-small-arabic-with-diacritics": "whisper",
      "whisper-yemeni": "whisper",
      "whisperlevantine": "whisper",
      "whisperv3-tunisian-codeswitch": "whisper",
      "xlm-roberta-morocco": "bert",
      "xtts-v2": "xtts",
      "yehia": "from-scratch",
      "yehia-7b-preview": "from-scratch",
      "yemeni-arabic-assistant": "gemma",
      "zarra": "other",
      "zipformer-p-quran": "other"
    }
  }
}
