{
  "version": "0.13.1",
  "generatedAt": "2026-08-16T09:58:58.317Z",
  "schema": "PERSIAN-LLM-REFERENCE/v1",
  "meta": {
    "scope": "global",
    "maintainer": "Persian LLM Reference maintainers",
    "canonicalRepo": "https://github.com/sinakazemnezhad/persian-llm-reference",
    "canonicalSite": "https://sinakazemnezhad.github.io/persian-llm-reference",
    "supersedes": "complements discovery lists and leaderboards with typed, verified records",
    "coverage": [
      "models",
      "datasets",
      "benchmarks",
      "leaderboards",
      "community-indexes",
      "research-programs"
    ],
    "citations": [
      {
        "type": "upstream-discovery",
        "title": "Awesome-Persian-LLM",
        "url": "https://github.com/MohammadHeydari/Awesome-Persian-LLM",
        "maintainer": "Awesome-Persian-LLM maintainers",
        "note": {
          "en": "Community discovery index — bibliographic credit for many PLR rows; complements, does not replace.",
          "fa": "فهرست کشف جامعه. اعتبار کتابشناختی برای بسیاری از ردیف‌های PLR؛ مکمل، نه جایگزین."
        }
      },
      {
        "type": "leaderboard",
        "title": "MIZAN LLM Leaderboard",
        "url": "https://huggingface.co/spaces/MCINext/mizan-llm-leaderboard",
        "maintainer": "MCINext",
        "note": {
          "en": "Persian LLM alignment and benchmark scores — PLR links scores, does not mirror live tables.",
          "fa": "جدول هم‌راستایی و معیار مدل‌های فارسی. PLR لینک می‌دهد، جدول زنده را کپی نمی‌کند."
        }
      },
      {
        "type": "leaderboard",
        "title": "Open Persian LLM Leaderboard (PartAI)",
        "url": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard",
        "maintainer": "PartAI",
        "note": {
          "en": "Public Persian instruct model rankings under 10B and beyond.",
          "fa": "رتبه‌بندی عمومی مدل‌های دستوری فارسی زیر ۱۰ میلیارد و بیشتر."
        }
      },
      {
        "type": "leaderboard",
        "title": "Matina Persian LLM Leaderboard",
        "url": "https://huggingface.co/spaces/MatinaAI/persian_llm_leaderboard",
        "maintainer": "MatinaAI",
        "note": {
          "en": "Persian LLM evaluation space from Matina corpus authors.",
          "fa": "فضای ارزیابی مدل فارسی از نویسندگان پیکرهٔ ماتینا."
        }
      },
      {
        "type": "benchmark",
        "title": "PersianMedQA",
        "url": "https://arxiv.org/abs/2506.00250",
        "maintainer": "University of Tehran · SBUMS",
        "note": {
          "en": "Primary measured-score receipt for 20+ model rows — bilingual medical QA from Iranian national exams.",
          "fa": "مدرک اصلی نمرهٔ مستند برای ۲۰+ مدل. پرسش‌وپاسخ پزشکی دوزبانه از آزمون‌های ملی ایران."
        },
        "extra": "https://mohammadjranjbar.github.io/PersianMedQA/"
      },
      {
        "type": "dataset-index",
        "title": "Hugging Face — Persian models & datasets",
        "url": "https://huggingface.co/models?language=fa",
        "maintainer": "Hugging Face community",
        "note": {
          "en": "Hosting layer for weights and cards — PLR adds taxonomy and verification gates on top.",
          "fa": "لایهٔ میزبانی وزن‌ها و کارت‌ها. PLR طبقه‌بندی و دروازهٔ تأیید را اضافه می‌کند."
        }
      }
    ]
  },
  "mission": {
    "en": "A structured, receipt-gated open registry for Persian (Farsi) language models — alongside discovery lists and leaderboards, with sourced records and honest nulls.",
    "fa": "مرجع باز و قابل‌استناد برای مدل‌های زبانی فارسی. در کنار فهرست‌های کشف و جدول‌های امتیاز، با رکوردهای منبع‌دار و جای خالی صادقانه."
  },
  "taxonomy": {
    "classes": [
      "native-foundation",
      "adapted-instruct",
      "multilingual-frontier",
      "encoder-only",
      "dataset",
      "leaderboard",
      "community-index",
      "program"
    ],
    "axes": [
      "scriptFidelity",
      "corpusLaw",
      "curriculumFit",
      "literaryDepth",
      "nativePreference"
    ]
  },
  "entries": [
    {
      "id": "persianmind-v1",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "PersianMind v1.0",
        "fa": "پرشین‌مایند نسخه ۱"
      },
      "org": "University of Tehran",
      "sizeB": 13.7,
      "license": "CC-BY-NC-SA-4.0",
      "status": "measured",
      "summary": {
        "en": "Cross-lingual Persian–English LLM; strong on Belebele Persian and ParsiNLU.",
        "fa": "مدل دوزبانهٔ فارسی–انگلیسی؛ قوی در بنچمارک‌های پرشین و پارسی‌ان‌ال‌یو."
      },
      "links": {
        "hf": "https://huggingface.co/universitytehran/PersianMind-v1.0",
        "paper": "https://arxiv.org/abs/2401.06466"
      },
      "origin": {
        "base": "LLaMA-class",
        "persianTraining": "cross-lingual-adaptation"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-instruction",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "Belebele (Persian)",
          "score": "73.9",
          "asOf": "2024-01",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2401.06466",
          "benchmark": "Belebele",
          "metric": "accuracy",
          "value": "73.9",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot, Persian (Fas), Belebele benchmark, arxiv:2401.06466",
          "source": "https://arxiv.org/abs/2401.06466",
          "publication": "https://arxiv.org/abs/2401.06466"
        },
        {
          "name": "PersianMedQA (Persian)",
          "score": "24.22",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "24.22",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 2,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-12",
      "firstSeen": "2024-01",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "dorna-llama3-8b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Dorna-Llama3-8B-Instruct",
        "fa": "درنا، لاما۳ ۸بی"
      },
      "org": "PartAI",
      "sizeB": 8,
      "license": "Llama-3-community",
      "status": "verified",
      "summary": {
        "en": "Widely used Persian instruct model; strong results on public Persian leaderboards under 10B.",
        "fa": "مدل دستوری فارسی پرکاربرد؛ نتایج قوی در جدول‌های عمومی فارسی زیر ۱۰ میلیارد پارامتر."
      },
      "links": {
        "hf": "https://huggingface.co/PartAI/Dorna-Llama3-8B-Instruct-GGUF",
        "ollama": "https://ollama.com/partai/dorna-llama3",
        "leaderboard": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
      },
      "origin": {
        "base": "Llama-3-8B-Instruct",
        "persianTraining": "continued-pretrain+SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open+instruction",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "Open Persian LLM Leaderboard",
          "score": null,
          "asOf": null,
          "receipt": "leaderboard-live",
          "url": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-12",
      "firstSeen": "2024-06",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ],
      "notes": "PersianMedQA paper evaluates Dorna2-LLaMA-3.1-8B separately (see dorna2-llama31-8b). Leaderboard link is live-receipt only."
    },
    {
      "id": "maral-7b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Maral-7B",
        "fa": "مارال ۷بی"
      },
      "org": "MaralGPT",
      "sizeB": 7,
      "license": "MIT",
      "status": "verified",
      "summary": {
        "en": "Early open Persian chat model in the Maral family.",
        "fa": "از نخستین مدل‌های گفتگوی فارسی باز در خانوادهٔ مارال."
      },
      "links": {
        "hf": "https://huggingface.co/MaralGPT/Maral-7B-alpha-1"
      },
      "origin": {
        "base": "Mistral-7B-v0.1",
        "persianTraining": "SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": "unknown"
      },
      "corpus": {
        "class": "unknown",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "firstSeen": "2023-06",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "persian-phi",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Persian-Phi",
        "fa": "پرشین‌فی"
      },
      "org": "Research",
      "sizeB": 3.8,
      "license": "Apache-2.0",
      "status": "verified",
      "summary": {
        "en": "Compact cross-lingual adaptation via curriculum learning on small base.",
        "fa": "سازگاری فشردهٔ دوزبانه با یادگیری برنامه‌ای روی پایهٔ کوچک."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2512.07454",
        "hf": "https://huggingface.co/amirakhlaghiqqq/PersianPhi"
      },
      "origin": {
        "base": "Phi-3-mini-4k-instruct",
        "persianTraining": "SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": "unknown"
      },
      "corpus": {
        "class": "instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": 1,
        "curriculumFit": 2,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "yasin-persian-base",
      "kind": "model",
      "class": "native-foundation",
      "name": {
        "en": "YASIN-Persian-Base",
        "fa": "یاسین، پایهٔ فارسی"
      },
      "org": "ysn-rfd",
      "sizeB": 0.15,
      "license": "YRSL-restricted",
      "status": "verified",
      "summary": {
        "en": "Claims from-scratch Persian decoder; verify architecture and eval independently.",
        "fa": "ادعای معماری فارسی از صفر — نیاز به تأیید مستقل معماری و ارزیابی."
      },
      "links": {
        "hf": "https://huggingface.co/ysn-rfd/YASIN-Persian-Base"
      },
      "origin": {
        "base": "claimed-native",
        "persianTraining": "from-scratch-claimed"
      },
      "script": {
        "rtl": true,
        "zwnj": "claimed",
        "morphology": "claimed"
      },
      "corpus": {
        "class": "Persian-QA-conversations",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "notes": "Small custom decoder (~152M params); YRSL v1.0 restricted license — not frontier native-foundation scale.",
      "gapTags": [
        "native-foundation"
      ]
    },
    {
      "id": "ava-llama3-v2",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "AVA-Llama-3-V2",
        "fa": "آوا، لاما۳ نسخه ۲"
      },
      "org": "Community",
      "sizeB": 8,
      "license": "MIT",
      "status": "verified",
      "summary": {
        "en": "Community Persian instruct fine-tune on Llama 3.",
        "fa": "تنظیم‌دقیق دستوری فارسی جامعه روی لاما ۳."
      },
      "links": {
        "hf": "https://huggingface.co/MehdiHosseiniMoghadam/AVA-Llama-3-V2"
      },
      "origin": {
        "base": "Llama-3-8B",
        "persianTraining": "fine-tune"
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": "unknown"
      },
      "corpus": {
        "class": "instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": 1,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "persian-llama-7b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "persian_llama_7b",
        "fa": "لامای فارسی ۷بی"
      },
      "org": "Community",
      "sizeB": 7,
      "license": "Llama-2-community",
      "status": "verified",
      "summary": {
        "en": "Early LLaMA adapter for Persian generation.",
        "fa": "از آداپترهای نخستین لاما برای تولید فارسی."
      },
      "links": {
        "hf": "https://huggingface.co/mostafaamiri/persian_llama_7b"
      },
      "origin": {
        "base": "Llama-2-7B",
        "persianTraining": "LoRA-SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": "unknown"
      },
      "corpus": {
        "class": "unknown",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "qwen25-7b-instruct",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Qwen2.5-7B-Instruct",
        "fa": "کوئن ۲٫۵، ۷بی دستوری"
      },
      "org": "Alibaba",
      "sizeB": 7,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Strong multilingual instruct; Persian supported but not Persian-first world.",
        "fa": "دستوری چندزبانهٔ قوی؛ فارسی پشتیبانی می‌شود اما جهان اول فارسی نیست."
      },
      "links": {
        "hf": "https://huggingface.co/Qwen/Qwen2.5-7B-Instruct"
      },
      "origin": {
        "base": "Qwen2.5",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "multilingual-mix",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "39.99",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "39.99",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 1,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-12",
      "firstSeen": "2024-09",
      "gapTags": [
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "gpt-4-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "GPT-4 class (proprietary)",
        "fa": "ردهٔ جی‌پی‌تی‑۴ (اختصاصی)"
      },
      "org": "OpenAI",
      "sizeB": null,
      "license": "proprietary",
      "status": "verified",
      "summary": {
        "en": "Frontier closed model; Persian capable but English-born — second-class script risk.",
        "fa": "مدل مرزی بسته؛ فارسی بلد است اما انگلیسی‌زاده — خط درجهٔ دوم."
      },
      "links": {
        "web": "https://platform.openai.com/docs"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-mix"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "weak"
      },
      "corpus": {
        "class": "proprietary",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "notes": "Legacy GPT-4 class row; see gpt-41-class for PersianMedQA measured receipt.",
      "gapTags": [
        "cataloged"
      ]
    },
    {
      "id": "parsbert",
      "kind": "model",
      "class": "encoder-only",
      "name": {
        "en": "ParsBERT",
        "fa": "پارس‌برت"
      },
      "org": "Hooshvare",
      "sizeB": null,
      "license": "open",
      "status": "verified",
      "summary": {
        "en": "Persian BERT encoder — classification/NER, not open-ended LLM generation.",
        "fa": "رمزگذار برت فارسی — طبقه‌بندی و NER، نه تولید متن باز."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2005.12515"
      },
      "origin": {
        "base": "BERT",
        "persianTraining": "Persian-MLM"
      },
      "script": {
        "rtl": true,
        "zwnj": "good",
        "morphology": "good"
      },
      "corpus": {
        "class": "Persian-web-corpus",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 1,
        "curriculumFit": 0,
        "literaryDepth": 0,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-12",
      "firstSeen": "2020-05",
      "gapTags": [
        "encoder-stack",
        "open-weights"
      ]
    },
    {
      "id": "matina-corpus",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Matina corpus",
        "fa": "پیکرهٔ ماتینا"
      },
      "org": "Research",
      "sizeB": null,
      "license": "CC-BY-NC-ND-4.0",
      "status": "verified",
      "summary": {
        "en": "~73B-token Persian text corpus for pretraining research.",
        "fa": "مجموعه‌دادهٔ فارسی — پیکرهٔ متنی فارسی حدود ۷۳ میلیارد توکن برای پژوهش پیش‌آموزش."
      },
      "links": {
        "paper": "https://arxiv.org/html/2502.09188v1",
        "hf": "https://huggingface.co/datasets/MatinaAI/matina_persian_text_corpus"
      },
      "origin": {
        "base": null,
        "persianTraining": "corpus"
      },
      "script": {
        "rtl": true,
        "zwnj": "varies",
        "morphology": "varies"
      },
      "corpus": {
        "class": "curated-persian-text",
        "licensedBooks": false,
        "tokensB": 73
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": 1,
        "curriculumFit": 0,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "mizan-leaderboard",
      "kind": "leaderboard",
      "class": "leaderboard",
      "name": {
        "en": "MIZAN Leaderboard",
        "fa": "جدول میزان"
      },
      "org": "MCINext",
      "sizeB": null,
      "license": "open",
      "status": "verified",
      "summary": {
        "en": "Persian LLM benchmark aggregation — cite scores with as-of date.",
        "fa": "تجمیع بنچمارک مدل‌های فارسی — نمره را با تاریخ ذکر کنید."
      },
      "links": {
        "web": "https://huggingface.co/spaces/MCINext/mizan-llm-leaderboard"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "public-benchmark"
      ]
    },
    {
      "id": "open-persian-llm-leaderboard",
      "kind": "leaderboard",
      "class": "leaderboard",
      "name": {
        "en": "Open Persian LLM Leaderboard",
        "fa": "جدول باز مدل‌های فارسی"
      },
      "org": "PartAI · AUT NLP Lab",
      "sizeB": null,
      "license": "open",
      "status": "verified",
      "summary": {
        "en": "Community leaderboard for open Persian generative models.",
        "fa": "جدول امتیاز عمومی برای مقایسهٔ مدل‌های فارسی در معیارهای منتشرشده."
      },
      "links": {
        "web": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "public-benchmark"
      ]
    },
    {
      "id": "awesome-persian-llm",
      "kind": "community-index",
      "class": "community-index",
      "name": {
        "en": "Awesome Persian LLM",
        "fa": "فهرست awesome فارسی"
      },
      "org": "MohammadHeydari",
      "sizeB": null,
      "license": "MIT",
      "status": "verified",
      "summary": {
        "en": "Primary discovery index for Persian LLMs — papers, models, datasets, tools. PLR ingests from here; reciprocal structured atlas outreach in governance/outreach/.",
        "fa": "فهرست اصلی کشف برای مدل‌های فارسی — مقاله، مدل، پیکره، ابزار. مرجع PLR از اینجا می‌خواند؛ همکاری متقابل در governance/outreach/."
      },
      "links": {
        "repo": "https://github.com/MohammadHeydari/Awesome-Persian-LLM",
        "web": "https://github.com/MohammadHeydari/Awesome-Persian-LLM#why-this-repo"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": null,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-12",
      "gapTags": [
        "ecosystem-index"
      ]
    },
    {
      "id": "elab-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "ELAB — Persian LLM alignment benchmark",
        "fa": "الاب، بنچمارک هم‌ترازی فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Safety, fairness, social norms for Persian — cultural alignment eval.",
        "fa": "معیار سنجش: Safety, fairness, social norms for Persian — cultural alignment ."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2504.12553",
        "web": "https://huggingface.co/spaces/MCILAB/LLM_Alignment_Evaluation"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-synthetic+collected",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": 2,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "taraz-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "TARAZ — Persian cultural SAQ",
        "fa": "تراز، پرسش کوتاه فرهنگی فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Short-answer cultural evaluation with Persian morphological variation.",
        "fa": "معیار سنجش: Short-answer cultural uation with Persian morphological variation."
      },
      "links": {
        "paper": "https://doi.org/10.48550/arxiv.2602.22827"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": "tested",
        "morphology": "tested"
      },
      "corpus": {
        "class": "benchmark-cultural",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": 3,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "gemma3-persian",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Gemma 3 Persian (community)",
        "fa": "جمّا ۳ فارسی (جامعه)"
      },
      "org": "Community",
      "sizeB": 4,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Community Persian fine-tunes on Gemma 3 — lightweight chat.",
        "fa": "تنظیم‌دقیق فارسی جامعه روی جمّا ۳ — گفتگوی سبک."
      },
      "links": {
        "ollama": "https://ollama.com/mshojaei77/gemma3persian",
        "hf": "https://huggingface.co/mshojaei77/gemma-3-4b-persian-v0"
      },
      "origin": {
        "base": "google/gemma-3-4b-it",
        "persianTraining": "fine-tune"
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": "unknown"
      },
      "corpus": {
        "class": "instruction",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "35.87",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "35.87",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": 1,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "dorna2-llama31-8b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Dorna2-Llama3.1-8B-Instruct",
        "fa": "درنا۲، لاما۳٫۱ ۸بی"
      },
      "org": "PartAI",
      "sizeB": 8,
      "license": "Llama-3.1-community",
      "status": "measured",
      "summary": {
        "en": "Successor in Dorna line on Llama 3.1 — cited in Persian alignment benchmarks.",
        "fa": "نسل بعدی درنا روی لاما ۳٫۱ — در بنچمارک‌های هم‌ترازی فارسی ذکر شده."
      },
      "links": {
        "hf": "https://huggingface.co/PartAI/Dorna2-Llama3.1-8B-Instruct",
        "leaderboard": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
      },
      "origin": {
        "base": "Llama-3.1-8B-Instruct",
        "persianTraining": "continued-pretrain+SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open+instruction",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "34.87",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "34.87",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "firstSeen": "2025-01",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "aya-expanse-8b",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Aya Expanse 8B",
        "fa": "آیا اکسپنس ۸بی"
      },
      "org": "Cohere For AI",
      "sizeB": 8,
      "license": "open-weights",
      "status": "measured",
      "summary": {
        "en": "Multilingual instruct model; competitive on Persian cultural alignment evals.",
        "fa": "مدل دستوری چندزبانه؛ رقابتی در ارزیابی هم‌ترازی فرهنگی فارسی."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2504.12553"
      },
      "origin": {
        "base": "Aya-Expanse",
        "persianTraining": "multilingual-instruct"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "multilingual-mix",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "40.60",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "40.60",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-12",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "parsbench",
      "kind": "leaderboard",
      "class": "leaderboard",
      "name": {
        "en": "ParsBench",
        "fa": "پارس‌بنچ"
      },
      "org": "Research",
      "sizeB": null,
      "license": "open",
      "status": "verified",
      "summary": {
        "en": "Persian LLM evaluation benchmark suite — canonical open-source project on GitHub.",
        "fa": "مجموعهٔ ارزیابی مدل‌های فارسی — پروژهٔ متن‌باز اصلی در گیت‌هاب."
      },
      "links": {
        "repo": "https://github.com/ParsBench/ParsBench",
        "web": "https://github.com/ParsBench/ParsBench"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "public-benchmark"
      ]
    },
    {
      "id": "parsi-nlu",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "ParsiNLU",
        "fa": "پارسی‌ان‌ال‌یو"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Persian natural language understanding benchmark — MCQ and comprehension tasks.",
        "fa": "معیار سنجش: Persian natural language understanding  — MCQ and comprehension tasks."
      },
      "links": {
        "repo": "https://github.com/persiannlp/parsinlu",
        "paper": "https://arxiv.org/abs/2012.06154"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-NLU",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "gemini-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Gemini class (proprietary)",
        "fa": "ردهٔ جمینای (اختصاصی)"
      },
      "org": "Google",
      "sizeB": null,
      "license": "proprietary",
      "status": "verified",
      "summary": {
        "en": "Frontier multilingual; Persian supported — not Persian-native foundation.",
        "fa": "مرز چندزبانه؛ فارسی دارد — بنیان فارسی نیست."
      },
      "links": {
        "web": "https://deepmind.google/technologies/gemini/"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-mix"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "proprietary",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "notes": "Generic Gemini class; see gemini-25-class and gemini-20-flash-class for PersianMedQA scores.",
      "gapTags": [
        "cataloged"
      ]
    },
    {
      "id": "claude-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Claude class (proprietary)",
        "fa": "ردهٔ کلاد (اختصاصی)"
      },
      "org": "Anthropic",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Frontier multilingual assistant; Persian capable, English-born stack.",
        "fa": "دستیار مرزی چندزبانه؛ فارسی بلد، پشتهٔ انگلیسی‌زاده."
      },
      "links": {
        "web": "https://docs.anthropic.com/"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-mix"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "proprietary",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "75.19",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "75.19",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "matina-llm-leaderboard",
      "kind": "leaderboard",
      "class": "leaderboard",
      "name": {
        "en": "Matina Persian LLM Leaderboard",
        "fa": "جدول مدل‌های فارسی ماتینا"
      },
      "org": "MatinaAI",
      "sizeB": null,
      "license": "open",
      "status": "verified",
      "summary": {
        "en": "Hugging Face Space ranking Persian LLMs — companion to MIZAN and PartAI boards.",
        "fa": "فضای Hugging Face برای رتبه‌بندی مدل‌های فارسی — مکمل MIZAN و PartAI."
      },
      "links": {
        "web": "https://huggingface.co/spaces/MatinaAI/persian_llm_leaderboard"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": null,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "public-benchmark"
      ]
    },
    {
      "id": "hakim-embedding",
      "kind": "model",
      "class": "encoder-only",
      "name": {
        "en": "Hakim — Farsi text embedding",
        "fa": "حکیم، تعبیهٔ متن فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "see-paper",
      "status": "verified",
      "summary": {
        "en": "Farsi text embedding model for retrieval and semantic search pipelines.",
        "fa": "مدل تعبیهٔ متن فارسی برای بازیابی و جستجوی معنایی."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2505.08435"
      },
      "origin": {
        "base": "encoder",
        "persianTraining": "persian-focused"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 1,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "encoder-stack"
      ]
    },
    {
      "id": "tooka-sbert",
      "kind": "model",
      "class": "encoder-only",
      "name": {
        "en": "Tooka-SBERT",
        "fa": "توکا-اس‌برت"
      },
      "org": "Research",
      "sizeB": null,
      "license": "see-paper",
      "status": "verified",
      "summary": {
        "en": "Lightweight Persian sentence embedding models for downstream NLP tasks.",
        "fa": "مدل‌های تعبیهٔ جملهٔ سبک فارسی برای وظایف پایین‌دستی."
      },
      "links": {
        "paper": "https://aclanthology.org/2025.findings-ijcnlp.147.pdf"
      },
      "origin": {
        "base": "SBERT-class",
        "persianTraining": "persian-focused"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 1,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "encoder-stack"
      ]
    },
    {
      "id": "famteb-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "FaMTEB — Persian embedding benchmark",
        "fa": "فام‌تی‌ای‌بی، بنچمارک تعبیهٔ فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Massive text embedding benchmark suite for Persian language models.",
        "fa": "معیار سنجش: Massive text embedding  suite for Persian language models."
      },
      "links": {
        "paper": "https://arxiv.org/pdf/2502.11571"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-embedding",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "public-benchmark"
      ]
    },
    {
      "id": "farsinstruct",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "FarsInstruct",
        "fa": "فارس‌اینستراکت"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Instruction dataset empowering Persian LLMs for instruction following.",
        "fa": "مجموعه‌دادهٔ فارسی — پیکرهٔ دستوری برای تقویت پیروی از دستور در مدل‌های فارسی."
      },
      "links": {
        "paper": "https://arxiv.org/html/2407.11186v1"
      },
      "origin": {
        "base": null,
        "persianTraining": "instruction-SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 2,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "khayyam-persianmmlu",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Khayyam Challenge — PersianMMLU",
        "fa": "چالش خیام، پرشین‌ام‌ام‌ال‌یو"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Multitask Persian knowledge benchmark — measures whether LLMs are truly wise in Persian.",
        "fa": "معیار سنجش: Multitask Persian knowledge  — measures whether LLMs are truly wise in Persian."
      },
      "links": {
        "paper": "https://openreview.net/forum?id=yIEyHP7AvH"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-knowledge",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 2,
        "literaryDepth": 2,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "farsi-synthetic-data",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "FarsiSyntheticData",
        "fa": "دادهٔ مصنوعی فارسی"
      },
      "org": "MohammadHeydari",
      "sizeB": null,
      "license": "see-repo",
      "status": "verified",
      "summary": {
        "en": "High-quality Farsi instruction-following synthetic datasets generated with LLMs.",
        "fa": "مجموعه‌دادهٔ فارسی — پیکره‌های مصنوعی دستوری فارسی با کیفیت بالا تولیدشده با مدل‌های زبانی."
      },
      "links": {
        "repo": "https://github.com/MohammadHeydari/FarsiSyntheticData"
      },
      "origin": {
        "base": null,
        "persianTraining": "synthetic-instruction"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "synthetic-instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 2,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "persian-synthetic-instruct",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Persian-Synthetic-Instruct (HF)",
        "fa": "دستور مصنوعی فارسی"
      },
      "org": "Heydaritoday",
      "sizeB": null,
      "license": "see-model-card",
      "status": "verified",
      "summary": {
        "en": "Hugging Face dataset of synthetic Persian instruction examples for SFT.",
        "fa": "مجموعه‌دادهٔ فارسی — پیکرهٔ Hugging Face از نمونه‌های دستوری مصنوعی فارسی برای SFT."
      },
      "links": {
        "hf": "https://huggingface.co/datasets/Heydaritoday/Persian-Synthetic-Instruct"
      },
      "origin": {
        "base": null,
        "persianTraining": "synthetic-instruction"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "synthetic-instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 2,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "tlpc-corpus",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "TLPC — Targoman Persian corpus",
        "fa": "پیکرهٔ تارگومان TLPC"
      },
      "org": "Targoman",
      "sizeB": null,
      "license": "see-model-card",
      "status": "verified",
      "summary": {
        "en": "Persian text corpus on Hugging Face — used in open Persian NLP pipelines.",
        "fa": "مجموعه‌دادهٔ فارسی — در خطوط پردازش باز فارسی."
      },
      "links": {
        "hf": "https://huggingface.co/datasets/Targoman/TLPC"
      },
      "origin": {
        "base": null,
        "persianTraining": "pretrain-corpus"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "web-mixed",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "alpaca-persian",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Alpaca Persian",
        "fa": "آلپاکا فارسی"
      },
      "org": "Community",
      "sizeB": null,
      "license": "see-model-card",
      "status": "verified",
      "summary": {
        "en": "Persian translation/adaptation of Alpaca-style instruction data for fine-tuning.",
        "fa": "نسخهٔ فارسی دادهٔ دستوری سبک آلپاکا برای فاین‌تیون."
      },
      "links": {
        "hf": "https://huggingface.co/datasets/sinarashidi/alpaca-persian"
      },
      "origin": {
        "base": null,
        "persianTraining": "instruction-SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 2,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "percul-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PERCUL — cultural eval (Persian)",
        "fa": "پرکول، ارزیابی فرهنگی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Story-driven cultural evaluation of LLMs in Persian literary and social context.",
        "fa": "معیار سنجش: Story-driven cultural uation of LLMs in Persian literary and social context."
      },
      "links": {
        "paper": "https://aclanthology.org/2025.naacl-long.631.pdf"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval-cultural"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-cultural",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 2,
        "literaryDepth": 3,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "native-preference",
        "public-benchmark"
      ]
    },
    {
      "id": "melac-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "MELAC — massive Persian LLM eval",
        "fa": "ملاک، ارزیابی گسترده"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Massive evaluation of large language models on Persian tasks.",
        "fa": "معیار سنجش: Massive uation of large language models on Persian tasks."
      },
      "links": {
        "paper": "https://aclanthology.org/2025.ijcnlp-long.105.pdf"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-suite",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 2,
        "literaryDepth": 2,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "taarof-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Taarof — Persian pragmatics eval",
        "fa": "تعارف، ارزیابی pragmaتیک فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Benchmark testing whether LLMs understand Persian taarof and politeness norms.",
        "fa": "معیار سنجش: testing whether LLMs understand Persian taarof and politeness norms."
      },
      "links": {
        "paper": "https://aclanthology.org/2025.emnlp-main.94.pdf"
      },
      "origin": {
        "base": null,
        "persianTraining": "eval-cultural"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "benchmark-cultural",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": null,
        "curriculumFit": 2,
        "literaryDepth": 3,
        "nativePreference": 3
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "native-preference",
        "public-benchmark"
      ]
    },
    {
      "id": "persianrag",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PersianRAG",
        "fa": "پرشین‌رگ"
      },
      "org": "Research",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "Retrieval-augmented generation system and resources for Persian language QA.",
        "fa": "معیار سنجش: Retri-augmented generation system and resources for Persian language QA."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2411.02832"
      },
      "origin": {
        "base": null,
        "persianTraining": "RAG"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "RAG-index",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "cataloged"
      ]
    },
    {
      "id": "persian-ollama-index",
      "kind": "community-index",
      "class": "community-index",
      "name": {
        "en": "Persian Ollama LLM index",
        "fa": "فهرست اولامای فارسی"
      },
      "org": "sepy-dev",
      "sizeB": null,
      "license": "GPL-3.0",
      "status": "verified",
      "summary": {
        "en": "Community index of Persian models packaged for Ollama local inference.",
        "fa": "فهرست جامعهٔ مدل‌های فارسی بسته‌بندی‌شده برای اولاما."
      },
      "links": {
        "repo": "https://github.com/sepy-dev/Persian-Ollama-LLm"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": null,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": null,
        "licensedBooks": null
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "ecosystem-index"
      ]
    },
    {
      "id": "dorna-4bit-quantized",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Dorna-Llama3-8B 4-bit (quantized)",
        "fa": "درنا ۴بیت کوانتیزه"
      },
      "org": "Community",
      "sizeB": 8,
      "license": "Llama-3-community",
      "status": "verified",
      "summary": {
        "en": "Quantized 4-bit variant of Dorna for consumer-GPU inference — sibling to PartAI GGUF release.",
        "fa": "نسخهٔ ۴بیت کوانتیزهٔ درنا برای GPU مصرفی — مکمل انتشار GGUF پارتی."
      },
      "links": {
        "hf": "https://huggingface.co/amirMohammadi/Dorna-Llama3-8B-Instruct-Quantized4Bit"
      },
      "origin": {
        "base": "PartAI/Dorna-Llama3-8B-Instruct",
        "persianTraining": "quantized-4bit"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open+instruction",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "dorna-llama3-8b-instruct",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Dorna-Llama3-8B-Instruct (full weights)",
        "fa": "درنا، وزن کامل"
      },
      "org": "PartAI",
      "sizeB": 8,
      "license": "Llama-3-community",
      "status": "verified",
      "summary": {
        "en": "Full-precision Dorna instruct weights on Hugging Face — distinct from GGUF distribution.",
        "fa": "وزن کامل دستوری درنا در Hugging Face — جدا از توزیع GGUF."
      },
      "links": {
        "hf": "https://huggingface.co/PartAI/Dorna-Llama3-8B-Instruct",
        "leaderboard": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
      },
      "origin": {
        "base": "Llama-3-8B-Instruct",
        "persianTraining": "continued-pretrain+SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open+instruction",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "Open Persian LLM Leaderboard",
          "score": null,
          "asOf": null,
          "receipt": "leaderboard-live",
          "url": "https://huggingface.co/spaces/PartAI/open-persian-llm-leaderboard"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "biopars",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "BioPars — Persian biomedical LLM",
        "fa": "بیوپارس، مدل زیست‌پزشکی فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "CC-BY-NC-ND-4.0",
      "status": "verified",
      "summary": {
        "en": "BioPars biomedical LLM for Persian clinical QA; Nature Scientific Reports + GitHub code.",
        "fa": "مدل زیست‌پزشکی بیوپارس برای پرسش‌وپاسخ بالینی فارسی؛ مقالهٔ Nature + گیت‌هاب."
      },
      "links": {
        "paper": "https://www.nature.com/articles/s41598-026-55970-3",
        "repo": "https://github.com/amirap80/BioPars"
      },
      "origin": {
        "base": "custom-biomedical",
        "persianTraining": "domain-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "domain-biomedical",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 0,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "parse-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PARSE — Persian reasoning QA",
        "fa": "پارس، پرسش‌وپاسخ استدلالی فارسی"
      },
      "org": "University of Innsbruck · IUST",
      "sizeB": null,
      "license": "see-repo",
      "status": "verified",
      "summary": {
        "en": "10,800 open-domain reasoning QA items (Boolean, multiple-choice, factoid) with multi-hop and unanswerable cases.",
        "fa": "۱۰٬۸۰۰ سؤال استدلالی دامنه‌باز (بله/خیر، چندگزینه‌ای، واقعی) با چندگامی و بدون‌پاسخ."
      },
      "links": {
        "repo": "https://github.com/DataScienceUIBK/Parse",
        "paper": "https://arxiv.org/abs/2602.01246",
        "hf": "https://huggingface.co/datasets/JamshidJDMY/Parse"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "eval-benchmark",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 2,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "ept-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "EPT — Persian trustworthiness eval",
        "fa": "ای‌پی‌تی، ارزیابی اعتمادپذیری فارسی"
      },
      "org": "Research",
      "sizeB": null,
      "license": "see-repo",
      "status": "verified",
      "summary": {
        "en": "1,200 expert-curated prompts across ethics, fairness, privacy, robustness, safety, and truthfulness in Persian context.",
        "fa": "۱٬۲۰۰ پرامپت خبره در شش بُعد اخلاق، انصاف، حریم خصوصی، استحکام، ایمنی و راست‌گویی."
      },
      "links": {
        "repo": "https://github.com/Rezamirbagheri110/EPT-Benchmark",
        "paper": "https://arxiv.org/abs/2509.06838",
        "hf": "https://huggingface.co/datasets/mirbagheri-mohammadreza/EPT-Benchmark"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": null
      },
      "corpus": {
        "class": "eval-benchmark",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 3
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "native-preference",
        "public-benchmark"
      ]
    },
    {
      "id": "persianmedqa",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PersianMedQA",
        "fa": "پرشین‌مدکیوای"
      },
      "org": "University of Tehran",
      "sizeB": null,
      "license": "see-paper",
      "status": "verified",
      "summary": {
        "en": "20,785 bilingual Persian–English medical MCQs from 14 years of Iranian board exams across 23 specialties.",
        "fa": "۲۰٬۷۸۵ سؤال چندگزینه‌ای پزشکی دوزبانه از ۱۴ سال آزمون‌های تخصصی ایران."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2506.00250",
        "web": "https://mohammadjranjbar.github.io/PersianMedQA/",
        "hf": "https://huggingface.co/datasets/MohammadJRanjbar/PersianMedQA"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 3,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 3
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "medical-eval"
      ]
    },
    {
      "id": "pquad-dataset",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PQuAD — Persian reading comprehension",
        "fa": "پیکیوای‌ای‌دی، درک مطلب فارسی"
      },
      "org": "AUT · Mabna Intelligent Computing",
      "sizeB": null,
      "license": "see-repo",
      "status": "verified",
      "summary": {
        "en": "80,000 span-extraction QA pairs on Persian Wikipedia; 25% adversarially unanswerable.",
        "fa": "۸۰٬۰۰۰ جفت پرسش‌وپاسخ استخراجی روی ویکی‌پدیای فارسی؛ ۲۵٪ بدون پاسخ."
      },
      "links": {
        "repo": "https://github.com/AUT-NLP/PQuAD",
        "paper": "https://arxiv.org/abs/2202.06219"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "wikipedia-mrc",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 2,
        "curriculumFit": 2,
        "literaryDepth": 2,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "literary-eval",
        "public-benchmark"
      ]
    },
    {
      "id": "persianmhqa-dataset",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "PersianMHQA — multi-hop QA",
        "fa": "پرشین‌ام‌اچ‌کیوای، پرسش چندگامی"
      },
      "org": "IUST",
      "sizeB": null,
      "license": "see-paper",
      "status": "verified",
      "summary": {
        "en": "7,000 open-domain multi-hop questions from Persian Wikipedia; partial public release on Hugging Face.",
        "fa": "۷٬۰۰۰ پرسش چندگامی دامنه‌باز از ویکی‌پدیای فارسی؛ انتشار جزئی در Hugging Face."
      },
      "links": {
        "hf": "https://huggingface.co/datasets/Arg1990/PersianMHQA",
        "web": "https://en.civilica.com/doc/2111883/"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "wikipedia-mhqa",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 2,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "notes": "Full 7k release pending per authors; HF hosts sample subset.",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "hooshvare-bert-fa",
      "kind": "model",
      "class": "encoder-only",
      "name": {
        "en": "BERT-FA base (HooshvareLab)",
        "fa": "برت پایهٔ فارسی (هوش‌ور)"
      },
      "org": "HooshvareLab",
      "sizeB": 0.12,
      "license": "MIT",
      "status": "verified",
      "summary": {
        "en": "Widely used Persian BERT base for classification, NER, and embedding pipelines.",
        "fa": "برت پایهٔ پرکاربرد فارسی برای طبقه‌بندی، شناسایی موجودیت و تعبیه."
      },
      "links": {
        "hf": "https://huggingface.co/HooshvareLab/bert-fa-base-uncased"
      },
      "origin": {
        "base": "BERT",
        "persianTraining": "persian-corpus"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-open",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 1,
        "curriculumFit": 1,
        "literaryDepth": 1,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "encoder-stack",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "islamicpcqa-benchmark",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "IslamicPCQA — Persian multi-hop Islamic QA",
        "fa": "اسلامیک‌پی‌سی‌کیوای"
      },
      "org": "IUST",
      "sizeB": null,
      "license": "research",
      "status": "verified",
      "summary": {
        "en": "12,282 multi-hop QA pairs from nine Persian Islamic encyclopedias; dataset not yet public.",
        "fa": "مجموعه‌دادهٔ فارسی — ۱۲٬۲۸۲ جفت پرسش‌وپاسخ چندگامی از نه دانشنامهٔ اسلامی فارسی؛ پیکره هنوز عمومی نیست."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2304.11664"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "domain-islamic",
        "licensedBooks": true
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 3,
        "curriculumFit": null,
        "literaryDepth": 3,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "notes": "Paper published; public dataset release pending per authors.",
      "gapTags": [
        "instruction-data",
        "public-benchmark"
      ]
    },
    {
      "id": "deepseek-v3-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "DeepSeek-V3 class (proprietary)",
        "fa": "ردهٔ دیپ‌سیک‑وی۳ (اختصاصی)"
      },
      "org": "DeepSeek",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Frontier multilingual model class cited on Persian leaderboards; no open weights.",
        "fa": "جدول امتیاز عمومی برای مقایسهٔ مدل‌های فارسی در معیارهای منتشرشده."
      },
      "links": {
        "web": "https://www.deepseek.com/"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "68.05",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "68.05",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "oscar-2201-corpus",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "OSCAR-2201 (multilingual incl. Persian)",
        "fa": "پیکرهٔ اسکار‑۲۲۰۱ (چندزبانه شامل فارسی)"
      },
      "org": "OSCAR / INRIA",
      "sizeB": null,
      "license": "CC0-1.0",
      "status": "verified",
      "summary": {
        "en": "Large multilingual web corpus with Persian (`fa`) shard — common pretrain source, not literary-grade alone.",
        "fa": "مجموعه‌دادهٔ فارسی — منبع پیش‌آموزش رایج، نه کتابخانهٔ ادبی به‌تنهایی."
      },
      "links": {
        "hf": "https://huggingface.co/datasets/oscar-corpus/OSCAR-2201"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "unknown",
        "morphology": null
      },
      "corpus": {
        "class": "web-scrape-filtered",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 1,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 0,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "qwen3-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Qwen3 class (proprietary / open mix)",
        "fa": "ردهٔ کوئن۳"
      },
      "org": "Alibaba",
      "sizeB": null,
      "license": "mixed",
      "status": "verified",
      "summary": {
        "en": "Qwen3 family on Persian leaderboards; mix of open weights and API-only tiers.",
        "fa": "جدول امتیاز عمومی برای مقایسهٔ مدل‌های فارسی در معیارهای منتشرشده."
      },
      "links": {
        "web": "https://qwenlm.github.io/"
      },
      "origin": {
        "base": "Qwen",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "open-weights"
      ]
    },
    {
      "id": "llama33-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Llama 3.3 class (community + API)",
        "fa": "ردهٔ لاما ۳٫۳"
      },
      "org": "Meta",
      "sizeB": null,
      "license": "Llama-3.3-community",
      "status": "measured",
      "summary": {
        "en": "Llama 3.3 instruct family — base for many Persian community fine-tunes including Dorna2.",
        "fa": "خانوادهٔ لاما ۳٫۳ دستوری — پایهٔ بسیاری از فاین‌تیون‌های فارسی از جمله درنا۲."
      },
      "links": {
        "hf": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct"
      },
      "origin": {
        "base": "Llama-3.3",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "66.63",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "66.63",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "wikimedia-fa-wikipedia",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "Persian Wikipedia dump",
        "fa": "دامپ ویکی‌پدیای فارسی"
      },
      "org": "Wikimedia",
      "sizeB": null,
      "license": "CC-BY-SA-3.0",
      "status": "verified",
      "summary": {
        "en": "Canonical open encyclopedia corpus for Persian NLP pretrain and QA dataset construction.",
        "fa": "مجموعه‌دادهٔ فارسی — پیکرهٔ دانشنامهٔ باز مرجع برای پیش‌آموزش و ساخت دادهٔ پرسش‌وپاسخ فارسی."
      },
      "links": {
        "web": "https://dumps.wikimedia.org/fawiki/latest/"
      },
      "origin": {
        "base": null,
        "persianTraining": null
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "wikipedia",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 2,
        "curriculumFit": 2,
        "literaryDepth": 2,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruction-data"
      ]
    },
    {
      "id": "gpt-41-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "GPT-4.1 class (proprietary)",
        "fa": "ردهٔ جی‌پی‌تی‑۴٫۱ (اختصاصی)"
      },
      "org": "OpenAI",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Frontier proprietary class; top PersianMedQA Persian accuracy in published eval (83.09%).",
        "fa": "ردهٔ اختصاصی مرزی؛ بالاترین دقت فارسی در ارزیابی منتشرشدهٔ PersianMedQA."
      },
      "links": {
        "web": "https://platform.openai.com/docs"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "83.09",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "83.09",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 3,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "gemini-25-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Gemini 2.5 class (proprietary)",
        "fa": "ردهٔ جمینای ۲٫۵ (اختصاصی)"
      },
      "org": "Google",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Google frontier class evaluated on EPT Persian trustworthiness benchmark.",
        "fa": "ردهٔ مرزی گوگل — ارزیابی‌شده در بنچمارک اعتمادپذیری فارسی EPT."
      },
      "links": {
        "web": "https://deepmind.google/technologies/gemini/"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "82.37",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "82.37",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "meditron3-8b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Meditron3-8B",
        "fa": "مدیترون۳‑۸بی"
      },
      "org": "EPFL · Yale",
      "sizeB": 8,
      "license": "Llama-2-community",
      "status": "measured",
      "summary": {
        "en": "Medical LLM family on Llama 2 — evaluated on PersianMedQA Persian split (39.70%).",
        "fa": "خانوادهٔ مدل پزشکی روی لاما ۲ — ارزیابی‌شده در PersianMedQA فارسی (۳۹٫۷۰٪)."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2311.16057",
        "repo": "https://github.com/epfLLM/Meditron",
        "hf": "https://huggingface.co/epfl-llm/meditron3-8b"
      },
      "origin": {
        "base": "Llama-2",
        "persianTraining": "medical-domain-continued-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "weak"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "38.67",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "38.67",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-13",
      "firstSeen": "2024-11",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "biomistral-7b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "BioMistral-7B",
        "fa": "بیومیسترال‑۷بی"
      },
      "org": "BioMistral",
      "sizeB": 7,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Open biomedical instruct model on Mistral — multilingual medical QA baseline, not Persian-native.",
        "fa": "مدل دستوری زیست‌پزشکی باز روی میسترال — خط پایهٔ پرسش‌وپاسخ پزشکی چندزبانه."
      },
      "links": {
        "hf": "https://huggingface.co/BioMistral/BioMistral-7B",
        "paper": "https://arxiv.org/abs/2402.10373"
      },
      "origin": {
        "base": "Mistral-7B",
        "persianTraining": "medical-domain-continued-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "weak"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "25.76",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "25.76",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 1
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "gaokerena-v",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Gaokerena-V — Persian medical assistant",
        "fa": "گاوکرنا‑وی، دستیار پزشکی فارسی"
      },
      "org": "Gaokerena",
      "sizeB": 8,
      "license": "CC-BY-NC-SA-4.0",
      "status": "verified",
      "summary": {
        "en": "First open Persian medical instruct model on Aya-Expanse-8B; 90M-token medical corpus + MF3QA physician Q&A.",
        "fa": "نخستین مدل دستوری پزشکی فارسی باز روی آیا‑اکسپنس‑۸بی؛ پیکرهٔ ۹۰M توکن + MF3QA."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2505.16000",
        "repo": "https://github.com/Mehrdadghassabi/Gaokerena-V",
        "hf": "https://huggingface.co/gaokerena/gaokerena-v1.0"
      },
      "origin": {
        "base": "aya-expanse-8b",
        "persianTraining": "domain-SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "firstSeen": "2025-05",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "gaokerena-mf3qa",
      "kind": "dataset",
      "class": "dataset",
      "name": {
        "en": "MF3QA — Persian medical free-form QA",
        "fa": "ام‌اف۳کیوای، پرسش‌وپاسخ آزاد پزشکی فارسی"
      },
      "org": "Gaokerena",
      "sizeB": null,
      "license": "see-repo",
      "status": "verified",
      "summary": {
        "en": "~186k crawled medical QA pairs plus 20k expert-cleaned physician Q&A for Persian medical LLM training.",
        "fa": "حدود ۱۸۶هزار جفت پرسش‌وپاسخ پزشکی خزیده‌شده به‌علاوه ۲۰هزار جفت پاک‌سازی‌شدهٔ پزشک."
      },
      "links": {
        "repo": "https://github.com/Mehrdadghassabi/Gaokerena/tree/main/dataset/MF3QA",
        "paper": "https://arxiv.org/abs/2505.16000",
        "hf": "https://huggingface.co/datasets/gaokerena/MF3QA"
      },
      "origin": {
        "base": null,
        "persianTraining": "instruction-SFT"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [],
      "alefbaAxes": {
        "scriptFidelity": 3,
        "corpusLaw": 2,
        "curriculumFit": null,
        "literaryDepth": 1,
        "nativePreference": 2
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "medical-eval"
      ]
    },
    {
      "id": "gemini-20-flash-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Gemini 2.0 Flash class (proprietary)",
        "fa": "ردهٔ جمینای ۲٫۰ فلش (اختصاصی)"
      },
      "org": "Google",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Google frontier flash tier — 76.86% Persian accuracy on PersianMedQA (Table 3 selective-answering baseline).",
        "fa": "ردهٔ فلش مرزی گوگل — ۷۶٫۸۶٪ دقت فارسی در PersianMedQA."
      },
      "links": {
        "web": "https://deepmind.google/technologies/gemini/",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "proprietary",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "76.86",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "76.86",
          "unit": "percent",
          "language": "fa",
          "conditions": "selective-answering baseline, Persian split (Fa), Table 3, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "llama-31-405b-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Llama 3.1 405B class (open weights)",
        "fa": "ردهٔ لاما ۳٫۱ ۴۰۵بی (وزن باز)"
      },
      "org": "Meta",
      "sizeB": 405,
      "license": "Llama-3.1-community",
      "status": "measured",
      "summary": {
        "en": "Best open-weight general model on PersianMedQA Persian split — 69.25% (paper §4).",
        "fa": "بهترین مدل عمومی با وزن باز در PersianMedQA فارسی — ۶۹٫۲۵٪."
      },
      "links": {
        "hf": "https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct",
        "paper": "https://arxiv.org/abs/2407.21783"
      },
      "origin": {
        "base": "Llama-3.1",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "67.02",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "67.02",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "gemma-3-27b-it-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Gemma 3 27B IT class",
        "fa": "ردهٔ جمّا ۳ ۲۷بی دستوری"
      },
      "org": "Google",
      "sizeB": 27,
      "license": "Gemma",
      "status": "measured",
      "summary": {
        "en": "Google Gemma 3 instruct tier — 59.06% Persian accuracy on PersianMedQA (paper Table 3).",
        "fa": "خانوادهٔ دستوری جمّا ۳ گوگل، ۵۹٫۰۶٪ دقت فارسی در PersianMedQA (جدول ۳ مقاله)."
      },
      "links": {
        "hf": "https://huggingface.co/google/gemma-3-27b-it",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Gemma-3",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "59.06",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "59.06",
          "unit": "percent",
          "language": "fa",
          "conditions": "selective-answering, Persian split (Fa), Table 3, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "firstSeen": "2025-06",
      "alefbaAxes": {
        "scriptFidelity": 2,
        "corpusLaw": 0,
        "curriculumFit": 0,
        "literaryDepth": 2,
        "nativePreference": 0
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "open-ensemble-persianmedqa",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Open-weight ensemble (PersianMedQA)",
        "fa": "مجموعهٔ مدل‌های باز (PersianMedQA)"
      },
      "org": "PersianMedQA authors",
      "sizeB": null,
      "license": "see-paper",
      "status": "measured",
      "summary": {
        "en": "Majority-vote ensemble of diverse open-weight families — 73.7% Persian on PersianMedQA (paper §4.4).",
        "fa": "ترکیب رأی‌گیری چند خانوادهٔ مدل باز، ۷۳٫۷٪ دقت فارسی در PersianMedQA (بخش ۴٫۴ مقاله)."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2506.00250",
        "web": "https://mohammadjranjbar.github.io/PersianMedQA/"
      },
      "origin": {
        "base": "ensemble",
        "persianTraining": "inference-only"
      },
      "script": {
        "rtl": true,
        "zwnj": null,
        "morphology": null
      },
      "corpus": {
        "class": "eval-benchmark",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "73.70",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "73.70",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "firstSeen": "2025-06",
      "notes": "Not a single checkpoint — documented ensemble result from PersianMedQA evaluation.",
      "alefbaAxes": {
        "scriptFidelity": null,
        "corpusLaw": null,
        "curriculumFit": null,
        "literaryDepth": null,
        "nativePreference": null
      },
      "verifiedAt": "2026-08-13",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "qwen25-72b-instruct",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Qwen 2.5 72B Instruct",
        "fa": "کوئن ۲٫۵ ۷۲بی دستوری"
      },
      "org": "Alibaba",
      "sizeB": 72,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Large open Qwen 2.5 instruct tier — 65.17% Persian accuracy on PersianMedQA (paper Table A).",
        "fa": "خانوادهٔ دستوری باز کوئن ۲٫۵ ۷۲بی، ۶۵٫۱۷٪ دقت فارسی در PersianMedQA (جدول پیوست مقاله)."
      },
      "links": {
        "hf": "https://huggingface.co/Qwen/Qwen2.5-72B-Instruct",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Qwen-2.5",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "65.17",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "65.17",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-13",
      "firstSeen": "2024-09",
      "gapTags": [
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "mixtral-8x22b-instruct",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Mixtral 8x22B Instruct",
        "fa": "میکسترال ۸×۲۲بی دستوری"
      },
      "org": "Mistral AI",
      "sizeB": 141,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Mistral sparse MoE instruct model — 36.78% Persian accuracy on PersianMedQA (paper Table A).",
        "fa": "مدل دستوری MoE میسترال، ۳۶٫۷۸٪ دقت فارسی در PersianMedQA (جدول پیوست مقاله)."
      },
      "links": {
        "hf": "https://huggingface.co/mistralai/Mixtral-8x22B-Instruct-v0.1",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Mixtral-8x22B",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "36.78",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "36.78",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-13",
      "firstSeen": "2024-04",
      "gapTags": [
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    },
    {
      "id": "mistral-saba-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Mistral Saba class (proprietary)",
        "fa": "ردهٔ میسترال سابا (اختصاصی)"
      },
      "org": "Mistral AI",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Mistral Saba proprietary tier — 61.85% Persian accuracy on PersianMedQA (paper Table A).",
        "fa": "خانوادهٔ اختصاصی میسترال سابا، ۶۱٫۸۵٪ دقت فارسی در PersianMedQA (جدول پیوست مقاله)."
      },
      "links": {
        "web": "https://mistral.ai/",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "proprietary",
        "persianTraining": "multilingual-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "partial"
      },
      "corpus": {
        "class": "mixed-web",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "61.85",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "61.85",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-13",
      "firstSeen": "2025-06",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "meditron3-qwen25-7b",
      "kind": "model",
      "class": "adapted-instruct",
      "name": {
        "en": "Meditron3-Qwen2.5-7B",
        "fa": "مدیترون۳‑کوئن۲٫۵‑۷بی"
      },
      "org": "EPFL · Yale",
      "sizeB": 7,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Medical Meditron3 variant on Qwen 2.5 7B — 37.62% Persian accuracy on PersianMedQA (paper Table A).",
        "fa": "نسخهٔ پزشکی مدیترون۳ روی کوئن ۲٫۵ ۷بی، ۳۷٫۶۲٪ دقت فارسی در PersianMedQA (جدول پیوست مقاله)."
      },
      "links": {
        "paper": "https://arxiv.org/abs/2506.00250",
        "repo": "https://github.com/epfLLM/Meditron"
      },
      "origin": {
        "base": "Qwen-2.5-7B",
        "persianTraining": "medical-domain-continued-pretrain"
      },
      "script": {
        "rtl": true,
        "zwnj": "partial",
        "morphology": "weak"
      },
      "corpus": {
        "class": "domain-medical",
        "licensedBooks": false
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "score": "37.62",
          "asOf": "2025-06",
          "receipt": "paper",
          "url": "https://arxiv.org/abs/2506.00250",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "37.62",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "source": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-13",
      "firstSeen": "2025-06",
      "gapTags": [
        "instruct-stack",
        "instruction-data",
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "gpt-41-mini-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "GPT-4.1 Mini class (proprietary)",
        "fa": "ردهٔ جی‌پی‌تی ۴.۱ مینی"
      },
      "org": "OpenAI",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "OpenAI GPT-4.1 Mini tier on PersianMedQA Persian split (74.76%, paper Table 4).",
        "fa": "ردهٔ GPT-4.1 Mini در PersianMedQA فارسی (۷۴٫۷۶٪، جدول ۴ مقاله)."
      },
      "links": {
        "web": "https://platform.openai.com/docs",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "GPT-4.1",
        "persianTraining": "multilingual-pretrain"
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "74.76",
          "score": "74.76",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "asOf": "2025-06",
          "receipt": "paper",
          "source": "https://arxiv.org/abs/2506.00250",
          "url": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-16",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "claude-35-haiku-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Claude 3.5 Haiku class (proprietary)",
        "fa": "ردهٔ کلود ۳.۵ هایکو"
      },
      "org": "Anthropic",
      "sizeB": null,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Anthropic Claude 3.5 Haiku on PersianMedQA Persian split (57.16%, paper Table 4).",
        "fa": "کلود ۳.۵ هایکو در PersianMedQA فارسی (۵۷٫۱۶٪، جدول ۴ مقاله)."
      },
      "links": {
        "web": "https://www.anthropic.com/",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Claude-3.5",
        "persianTraining": "multilingual-pretrain"
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "57.16",
          "score": "57.16",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "asOf": "2025-06",
          "receipt": "paper",
          "source": "https://arxiv.org/abs/2506.00250",
          "url": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-16",
      "gapTags": [
        "measured-evidence",
        "medical-eval"
      ]
    },
    {
      "id": "gemma-3-12b-it-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Gemma 3 12B IT class",
        "fa": "جمّا ۳ ۱۲بی دستوری"
      },
      "org": "Google",
      "sizeB": 12,
      "license": "open-weights",
      "status": "measured",
      "summary": {
        "en": "Gemma 3 12B instruct on PersianMedQA Persian split (52.22%, paper Table 4).",
        "fa": "جمّا ۳ ۱۲بی دستوری در PersianMedQA فارسی (۵۲٫۲۲٪، جدول ۴ مقاله)."
      },
      "links": {
        "hf": "https://huggingface.co/google/gemma-3-12b-it",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Gemma-3",
        "persianTraining": "multilingual-pretrain"
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "52.22",
          "score": "52.22",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "asOf": "2025-06",
          "receipt": "paper",
          "source": "https://arxiv.org/abs/2506.00250",
          "url": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-16",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights",
        "small-model"
      ]
    },
    {
      "id": "cohere-command-r7b-class",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Cohere Command R7B class",
        "fa": "کوهیر کامند R7B"
      },
      "org": "Cohere",
      "sizeB": 7,
      "license": "proprietary",
      "status": "measured",
      "summary": {
        "en": "Cohere Command R7B on PersianMedQA Persian split (38.77%, paper Table 4).",
        "fa": "کوهیر کامند R7B در PersianMedQA فارسی (۳۸٫۷۷٪، جدول ۴ مقاله)."
      },
      "links": {
        "web": "https://cohere.com/",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Command-R",
        "persianTraining": "multilingual-pretrain"
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "38.77",
          "score": "38.77",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "asOf": "2025-06",
          "receipt": "paper",
          "source": "https://arxiv.org/abs/2506.00250",
          "url": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-16",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "small-model"
      ]
    },
    {
      "id": "mistral-nemo-instruct",
      "kind": "model",
      "class": "multilingual-frontier",
      "name": {
        "en": "Mistral Nemo Instruct",
        "fa": "میسترال نمو دستوری"
      },
      "org": "Mistral AI",
      "sizeB": 12,
      "license": "Apache-2.0",
      "status": "measured",
      "summary": {
        "en": "Mistral Nemo instruct on PersianMedQA Persian split (36.23%, paper Table 4).",
        "fa": "میسترال نمو دستوری در PersianMedQA فارسی (۳۶٫۲۳٪، جدول ۴ مقاله)."
      },
      "links": {
        "hf": "https://huggingface.co/mistralai/Mistral-Nemo-Instruct-2407",
        "paper": "https://arxiv.org/abs/2506.00250"
      },
      "origin": {
        "base": "Mistral-Nemo",
        "persianTraining": "multilingual-pretrain"
      },
      "benchmarks": [
        {
          "name": "PersianMedQA (Persian)",
          "benchmark": "PersianMedQA",
          "metric": "accuracy",
          "value": "36.23",
          "score": "36.23",
          "unit": "percent",
          "language": "fa",
          "conditions": "zero-shot MCQ, Persian split (Fa), Table 4, arxiv:2506.00250",
          "asOf": "2025-06",
          "receipt": "paper",
          "source": "https://arxiv.org/abs/2506.00250",
          "url": "https://arxiv.org/abs/2506.00250",
          "publication": "https://arxiv.org/abs/2506.00250"
        }
      ],
      "verifiedAt": "2026-08-16",
      "gapTags": [
        "measured-evidence",
        "medical-eval",
        "open-weights"
      ]
    }
  ],
  "gapMap": {
    "en": [
      "No frontier-scale model whose first world is licensed Persian literature",
      "Few native-rater preference loops at production scale",
      "Literary register and book-memory evals remain sparse",
      "Most open models are English-base adaptations under 15B",
      "Medical QA evidence is thin outside PersianMedQA-style receipts",
      "Few open models under 8B with citable public eval scores",
      "Many rows still lack a published benchmark we can cite with asOf",
      "Instruction data and SFT recipes are scattered across repos"
    ],
    "fa": [
      "هنوز مدل بزرگی نداریم که اولویت اولش ادبیات مجازدار فارسی باشد",
      "نظرسنجی ارزیاب بومی در مقیاس محصول واقعی کم است",
      "معیار سنجش سبک ادبی و حافظهٔ کتاب در فارسی هنوز پراکنده است",
      "بیشتر مدل‌های باز، نسخهٔ فارسی‌شدهٔ مدل انگلیسی زیر ۱۵ میلیارد پارامترند",
      "شواهد پرسش‌وپاسخ پزشکی فارسی خارج از رسیدهای PersianMedQA کم است",
      "مدل باز زیر ۸ میلیارد با نمرهٔ عمومی قابل استناد کم داریم",
      "هنوز ردیف‌های زیادی بدون معیار منتشرشده با تاریخ و منبع هستند",
      "داده و دستور آموزش پراکنده بین مخازن مختلف پخش شده است"
    ],
    "tags": [
      "native-foundation",
      "native-preference",
      "literary-eval",
      "instruct-stack",
      "medical-eval",
      "small-model",
      "public-benchmark",
      "instruction-data"
    ]
  },
  "stats": {
    "total": 72,
    "byKind": {
      "model": 42,
      "dataset": 24,
      "leaderboard": 4,
      "community-index": 2
    },
    "byClass": {
      "adapted-instruct": 15,
      "native-foundation": 1,
      "multilingual-frontier": 22,
      "encoder-only": 4,
      "dataset": 24,
      "leaderboard": 4,
      "community-index": 2
    },
    "byStatus": {
      "measured": 25,
      "verified": 47
    }
  }
}