{
  "$schema": "https://www.yerel.ai/schemas/methodology-citations-v1.json",
  "version": "methodology-citations-v1",
  "generatedAt": "2026-05-29",
  "description": "Türkçe AI evaluation frameworks YerelAI cites for methodology but has NOT imported row-level data from. These are referenced as broader landscape context; readers should consult the upstream source directly for current rankings.",
  "honestyStatement": "Bu liste, YerelAI'nin metodolojik kaynak olarak referans verdiği harici Türkçe AI değerlendirme çerçevelerini gösterir. Her birinin kendi puanlama kuralları, veri setleri ve güncelleme kadansları vardır. YerelAI bu kaynaklardan satır-seviyesinde veri ÇEKMEMİŞTİR (alibayram-tr-mmlu hariç — bkz. external-benchmarks-index.json). Aşağıdaki kaynaklara doğrudan başvurun ya da kendi YerelAI run'larımızı /benchmark sayfamızdan inceleyin.",
  "citations": [
    {
      "id": "cetvel",
      "name": "Cetvel",
      "fullName": "Cetvel — Comprehensive Evaluation Benchmark for Turkish NLP",
      "organization": "Koç Üniversitesi Yapay Zeka Lab (KUIS-AI)",
      "homepage": "https://github.com/KUIS-AI/cetvel",
      "scope": "22 dataset across 7 NLP tasks (NER, NLI, QA, summarization, classification, etc.)",
      "language": "tr-only",
      "license": "Per-task licenses vary — see upstream",
      "lastChecked": "2026-05-29",
      "whyWeCite": "Sektör vertical pages ve `/benchmark/metodoloji` Türkçe NLP değerlendirme manzarasını anlatırken Cetvel'i referans gösterir. YerelAI 12-boyutlu rubric'i ile yan yana gelir: Cetvel klasik NLP görevlerini, YerelAI rubric ise uygulama-seviyesi kalite + hız + maliyet boyutlarını kapsar.",
      "relationshipToYerelai": "complement",
      "rowImport": null,
      "notes": "Akademik repository; canlı leaderboard JSON yayınlanmıyor. Kendi modelinizi çalıştırmak için lm-evaluation-harness benzeri harness gerekir."
    },
    {
      "id": "turkishmmlu",
      "name": "TurkishMMLU",
      "fullName": "TurkishMMLU — Turkish Massive Multitask Language Understanding",
      "organization": "Yüksel et al. (2024)",
      "homepage": "https://huggingface.co/datasets/AYueksel/TurkishMMLU",
      "paperUrl": "https://arxiv.org/abs/2407.12402",
      "scope": "10,032 multiple-choice sorular, 9 lise/üniversite dersi",
      "language": "tr-only",
      "license": "CC-BY-4.0",
      "lastChecked": "2026-05-29",
      "whyWeCite": "TR-MMLU boyutumuzun (`tr-mmlu`) doğrudan veri seti kaynağı. Her TR-MMLU YerelAI run'ı bu veri setinden seed=42 ile örnekler.",
      "relationshipToYerelai": "primary-data-source",
      "rowImport": "via-alibayram-leaderboard",
      "notes": "Veri setini biz koşuyoruz; sonuçların bağımsız doğrulaması için alibayram'ın leaderboard'ına attributedSources üzerinden link veriyoruz."
    },
    {
      "id": "turkbench",
      "name": "TurkBench",
      "fullName": "TurkBench — Türkçe Genel-amaçlı Benchmark",
      "organization": "TURKBench araştırmacıları (varyantlar mevcut)",
      "homepage": "https://huggingface.co/datasets/TURKBench",
      "scope": "21 alt-görev × 6 kategori, 8,151 toplam soru",
      "language": "tr-only",
      "license": "Per-task — see upstream",
      "lastChecked": "2026-05-29",
      "whyWeCite": "`turkbench` boyutumuzun veri seti kaynağı. 21 alt-görevin F1 / exact-match / ROUGE metriklerini ortalama alarak puan üretir.",
      "relationshipToYerelai": "primary-data-source",
      "rowImport": null,
      "notes": "Görevler arasında farklı puanlama metrikleri var — `editorial-estimate` statüsünde puanlanır çünkü YerelAI henüz tam harness koşmamıştır."
    },
    {
      "id": "mmlu-pro-tr",
      "name": "MMLU-Pro-TR",
      "fullName": "MMLU-Pro — Türkçe Lokalize Sürüm",
      "organization": "malhajar (Hugging Face)",
      "homepage": "https://huggingface.co/datasets/malhajar/mmlu-pro-tr",
      "scope": "12,000 soru, 10-seçenekli, profesyonel çeviri",
      "language": "tr-only",
      "license": "Apache-2.0 (typical)",
      "lastChecked": "2026-05-29",
      "whyWeCite": "`mmlu-pro-tr` boyutumuzun veri seti kaynağı. MMLU'nun 4-seçenekli klasik versiyonundan daha zor; Türkçe model değerlendirmesinin en kapsamlı sınavlarından.",
      "relationshipToYerelai": "primary-data-source",
      "rowImport": null,
      "notes": "10-şıklı logprob seçim; mikro-ortalama. Editorial-estimate kategorisinde."
    },
    {
      "id": "open-llm-turkish-leaderboard",
      "name": "Open LLM Turkish Leaderboard",
      "fullName": "Açık Türkçe LLM Liderlik Tablosu",
      "organization": "Topluluk (community-maintained)",
      "homepage": "https://huggingface.co/spaces/malhajar/OpenLLMTurkishLeaderboard",
      "scope": "Türkçe-uyumlu modellerin bayrak benchmark'lar üzerinde sürekli güncellenen rankingi",
      "language": "tr-only",
      "license": "Per-leaderboard listing — see upstream",
      "lastChecked": "2026-05-29",
      "whyWeCite": "Türkçe açık-ağırlık model ekosisteminin sezgi ölçütü olarak referans gösterilir. YerelAI'nin 12-boyutlu rubric'i daha derin; topluluk leaderboard'ı daha geniş model coverage'i sağlar.",
      "relationshipToYerelai": "complement",
      "rowImport": null,
      "notes": "Leaderboard sıralaması topluluk PR'larına bağlı — güncelleme kadansı düzensiz olabilir."
    },
    {
      "id": "mizan",
      "name": "Mizan",
      "fullName": "Mizan — Türkçe Embedding Benchmark",
      "organization": "Topluluk araştırmacıları",
      "homepage": "https://huggingface.co/spaces/mizan-tr",
      "scope": "Türkçe embedding kalitesi (retrieval, clustering, classification)",
      "language": "tr-only",
      "license": "Per-listing — see upstream",
      "lastChecked": "2026-05-29",
      "whyWeCite": "`/turkce-rag` rehberi Türkçe embedding seçimi konusunu kapsarken Mizan'a yönlendirir.",
      "relationshipToYerelai": "tangent",
      "rowImport": null,
      "notes": "YerelAI'nin temel benchmark rubric'i jeneratif modeller için; Mizan embedding-spesifik olduğu için ortogonal."
    },
    {
      "id": "tr-mteb",
      "name": "TR-MTEB",
      "fullName": "TR-MTEB — Turkish Massive Text Embedding Benchmark",
      "organization": "Topluluk araştırmacıları (MTEB Türkçe varyantı)",
      "homepage": "https://huggingface.co/spaces/mteb/leaderboard",
      "scope": "Türkçe-spesifik retrieval/clustering/classification görevleri",
      "language": "tr-only",
      "license": "Per-task — see upstream",
      "lastChecked": "2026-05-29",
      "whyWeCite": "Embedding seçimine yardımcı; `/turkce-rag` rehberinin BGE-M3, multilingual-e5, jina-v3 karşılaştırmasının zemini.",
      "relationshipToYerelai": "complement",
      "rowImport": null,
      "notes": "MTEB ana liderlik tablosunun Türkçe filtreli görünümü."
    }
  ],
  "summary": {
    "totalCitations": 7,
    "withRowImport": 1,
    "complementCount": 4,
    "primaryDataSourceCount": 3
  }
}
