{
 "body": "\nPart of the [FLORES-200](flores.md) family.\n\n## What it measures\n\nflores_en_es scores how well a model translates the FLORES-200 devtest sentences from English\ninto Spanish, one of the highest-resource pairs in the benchmark and one of the most heavily\nrepresented languages in web training data. FLORES-200 evaluates against a single standard\nSpanish reference and does not separately score regional variants.\n\n## Reading the numbers\n\nBecause English-Spanish is close to the ceiling of what current systems achieve on FLORES\nrelative to other pairs, scores here tend to run higher and cluster more tightly than harder\npairs such as flores_en_ja or flores_en_zh, so a small gap is less likely to represent a\nmeaningful capability difference. This repository reports plain BLEU rather than the chrF++ or\nspBLEU the FLORES/NLLB team recommends, so do not compare these numbers directly to chrF++\nfigures from other sources.\n",
 "build": {
  "built_at": "2026-09-09T16:56:50+00:00",
  "commit": "0a599558854c0e238c03a0f0d725239cb28f9d11",
  "eligibility_as_of": "2026-09-09"
 },
 "disposition": {
  "canonical_id": "flores_en_es",
  "reasons": [],
  "status": "unassessed",
  "verified_results": []
 },
 "models_covered": [
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4",
   "model_id": "openai/gpt-4",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4 Turbo",
   "model_id": "openai/gpt-4-turbo",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o",
   "model_id": "openai/gpt-4o",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-05-13)",
   "model_id": "openai/gpt-4o-2024-05-13",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-08-06)",
   "model_id": "openai/gpt-4o-2024-08-06",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-11-20)",
   "model_id": "openai/gpt-4o-2024-11-20",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o mini",
   "model_id": "openai/gpt-4o-mini",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 71.2,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Claude Sonnet 3.5",
   "model_id": "anthropic/claude-3-5-sonnet-20240620",
   "provider": "anthropic",
   "provider_display": "Anthropic",
   "score": 70.8,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Claude Sonnet 3.5 v2",
   "model_id": "anthropic/claude-3-5-sonnet-20241022",
   "provider": "anthropic",
   "provider_display": "Anthropic",
   "score": 70.8,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Flash",
   "model_id": "google/gemini-1-5-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Flash-8B",
   "model_id": "google/gemini-1-5-flash-8b",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Pro",
   "model_id": "google/gemini-1-5-pro",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, llm-stats, intlpull, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.0 Flash",
   "model_id": "google/gemini-2-0-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, llm-stats, intlpull, domain-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.5 Flash",
   "model_id": "google/gemini-2-5-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.5 Pro",
   "model_id": "google/gemini-2-5-pro",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 68.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  }
 ],
 "page": {
  "aliases": [
   "flores200 eng_Latn-spa_Latn"
  ],
  "category": "translation",
  "contamination": {
   "note": "FLORES devtest is fully public with reference translations and has been online since 2022; see the flores family page.",
   "risk": "high"
  },
  "dataset": {
   "languages": [
    "English",
    "Spanish"
   ],
   "license": "CC BY-SA 4.0",
   "modalities": [
    "text"
   ],
   "public_test_set": true,
   "size": 1012,
   "size_note": "1012 devtest sentences translated from English into Spanish, part of the shared FLORES-200 3,001-sentence corpus. FLORES-200 evaluates against a single standard Spanish reference, not regional variants.",
   "splits": "devtest",
   "url": "https://github.com/openlanguagedata/flores"
  },
  "freshness": {
   "researched": "2026-09-07",
   "researched_by": "sonnet-5 agent, batch 1, slice D",
   "reviewed": "",
   "reviewed_by": ""
  },
  "harness": {
   "bigbench": "",
   "helm": "",
   "inspect_evals": "",
   "lm_eval": "",
   "opencompass": "",
   "other": "No confirmed registered task in lm-evaluation-harness, HELM, OpenCompass or BIG-bench; see the flores family page."
  },
  "id": "flores_en_es",
  "last_updated": "",
  "leaderboard_url": "",
  "lineage": {
   "family": "flores",
   "predecessor": "",
   "successors": [],
   "variants": []
  },
  "measures": "flores_en_es scores how well a model translates the FLORES-200 devtest sentences from English into Spanish, one of the highest-resource pairs in the benchmark and one of the most heavily represented languages in web training data.\n",
  "metric": {
   "baseline_note": "This repository reports plain BLEU (English-source direction), not the chrF++ or spBLEU the FLORES/NLLB team itself recommends.",
   "direction": "higher_is_better",
   "human_baseline": null,
   "max_score": 100,
   "name": "BLEU",
   "random_baseline": null,
   "unit": "%"
  },
  "name": "FLORES English-to-Spanish",
  "page_kind": "subset",
  "paper": {
   "arxiv": "2207.04672",
   "title": "No Language Left Behind: Scaling Human-Centered Machine Translation",
   "url": "https://arxiv.org/abs/2207.04672",
   "year": 2022
  },
  "publisher": {
   "authors": [],
   "org": "Meta AI (FAIR), NLLB Team; now maintained by the Open Language Data Initiative (OLDI)",
   "url": "https://github.com/openlanguagedata/flores"
  },
  "released": "2022-07",
  "repo_url": "https://github.com/openlanguagedata/flores",
  "saturation": {
   "as_of": "",
   "note": "Not established from a source read for this page; English-Spanish is among the highest-resource pairs, so scores are expected to sit near the top of what current systems achieve on FLORES.",
   "status": "unknown",
   "top_score": null
  },
  "sources": [
   {
    "accessed": "2026-09-07",
    "title": "No Language Left Behind: Scaling Human-Centered Machine Translation (arXiv)",
    "url": "https://arxiv.org/abs/2207.04672"
   },
   {
    "accessed": "2026-09-07",
    "title": "facebook/flores dataset card (Hugging Face)",
    "url": "https://huggingface.co/datasets/facebook/flores"
   },
   {
    "accessed": "2026-09-07",
    "title": "openlanguagedata/flores GitHub repository (current maintainer, FLORES+)",
    "url": "https://github.com/openlanguagedata/flores"
   }
  ],
  "status": "active",
  "subcategory": "en-es",
  "summary": "BLEU score for English-to-Spanish translation on the FLORES-200 devtest set, as reported in this repository's model cards.",
  "tags": [
   "translation",
   "spanish"
  ],
  "task_format": "Translate the 1012 FLORES-200 devtest sentences from English into Spanish; score against the human Spanish reference."
 }
}