{
 "body": "\nPart of the [FLORES-200](flores.md) family.\n\n## What it measures\n\nflores_en_ja scores how well a model translates the FLORES-200 devtest sentences from English\ninto Japanese. Japanese's subject-object-verb order, lack of whitespace-delimited words, and\nmixed kanji/kana/katakana script make it more distant from English than the benchmark's\nRomance-language pairs, and it is generally treated as a harder direction for both translation\nquality and for automatic metrics that rely on surface n-gram overlap after tokenization.\n\n## Reading the numbers\n\nWord-level BLEU is more sensitive to tokenization choices for Japanese than for\nspace-delimited languages, since a tokenizer's segmentation decisions directly change what\ncounts as a matching n-gram; this repository does not state which tokenizer produced its\nflores_en_ja scores, so treat comparisons against other sources' Japanese BLEU numbers with\ncaution unless the tokenizer matches. Expect lower absolute scores here than on flores_en_de\nor flores_en_es even for strong models \u2014 that is consistent with the pair's difficulty, not\nnecessarily weaker translation quality.\n",
 "build": {
  "built_at": "2026-09-09T16:56:50+00:00",
  "commit": "0a599558854c0e238c03a0f0d725239cb28f9d11",
  "eligibility_as_of": "2026-09-09"
 },
 "disposition": {
  "canonical_id": "flores_en_ja",
  "reasons": [],
  "status": "unassessed",
  "verified_results": []
 },
 "models_covered": [
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4",
   "model_id": "openai/gpt-4",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4 Turbo",
   "model_id": "openai/gpt-4-turbo",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o",
   "model_id": "openai/gpt-4o",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-05-13)",
   "model_id": "openai/gpt-4o-2024-05-13",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-08-06)",
   "model_id": "openai/gpt-4o-2024-08-06",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o (2024-11-20)",
   "model_id": "openai/gpt-4o-2024-11-20",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "GPT-4o mini",
   "model_id": "openai/gpt-4o-mini",
   "provider": "openai",
   "provider_display": "OpenAI",
   "score": 61.4,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Claude Sonnet 3.5",
   "model_id": "anthropic/claude-3-5-sonnet-20240620",
   "provider": "anthropic",
   "provider_display": "Anthropic",
   "score": 60.8,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Claude Sonnet 3.5 v2",
   "model_id": "anthropic/claude-3-5-sonnet-20241022",
   "provider": "anthropic",
   "provider_display": "Anthropic",
   "score": 60.8,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Flash",
   "model_id": "google/gemini-1-5-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Flash-8B",
   "model_id": "google/gemini-1-5-flash-8b",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull, multimodal-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 1.5 Pro",
   "model_id": "google/gemini-1-5-pro",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, llm-stats, intlpull, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.0 Flash",
   "model_id": "google/gemini-2-0-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, llm-stats, intlpull, domain-evals, safety-evals"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.5 Flash",
   "model_id": "google/gemini-2-5-flash",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, llm-stats, intlpull"
  },
  {
   "as_of": "2026-04",
   "attribution": "unverified-legacy",
   "display_name": "Gemini 2.5 Pro",
   "model_id": "google/gemini-2-5-pro",
   "provider": "google",
   "provider_display": "Google DeepMind",
   "score": 58.1,
   "source": "lmarena.ai, provider-reports, multimodal-evals, safety-evals, preference-evals, domain-evals, llm-stats, intlpull"
  }
 ],
 "page": {
  "aliases": [
   "flores200 eng_Latn-jpn_Jpan"
  ],
  "category": "translation",
  "contamination": {
   "note": "FLORES devtest is fully public with reference translations and has been online since 2022; see the flores family page.",
   "risk": "high"
  },
  "dataset": {
   "languages": [
    "English",
    "Japanese"
   ],
   "license": "CC BY-SA 4.0",
   "modalities": [
    "text"
   ],
   "public_test_set": true,
   "size": 1012,
   "size_note": "1012 devtest sentences translated from English into Japanese, part of the shared FLORES-200 3,001-sentence corpus.",
   "splits": "devtest",
   "url": "https://github.com/openlanguagedata/flores"
  },
  "freshness": {
   "researched": "2026-09-07",
   "researched_by": "sonnet-5 agent, batch 1, slice D",
   "reviewed": "",
   "reviewed_by": ""
  },
  "harness": {
   "bigbench": "",
   "helm": "",
   "inspect_evals": "",
   "lm_eval": "",
   "opencompass": "",
   "other": "No confirmed registered task in lm-evaluation-harness, HELM, OpenCompass or BIG-bench; see the flores family page."
  },
  "id": "flores_en_ja",
  "last_updated": "",
  "leaderboard_url": "",
  "lineage": {
   "family": "flores",
   "predecessor": "",
   "successors": [],
   "variants": []
  },
  "measures": "flores_en_ja scores how well a model translates the FLORES-200 devtest sentences from English into Japanese. Japanese's subject-object-verb order, lack of whitespace-delimited words, and mixed kanji/kana/katakana script make it more distant from English than the benchmark's Romance-language pairs.\n",
  "metric": {
   "baseline_note": "This repository reports plain BLEU (English-source direction), not the chrF++ or spBLEU the FLORES/NLLB team itself recommends. The tokenizer or word segmenter used for this repository's Japanese scores is not documented.",
   "direction": "higher_is_better",
   "human_baseline": null,
   "max_score": 100,
   "name": "BLEU",
   "random_baseline": null,
   "unit": "%"
  },
  "name": "FLORES English-to-Japanese",
  "page_kind": "subset",
  "paper": {
   "arxiv": "2207.04672",
   "title": "No Language Left Behind: Scaling Human-Centered Machine Translation",
   "url": "https://arxiv.org/abs/2207.04672",
   "year": 2022
  },
  "publisher": {
   "authors": [],
   "org": "Meta AI (FAIR), NLLB Team; now maintained by the Open Language Data Initiative (OLDI)",
   "url": "https://github.com/openlanguagedata/flores"
  },
  "released": "2022-07",
  "repo_url": "https://github.com/openlanguagedata/flores",
  "saturation": {
   "as_of": "",
   "note": "Not established from a source read for this page; English-Japanese is more linguistically distant than the benchmark's Romance-language pairs, so lower absolute scores are expected by design.",
   "status": "unknown",
   "top_score": null
  },
  "sources": [
   {
    "accessed": "2026-09-07",
    "title": "No Language Left Behind: Scaling Human-Centered Machine Translation (arXiv)",
    "url": "https://arxiv.org/abs/2207.04672"
   },
   {
    "accessed": "2026-09-07",
    "title": "facebook/flores dataset card (Hugging Face)",
    "url": "https://huggingface.co/datasets/facebook/flores"
   },
   {
    "accessed": "2026-09-07",
    "title": "openlanguagedata/flores GitHub repository (current maintainer, FLORES+)",
    "url": "https://github.com/openlanguagedata/flores"
   }
  ],
  "status": "active",
  "subcategory": "en-ja",
  "summary": "BLEU score for English-to-Japanese translation on the FLORES-200 devtest set, as reported in this repository's model cards.",
  "tags": [
   "translation",
   "japanese"
  ],
  "task_format": "Translate the 1012 FLORES-200 devtest sentences from English into Japanese; score against the human Japanese reference."
 }
}