{
 "slug": "The_Pile",
 "title": "The Pile",
 "type": "dataset",
 "short_desc": "EleutherAI's December 2020 corpus of 825 GiB of curated English text drawn from 22 sources, the training set behind GPT-Neo, GPT-J, GPT-NeoX-20B, and Pythia, later partly withdrawn amid the Books3 copyright dispute.",
 "categories": [
  "Training datasets",
  "Open-source AI"
 ],
 "infobox": {
  "Released": "December 31, 2020",
  "Creator": "EleutherAI",
  "Size": "825 GiB (roughly 886 GB) of English text",
  "Composition": "22 sub-datasets spanning web scrapes, academic text, books, code, legal and government documents, and dialogue",
  "License / access": "Mixed component licensing; original downloads taken offline in 2023, partial community mirror (Pile-uncopyrighted) remains",
  "Paper/report": "arXiv:2101.00027",
  "Successor": "Common Pile"
 },
 "infobox_links": {
  "Creator": [
   "EleutherAI"
  ],
  "Successor": [
   "Common_Pile"
  ]
 },
 "aliases": [],
 "family": "",
 "org": "",
 "registry_status": "stub",
 "words": 1146,
 "references": 14,
 "outbound": [
  "Cerebras-GPT",
  "Cerebras_Systems",
  "Chinchilla",
  "Comma_(models)",
  "Common_Crawl",
  "Common_Pile",
  "EleutherAI",
  "GPT-2",
  "GPT-3",
  "GPT-J",
  "GPT-Neo",
  "GPT-NeoX-20B",
  "Google_DeepMind",
  "Hugging_Face",
  "LLaMA",
  "Large_language_model",
  "Llama_2",
  "Megatron-Turing_NLG",
  "Meta_AI",
  "NVIDIA",
  "OpenAI",
  "Pythia",
  "YaLM-100B",
  "Yandex"
 ],
 "inbound": [
  "Cerebras-GPT",
  "Cerebras_Systems",
  "Comma_(models)",
  "Common_Crawl",
  "Common_Pile",
  "EleutherAI",
  "GPT-J",
  "GPT-Neo",
  "GPT-NeoX-20B",
  "Megatron-Turing_NLG",
  "OPT",
  "Pretraining",
  "Pythia",
  "StarCoder",
  "State-space_model",
  "Technology_Innovation_Institute",
  "YaLM-100B",
  "Yandex"
 ],
 "url": "wiki/The_Pile.html",
 "built": "2026-07-24"
}
