{
 "slug": "Scaling_laws",
 "title": "Scaling laws",
 "type": "concept",
 "short_desc": "Empirical power-law relationships between model performance and parameters, data, and compute; the planning instrument of the LLM era.",
 "categories": [
  "Concepts",
  "Training methods"
 ],
 "infobox": {
  "Field": "Machine learning",
  "Introduced": "Kaplan et al. 2020; revised by Hoffmann et al. (Chinchilla) 2022",
  "Used in": "Compute allocation for essentially every frontier training run",
  "Key papers": "arXiv:2001.08361, arXiv:2203.15556"
 },
 "infobox_links": {
  "Introduced": [
   "Chinchilla"
  ]
 },
 "aliases": [
  "Kaplan scaling laws",
  "Chinchilla scaling"
 ],
 "family": "",
 "org": "",
 "registry_status": "stub",
 "words": 408,
 "references": 7,
 "outbound": [
  "AI_and_Compute",
  "Anthropic",
  "Chinchilla",
  "DeepSeek",
  "GPT-3",
  "Google_DeepMind",
  "Gopher",
  "Large_language_model",
  "Llama_(model_family)",
  "MMLU",
  "Meta_AI",
  "OpenAI",
  "Richard_Sutton",
  "Situational_Awareness_(essay)",
  "Test-time_compute",
  "The_Bitter_Lesson",
  "The_Scaling_Hypothesis",
  "Transformer_(architecture)",
  "o1"
 ],
 "inbound": [
  "01.AI",
  "AI_and_Compute",
  "Aleph_Alpha",
  "Arcee_AI",
  "Backpropagation",
  "Cerebras-GPT",
  "Cerebras_Systems",
  "Chinchilla",
  "Circuits_(interpretability)",
  "Compositional_generalization",
  "Cross-entropy_loss",
  "DeepSeek-R1",
  "DeepSeek_LLM",
  "Demis_Hassabis",
  "Diffusion_language_model",
  "Dropout",
  "FlashAttention",
  "François_Chollet",
  "GPT-1",
  "GPT-2",
  "GPT-4.5",
  "GPT-5",
  "Genie",
  "Geoffrey_Hinton",
  "Gopher",
  "Ilya_Sutskever",
  "Knowledge_distillation",
  "LLaMA",
  "LSTM",
  "Large_language_model",
  "Machines_of_Loving_Grace",
  "Mechanistic_interpretability",
  "Meena",
  "Megatron-LM",
  "Megatron-Turing_NLG",
  "Mixture_of_experts",
  "Ndea",
  "Next-token_prediction",
  "OLMo",
  "PaLM",
  "PaLM_2",
  "Phi_(model_family)",
  "Pretraining",
  "Pythia",
  "Richard_Sutton",
  "RoBERTa",
  "Safe_Superintelligence",
  "Self-attention",
  "Seminal_AI_essays",
  "Situational_Awareness_(essay)",
  "Sparse_autoencoder",
  "State-space_model",
  "Superposition_(interpretability)",
  "Test-time_compute",
  "The_Bitter_Lesson",
  "The_Most_Important_Century",
  "The_Scaling_Hypothesis",
  "Transformer_(architecture)",
  "Trinity_(model_family)",
  "Trinity_Large",
  "Turing-NLG",
  "Vision_Transformer"
 ],
 "url": "wiki/Scaling_laws.html",
 "built": "2026-07-24"
}
