{
 "slug": "Large_language_model",
 "title": "Large language model",
 "type": "concept",
 "short_desc": "A neural language model, typically a decoder-only Transformer with billions of parameters, pretrained on internet-scale text.",
 "categories": [
  "Concepts",
  "Models"
 ],
 "infobox": {
  "Field": "Natural language processing, deep learning",
  "Defining architecture": "Decoder-only Transformer",
  "Defining method": "Self-supervised pretraining on next-token prediction at scale"
 },
 "infobox_links": {
  "Defining architecture": [
   "Transformer_(architecture)"
  ]
 },
 "aliases": [
  "LLM"
 ],
 "family": "",
 "org": "",
 "registry_status": "stub",
 "words": 1972,
 "references": 32,
 "outbound": [
  "Adam_(optimizer)",
  "Anthropic",
  "BERT",
  "Backpropagation",
  "CLIP",
  "Chain-of-thought_prompting",
  "Chinchilla",
  "Cross-entropy_loss",
  "DeepSeek",
  "DeepSeek-R1",
  "DeepSeek-V3",
  "Diffusion_language_model",
  "Direct_preference_optimization",
  "ELMo",
  "Embedding_(machine_learning)",
  "GPT-2",
  "GPT-3",
  "GPT-4",
  "GPT_(model_family)",
  "Google_DeepMind",
  "Gopher",
  "Hugging_Face",
  "In-context_learning",
  "InstructGPT",
  "Instruction_tuning",
  "Knowledge_distillation",
  "LLaMA",
  "LLaVA",
  "Llama_(model_family)",
  "MMLU",
  "Masked_language_modeling",
  "Meta_AI",
  "Mixtral_8x7B",
  "Mixture_of_experts",
  "Moonshot_AI",
  "Next-token_prediction",
  "OpenAI",
  "Pretraining",
  "Qwen_(model_family)",
  "Reinforcement_learning_from_human_feedback",
  "Reinforcement_learning_with_verifiable_rewards",
  "Scaling_laws",
  "Self-attention",
  "Softmax",
  "T5",
  "Teacher_forcing",
  "Tokenization",
  "Transformer_(architecture)",
  "ULMFiT",
  "Word2vec",
  "Zhipu_AI",
  "o1",
  "xAI"
 ],
 "inbound": [
  "AFM-4.5B",
  "AI21_Labs",
  "AI_2027",
  "ALBERT",
  "Activation_steering",
  "Activation_verbalizer",
  "Adam_(optimizer)",
  "Ai2",
  "Aleph_Alpha",
  "Alex_Krizhevsky",
  "AlphaFold",
  "AlphaGo",
  "AlphaZero",
  "Amazon_Titan",
  "Andrej_Karpathy",
  "Andrew_Ng",
  "Anthropic",
  "Apple_foundation_models",
  "Arcee_AI",
  "Arthur_Samuel",
  "Attention_sink",
  "Aya",
  "BERT",
  "BLIP",
  "BLOOM",
  "Backpropagation",
  "Baidu",
  "Cerebras_Systems",
  "Chain-of-thought_prompting",
  "ChatGLM",
  "Circuits_(interpretability)",
  "Claude_(model_family)",
  "Claude_1",
  "Claude_3.5_Haiku",
  "Claude_Instant",
  "Claude_Opus_4.1",
  "Claude_Opus_4.6",
  "Claude_Opus_4.8",
  "Claude_Opus_5",
  "Claude_Sonnet_4.5",
  "Claude_Sonnet_5",
  "Code_Llama",
  "Codestral",
  "Cohere",
  "Comma_(models)",
  "Common_Crawl",
  "Common_Pile",
  "Compositional_generalization",
  "Constitutional_AI",
  "Cross-entropy_loss",
  "Cursor",
  "DALL-E",
  "DBRX",
  "Databricks",
  "David_Rumelhart",
  "David_Silver",
  "DeepSeek",
  "DeepSeek-V3.1",
  "Demis_Hassabis",
  "Diffusion_language_model",
  "Direct_preference_optimization",
  "DistilBERT",
  "Distributed_training",
  "ELECTRA",
  "ELMo",
  "ERNIE_5",
  "ERNIE_Bot",
  "EleutherAI",
  "Eliciting_Latent_Knowledge",
  "Embedding_(machine_learning)",
  "FLUX",
  "Feed-forward_network",
  "Flamingo",
  "Flan-T5",
  "FlashAttention",
  "Frank_Rosenblatt",
  "François_Chollet",
  "GLM-130B",
  "GLM-4",
  "GLM-4.5",
  "GLM-4.6",
  "GLM-5",
  "GPT-1",
  "GPT-2",
  "GPT-3",
  "GPT-3.5",
  "GPT-4",
  "GPT-4.5",
  "GPT-4_Turbo",
  "GPT-4o",
  "GPT-4o_mini",
  "GPT-5.1",
  "GPT-5.2",
  "GPT-J",
  "GPT-Neo",
  "GPT-NeoX-20B",
  "GPT_(model_family)",
  "Gemini_(model_family)",
  "Gemini_1.0",
  "Gemini_3.5",
  "Genie",
  "Geoffrey_Hinton",
  "GloVe",
  "Google_DeepMind",
  "Gopher",
  "Grok-1.5",
  "Grok-2",
  "Grok_3",
  "Grok_4",
  "Grok_4.1",
  "Grok_4.5",
  "Hailuo",
  "Hugging_Face",
  "HunyuanVideo",
  "Hunyuan_3.0",
  "Hybrid_architecture_(LLM)",
  "Ilya_Sutskever",
  "In-context_learning",
  "Instruction_tuning",
  "InternLM",
  "InternVL",
  "Jurassic_(models)",
  "Keras",
  "Knowledge_distillation",
  "LFM_(models)",
  "LLaMA",
  "LLaVA",
  "LSTM",
  "LaMDA",
  "Layer_normalization",
  "Ling_(models)",
  "Liquid_AI",
  "Llama_(model_family)",
  "Llama_3.1",
  "Llama_3.3",
  "Logit_lens",
  "MAI-1",
  "MMLU",
  "Mechanistic_interpretability",
  "Meena",
  "Megatron-LM",
  "Megatron-Turing_NLG",
  "Meta_AI",
  "Microsoft_AI",
  "MiniMax-M2",
  "MiniMax-Text-01",
  "Mistral_AI",
  "Mistral_Large",
  "Mistral_Medium_3",
  "Mistral_Small",
  "Mixtral_8x22B",
  "Mixture_of_experts",
  "Molmo",
  "Moondream",
  "Moondream_(model_family)",
  "Moonshot_AI",
  "Multi-head_attention",
  "Multi-head_latent_attention",
  "NETtalk",
  "NVIDIA",
  "Ndea",
  "Nemotron",
  "Next-token_prediction",
  "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "Nous_Research",
  "OLMo",
  "OPT",
  "OpenAI",
  "PaLM",
  "PaLM_2",
  "PagedAttention",
  "Paul_Werbos",
  "Pixtral",
  "Pretraining",
  "Prometheus",
  "Qwen2",
  "Qwen2.5-Max",
  "Qwen2.5-VL",
  "Qwen3",
  "Qwen3-Coder",
  "Qwen3-Coder-Next",
  "Qwen_1",
  "RMSNorm",
  "Reinforcement_learning_from_human_feedback",
  "Reinforcement_learning_with_verifiable_rewards",
  "Richard_Sutton",
  "RoBERTa",
  "Safe_Superintelligence",
  "Scaling_laws",
  "Seed-OSS",
  "Segment_Anything",
  "Self-attention",
  "Seminal_AI_essays",
  "SenseNova",
  "SenseTime",
  "Simulators_(essay)",
  "Sliding-window_attention",
  "SmolLM",
  "Softmax",
  "Software_2.0",
  "Sora",
  "Sparse_attention",
  "Sparse_autoencoder",
  "Stability_AI",
  "State-space_model",
  "Step-2",
  "Superposition_(interpretability)",
  "Supervised_fine-tuning",
  "SynthID",
  "T5",
  "Teacher_forcing",
  "Technology_Innovation_Institute",
  "The_Bitter_Lesson",
  "The_Illustrated_Transformer",
  "The_Pile",
  "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "Thinking_Machines_Lab",
  "Tokenization",
  "Transformer_(architecture)",
  "Trinity_(model_family)",
  "Trinity_Large",
  "Trinity_Large_Thinking",
  "Trinity_Mini",
  "Trinity_Nano",
  "Turing-NLG",
  "ULMFiT",
  "Vicuna",
  "Vision_Transformer",
  "Walter_Pitts",
  "Warren_McCulloch",
  "What_Failure_Looks_Like",
  "Whisper",
  "Word2vec",
  "XLNet",
  "YaLM-100B",
  "Yandex",
  "YandexGPT",
  "Yann_LeCun",
  "Yi_(model_family)",
  "Yoshua_Bengio",
  "Zhipu_AI",
  "iFlytek_Spark",
  "o4-mini",
  "xAI"
 ],
 "url": "wiki/Large_language_model.html",
 "built": "2026-07-24"
}
