{
 "canonical": {
  "01.AI": "01.AI",
  "AFM": "Apple_foundation_models",
  "AFM 4.5B": "AFM-4.5B",
  "AFM-4.5B": "AFM-4.5B",
  "AI & Compute": "AI_and_Compute",
  "AI 2027": "AI_2027",
  "AI 2027 scenario": "AI_2027",
  "AI and Compute": "AI_and_Compute",
  "AI and Compute (OpenAI)": "AI_and_Compute",
  "AI-2027": "AI_2027",
  "AI2": "Ai2",
  "AI2027": "AI_2027",
  "AI21": "AI21_Labs",
  "AI21 Labs": "AI21_Labs",
  "AI21_Labs": "AI21_Labs",
  "AI_2027": "AI_2027",
  "AI_and_Compute": "AI_and_Compute",
  "ALBERT": "ALBERT",
  "AV explanations": "Activation_verbalizer",
  "Activation oracle": "Activation_verbalizer",
  "Activation steering": "Activation_steering",
  "Activation verbalizer": "Activation_verbalizer",
  "Activation_steering": "Activation_steering",
  "Activation_verbalizer": "Activation_verbalizer",
  "Adam (optimizer)": "Adam_(optimizer)",
  "Adam optimizer": "Adam_(optimizer)",
  "AdamW": "Adam_(optimizer)",
  "Adam_(optimizer)": "Adam_(optimizer)",
  "Ai2": "Ai2",
  "Aleph Alpha": "Aleph_Alpha",
  "Aleph_Alpha": "Aleph_Alpha",
  "Alex Krizhevsky": "Alex_Krizhevsky",
  "AlexNet": "AlexNet",
  "Alex_Krizhevsky": "Alex_Krizhevsky",
  "Alibaba Qwen": "Qwen_team",
  "Allen Institute for AI": "Ai2",
  "Allen Institute for Artificial Intelligence": "Ai2",
  "Alpaca": "Alpaca",
  "AlphaFold": "AlphaFold",
  "AlphaFold 2": "AlphaFold",
  "AlphaFold2": "AlphaFold",
  "AlphaGo": "AlphaGo",
  "AlphaGo Lee": "AlphaGo",
  "AlphaGo Master": "AlphaGo",
  "AlphaGo Zero": "AlphaGo",
  "AlphaZero": "AlphaZero",
  "Amazon Nova": "Amazon_Nova",
  "Amazon Titan": "Amazon_Titan",
  "Amazon_Nova": "Amazon_Nova",
  "Amazon_Titan": "Amazon_Titan",
  "Analytic distillation": "Analytic_distillation",
  "Analytic_distillation": "Analytic_distillation",
  "Andrej Karpathy": "Andrej_Karpathy",
  "Andrej_Karpathy": "Andrej_Karpathy",
  "Andrew Ng": "Andrew_Ng",
  "Andrew_Ng": "Andrew_Ng",
  "Ant Group": "Ant_Group",
  "Ant Ling": "Ant_Group",
  "Ant_Group": "Ant_Group",
  "Anthropic": "Anthropic",
  "Anysphere": "Cursor",
  "Apple Intelligence models": "Apple_foundation_models",
  "Apple foundation models": "Apple_foundation_models",
  "Apple_foundation_models": "Apple_foundation_models",
  "Arcee": "Arcee_AI",
  "Arcee AI": "Arcee_AI",
  "Arcee Foundation Model": "AFM-4.5B",
  "Arcee_AI": "Arcee_AI",
  "Arthur L. Samuel": "Arthur_Samuel",
  "Arthur Samuel": "Arthur_Samuel",
  "Arthur_Samuel": "Arthur_Samuel",
  "Attention Is All You Need": "Transformer_(architecture)",
  "Attention sink": "Attention_sink",
  "Attention_sink": "Attention_sink",
  "Attribution graphs": "Circuits_(interpretability)",
  "Autoregressive language modeling": "Next-token_prediction",
  "Aya": "Aya",
  "BERT": "BERT",
  "BLIP": "BLIP",
  "BLIP-2": "BLIP",
  "BLOOM": "BLOOM",
  "BPE": "Tokenization",
  "Back-propagation": "Backpropagation",
  "Backprop": "Backpropagation",
  "Backpropagation": "Backpropagation",
  "Baichuan": "Baichuan",
  "Baichuan Intelligence": "Baichuan",
  "Baidu": "Baidu",
  "Baidu ERNIE team": "Baidu",
  "Bailing": "Ant_Group",
  "Bailing (models)": "Ling_(models)",
  "BigScience BLOOM": "BLOOM",
  "Bitter Lesson": "The_Bitter_Lesson",
  "Boltzmann machine": "Boltzmann_machine",
  "Boltzmann_machine": "Boltzmann_machine",
  "Byte-pair encoding": "Tokenization",
  "ByteDance": "ByteDance",
  "ByteDance Seed": "ByteDance",
  "CAI": "Constitutional_AI",
  "CLIP": "CLIP",
  "Cerebras": "Cerebras_Systems",
  "Cerebras Systems": "Cerebras_Systems",
  "Cerebras-GPT": "Cerebras-GPT",
  "Cerebras_Systems": "Cerebras_Systems",
  "Chain of thought": "Chain-of-thought_prompting",
  "Chain-of-thought prompting": "Chain-of-thought_prompting",
  "Chain-of-thought_prompting": "Chain-of-thought_prompting",
  "ChatGLM": "ChatGLM",
  "ChatGPT (initial model)": "GPT-3.5",
  "Chinchilla": "Chinchilla",
  "Chinchilla scaling": "Scaling_laws",
  "Chinchilla scaling laws (model)": "Chinchilla",
  "Chinese AI labs": "Chinese_AI_labs",
  "Chinese labs": "Chinese_AI_labs",
  "Chinese_AI_labs": "Chinese_AI_labs",
  "Circuit tracing": "Circuits_(interpretability)",
  "Circuits (interpretability)": "Circuits_(interpretability)",
  "Circuits_(interpretability)": "Circuits_(interpretability)",
  "Claude (2023 model)": "Claude_1",
  "Claude (model family)": "Claude_(model_family)",
  "Claude 1": "Claude_1",
  "Claude 2": "Claude_2",
  "Claude 3": "Claude_3",
  "Claude 3 Haiku": "Claude_3",
  "Claude 3 Opus": "Claude_3",
  "Claude 3 Sonnet": "Claude_3",
  "Claude 3.5 Haiku": "Claude_3.5_Haiku",
  "Claude 3.5 Sonnet": "Claude_3.5_Sonnet",
  "Claude 3.7 Sonnet": "Claude_3.7_Sonnet",
  "Claude 4": "Claude_4",
  "Claude Fable 5": "Claude_Fable_5",
  "Claude Haiku 4.5": "Claude_Haiku_4.5",
  "Claude Instant": "Claude_Instant",
  "Claude Mythos 5": "Claude_Fable_5",
  "Claude Opus 4": "Claude_4",
  "Claude Opus 4.1": "Claude_Opus_4.1",
  "Claude Opus 4.5": "Claude_Opus_4.5",
  "Claude Opus 4.6": "Claude_Opus_4.6",
  "Claude Opus 4.8": "Claude_Opus_4.8",
  "Claude Opus 5": "Claude_Opus_5",
  "Claude Sonnet 4": "Claude_4",
  "Claude Sonnet 4.5": "Claude_Sonnet_4.5",
  "Claude Sonnet 5": "Claude_Sonnet_5",
  "Claude_(model_family)": "Claude_(model_family)",
  "Claude_1": "Claude_1",
  "Claude_2": "Claude_2",
  "Claude_3": "Claude_3",
  "Claude_3.5_Haiku": "Claude_3.5_Haiku",
  "Claude_3.5_Sonnet": "Claude_3.5_Sonnet",
  "Claude_3.7_Sonnet": "Claude_3.7_Sonnet",
  "Claude_4": "Claude_4",
  "Claude_Fable_5": "Claude_Fable_5",
  "Claude_Haiku_4.5": "Claude_Haiku_4.5",
  "Claude_Instant": "Claude_Instant",
  "Claude_Opus_4.1": "Claude_Opus_4.1",
  "Claude_Opus_4.5": "Claude_Opus_4.5",
  "Claude_Opus_4.6": "Claude_Opus_4.6",
  "Claude_Opus_4.8": "Claude_Opus_4.8",
  "Claude_Opus_5": "Claude_Opus_5",
  "Claude_Sonnet_4.5": "Claude_Sonnet_4.5",
  "Claude_Sonnet_5": "Claude_Sonnet_5",
  "CoT": "Chain-of-thought_prompting",
  "Code Llama": "Code_Llama",
  "Code_Llama": "Code_Llama",
  "Codestral": "Codestral",
  "Codex (2021 model)": "Codex_(2021_model)",
  "Codex_(2021_model)": "Codex_(2021_model)",
  "Cohere": "Cohere",
  "Comma (models)": "Comma_(models)",
  "Comma v0.1": "Comma_(models)",
  "Comma_(models)": "Comma_(models)",
  "Command (models)": "Command_(models)",
  "Command A": "Command_(models)",
  "Command R": "Command_(models)",
  "Command R+": "Command_(models)",
  "Command_(models)": "Command_(models)",
  "Common Crawl": "Common_Crawl",
  "Common Pile": "Common_Pile",
  "Common Pile v0.1": "Common_Pile",
  "Common_Crawl": "Common_Crawl",
  "Common_Pile": "Common_Pile",
  "Compositional generalization": "Compositional_generalization",
  "Compositional_generalization": "Compositional_generalization",
  "Constitutional AI": "Constitutional_AI",
  "Constitutional_AI": "Constitutional_AI",
  "Cross-entropy": "Cross-entropy_loss",
  "Cross-entropy loss": "Cross-entropy_loss",
  "Cross-entropy_loss": "Cross-entropy_loss",
  "Cursor": "Cursor",
  "Cursor (company)": "Cursor",
  "DALL-E": "DALL-E",
  "DALL-E 1": "DALL-E",
  "DALL-E 2": "DALL-E_2",
  "DALL-E 3": "DALL-E_3",
  "DALL-E_2": "DALL-E_2",
  "DALL-E_3": "DALL-E_3",
  "DBRX": "DBRX",
  "DPO": "Direct_preference_optimization",
  "DSA": "Sparse_attention",
  "Data parallelism": "Distributed_training",
  "Databricks": "Databricks",
  "David E. Rumelhart": "David_Rumelhart",
  "David Everett Rumelhart": "David_Rumelhart",
  "David Rumelhart": "David_Rumelhart",
  "David Silver": "David_Silver",
  "David Silver (computer scientist)": "David_Silver",
  "David_Rumelhart": "David_Rumelhart",
  "David_Silver": "David_Silver",
  "DeepMind": "Google_DeepMind",
  "DeepSeek": "DeepSeek",
  "DeepSeek (model family)": "DeepSeek_(model_family)",
  "DeepSeek LLM": "DeepSeek_LLM",
  "DeepSeek R1 release shock": "DeepSeek_R1_release_shock",
  "DeepSeek-Coder-V2": "DeepSeek-Coder-V2",
  "DeepSeek-R1": "DeepSeek-R1",
  "DeepSeek-V2": "DeepSeek-V2",
  "DeepSeek-V3": "DeepSeek-V3",
  "DeepSeek-V3.1": "DeepSeek-V3.1",
  "DeepSeekMoE": "Mixture_of_experts",
  "DeepSeek_(model_family)": "DeepSeek_(model_family)",
  "DeepSeek_LLM": "DeepSeek_LLM",
  "DeepSeek_R1_release_shock": "DeepSeek_R1_release_shock",
  "Demis Hassabis": "Demis_Hassabis",
  "Demis_Hassabis": "Demis_Hassabis",
  "Devstral": "Devstral",
  "Dictionary learning": "Sparse_autoencoder",
  "Diffusion language model": "Diffusion_language_model",
  "Diffusion_language_model": "Diffusion_language_model",
  "Direct preference optimization": "Direct_preference_optimization",
  "Direct_preference_optimization": "Direct_preference_optimization",
  "DistilBERT": "DistilBERT",
  "Distillation": "Knowledge_distillation",
  "Distributed training": "Distributed_training",
  "Distributed_training": "Distributed_training",
  "Doubao": "Doubao",
  "Dropout": "Dropout",
  "ELECTRA": "ELECTRA",
  "ELK": "Eliciting_Latent_Knowledge",
  "ELMo": "ELMo",
  "ERNIE 3.0": "ERNIE_Bot",
  "ERNIE 4.0": "ERNIE_Bot",
  "ERNIE 4.5": "ERNIE_Bot",
  "ERNIE 5": "ERNIE_5",
  "ERNIE 5.0": "ERNIE_5",
  "ERNIE 5.1": "ERNIE_5",
  "ERNIE Bot": "ERNIE_Bot",
  "ERNIE X1": "ERNIE_Bot",
  "ERNIE_5": "ERNIE_5",
  "ERNIE_Bot": "ERNIE_Bot",
  "EleutherAI": "EleutherAI",
  "Eliciting Latent Knowledge": "Eliciting_Latent_Knowledge",
  "Eliciting_Latent_Knowledge": "Eliciting_Latent_Knowledge",
  "Embedding (machine learning)": "Embedding_(machine_learning)",
  "Embedding_(machine_learning)": "Embedding_(machine_learning)",
  "Expert parallelism": "Distributed_training",
  "Expert routing": "Mixture_of_experts",
  "FAIR": "Meta_AI",
  "FFN": "Feed-forward_network",
  "FLAN": "Flan-T5",
  "FLUX": "FLUX",
  "FLUX.1": "FLUX",
  "FSDP": "Distributed_training",
  "Falcon (models)": "Falcon_(models)",
  "Falcon 180B": "Falcon_(models)",
  "Falcon 40B": "Falcon_(models)",
  "Falcon Mamba": "Falcon_(models)",
  "Falcon_(models)": "Falcon_(models)",
  "Feed-forward network": "Feed-forward_network",
  "Feed-forward_network": "Feed-forward_network",
  "Few-shot learning (LLMs)": "In-context_learning",
  "Fire alarm for AGI": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "Flamingo": "Flamingo",
  "Flamingo (model)": "Flamingo",
  "Flan-PaLM": "Flan-T5",
  "Flan-T5": "Flan-T5",
  "FlashAttention": "FlashAttention",
  "FlashAttention-2": "FlashAttention",
  "FlashAttention-3": "FlashAttention",
  "Francois Chollet": "François_Chollet",
  "Frank Rosenblatt": "Frank_Rosenblatt",
  "Frank_Rosenblatt": "Frank_Rosenblatt",
  "François Chollet": "François_Chollet",
  "François_Chollet": "François_Chollet",
  "GLM (model family)": "GLM_(model_family)",
  "GLM-130B": "GLM-130B",
  "GLM-4": "GLM-4",
  "GLM-4.5": "GLM-4.5",
  "GLM-4.6": "GLM-4.6",
  "GLM-5": "GLM-5",
  "GLM-5.1": "GLM-5",
  "GLM-5.2": "GLM-5",
  "GLM_(model_family)": "GLM_(model_family)",
  "GPT": "GPT-1",
  "GPT (model family)": "GPT_(model_family)",
  "GPT-1": "GPT-1",
  "GPT-2": "GPT-2",
  "GPT-3": "GPT-3",
  "GPT-3.5": "GPT-3.5",
  "GPT-4": "GPT-4",
  "GPT-4 Turbo": "GPT-4_Turbo",
  "GPT-4.5": "GPT-4.5",
  "GPT-4_Turbo": "GPT-4_Turbo",
  "GPT-4o": "GPT-4o",
  "GPT-4o mini": "GPT-4o_mini",
  "GPT-4o_mini": "GPT-4o_mini",
  "GPT-5": "GPT-5",
  "GPT-5.1": "GPT-5.1",
  "GPT-5.2": "GPT-5.2",
  "GPT-5.6": "GPT-5.6",
  "GPT-5.6 Luna": "GPT-5.6",
  "GPT-5.6 Sol": "GPT-5.6",
  "GPT-5.6 Terra": "GPT-5.6",
  "GPT-J": "GPT-J",
  "GPT-J-6B": "GPT-J",
  "GPT-Neo": "GPT-Neo",
  "GPT-NeoX": "GPT-NeoX-20B",
  "GPT-NeoX-20B": "GPT-NeoX-20B",
  "GPT3": "GPT-3",
  "GPT4": "GPT-4",
  "GPT_(model_family)": "GPT_(model_family)",
  "GQA": "Multi-head_attention",
  "GShard": "Mixture_of_experts",
  "Gemini (2023 model)": "Gemini_1.0",
  "Gemini (model family)": "Gemini_(model_family)",
  "Gemini 1.0": "Gemini_1.0",
  "Gemini 1.5": "Gemini_1.5",
  "Gemini 2.0": "Gemini_2.0",
  "Gemini 2.5": "Gemini_2.5",
  "Gemini 3": "Gemini_3",
  "Gemini 3.5": "Gemini_3.5",
  "Gemini 3.5 Flash": "Gemini_3.5",
  "Gemini 3.5 Pro": "Gemini_3.5",
  "Gemini_(model_family)": "Gemini_(model_family)",
  "Gemini_1.0": "Gemini_1.0",
  "Gemini_1.5": "Gemini_1.5",
  "Gemini_2.0": "Gemini_2.0",
  "Gemini_2.5": "Gemini_2.5",
  "Gemini_3": "Gemini_3",
  "Gemini_3.5": "Gemini_3.5",
  "Gemma": "Gemma",
  "Gemma 1": "Gemma",
  "Gemma 2": "Gemma",
  "Gemma 3": "Gemma",
  "Genie": "Genie",
  "Genie 2": "Genie",
  "Genie 3": "Genie",
  "Geoff Hinton": "Geoffrey_Hinton",
  "Geoffrey Hinton": "Geoffrey_Hinton",
  "Geoffrey_Hinton": "Geoffrey_Hinton",
  "GloVe": "GloVe",
  "Golden Gate Claude": "Activation_steering",
  "Google DeepMind": "Google_DeepMind",
  "Google_DeepMind": "Google_DeepMind",
  "Gopher": "Gopher",
  "Granite 3.x": "IBM_Granite",
  "Griffin (architecture)": "Hybrid_architecture_(LLM)",
  "Grok (model family)": "Grok_(model_family)",
  "Grok 3": "Grok_3",
  "Grok 4": "Grok_4",
  "Grok 4.1": "Grok_4.1",
  "Grok 4.5": "Grok_4.5",
  "Grok V9-Medium": "Grok_4.5",
  "Grok-1": "Grok-1",
  "Grok-1.5": "Grok-1.5",
  "Grok-2": "Grok-2",
  "Grok_(model_family)": "Grok_(model_family)",
  "Grok_3": "Grok_3",
  "Grok_4": "Grok_4",
  "Grok_4.1": "Grok_4.1",
  "Grok_4.5": "Grok_4.5",
  "Grouped-query attention": "Multi-head_attention",
  "Hailuo": "Hailuo",
  "Harmonium": "Boltzmann_machine",
  "High-Flyer": "DeepSeek",
  "Hopfield model": "Hopfield_network",
  "Hopfield net": "Hopfield_network",
  "Hopfield network": "Hopfield_network",
  "Hopfield_network": "Hopfield_network",
  "Huawei": "Huawei",
  "Huawei Noah's Ark": "Huawei",
  "Hugging Face": "Hugging_Face",
  "Hugging_Face": "Hugging_Face",
  "HunYuan 3.0": "Hunyuan_3.0",
  "Hunyuan": "Hunyuan",
  "Hunyuan 3.0": "Hunyuan_3.0",
  "Hunyuan-Large": "Hunyuan",
  "HunyuanVideo": "HunyuanVideo",
  "Hunyuan_3.0": "Hunyuan_3.0",
  "Hy3": "Hunyuan_3.0",
  "Hybrid architecture (LLM)": "Hybrid_architecture_(LLM)",
  "Hybrid_architecture_(LLM)": "Hybrid_architecture_(LLM)",
  "IBM Granite": "IBM_Granite",
  "IBM_Granite": "IBM_Granite",
  "IDEFICS": "IDEFICS",
  "ILSVRC": "ImageNet",
  "IPO": "Direct_preference_optimization",
  "Ian Goodfellow": "Ian_Goodfellow",
  "Ian_Goodfellow": "Ian_Goodfellow",
  "Illustrated Transformer": "The_Illustrated_Transformer",
  "Ilya Sutskever": "Ilya_Sutskever",
  "Ilya_Sutskever": "Ilya_Sutskever",
  "ImageNet": "ImageNet",
  "ImageNet Large Scale Visual Recognition Challenge": "ImageNet",
  "Imagen": "Imagen",
  "Imagen 2": "Imagen",
  "Imagen 3": "Imagen",
  "Imagen 4": "Imagen",
  "In-context learning": "In-context_learning",
  "In-context_learning": "In-context_learning",
  "Inference-time scaling": "Test-time_compute",
  "InstructGPT": "InstructGPT",
  "Instruction tuning": "Instruction_tuning",
  "Instruction_tuning": "Instruction_tuning",
  "InternLM": "InternLM",
  "InternLM (org)": "Shanghai_AI_Laboratory",
  "InternLM2": "InternLM",
  "InternLM3": "InternLM",
  "InternVL": "InternVL",
  "Intra-attention": "Self-attention",
  "Jamba": "Jamba",
  "January 2025 DeepSeek shock": "DeepSeek_R1_release_shock",
  "John Hopfield": "John_Hopfield",
  "John J. Hopfield": "John_Hopfield",
  "John Joseph Hopfield": "John_Hopfield",
  "John Jumper": "John_Jumper",
  "John M. Jumper": "John_Jumper",
  "John Michael Jumper": "John_Jumper",
  "John_Hopfield": "John_Hopfield",
  "John_Jumper": "John_Jumper",
  "Jurassic (models)": "Jurassic_(models)",
  "Jurassic-1": "Jurassic_(models)",
  "Jurassic-2": "Jurassic_(models)",
  "Jurassic_(models)": "Jurassic_(models)",
  "KTO": "Direct_preference_optimization",
  "Kaplan scaling laws": "Scaling_laws",
  "Keras": "Keras",
  "Keras 3": "Keras",
  "KerasHub": "Keras",
  "Kimi (model family)": "Kimi_(model_family)",
  "Kimi K2": "Kimi_K2",
  "Kimi K3": "Kimi_K3",
  "Kimi k1.5": "Kimi_k1.5",
  "Kimi_(model_family)": "Kimi_(model_family)",
  "Kimi_K2": "Kimi_K2",
  "Kimi_K3": "Kimi_K3",
  "Kimi_k1.5": "Kimi_k1.5",
  "Kling": "Kling",
  "Knowledge distillation": "Knowledge_distillation",
  "Knowledge_distillation": "Knowledge_distillation",
  "LFM (models)": "LFM_(models)",
  "LFM2": "LFM_(models)",
  "LFM_(models)": "LFM_(models)",
  "LLM": "Large_language_model",
  "LLaMA": "LLaMA",
  "LLaMA 1": "LLaMA",
  "LLaVA": "LLaVA",
  "LSTM": "LSTM",
  "LaMDA": "LaMDA",
  "Lab missions": "Stated_missions_of_AI_labs",
  "Language model pretraining": "Pretraining",
  "Large language model": "Large_language_model",
  "Large_language_model": "Large_language_model",
  "LatentQA": "Activation_verbalizer",
  "Layer normalization": "Layer_normalization",
  "LayerNorm": "Layer_normalization",
  "Layer_normalization": "Layer_normalization",
  "Ling (models)": "Ling_(models)",
  "Ling 2.0": "Ling_(models)",
  "Ling_(models)": "Ling_(models)",
  "Liquid AI": "Liquid_AI",
  "Liquid Foundation Models": "LFM_(models)",
  "Liquid_AI": "Liquid_AI",
  "Llama (model family)": "Llama_(model_family)",
  "Llama 2": "Llama_2",
  "Llama 3": "Llama_3",
  "Llama 3.1": "Llama_3.1",
  "Llama 3.1 405B": "Llama_3.1",
  "Llama 3.2": "Llama_3.2",
  "Llama 3.3": "Llama_3.3",
  "Llama 4": "Llama_4",
  "Llama 4 Behemoth": "Llama_4",
  "Llama 4 Maverick": "Llama_4",
  "Llama 4 Scout": "Llama_4",
  "Llama_(model_family)": "Llama_(model_family)",
  "Llama_2": "Llama_2",
  "Llama_3": "Llama_3",
  "Llama_3.1": "Llama_3.1",
  "Llama_3.2": "Llama_3.2",
  "Llama_3.3": "Llama_3.3",
  "Llama_4": "Llama_4",
  "Logit lens": "Logit_lens",
  "Logit_lens": "Logit_lens",
  "Long short-term memory": "LSTM",
  "LongCat": "LongCat",
  "LongCat (org)": "Meituan",
  "LongCat-2.0": "LongCat",
  "LongCat-Flash": "LongCat",
  "M1 (model)": "MiniMax-M1",
  "M87 Labs": "Moondream",
  "MAI-1": "MAI-1",
  "MAI-1-preview": "MAI-1",
  "MLA": "Multi-head_latent_attention",
  "MLM": "Masked_language_modeling",
  "MMLU": "MMLU",
  "MPT (models)": "MPT_(models)",
  "MPT-30B": "MPT_(models)",
  "MPT-7B": "MPT_(models)",
  "MPT_(models)": "MPT_(models)",
  "MQA": "Multi-head_attention",
  "MSL": "Meta_AI",
  "MT-NLG 530B": "Megatron-Turing_NLG",
  "Machines of Loving Grace": "Machines_of_Loving_Grace",
  "Machines_of_Loving_Grace": "Machines_of_Loving_Grace",
  "Magistral": "Magistral",
  "Mamba (architecture)": "State-space_model",
  "Mark 1 Perceptron": "Perceptron",
  "Mark I Perceptron": "Perceptron",
  "Masked language modeling": "Masked_language_modeling",
  "Masked_language_modeling": "Masked_language_modeling",
  "Massive Multitask Language Understanding": "MMLU",
  "Mechanistic interpretability": "Mechanistic_interpretability",
  "Mechanistic_interpretability": "Mechanistic_interpretability",
  "Meena": "Meena",
  "Megatron-LM": "Megatron-LM",
  "Megatron-Turing NLG": "Megatron-Turing_NLG",
  "Megatron-Turing_NLG": "Megatron-Turing_NLG",
  "Meituan": "Meituan",
  "Meta AI": "Meta_AI",
  "Meta Superintelligence Labs": "Meta_AI",
  "Meta_AI": "Meta_AI",
  "MiMo": "MiMo",
  "MiMo (models)": "MiMo",
  "Microsoft": "Microsoft_AI",
  "Microsoft AI": "Microsoft_AI",
  "Microsoft_AI": "Microsoft_AI",
  "MiniMax": "MiniMax",
  "MiniMax-M1": "MiniMax-M1",
  "MiniMax-M2": "MiniMax-M2",
  "MiniMax-Text-01": "MiniMax-Text-01",
  "Mistral (model family)": "Mistral_(model_family)",
  "Mistral 7B": "Mistral_7B",
  "Mistral AI": "Mistral_AI",
  "Mistral Large": "Mistral_Large",
  "Mistral Large 2": "Mistral_Large",
  "Mistral Medium 3": "Mistral_Medium_3",
  "Mistral Small": "Mistral_Small",
  "Mistral_(model_family)": "Mistral_(model_family)",
  "Mistral_7B": "Mistral_7B",
  "Mistral_AI": "Mistral_AI",
  "Mistral_Large": "Mistral_Large",
  "Mistral_Medium_3": "Mistral_Medium_3",
  "Mistral_Small": "Mistral_Small",
  "Mixtral": "Mixtral_8x7B",
  "Mixtral 8x22B": "Mixtral_8x22B",
  "Mixtral 8x7B": "Mixtral_8x7B",
  "Mixtral_8x22B": "Mixtral_8x22B",
  "Mixtral_8x7B": "Mixtral_8x7B",
  "Mixture of experts": "Mixture_of_experts",
  "Mixture_of_experts": "Mixture_of_experts",
  "MoE": "Mixture_of_experts",
  "Molmo": "Molmo",
  "Moondream": "Moondream",
  "Moondream (model family)": "Moondream_(model_family)",
  "Moondream 2": "Moondream_(model_family)",
  "Moondream 3": "Moondream_(model_family)",
  "Moondream 3 Preview": "Moondream_(model_family)",
  "Moondream_(model_family)": "Moondream_(model_family)",
  "Moonshot": "Moonshot_AI",
  "Moonshot AI": "Moonshot_AI",
  "Moonshot_AI": "Moonshot_AI",
  "Mosaic Research": "MosaicML",
  "MosaicML": "MosaicML",
  "Most Important Century": "The_Most_Important_Century",
  "Movie Gen": "Movie_Gen",
  "Movie_Gen": "Movie_Gen",
  "Multi-head attention": "Multi-head_attention",
  "Multi-head latent attention": "Multi-head_latent_attention",
  "Multi-head_attention": "Multi-head_attention",
  "Multi-head_latent_attention": "Multi-head_latent_attention",
  "Multi-query attention": "Multi-head_attention",
  "Mythos-class": "Claude_Fable_5",
  "NETtalk": "NETtalk",
  "NETtalk (neural network)": "NETtalk",
  "NVIDIA": "NVIDIA",
  "Ndea": "Ndea",
  "Ndea Labs": "Ndea",
  "Nemotron": "Nemotron",
  "Nemotron-4": "Nemotron",
  "NetTalk": "NETtalk",
  "Nettalk": "NETtalk",
  "Next-token prediction": "Next-token_prediction",
  "Next-token_prediction": "Next-token_prediction",
  "No Fire Alarm for AGI": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "No Fire Alarm for Artificial General Intelligence": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "NoPE": "Positional_encoding",
  "No_Fire_Alarm_for_Artificial_General_Intelligence": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "Nous": "Nous_Research",
  "Nous Research": "Nous_Research",
  "Nous_Research": "Nous_Research",
  "Nova (models)": "Amazon_Nova",
  "Nova Premier": "Amazon_Nova",
  "OLMo": "OLMo",
  "OLMo 2": "OLMo",
  "OLMo 3": "OLMo",
  "OPT": "OPT",
  "OPT-175B": "OPT",
  "ORPO": "Direct_preference_optimization",
  "OpenAI": "OpenAI",
  "OpenAI Codex": "Codex_(2021_model)",
  "OpenAI board crisis annotated edition": "Removal_of_Sam_Altman_from_OpenAI_(annotated_edition)",
  "OpenAI o1": "o1",
  "OpenAI o3": "o3",
  "OpenAI o4-mini": "o4-mini",
  "OpenGVLab": "Shanghai_AI_Laboratory",
  "Opus 5": "Claude_Opus_5",
  "Outcome reward model": "Reward_model",
  "PaLM": "PaLM",
  "PaLM 2": "PaLM_2",
  "PaLM_2": "PaLM_2",
  "PagedAttention": "PagedAttention",
  "PaliGemma": "PaliGemma",
  "Paul J. Werbos": "Paul_Werbos",
  "Paul John Werbos": "Paul_Werbos",
  "Paul Werbos": "Paul_Werbos",
  "Paul_Werbos": "Paul_Werbos",
  "Perceptron": "Perceptron",
  "Phi (model family)": "Phi_(model_family)",
  "Phi-1": "Phi_(model_family)",
  "Phi-2": "Phi_(model_family)",
  "Phi-3": "Phi_(model_family)",
  "Phi-4": "Phi_(model_family)",
  "Phi-4-reasoning": "Phi_(model_family)",
  "Phi_(model_family)": "Phi_(model_family)",
  "Pipeline parallelism": "Distributed_training",
  "Pixtral": "Pixtral",
  "Position-wise feed-forward network": "Feed-forward_network",
  "Positional encoding": "Positional_encoding",
  "Positional_encoding": "Positional_encoding",
  "Pre-training": "Pretraining",
  "Pretraining": "Pretraining",
  "Probing (interpretability)": "Logit_lens",
  "Process reward model": "Reward_model",
  "Project Prometheus": "Prometheus",
  "Prometheus": "Prometheus",
  "Pythia": "Pythia",
  "Pythia (model suite)": "Pythia",
  "QwQ-32B": "QwQ-32B",
  "Qwen (model family)": "Qwen_(model_family)",
  "Qwen 1": "Qwen_1",
  "Qwen 1.5": "Qwen_1",
  "Qwen 3.6": "Qwen3-Coder-Next",
  "Qwen team": "Qwen_team",
  "Qwen2": "Qwen2",
  "Qwen2.5": "Qwen2.5",
  "Qwen2.5-Max": "Qwen2.5-Max",
  "Qwen2.5-VL": "Qwen2.5-VL",
  "Qwen3": "Qwen3",
  "Qwen3-Coder": "Qwen3-Coder",
  "Qwen3-Coder-Next": "Qwen3-Coder-Next",
  "Qwen3-Max": "Qwen3-Max",
  "Qwen_(model_family)": "Qwen_(model_family)",
  "Qwen_1": "Qwen_1",
  "Qwen_team": "Qwen_team",
  "R1": "DeepSeek-R1",
  "RBM": "Boltzmann_machine",
  "RLHF": "Reinforcement_learning_from_human_feedback",
  "RLVR": "Reinforcement_learning_with_verifiable_rewards",
  "RMSNorm": "RMSNorm",
  "RWKV": "RWKV",
  "Reasoning models": "Test-time_compute",
  "Reinforcement learning from human feedback": "Reinforcement_learning_from_human_feedback",
  "Reinforcement learning with verifiable rewards": "Reinforcement_learning_with_verifiable_rewards",
  "Reinforcement_learning_from_human_feedback": "Reinforcement_learning_from_human_feedback",
  "Reinforcement_learning_with_verifiable_rewards": "Reinforcement_learning_with_verifiable_rewards",
  "Removal of Sam Altman from OpenAI (annotated edition)": "Removal_of_Sam_Altman_from_OpenAI_(annotated_edition)",
  "Removal_of_Sam_Altman_from_OpenAI_(annotated_edition)": "Removal_of_Sam_Altman_from_OpenAI_(annotated_edition)",
  "Representation engineering": "Activation_steering",
  "Restricted Boltzmann machine": "Boltzmann_machine",
  "Reward model": "Reward_model",
  "Reward_model": "Reward_model",
  "Rich Sutton": "Richard_Sutton",
  "Richard S. Sutton": "Richard_Sutton",
  "Richard Sutton": "Richard_Sutton",
  "Richard_Sutton": "Richard_Sutton",
  "RoBERTa": "RoBERTa",
  "Root mean square layer normalization": "RMSNorm",
  "S4": "State-space_model",
  "SAE": "Sparse_autoencoder",
  "SAM": "Segment_Anything",
  "SDXL": "Stable_Diffusion",
  "SFT": "Supervised_fine-tuning",
  "SSI": "Safe_Superintelligence",
  "SSM": "State-space_model",
  "Safe Superintelligence": "Safe_Superintelligence",
  "Safe_Superintelligence": "Safe_Superintelligence",
  "Sam Altman removal annotated edition": "Removal_of_Sam_Altman_from_OpenAI_(annotated_edition)",
  "Scaling hypothesis": "The_Scaling_Hypothesis",
  "Scaling laws": "Scaling_laws",
  "Scaling_laws": "Scaling_laws",
  "Seed-OSS": "Seed-OSS",
  "Seedance": "Seedream",
  "Seedream": "Seedream",
  "Segment Anything": "Segment_Anything",
  "Segment_Anything": "Segment_Anything",
  "Self-Instruct": "Instruction_tuning",
  "Self-attention": "Self-attention",
  "Seminal AI essays": "Seminal_AI_essays",
  "Seminal_AI_essays": "Seminal_AI_essays",
  "SenseNova": "SenseNova",
  "SenseNova (org)": "SenseTime",
  "SenseTime": "SenseTime",
  "SentencePiece": "Tokenization",
  "Seq2seq": "Seq2seq",
  "Sequence-to-sequence learning": "Seq2seq",
  "Shanghai AI Laboratory": "Shanghai_AI_Laboratory",
  "Shanghai_AI_Laboratory": "Shanghai_AI_Laboratory",
  "SigLIP": "Vision_Transformer",
  "Simulators (essay)": "Simulators_(essay)",
  "Simulators (janus)": "Simulators_(essay)",
  "Simulators_(essay)": "Simulators_(essay)",
  "Sinusoidal positional encoding": "Positional_encoding",
  "Sir Demis Hassabis": "Demis_Hassabis",
  "Situational Awareness (essay)": "Situational_Awareness_(essay)",
  "Situational Awareness: The Decade Ahead": "Situational_Awareness_(essay)",
  "Situational_Awareness_(essay)": "Situational_Awareness_(essay)",
  "Sliding-window attention": "Sliding-window_attention",
  "Sliding-window_attention": "Sliding-window_attention",
  "SmolLM": "SmolLM",
  "Softmax": "Softmax",
  "Softmax function": "Softmax",
  "Software 1.0": "Software_2.0",
  "Software 2.0": "Software_2.0",
  "Software 2.0 (essay)": "Software_2.0",
  "Software_2.0": "Software_2.0",
  "Sora": "Sora",
  "Sora (video model)": "Sora",
  "Sora 2": "Sora_2",
  "Sora_2": "Sora_2",
  "Span corruption": "Masked_language_modeling",
  "Spark (iFlytek)": "iFlytek_Spark",
  "Sparse attention": "Sparse_attention",
  "Sparse autoencoder": "Sparse_autoencoder",
  "Sparse_attention": "Sparse_attention",
  "Sparse_autoencoder": "Sparse_autoencoder",
  "Stability AI": "Stability_AI",
  "Stability_AI": "Stability_AI",
  "Stable Diffusion": "Stable_Diffusion",
  "Stable Diffusion 1.5": "Stable_Diffusion",
  "Stable Diffusion 3": "Stable_Diffusion",
  "Stable_Diffusion": "Stable_Diffusion",
  "Stanford Alpaca": "Alpaca",
  "StarCoder": "StarCoder",
  "StarCoder2": "StarCoder",
  "State-space model": "State-space_model",
  "State-space_model": "State-space_model",
  "Stated missions of AI labs": "Stated_missions_of_AI_labs",
  "Stated_missions_of_AI_labs": "Stated_missions_of_AI_labs",
  "Step-2": "Step-2",
  "StepFun": "StepFun",
  "Superposition (interpretability)": "Superposition_(interpretability)",
  "Superposition_(interpretability)": "Superposition_(interpretability)",
  "Supervised fine-tuning": "Supervised_fine-tuning",
  "Supervised_fine-tuning": "Supervised_fine-tuning",
  "Switch Transformer": "Mixture_of_experts",
  "SynthID": "SynthID",
  "Systematic generalization": "Compositional_generalization",
  "Systematicity": "Compositional_generalization",
  "T0": "Instruction_tuning",
  "T5": "T5",
  "TD-Gammon": "TD-Gammon",
  "TD-Gammon (backgammon)": "TD-Gammon",
  "TDGammon": "TD-Gammon",
  "TII": "Technology_Innovation_Institute",
  "Teacher forcing": "Teacher_forcing",
  "Teacher_forcing": "Teacher_forcing",
  "Technology Innovation Institute": "Technology_Innovation_Institute",
  "Technology_Innovation_Institute": "Technology_Innovation_Institute",
  "Tencent": "Tencent",
  "Tencent Hunyuan": "Tencent",
  "Tensor parallelism": "Distributed_training",
  "Test-time compute": "Test-time_compute",
  "Test-time_compute": "Test-time_compute",
  "Text embeddings": "Embedding_(machine_learning)",
  "Text-to-Text Transfer Transformer": "T5",
  "The Bitter Lesson": "The_Bitter_Lesson",
  "The Decade Ahead": "Situational_Awareness_(essay)",
  "The Illustrated Transformer": "The_Illustrated_Transformer",
  "The Most Important Century": "The_Most_Important_Century",
  "The Pile": "The_Pile",
  "The Scaling Hypothesis": "The_Scaling_Hypothesis",
  "The Unreasonable Effectiveness of RNNs": "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "The Unreasonable Effectiveness of Recurrent Neural Networks": "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "The most important century series": "The_Most_Important_Century",
  "The_Bitter_Lesson": "The_Bitter_Lesson",
  "The_Illustrated_Transformer": "The_Illustrated_Transformer",
  "The_Most_Important_Century": "The_Most_Important_Century",
  "The_Pile": "The_Pile",
  "The_Scaling_Hypothesis": "The_Scaling_Hypothesis",
  "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks": "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "There's No Fire Alarm for AGI": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "There's No Fire Alarm for Artificial General Intelligence": "No_Fire_Alarm_for_Artificial_General_Intelligence",
  "Thinking Machines Lab": "Thinking_Machines_Lab",
  "Thinking_Machines_Lab": "Thinking_Machines_Lab",
  "Titan (models)": "Amazon_Titan",
  "Tokenization": "Tokenization",
  "Transformer": "Transformer_(architecture)",
  "Transformer (architecture)": "Transformer_(architecture)",
  "Transformer_(architecture)": "Transformer_(architecture)",
  "Trinity": "Trinity_(model_family)",
  "Trinity (model family)": "Trinity_(model_family)",
  "Trinity Large": "Trinity_Large",
  "Trinity Large Base": "Trinity_Large",
  "Trinity Large Preview": "Trinity_Large",
  "Trinity Large Thinking": "Trinity_Large_Thinking",
  "Trinity Mini": "Trinity_Mini",
  "Trinity Nano": "Trinity_Nano",
  "Trinity-Large": "Trinity_Large",
  "Trinity-Large-Base": "Trinity_Large",
  "Trinity-Large-Preview": "Trinity_Large",
  "Trinity-Large-Thinking": "Trinity_Large_Thinking",
  "Trinity-Mini": "Trinity_Mini",
  "Trinity-Nano": "Trinity_Nano",
  "Trinity_(model_family)": "Trinity_(model_family)",
  "Trinity_Large": "Trinity_Large",
  "Trinity_Large_Thinking": "Trinity_Large_Thinking",
  "Trinity_Mini": "Trinity_Mini",
  "Trinity_Nano": "Trinity_Nano",
  "TrueBase": "Trinity_Large",
  "Tulu 3": "Tulu_3",
  "Tulu_3": "Tulu_3",
  "Turing-NLG": "Turing-NLG",
  "Tülu 3": "Tulu_3",
  "ULMFiT": "ULMFiT",
  "Unreasonable Effectiveness of Recurrent Neural Networks": "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "Vector search": "Embedding_(machine_learning)",
  "Veo": "Veo",
  "Veo 2": "Veo",
  "Veo 3": "Veo",
  "ViT": "Vision_Transformer",
  "Vicuna": "Vicuna",
  "Vision Transformer": "Vision_Transformer",
  "Vision_Transformer": "Vision_Transformer",
  "Walter H. Pitts": "Walter_Pitts",
  "Walter Pitts": "Walter_Pitts",
  "Walter_Pitts": "Walter_Pitts",
  "Warren McCulloch": "Warren_McCulloch",
  "Warren S. McCulloch": "Warren_McCulloch",
  "Warren_McCulloch": "Warren_McCulloch",
  "What Failure Looks Like": "What_Failure_Looks_Like",
  "What failure looks like (essay)": "What_Failure_Looks_Like",
  "What_Failure_Looks_Like": "What_Failure_Looks_Like",
  "Whisper": "Whisper",
  "Whisper (speech model)": "Whisper",
  "Word2vec": "Word2vec",
  "XLNet": "XLNet",
  "Xiaomi": "Xiaomi",
  "Xiaomi MiMo (org)": "Xiaomi",
  "YaLM": "YaLM-100B",
  "YaLM-100B": "YaLM-100B",
  "Yandex": "Yandex",
  "Yandex LLC": "Yandex",
  "Yandex N.V.": "Yandex",
  "YandexGPT": "YandexGPT",
  "YandexGPT 5": "YandexGPT",
  "Yann Le Cun": "Yann_LeCun",
  "Yann LeCun": "Yann_LeCun",
  "Yann_LeCun": "Yann_LeCun",
  "Yi (01.AI)": "01.AI",
  "Yi (model family)": "Yi_(model_family)",
  "Yi-34B": "Yi_(model_family)",
  "Yi-Large": "Yi_(model_family)",
  "Yi_(model_family)": "Yi_(model_family)",
  "Yoshua Bengio": "Yoshua_Bengio",
  "Yoshua_Bengio": "Yoshua_Bengio",
  "Z.ai": "Zhipu_AI",
  "ZeRO": "Distributed_training",
  "Zhipu": "Zhipu_AI",
  "Zhipu AI": "Zhipu_AI",
  "Zhipu_AI": "Zhipu_AI",
  "char-rnn": "The_Unreasonable_Effectiveness_of_Recurrent_Neural_Networks",
  "comma-v0.1-1t": "Comma_(models)",
  "comma-v0.1-2t": "Comma_(models)",
  "gpt-oss": "gpt-oss",
  "gpt-oss-120b": "gpt-oss",
  "gpt-oss-20b": "gpt-oss",
  "iFlytek": "iFlytek",
  "iFlytek Spark": "iFlytek_Spark",
  "iFlytek_Spark": "iFlytek_Spark",
  "o1": "o1",
  "o3": "o3",
  "o4-mini": "o4-mini",
  "tf.keras": "Keras",
  "tiktoken": "Tokenization",
  "vLLM": "PagedAttention",
  "xAI": "xAI"
 },
 "redirects": {}
}
