{
 "slug": "LLaVA",
 "title": "LLaVA",
 "type": "model",
 "short_desc": "The 2023 open vision-language recipe: a frozen vision encoder bridged to an open LLM with GPT-4-generated instruction data.",
 "categories": [
  "Models",
  "Open-weight models",
  "Vision-language models",
  "2023 model releases"
 ],
 "infobox": {
  "Developer": "University of Wisconsin-Madison, Microsoft Research (Liu et al.)",
  "Predecessor": "None (original open VLM recipe)",
  "Successor": "LLaVA-1.5, LLaVA-NeXT, and the open VLM wave",
  "Announced": "April 17, 2023",
  "Architecture": "CLIP ViT encoder + projection into an open LLM (Vicuna)",
  "Parameters": "7B and 13B variants",
  "Training data": "158K GPT-4-generated visual instruction samples",
  "License / access": "Open weights and data (research license at release)",
  "Paper/report": "arXiv:2304.08485"
 },
 "infobox_links": {
  "Developer": [
   "Microsoft_AI"
  ],
  "Architecture": [
   "Vision_Transformer",
   "Vicuna"
  ],
  "Training data": [
   "GPT-4"
  ]
 },
 "aliases": [],
 "family": "",
 "org": "",
 "registry_status": "stub",
 "words": 322,
 "references": 3,
 "outbound": [
  "Alpaca",
  "CLIP",
  "GPT-4",
  "Hugging_Face",
  "Instruction_tuning",
  "InternVL",
  "Knowledge_distillation",
  "Large_language_model",
  "MMLU",
  "Microsoft_AI",
  "Moondream_(model_family)",
  "Qwen_(model_family)",
  "Vicuna",
  "Vision_Transformer"
 ],
 "inbound": [
  "BLIP",
  "Large_language_model",
  "Vision_Transformer"
 ],
 "url": "wiki/LLaVA.html",
 "built": "2026-07-24"
}
