{
 "slug": "DALL-E_2",
 "title": "DALL-E 2",
 "type": "model",
 "short_desc": "OpenAI's April 2022 text-to-image diffusion model, which generated 1024x1024 images from CLIP embeddings and popularized AI image generation.",
 "categories": [
  "OpenAI models",
  "Models",
  "Multimodal models",
  "2022 model releases"
 ],
 "infobox": {
  "Developer": "OpenAI",
  "Family": "DALL-E",
  "Announced": "April 2022",
  "Architecture": "Two-stage 'unCLIP' system: a prior mapping text to CLIP image embeddings, plus a diffusion decoder with upsamplers",
  "Parameters": "Approximately 3.5 billion (diffusion decoder; developer-reported)",
  "Post-training": "Safety filtering of training data; deployment-side prompt and output filters",
  "License / access": "Proprietary; research preview, then paid credits and API (November 2022); later retired in favor of DALL-E 3",
  "Paper/report": "arXiv:2204.06125 (\"Hierarchical Text-Conditional Image Generation with CLIP Latents\")"
 },
 "infobox_links": {
  "Developer": [
   "OpenAI"
  ],
  "Family": [
   "DALL-E"
  ],
  "Architecture": [
   "CLIP"
  ],
  "License / access": [
   "DALL-E_3"
  ],
  "Paper/report": [
   "CLIP"
  ]
 },
 "aliases": [],
 "family": "",
 "org": "OpenAI",
 "registry_status": "stub",
 "words": 378,
 "references": 3,
 "outbound": [
  "CLIP",
  "DALL-E",
  "DALL-E_3",
  "FLUX",
  "GPT-3",
  "GPT-4o",
  "Imagen",
  "OpenAI",
  "Sora",
  "Stability_AI",
  "Stable_Diffusion",
  "Transformer_(architecture)"
 ],
 "inbound": [
  "CLIP",
  "DALL-E",
  "DALL-E_3",
  "Diffusion_language_model",
  "Stable_Diffusion",
  "Vision_Transformer"
 ],
 "url": "wiki/DALL-E_2.html",
 "built": "2026-07-24"
}
