{"url":"/method/vq-vae-2","slug":"vq-vae-2","name":"VQ-VAE-2","full_name":"VQ-VAE-2","full_name_withheld":false,"description_markdown":"**VQ-VAE-2** is a type of variational autoencoder that combines a a two-level hierarchical VQ-[VAE](https://paperswithcode.com/method/vae) with a self-attention autoregressive model ([PixelCNN](https://paperswithcode.com/method/pixelcnn)) as a prior. The encoder and decoder architectures are kept simple and light-weight as in the original [VQ-VAE](https://paperswithcode.com/method/vq-vae), with the only difference that hierarchical multi-scale latent maps are used for increased resolution.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Generating Diverse High-Fidelity Images with VQ-VAE-2","paper":"/paper/190600446","first_author":"Ali Razavi","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/190600446"},"source":{"url":"https://arxiv.org/abs/1906.00446v1","title":"Generating Diverse High-Fidelity Images with VQ-VAE-2","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"3D Face Mesh Models","url":"/methods/category/3d-face-mesh-models","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Likelihood-Based Generative Models","url":"/methods/category/likelihood-based-generative-models","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Generative Models","url":"/methods/category/generative-models","pwc_aliases":[]}],"n_papers_tagged":7,"archive_num_papers":7,"papers_newest_first":[{"paper":null,"title":"HiTVideo: Hierarchical Tokenizers for Enhancing Text-to-Video Generation with Autoregressive Large Language Models","date":"2025-03-14","arxiv_id":"2503.11513","n_code_links":0,"syntology":null},{"paper":null,"title":"HQ-VAE: Hierarchical Discrete Representation Learning with Variational Bayes","date":"2023-12-31","arxiv_id":"2401.00365","n_code_links":0,"syntology":null},{"paper":null,"title":"SeaDSC: A video-based unsupervised method for dynamic scene change detection in unmanned surface vehicles","date":"2023-11-20","arxiv_id":"2311.11580","n_code_links":0,"syntology":null},{"paper":null,"title":"Phased Data Augmentation for Training a Likelihood-Based Generative Model with Limited Data","date":"2023-05-22","arxiv_id":"2305.12681","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-residual-learning-based-vector","title":"Hierarchical Residual Learning Based Vector Quantized Variational Autoencoder for Image Reconstruction and Generation","date":"2022-08-09","arxiv_id":"2208.04554","n_code_links":1,"syntology":null},{"paper":"/paper/an-unsupervised-video-game-playstyle-metric","title":"An Unsupervised Video Game Playstyle Metric via State Discretization","date":"2021-10-03","arxiv_id":"2110.00950","n_code_links":1,"syntology":null},{"paper":"/paper/190600446","title":"Generating Diverse High-Fidelity Images with VQ-VAE-2","date":"2019-06-02","arxiv_id":"1906.00446","n_code_links":15,"syntology":{"ran":2,"of":9,"unverified":7,"pointer_only":4}}],"papers_shown":7,"tasks":[{"task":"/task/decoder","name":"Decoder","papers":2},{"task":"/task/image-generation","name":"Image Generation","papers":2},{"task":"/task/atari-games","name":"Atari Games","papers":1},{"task":"/task/car-racing","name":"Car Racing","papers":1},{"task":"/task/change-detection","name":"Change Detection","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/diversity","name":"Diversity","papers":1},{"task":"/task/image-reconstruction","name":"Image Reconstruction","papers":1},{"task":"/task/motion-planning","name":"Motion Planning","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/object-tracking","name":"Object Tracking","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/scene-change-detection","name":"Scene Change Detection","papers":1},{"task":"/task/scene-understanding","name":"Scene Understanding","papers":1},{"task":"/task/text-to-video-generation","name":"Text-to-Video Generation","papers":1},{"task":"/task/video-generation","name":"Video Generation","papers":1},{"task":"/task/high","name":"Vocal Bursts Intensity Prediction","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":20,"n_tasks":20,"usage_by_year":[{"year":"2019","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2023","papers":3},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/vq-vae-2"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}