{"url":"/method/gumbel-softmax","slug":"gumbel-softmax","name":"Gumbel Softmax","full_name":"Gumbel Softmax","full_name_withheld":false,"description_markdown":"**Gumbel-Softmax** is a continuous distribution that has the property that it can be smoothly annealed into a categorical distribution, and whose parameter gradients can be easily computed via the reparameterization trick.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"http://arxiv.org/abs/1611.01144v5","title":"Categorical Reparameterization with Gumbel-Softmax","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/ericjang/gumbel-softmax/blob/master/gumbel_softmax_vae_v2.ipynb","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Distributions","url":"/methods/category/distributions","pwc_aliases":[]}],"n_papers_tagged":52,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"An LMM for Efficient Video Understanding via Reinforced Compression of Video Cubes","date":"2025-04-21","arxiv_id":"2504.15270","n_code_links":0,"syntology":null},{"paper":null,"title":"Learning Novel Transformer Architecture for Time-series Forecasting","date":"2025-02-19","arxiv_id":"2502.13721","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Text Generation in Joint NLG/NLU Learning Through Curriculum Learning, Semi-Supervised Training, and Advanced Optimization Techniques","date":"2024-10-17","arxiv_id":"2410.13498","n_code_links":0,"syntology":null},{"paper":null,"title":"Quantum-Enhanced Detection of Viral cDNA via Luminescence Resonance Energy Transfer Using Upconversion and Gold Nanoparticles","date":"2024-10-14","arxiv_id":"2410.10911","n_code_links":0,"syntology":null},{"paper":"/paper/maskllm-learnable-semi-structured-sparsity","title":"MaskLLM: Learnable Semi-Structured Sparsity for Large Language Models","date":"2024-09-26","arxiv_id":"2409.17481","n_code_links":1,"syntology":{"ran":5,"of":16,"unverified":11,"pointer_only":16}},{"paper":null,"title":"MicroMIL: Graph-based Contextual Multiple Instance Learning for Patient Diagnosis Using Microscopy Images","date":"2024-07-31","arxiv_id":"2407.21604","n_code_links":0,"syntology":null},{"paper":"/paper/meta-learning-an-evolvable-developmental","title":"Meta-Learning an Evolvable Developmental Encoding","date":"2024-06-13","arxiv_id":"2406.09020","n_code_links":1,"syntology":null},{"paper":null,"title":"Heterogeneous Learning Rate Scheduling for Neural Architecture Search on Long-Tailed Datasets","date":"2024-06-11","arxiv_id":"2406.07028","n_code_links":0,"syntology":null},{"paper":null,"title":"The devil is in discretization discrepancy. Robustifying Differentiable NAS with Single-Stage Searching Protocol","date":"2024-05-26","arxiv_id":"2405.16610","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-concept-bottleneck-models-gumbel","title":"Sparse Concept Bottleneck Models: Gumbel Tricks in Contrastive Learning","date":"2024-04-04","arxiv_id":"2404.03323","n_code_links":2,"syntology":null},{"paper":null,"title":"MicroNAS: Memory and Latency Constrained Hardware-Aware Neural Architecture Search for Time Series Classification on Microcontrollers","date":"2023-10-27","arxiv_id":"2310.18384","n_code_links":0,"syntology":null},{"paper":"/paper/minimizing-return-gaps-with-discrete","title":"RGMComm: Return Gap Minimization via Discrete Communications in Multi-Agent Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03358","n_code_links":1,"syntology":null},{"paper":null,"title":"Feature-aware conditional GAN for category text generation","date":"2023-08-02","arxiv_id":"2308.00939","n_code_links":0,"syntology":null},{"paper":"/paper/stable-diffusion-is-unstable-1","title":"Stable Diffusion is Unstable","date":"2023-06-05","arxiv_id":"2306.02583","n_code_links":1,"syntology":{"ran":6,"of":9,"unverified":3,"pointer_only":0}},{"paper":null,"title":"Visual DNA: Representing and Comparing Images using Distributions of Neuron Activations","date":"2023-04-20","arxiv_id":"2304.10036","n_code_links":0,"syntology":null},{"paper":null,"title":"Efficient Automation of Neural Network Design: A Survey on Differentiable Neural Architecture Search","date":"2023-04-11","arxiv_id":"2304.05405","n_code_links":0,"syntology":null},{"paper":"/paper/fully-differentiable-ransac","title":"Generalized Differentiable RANSAC","date":"2022-12-26","arxiv_id":"2212.13185","n_code_links":2,"syntology":null},{"paper":"/paper/adatriplet-ra-domain-matching-via-adaptive","title":"AdaTriplet-RA: Domain Matching via Adaptive Triplet and Reinforced Attention for Unsupervised Domain Adaptation","date":"2022-11-16","arxiv_id":"2211.08894","n_code_links":1,"syntology":null},{"paper":"/paper/regularized-graph-structure-learning-with","title":"Regularized Graph Structure Learning with Semantic Knowledge for Multi-variates Time-Series Forecasting","date":"2022-10-12","arxiv_id":"2210.06126","n_code_links":1,"syntology":{"ran":5,"of":6,"unverified":1,"pointer_only":6}},{"paper":"/paper/patch-based-knowledge-distillation-for","title":"Patch-based Knowledge Distillation for Lifelong Person Re-Identification","date":"2022-10-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Tiered Pruning for Efficient Differentialble Inference-Aware Neural Architecture Search","date":"2022-09-23","arxiv_id":"2209.11785","n_code_links":0,"syntology":null},{"paper":"/paper/multi-complexity-loss-dnas-for-energy","title":"Multi-Complexity-Loss DNAS for Energy-Efficient and Memory-Constrained Deep Neural Networks","date":"2022-06-01","arxiv_id":"2206.00302","n_code_links":1,"syntology":null},{"paper":null,"title":"Approaches to the classification of complex systems: Words, texts, and more","date":"2022-05-09","arxiv_id":"2205.04060","n_code_links":0,"syntology":null},{"paper":null,"title":"An Analysis of Discretization Methods for Communication Learning with Multi-Agent Reinforcement Learning","date":"2022-04-12","arxiv_id":"2204.05669","n_code_links":0,"syntology":null},{"paper":"/paper/extracting-effective-subnetworks-with-gumebel","title":"Extracting Effective Subnetworks with Gumbel-Softmax","date":"2022-02-25","arxiv_id":"2202.12986","n_code_links":1,"syntology":null},{"paper":null,"title":"UDC: Unified DNAS for Compressible TinyML Models","date":"2022-01-15","arxiv_id":"2201.05842","n_code_links":0,"syntology":null},{"paper":"/paper/eh-dnas-end-to-end-hardware-aware","title":"EH-DNAS: End-to-End Hardware-aware Differentiable Neural Architecture Search","date":"2021-11-24","arxiv_id":"2111.12299","n_code_links":1,"syntology":null},{"paper":"/paper/linear-or-non-linear-that-is-the-question","title":"Linear, or Non-Linear, That is the Question!","date":"2021-11-14","arxiv_id":"2111.07265","n_code_links":2,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/differentiable-nas-framework-and-application","title":"Differentiable NAS Framework and Application to Ads CTR Prediction","date":"2021-10-25","arxiv_id":"2110.14812","n_code_links":1,"syntology":null},{"paper":null,"title":"RADARS: Memory Efficient Reinforcement Learning Aided Differentiable Neural Architecture Search","date":"2021-09-13","arxiv_id":"2109.05691","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":16},{"task":"/task/clustering","name":"Clustering","papers":6},{"task":null,"name":"GPU","papers":4},{"task":"/task/classification","name":"General Classification","papers":4},{"task":"/task/image-classification","name":"Image Classification","papers":4},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":4},{"task":"/task/time-series-1","name":"Time Series","papers":4},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":4},{"task":"/task/quantization","name":"Quantization","papers":3},{"task":"/task/recommendation-systems","name":"Recommendation Systems","papers":3},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":3},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":3},{"task":"/task/image-classification","name":"image-classification","papers":3},{"task":"/task/benchmarking","name":"Benchmarking","papers":2},{"task":"/task/diversity","name":"Diversity","papers":2},{"task":"/task/evolutionary-algorithms","name":"Evolutionary Algorithms","papers":2},{"task":"/task/graph-clustering","name":"Graph Clustering","papers":2},{"task":"/task/meta-learning","name":"Meta-Learning","papers":2},{"task":"/task/model-compression","name":"Model Compression","papers":2},{"task":"/task/multi-agent-reinforcement-learning","name":"Multi-agent Reinforcement Learning","papers":2}],"tasks_shown":20,"n_tasks":87,"usage_by_year":[{"year":"2016","papers":1},{"year":"2018","papers":5},{"year":"2019","papers":3},{"year":"2020","papers":7},{"year":"2021","papers":10},{"year":"2022","papers":10},{"year":"2023","papers":6},{"year":"2024","papers":8},{"year":"2025","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gumbel-softmax"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}