{"url":"/method/mlstm","slug":"mlstm","name":"mLSTM","full_name":"Multiplicative LSTM","full_name_withheld":false,"description_markdown":"A **Multiplicative LSTM (mLSTM)** is a  recurrent neural network architecture for sequence modelling that combines the long short-term memory ([LSTM](https://paperswithcode.com/method/lstm)) and multiplicative recurrent neural network ([mRNN](https://paperswithcode.com/method/mrnn)) architectures. The mRNN and LSTM architectures can be combined by adding connections from the mRNN’s intermediate state $m\\_{t}$ to each gating units in the LSTM.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Multiplicative LSTM for sequence modelling","paper":"/paper/multiplicative-lstm-for-sequence-modelling","first_author":"Ben Krause","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/multiplicative-lstm-for-sequence-modelling"},"source":{"url":"http://arxiv.org/abs/1609.07959v3","title":"Multiplicative LSTM for sequence modelling","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Sequential","area_id":"sequential","collection":"Recurrent Neural Networks","url":"/methods/category/recurrent-neural-networks","pwc_aliases":[]}],"n_papers_tagged":7,"archive_num_papers":7,"papers_newest_first":[{"paper":"/paper/nr4der-neural-re-ranking-for-diversified","title":"NR4DER: Neural Re-ranking for Diversified Exercise Recommendation","date":"2025-06-01","arxiv_id":"2506.06341","n_code_links":1,"syntology":null},{"paper":null,"title":"MoxE: Mixture of xLSTM Experts with Entropy-Aware Routing for Efficient Language Modeling","date":"2025-05-01","arxiv_id":"2505.01459","n_code_links":0,"syntology":null},{"paper":"/paper/tiled-flash-linear-attention-more-efficient","title":"Tiled Flash Linear Attention: More Efficient Linear RNN and xLSTM Kernels","date":"2025-03-18","arxiv_id":"2503.14376","n_code_links":1,"syntology":null},{"paper":"/paper/deltaproduct-increasing-the-expressivity-of","title":"DeltaProduct: Improving State-Tracking in Linear RNNs via Householder Products","date":"2025-02-14","arxiv_id":"2502.10297","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":0}},{"paper":"/paper/unlocking-state-tracking-in-linear-rnns","title":"Unlocking State-Tracking in Linear RNNs Through Negative Eigenvalues","date":"2024-11-19","arxiv_id":"2411.12537","n_code_links":2,"syntology":{"ran":1,"of":8,"unverified":7,"pointer_only":8}},{"paper":"/paper/xlstm-extended-long-short-term-memory","title":"xLSTM: Extended Long Short-Term Memory","date":"2024-05-07","arxiv_id":"2405.04517","n_code_links":5,"syntology":{"ran":0,"of":9,"unverified":9,"pointer_only":0}},{"paper":"/paper/multiplicative-lstm-for-sequence-modelling","title":"Multiplicative LSTM for sequence modelling","date":"2016-09-26","arxiv_id":"1609.07959","n_code_links":1,"syntology":null}],"papers_shown":7,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":6},{"task":"/task/language-modeling","name":"Language Modeling","papers":5},{"task":"/task/mamba","name":"Mamba","papers":3},{"task":"/task/density-estimation","name":"Density Estimation","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/math","name":"Math","papers":1},{"task":"/task/mixture-of-experts","name":"Mixture-of-Experts","papers":1},{"task":"/task/re-ranking","name":"Re-Ranking","papers":1},{"task":"/task/state-space-models","name":"State Space Models","papers":1}],"tasks_shown":9,"n_tasks":9,"usage_by_year":[{"year":"2016","papers":1},{"year":"2024","papers":2},{"year":"2025","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mlstm"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}