{"url":"/method/multi-query-attention","slug":"multi-query-attention","name":"Multi-Query Attention","full_name":"Multi-Query Attention","full_name_withheld":false,"description_markdown":"Multi-head attention consists of multiple attention layers (heads) in parallel with different linear\r\ntransformations on the queries, keys, values and outputs. **Multi-query attention** is identical except that the\r\ndifferent heads share a single set of keys and values.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/1911.02150v1","title":"Fast Transformer Decoding: One Write-Head is All You Need","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Attention","url":"/methods/category/attention","pwc_aliases":[]}],"n_papers_tagged":13,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"The Nature of Mathematical Modeling and Probabilistic Optimization Engineering in Generative AI","date":"2024-10-24","arxiv_id":"2410.18441","n_code_links":0,"syntology":null},{"paper":null,"title":"Weighted Grouped Query Attention in Transformers","date":"2024-07-15","arxiv_id":"2407.10855","n_code_links":0,"syntology":null},{"paper":"/paper/mlkv-multi-layer-key-value-heads-for-memory","title":"MLKV: Multi-Layer Key-Value Heads for Memory Efficient Transformer Decoding","date":"2024-06-13","arxiv_id":"2406.09297","n_code_links":1,"syntology":null},{"paper":null,"title":"Effectively Compress KV Heads for LLM","date":"2024-06-11","arxiv_id":"2406.07056","n_code_links":0,"syntology":null},{"paper":null,"title":"QCQA: Quality and Capacity-aware grouped Query Attention","date":"2024-06-08","arxiv_id":"2406.10247","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-transformer-key-value-cache-size","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","date":"2024-05-21","arxiv_id":"2405.12981","n_code_links":2,"syntology":null},{"paper":null,"title":"Bifurcated Attention: Accelerating Massively Parallel Decoding with Shared Prefixes in LLMs","date":"2024-03-13","arxiv_id":"2403.08845","n_code_links":0,"syntology":null},{"paper":"/paper/griffin-mixing-gated-linear-recurrences-with","title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","date":"2024-02-29","arxiv_id":"2402.19427","n_code_links":4,"syntology":{"ran":0,"of":10,"unverified":10,"pointer_only":0}},{"paper":"/paper/gqa-training-generalized-multi-query","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","date":"2023-05-22","arxiv_id":"2305.13245","n_code_links":4,"syntology":{"ran":4,"of":5,"unverified":1,"pointer_only":3}},{"paper":"/paper/flashattention-fast-and-memory-efficient","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","date":"2022-05-27","arxiv_id":"2205.14135","n_code_links":13,"syntology":{"ran":9,"of":30,"unverified":21,"pointer_only":1}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","n_code_links":7,"syntology":{"ran":30,"of":37,"unverified":7,"pointer_only":0}},{"paper":"/paper/fast-transformer-decoding-one-write-head-is","title":"Fast Transformer Decoding: One Write-Head is All You Need","date":"2019-11-06","arxiv_id":"1911.02150","n_code_links":4,"syntology":{"ran":0,"of":3,"unverified":3,"pointer_only":0}},{"paper":"/paper/mastering-chess-and-shogi-by-self-play-with-a","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","date":"2017-12-05","arxiv_id":"1712.01815","n_code_links":62,"syntology":{"ran":13,"of":17,"unverified":4,"pointer_only":9}}],"papers_shown":13,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":6},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/decoder","name":"Decoder","papers":2},{"task":"/task/16k","name":"16k","papers":1},{"task":"/task/4k","name":"4k","papers":1},{"task":null,"name":"8k","papers":1},{"task":"/task/all","name":"All","papers":1},{"task":"/task/answer-generation","name":"Answer Generation","papers":1},{"task":"/task/auto-debugging","name":"Auto Debugging","papers":1},{"task":"/task/code-generation","name":"Code Generation","papers":1},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":1},{"task":"/task/coreference-resolution","name":"Coreference Resolution","papers":1},{"task":"/task/cross-lingual-question-answering","name":"Cross-Lingual Question Answering","papers":1},{"task":"/task/document-classification","name":"Document Classification","papers":1},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/game-of-chess","name":"Game of Chess","papers":1},{"task":"/task/game-of-go","name":"Game of Go","papers":1},{"task":"/task/game-of-shogi","name":"Game of Shogi","papers":1},{"task":"/task/general-reinforcement-learning","name":"General Reinforcement Learning","papers":1}],"tasks_shown":20,"n_tasks":44,"usage_by_year":[{"year":"2017","papers":1},{"year":"2019","papers":1},{"year":"2022","papers":2},{"year":"2023","papers":1},{"year":"2024","papers":8}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/multi-query-attention"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}