{"url":"/method/attention-sinks","slug":"attention-sinks","name":"Attention Sinks","full_name":"Attention Sinks","full_name_withheld":false,"description_markdown":null,"description_state":"placeholder","introduced_year":null,"introduced_by":{"title":"Efficient Streaming Language Models with Attention Sinks","paper":"/paper/efficient-streaming-language-models-with","first_author":"Guangxuan Xiao","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/efficient-streaming-language-models-with"},"source":{"url":"https://arxiv.org/abs/2309.17453v4","title":"Efficient Streaming Language Models with Attention Sinks","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Attention","url":"/methods/category/attention","pwc_aliases":[]}],"n_papers_tagged":13,"archive_num_papers":13,"papers_newest_first":[{"paper":null,"title":"MARS-Bench: A Multi-turn Athletic Real-world Scenario Benchmark for Dialogue Evaluation","date":"2025-05-27","arxiv_id":"2505.23810","n_code_links":0,"syntology":null},{"paper":null,"title":"Analysis of Attention in Video Diffusion Transformers","date":"2025-04-14","arxiv_id":"2504.10317","n_code_links":0,"syntology":null},{"paper":"/paper/why-do-llms-attend-to-the-first-token","title":"Why do LLMs attend to the first token?","date":"2025-04-03","arxiv_id":"2504.02732","n_code_links":1,"syntology":{"ran":0,"of":7,"unverified":7,"pointer_only":0}},{"paper":"/paper/interpreting-the-repeated-token-phenomenon-in","title":"Interpreting the Repeated Token Phenomenon in Large Language Models","date":"2025-03-11","arxiv_id":"2503.08908","n_code_links":1,"syntology":null},{"paper":null,"title":"Attention Sinks and Outlier Features: A 'Catch, Tag, and Release' Mechanism for Embeddings","date":"2025-02-02","arxiv_id":"2502.00919","n_code_links":0,"syntology":null},{"paper":null,"title":"Task-KV: Task-aware KV Cache Optimization via Semantic Differentiation of Attention Heads","date":"2025-01-25","arxiv_id":"2501.15113","n_code_links":0,"syntology":null},{"paper":null,"title":"Attention Entropy is a Key Factor: An Analysis of Parallel Context Encoding with Full-attention-based Pre-trained Language Models","date":"2024-12-21","arxiv_id":"2412.16545","n_code_links":0,"syntology":null},{"paper":null,"title":"Seeing Clearly by Layer Two: Enhancing Attention Heads to Alleviate Hallucination in LVLMs","date":"2024-11-15","arxiv_id":"2411.09968","n_code_links":0,"syntology":null},{"paper":"/paper/value-residual-learning-for-alleviating","title":"Value Residual Learning For Alleviating Attention Concentration In Transformers","date":"2024-10-23","arxiv_id":"2410.17897","n_code_links":1,"syntology":{"ran":3,"of":12,"unverified":9,"pointer_only":0}},{"paper":"/paper/when-attention-sink-emerges-in-language","title":"When Attention Sink Emerges in Language Models: An Empirical View","date":"2024-10-14","arxiv_id":"2410.10781","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}},{"paper":"/paper/does-roberta-perform-better-than-bert-in","title":"Does RoBERTa Perform Better than BERT in Continual Learning: An Attention Sink Perspective","date":"2024-10-08","arxiv_id":"2410.05648","n_code_links":1,"syntology":null},{"paper":"/paper/unveiling-and-harnessing-hidden-attention","title":"Unveiling and Harnessing Hidden Attention Sinks: Enhancing Large Language Models without Training through Attention Calibration","date":"2024-06-22","arxiv_id":"2406.15765","n_code_links":1,"syntology":{"ran":5,"of":13,"unverified":8,"pointer_only":13}},{"paper":"/paper/efficient-streaming-language-models-with","title":"Efficient Streaming Language Models with Attention Sinks","date":"2023-09-29","arxiv_id":"2309.17453","n_code_links":6,"syntology":{"ran":8,"of":11,"unverified":3,"pointer_only":2}}],"papers_shown":13,"tasks":[{"task":"/task/continual-learning","name":"Continual Learning","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/dialogue-evaluation","name":"Dialogue Evaluation","papers":1},{"task":"/task/diversity","name":"Diversity","papers":1},{"task":"/task/hallucination","name":"Hallucination","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/model-compression","name":"Model Compression","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/tag","name":"TAG","papers":1},{"task":"/task/video-editing","name":"Video Editing","papers":1}],"tasks_shown":11,"n_tasks":11,"usage_by_year":[{"year":"2023","papers":1},{"year":"2024","papers":6},{"year":"2025","papers":6}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/attention-sinks"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}