{"url":"/method/longformer","slug":"longformer","name":"Longformer","full_name":"Longformer","full_name_withheld":false,"description_markdown":"**Longformer** is a modified [Transformer](https://paperswithcode.com/method/transformer) architecture. Traditional [Transformer-based models](https://paperswithcode.com/methods/category/transformers) are unable to process long sequences due to their self-attention operation, which scales quadratically with the sequence length. To address this, **Longformer** uses an attention pattern that scales linearly with sequence length, making it easy to process documents of thousands of tokens or longer. The attention mechanism is a drop-in replacement for the standard self-attention and combines a local windowed attention with a task motivated global attention.\r\n\r\nThe attention patterns utilised include: [sliding window attention](https://paperswithcode.com/method/sliding-window-attention), [dilated sliding window attention](https://paperswithcode.com/method/dilated-sliding-window-attention) and global + sliding window. These can be viewed in the components section of this page.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Longformer: The Long-Document Transformer","paper":"/paper/longformer-the-long-document-transformer","first_author":"Iz Beltagy","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/longformer-the-long-document-transformer"},"source":{"url":"https://arxiv.org/abs/2004.05150v2","title":"Longformer: The Long-Document Transformer","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoencoding Transformers","url":"/methods/category/autoencoding-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":87,"archive_num_papers":87,"papers_newest_first":[{"paper":null,"title":"I Know Which LLM Wrote Your Code Last Summer: LLM generated Code Stylometry for Authorship Attribution","date":"2025-06-18","arxiv_id":"2506.17323","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-abstractive-summarization-of","title":"Enhancing Abstractive Summarization of Scientific Papers Using Structure Information","date":"2025-05-20","arxiv_id":"2505.14179","n_code_links":1,"syntology":null},{"paper":null,"title":"CacheFormer: High Attention-Based Segment Caching","date":"2025-04-18","arxiv_id":"2504.13981","n_code_links":0,"syntology":null},{"paper":null,"title":"ARLED: Leveraging LED-based ARMAN Model for Abstractive Summarization of Persian Long Documents","date":"2025-03-13","arxiv_id":"2503.10233","n_code_links":0,"syntology":null},{"paper":null,"title":"Understanding Players as if They Are Talking to the Game in a Customized Language: A Pilot Study","date":"2024-10-24","arxiv_id":"2410.18605","n_code_links":0,"syntology":null},{"paper":null,"title":"Extra Global Attention Designation Using Keyword Detection in Sparse Transformer Architectures","date":"2024-10-11","arxiv_id":"2410.08971","n_code_links":0,"syntology":null},{"paper":"/paper/the-accuracy-paradox-in-rlhf-when-better","title":"The Accuracy Paradox in RLHF: When Better Reward Models Don't Yield Better Language Models","date":"2024-10-09","arxiv_id":"2410.06554","n_code_links":1,"syntology":{"ran":5,"of":9,"unverified":4,"pointer_only":0}},{"paper":"/paper/rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Transfer Learning in Pre-Trained Large Language Models for Malware Detection Based on System Calls","date":"2024-05-15","arxiv_id":"2405.09318","n_code_links":0,"syntology":null},{"paper":null,"title":"Advancing AI with Integrity: Ethical Challenges and Solutions in Neural Machine Translation","date":"2024-04-01","arxiv_id":"2404.01070","n_code_links":0,"syntology":null},{"paper":null,"title":"A multi-cohort study on prediction of acute brain dysfunction states using selective state space models","date":"2024-03-11","arxiv_id":"2403.07201","n_code_links":0,"syntology":null},{"paper":"/paper/adaptation-of-biomedical-and-clinical","title":"Adaptation of Biomedical and Clinical Pretrained Models to French Long Documents: A Comparative Study","date":"2024-02-26","arxiv_id":"2402.16689","n_code_links":1,"syntology":null},{"paper":"/paper/accurate-and-well-calibrated-icd-code","title":"Accurate and Well-Calibrated ICD Code Assignment Through Attention Over Diverse Label Embeddings","date":"2024-02-05","arxiv_id":"2402.03172","n_code_links":1,"syntology":{"ran":7,"of":7,"unverified":0,"pointer_only":0}},{"paper":"/paper/unimem-towards-a-unified-view-of-long-context","title":"UniMem: Towards a Unified View of Long-Context Large Language Models","date":"2024-02-05","arxiv_id":"2402.03009","n_code_links":1,"syntology":null},{"paper":null,"title":"Enhanced Labeling Technique for Reddit Text and Fine-Tuned Longformer Models for Classifying Depression Severity in English and Luganda","date":"2024-01-25","arxiv_id":"2401.14240","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-automatic-text-simplification-of","title":"Exploring Automatic Text Simplification of German Narrative Documents","date":"2023-12-15","arxiv_id":"2312.09907","n_code_links":1,"syntology":null},{"paper":"/paper/llvms4protest-harnessing-the-power-of-large","title":"LLVMs4Protest: Harnessing the Power of Large Language and Vision Models for Deciphering Protests in the News","date":"2023-11-30","arxiv_id":"2311.18241","n_code_links":1,"syntology":null},{"paper":null,"title":"Towards Harmful Erotic Content Detection through Coreference-Driven Contextual Analysis","date":"2023-10-22","arxiv_id":"2310.14325","n_code_links":0,"syntology":null},{"paper":"/paper/multi-level-contrastive-learning-for-script","title":"Multi-level Contrastive Learning for Script-based Character Understanding","date":"2023-10-20","arxiv_id":"2310.13231","n_code_links":1,"syntology":null},{"paper":"/paper/improving-long-document-topic-segmentation","title":"Improving Long Document Topic Segmentation Models With Enhanced Coherence Modeling","date":"2023-10-18","arxiv_id":"2310.11772","n_code_links":1,"syntology":null},{"paper":"/paper/hallucination-reduction-in-long-input-text","title":"Hallucination Reduction in Long Input Text Summarization","date":"2023-09-28","arxiv_id":"2309.16781","n_code_links":1,"syntology":null},{"paper":null,"title":"An NLP Benchmark Dataset for Assessing Corporate Climate Policy Engagement","date":"2023-09-26","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Language Models for Novelty Detection in System Call Traces","date":"2023-09-05","arxiv_id":"2309.02206","n_code_links":0,"syntology":null},{"paper":"/paper/local-large-language-models-for-complex","title":"Local Large Language Models for Complex Structured Medical Tasks","date":"2023-08-03","arxiv_id":"2308.01727","n_code_links":1,"syntology":null},{"paper":"/paper/can-model-fusing-help-transformers-in-long","title":"Can Model Fusing Help Transformers in Long Document Classification? An Empirical Study","date":"2023-07-18","arxiv_id":"2307.09532","n_code_links":1,"syntology":null},{"paper":null,"title":"NOWJ at COLIEE 2023 -- Multi-Task and Ensemble Approaches in Legal Information Processing","date":"2023-06-08","arxiv_id":"2306.04903","n_code_links":0,"syntology":null},{"paper":null,"title":"MultiLegalPile: A 689GB Multilingual Legal Corpus","date":"2023-06-03","arxiv_id":"2306.02069","n_code_links":0,"syntology":null},{"paper":"/paper/deplain-a-german-parallel-corpus-with","title":"DEPLAIN: A German Parallel Corpus with Intralingual Translations into Plain Language for Sentence and Document Simplification","date":"2023-05-30","arxiv_id":"2305.18939","n_code_links":1,"syntology":null},{"paper":null,"title":"Incorporating Distributions of Discourse Structure for Long Document Abstractive Summarization","date":"2023-05-26","arxiv_id":"2305.16784","n_code_links":0,"syntology":null},{"paper":null,"title":"Neural Summarization of Electronic Health Records","date":"2023-05-24","arxiv_id":"2305.15222","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":13},{"task":"/task/decoder","name":"Decoder","papers":11},{"task":"/task/language-modeling","name":"Language Modeling","papers":11},{"task":"/task/sentence","name":"Sentence","papers":11},{"task":"/task/document-classification","name":"Document Classification","papers":10},{"task":"/task/question-answering","name":"Question Answering","papers":9},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":8},{"task":"/task/articles","name":"Articles","papers":7},{"task":"/task/classification-1","name":"Classification","papers":6},{"task":"/task/text-classification","name":"Text Classification","papers":6},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":5},{"task":"/task/text-summarization","name":"Text Summarization","papers":5},{"task":"/task/text-classification-1","name":"text-classification","papers":5},{"task":"/task/retrieval","name":"Retrieval","papers":4},{"task":"/task/document-summarization","name":"Document Summarization","papers":3},{"task":null,"name":"GPU","papers":3},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":3},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":3},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":3},{"task":"/task/named-entity-recognition","name":"named-entity-recognition","papers":3}],"tasks_shown":20,"n_tasks":111,"usage_by_year":[{"year":"2020","papers":7},{"year":"2021","papers":20},{"year":"2022","papers":17},{"year":"2023","papers":28},{"year":"2024","papers":11},{"year":"2025","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/longformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}