{"url":"/method/bigbird","slug":"bigbird","name":"BigBird","full_name":"BigBird","full_name_withheld":false,"description_markdown":"**BigBird** is a [Transformer](https://paperswithcode.com/method/transformer) with a sparse attention mechanism that reduces the quadratic dependency of self-attention to linear in the number of tokens. BigBird is a universal approximator of sequence functions and is Turing complete, thereby preserving these properties of the quadratic, full attention model.  In particular, BigBird consists of three main parts:\r\n\r\n- A set of $g$ global tokens attending on all parts of the sequence.\r\n- All tokens attending to a set of $w$ local neighboring tokens.\r\n- All tokens attending to a set of $r$ random tokens.\r\n\r\nThis leads to a high performing attention mechanism scaling to much longer sequence lengths (8x).","description_state":"present","introduced_year":null,"introduced_by":{"title":"Big Bird: Transformers for Longer Sequences","paper":"/paper/big-bird-transformers-for-longer-sequences","first_author":"Manzil Zaheer","n_authors":11,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/big-bird-transformers-for-longer-sequences"},"source":{"url":"https://arxiv.org/abs/2007.14062v2","title":"Big Bird: Transformers for Longer Sequences","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Attention Patterns","url":"/methods/category/attention-patterns","pwc_aliases":["factorized-attention"]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":16,"archive_num_papers":16,"papers_newest_first":[{"paper":null,"title":"Convolutional vs Large Language Models for Software Log Classification in Edge-Deployable Cellular Network Testing","date":"2024-07-04","arxiv_id":"2407.03759","n_code_links":0,"syntology":null},{"paper":null,"title":"Transfer Learning in Pre-Trained Large Language Models for Malware Detection Based on System Calls","date":"2024-05-15","arxiv_id":"2405.09318","n_code_links":0,"syntology":null},{"paper":"/paper/multi-level-contrastive-learning-for-script","title":"Multi-level Contrastive Learning for Script-based Character Understanding","date":"2023-10-20","arxiv_id":"2310.13231","n_code_links":1,"syntology":null},{"paper":null,"title":"KoBigBird-large: Transformation of Transformer for Korean Language Understanding","date":"2023-09-19","arxiv_id":"2309.10339","n_code_links":0,"syntology":null},{"paper":null,"title":"BudgetLongformer: Can we Cheaply Pretrain a SotA Legal Language Model From Scratch?","date":"2022-11-30","arxiv_id":"2211.17135","n_code_links":0,"syntology":null},{"paper":null,"title":"Processing Long Legal Documents with Pre-trained Transformers: Modding LegalBERT and Longformer","date":"2022-11-02","arxiv_id":"2211.00974","n_code_links":0,"syntology":null},{"paper":null,"title":"LittleBird: Efficient Faster & Longer Transformer for Question Answering","date":"2022-10-21","arxiv_id":"2210.11870","n_code_links":0,"syntology":null},{"paper":"/paper/factorizing-content-and-budget-decisions-in","title":"Factorizing Content and Budget Decisions in Abstractive Summarization of Long Documents","date":"2022-05-25","arxiv_id":"2205.12486","n_code_links":1,"syntology":null},{"paper":null,"title":"ICDBigBird: A Contextual Embedding Model for ICD Code Classification","date":"2022-04-21","arxiv_id":"2204.10408","n_code_links":0,"syntology":null},{"paper":"/paper/clinical-longformer-and-clinical-bigbird","title":"Clinical-Longformer and Clinical-BigBird: Transformers for long clinical sequences","date":"2022-01-27","arxiv_id":"2201.11838","n_code_links":1,"syntology":null},{"paper":null,"title":"Hierarchical Neural Network Approaches for Long Document Classification","date":"2022-01-18","arxiv_id":"2201.06774","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-token-normalization-improves-vision-1","title":"Dynamic Token Normalization Improves Vision Transformers","date":"2021-12-05","arxiv_id":"2112.02624","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":1}},{"paper":null,"title":"Hierarchical Transformer Networks for Long-sequence and Multiple Clinical Documents Classification","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-dataset-for-answering-time-sensitive","title":"A Dataset for Answering Time-Sensitive Questions","date":"2021-08-13","arxiv_id":"2108.06314","n_code_links":1,"syntology":{"ran":4,"of":4,"unverified":0,"pointer_only":0}},{"paper":"/paper/hierarchical-transformer-networks-for","title":"Three-level Hierarchical Transformer Networks for Long-sequence and Multiple Clinical Documents Classification","date":"2021-04-17","arxiv_id":"2104.08444","n_code_links":1,"syntology":null},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","arxiv_id":"2007.14062","n_code_links":14,"syntology":{"ran":10,"of":15,"unverified":5,"pointer_only":11}}],"papers_shown":16,"tasks":[{"task":"/task/document-classification","name":"Document Classification","papers":5},{"task":"/task/question-answering","name":"Question Answering","papers":4},{"task":"/task/sentence","name":"Sentence","papers":3},{"task":"/task/classification-1","name":"Classification","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":2},{"task":"/task/text-classification","name":"Text Classification","papers":2},{"task":"/task/text-summarization","name":"Text Summarization","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/clinical-knowledge","name":"Clinical Knowledge","papers":1},{"task":"/task/code-classification","name":"Code Classification","papers":1},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/disentanglement","name":"Disentanglement","papers":1},{"task":"/task/document-summarization","name":"Document Summarization","papers":1},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":1},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":1},{"task":"/task/classification","name":"General Classification","papers":1}],"tasks_shown":20,"n_tasks":35,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":4},{"year":"2022","papers":7},{"year":"2023","papers":2},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/bigbird"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}