{"url":"/method/rope","slug":"rope","name":"Rotary Embeddings","full_name":"Rotary Position Embedding","full_name_withheld":false,"description_markdown":"**Rotary Position Embedding**, or **RoPE**, is a type of position embedding which encodes absolute positional information with rotation matrix and naturally incorporates explicit relative position dependency in self-attention formulation. Notably, RoPE comes with valuable properties such as flexibility of being expand to any sequence lengths, decaying inter-token dependency with increasing relative distances, and capability of equipping the linear self-attention with relative position encoding.","description_state":"present","introduced_year":null,"introduced_by":{"title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","paper":"/paper/roformer-enhanced-transformer-with-rotary","first_author":"Jianlin Su","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/roformer-enhanced-transformer-with-rotary"},"source":{"url":"https://arxiv.org/abs/2104.09864v5","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Position Embeddings","url":"/methods/category/position-embeddings","pwc_aliases":[]}],"n_papers_tagged":8,"archive_num_papers":8,"papers_newest_first":[{"paper":"/paper/yourmt3-multi-instrument-music-transcription","title":"YourMT3+: Multi-instrument Music Transcription with Enhanced Transformer Architectures and Cross-dataset Stem Augmentation","date":"2024-07-05","arxiv_id":"2407.04822","n_code_links":1,"syntology":null},{"paper":"/paper/mitigate-position-bias-in-large-language","title":"Mitigate Position Bias in Large Language Models via Scaling a Single Dimension","date":"2024-06-04","arxiv_id":"2406.02536","n_code_links":1,"syntology":{"ran":16,"of":22,"unverified":6,"pointer_only":0}},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","n_code_links":19,"syntology":{"ran":31,"of":52,"unverified":21,"pointer_only":16}},{"paper":"/paper/flashattention-fast-and-memory-efficient","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","date":"2022-05-27","arxiv_id":"2205.14135","n_code_links":13,"syntology":{"ran":9,"of":30,"unverified":21,"pointer_only":1}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","n_code_links":7,"syntology":{"ran":30,"of":37,"unverified":7,"pointer_only":0}},{"paper":"/paper/hierarchical-transformers-are-more-efficient","title":"Hierarchical Transformers Are More Efficient Language Models","date":"2021-10-26","arxiv_id":"2110.13711","n_code_links":3,"syntology":{"ran":2,"of":4,"unverified":2,"pointer_only":2}},{"paper":null,"title":"Conformer-based End-to-end Speech Recognition With Rotary Position Embedding","date":"2021-07-13","arxiv_id":"2107.05907","n_code_links":0,"syntology":null},{"paper":"/paper/roformer-enhanced-transformer-with-rotary","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","date":"2021-04-20","arxiv_id":"2104.09864","n_code_links":20,"syntology":{"ran":14,"of":16,"unverified":2,"pointer_only":0}}],"papers_shown":8,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":null,"name":"Position","papers":3},{"task":"/task/code-generation","name":"Code Generation","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/multi-task-language-understanding","name":"Multi-task Language Understanding","papers":2},{"task":"/task/multiple-choice-qa","name":"Multiple Choice Question Answering (MCQA)","papers":2},{"task":"/task/question-answering","name":"Question Answering","papers":2},{"task":"/task/sentence-completion","name":"Sentence Completion","papers":2},{"task":"/task/16k","name":"16k","papers":1},{"task":"/task/4k","name":"4k","papers":1},{"task":"/task/arithmetic-reasoning","name":"Arithmetic Reasoning","papers":1},{"task":"/task/auto-debugging","name":"Auto Debugging","papers":1},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":1},{"task":"/task/coreference-resolution","name":"Coreference Resolution","papers":1},{"task":"/task/cross-lingual-question-answering","name":"Cross-Lingual Question Answering","papers":1},{"task":"/task/document-classification","name":"Document Classification","papers":1},{"task":"/task/drum-transcription","name":"Drum Transcription","papers":1},{"task":"/task/drum-transcription-in-music-dtm","name":"Drum Transcription in Music (DTM)","papers":1},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":1},{"task":null,"name":"GPU","papers":1}],"tasks_shown":20,"n_tasks":42,"usage_by_year":[{"year":"2021","papers":3},{"year":"2022","papers":2},{"year":"2023","papers":1},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/rope"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}