{"url":"/method/attention-dropout","slug":"attention-dropout","name":"Attention Dropout","full_name":"Attention Dropout","full_name_withheld":false,"description_markdown":"**Attention Dropout** is a type of [dropout](https://paperswithcode.com/method/dropout) used in attention-based architectures, where elements are randomly dropped out of the [softmax](https://paperswithcode.com/method/softmax) in the attention equation. For example, for scaled-dot product attention, we would drop elements from the first term:\r\n\r\n$$ {\\text{Attention}}(Q, K, V) = \\text{softmax}\\left(\\frac{QK^{T}}{\\sqrt{d_k}}\\right)V $$","description_state":"present","introduced_year":2018,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":"https://github.com/huggingface/transformers/blob/4dc65591b5c61d75c3ef3a2a883bf1433e08fc45/src/transformers/modeling_tf_bert.py#L271","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Regularization","url":"/methods/category/regularization","pwc_aliases":[]}],"n_papers_tagged":10892,"archive_num_papers":10892,"papers_newest_first":[{"paper":"/paper/making-language-model-a-hierarchical","title":"Making Language Model a Hierarchical Classifier and Generator","date":"2025-07-17","arxiv_id":"2507.12930","n_code_links":1,"syntology":null},{"paper":null,"title":"Generative Click-through Rate Prediction with Applications to Search Advertising","date":"2025-07-15","arxiv_id":"2507.11246","n_code_links":0,"syntology":null},{"paper":null,"title":"Chat-Ghosting: A Comparative Study of Methods for Auto-Completion in Dialog Systems","date":"2025-07-08","arxiv_id":"2507.05940","n_code_links":0,"syntology":null},{"paper":null,"title":"SARA: Selective and Adaptive Retrieval-augmented Generation with Context Compression","date":"2025-07-08","arxiv_id":"2507.05633","n_code_links":0,"syntology":null},{"paper":null,"title":"AI Generated Text Detection Using Instruction Fine-tuned Large Language and Transformer-Based Models","date":"2025-07-07","arxiv_id":"2507.05157","n_code_links":0,"syntology":null},{"paper":null,"title":"Behaviour Space Analysis of LLM-driven Meta-heuristic Discovery","date":"2025-07-04","arxiv_id":"2507.03605","n_code_links":0,"syntology":null},{"paper":null,"title":"CyberRAG: An agentic RAG cyber attack classification and reporting tool","date":"2025-07-03","arxiv_id":"2507.02424","n_code_links":0,"syntology":null},{"paper":null,"title":"Knowledge Protocol Engineering: A New Paradigm for AI in Domain-Specific Knowledge Work","date":"2025-07-03","arxiv_id":"2507.02760","n_code_links":0,"syntology":null},{"paper":"/paper/robustness-of-misinformation-classification","title":"Robustness of Misinformation Classification Systems to Adversarial Examples Through BeamAttack","date":"2025-06-30","arxiv_id":"2506.23661","n_code_links":1,"syntology":null},{"paper":null,"title":"Agent-to-Agent Theory of Mind: Testing Interlocutor Awareness among Large Language Models","date":"2025-06-28","arxiv_id":"2506.22957","n_code_links":0,"syntology":null},{"paper":null,"title":"Cat and Mouse -- Can Fake Text Generation Outpace Detector Systems?","date":"2025-06-26","arxiv_id":"2506.21274","n_code_links":0,"syntology":null},{"paper":"/paper/erarag-efficient-and-incremental-retrieval","title":"EraRAG: Efficient and Incremental Retrieval Augmented Generation for Growing Corpora","date":"2025-06-26","arxiv_id":"2506.20963","n_code_links":1,"syntology":null},{"paper":null,"title":"Large Language Models Acing Chartered Accountancy","date":"2025-06-26","arxiv_id":"2506.21031","n_code_links":0,"syntology":null},{"paper":null,"title":"Leveraging LLM-Assisted Query Understanding for Live Retrieval-Augmented Generation","date":"2025-06-26","arxiv_id":"2506.21384","n_code_links":0,"syntology":null},{"paper":"/paper/psylite-technical-report","title":"PsyLite Technical Report","date":"2025-06-26","arxiv_id":"2506.21536","n_code_links":1,"syntology":null},{"paper":"/paper/response-quality-assessment-for-retrieval","title":"Response Quality Assessment for Retrieval-Augmented Generation via Conditional Conformal Factuality","date":"2025-06-26","arxiv_id":"2506.20978","n_code_links":1,"syntology":{"ran":0,"of":3,"unverified":3,"pointer_only":3}},{"paper":null,"title":"AI Assistants to Enhance and Exploit the PETSc Knowledge Base","date":"2025-06-25","arxiv_id":"2506.20608","n_code_links":0,"syntology":null},{"paper":null,"title":"CCRS: A Zero-Shot LLM-as-a-Judge Framework for Comprehensive RAG Evaluation","date":"2025-06-25","arxiv_id":"2506.20128","n_code_links":0,"syntology":null},{"paper":null,"title":"Engineering RAG Systems for Real-World Applications: Design, Development, and Evaluation","date":"2025-06-25","arxiv_id":"2506.20869","n_code_links":0,"syntology":null},{"paper":null,"title":"Knowledge-Aware Diverse Reranking for Cross-Source Question Answering","date":"2025-06-25","arxiv_id":"2506.20476","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Language Model-Driven Code Compliance Checking in Building Information Modeling","date":"2025-06-25","arxiv_id":"2506.20551","n_code_links":0,"syntology":null},{"paper":null,"title":"Memento: Note-Taking for Your Future Self","date":"2025-06-25","arxiv_id":"2506.20642","n_code_links":0,"syntology":null},{"paper":null,"title":"Accurate and Energy Efficient: Local Retrieval-Augmented Generation Models Outperform Commercial Large Language Models in Medical Tasks","date":"2025-06-24","arxiv_id":"2506.20009","n_code_links":0,"syntology":null},{"paper":null,"title":"Controlled Retrieval-augmented Context Evaluation for Long-form RAG","date":"2025-06-24","arxiv_id":"2506.20051","n_code_links":0,"syntology":null},{"paper":null,"title":"Inference Scaled GraphRAG: Improving Multi Hop Question Answering on Knowledge Graphs","date":"2025-06-24","arxiv_id":"2506.19967","n_code_links":0,"syntology":null},{"paper":null,"title":"KunLunBaizeRAG: Reinforcement Learning Driven Inference Performance Leap for Large Language Models","date":"2025-06-24","arxiv_id":"2506.19466","n_code_links":0,"syntology":null},{"paper":null,"title":"Unlocking Insights Addressing Alcohol Inference Mismatch through Database-Narrative Alignment","date":"2025-06-24","arxiv_id":"2506.19342","n_code_links":0,"syntology":null},{"paper":null,"title":"An Audio-centric Multi-task Learning Framework for Streaming Ads Targeting on Spotify","date":"2025-06-23","arxiv_id":"2506.18735","n_code_links":0,"syntology":null},{"paper":null,"title":"Semantic similarity estimation for domain specific data using BERT and other techniques","date":"2025-06-23","arxiv_id":"2506.18602","n_code_links":0,"syntology":null},{"paper":null,"title":"T-CPDL: A Temporal Causal Probabilistic Description Logic for Developing Logic-RAG Agent","date":"2025-06-23","arxiv_id":"2506.18559","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":1902},{"task":"/task/language-modeling","name":"Language Modeling","papers":1489},{"task":"/task/retrieval","name":"Retrieval","papers":1433},{"task":"/task/rag","name":"RAG","papers":1308},{"task":"/task/retrieval-augmented-generation","name":"Retrieval-augmented Generation","papers":1122},{"task":"/task/question-answering","name":"Question Answering","papers":1070},{"task":"/task/sentence","name":"Sentence","papers":920},{"task":"/task/large-language-model","name":"Large Language Model","papers":501},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":470},{"task":"/task/text-generation","name":"Text Generation","papers":470},{"task":"/task/text-classification","name":"Text Classification","papers":459},{"task":"/task/text-classification-1","name":"text-classification","papers":409},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":393},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":311},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":302},{"task":"/task/classification-1","name":"Classification","papers":299},{"task":"/task/decoder","name":"Decoder","papers":295},{"task":"/task/translation","name":"Translation","papers":294},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":280},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":273}],"tasks_shown":20,"n_tasks":1501,"usage_by_year":[{"year":"2015","papers":1},{"year":"2018","papers":10},{"year":"2019","papers":609},{"year":"2020","papers":1367},{"year":"2021","papers":1623},{"year":"2022","papers":1298},{"year":"2023","papers":2079},{"year":"2024","papers":2727},{"year":"2025","papers":1178}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/attention-dropout"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}