{"url":"/method/content-based-attention","slug":"content-based-attention","name":"Content-based Attention","full_name":"Content-based Attention","full_name_withheld":false,"description_markdown":"**Content-based attention** is an attention mechanism based on cosine similarity:\r\n\r\n$$f_{att}\\left(\\textbf{h}_{i}, \\textbf{s}\\_{j}\\right) = \\cos\\left[\\textbf{h}\\_{i};\\textbf{s}\\_{j}\\right] $$\r\n\r\nIt was utilised in [Neural Turing Machines](https://paperswithcode.com/method/neural-turing-machine) as part of the Addressing Mechanism.\r\n\r\nWe produce a normalized attention weighting by taking a [softmax](https://paperswithcode.com/method/softmax) over these attention alignment scores.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Neural Turing Machines","paper":"/paper/neural-turing-machines","first_author":"Alex Graves","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/neural-turing-machines"},"source":{"url":"http://arxiv.org/abs/1410.5401v2","title":"Neural Turing Machines","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/loudinthecloud/pytorch-ntm/blob/3c64937cb075e737528ff7ecf02b2c1c6d8347a1/ntm/memory.py#L82","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Attention Mechanisms","url":"/methods/category/attention-mechanisms","pwc_aliases":["attention-mechanisms-1"]}],"n_papers_tagged":33,"archive_num_papers":33,"papers_newest_first":[{"paper":null,"title":"Intelligent DoS and DDoS Detection: A Hybrid GRU-NTM Approach to Network Security","date":"2025-04-10","arxiv_id":"2504.07478","n_code_links":0,"syntology":null},{"paper":"/paper/memory-augmented-conformer-for-improved-end","title":"Memory-augmented conformer for improved end-to-end long-form ASR","date":"2023-09-22","arxiv_id":"2309.13029","n_code_links":1,"syntology":null},{"paper":null,"title":"FashionNTM: Multi-turn Fashion Image Retrieval via Cascaded Memory","date":"2023-08-20","arxiv_id":"2308.10170","n_code_links":0,"syntology":null},{"paper":"/paper/tps-attention-enhanced-thin-plate-spline-for","title":"TPS++: Attention-Enhanced Thin-Plate Spline for Scene Text Recognition","date":"2023-05-09","arxiv_id":"2305.05322","n_code_links":1,"syntology":null},{"paper":"/paper/token-turing-machines","title":"Token Turing Machines","date":"2022-11-16","arxiv_id":"2211.09119","n_code_links":1,"syntology":null},{"paper":null,"title":"Similarity and Content-based Phonetic Self Attention for Speech Recognition","date":"2022-03-19","arxiv_id":"2203.10252","n_code_links":0,"syntology":null},{"paper":null,"title":"HiMA: A Fast and Scalable History-based Memory Access Engine for Differentiable Neural Computer","date":"2022-02-15","arxiv_id":"2202.07275","n_code_links":0,"syntology":null},{"paper":"/paper/attentionhtr-handwritten-text-recognition","title":"AttentionHTR: Handwritten Text Recognition Based on Attention Encoder-Decoder Networks","date":"2022-01-23","arxiv_id":"2201.09390","n_code_links":1,"syntology":null},{"paper":"/paper/hard-attention-for-scalable-image","title":"Hard-Attention for Scalable Image Classification","date":"2021-02-20","arxiv_id":"2102.10212","n_code_links":1,"syntology":null},{"paper":null,"title":"Robust High-dimensional Memory-augmented Neural Networks","date":"2020-10-05","arxiv_id":"2010.01939","n_code_links":0,"syntology":null},{"paper":"/paper/explainable-inference-on-sequential-data-via","title":"Explainable Inference on Sequential Data via Memory-Tracking","date":"2020-07-11","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Unsupervised Speaker Adaptation using Attention-based Speaker Memory for End-to-End ASR","date":"2020-02-14","arxiv_id":"2002.06165","n_code_links":0,"syntology":null},{"paper":null,"title":"Sequence-to-sequence Singing Synthesis Using the Feed-forward Transformer","date":"2019-10-22","arxiv_id":"1910.09989","n_code_links":0,"syntology":null},{"paper":null,"title":"Memory-Augmented Recurrent Networks for Dialogue Coherence","date":"2019-10-16","arxiv_id":"1910.10487","n_code_links":0,"syntology":null},{"paper":null,"title":"A Neural Turing~Machine for Conditional Transition Graph Modeling","date":"2019-07-15","arxiv_id":"1907.06432","n_code_links":0,"syntology":null},{"paper":null,"title":"Understanding Memory Modules on Learning Simple Algorithms","date":"2019-07-01","arxiv_id":"1907.00820","n_code_links":0,"syntology":null},{"paper":null,"title":"A review on Neural Turing Machine","date":"2019-04-10","arxiv_id":"1904.05061","n_code_links":0,"syntology":null},{"paper":"/paper/few-shot-generalization-across-dialogue-tasks","title":"Few-Shot Generalization Across Dialogue Tasks","date":"2018-11-28","arxiv_id":"1811.11707","n_code_links":2,"syntology":null},{"paper":null,"title":"Context-Aware Neural Model for Temporal Information Extraction","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"A Taxonomy for Neural Memory Networks","date":"2018-05-01","arxiv_id":"1805.00327","n_code_links":0,"syntology":null},{"paper":null,"title":"The Set Autoencoder: Unsupervised Representation Learning for Sets","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Meta-Learning via Feature-Label Memory Network","date":"2017-10-19","arxiv_id":"1710.07110","n_code_links":0,"syntology":null},{"paper":null,"title":"Efficient Attention using a Fixed-Size Memory Representation","date":"2017-07-01","arxiv_id":"1707.00110","n_code_links":0,"syntology":null},{"paper":null,"title":"Attention-Set based Metric Learning for Video Face Recognition","date":"2017-04-12","arxiv_id":"1704.03805","n_code_links":0,"syntology":null},{"paper":"/paper/tracking-the-world-state-with-recurrent","title":"Tracking the World State with Recurrent Entity Networks","date":"2016-12-12","arxiv_id":"1612.03969","n_code_links":5,"syntology":{"ran":0,"of":3,"unverified":3,"pointer_only":0}},{"paper":null,"title":"Neural Turing Machines: Convergence of Copy Tasks","date":"2016-12-07","arxiv_id":"1612.02336","n_code_links":0,"syntology":null},{"paper":null,"title":"A Cheap Linear Attention Mechanism with Fast Lookups and Fixed-Size Representations","date":"2016-09-19","arxiv_id":"1609.05866","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-neural-turing-machine-with-soft-and","title":"Dynamic Neural Turing Machine with Soft and Hard Addressing Schemes","date":"2016-06-30","arxiv_id":"1607.00036","n_code_links":0,"syntology":null},{"paper":null,"title":"Lie Access Neural Turing Machine","date":"2016-02-28","arxiv_id":"1602.08671","n_code_links":0,"syntology":null},{"paper":null,"title":"Empirical Study on Deep Learning Models for Question Answering","date":"2015-10-26","arxiv_id":"1510.07526","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/decoder","name":"Decoder","papers":5},{"task":"/task/question-answering","name":"Question Answering","papers":5},{"task":"/task/machine-translation","name":"Machine Translation","papers":3},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":3},{"task":"/task/translation","name":"Translation","papers":3},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":3},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":2},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":2},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":2},{"task":"/task/image-classification","name":"Image Classification","papers":2},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":2},{"task":"/task/retrieval","name":"Retrieval","papers":2},{"task":"/task/sentence","name":"Sentence","papers":2},{"task":"/task/image-classification","name":"image-classification","papers":2},{"task":"/task/action-detection","name":"Action Detection","papers":1},{"task":"/task/activity-detection","name":"Activity Detection","papers":1},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/cloze-test","name":"Cloze Test","papers":1},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":1},{"task":"/task/deep-attention","name":"Deep Attention","papers":1}],"tasks_shown":20,"n_tasks":51,"usage_by_year":[{"year":"2014","papers":1},{"year":"2015","papers":3},{"year":"2016","papers":5},{"year":"2017","papers":3},{"year":"2018","papers":4},{"year":"2019","papers":5},{"year":"2020","papers":3},{"year":"2021","papers":1},{"year":"2022","papers":4},{"year":"2023","papers":3},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/content-based-attention"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}