{"url":"/method/lama","slug":"lama","name":"LAMA","full_name":"Low-Rank Factorization-based Multi-Head Attention","full_name_withheld":false,"description_markdown":"**Low-Rank Factorization-based Multi-head Attention Mechanism**, or **LAMA**, is a type of attention module that uses low-rank factorization to reduce computational complexity. It uses low-rank bilinear pooling to construct a structured sentence representation that attends to multiple aspects of a sentence.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/1912.00835v2","title":"Low Rank Factorization for Compact Multi-Head Self-Attention","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/JohnGiorgi/compact-multi-head-self-attention-pytorch/blob/56358eaddde176d539916896d6836c00d1dc5f0c/modules/lama.py#L10","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Attention Modules","url":"/methods/category/attention-modules","pwc_aliases":[]}],"n_papers_tagged":42,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"HumanDreamer: Generating Controllable Human-Motion Videos via Decoupled Generation","date":"2025-03-31","arxiv_id":"2503.24026","n_code_links":0,"syntology":null},{"paper":null,"title":"SenseExpo: Efficient Autonomous Exploration with Prediction Information from Lightweight Neural Networks","date":"2025-03-20","arxiv_id":"2503.16000","n_code_links":0,"syntology":null},{"paper":null,"title":"One Mind, Many Tongues: A Deep Dive into Language-Agnostic Knowledge Neurons in Large Language Models","date":"2024-11-26","arxiv_id":"2411.17401","n_code_links":0,"syntology":null},{"paper":null,"title":"LAMA: Stable Dual-Domain Deep Reconstruction For Sparse-View CT","date":"2024-10-28","arxiv_id":"2410.21111","n_code_links":0,"syntology":null},{"paper":"/paper/discovering-leitmotifs-in-multidimensional","title":"Discovering Leitmotifs in Multidimensional Time Series","date":"2024-10-16","arxiv_id":"2410.12293","n_code_links":1,"syntology":null},{"paper":null,"title":"Consolidating LAMA with Best-First Width Search","date":"2024-04-26","arxiv_id":"2404.17648","n_code_links":0,"syntology":null},{"paper":"/paper/return-to-tradition-learning-reliable","title":"Return to Tradition: Learning Reliable Heuristics with Classical Machine Learning","date":"2024-03-25","arxiv_id":"2403.16508","n_code_links":1,"syntology":null},{"paper":"/paper/take-care-of-your-prompt-bias-investigating","title":"Take Care of Your Prompt Bias! Investigating and Mitigating Prompt Bias in Factual Knowledge Extraction","date":"2024-03-15","arxiv_id":"2403.09963","n_code_links":1,"syntology":null},{"paper":"/paper/the-cost-of-compression-investigating-the","title":"The Cost of Compression: Investigating the Impact of Compression on Parametric Knowledge in Language Models","date":"2023-12-01","arxiv_id":"2312.00960","n_code_links":1,"syntology":{"ran":1,"of":2,"unverified":1,"pointer_only":2}},{"paper":null,"title":"The minimal computational substrate of fluid intelligence","date":"2023-08-14","arxiv_id":"2308.07039","n_code_links":0,"syntology":null},{"paper":"/paper/dlama-a-framework-for-curating-culturally","title":"DLAMA: A Framework for Curating Culturally Diverse Facts for Probing the Knowledge of Pretrained Language Models","date":"2023-06-08","arxiv_id":"2306.05076","n_code_links":1,"syntology":null},{"paper":"/paper/learned-alternating-minimization-algorithm","title":"Learned Alternating Minimization Algorithm for Dual-domain Sparse-View CT Reconstruction","date":"2023-06-05","arxiv_id":"2306.02644","n_code_links":1,"syntology":null},{"paper":null,"title":"Locomotion-Action-Manipulation: Synthesizing Human-Scene Interactions in Complex 3D Environments","date":"2023-01-09","arxiv_id":"2301.02667","n_code_links":0,"syntology":null},{"paper":null,"title":"Rethinking Fast Fourier Convolution in Image Inpainting","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"In-context Learning Distillation: Transferring Few-shot Learning Ability of Pre-trained Language Models","date":"2022-12-20","arxiv_id":"2212.10670","n_code_links":0,"syntology":null},{"paper":null,"title":"Context Variance Evaluation of Pretrained Language Models for Prompt-based Biomedical Knowledge Probing","date":"2022-11-18","arxiv_id":"2211.10265","n_code_links":0,"syntology":null},{"paper":null,"title":"SPE: Symmetrical Prompt Enhancement for Fact Probing","date":"2022-11-14","arxiv_id":"2211.07078","n_code_links":0,"syntology":null},{"paper":"/paper/kamel-knowledge-analysis-with-multitoken","title":"KAMEL : Knowledge Analysis with Multitoken Entities in Language Models","date":"2022-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/informask-unsupervised-informative-masking","title":"InforMask: Unsupervised Informative Masking for Language Model Pretraining","date":"2022-10-21","arxiv_id":"2210.11771","n_code_links":1,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":1}},{"paper":"/paper/the-effectiveness-of-masked-language-modeling","title":"The Effectiveness of Masked Language Modeling and Adapters for Factual Knowledge Injection","date":"2022-10-03","arxiv_id":"2210.00907","n_code_links":1,"syntology":null},{"paper":null,"title":"Inpainting at Modern Camera Resolution by Guided PatchMatch with Auto-Curation","date":"2022-08-06","arxiv_id":"2208.03552","n_code_links":0,"syntology":null},{"paper":null,"title":"GLaMa: Joint Spatial and Frequency Loss for General Image Inpainting","date":"2022-05-15","arxiv_id":"2205.07162","n_code_links":0,"syntology":null},{"paper":null,"title":"Comparison of CoModGANs, LaMa and GLIDE for Art Inpainting- Completing M.C Escher's Print Gallery","date":"2022-05-03","arxiv_id":"2205.01741","n_code_links":0,"syntology":null},{"paper":null,"title":"Meta-learning via Language Model In-context Tuning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"SPE: Symmetrical Prompt Enhancement for Factual Knowledge Retrieval","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/meta-learning-via-language-model-in-context","title":"Meta-learning via Language Model In-context Tuning","date":"2021-10-15","arxiv_id":"2110.07814","n_code_links":1,"syntology":null},{"paper":"/paper/rewire-then-probe-a-contrastive-recipe-for","title":"Rewire-then-Probe: A Contrastive Recipe for Probing Biomedical Knowledge of Pre-trained Language Models","date":"2021-10-15","arxiv_id":"2110.08173","n_code_links":1,"syntology":null},{"paper":"/paper/p-adapters-robustly-extracting-factual-1","title":"P-Adapters: Robustly Extracting Factual Information from Language Models with Diverse Prompts","date":"2021-10-14","arxiv_id":"2110.07280","n_code_links":1,"syntology":null},{"paper":"/paper/resolution-robust-large-mask-inpainting-with","title":"Resolution-robust Large Mask Inpainting with Fourier Convolutions","date":"2021-09-15","arxiv_id":"2109.07161","n_code_links":8,"syntology":{"ran":0,"of":4,"unverified":4,"pointer_only":0}},{"paper":null,"title":"No Need to Know Everything! Efficiently Augmenting Language Models With External Knowledge","date":"2021-09-03","arxiv_id":null,"n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":11},{"task":"/task/language-modeling","name":"Language Modeling","papers":10},{"task":"/task/image-inpainting","name":"Image Inpainting","papers":4},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":3},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":3},{"task":"/task/knowledge-probing","name":"Knowledge Probing","papers":3},{"task":"/task/question-answering","name":"Question Answering","papers":3},{"task":"/task/text-classification","name":"Text Classification","papers":3},{"task":"/task/model","name":"model","papers":3},{"task":"/task/text-classification-1","name":"text-classification","papers":3},{"task":"/task/masked-language-modeling","name":"Masked Language Modeling","papers":2},{"task":"/task/meta-learning","name":"Meta-Learning","papers":2},{"task":"/task/object","name":"Object","papers":2},{"task":"/task/retrieval","name":"Retrieval","papers":2},{"task":"/task/ssim","name":"SSIM","papers":2},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":2},{"task":"/task/4k","name":"4k","papers":1},{"task":"/task/anomaly-detection","name":"Anomaly Detection","papers":1},{"task":"/task/articles","name":"Articles","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1}],"tasks_shown":20,"n_tasks":56,"usage_by_year":[{"year":"2019","papers":1},{"year":"2020","papers":4},{"year":"2021","papers":14},{"year":"2022","papers":9},{"year":"2023","papers":6},{"year":"2024","papers":6},{"year":"2025","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/lama"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}