{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/294","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":294,"pages_in_order":316,"rows_per_page":100,"rows":[29301,29400],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/293","next":"/method/attention/papers/295","papers":[{"paper":"/paper/the-lottery-ticket-hypothesis-for-pre-trained","slug":"the-lottery-ticket-hypothesis-for-pre-trained","title":"The Lottery Ticket Hypothesis for Pre-trained BERT Networks","date":"2020-07-23","arxiv_id":"2007.12223","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":4,"n_instrument":5,"unverified":2,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["TAMU-VITA/BERT-Tickets","VITA-Group/BERT-Tickets"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analogical-reasoning-for-visually-grounded","title":"Analogical Reasoning for Visually Grounded Language Acquisition","date":"2020-07-22","arxiv_id":"2007.11668","n_code_links":0,"syntology":null},{"paper":"/paper/crosstransformers-spatially-aware-few-shot","slug":"crosstransformers-spatially-aware-few-shot","title":"CrossTransformers: spatially-aware few-shot transfer","date":"2020-07-22","arxiv_id":"2007.11498","n_code_links":6,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/meta-dataset"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","listed"]}}},{"paper":null,"slug":"iitk-at-the-finsim-task-hypernym-detection-in","title":"IITK at the FinSim Task: Hypernym Detection in Financial Domain via Context-Free and Contextualized Word Embeddings","date":"2020-07-22","arxiv_id":"2007.11201","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-learning-for-natural-language-1","title":"Multi-task learning for natural language processing in the 2020s: where are we going?","date":"2020-07-22","arxiv_id":"2007.16008","n_code_links":0,"syntology":null},{"paper":"/paper/check-square-at-checkthat-2020-claim","slug":"check-square-at-checkthat-2020-claim","title":"Check_square at CheckThat! 2020: Claim Detection in Social Media via Fusion of Transformer and Syntactic Features","date":"2020-07-21","arxiv_id":"2007.10534","n_code_links":1,"syntology":null},{"paper":"/paper/neural-machine-translation-with-error","slug":"neural-machine-translation-with-error","title":"Neural Machine Translation with Error Correction","date":"2020-07-21","arxiv_id":"2007.10681","n_code_links":1,"syntology":null},{"paper":"/paper/newssweeper-at-semeval-2020-task-11-context","slug":"newssweeper-at-semeval-2020-task-11-context","title":"newsSweeper at SemEval-2020 Task 11: Context-Aware Rich Feature Representations For Propaganda Classification","date":"2020-07-21","arxiv_id":"2007.10827","n_code_links":1,"syntology":null},{"paper":"/paper/problemconquero-at-semeval-2020-task-12","slug":"problemconquero-at-semeval-2020-task-12","title":"problemConquero at SemEval-2020 Task 12: Transformer and Soft label-based approaches","date":"2020-07-21","arxiv_id":"2007.10877","n_code_links":1,"syntology":null},{"paper":null,"slug":"sliceout-training-transformers-and-cnns","title":"Improving compute efficacy frontiers with SliceOut","date":"2020-07-21","arxiv_id":"2007.10909","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-bert-rankers-under-distillation","title":"Understanding BERT Rankers Under Distillation","date":"2020-07-21","arxiv_id":"2007.11088","n_code_links":0,"syntology":null},{"paper":"/paper/word-representation-for-rhythms","slug":"word-representation-for-rhythms","title":"Word Representation for Rhythms","date":"2020-07-21","arxiv_id":"2007.10610","n_code_links":1,"syntology":null},{"paper":"/paper/a-comparison-of-supervised-learning-to-match","slug":"a-comparison-of-supervised-learning-to-match","title":"A Comparison of Supervised Learning to Match Methods for Product Search","date":"2020-07-20","arxiv_id":"2007.10296","n_code_links":1,"syntology":null},{"paper":"/paper/conformer-kernel-with-query-term-independence","slug":"conformer-kernel-with-query-term-independence","title":"Conformer-Kernel with Query Term Independence for Document Retrieval","date":"2020-07-20","arxiv_id":"2007.10434","n_code_links":1,"syntology":null},{"paper":"/paper/learning-joint-spatial-temporal","slug":"learning-joint-spatial-temporal","title":"Learning Joint Spatial-Temporal Transformations for Video Inpainting","date":"2020-07-20","arxiv_id":"2007.10247","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":2,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["researchmm/STTN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/mono-vs-multilingual-transformer-based-models","slug":"mono-vs-multilingual-transformer-based-models","title":"Mono vs Multilingual Transformer-based Models: a Comparison across Several Language Tasks","date":"2020-07-19","arxiv_id":"2007.09757","n_code_links":1,"syntology":null},{"paper":null,"slug":"feature-pyramid-transformer","title":"Feature Pyramid Transformer","date":"2020-07-18","arxiv_id":"2007.09451","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-pointwise-convolutional-networks-for","slug":"temporal-pointwise-convolutional-networks-for","title":"Temporal Pointwise Convolutional Networks for Length of Stay Prediction in the Intensive Care Unit","date":"2020-07-18","arxiv_id":"2007.09483","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-based-traffic-surveillance","title":"Deep Learning Based Traffic Surveillance System For Missing and Suspicious Car Detection","date":"2020-07-17","arxiv_id":"2007.08783","n_code_links":0,"syntology":null},{"paper":"/paper/generative-pretraining-from-pixels","slug":"generative-pretraining-from-pixels","title":"Generative Pretraining from Pixels","date":"2020-07-17","arxiv_id":null,"n_code_links":4,"syntology":null},{"paper":null,"slug":"multi-perspective-semantic-information","title":"Multi-Perspective Semantic Information Retrieval in the Biomedical Domain","date":"2020-07-17","arxiv_id":"2008.01526","n_code_links":0,"syntology":null},{"paper":"/paper/hopfield-networks-is-all-you-need","slug":"hopfield-networks-is-all-you-need","title":"Hopfield Networks is All You Need","date":"2020-07-16","arxiv_id":"2008.02217","n_code_links":3,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ml-jku/hopfield-layers"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/investigating-pretrained-language-models-for","slug":"investigating-pretrained-language-models-for","title":"Investigating Pretrained Language Models for Graph-to-Text Generation","date":"2020-07-16","arxiv_id":"2007.08426","n_code_links":3,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["UKPLab/plms-graph2text"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-debiasing-sentence-representations-1","slug":"towards-debiasing-sentence-representations-1","title":"Towards Debiasing Sentence Representations","date":"2020-07-16","arxiv_id":"2007.08100","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pliang279/sent_debias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"translate-reverberated-speech-to-anechoic","title":"Translate Reverberated Speech to Anechoic Ones: Speech Dereverberation with BERT","date":"2020-07-16","arxiv_id":"2007.08052","n_code_links":0,"syntology":null},{"paper":"/paper/adapterhub-a-framework-for-adapting","slug":"adapterhub-a-framework-for-adapting","title":"AdapterHub: A Framework for Adapting Transformers","date":"2020-07-15","arxiv_id":"2007.07779","n_code_links":9,"syntology":{"ran":11,"of":15,"n_ran_checked":11,"n_instrument":0,"unverified":4,"pointer_only":12,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Adapter-Hub/Hub","Adapter-Hub/adapter-transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"deep-reinforced-query-reformulation-for","title":"Deep Reinforced Query Reformulation for Information Retrieval","date":"2020-07-15","arxiv_id":"2007.07987","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tune-longformer-for-jointly-predicting","title":"Fine-Tune Longformer for Jointly Predicting Rumor Stance and Veracity","date":"2020-07-15","arxiv_id":"2007.07803","n_code_links":0,"syntology":null},{"paper":"/paper/infoxlm-an-information-theoretic-framework","slug":"infoxlm-an-information-theoretic-framework","title":"InfoXLM: An Information-Theoretic Framework for Cross-Lingual Language Model Pre-Training","date":"2020-07-15","arxiv_id":"2007.07834","n_code_links":4,"syntology":null},{"paper":"/paper/logic-constrained-pointer-networks-for","slug":"logic-constrained-pointer-networks-for","title":"Logic Constrained Pointer Networks for Interpretable Textual Similarity","date":"2020-07-15","arxiv_id":"2007.07670","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-word-sense-disambiguation-in","slug":"multimodal-word-sense-disambiguation-in","title":"Multimodal Word Sense Disambiguation in Creative Practice","date":"2020-07-15","arxiv_id":"2007.07758","n_code_links":1,"syntology":null},{"paper":"/paper/overview-of-checkthat-2020-automatic","slug":"overview-of-checkthat-2020-automatic","title":"Overview of CheckThat! 2020: Automatic Identification and Verification of Claims in Social Media","date":"2020-07-15","arxiv_id":"2007.07997","n_code_links":3,"syntology":null},{"paper":null,"slug":"predicting-clinical-diagnosis-from-patients","title":"Predicting Clinical Diagnosis from Patients Electronic Health Records Using BERT-based Neural Networks","date":"2020-07-15","arxiv_id":"2007.07562","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-monte-carlo-transformer-a-stochastic-self","title":"The Monte Carlo Transformer: a stochastic self-attention model for sequence prediction","date":"2020-07-15","arxiv_id":"2007.08620","n_code_links":0,"syntology":null},{"paper":null,"slug":"add-a-sidenet-to-your-mainnet","title":"Add a SideNet to your MainNet","date":"2020-07-14","arxiv_id":"2007.13512","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-on-robustness-to-spurious","slug":"an-empirical-study-on-robustness-to-spurious","title":"An Empirical Study on Robustness to Spurious Correlations using Pre-trained Language Models","date":"2020-07-14","arxiv_id":"2007.06778","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-neural-networks-acquire-a-structural-bias","title":"Can neural networks acquire a structural bias from raw linguistic data?","date":"2020-07-14","arxiv_id":"2007.06761","n_code_links":0,"syntology":null},{"paper":"/paper/contextualized-code-representation-learning","slug":"contextualized-code-representation-learning","title":"CoreGen: Contextualized Code Representation Learning for Commit Message Generation","date":"2020-07-14","arxiv_id":"2007.06934","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-transformer-based-data-augmentation-with","title":"Deep Transformer based Data Augmentation with Subword Units for Morphologically Rich Online ASR","date":"2020-07-14","arxiv_id":"2007.06949","n_code_links":0,"syntology":null},{"paper":"/paper/emoji-prediction-extensions-and-benchmarking","slug":"emoji-prediction-extensions-and-benchmarking","title":"Emoji Prediction: Extensions and Benchmarking","date":"2020-07-14","arxiv_id":"2007.07389","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-memory-placement-using","title":"Optimizing Memory Placement using Evolutionary Graph Reinforcement Learning","date":"2020-07-14","arxiv_id":"2007.07298","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-s-in-a-name-are-bert-named-entity-1","title":"What's in a Name? Are BERT Named Entity Representations just as Good for any other Name?","date":"2020-07-14","arxiv_id":"2007.06897","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-enhanced-text-classification-to-explore","title":"An Enhanced Text Classification to Explore Health based Indian Government Policy Tweets","date":"2020-07-13","arxiv_id":"2007.06511","n_code_links":0,"syntology":null},{"paper":"/paper/paranoid-transformer-reading-narrative-of","slug":"paranoid-transformer-reading-narrative-of","title":"Paranoid Transformer: Reading Narrative of Madness as Computational Approach to Creativity","date":"2020-07-13","arxiv_id":"2007.06290","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-with-depth-wise-lstm","title":"Rewiring the Transformer with Depth-Wise LSTMs","date":"2020-07-13","arxiv_id":"2007.06257","n_code_links":0,"syntology":null},{"paper":null,"slug":"hypergrid-efficient-multi-task-transformers","title":"HyperGrid: Efficient Multi-Task Transformers with Grid-wise Decomposable Hyper Projections","date":"2020-07-12","arxiv_id":"2007.05891","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-graph-to-sequence-learning-for-vision","title":"Sparse Graph to Sequence Learning for Vision Conditioned Long Textual Sequence Generation","date":"2020-07-12","arxiv_id":"2007.06077","n_code_links":0,"syntology":null},{"paper":"/paper/tera-self-supervised-learning-of-transformer","slug":"tera-self-supervised-learning-of-transformer","title":"TERA: Self-Supervised Learning of Transformer Encoder Representation for Speech","date":"2020-07-12","arxiv_id":"2007.06028","n_code_links":7,"syntology":{"ran":11,"of":14,"n_ran_checked":7,"n_instrument":4,"unverified":3,"pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["andi611/Self-Supervised-Speech-Pretraining-and-Representation-Learning","s3prl/s3prl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"bert-learns-and-teaches-chemistry","title":"BERT Learns (and Teaches) Chemistry","date":"2020-07-11","arxiv_id":"2007.16012","n_code_links":0,"syntology":null},{"paper":"/paper/generative-graph-perturbations-for-scene","slug":"generative-graph-perturbations-for-scene","title":"Generative Compositional Augmentations for Scene Graph Prediction","date":"2020-07-11","arxiv_id":"2007.05756","n_code_links":1,"syntology":null},{"paper":"/paper/sequence-generation-with-mixed","slug":"sequence-generation-with-mixed","title":"Sequence Generation with Mixed Representations","date":"2020-07-11","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/bison-bm25-weighted-self-attention-framework","slug":"bison-bm25-weighted-self-attention-framework","title":"GLOW : Global Weighted Self-Attention Network for Web Search","date":"2020-07-10","arxiv_id":"2007.05186","n_code_links":1,"syntology":null},{"paper":"/paper/multi-dialect-arabic-bert-for-country-level","slug":"multi-dialect-arabic-bert-for-country-level","title":"Multi-Dialect Arabic BERT for Country-Level Dialect Identification","date":"2020-07-10","arxiv_id":"2007.05612","n_code_links":1,"syntology":null},{"paper":null,"slug":"to-ban-or-not-to-ban-bayesian-attention","title":"To BAN or not to BAN: Bayesian Attention Networks for Reliable Hate Speech Detection","date":"2020-07-10","arxiv_id":"2007.05304","n_code_links":0,"syntology":null},{"paper":"/paper/advances-of-transformer-based-models-for-news","slug":"advances-of-transformer-based-models-for-news","title":"Advances of Transformer-Based Models for News Headline Generation","date":"2020-07-09","arxiv_id":"2007.05044","n_code_links":2,"syntology":null},{"paper":"/paper/contrastive-code-representation-learning","slug":"contrastive-code-representation-learning","title":"Contrastive Code Representation Learning","date":"2020-07-09","arxiv_id":"2007.04973","n_code_links":1,"syntology":null},{"paper":null,"slug":"deepsinger-singing-voice-synthesis-with-data","title":"DeepSinger: Singing Voice Synthesis with Data Mined From the Web","date":"2020-07-09","arxiv_id":"2007.04590","n_code_links":0,"syntology":null},{"paper":"/paper/fast-transformers-with-clustered-attention","slug":"fast-transformers-with-clustered-attention","title":"Fast Transformers with Clustered Attention","date":"2020-07-09","arxiv_id":"2007.04825","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatio-temporal-scene-graphs-for-video-dialog","title":"Dynamic Graph Representation Learning for Video Dialog via Multi-Modal Shuffled Transformers","date":"2020-07-08","arxiv_id":"2007.03848","n_code_links":0,"syntology":null},{"paper":"/paper/continual-bert-continual-learning-for","slug":"continual-bert-continual-learning-for","title":"Continual BERT: Continual Learning for Adaptive Extractive Summarization of COVID-19 Literature","date":"2020-07-07","arxiv_id":"2007.03405","n_code_links":1,"syntology":null},{"paper":"/paper/do-transformers-need-deep-long-range-memory-1","slug":"do-transformers-need-deep-long-range-memory-1","title":"Do Transformers Need Deep Long-Range Memory","date":"2020-07-07","arxiv_id":"2007.03356","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-heterogeneous-information-networks","title":"Pre-Trained Models for Heterogeneous Information Networks","date":"2020-07-07","arxiv_id":"2007.03184","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-go-transformer-natural-language-modeling","title":"The Go Transformer: Natural Language Modeling for Game Play","date":"2020-07-07","arxiv_id":"2007.03500","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-contextual-embeddings-for-address","title":"Deep Contextual Embeddings for Address Classification in E-commerce","date":"2020-07-06","arxiv_id":"2007.03020","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-segment-anatomical-structures","title":"Learning to Segment Anatomical Structures Accurately from One Exemplar","date":"2020-07-06","arxiv_id":"2007.03052","n_code_links":0,"syntology":null},{"paper":null,"slug":"lmve-at-semeval-2020-task-4-commonsense","title":"LMVE at SemEval-2020 Task 4: Commonsense Validation and Explanation using Pretraining Language Model","date":"2020-07-06","arxiv_id":"2007.02540","n_code_links":0,"syntology":null},{"paper":null,"slug":"relevance-transformer-generating-concise-code","title":"Relevance Transformer: Generating Concise Code Snippets with Relevance Feedback","date":"2020-07-06","arxiv_id":"2007.02609","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-autocomplete-me-poisoning-vulnerabilities","title":"You Autocomplete Me: Poisoning Vulnerabilities in Neural Code Completion","date":"2020-07-05","arxiv_id":"2007.02220","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-prediction-of-punctuation-and-1","title":"Robust Prediction of Punctuation and Truecasing for Medical ASR","date":"2020-07-04","arxiv_id":"2007.02025","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-data-augmentation-towards-better","title":"Text Data Augmentation: Towards better detection of spear-phishing emails","date":"2020-07-04","arxiv_id":"2007.02033","n_code_links":0,"syntology":null},{"paper":null,"slug":"abstractive-and-mixed-summarization-for-long","title":"Abstractive and mixed summarization for long-single documents","date":"2020-07-03","arxiv_id":"2007.01918","n_code_links":0,"syntology":null},{"paper":"/paper/language-agnostic-bert-sentence-embedding","slug":"language-agnostic-bert-sentence-embedding","title":"Language-agnostic BERT Sentence Embedding","date":"2020-07-03","arxiv_id":"2007.01852","n_code_links":6,"syntology":null},{"paper":null,"slug":"mira-leveraging-multi-intention-co-click","title":"MIRA: Leveraging Multi-Intention Co-click Information in Web-scale Document Retrieval using Deep Neural Networks","date":"2020-07-03","arxiv_id":"2007.01510","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-fly-information-retrieval-augmentation-1","title":"On-The-Fly Information Retrieval Augmentation for Language Models","date":"2020-07-03","arxiv_id":"2007.01528","n_code_links":0,"syntology":null},{"paper":"/paper/playing-with-words-at-the-national-library-of","slug":"playing-with-words-at-the-national-library-of","title":"Playing with Words at the National Library of Sweden -- Making a Swedish BERT","date":"2020-07-03","arxiv_id":"2007.01658","n_code_links":1,"syntology":null},{"paper":null,"slug":"pretrained-semantic-speech-embeddings-for-end","title":"Pretrained Semantic Speech Embeddings for End-to-End Spoken Language Understanding via Cross-Modal Teacher-Student Learning","date":"2020-07-03","arxiv_id":"2007.01836","n_code_links":0,"syntology":null},{"paper":null,"slug":"reading-comprehension-in-czech-via-machine","title":"Reading Comprehension in Czech via Machine Translation and Cross-lingual Transfer","date":"2020-07-03","arxiv_id":"2007.01667","n_code_links":0,"syntology":null},{"paper":null,"slug":"bidirectional-encoder-representations-from","title":"Bidirectional Encoder Representations from Transformers (BERT): A sentiment analysis odyssey","date":"2020-07-02","arxiv_id":"2007.01127","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-event-detection-using-contextual","title":"Detecting Ongoing Events Using Contextual Word and Sentence Embeddings","date":"2020-07-02","arxiv_id":"2007.01379","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-explanations-on-ai-competency","title":"The Impact of Explanations on AI Competency Prediction in VQA","date":"2020-07-02","arxiv_id":"2007.00900","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bert-based-one-pass-multi-task-model-for","title":"A BERT-based One-Pass Multi-Task Model for Clinical Temporal Relation Extraction","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-generate-and-rank-framework-with-semantic","title":"A Generate-and-Rank Framework with Semantic Type Regularization for Biomedical Concept Normalization","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-metric-learning-approach-to-misogyny","title":"A Metric Learning Approach to Misogyny Categorization","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mixture-of-h-1-heads-is-better-than-h-heads-1","title":"A Mixture of h - 1 Heads is Better than h Heads","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-and-effective-dependency-parser-for","title":"A Simple and Effective Dependency Parser for Telugu","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-transformer-approach-to-contextual-sarcasm","title":"A Transformer Approach to Contextual Sarcasm Detection in Twitter","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptation-of-multilingual-transformer","title":"Adaptation of Multilingual Transformer Encoder for Robust Enhanced Universal Dependency Parsing","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"addressing-posterior-collapse-with-mutual","title":"Addressing Posterior Collapse with Mutual Information for Improved Variational Neural Machine Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-and-domain-aware-bert-for-cross","title":"Adversarial and Domain-Aware BERT for Cross-Domain Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-evaluation-of-bert-for-biomedical","title":"Adversarial Evaluation of BERT for Biomedical Named Entity Recognition","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-investigation-of-neural-methods","title":"An empirical investigation of neural methods for content scoring of science explanations","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-generation-of-citation-texts-in","title":"Automatic Generation of Citation Texts in Scholarly Papers: A Pilot Study","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"casia-s-system-for-iwslt-2020-open-domain","title":"CASIA's System for IWSLT 2020 Open Domain Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"character-aware-models-with-similarity","title":"Character aware models with similarity learning for metaphor detection","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-subword-representations-into-word","title":"Combining Subword Representations into Word-level Representations in the Transformer Architecture","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-neural-machine-translation-models","title":"Compressing Neural Machine Translation Models with 4-bit Precision","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-sarcasm-detection-using-bert","title":"Context-Aware Sarcasm Detection Using BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-and-non-contextual-word-embeddings","title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/contextualized-emotion-recognition-in","slug":"contextualized-emotion-recognition-in","title":"Contextualized Emotion Recognition in Conversation as Sequence Tagging","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"copybert-a-unified-approach-to-question","title":"CopyBERT: A Unified Approach to Question Generation with Self-Attention","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"06d100a9847589825a5df89bc1f92f9a5e2b8869c6b839df8a7cdf8a025b4b16","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}