{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/85","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":85,"pages_in_order":109,"rows_per_page":100,"rows":[8401,8500],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/84","next":"/method/attention-dropout/papers/86","papers":[{"paper":"/paper/compressed-communication-for-distributed","slug":"compressed-communication-for-distributed","title":"Compressed Communication for Distributed Training: Adaptive Methods and System","date":"2021-05-17","arxiv_id":"2105.07829","n_code_links":1,"syntology":null},{"paper":"/paper/pay-attention-to-mlps","slug":"pay-attention-to-mlps","title":"Pay Attention to MLPs","date":"2021-05-17","arxiv_id":"2105.08050","n_code_links":20,"syntology":{"ran":34,"of":44,"n_ran_checked":31,"n_instrument":3,"unverified":10,"pointer_only":11,"phrase":"34 ran (of which 15 constructed an object rather than computing a result; 31 with no instrument failure: 1 honoured, 4 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","official":null}},{"paper":"/paper/stage-wise-fine-tuning-for-graph-to-text","slug":"stage-wise-fine-tuning-for-graph-to-text","title":"Stage-wise Fine-tuning for Graph-to-Text Generation","date":"2021-05-17","arxiv_id":"2105.08021","n_code_links":1,"syntology":null},{"paper":null,"slug":"bdlan-bertdoc-label-attention-networks-for","title":"BdLAN:BERTdoc Label Attention Networks for Multi-label text classification","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"coming-to-its-senses-lessons-learned-from","title":"Coming to its senses: Lessons learned from Approximating Retrofitted BERT representations for Word Sense information","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-transformers-show-clusters-of-1","title":"Fine-Tuned Transformers Show Clusters of Similar Representations Across Layers","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/how-is-bert-surprised-layerwise-detection-of","slug":"how-is-bert-surprised-layerwise-detection-of","title":"How is BERT surprised? Layerwise detection of linguistic anomalies","date":"2021-05-16","arxiv_id":"2105.07452","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-predictive-text-for-grammatical-error","title":"Neural Predictive Text for Grammatical Error Prevention","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sina-bert-a-pre-trained-language-model-for-1","title":"SINA-BERT: A Pre-Trained Language Model for Analysis of Medical Texts in Persian","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slgpt-using-transfer-learning-to-directly","slug":"slgpt-using-transfer-learning-to-directly","title":"SLGPT: Using Transfer Learning to Directly Generate Simulink Model Files and Find Bugs in the Simulink Toolchain","date":"2021-05-16","arxiv_id":"2105.07465","n_code_links":1,"syntology":null},{"paper":null,"slug":"subtopic-clustering-with-a-query-specific","title":"Subtopic Clustering with a Query-Specific Siamese Similarity Metric","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"directqe-direct-pretraining-for-machine","title":"DirectQE: Direct Pretraining for Machine Translation Quality Estimation","date":"2021-05-15","arxiv_id":"2105.07149","n_code_links":0,"syntology":null},{"paper":"/paper/lexicon-enhanced-chinese-sequence-labelling","slug":"lexicon-enhanced-chinese-sequence-labelling","title":"Lexicon Enhanced Chinese Sequence Labeling Using BERT Adapter","date":"2021-05-15","arxiv_id":"2105.07148","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-low-dimensional-linear-geometry-of","title":"The Low-Dimensional Linear Geometry of Contextualized Word Representations","date":"2021-05-15","arxiv_id":"2105.07109","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-busters-outlier-layernorm-dimensions","title":"BERT Busters: Outlier Dimensions that Disrupt Transformers","date":"2021-05-14","arxiv_id":"2105.06990","n_code_links":0,"syntology":null},{"paper":null,"slug":"counterfactual-interventions-reveal-the","title":"Counterfactual Interventions Reveal the Causal Effect of Relative Clause Representations on Agreement Prediction","date":"2021-05-14","arxiv_id":"2105.06965","n_code_links":0,"syntology":null},{"paper":null,"slug":"dalaj-a-dataset-for-linguistic-acceptability","title":"DaLAJ - a dataset for linguistic acceptability judgments for Swedish: Format, baseline, sharing","date":"2021-05-14","arxiv_id":"2105.06681","n_code_links":0,"syntology":null},{"paper":"/paper/joint-retrieval-and-generation-training-for","slug":"joint-retrieval-and-generation-training-for","title":"RetGen: A Joint framework for Retrieval and Grounded Text Generation Modeling","date":"2021-05-14","arxiv_id":"2105.06597","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-bert-for-low-complexity-network","title":"Distilling BERT for low complexity network training","date":"2021-05-13","arxiv_id":"2105.06514","n_code_links":0,"syntology":null},{"paper":"/paper/bertgcn-transductive-text-classification-by","slug":"bertgcn-transductive-text-classification-by","title":"BertGCN: Transductive Text Classification by Combining GCN and BERT","date":"2021-05-12","arxiv_id":"2105.05727","n_code_links":1,"syntology":null},{"paper":null,"slug":"better-than-bert-but-worse-than-baseline","title":"Better than BERT but Worse than Baseline","date":"2021-05-12","arxiv_id":"2105.05915","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-a-question-and-answer-system-for","title":"Building a Question and Answer System for News Domain","date":"2021-05-12","arxiv_id":"2105.05744","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-gender-bias-in-natural-language-1","slug":"evaluating-gender-bias-in-natural-language-1","title":"Evaluating Gender Bias in Natural Language Inference","date":"2021-05-12","arxiv_id":"2105.05541","n_code_links":1,"syntology":null},{"paper":null,"slug":"go-beyond-plain-fine-tuning-improving","title":"Go Beyond Plain Fine-tuning: Improving Pretrained Models for Social Commonsense","date":"2021-05-12","arxiv_id":"2105.05913","n_code_links":0,"syntology":null},{"paper":"/paper/kleister-key-information-extraction-datasets","slug":"kleister-key-information-extraction-datasets","title":"Kleister: Key Information Extraction Datasets Involving Long Documents with Complex Layouts","date":"2021-05-12","arxiv_id":"2105.05796","n_code_links":0,"syntology":null},{"paper":"/paper/mate-kd-masked-adversarial-text-a-companion","slug":"mate-kd-masked-adversarial-text-a-companion","title":"MATE-KD: Masked Adversarial TExt, a Companion to Knowledge Distillation","date":"2021-05-12","arxiv_id":"2105.05912","n_code_links":1,"syntology":null},{"paper":null,"slug":"ochadai-kyodai-at-semeval-2021-task-1","title":"OCHADAI-KYOTO at SemEval-2021 Task 1: Enhancing Model Generalization and Robustness for Lexical Complexity Prediction","date":"2021-05-12","arxiv_id":"2105.05535","n_code_links":0,"syntology":null},{"paper":"/paper/playing-codenames-with-language-graphs-and","slug":"playing-codenames-with-language-graphs-and","title":"Playing Codenames with Language Graphs and Word Embeddings","date":"2021-05-12","arxiv_id":"2105.05885","n_code_links":1,"syntology":null},{"paper":"/paper/priberam-at-mesinesp-multi-label","slug":"priberam-at-mesinesp-multi-label","title":"Priberam at MESINESP Multi-label Classification of Medical Texts Task","date":"2021-05-12","arxiv_id":"2105.05614","n_code_links":1,"syntology":null},{"paper":null,"slug":"priberam-labs-at-the-ntcir-15-shinra2020-ml","title":"Priberam Labs at the NTCIR-15 SHINRA2020-ML: Classification Task","date":"2021-05-12","arxiv_id":"2105.05605","n_code_links":0,"syntology":null},{"paper":"/paper/addressing-documentation-debt-in-machine","slug":"addressing-documentation-debt-in-machine","title":"Addressing \"Documentation Debt\" in Machine Learning Research: A Retrospective Datasheet for BookCorpus","date":"2021-05-11","arxiv_id":"2105.05241","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jackbandy/bookcorpus-datasheet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","slug":"bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","title":"BERT is to NLP what AlexNet is to CV: Can Pre-Trained Language Models Identify Analogies?","date":"2021-05-11","arxiv_id":"2105.04949","n_code_links":1,"syntology":null},{"paper":"/paper/el-attention-memory-efficient-lossless","slug":"el-attention-memory-efficient-lossless","title":"EL-Attention: Memory Efficient Lossless Attention for Generation","date":"2021-05-11","arxiv_id":"2105.04779","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/fastseq"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"integrating-extracted-information-from-bert","title":"Integrating extracted information from bert and multiple embedding methods with the deep neural network for humour detection","date":"2021-05-11","arxiv_id":"2105.05112","n_code_links":0,"syntology":null},{"paper":null,"slug":"role-of-artificial-intelligence-in-detection","title":"Role of Artificial Intelligence in Detection of Hateful Speech for Hinglish Data on Social Media","date":"2021-05-11","arxiv_id":"2105.04913","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-syntactic-capabilities-of","title":"Assessing the Syntactic Capabilities of Transformer-based Multilingual Language Models","date":"2021-05-10","arxiv_id":"2105.04688","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-classification-of-human-translation","title":"Automatic Classification of Human Translation and Machine Translation: A Study from the Perspective of Lexical Diversity","date":"2021-05-10","arxiv_id":"2105.04616","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-learning-with-swin","slug":"self-supervised-learning-with-swin","title":"Self-Supervised Learning with Swin Transformers","date":"2021-05-10","arxiv_id":"2105.04553","n_code_links":6,"syntology":null},{"paper":null,"slug":"srlf-a-stance-aware-reinforcement-learning","title":"SRLF: A Stance-aware Reinforcement Learning Framework for Content-based Rumor Detection on Social Media","date":"2021-05-10","arxiv_id":"2105.04098","n_code_links":0,"syntology":null},{"paper":"/paper/fnet-mixing-tokens-with-fourier-transforms","slug":"fnet-mixing-tokens-with-fourier-transforms","title":"FNet: Mixing Tokens with Fourier Transforms","date":"2021-05-09","arxiv_id":"2105.03824","n_code_links":12,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"which-transformer-architecture-fits-my-data-a","title":"Which transformer architecture fits my data? A vocabulary bottleneck in self-attention","date":"2021-05-09","arxiv_id":"2105.03928","n_code_links":0,"syntology":null},{"paper":"/paper/e-vil-a-dataset-and-benchmark-for-natural","slug":"e-vil-a-dataset-and-benchmark-for-natural","title":"e-ViL: A Dataset and Benchmark for Natural Language Explanations in Vision-Language Tasks","date":"2021-05-08","arxiv_id":"2105.03761","n_code_links":2,"syntology":{"ran":7,"of":15,"n_ran_checked":6,"n_instrument":1,"unverified":8,"pointer_only":15,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["maximek3/e-ViL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/improving-named-entity-recognition-by","slug":"improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","arxiv_id":"2105.03654","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["modelscope/adaseq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nlp-iis-ut-at-semeval-2021-task-4-machine","title":"NLP-IIS@UT at SemEval-2021 Task 4: Machine Reading Comprehension using the Long Document Transformer","date":"2021-05-08","arxiv_id":"2105.03775","n_code_links":0,"syntology":null},{"paper":"/paper/adapting-by-pruning-a-case-study-on-bert","slug":"adapting-by-pruning-a-case-study-on-bert","title":"Adapting by Pruning: A Case Study on BERT","date":"2021-05-07","arxiv_id":"2105.03343","n_code_links":1,"syntology":null},{"paper":"/paper/empirical-evaluation-of-pre-trained","slug":"empirical-evaluation-of-pre-trained","title":"Empirical Evaluation of Pre-trained Transformers for Human-Level NLP: The Role of Sample Size and Dimensionality","date":"2021-05-07","arxiv_id":"2105.03484","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-fly-controlled-text-generation-with","slug":"on-the-fly-controlled-text-generation-with","title":"DExperts: Decoding-Time Controlled Text Generation with Experts and Anti-Experts","date":"2021-05-07","arxiv_id":"2105.03023","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alisawuffles/DExperts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-by-understanding-not-modeling","slug":"understanding-by-understanding-not-modeling","title":"Understanding by Understanding Not: Modeling Negation in Language Models","date":"2021-05-07","arxiv_id":"2105.03519","n_code_links":1,"syntology":null},{"paper":"/paper/adapting-monolingual-models-data-can-be","slug":"adapting-monolingual-models-data-can-be","title":"Adapting Monolingual Models: Data can be Scarce when Language Similarity is High","date":"2021-05-06","arxiv_id":"2105.02855","n_code_links":1,"syntology":null},{"paper":null,"slug":"aligning-subtitles-in-sign-language-videos","title":"Aligning Subtitles in Sign Language Videos","date":"2021-05-06","arxiv_id":"2105.02877","n_code_links":0,"syntology":null},{"paper":"/paper/bird-s-eye-probing-for-linguistic-graph","slug":"bird-s-eye-probing-for-linguistic-graph","title":"Bird's Eye: Probing for Linguistic Graph Structures with a Simple Information-Theoretic Approach","date":"2021-05-06","arxiv_id":"2105.02629","n_code_links":1,"syntology":null},{"paper":"/paper/introducing-information-retrieval-for","slug":"introducing-information-retrieval-for","title":"Introducing Information Retrieval for Biomedical Informatics Students","date":"2021-05-06","arxiv_id":"2105.02746","n_code_links":1,"syntology":null},{"paper":"/paper/tabbie-pretrained-representations-of-tabular","slug":"tabbie-pretrained-representations-of-tabular","title":"TABBIE: Pretrained Representations of Tabular Data","date":"2021-05-06","arxiv_id":"2105.02584","n_code_links":2,"syntology":null},{"paper":null,"slug":"goldilocks-just-right-tuning-of-bert-for","title":"Goldilocks: Just-Right Tuning of BERT for Technology-Assisted Review","date":"2021-05-03","arxiv_id":"2105.01044","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-model-to-rule-them-all-towards-zero-shot","title":"One Model to Rule them All: Towards Zero-Shot Learning for Databases","date":"2021-05-03","arxiv_id":"2105.00642","n_code_links":0,"syntology":null},{"paper":"/paper/smoothi-smooth-rank-indicators-for","slug":"smoothi-smooth-rank-indicators-for","title":"SmoothI: Smooth Rank Indicators for Differentiable IR Metrics","date":"2021-05-03","arxiv_id":"2105.00942","n_code_links":1,"syntology":null},{"paper":"/paper/unreasonable-effectiveness-of-rule-based","slug":"unreasonable-effectiveness-of-rule-based","title":"Unreasonable Effectiveness of Rule-Based Heuristics in Solving Russian SuperGLUE Tasks","date":"2021-05-03","arxiv_id":"2105.01192","n_code_links":0,"syntology":null},{"paper":null,"slug":"mathbert-a-pre-trained-model-for-mathematical","title":"MathBERT: A Pre-Trained Model for Mathematical Formula Understanding","date":"2021-05-02","arxiv_id":"2105.00377","n_code_links":0,"syntology":null},{"paper":"/paper/mrcbert-a-machine-reading","slug":"mrcbert-a-machine-reading","title":"MRCBert: A Machine Reading ComprehensionApproach for Unsupervised Summarization","date":"2021-05-01","arxiv_id":"2105.00239","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-meets-relational-db-contextual","title":"BERT Meets Relational DB: Contextual Representations of Relational Databases","date":"2021-04-30","arxiv_id":"2104.14914","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-political-bias-in-language-models","title":"Mitigating Political Bias in Language Models Through Reinforced Calibration","date":"2021-04-30","arxiv_id":"2104.14795","n_code_links":0,"syntology":null},{"paper":"/paper/word-sense-disambiguation-with-transformer","slug":"word-sense-disambiguation-with-transformer","title":"Word Sense Disambiguation with Transformer Models","date":"2021-04-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/entailment-as-few-shot-learner","slug":"entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","arxiv_id":"2104.14690","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/let-s-play-mono-poly-bert-can-reveal-words","slug":"let-s-play-mono-poly-bert-can-reveal-words","title":"Let's Play Mono-Poly: BERT Can Reveal Words' Polysemy Level and Partitionability into Senses","date":"2021-04-29","arxiv_id":"2104.14694","n_code_links":1,"syntology":null},{"paper":"/paper/improving-bert-model-using-contrastive","slug":"improving-bert-model-using-contrastive","title":"Improving BERT Model Using Contrastive Learning for Biomedical Relation Extraction","date":"2021-04-28","arxiv_id":"2104.13913","n_code_links":1,"syntology":null},{"paper":"/paper/melbert-metaphor-detection-via-contextualized","slug":"melbert-metaphor-detection-via-contextualized","title":"MelBERT: Metaphor Detection via Contextualized Late Interaction using Metaphorical Identification Theories","date":"2021-04-28","arxiv_id":"2104.13615","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["jin530/MelBERT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"multi-task-learning-of-query-intent-and-named","title":"Multi-Task Learning of Query Intent and Named Entities using Transfer Learning","date":"2021-04-28","arxiv_id":"2105.03316","n_code_links":0,"syntology":null},{"paper":"/paper/societal-biases-in-retrieved-contents","slug":"societal-biases-in-retrieved-contents","title":"Societal Biases in Retrieved Contents: Measurement Framework and Adversarial Mitigation for BERT Rankers","date":"2021-04-28","arxiv_id":"2104.13640","n_code_links":1,"syntology":null},{"paper":null,"slug":"extractive-and-abstractive-explanations-for","title":"Extractive and Abstractive Explanations for Fact-Checking and Evaluation of News","date":"2021-04-27","arxiv_id":"2104.12918","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-class-text-classification-using-bert","title":"Multi-class Text Classification using BERT-based Active Learning","date":"2021-04-27","arxiv_id":"2104.14289","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-supervised-interactive-intent-labeling","title":"Semi-supervised Interactive Intent Labeling","date":"2021-04-27","arxiv_id":"2104.13406","n_code_links":0,"syntology":null},{"paper":null,"slug":"uot-uwf-partai-at-semeval-2021-task-5-self","title":"UoT-UWF-PartAI at SemEval-2021 Task 5: Self Attention Based Bi-GRU with Multi-Embedding Representation for Toxicity Highlighter","date":"2021-04-27","arxiv_id":"2104.13164","n_code_links":0,"syntology":null},{"paper":null,"slug":"accounting-for-agreement-phenomena-in","title":"Accounting for Agreement Phenomena in Sentence Comprehension with Transformer Language Models: Effects of Similarity-based Interference on Surprisal and Attention","date":"2021-04-26","arxiv_id":"2104.12874","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-image-inpainting-with-bidirectional","title":"Diverse Image Inpainting with Bidirectional and Autoregressive Transformers","date":"2021-04-26","arxiv_id":"2104.12335","n_code_links":0,"syntology":null},{"paper":"/paper/easy-and-efficient-transformer-scalable","slug":"easy-and-efficient-transformer-scalable","title":"Easy and Efficient Transformer : Scalable Inference Solution For large NLP model","date":"2021-04-26","arxiv_id":"2104.12470","n_code_links":1,"syntology":null},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","slug":"mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","n_code_links":5,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ashkamath/mdetr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/pangu-a-large-scale-autoregressive-pretrained","slug":"pangu-a-large-scale-autoregressive-pretrained","title":"PanGu-$α$: Large-scale Autoregressive Pretrained Chinese Language Models with Auto-parallel Computation","date":"2021-04-26","arxiv_id":"2104.12369","n_code_links":5,"syntology":null},{"paper":"/paper/phrase-break-prediction-with-bidirectional","slug":"phrase-break-prediction-with-bidirectional","title":"Phrase break prediction with bidirectional encoder representations in Japanese text-to-speech synthesis","date":"2021-04-26","arxiv_id":"2104.12395","n_code_links":1,"syntology":null},{"paper":"/paper/potential-idiomatic-expression-pie-english","slug":"potential-idiomatic-expression-pie-english","title":"Potential Idiomatic Expression (PIE)-English: Corpus for Classes of Idioms","date":"2021-04-25","arxiv_id":"2105.03280","n_code_links":2,"syntology":null},{"paper":null,"slug":"extract-then-distill-efficient-and-effective","title":"Extract then Distill: Efficient and Effective Task-Agnostic BERT Distillation","date":"2021-04-24","arxiv_id":"2104.11928","n_code_links":0,"syntology":null},{"paper":"/paper/learning-passage-impacts-for-inverted-indexes","slug":"learning-passage-impacts-for-inverted-indexes","title":"Learning Passage Impacts for Inverted Indexes","date":"2021-04-24","arxiv_id":"2104.12016","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DI4IR/SIGIR2021"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analysing-cyberbullying-using-natural","title":"Analysing Cyberbullying using Natural Language Processing by Understanding Jargon in Social Media","date":"2021-04-23","arxiv_id":"2107.08902","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-coqac-bert-based-conversational-question","title":"BERT-CoQAC: BERT-based Conversational Question Answering in Context","date":"2021-04-23","arxiv_id":"2104.11394","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-machine-learning-and","title":"Comparative Analysis of Machine Learning and Deep Learning Algorithms for Detection of Online Hate Speech","date":"2021-04-23","arxiv_id":"2108.01063","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-fusion-with-bert-and-attention","slug":"multimodal-fusion-with-bert-and-attention","title":"Multimodal Fusion with BERT and Attention Mechanism for Fake News Detection","date":"2021-04-23","arxiv_id":"2104.11476","n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-small-berts-trained-for-german-ner","slug":"optimizing-small-berts-trained-for-german-ner","title":"Optimizing small BERTs trained for German NER","date":"2021-04-23","arxiv_id":"2104.11559","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-trustworthy-deception-detection","title":"Towards Trustworthy Deception Detection: Benchmarking Model Robustness across Domains, Modalities, and Languages","date":"2021-04-23","arxiv_id":"2104.11761","n_code_links":0,"syntology":null},{"paper":"/paper/on-geodesic-distances-and-contextual","slug":"on-geodesic-distances-and-contextual","title":"On Geodesic Distances and Contextual Embedding Compression for Text Classification","date":"2021-04-22","arxiv_id":"2104.11295","n_code_links":1,"syntology":null},{"paper":null,"slug":"discriminative-self-training-for-punctuation","title":"Discriminative Self-training for Punctuation Prediction","date":"2021-04-21","arxiv_id":"2104.10339","n_code_links":0,"syntology":null},{"paper":null,"slug":"disfluency-detection-with-unlabeled-data-and","title":"Disfluency Detection with Unlabeled Data and Small BERT Models","date":"2021-04-21","arxiv_id":"2104.10769","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-covid-19-tweets-with-transformer","title":"Analyzing COVID-19 Tweets with Transformer-based Language Models","date":"2021-04-20","arxiv_id":"2104.10259","n_code_links":0,"syntology":null},{"paper":"/paper/b-prop-bootstrapped-pre-training-with","slug":"b-prop-bootstrapped-pre-training-with","title":"B-PROP: Bootstrapped Pre-training with Representative Words Prediction for Ad-hoc Retrieval","date":"2021-04-20","arxiv_id":"2104.09791","n_code_links":1,"syntology":{"ran":10,"of":16,"n_ran_checked":7,"n_instrument":3,"unverified":6,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["Albert-Ma/PROP"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-pre-training-objectives-for","title":"Efficient pre-training objectives for Transformers","date":"2021-04-20","arxiv_id":"2104.09694","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-shifts-in-attitudes-towards-covid","slug":"measuring-shifts-in-attitudes-towards-covid","title":"Measuring Shifts in Attitudes Towards COVID-19 Measures in Belgium Using Multilingual BERT","date":"2021-04-20","arxiv_id":"2104.09947","n_code_links":1,"syntology":null},{"paper":"/paper/subsentence-extraction-from-text-using","slug":"subsentence-extraction-from-text-using","title":"Subsentence Extraction from Text Using Coverage-Based Deep Learning Language Models","date":"2021-04-20","arxiv_id":"2104.09777","n_code_links":1,"syntology":null},{"paper":"/paper/uit-ise-nlp-at-semeval-2021-task-5-toxic","slug":"uit-ise-nlp-at-semeval-2021-task-5-toxic","title":"UIT-ISE-NLP at SemEval-2021 Task 5: Toxic Spans Detection with BiLSTM-CRF and ToxicBERT Comment Classification","date":"2021-04-20","arxiv_id":"2104.10100","n_code_links":1,"syntology":null},{"paper":null,"slug":"wassa-iitk-at-wassa-2021-multi-task-learning","title":"WASSA@IITK at WASSA 2021: Multi-task Learning and Transformer Finetuning for Emotion Classification and Empathy Prediction","date":"2021-04-20","arxiv_id":"2104.09827","n_code_links":0,"syntology":null},{"paper":"/paper/attention-in-attention-network-for-image","slug":"attention-in-attention-network-for-image","title":"Attention in Attention Network for Image Super-Resolution","date":"2021-04-19","arxiv_id":"2104.09497","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haoyuc/A2N"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/biggreen-at-semeval-2021-task-1-lexical","slug":"biggreen-at-semeval-2021-task-1-lexical","title":"BigGreen at SemEval-2021 Task 1: Lexical Complexity Prediction with Assembly Models","date":"2021-04-19","arxiv_id":"2104.09040","n_code_links":1,"syntology":null},{"paper":"/paper/electramed-a-new-pre-trained-language","slug":"electramed-a-new-pre-trained-language","title":"ELECTRAMed: a new pre-trained language representation model for biomedical NLP","date":"2021-04-19","arxiv_id":"2104.09585","n_code_links":2,"syntology":null}],"record_sha256":"597a92666c4d7d2e2e249e30886c926c4dc34edd419bd8cb214dbea1e44e2c8e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}