{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/87","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":87,"pages_in_order":109,"rows_per_page":100,"rows":[8601,8700],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/86","next":"/method/attention-dropout/papers/88","papers":[{"paper":"/paper/efficient-transfer-learning-for-nlp-with","slug":"efficient-transfer-learning-for-nlp-with","title":"Efficient transfer learning for NLP with ELECTRA","date":"2021-04-06","arxiv_id":"2104.02756","n_code_links":1,"syntology":null},{"paper":"/paper/hbert-biascorp-fighting-racism-on-the-web","slug":"hbert-biascorp-fighting-racism-on-the-web","title":"HBert + BiasCorp -- Fighting Racism on the Web","date":"2021-04-06","arxiv_id":"2104.02242","n_code_links":0,"syntology":null},{"paper":null,"slug":"muslcat-multi-scale-multi-level-convolutional","title":"MuSLCAT: Multi-Scale Multi-Level Convolutional Attention Transformer for Discriminative Music Modeling on Raw Waveforms","date":"2021-04-06","arxiv_id":"2104.02309","n_code_links":0,"syntology":null},{"paper":null,"slug":"covid-19-sentiment-analysis-via-deep-learning","title":"COVID-19 sentiment analysis via deep learning during the rise of novel cases","date":"2021-04-05","arxiv_id":"2104.10662","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-transformers-in-emotion-recognition","title":"Exploring Transformers in Emotion Recognition: a comparison of BERT, DistillBERT, RoBERTa, XLNet and ELECTRA","date":"2021-04-05","arxiv_id":"2104.02041","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-distance-a-new-metric-for-asr","title":"Semantic Distance: A New Metric for ASR Performance Analysis Towards Spoken Language Understanding","date":"2021-04-05","arxiv_id":"2104.02138","n_code_links":0,"syntology":null},{"paper":"/paper/what-s-the-best-place-for-an-ai-conference","slug":"what-s-the-best-place-for-an-ai-conference","title":"What's the best place for an AI conference, Vancouver or ______: Why completing comparative questions is difficult","date":"2021-04-05","arxiv_id":"2104.01940","n_code_links":0,"syntology":null},{"paper":"/paper/improving-pretrained-models-for-zero-shot","slug":"improving-pretrained-models-for-zero-shot","title":"Improving Pretrained Models for Zero-shot Multi-label Text Classification through Reinforced Label Hierarchy Reasoning","date":"2021-04-04","arxiv_id":"2104.01666","n_code_links":1,"syntology":null},{"paper":null,"slug":"mcl-iitk-at-semeval-2021-task-2-multilingual","title":"MCL@IITK at SemEval-2021 Task 2: Multilingual and Cross-lingual Word-in-Context Disambiguation using Augmented Data, Signals, and Transformers","date":"2021-04-04","arxiv_id":"2104.01567","n_code_links":0,"syntology":null},{"paper":"/paper/recam-iitk-at-semeval-2021-task-4-bert-and","slug":"recam-iitk-at-semeval-2021-task-4-bert-and","title":"ReCAM@IITK at SemEval-2021 Task 4: BERT and ALBERT based Ensemble for Abstract Word Prediction","date":"2021-04-04","arxiv_id":"2104.01563","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-the-role-of-bert-token","slug":"exploring-the-role-of-bert-token","title":"Exploring the Role of BERT Token Representations to Explain Sentence Probing Results","date":"2021-04-03","arxiv_id":"2104.01477","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-domain-adaptation-with-global","title":"Unsupervised Domain Adaptation with Global and Local Graph Neural Networks in Limited Labeled Data Scenario: Application to Disaster Management","date":"2021-04-03","arxiv_id":"2104.01436","n_code_links":0,"syntology":null},{"paper":"/paper/iitk-lcp-at-semeval-2021-task-1","slug":"iitk-lcp-at-semeval-2021-task-1","title":"IITK@LCP at SemEval 2021 Task 1: Classification for Lexical Complexity Regression Task","date":"2021-04-02","arxiv_id":"2104.01046","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-coronavirus-is-a-bioweapon-analysing","title":"The Coronavirus is a Bioweapon: Analysing Coronavirus Fact-Checked Stories","date":"2021-04-02","arxiv_id":"2104.01215","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-gpt-2-to-create-synthetic-data-to","title":"Using GPT-2 to Create Synthetic Data to Improve the Prediction Performance of NLP Machine Learning Classification Models","date":"2021-04-02","arxiv_id":"2104.10658","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dashboard-for-mitigating-the-covid-19","title":"A Dashboard for Mitigating the COVID-19 Misinfodemic","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/are-neural-networks-extracting-linguistic","slug":"are-neural-networks-extracting-linguistic","title":"Are Neural Networks Extracting Linguistic Properties or Memorizing Training Data? An Observation with a Multilingual Probe for Predicting Tense","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-meets-cranfield-uncovering-the","title":"BERT meets Cranfield: Uncovering the Properties of Full Ranking on Fully Labeled Data","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bert-prescriptions-to-avoid-unwanted","slug":"bert-prescriptions-to-avoid-unwanted","title":"BERT Prescriptions to Avoid Unwanted Headaches: A Comparison of Transformer Architectures for Adverse Drug Event Detection","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bertective-language-models-and-contextual","title":"BERTective: Language Models and Contextual Information for Deception Detection","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/berxit-early-exiting-for-bert-with-better","slug":"berxit-early-exiting-for-bert-with-better","title":"BERxiT: Early Exiting for BERT with Better Fine-Tuning and Extension to Regression","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"complex-question-answering-on-knowledge","title":"Complex Question Answering on knowledge graphs using machine translation and multi-task learning","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"content-based-models-of-quotation","title":"Content-based Models of Quotation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-scenes-in-fiction-a-new","title":"Detecting Scenes in Fiction: A new Segmentation Task","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enpar-enhancing-entity-and-entity-pair","slug":"enpar-enhancing-entity-and-entity-pair","title":"ENPAR:Enhancing Entity and Entity Pair Representations for Joint Entity Relation Extraction","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-language-models-for-the-retrieval","slug":"evaluating-language-models-for-the-retrieval","title":"Evaluating language models for the retrieval and categorization of lexical collocations","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-neural-model-robustness-for","title":"Evaluating Neural Model Robustness for Machine Comprehension","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hle-upc-at-semeval-2021-task-5-multi-depth","slug":"hle-upc-at-semeval-2021-task-5-multi-depth","title":"HLE-UPC at SemEval-2021 Task 5: Multi-Depth DistilBERT for Toxic Spans Detection","date":"2021-04-01","arxiv_id":"2104.00639","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-fast-can-bert-learn-simple-natural","title":"How Fast can BERT Learn Simple Natural Language Inference?","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/keep-learning-self-supervised-meta-learning","slug":"keep-learning-self-supervised-meta-learning","title":"Keep Learning: Self-supervised Meta-learning for Learning from Inference","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"maximal-multiverse-learning-for-promoting","title":"Maximal Multiverse Learning for Promoting Cross-Task Generalization of Fine-Tuned Language Models","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-entity-and-relation-extraction","slug":"multilingual-entity-and-relation-extraction","title":"Multilingual Entity and Relation Extraction Dataset and Model","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-driven-search-based-paraphrase","title":"Neural-Driven Search-Based Paraphrase Generation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/nlquad-a-non-factoid-long-question-answering","slug":"nlquad-a-non-factoid-long-question-answering","title":"NLQuAD: A Non-Factoid Long Question Answering Data Set","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-in-effectiveness-of-images-for-text","title":"On the (In)Effectiveness of Images for Text Classification","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/probing-for-idiomaticity-in-vector-space","slug":"probing-for-idiomaticity-in-vector-space","title":"Probing for idiomaticity in vector space models","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieval-re-ranking-and-multi-task-learning","title":"Retrieval, Re-ranking and Multi-task Learning for Knowledge-Base Question Answering","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/russian-paraphrasers-paraphrase-with","slug":"russian-paraphrasers-paraphrase-with","title":"Russian Paraphrasers: Paraphrase with Transformers","date":"2021-04-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/an-in-depth-analysis-of-passage-level-label","slug":"an-in-depth-analysis-of-passage-level-label","title":"An In-depth Analysis of Passage-Level Label Transfer for Contextual Document Ranking","date":"2021-03-30","arxiv_id":"2103.16669","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-graph-partitioning-for-very-large","title":"Automatic Graph Partitioning for Very Large-scale Deep Learning","date":"2021-03-30","arxiv_id":"2103.16063","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-dialogue-systems-via-knowledge","slug":"grounding-dialogue-systems-via-knowledge","title":"Grounding Dialogue Systems via Knowledge Graph Aware Decoding with Pre-trained Transformers","date":"2021-03-30","arxiv_id":"2103.16289","n_code_links":1,"syntology":null},{"paper":"/paper/kaleido-bert-vision-language-pre-training-on","slug":"kaleido-bert-vision-language-pre-training-on","title":"Kaleido-BERT: Vision-Language Pre-training on Fashion Domain","date":"2021-03-30","arxiv_id":"2103.16110","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mczhuge/Kaleido-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/2103-15358","slug":"2103-15358","title":"Multi-Scale Vision Longformer: A New Vision Transformer for High-Resolution Image Encoding","date":"2021-03-29","arxiv_id":"2103.15358","n_code_links":3,"syntology":{"ran":8,"of":10,"n_ran_checked":5,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/vision-longformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"contextual-text-embeddings-for-twi","title":"Contextual Text Embeddings for Twi","date":"2021-03-29","arxiv_id":"2103.15963","n_code_links":0,"syntology":null},{"paper":null,"slug":"retraining-distilbert-for-a-voice-shopping","title":"Retraining DistilBERT for a Voice Shopping Assistant by Using Universal Dependencies","date":"2021-03-29","arxiv_id":"2103.15737","n_code_links":0,"syntology":null},{"paper":"/paper/whitening-sentence-representations-for-better","slug":"whitening-sentence-representations-for-better","title":"Whitening Sentence Representations for Better Semantics and Faster Retrieval","date":"2021-03-29","arxiv_id":"2103.15316","n_code_links":3,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bojone/BERT-whitening"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"png-bert-augmented-bert-on-phonemes-and","title":"PnG BERT: Augmented BERT on Phonemes and Graphemes for Neural TTS","date":"2021-03-28","arxiv_id":"2103.15060","n_code_links":0,"syntology":null},{"paper":"/paper/2103-14899","slug":"2103-14899","title":"CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image Classification","date":"2021-03-27","arxiv_id":"2103.14899","n_code_links":15,"syntology":{"ran":17,"of":26,"n_ran_checked":17,"n_instrument":0,"unverified":9,"pointer_only":0,"phrase":"17 ran (of which 10 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["IBM/CrossViT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"machine-learning-meets-natural-language","title":"Machine Learning Meets Natural Language Processing -- The story so far","date":"2021-03-27","arxiv_id":"2104.10213","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-self-training-for-sentiment","title":"Unsupervised Self-Training for Sentiment Analysis of Code-Switched Data","date":"2021-03-27","arxiv_id":"2103.14797","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-practical-survey-on-faster-and-lighter","title":"A Practical Survey on Faster and Lighter Transformers","date":"2021-03-26","arxiv_id":"2103.14636","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert4so-neural-sentence-ordering-by-fine","title":"BERT4SO: Neural Sentence Ordering by Fine-tuning BERT","date":"2021-03-25","arxiv_id":"2103.13584","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertinho-galician-bert-representations","title":"Bertinho: Galician BERT Representations","date":"2021-03-25","arxiv_id":"2103.13799","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-xlnet-a-general-method-for-combining","title":"K-XLNet: A General Method for Combining Explicit Knowledge with Language Model Pretraining","date":"2021-03-25","arxiv_id":"2104.10649","n_code_links":0,"syntology":null},{"paper":"/paper/predicting-directionality-in-causal-relations","slug":"predicting-directionality-in-causal-relations","title":"Predicting Directionality in Causal Relations in Text","date":"2021-03-25","arxiv_id":"2103.13606","n_code_links":2,"syntology":null},{"paper":null,"slug":"visual-grounding-strategies-for-text-only","title":"Visual Grounding Strategies for Text-Only Natural Language Processing","date":"2021-03-25","arxiv_id":"2103.13942","n_code_links":0,"syntology":null},{"paper":"/paper/czert-czech-bert-like-model-for-language","slug":"czert-czech-bert-like-model-for-language","title":"Czert -- Czech BERT-like Model for Language Representation","date":"2021-03-24","arxiv_id":"2103.13031","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-resource-machine-translation-for-low","title":"Low-Resource Machine Translation Training Curriculum Fit for Low-Resource Languages","date":"2021-03-24","arxiv_id":"2103.13272","n_code_links":0,"syntology":null},{"paper":null,"slug":"thinking-aloud-dynamic-context-generation","title":"Thinking Aloud: Dynamic Context Generation Improves Zero-Shot Reasoning Performance of GPT-2","date":"2021-03-24","arxiv_id":"2103.13033","n_code_links":0,"syntology":null},{"paper":"/paper/are-neural-language-models-good-plagiarists-a","slug":"are-neural-language-models-good-plagiarists-a","title":"Are Neural Language Models Good Plagiarists? A Benchmark for Neural Paraphrase Detection","date":"2021-03-23","arxiv_id":"2103.12450","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-hate-speech-with-gpt-3","slug":"detecting-hate-speech-with-gpt-3","title":"Detecting Hate Speech with GPT-3","date":"2021-03-23","arxiv_id":"2103.12407","n_code_links":2,"syntology":null},{"paper":null,"slug":"repairing-pronouns-in-translation-with-bert","title":"Repairing Pronouns in Translation with BERT-Based Post-Editing","date":"2021-03-23","arxiv_id":"2103.12838","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-nlp-cookbook-modern-recipes-for","title":"The NLP Cookbook: Modern Recipes for Transformer based Deep Learning Architectures","date":"2021-03-23","arxiv_id":"2104.10640","n_code_links":0,"syntology":null},{"paper":null,"slug":"tmr-evaluating-ner-recall-on-tough-mentions","title":"TMR: Evaluating NER Recall on Tough Mentions","date":"2021-03-23","arxiv_id":"2103.12312","n_code_links":0,"syntology":null},{"paper":null,"slug":"variable-name-recovery-in-decompiled-binary","title":"Variable Name Recovery in Decompiled Binary Code using Constrained Masked Language Modeling","date":"2021-03-23","arxiv_id":"2103.12801","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-a-review-of-applications-in-natural","title":"BERT: A Review of Applications in Natural Language Processing and Understanding","date":"2021-03-22","arxiv_id":"2103.11943","n_code_links":0,"syntology":null},{"paper":null,"slug":"bridging-the-gap-between-supervised","title":"Bridging the gap between supervised classification and unsupervised topic modelling for social-media assisted crisis management","date":"2021-03-22","arxiv_id":"2103.11835","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-model-for-patent-classification-using","slug":"hybrid-model-for-patent-classification-using","title":"PatentSBERTa: A Deep NLP based Hybrid Model for Patent Distance and Classification using Augmented SBERT","date":"2021-03-22","arxiv_id":"2103.11933","n_code_links":2,"syntology":null},{"paper":"/paper/identifying-machine-paraphrased-plagiarism","slug":"identifying-machine-paraphrased-plagiarism","title":"Identifying Machine-Paraphrased Plagiarism","date":"2021-03-22","arxiv_id":"2103.11909","n_code_links":2,"syntology":null},{"paper":"/paper/open-domain-question-answering-over-tables","slug":"open-domain-question-answering-over-tables","title":"Open Domain Question Answering over Tables via Dense Retrieval","date":"2021-03-22","arxiv_id":"2103.12011","n_code_links":1,"syntology":null},{"paper":null,"slug":"namerec-highly-accurate-and-fine-grained","title":"NameRec*: Highly Accurate and Fine-grained Person Name Recognition","date":"2021-03-21","arxiv_id":"2103.11360","n_code_links":0,"syntology":null},{"paper":"/paper/rosita-refined-bert-compression-with","slug":"rosita-refined-bert-compression-with","title":"ROSITA: Refined BERT cOmpreSsion with InTegrAted techniques","date":"2021-03-21","arxiv_id":"2103.11367","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["llyx97/Rosita"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/convit-improving-vision-transformers-with","slug":"convit-improving-vision-transformers-with","title":"ConViT: Improving Vision Transformers with Soft Convolutional Inductive Biases","date":"2021-03-19","arxiv_id":"2103.10697","n_code_links":9,"syntology":null},{"paper":null,"slug":"cost-effective-deployment-of-bert-models-in","title":"Cost-effective Deployment of BERT Models in Serverless Environment","date":"2021-03-19","arxiv_id":"2103.10673","n_code_links":0,"syntology":null},{"paper":"/paper/let-your-heart-speak-in-its-mother-tongue","slug":"let-your-heart-speak-in-its-mother-tongue","title":"Let Your Heart Speak in its Mother Tongue: Multilingual Captioning of Cardiac Signals","date":"2021-03-19","arxiv_id":"2103.11011","n_code_links":1,"syntology":null},{"paper":"/paper/muril-multilingual-representations-for-indian","slug":"muril-multilingual-representations-for-indian","title":"MuRIL: Multilingual Representations for Indian Languages","date":"2021-03-19","arxiv_id":"2103.10730","n_code_links":1,"syntology":null},{"paper":null,"slug":"play-the-shannon-game-with-language-models-a","title":"Play the Shannon Game With Language Models: A Human-Free Approach to Summary Evaluation","date":"2021-03-19","arxiv_id":"2103.10918","n_code_links":0,"syntology":null},{"paper":"/paper/all-nlp-tasks-are-generation-tasks-a-general","slug":"all-nlp-tasks-are-generation-tasks-a-general","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-03-18","arxiv_id":"2103.10360","n_code_links":8,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/GLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"contextual-biasing-of-language-models-for","title":"Contextual Biasing of Language Models for Speech Recognition in Goal-Oriented Conversational Agents","date":"2021-03-18","arxiv_id":"2103.10325","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-understands-too","slug":"gpt-understands-too","title":"GPT Understands, Too","date":"2021-03-18","arxiv_id":"2103.10385","n_code_links":10,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/P-tuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/model-extraction-and-adversarial","slug":"model-extraction-and-adversarial","title":"Model Extraction and Adversarial Transferability, Your BERT is Vulnerable!","date":"2021-03-18","arxiv_id":"2103.10013","n_code_links":1,"syntology":null},{"paper":null,"slug":"code-word-detection-in-fraud-investigations","title":"Code Word Detection in Fraud Investigations using a Deep-Learning Approach","date":"2021-03-17","arxiv_id":"2103.09606","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-role-of-images-for-analyzing-claims-in","slug":"on-the-role-of-images-for-analyzing-claims-in","title":"On the Role of Images for Analyzing Claims in Social Media","date":"2021-03-17","arxiv_id":"2103.09602","n_code_links":1,"syntology":null},{"paper":"/paper/uniparma-semeval-2021-task-5-toxic-spans","slug":"uniparma-semeval-2021-task-5-toxic-spans","title":"UniParma at SemEval-2021 Task 5: Toxic Spans Detection Using CharacterBERT and Bag-of-Words Model","date":"2021-03-17","arxiv_id":"2103.09645","n_code_links":1,"syntology":null},{"paper":null,"slug":"kgsynnet-a-novel-entity-synonyms-discovery","title":"KGSynNet: A Novel Entity Synonyms Discovery Framework with Knowledge Graph","date":"2021-03-16","arxiv_id":"2103.08893","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustly-optimized-and-distilled-training-for","title":"Robustly Optimized and Distilled Training for Natural Language Understanding","date":"2021-03-16","arxiv_id":"2103.08809","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-mining-of-stocktwits-data-for-predicting","title":"Text Mining of Stocktwits Data for Predicting Stock Prices","date":"2021-03-13","arxiv_id":"2103.16388","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-the-performance-of-nlp-toolkits-and","title":"Comparing the Performance of NLP Toolkits and Evaluation measures in Legal Tech","date":"2021-03-12","arxiv_id":"2103.11792","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-and-improving-bert-performance-on","title":"Explaining and Improving BERT Performance on Lexical Semantic Change Detection","date":"2021-03-12","arxiv_id":"2103.07259","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-bert-a-cross-disciplinary-knowledge","title":"Is BERT a Cross-Disciplinary Knowledge Learner? A Surprising Finding of Pre-trained Models' Transferability","date":"2021-03-12","arxiv_id":"2103.07162","n_code_links":0,"syntology":null},{"paper":null,"slug":"composite-re-ranking-for-efficient-document","title":"Composite Re-Ranking for Efficient Document Search with BERT","date":"2021-03-11","arxiv_id":"2103.06499","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-morphological-embeddings-for-1","title":"Evaluation of Morphological Embeddings for the Russian Language","date":"2021-03-11","arxiv_id":"2103.06628","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairfil-contrastive-neural-debiasing-method-1","title":"FairFil: Contrastive Neural Debiasing Method for Pretrained Text Encoders","date":"2021-03-11","arxiv_id":"2103.06413","n_code_links":0,"syntology":null},{"paper":"/paper/improving-bi-encoder-document-ranking-models","slug":"improving-bi-encoder-document-ranking-models","title":"Improving Bi-encoder Document Ranking Models with Two Rankers and Multi-teacher Distillation","date":"2021-03-11","arxiv_id":"2103.06523","n_code_links":1,"syntology":null},{"paper":null,"slug":"lightmbert-a-simple-yet-effective-method-for","title":"LightMBERT: A Simple Yet Effective Method for Multilingual BERT Distillation","date":"2021-03-11","arxiv_id":"2103.06418","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-text-to-sql-learning-with","title":"Self-supervised Text-to-SQL Learning with Header Alignment Training","date":"2021-03-11","arxiv_id":"2103.06402","n_code_links":0,"syntology":null},{"paper":"/paper/towards-multi-sense-cross-lingual-alignment-1","slug":"towards-multi-sense-cross-lingual-alignment-1","title":"Towards Multi-Sense Cross-Lingual Alignment of Contextual Embeddings","date":"2021-03-11","arxiv_id":"2103.06459","n_code_links":1,"syntology":null},{"paper":"/paper/majority-voting-with-bidirectional-pre","slug":"majority-voting-with-bidirectional-pre","title":"Majority Voting with Bidirectional Pre-translation For Bitext Retrieval","date":"2021-03-10","arxiv_id":"2103.06369","n_code_links":1,"syntology":null},{"paper":null,"slug":"ceqe-contextualized-embeddings-for-query","title":"CEQE: Contextualized Embeddings for Query Expansion","date":"2021-03-09","arxiv_id":"2103.05256","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-have-a-moral-dimension","slug":"language-models-have-a-moral-dimension","title":"Large Pre-trained Language Models Contain Human-like Biases of What is Right and Wrong to Do","date":"2021-03-08","arxiv_id":"2103.11790","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/MoRT_NMI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"53569ec25957e734ff93c8d9a9f92d48d6cdb5c405cf9a2f768b6521ec09fe9c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}