{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/80","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":80,"pages_in_order":109,"rows_per_page":100,"rows":[7901,8000],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/79","next":"/method/attention-dropout/papers/81","papers":[{"paper":"/paper/text-free-prosody-aware-generative-spoken","slug":"text-free-prosody-aware-generative-spoken","title":"Text-Free Prosody-Aware Generative Spoken Language Modeling","date":"2021-09-07","arxiv_id":"2109.03264","n_code_links":1,"syntology":null},{"paper":"/paper/does-bert-learn-as-humans-perceive","slug":"does-bert-learn-as-humans-perceive","title":"Does BERT Learn as Humans Perceive? Understanding Linguistic Styles through Lexica","date":"2021-09-06","arxiv_id":"2109.02738","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sweetpeach/hummingbird"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-language-models-with-plug-and-play","slug":"enhancing-language-models-with-plug-and-play","title":"Enhancing Natural Language Representation with Large-Scale Out-of-Domain Commonsense","date":"2021-09-06","arxiv_id":"2109.02572","n_code_links":1,"syntology":null},{"paper":"/paper/general-purpose-question-answering-with-macaw","slug":"general-purpose-question-answering-with-macaw","title":"General-Purpose Question-Answering with Macaw","date":"2021-09-06","arxiv_id":"2109.02593","n_code_links":2,"syntology":null},{"paper":"/paper/gpt-3-models-are-poor-few-shot-learners-in","slug":"gpt-3-models-are-poor-few-shot-learners-in","title":"GPT-3 Models are Poor Few-Shot Learners in the Biomedical Domain","date":"2021-09-06","arxiv_id":"2109.02555","n_code_links":1,"syntology":null},{"paper":null,"slug":"ss-bert-mitigating-identity-terms-bias-in","title":"SS-BERT: Mitigating Identity Terms Bias in Toxic Comment Classification by Utilising the Notion of \"Subjectivity\" and \"Identity Terms\"","date":"2021-09-06","arxiv_id":"2109.02691","n_code_links":0,"syntology":null},{"paper":null,"slug":"error-detection-in-large-scale-natural","title":"Error Detection in Large-Scale Natural Language Understanding Systems Using Transformer Models","date":"2021-09-04","arxiv_id":"2109.01754","n_code_links":0,"syntology":null},{"paper":"/paper/uncovering-the-limits-of-text-based-emotion","slug":"uncovering-the-limits-of-text-based-emotion","title":"Uncovering the Limits of Text-based Emotion Detection","date":"2021-09-04","arxiv_id":"2109.01900","n_code_links":2,"syntology":null},{"paper":"/paper/a-context-aware-hierarchical-bert-fusion","slug":"a-context-aware-hierarchical-bert-fusion","title":"A Context-Aware Hierarchical BERT Fusion Network for Multi-turn Dialog Act Detection","date":"2021-09-03","arxiv_id":"2109.01267","n_code_links":1,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","slug":"finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","arxiv_id":"2109.01652","n_code_links":8,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["google-research/flan"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/codet5-identifier-aware-unified-pre-trained","slug":"codet5-identifier-aware-unified-pre-trained","title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","date":"2021-09-02","arxiv_id":"2109.00859","n_code_links":5,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["salesforce/codet5"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"conqx-semantic-expansion-of-spoken-queries","title":"ConQX: Semantic Expansion of Spoken Queries for Intent Detection based on Conditioned Text Generation","date":"2021-09-02","arxiv_id":"2109.00729","n_code_links":0,"syntology":null},{"paper":null,"slug":"legalmfit-efficient-short-legal-text","title":"LegaLMFiT: Efficient Short Legal Text Classification with LSTM Language Model Pre-Training","date":"2021-09-02","arxiv_id":"2109.00993","n_code_links":0,"syntology":null},{"paper":null,"slug":"so-cloze-yet-so-far-n400-amplitude-is-better","title":"So Cloze yet so Far: N400 Amplitude is Better Predicted by Distributional Information than Human Predictability Judgements","date":"2021-09-02","arxiv_id":"2109.01226","n_code_links":0,"syntology":null},{"paper":null,"slug":"travelbert-pre-training-language-model","title":"Pre-training Language Model Incorporating Domain-specific Heterogeneous Knowledge into A Unified Representation","date":"2021-09-02","arxiv_id":"2109.01048","n_code_links":0,"syntology":null},{"paper":"/paper/dilbert-customized-pre-training-for-domain","slug":"dilbert-customized-pre-training-for-domain","title":"DILBERT: Customized Pre-Training for Domain Adaptation withCategory Shift, with an Application to Aspect Extraction","date":"2021-09-01","arxiv_id":"2109.00571","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-deep-learning-methods-for","slug":"exploring-deep-learning-methods-for","title":"Exploring deep learning methods for recognizing rare diseases and their clinical manifestations from texts","date":"2021-09-01","arxiv_id":"2109.00343","n_code_links":2,"syntology":null},{"paper":null,"slug":"fight-fire-with-fire-fine-tuning-hate","title":"Fight Fire with Fire: Fine-tuning Hate Detectors using Large Samples of Generated Hate Speech","date":"2021-09-01","arxiv_id":"2109.00591","n_code_links":0,"syntology":null},{"paper":"/paper/optagan-entropy-based-finetuning-on-text-vae","slug":"optagan-entropy-based-finetuning-on-text-vae","title":"OptAGAN: Entropy-based finetuning on text VAE-GAN","date":"2021-09-01","arxiv_id":"2109.00239","n_code_links":1,"syntology":null},{"paper":"/paper/searching-for-efficient-multi-stage-vision","slug":"searching-for-efficient-multi-stage-vision","title":"Searching for Efficient Multi-Stage Vision Transformers","date":"2021-09-01","arxiv_id":"2109.00642","n_code_links":1,"syntology":null},{"paper":"/paper/towards-improving-adversarial-training-of-nlp","slug":"towards-improving-adversarial-training-of-nlp","title":"Towards Improving Adversarial Training of NLP Models","date":"2021-09-01","arxiv_id":"2109.00544","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["QData/TextAttack-A2T"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/arat5-text-to-text-transformers-for-arabic","slug":"arat5-text-to-text-transformers-for-arabic","title":"AraT5: Text-to-Text Transformers for Arabic Language Generation","date":"2021-08-31","arxiv_id":"2109.12068","n_code_links":1,"syntology":null},{"paper":null,"slug":"effectiveness-of-deep-networks-in-nlp-using","title":"Effectiveness of Deep Networks in NLP using BiDAF as an example architecture","date":"2021-08-31","arxiv_id":"2109.00074","n_code_links":0,"syntology":null},{"paper":"/paper/enjoy-the-salience-towards-better-transformer","slug":"enjoy-the-salience-towards-better-transformer","title":"Enjoy the Salience: Towards Better Transformer-based Faithful Explanations with Word Salience","date":"2021-08-31","arxiv_id":"2108.13759","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-does-adversarial-fine-tuning-benefit-bert","title":"How Does Adversarial Fine-Tuning Benefit BERT?","date":"2021-08-31","arxiv_id":"2108.13602","n_code_links":0,"syntology":null},{"paper":"/paper/minif2f-a-cross-system-benchmark-for-formal","slug":"minif2f-a-cross-system-benchmark-for-formal","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","date":"2021-08-31","arxiv_id":"2109.00110","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openai/minif2f"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"monolingual-versus-multilingual-bertology-for","title":"Monolingual versus Multilingual BERTology for Vietnamese Extractive Multi-Document Summarization","date":"2021-08-31","arxiv_id":"2108.13741","n_code_links":0,"syntology":null},{"paper":null,"slug":"sense-representations-for-portuguese","title":"Sense representations for Portuguese: experiments with sense embeddings and deep neural language models","date":"2021-08-31","arxiv_id":"2109.00025","n_code_links":0,"syntology":null},{"paper":"/paper/task-oriented-dialogue-system-as-natural","slug":"task-oriented-dialogue-system-as-natural","title":"Task-Oriented Dialogue System as Natural Language Generation","date":"2021-08-31","arxiv_id":"2108.13679","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["victorwz/tod_as_nlg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-and-improving-mobile-level-vision","title":"Exploring and Improving Mobile Level Vision Transformers","date":"2021-08-30","arxiv_id":"2108.13015","n_code_links":0,"syntology":null},{"paper":"/paper/improving-query-representations-for-dense","slug":"improving-query-representations-for-dense","title":"Improving Query Representations for Dense Retrieval with Pseudo Relevance Feedback","date":"2021-08-30","arxiv_id":"2108.13454","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["yuhongqian/ance-prf"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/knowledge-base-completion-meets-transfer","slug":"knowledge-base-completion-meets-transfer","title":"Knowledge Base Completion Meets Transfer Learning","date":"2021-08-30","arxiv_id":"2108.13073","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["vid-koci/kbctransferlearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-multilingual-capabilities-of-very","slug":"on-the-multilingual-capabilities-of-very","title":"On the Multilingual Capabilities of Very Large-Scale English Language Models","date":"2021-08-30","arxiv_id":"2108.13349","n_code_links":2,"syntology":null},{"paper":null,"slug":"shatter-an-efficient-transformer-encoder-with","title":"Shatter: An Efficient Transformer Encoder with Single-Headed Self-Attention and Relative Sequence Partitioning","date":"2021-08-30","arxiv_id":"2108.13032","n_code_links":0,"syntology":null},{"paper":"/paper/want-to-reduce-labeling-cost-gpt-3-can-help","slug":"want-to-reduce-labeling-cost-gpt-3-can-help","title":"Want To Reduce Labeling Cost? GPT-3 Can Help","date":"2021-08-30","arxiv_id":"2108.13487","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-and-mitigating-interference-in","title":"Analyzing and Mitigating Interference in Neural Architecture Search","date":"2021-08-29","arxiv_id":"2108.12821","n_code_links":0,"syntology":null},{"paper":null,"slug":"noier-an-approach-for-training-more-reliable","title":"NoiER: An Approach for Training more Reliable Fine-TunedDownstream Task Models","date":"2021-08-29","arxiv_id":"2110.02054","n_code_links":0,"syntology":null},{"paper":null,"slug":"dkm-differentiable-k-means-clustering-layer","title":"DKM: Differentiable K-Means Clustering Layer for Neural Network Compression","date":"2021-08-28","arxiv_id":"2108.12659","n_code_links":0,"syntology":null},{"paper":"/paper/headlinecause-a-dataset-of-news-headlines-for","slug":"headlinecause-a-dataset-of-news-headlines-for","title":"HeadlineCause: A Dataset of News Headlines for Detecting Causalities","date":"2021-08-28","arxiv_id":"2108.12626","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-text-evaluation-through-the-lens-of","slug":"automatic-text-evaluation-through-the-lens-of","title":"Automatic Text Evaluation through the Lens of Wasserstein Barycenters","date":"2021-08-27","arxiv_id":"2108.12463","n_code_links":2,"syntology":null},{"paper":"/paper/dealing-with-typos-for-bert-based-passage","slug":"dealing-with-typos-for-bert-based-passage","title":"Dealing with Typos for BERT-based Passage Retrieval and Ranking","date":"2021-08-27","arxiv_id":"2108.12139","n_code_links":2,"syntology":null},{"paper":"/paper/evaluating-the-robustness-of-neural-language","slug":"evaluating-the-robustness-of-neural-language","title":"Evaluating the Robustness of Neural Language Models to Input Perturbations","date":"2021-08-27","arxiv_id":"2108.12237","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mmoradi-iut/nlp-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/query-focused-extractive-summarisation-for","slug":"query-focused-extractive-summarisation-for","title":"Query-Focused Extractive Summarisation for Finding Ideal Answers to Biomedical and COVID-19 Questions","date":"2021-08-27","arxiv_id":"2108.12189","n_code_links":1,"syntology":null},{"paper":"/paper/a-computational-approach-to-measure-empathy","slug":"a-computational-approach-to-measure-empathy","title":"A Computational Approach to Measure Empathy and Theory-of-Mind from Written Texts","date":"2021-08-26","arxiv_id":"2108.11810","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-new-sentence-ordering-method-using-bert","title":"A New Sentence Ordering Method Using BERT Pretrained Model","date":"2021-08-26","arxiv_id":"2108.11994","n_code_links":0,"syntology":null},{"paper":"/paper/emoberta-speaker-aware-emotion-recognition-in","slug":"emoberta-speaker-aware-emotion-recognition-in","title":"EmoBERTa: Speaker-Aware Emotion Recognition in Conversation with RoBERTa","date":"2021-08-26","arxiv_id":"2108.12009","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-why-intermediate-task-fine-tuning","slug":"rethinking-why-intermediate-task-fine-tuning","title":"Rethinking Why Intermediate-Task Fine-Tuning Works","date":"2021-08-26","arxiv_id":"2108.11696","n_code_links":1,"syntology":null},{"paper":"/paper/slim-explicit-slot-intent-mapping-with-bert","slug":"slim-explicit-slot-intent-mapping-with-bert","title":"SLIM: Explicit Slot-Intent Mapping with BERT for Joint Multi-Intent Detection and Slot Filling","date":"2021-08-26","arxiv_id":"2108.11711","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-detail-simple-tricks","slug":"the-devil-is-in-the-detail-simple-tricks","title":"The Devil is in the Detail: Simple Tricks Improve Systematic Generalization of Transformers","date":"2021-08-26","arxiv_id":"2108.12284","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["robertcsordas/transformer_generalization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-attention-in-machine-reading","slug":"understanding-attention-in-machine-reading","title":"Multilingual Multi-Aspect Explainability Analyses on Machine Reading Comprehension Models","date":"2021-08-26","arxiv_id":"2108.11574","n_code_links":1,"syntology":null},{"paper":"/paper/models-in-a-spelling-bee-language-models","slug":"models-in-a-spelling-bee-language-models","title":"Models In a Spelling Bee: Language Models Implicitly Learn the Character Composition of Tokens","date":"2021-08-25","arxiv_id":"2108.11193","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["itay1itzhak/spellingbee"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-approximate-nearest-neighbour-selection","slug":"on-approximate-nearest-neighbour-selection","title":"On Approximate Nearest Neighbour Selection for Multi-Stage Dense Retrieval","date":"2021-08-25","arxiv_id":"2108.11480","n_code_links":1,"syntology":null},{"paper":null,"slug":"ontology-enhanced-slot-filling","title":"Ontology-Enhanced Slot Filling","date":"2021-08-25","arxiv_id":"2108.11275","n_code_links":0,"syntology":null},{"paper":"/paper/what-do-pre-trained-code-models-know-about","slug":"what-do-pre-trained-code-models-know-about","title":"What do pre-trained code models know about code?","date":"2021-08-25","arxiv_id":"2108.11308","n_code_links":1,"syntology":null},{"paper":"/paper/sigmoidf1-a-smooth-f1-score-surrogate-loss","slug":"sigmoidf1-a-smooth-f1-score-surrogate-loss","title":"sigmoidF1: A Smooth F1 Score Surrogate Loss for Multilabel Classification","date":"2021-08-24","arxiv_id":"2108.10566","n_code_links":1,"syntology":null},{"paper":"/paper/sn-computer-science-towards-offensive","slug":"sn-computer-science-towards-offensive","title":"Towards Offensive Language Identification for Tamil Code-Mixed YouTube Comments and Posts","date":"2021-08-24","arxiv_id":"2108.10939","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-bert-encoding-and-sentence-level","title":"Using BERT Encoding and Sentence-Level Language Model for Sentence Ordering","date":"2021-08-24","arxiv_id":"2108.10986","n_code_links":0,"syntology":null},{"paper":null,"slug":"weakly-supervised-cross-platform-teenager","title":"Weakly Supervised Cross-platform Teenager Detection with Adversarial BERT","date":"2021-08-24","arxiv_id":"2108.10619","n_code_links":0,"syntology":null},{"paper":null,"slug":"cgems-a-metric-model-for-automatic-code","title":"CGEMs: A Metric Model for Automatic Code Generation using GPT-3","date":"2021-08-23","arxiv_id":"2108.10168","n_code_links":0,"syntology":null},{"paper":null,"slug":"deploying-a-bert-based-query-title-relevance","title":"Deploying a BERT-based Query-Title Relevance Classifier in a Production System: a View from the Trenches","date":"2021-08-23","arxiv_id":"2108.10197","n_code_links":0,"syntology":null},{"paper":"/paper/query-embedding-pruning-for-dense-retrieval","slug":"query-embedding-pruning-for-dense-retrieval","title":"Query Embedding Pruning for Dense Retrieval","date":"2021-08-23","arxiv_id":"2108.10341","n_code_links":1,"syntology":null},{"paper":"/paper/regularizing-transformers-with-deep","slug":"regularizing-transformers-with-deep","title":"Regularizing Transformers With Deep Probabilistic Layers","date":"2021-08-23","arxiv_id":"2108.10764","n_code_links":0,"syntology":null},{"paper":"/paper/sarcasm-detection-in-twitter-performance","slug":"sarcasm-detection-in-twitter-performance","title":"Sarcasm Detection in Twitter -- Performance Impact while using Data Augmentation: Word Embeddings","date":"2021-08-23","arxiv_id":"2108.09924","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-large-pre-trained-models-with-cross","title":"Using Large Pre-Trained Models with Cross-Modal Attention for Multi-Modal Emotion Recognition","date":"2021-08-22","arxiv_id":"2108.09669","n_code_links":0,"syntology":null},{"paper":null,"slug":"uzbert-pretraining-a-bert-model-for-uzbek","title":"UzBERT: pretraining a BERT model for Uzbek","date":"2021-08-22","arxiv_id":"2108.09814","n_code_links":0,"syntology":null},{"paper":null,"slug":"extracting-radiological-findings-with","title":"Extracting Radiological Findings With Normalized Anatomical Information Using a Span-Based BERT Relation Extraction Model","date":"2021-08-20","arxiv_id":"2108.09211","n_code_links":0,"syntology":null},{"paper":"/paper/frozen-pretrained-transformers-for-neural","slug":"frozen-pretrained-transformers-for-neural","title":"Frozen Pretrained Transformers for Neural Sign Language Translation","date":"2021-08-20","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-communication-with-adaptive","title":"Semantic Communication with Adaptive Universal Transformer","date":"2021-08-20","arxiv_id":"2108.09119","n_code_links":0,"syntology":null},{"paper":"/paper/a-framework-for-neural-topic-modeling-of-text","slug":"a-framework-for-neural-topic-modeling-of-text","title":"A Framework for Neural Topic Modeling of Text Corpora","date":"2021-08-19","arxiv_id":"2108.08946","n_code_links":1,"syntology":null},{"paper":null,"slug":"detection-of-illicit-drug-trafficking-events","title":"Detection of Illicit Drug Trafficking Events on Instagram: A Deep Multimodal Multilabel Learning Approach","date":"2021-08-19","arxiv_id":"2108.08920","n_code_links":0,"syntology":null},{"paper":"/paper/fast-passage-re-ranking-with-contextualized","slug":"fast-passage-re-ranking-with-contextualized","title":"Fast Passage Re-ranking with Contextualized Exact Term Matching and Efficient Passage Expansion","date":"2021-08-19","arxiv_id":"2108.08513","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-grained-element-identification-in","title":"Fine-Grained Element Identification in Complaint Text of Internet Fraud","date":"2021-08-19","arxiv_id":"2108.08676","n_code_links":0,"syntology":null},{"paper":"/paper/how-hateful-are-movies-a-study-and-prediction","slug":"how-hateful-are-movies-a-study-and-prediction","title":"How Hateful are Movies? A Study and Prediction on Movie Subtitles","date":"2021-08-19","arxiv_id":"2108.10724","n_code_links":1,"syntology":null},{"paper":"/paper/sentence-t5-scalable-sentence-encoders-from","slug":"sentence-t5-scalable-sentence-encoders-from","title":"Sentence-T5: Scalable Sentence Encoders from Pre-trained Text-to-Text Models","date":"2021-08-19","arxiv_id":"2108.08877","n_code_links":2,"syntology":null},{"paper":"/paper/uniqorn-unified-question-answering-over-rdf","slug":"uniqorn-unified-question-answering-over-rdf","title":"UNIQORN: Unified Question Answering over RDF Knowledge Graphs and Natural Language Text","date":"2021-08-19","arxiv_id":"2108.08614","n_code_links":1,"syntology":null},{"paper":null,"slug":"contributions-of-transformer-attention-heads","title":"Contributions of Transformer Attention Heads in Multi- and Cross-lingual Tasks","date":"2021-08-18","arxiv_id":"2108.08375","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-dialog-history-into-end-to-end","title":"Integrating Dialog History into End-to-End Spoken Language Understanding Systems","date":"2021-08-18","arxiv_id":"2108.08405","n_code_links":0,"syntology":null},{"paper":null,"slug":"sifn-a-sentiment-aware-interactive-fusion","title":"SIFN: A Sentiment-aware Interactive Fusion Network for Review-based Item Recommendation","date":"2021-08-18","arxiv_id":"2108.08022","n_code_links":0,"syntology":null},{"paper":"/paper/table-caption-generation-in-scholarly","slug":"table-caption-generation-in-scholarly","title":"Table Caption Generation in Scholarly Documents Leveraging Pre-trained Language Models","date":"2021-08-18","arxiv_id":"2108.08111","n_code_links":1,"syntology":null},{"paper":null,"slug":"tsi-an-ad-text-strength-indicator-using-text","title":"TSI: an Ad Text Strength Indicator using Text-to-CTR and Semantic-Ad-Similarity","date":"2021-08-18","arxiv_id":"2108.08226","n_code_links":0,"syntology":null},{"paper":null,"slug":"enct5-fine-tuning-t5-encoder-for","title":"EncT5: Fine-tuning T5 Encoder for Discriminative Tasks","date":"2021-08-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/response-ranking-with-multi-types-of-deep","slug":"response-ranking-with-multi-types-of-deep","title":"Response Ranking with Multi-types of Deep Interactive Representations in Retrieval-based Dialogues","date":"2021-08-17","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-natural-language-processing-for-linkedin-1","title":"Deep Natural Language Processing for LinkedIn Search","date":"2021-08-16","arxiv_id":"2108.13300","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-opportunities-and-risks-of-foundation","slug":"on-the-opportunities-and-risks-of-foundation","title":"On the Opportunities and Risks of Foundation Models","date":"2021-08-16","arxiv_id":"2108.07258","n_code_links":2,"syntology":null},{"paper":null,"slug":"maps-search-misspelling-detection-leveraging","title":"Maps Search Misspelling Detection Leveraging Domain-Augmented Contextual Representations","date":"2021-08-15","arxiv_id":"2108.06842","n_code_links":0,"syntology":null},{"paper":null,"slug":"sapphire-approaches-for-enhanced-concept-to","title":"SAPPHIRE: Approaches for Enhanced Concept-to-Text Generation","date":"2021-08-15","arxiv_id":"2108.06643","n_code_links":0,"syntology":null},{"paper":"/paper/curriculum-learning-a-regularization-method","slug":"curriculum-learning-a-regularization-method","title":"The Stability-Efficiency Dilemma: Investigating Sequence Length Warmup for Training GPT Models","date":"2021-08-13","arxiv_id":"2108.06084","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-structured-dynamic-sparse-pre","title":"Towards Structured Dynamic Sparse Pre-Training of BERT","date":"2021-08-13","arxiv_id":"2108.06277","n_code_links":0,"syntology":null},{"paper":"/paper/ammus-a-survey-of-transformer-based","slug":"ammus-a-survey-of-transformer-based","title":"AMMUS : A Survey of Transformer-based Pretrained Models in Natural Language Processing","date":"2021-08-12","arxiv_id":"2108.05542","n_code_links":1,"syntology":null},{"paper":"/paper/how-optimal-is-greedy-decoding-for-extractive","slug":"how-optimal-is-greedy-decoding-for-extractive","title":"How Optimal is Greedy Decoding for Extractive Question Answering?","date":"2021-08-12","arxiv_id":"2108.05857","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ocastel/exact-extract"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modeling-relevance-ranking-under-the-pre","title":"Modeling Relevance Ranking under the Pre-training and Fine-tuning Paradigm","date":"2021-08-12","arxiv_id":"2108.05652","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-analysis-of-the-predictability-of","title":"Multimodal analysis of the predictability of hand-gesture properties","date":"2021-08-12","arxiv_id":"2108.05762","n_code_links":0,"syntology":null},{"paper":null,"slug":"overview-of-the-hasoc-track-at-fire-2020-hate","title":"Overview of the HASOC track at FIRE 2020: Hate Speech and Offensive Content Identification in Indo-European Languages","date":"2021-08-12","arxiv_id":"2108.05927","n_code_links":0,"syntology":null},{"paper":"/paper/patrickstar-parallel-training-of-pre-trained","slug":"patrickstar-parallel-training-of-pre-trained","title":"PatrickStar: Parallel Training of Pre-trained Models via Chunk-based Memory Management","date":"2021-08-12","arxiv_id":"2108.05818","n_code_links":1,"syntology":null},{"paper":null,"slug":"medical-vlbert-medical-visual-language-bert","title":"Medical-VLBERT: Medical Visual Language BERT for COVID-19 CT Report Generation With Alternate Learning","date":"2021-08-11","arxiv_id":"2108.05067","n_code_links":0,"syntology":null},{"paper":null,"slug":"nofake-at-checkthat-2021-fake-news-detection","title":"NoFake at CheckThat! 2021: Fake News Detection Using BERT","date":"2021-08-11","arxiv_id":"2108.05419","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-of-social-and-behavioral-determinants","title":"A Study of Social and Behavioral Determinants of Health in Lung Cancer Patients Using Transformers-based Natural Language Processing Models","date":"2021-08-10","arxiv_id":"2108.04949","n_code_links":0,"syntology":null},{"paper":"/paper/bros-a-layout-aware-pre-trained-language","slug":"bros-a-layout-aware-pre-trained-language","title":"BROS: A Pre-trained Language Model Focusing on Text and Layout for Better Key Information Extraction from Documents","date":"2021-08-10","arxiv_id":"2108.04539","n_code_links":2,"syntology":{"ran":13,"of":13,"n_ran_checked":11,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clovaai/bros"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clsebert-contrastive-learning-for-syntax","title":"SynCoBERT: Syntax-Guided Multi-Modal Contrastive Pre-Training for Code Representation","date":"2021-08-10","arxiv_id":"2108.04556","n_code_links":0,"syntology":null},{"paper":"/paper/embodied-bert-a-transformer-model-for","slug":"embodied-bert-a-transformer-model-for","title":"Embodied BERT: A Transformer Model for Embodied, Language-guided Visual Task Completion","date":"2021-08-10","arxiv_id":"2108.04927","n_code_links":1,"syntology":null}],"record_sha256":"772e2b6034e5bba35f5397656d5eb5e0fad8e02e786f3d61c66c1a37327b041b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}