{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/78","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":78,"pages_in_order":109,"rows_per_page":100,"rows":[7701,7800],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/77","next":"/method/attention-dropout/papers/79","papers":[{"paper":null,"slug":"token-pooling-in-visual-transformers","title":"Token Pooling in Vision Transformers","date":"2021-10-08","arxiv_id":"2110.03860","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-transformer-based","title":"A Comparative Study of Transformer-Based Language Models on Extractive Question Answering","date":"2021-10-07","arxiv_id":"2110.03142","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-is-all-you-need-good-embeddings","title":"Attention is All You Need? Good Embeddings with Statistics are enough:Large Scale Audio Understanding without Transformers/ Convolutions/ BERTs/ Mixers/ Attention/ RNNs or ....","date":"2021-10-07","arxiv_id":"2110.03183","n_code_links":0,"syntology":null},{"paper":"/paper/cross-language-learning-for-entity-matching","slug":"cross-language-learning-for-entity-matching","title":"Cross-Language Learning for Entity Matching","date":"2021-10-07","arxiv_id":"2110.03338","n_code_links":1,"syntology":null},{"paper":null,"slug":"universality-of-deep-neural-network-lottery","title":"Universality of Winning Tickets: A Renormalization Group Perspective","date":"2021-10-07","arxiv_id":"2110.03210","n_code_links":0,"syntology":null},{"paper":"/paper/8-bit-optimizers-via-block-wise-quantization","slug":"8-bit-optimizers-via-block-wise-quantization","title":"8-bit Optimizers via Block-wise Quantization","date":"2021-10-06","arxiv_id":"2110.02861","n_code_links":3,"syntology":null},{"paper":"/paper/nus-ids-at-fincausal-2021-dependency-tree-in","slug":"nus-ids-at-fincausal-2021-dependency-tree-in","title":"NUS-IDS at FinCausal 2021: Dependency Tree in Graph Neural Network for Better Cause-Effect Span Detection","date":"2021-10-06","arxiv_id":"2110.02991","n_code_links":1,"syntology":null},{"paper":"/paper/ponet-pooling-network-for-efficient-token","slug":"ponet-pooling-network-for-efficient-token","title":"PoNet: Pooling Network for Efficient Token Mixing in Long Sequences","date":"2021-10-06","arxiv_id":"2110.02442","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lxchtan/ponet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/psg-hasoc-dravidian-codemixfire2021","slug":"psg-hasoc-dravidian-codemixfire2021","title":"Pretrained Transformers for Offensive Language Identification in Tanglish","date":"2021-10-06","arxiv_id":"2110.02852","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-the-impact-of-covid-19-on-economy","title":"Analyzing the Impact of COVID-19 on Economy from the Perspective of Users Reviews","date":"2021-10-05","arxiv_id":"2110.02198","n_code_links":0,"syntology":null},{"paper":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","n_code_links":0,"syntology":null},{"paper":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","n_code_links":1,"syntology":null},{"paper":"/paper/distilhubert-speech-representation-learning","slug":"distilhubert-speech-representation-learning","title":"DistilHuBERT: Speech Representation Learning by Layer-wise Distillation of Hidden-unit BERT","date":"2021-10-05","arxiv_id":"2110.01900","n_code_links":1,"syntology":null},{"paper":"/paper/foodchem-a-food-chemical-relation-extraction","slug":"foodchem-a-food-chemical-relation-extraction","title":"FoodChem: A food-chemical relation extraction model","date":"2021-10-05","arxiv_id":"2110.02019","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-sense-specific-static-embeddings","title":"Learning Sense-Specific Static Embeddings using Contextualised Word Embeddings as a Proxy","date":"2021-10-05","arxiv_id":"2110.02204","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-the-inductive-bias-of-large","title":"Leveraging the Inductive Bias of Large Language Models for Abstract Textual Reasoning","date":"2021-10-05","arxiv_id":"2110.02370","n_code_links":0,"syntology":null},{"paper":"/paper/mobilevit-light-weight-general-purpose-and","slug":"mobilevit-light-weight-general-purpose-and","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","date":"2021-10-05","arxiv_id":"2110.02178","n_code_links":31,"syntology":{"ran":53,"of":68,"n_ran_checked":44,"n_instrument":9,"unverified":15,"pointer_only":18,"phrase":"53 ran (of which 31 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 9 where Syntology's instrument failed) · 15 unverified","official":{"repos":["apple/ml-cvnets"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"ur-iw-hnt-at-germeval-2021-an-ensembling","title":"ur-iw-hnt at GermEval 2021: An Ensembling Strategy with Multiple BERT Models","date":"2021-10-05","arxiv_id":"2110.02042","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":"/paper/deepa2-a-modular-framework-for-deep-argument","slug":"deepa2-a-modular-framework-for-deep-argument","title":"DeepA2: A Modular Framework for Deep Argument Analysis with Pretrained Neural Text2Text Language Models","date":"2021-10-04","arxiv_id":"2110.01509","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","n_code_links":0,"syntology":null},{"paper":"/paper/juribert-a-masked-language-model-adaptation","slug":"juribert-a-masked-language-model-adaptation","title":"JuriBERT: A Masked-Language Model Adaptation for French Legal Text","date":"2021-10-04","arxiv_id":"2110.01485","n_code_links":1,"syntology":null},{"paper":null,"slug":"perhaps-ptlms-should-go-to-school-a-task-to","title":"Perhaps PTLMs Should Go to School -- A Task to Assess Open Book and Closed Book QA","date":"2021-10-04","arxiv_id":"2110.01552","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-examples-generation-for-reducing","title":"Adversarial Examples Generation for Reducing Implicit Gender Bias in Pre-trained Models","date":"2021-10-03","arxiv_id":"2110.01094","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-paradigm-for-information","title":"Unsupervised paradigm for information extraction from transcripts using BERT","date":"2021-10-03","arxiv_id":"2110.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-for-sustainable","title":"Artificial intelligence for Sustainable Energy: A Contextual Topic Modeling and Content Analysis","date":"2021-10-02","arxiv_id":"2110.00828","n_code_links":0,"syntology":null},{"paper":"/paper/swiss-judgment-prediction-a-multilingual","slug":"swiss-judgment-prediction-a-multilingual","title":"Swiss-Judgment-Prediction: A Multilingual Legal Judgment Prediction Benchmark","date":"2021-10-02","arxiv_id":"2110.00806","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert4gcn-using-bert-intermediate-layers-to","title":"BERT4GCN: Using BERT Intermediate Layers to Augment GCN for Aspect-based Sentiment Classification","date":"2021-10-01","arxiv_id":"2110.00171","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"span-labeling-approach-for-vietnamese-and","title":"Span Labeling Approach for Vietnamese and Chinese Word Segmentation","date":"2021-10-01","arxiv_id":"2110.00156","n_code_links":0,"syntology":null},{"paper":null,"slug":"unpacking-the-interdependent-systems-of","title":"Unpacking the Interdependent Systems of Discrimination: Ableist Bias in NLP Systems through an Intersectional Lens","date":"2021-10-01","arxiv_id":"2110.00521","n_code_links":0,"syntology":null},{"paper":"/paper/bert-got-a-date-introducing-transformers-to","slug":"bert-got-a-date-introducing-transformers-to","title":"BERT got a Date: Introducing Transformers to Temporal Tagging","date":"2021-09-30","arxiv_id":"2109.14927","n_code_links":1,"syntology":null},{"paper":"/paper/covid-19-fake-news-detection-using","slug":"covid-19-fake-news-detection-using","title":"COVID-19 Fake News Detection Using Bidirectional Encoder Representations from Transformers Based Models","date":"2021-09-30","arxiv_id":"2109.14816","n_code_links":1,"syntology":null},{"paper":"/paper/prose2poem-the-blessing-of-transformer-based","slug":"prose2poem-the-blessing-of-transformer-based","title":"Prose2Poem: The Blessing of Transformers in Translating Prose to Persian Poetry","date":"2021-09-30","arxiv_id":"2109.14934","n_code_links":1,"syntology":null},{"paper":null,"slug":"are-bert-families-zero-shot-learners-a-study","title":"Are BERT Families Zero-Shot Learners? A Study on Their Potential and Limitations","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"collaborative-storytelling-with-human-actors","title":"Collaborative Storytelling with Human Actors and AI Narrators","date":"2021-09-29","arxiv_id":"2109.14728","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-transformer-based-sequence-to","title":"Compressing Transformer-Based Sequence to Sequence Models With Pre-trained Autoencoders for Text Summarization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-pre-training-for-zero-shot","title":"Contrastive Pre-training for Zero-Shot Information Retrieval","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-architecture-distillation-using","title":"Cross-Architecture Distillation Using Bidirectional CMOW Embeddings","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-packing-towards-2x-nlp-speed-up","title":"Efficient Packing: Towards 2x NLP Speed-Up without Loss of Accuracy for BERT","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-models-through-the-lens-of-stable","title":"Embedding models through the lens of Stable Coloring","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-sparse-robust-efficient-transformer","title":"ERNIE-SPARSE: Robust Efficient Transformer Through Hierarchically Unifying Isolated Information","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-broadcast-adaptation-defending","title":"Gradient Broadcast Adaptation: Defending against the backdoor attack in pre-trained models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/group-based-interleaved-pipeline-parallelism","slug":"group-based-interleaved-pipeline-parallelism","title":"Group-based Interleaved Pipeline Parallelism for Large-scale DNN Training","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-character-tagger-for-short-text","title":"Hierarchical Character Tagger for Short Text Spelling Error Correction","date":"2021-09-29","arxiv_id":"2109.14259","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-bert-address-polysemy-of-korean","title":"How does BERT address polysemy of Korean adverbial postpositions -ey, -eyse, and -(u)lo?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"illiterate-dall-cdot-e-learns-to-compose","title":"Illiterate DALL$\\cdot$E Learns to Compose","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-sentiment-classification-using-0","title":"Improving Sentiment Classification Using 0-Shot Generated Labels for Custom Transformer Embeddings","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"in-defense-of-dual-encoders-for-neural","title":"In defense of dual-encoders for neural ranking","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"language-model-pre-training-improves","title":"Language Model Pre-training Improves Generalization in Policy Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-rate-grafting-transferability-of","title":"Learning Rate Grafting: Transferability of Optimizer Tuning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-visual-linguistic-adequacy-fidelity","title":"Learning Visual-Linguistic Adequacy, Fidelity, and Fluency for Novel Object Captioning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/mait-integrating-spatial-locality-into-image","slug":"mait-integrating-spatial-locality-into-image","title":"MaiT: integrating spatial locality into image transformers with attention masks","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"mapping-language-models-to-grounded","title":"Mapping Language Models to Grounded Conceptual Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"modeling-label-correlations-implicitly","title":"Modeling label correlations implicitly through latent label encodings for multi-label text classification","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-for-large","title":"Offline Reinforcement Learning for Large Scale Language Action Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rank4class-examining-multiclass","title":"Rank4Class: Examining Multiclass Classification through the Lens of Learning to Rank","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robot-intent-recognition-method-based-on","title":"Robot Intent Recognition Method Based on State Grid Business Office","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scala-speeding-up-fine-tuning-of-pre-trained","title":"ScaLA: Speeding-Up Fine-tuning of Pre-trained Transformer Networks via Efficient and Scalable Adversarial Perturbation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scale-efficiently-insights-from-pretraining","title":"Scale Efficiently: Insights from Pretraining and Finetuning Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"seqpate-differentially-private-text","title":"SeqPATE: Differentially Private Text Generation via Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"specialized-transformers-faster-smaller-and","title":"Specialized Transformers: Faster, Smaller and more Accurate NLP Models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-sequence-labeling-models-using-prior","title":"Training sequence labeling models using prior knowledge","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transslowdown-efficiency-attacks-on-neural","title":"TransSlowDown: Efficiency Attacks on Neural Machine Translation Systems","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-different-text-preprocessing-techniques","title":"How Different Text-preprocessing Techniques Using The BERT Model Affect The Gender Profiling of Authors","date":"2021-09-28","arxiv_id":"2109.13890","n_code_links":0,"syntology":null},{"paper":"/paper/raft-a-real-world-few-shot-text","slug":"raft-a-real-world-few-shot-text","title":"RAFT: A Real-World Few-Shot Text Classification Benchmark","date":"2021-09-28","arxiv_id":"2109.14076","n_code_links":1,"syntology":null},{"paper":"/paper/effective-use-of-graph-convolution-network","slug":"effective-use-of-graph-convolution-network","title":"Effective Use of Graph Convolution Network and Contextual Sub-Tree forCommodity News Event Extraction","date":"2021-09-27","arxiv_id":"2109.12781","n_code_links":1,"syntology":null},{"paper":"/paper/patterns-of-lexical-ambiguity-in","slug":"patterns-of-lexical-ambiguity-in","title":"Patterns of Lexical Ambiguity in Contextualised Language Models","date":"2021-09-27","arxiv_id":"2109.13032","n_code_links":0,"syntology":null},{"paper":"/paper/turingbench-a-benchmark-environment-for","slug":"turingbench-a-benchmark-environment-for","title":"TURINGBENCH: A Benchmark Environment for Turing Test in the Age of Neural Text Generation","date":"2021-09-27","arxiv_id":"2109.13296","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/understanding-and-overcoming-the-challenges","slug":"understanding-and-overcoming-the-challenges","title":"Understanding and Overcoming the Challenges of Efficient Transformer Quantization","date":"2021-09-27","arxiv_id":"2109.12948","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qualcomm-ai-research/transformer-quantization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-question-answering-performance","slug":"improving-question-answering-performance","title":"Improving Question Answering Performance Using Knowledge Distillation and Active Learning","date":"2021-09-26","arxiv_id":"2109.12662","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mirbostani/QA-KD-AL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-prunability-of-attention-heads-in","title":"On the Prunability of Attention Heads in Multilingual BERT","date":"2021-09-26","arxiv_id":"2109.12683","n_code_links":0,"syntology":null},{"paper":null,"slug":"finetuning-transformer-models-to-build-asag","title":"Finetuning Transformer Models to Build ASAG System","date":"2021-09-25","arxiv_id":"2109.12300","n_code_links":0,"syntology":null},{"paper":null,"slug":"aes-are-both-overstable-and-oversensitive","title":"AES Systems Are Both Overstable And Oversensitive: Explaining Why And Proposing Defenses","date":"2021-09-24","arxiv_id":"2109.11728","n_code_links":0,"syntology":null},{"paper":null,"slug":"dense-contrastive-visual-linguistic","title":"Dense Contrastive Visual-Linguistic Pretraining","date":"2021-09-24","arxiv_id":"2109.11778","n_code_links":0,"syntology":null},{"paper":null,"slug":"lacking-the-embedding-of-a-word-look-it-up","title":"Lacking the embedding of a word? Look it up into a traditional dictionary","date":"2021-09-24","arxiv_id":"2109.11763","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-and-sensitivity-of-bert-models","title":"Robustness and Sensitivity of BERT Models Predicting Alzheimer's Disease from Text","date":"2021-09-24","arxiv_id":"2109.11888","n_code_links":0,"syntology":null},{"paper":"/paper/breaking-bert-understanding-its","slug":"breaking-bert-understanding-its","title":"Breaking BERT: Understanding its Vulnerabilities for Named Entity Recognition through Adversarial Attack","date":"2021-09-23","arxiv_id":"2109.11308","n_code_links":1,"syntology":null},{"paper":"/paper/putting-words-in-bert-s-mouth-navigating","slug":"putting-words-in-bert-s-mouth-navigating","title":"Putting Words in BERT's Mouth: Navigating Contextualized Vector Spaces with Pseudowords","date":"2021-09-23","arxiv_id":"2109.11491","n_code_links":1,"syntology":null},{"paper":null,"slug":"alzheimers-dementia-detection-using-acoustic","title":"Alzheimers Dementia Detection using Acoustic & Linguistic features and Pre-Trained BERT","date":"2021-09-22","arxiv_id":"2109.11010","n_code_links":0,"syntology":null},{"paper":null,"slug":"dialoguebert-a-self-supervised-learning-based","title":"DialogueBERT: A Self-Supervised Learning based Dialogue Pre-training Encoder","date":"2021-09-22","arxiv_id":"2109.10480","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-as-recommender-systems","title":"Language Models as Recommender Systems: Evaluations and Limitations","date":"2021-09-22","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-efficiency-effectiveness-trade","title":"Predicting Efficiency/Effectiveness Trade-offs for Dense vs. Sparse Retrieval Strategy Selection","date":"2021-09-22","arxiv_id":"2109.10739","n_code_links":0,"syntology":null},{"paper":null,"slug":"recursively-summarizing-books-with-human","title":"Recursively Summarizing Books with Human Feedback","date":"2021-09-22","arxiv_id":"2109.10862","n_code_links":0,"syntology":null},{"paper":"/paper/scale-efficiently-insights-from-pre-training","slug":"scale-efficiently-insights-from-pre-training","title":"Scale Efficiently: Insights from Pre-training and Fine-tuning Transformers","date":"2021-09-22","arxiv_id":"2109.10686","n_code_links":3,"syntology":{"ran":12,"of":12,"n_ran_checked":12,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 4 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/unsupervised-contextualized-document","slug":"unsupervised-contextualized-document","title":"Unsupervised Contextualized Document Representation","date":"2021-09-22","arxiv_id":"2109.10509","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-review-on-summarizing","title":"A Comprehensive Review on Summarizing Financial News Using Deep Learning","date":"2021-09-21","arxiv_id":"2109.10118","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertweetfr-domain-adaptation-of-pre-trained","title":"BERTweetFR : Domain Adaptation of Pre-Trained Language Models for French Tweets","date":"2021-09-21","arxiv_id":"2109.10234","n_code_links":0,"syntology":null},{"paper":null,"slug":"invbert-text-reconstruction-from","title":"InvBERT: Reconstructing Text from Contextualized Word Embeddings by inverting the BERT pipeline","date":"2021-09-21","arxiv_id":"2109.10104","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-learning-with-sentiment-emotion","title":"Multi-Task Learning with Sentiment, Emotion, and Target Detection to Recognize Hate Speech and Offensive Language","date":"2021-09-21","arxiv_id":"2109.10255","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-short-text","title":"Representation Learning for Short Text Clustering","date":"2021-09-21","arxiv_id":"2109.09894","n_code_links":0,"syntology":null},{"paper":"/paper/a-plug-and-play-method-for-controlled-text","slug":"a-plug-and-play-method-for-controlled-text","title":"A Plug-and-Play Method for Controlled Text Generation","date":"2021-09-20","arxiv_id":"2109.09707","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-cannot-align-characters","title":"BERT Cannot Align Characters","date":"2021-09-20","arxiv_id":"2109.09700","n_code_links":0,"syntology":null},{"paper":"/paper/bert-has-uncommon-sense-similarity-ranking","slug":"bert-has-uncommon-sense-similarity-ranking","title":"BERT Has Uncommon Sense: Similarity Ranking for Word Sense BERTology","date":"2021-09-20","arxiv_id":"2109.09780","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-bias-in-nlp-application-to-hate-speech","title":"Model Bias in NLP -- Application to Hate Speech Classification using transfer learning techniques","date":"2021-09-20","arxiv_id":"2109.09725","n_code_links":0,"syntology":null},{"paper":"/paper/mirrorwic-on-eliciting-word-in-context","slug":"mirrorwic-on-eliciting-word-in-context","title":"MirrorWiC: On Eliciting Word-in-Context Representations from Pretrained Language Models","date":"2021-09-19","arxiv_id":"2109.09237","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-zero-label-language-learning","title":"Towards Zero-Label Language Learning","date":"2021-09-19","arxiv_id":"2109.09193","n_code_links":0,"syntology":null},{"paper":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-bert-based-language-models-learn-in","title":"What BERT Based Language Models Learn in Spoken Transcripts: An Empirical Study","date":"2021-09-19","arxiv_id":"2109.09105","n_code_links":0,"syntology":null}],"record_sha256":"34623d62e0845e885651bbcd28c06ec5abe94a157a2d339ce9e9722ec3ad1b19","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}