{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dropout/papers/194","list_of":"/method/dropout","method":"Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":194,"pages_in_order":275,"rows_per_page":100,"rows":[19301,19400],"of":27472,"counts":{"archive_papers_tagged":27472,"with_a_code_link":12129,"where_syntology_ran_a_sample":3620,"not_listed_spam_title":0,"listed":27472,"listed_where_code_ran":3620,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3044,"every_run_a_failure_of_syntologys_instrument":576,"listed_with_a_run_with_no_instrument_failure":3044,"listed_every_run_a_failure_of_syntologys_instrument":576,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dropout","prev":"/method/dropout/papers/193","next":"/method/dropout/papers/195","papers":[{"paper":null,"slug":"elle-efficient-lifelong-pre-training-for","title":"ELLE: Efficient Lifelong Pre-training for Emerging Data","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-analysis-of-training-strategies-of-1","title":"Empirical Analysis of Training Strategies of Transformer-based Japanese Chit-chat Systems","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-sign-language-translation-via","title":"End-To-End Sign Language Translation via Multitask Learning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-task-oriented-dialog-policy","title":"End-to-end Task-oriented Dialog Policy Learning based on Pre-trained Language Model","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-the-nonlinear-mutual-dependencies","title":"Enhancing the Nonlinear Mutual Dependencies in Transformers with Mutual Information","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-sparse-learning-hierarchical-efficient","title":"ERNIE-SPARSE: Learning Hierarchical Efficient Transformer Through Regularized Self-Attention","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"event-detection-via-derangement-question","title":"Event Detection via Derangement Question Answering","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"eventbert","title":"EventBERT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"explicit-modeling-the-context-for-chinese-ner","title":"Explicit Modeling the Context for Chinese NER","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-and-adapting-chinese-gpt-to-pinyin","title":"Exploring and Adapting Chinese GPT to Pinyin Input Method","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"extreme-multi-label-text-classification-with-1","title":"Extreme Multi-label Text Classification with Multi-layer Experts","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"eye-gaze-and-self-attention-how-humans-and","title":"Eye Gaze and Self-attention: How Humans and Transformers Attend Words in Sentences","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-and-accurate-transformer-based","title":"Fast and Accurate Transformer-based Translation with Character-Level Encoding and Subword-Level Decoding","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-rich-open-vocabulary-interpretable","title":"Feature-rich Open-vocabulary Interpretable Neural Representations for All of the World’s 7000 Languages","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-structure-distillation-for-bert","title":"Feature Structure Distillation for BERT Transferring","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fix-bugs-with-transformer-through-a-neural","title":"Fix Bugs with Transformer through a Neural-Symbolic Edit Grammar","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"frsum-towards-faithful-abstractive","title":"FRSUM: Towards Faithful Abstractive Summarization via Enhancing Factual Robustness","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gabert-an-irish-language-model-1","title":"gaBERT — an Irish Language Model","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-diverse-and-high-quality","title":"Generating Diverse and High-Quality Abstractive Summaries with Variational Transformers","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generation-of-news-articles-from-tweets-an","title":"Generation of News Articles from Tweets : An Experiment","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-pre-trained-transformer-for-design","title":"Generative Pre-Trained Transformer for Design Concept Generation: An Exploration","date":"2021-11-16","arxiv_id":"2111.08489","n_code_links":0,"syntology":null},{"paper":null,"slug":"get-the-point-graph-enhanced-candidate","title":"Get the Point! Graph Enhanced Candidate Retrieval for Zero-shot Entity Linking","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"glm-general-language-model-pretraining-with","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-based-fine-grained-multimodal-attention","title":"Graph-based Fine-grained Multimodal Attention Mechanism for Sentiment Analysis","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-the-pre-training-objective-affect","title":"How does the pre-training objective affect what large language models learn about linguistic properties?","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-tokenization-on-language-models-an","title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-compositional-generalization-with-1","title":"Improving Compositional Generalization with Self-Training for Data-to-Text Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-gpt-3-after-deployment-with-a","title":"Improving GPT-3 after deployment with a dynamic memory of feedback","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-neural-models-for-radiology-report","title":"Improving Neural Models for Radiology Report Retrieval with Lexicon-based Automated Annotation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-unsupervised-sentence","title":"Improving Unsupervised Sentence Simplification Using Fine-Tuned Masked Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"input-specific-attention-subnetworks-for","title":"Input-specific Attention Subnetworks for Adversarial Detection","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/integrated-semantic-and-phonetic-post-1","slug":"integrated-semantic-and-phonetic-post-1","title":"Integrated Semantic and Phonetic Post-correction for Chinese Speech Recognition","date":"2021-11-16","arxiv_id":"2111.08400","n_code_links":1,"syntology":null},{"paper":"/paper/interpreting-language-models-through","slug":"interpreting-language-models-through","title":"Interpreting Language Models Through Knowledge Graph Extraction","date":"2021-11-16","arxiv_id":"2111.08546","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpreting-the-robustness-of-neural-nlp","title":"Interpreting the Robustness of Neural NLP Models to Textual Perturbations","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-use-of-bert-anchors-for","title":"Investigating the Use of BERT Anchors for Bilingual Lexicon Induction with Minimal Supervision","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"is-neural-topic-modelling-better-than","title":"Is Neural Topic Modelling Better than Clustering? An Empirical Study on Clustering with Contextual Embeddings for Topics","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"is-whole-word-masking-always-better-for","title":"\"Is Whole Word Masking Always Better for Chinese BERT?\": Probing on Chinese Grammatical Error Correction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kinyabert-a-morphology-aware-kinyarwanda","title":"KinyaBERT: a Morphology-aware Kinyarwanda Language Model","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-enhanced-embedding-improve-model","title":"Knowledge Enhanced Embedding: Improve Model Generalization Through Knowledge Graphs","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-is-in-rescue-task-oriented","title":"Knowledge Graph is in Rescue: Task Oriented Dialogue System for Response Generation without NLU and DM","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-guided-transformer-for-joint-theme","title":"Knowledge-guided Transformer for Joint Theme and Emotion Classification of Chinese Classical Poetry","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"language-level-classification-on-german-texts","title":"Language Level Classification on German Texts using a Neural Approach","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-more-from-less-improving-conversational","title":"Learn More from Less: Improving Conversational Recommender Systems via Contextual and Time-Aware Modeling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-methods-for-solving-astronomy-course","title":"Learning Methods for Solving Astronomy Course Problems","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-non-autoregressive-models-from","title":"Learning Non-Autoregressive Models from Search for Unsupervised Sentence Summarization","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-ignore-adversarial-attacks","title":"Learning to Ignore Adversarial Attacks","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"life-after-bert-what-do-other-muppets","title":"Life after BERT: What do Other Muppets Understand about Language?","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"listen-to-both-sides-and-be-enlightened","title":"Listen to Both Sides and be Enlightened! -- Hierarchical Modality Fusion Network for Entity and Relation Extraction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"looking-into-the-black-box-how-are-idioms","title":"Looking Into the Black Box - How Are Idioms Processed in BERT?","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lordbert-embedding-long-text-by-segment","title":"LordBERT: Embedding Long Text by Segment Ordering with BERT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"making-transformers-solve-compositional-tasks-1","title":"Making Transformers Solve Compositional Tasks","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"marcqap-effective-context-modeling-for","title":"MarCQAp: Effective Context Modeling for Conversational Question Answering","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/markbert-marking-word-boundaries-improves","slug":"markbert-marking-word-boundaries-improves","title":"MarkBERT: Marking Word Boundaries Improves Chinese BERT","date":"2021-11-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"mderank-a-masked-document-embedding-rank-1","title":"MDERank: A Masked Document Embedding Rank Approach for Unsupervised Keyphrase Extraction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/meeting-summarization-with-pre-training-and","slug":"meeting-summarization-with-pre-training-and","title":"Meeting Summarization with Pre-training and Clustering Methods","date":"2021-11-16","arxiv_id":"2111.08210","n_code_links":1,"syntology":null},{"paper":null,"slug":"metadata-shaping-natural-language-annotations-1","title":"Metadata Shaping: Natural Language Annotations for the Long Tail","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/modeling-hierarchical-syntax-structure-with","slug":"modeling-hierarchical-syntax-structure-with","title":"Modeling Hierarchical Syntax Structure with Triplet Position for Source Code Summarization","date":"2021-11-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"moving-the-eiffel-tower-to-rome-tracing-and","title":"Moving the Eiffel Tower to ROME: Tracing and Editing Facts in GPT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-head-or-single-head-an-empirical-1","title":"Multi-head or Single-head? An Empirical Comparison for Transformer Training","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"n-grammer-augmenting-transformers-with-latent","title":"N-grammer: Augmenting Transformers with latent n-grams","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-keyphrase-generation-analysis-and","title":"Neural Keyphrase Generation: Analysis and Evaluation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"nsp-bert-a-prompt-based-zero-shot-learner-1","title":"NSP-BERT: A Prompt-based Zero-Shot Learner Through an Original Pre-training Task —— Next Sentence Prediction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-multilingual-capabilities-of-very-1","title":"On the Multilingual Capabilities of Very Large-Scale English Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-robustness-of-reading-comprehension-1","title":"On the Robustness of Reading Comprehension Models to Entity Renaming","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-vision-features-in-multimodal-machine","title":"On Vision Features in Multimodal Machine Translation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-advertising-revenue-forecasting-an","title":"Online Advertising Revenue Forecasting: An Interpretable Deep Learning Approach","date":"2021-11-16","arxiv_id":"2111.08840","n_code_links":0,"syntology":null},{"paper":null,"slug":"pare-a-simple-and-strong-baseline-for","title":"PARE: A Simple and Strong Baseline for Monolingual and Multilingual Distantly Supervised Relation Extraction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"perturbations-in-the-wild-leveraging-human","title":"Perturbations in the Wild: Leveraging Human-Written Text Perturbations for Realistic Adversarial Attack and Defense","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pesto-a-post-user-fusion-network-for-rumour","title":"PESTO: A Post-User Fusion Network for Rumour Detection on Social Media","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pinyin-bert-a-new-solution-to-chinese-pinyin","title":"Pinyin-bert: A new solution to Chinese pinyin to character conversion task","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"polise-reinforcing-politeness-using-user","title":"PoliSe: Reinforcing Politeness using User Sentiment for Customer Care Response Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-berts-priors-with-serial-reproduction","title":"Probing BERT’s priors with serial reproduction chains","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"promptbert-improving-bert-sentence-embeddings","title":"PromptBERT: Improving BERT Sentence Embeddings with Prompts","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoning-like-program-executors","title":"Reasoning Like Program Executors","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reco-reliable-multi-hop-causal-reasoning-via","title":"ReCo: Reliable Multi-hop Causal Reasoning via Structural Causal Recurrent Unit","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-of-ambiguity-in-pre-trained","title":"Representation of Ambiguity in Pre-Trained Sentence Embeddings","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-based-layer-wise-adaptive","title":"Retrieval-based Layer-wise Adaptive Transformer for Source Code Summarization","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-softmax-for-uncertainty","title":"Revisiting Softmax for Uncertainty Approximation in Text Classification","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-of-bayesian-neural-networks-to-1","title":"Robustness of Bayesian Neural Networks to White-Box Adversarial Attacks","date":"2021-11-16","arxiv_id":"2111.08591","n_code_links":0,"syntology":null},{"paper":null,"slug":"sambert-improve-aspect-sentiment-triplet","title":"SAMBERT: Improve Aspect Sentiment Triplet Extraction by Segmenting the Attention Maps of BERT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-contrastive-learning-with-1","title":"Self-Supervised Contrastive Learning with Adversarial Perturbations for Robust Pretrained Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sequence-to-sequence-amr-parsing-with","title":"Sequence-to-sequence AMR Parsing with Ancestor Information","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sequence-to-sequence-knowledge-graph","title":"Sequence-to-Sequence Knowledge Graph Completion and Question Answering","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shct-a-successively-hierarchical-conditional","title":"SHCT: A Successively Hierarchical Conditional Transformer for Controllable Paraphrase Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shield-defending-textual-neural-networks","title":"SHIELD: Defending Textual Neural Networks against Black-Box Adversarial Attacks with Stochastic Multi-Expert Patcher","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shrinknas-single-path-one-shot-operator","title":"ShrinkNAS : Single-Path One-Shot Operator Exploratory Training for Transformer with Dynamic Space Shrinking","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-bottleneck-makes-language-models","title":"Softmax Bottleneck Makes Language Models Unable to Represent Multi-mode Word Distributions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-probability-and-statistics-problems","title":"Solving Probability and Statistics Problems by Program Synthesis","date":"2021-11-16","arxiv_id":"2111.08267","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-probability-and-statistics-problems-1","title":"Solving Probability and Statistics Problems by Program Synthesis","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/sparsifying-transformer-models-with-trainable","slug":"sparsifying-transformer-models-with-trainable","title":"Sparsifying Transformer Models with Trainable Representation Pooling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"speaker-profiling-in-multi-party","title":"Speaker Profiling in Multi-party Conversations","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-pruning-learns-compact-and","title":"Structured Pruning Learns Compact and Accurate Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"supershaper-task-agnostic-super-pre-training-1","title":"SuperShaper: Task-Agnostic Super Pre-training of BERT Models with Variable Hidden Dimensions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"taco-pre-training-of-deep-transformers-with","title":"TACO: Pre-training of Deep Transformers with Attention Convolution using Disentangled Positional Representation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-bert-to-wait-balancing-accuracy-and","title":"Teaching BERT to Wait: Balancing Accuracy and Latency for Streaming Disfluency Detection","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tell-me-who-you-are-and-i-ll-tell-you-what-to","title":"Tell me who you are and i'll tell you what to do: A Persona Grounded Task Oriented Dialogue Generation System","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-lexical-and-grammatical","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource-1","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-coding-social-science-datasets-with","title":"Towards Coding Social Science Datasets with Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-fully-self-supervised-learning-of","title":"Towards Fully Self-Supervised Learning of Knowledge from Unstructured Text","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"91d35a9ff9d330df20cb6e9de564192180acd284f9dc696d70c4c3223edbbb55","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}