{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dropout/papers/225","list_of":"/method/dropout","method":"Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":225,"pages_in_order":275,"rows_per_page":100,"rows":[22401,22500],"of":27472,"counts":{"archive_papers_tagged":27472,"with_a_code_link":12129,"where_syntology_ran_a_sample":3620,"not_listed_spam_title":0,"listed":27472,"listed_where_code_ran":3620,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3044,"every_run_a_failure_of_syntologys_instrument":576,"listed_with_a_run_with_no_instrument_failure":3044,"listed_every_run_a_failure_of_syntologys_instrument":576,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dropout","prev":"/method/dropout/papers/224","next":"/method/dropout/papers/226","papers":[{"paper":null,"slug":"on-position-embeddings-in-bert","title":"On Position Embeddings in BERT","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"parameterization-of-hypercomplex","title":"Parameterization of Hypercomplex Multiplications","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/phrasetransformer-self-attention-using-local","slug":"phrasetransformer-self-attention-using-local","title":"PhraseTransformer: Self-Attention using Local Context for Semantic Parsing","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/polyjuice-automated-general-purpose","slug":"polyjuice-automated-general-purpose","title":"Polyjuice: Generating Counterfactuals for Explaining, Evaluating, and Improving Models","date":"2021-01-01","arxiv_id":"2101.00288","n_code_links":1,"syntology":null},{"paper":null,"slug":"post-training-weighted-quantization-of-neural","title":"Post-Training Weighted Quantization of Neural Networks for Language Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-training-text-to-text-transformers-to","title":"Pre-training Text-to-Text Transformers to Write and Reason with Concepts","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-attention-transformer-improving","title":"Predictive Attention Transformer: Improving Transformer with Attention Map Prediction","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/prefix-tuning-optimizing-continuous-prompts","slug":"prefix-tuning-optimizing-continuous-prompts","title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","date":"2021-01-01","arxiv_id":"2101.00190","n_code_links":13,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["XiangLi1999/PrefixTuning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"pretrain-knowledge-aware-language-models","title":"Pretrain Knowledge-Aware Language Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"quantile-regularization-towards-implicit-1","title":"Quantile Regularization : Towards Implicit Calibration of Regression Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-and-bias-in-multilingual-nlp","title":"Representation and Bias in Multilingual NLP: Insights from Controlled Experiments on Conditional Language Modeling","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representational-correlates-of-hierarchical","title":"Representational correlates of hierarchical phrase structure in deep language models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/scene-context-aware-salient-object-detection","slug":"scene-context-aware-salient-object-detection","title":"Scene Context-Aware Salient Object Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sedona-search-for-decoupled-neural-networks","slug":"sedona-search-for-decoupled-neural-networks","title":"SEDONA: Search for Decoupled Neural Networks toward Greedy Block-wise Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"self-born-wiring-for-neural-trees","title":"Self-Born Wiring for Neural Trees","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-relaxed-quantization-with-dropbits-1","title":"Semi-Relaxed Quantization with DropBits: Training Low-Bit Neural Networks via Bitwise Regularization","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"share-or-not-learning-to-schedule-language","title":"Share or Not? Learning to Schedule Language-Specific Capacity for Multilingual Translation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"single-layers-of-attention-suffice-to-predict","title":"Single Layers of Attention Suffice to Predict Protein Contacts","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"skillbert-skilling-the-bert-to-classify","title":"SkillBERT: “Skilling” the BERT to classify skills!","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"speeding-up-deep-learning-training-by-sharing","title":"Speeding up Deep Learning Training by Sharing Weights and Then Unsharing","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/star-a-structure-aware-lightweight","slug":"star-a-structure-aware-lightweight","title":"STAR: A Structure-Aware Lightweight Transformer for Real-Time Image Enhancement","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/stochastic-partial-swap-enhanced-model","slug":"stochastic-partial-swap-enhanced-model","title":"Stochastic Partial Swap: Enhanced Model Generalization and Interpretability for Fine-Grained Recognition","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/subformer-a-parameter-reduced-transformer","slug":"subformer-a-parameter-reduced-transformer","title":"Subformer: A Parameter Reduced Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/subformer-exploring-weight-sharing-for","slug":"subformer-exploring-weight-sharing-for","title":"Subformer: Exploring Weight Sharing for Parameter Efficiency in Generative Transformers","date":"2021-01-01","arxiv_id":"2101.00234","n_code_links":1,"syntology":null},{"paper":null,"slug":"syntactic-relevance-xlnet-word-embedding","title":"Syntactic Relevance XLNet Word Embedding Generation in Low-Resource Machine Translation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/synthesising-realistic-calcium-imaging-data","slug":"synthesising-realistic-calcium-imaging-data","title":"Synthesising Realistic Calcium Imaging Data of Neuronal Populations Using GAN","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"synthesizer-rethinking-self-attention-for","title":"Synthesizer: Rethinking Self-Attention for Transformer Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"taking-notes-on-the-fly-helps-language-pre","title":"Taking Notes on the Fly Helps Language Pre-Training","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"task-agnostic-and-adaptive-size-bert","title":"Task-Agnostic and Adaptive-Size BERT Compression","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-practical-second-order-optimization","slug":"towards-practical-second-order-optimization","title":"Towards Practical Second Order Optimization for Deep Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-and-improving-dropout","title":"Towards Understanding and Improving Dropout in Game Theory","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"trans-caps-transformer-capsule-networks-with","title":"Trans-Caps: Transformer Capsule Networks with Self-attention Routing","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-protein-language-models-are","title":"Transformer protein language models are unsupervised structure learners","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-ql-a-step-towards-making","title":"Transformer-QL: A Step Towards Making Transformer Network Quadratically Large","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformers-satisfy","title":"Transformers satisfy","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transforming-recurrent-neural-networks-with","title":"Transforming Recurrent Neural Networks with Attention and Fixed-point Equations","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/trar-routing-the-attention-spans-in","slug":"trar-routing-the-attention-spans-in","title":"TRAR: Routing the Attention Spans in Transformer for Visual Question Answering","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"u-bert-pre-training-user-representations-for","title":"U-BERT: Pre-training User Representations for Improved Recommendation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"updet-universal-multi-agent-rl-via-policy","title":"UPDeT: Universal Multi-agent RL via Policy Decoupling with Transformers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"userbert-self-supervised-user-representation","title":"UserBERT: Self-supervised User Representation Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-transformers-where-do-transformers","title":"Visual Transformers: Where Do Transformers Really Belong in Vision Models?","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/visualsparta-sparse-transformer-fragment","slug":"visualsparta-sparse-transformer-fragment","title":"VisualSparta: An Embarrassingly Simple Approach to Large-scale Text-to-Image Search with Weighted Bag-of-words","date":"2021-01-01","arxiv_id":"2101.00265","n_code_links":1,"syntology":null},{"paper":"/paper/warp-word-level-adversarial-reprogramming","slug":"warp-word-level-adversarial-reprogramming","title":"WARP: Word-level Adversarial ReProgramming","date":"2021-01-01","arxiv_id":"2101.00121","n_code_links":1,"syntology":null},{"paper":null,"slug":"wb-detr-transformer-based-detector-without","title":"WB-DETR: Transformer-Based Detector Without Backbone","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-closer-look-at-few-shot-crosslingual","title":"A Closer Look at Few-Shot Crosslingual Transfer: The Choice of Shots Matters","date":"2020-12-31","arxiv_id":"2012.15682","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multi-modal-deep-learning-model-for-video","title":"A Multi-modal Deep Learning Model for Video Thumbnail Selection","date":"2020-12-31","arxiv_id":"2101.00073","n_code_links":0,"syntology":null},{"paper":"/paper/better-robustness-by-more-coverage","slug":"better-robustness-by-more-coverage","title":"Better Robustness by More Coverage: Adversarial Training with Mixup Augmentation for Robust Fine-tuning","date":"2020-12-31","arxiv_id":"2012.15699","n_code_links":1,"syntology":null},{"paper":"/paper/binarybert-pushing-the-limit-of-bert","slug":"binarybert-pushing-the-limit-of-bert","title":"BinaryBERT: Pushing the Limit of BERT Quantization","date":"2020-12-31","arxiv_id":"2012.15701","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["huawei-noah/Pretrained-Language-Model"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/cocolm-complex-commonsense-enhanced-language","slug":"cocolm-complex-commonsense-enhanced-language","title":"CoCoLM: COmplex COmmonsense Enhanced Language Model with Discourse Relations","date":"2020-12-31","arxiv_id":"2012.15643","n_code_links":1,"syntology":null},{"paper":null,"slug":"conditional-generation-of-temporally-ordered","title":"Conditional Generation of Temporally-ordered Event Sequences","date":"2020-12-31","arxiv_id":"2012.15786","n_code_links":0,"syntology":null},{"paper":"/paper/directed-beam-search-plug-and-play-lexically","slug":"directed-beam-search-plug-and-play-lexically","title":"Directed Beam Search: Plug-and-Play Lexically Constrained Language Generation","date":"2020-12-31","arxiv_id":"2012.15416","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dapascual/DirectedBeamSearch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/earlybert-efficient-bert-training-via-early-1","slug":"earlybert-efficient-bert-training-via-early-1","title":"EarlyBERT: Efficient BERT Training via Early-bird Lottery Tickets","date":"2020-12-31","arxiv_id":"2101.00063","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VITA-Group/EarlyBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"estimating-uncertainty-in-neural-networks-for","title":"Estimating Uncertainty in Neural Networks for Cardiac MRI Segmentation: A Benchmark Study","date":"2020-12-31","arxiv_id":"2012.15772","n_code_links":0,"syntology":null},{"paper":"/paper/factual-error-correction-of-claims","slug":"factual-error-correction-of-claims","title":"Evidence-based Factual Error Correction","date":"2020-12-31","arxiv_id":"2012.15788","n_code_links":3,"syntology":null},{"paper":"/paper/fully-non-autoregressive-neural-machine","slug":"fully-non-autoregressive-neural-machine","title":"Fully Non-autoregressive Neural Machine Translation: Tricks of the Trade","date":"2020-12-31","arxiv_id":"2012.15833","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-networks-with-spectral-message-passing","title":"Graph Networks with Spectral Message Passing","date":"2020-12-31","arxiv_id":"2101.00079","n_code_links":0,"syntology":null},{"paper":"/paper/kart-privacy-leakage-framework-of-language","slug":"kart-privacy-leakage-framework-of-language","title":"KART: Parameterization of Privacy Leakage Scenarios from Pre-trained Language Models","date":"2020-12-31","arxiv_id":"2101.00036","n_code_links":1,"syntology":null},{"paper":"/paper/linear-time-wordpiece-tokenization","slug":"linear-time-wordpiece-tokenization","title":"Fast WordPiece Tokenization","date":"2020-12-31","arxiv_id":"2012.15524","n_code_links":1,"syntology":null},{"paper":"/paper/making-pre-trained-language-models-better-few","slug":"making-pre-trained-language-models-better-few","title":"Making Pre-trained Language Models Better Few-shot Learners","date":"2020-12-31","arxiv_id":"2012.15723","n_code_links":9,"syntology":{"ran":2,"of":9,"n_ran_checked":2,"n_instrument":0,"unverified":7,"pointer_only":7,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["princeton-nlp/LM-BFF"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/minilmv2-multi-head-self-attention-relation","slug":"minilmv2-multi-head-self-attention-relation","title":"MiniLMv2: Multi-Head Self-Attention Relation Distillation for Compressing Pretrained Transformers","date":"2020-12-31","arxiv_id":"2012.15828","n_code_links":2,"syntology":null},{"paper":null,"slug":"revisiting-robust-neural-machine-translation","title":"Revisiting Robust Neural Machine Translation: A Transformer Case Study","date":"2020-12-31","arxiv_id":"2012.15710","n_code_links":0,"syntology":null},{"paper":null,"slug":"studying-strategically-learning-to-mask-for","title":"Studying Strategically: Learning to Mask for Closed-book QA","date":"2020-12-31","arxiv_id":"2012.15856","n_code_links":0,"syntology":null},{"paper":"/paper/the-pile-an-800gb-dataset-of-diverse-text-for","slug":"the-pile-an-800gb-dataset-of-diverse-text-for","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","date":"2020-12-31","arxiv_id":"2101.00027","n_code_links":22,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["EleutherAI/The-Pile"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transtrack-multiple-object-tracking-with","slug":"transtrack-multiple-object-tracking-with","title":"TransTrack: Multiple Object Tracking with Transformer","date":"2020-12-31","arxiv_id":"2012.15460","n_code_links":2,"syntology":null},{"paper":"/paper/unified-mandarin-tts-front-end-based-on","slug":"unified-mandarin-tts-front-end-based-on","title":"Unified Mandarin TTS Front-end Based on Distilled BERT Model","date":"2020-12-31","arxiv_id":"2012.15404","n_code_links":1,"syntology":null},{"paper":"/paper/unks-everywhere-adapting-multilingual","slug":"unks-everywhere-adapting-multilingual","title":"UNKs Everywhere: Adapting Multilingual Language Models to New Scripts","date":"2020-12-31","arxiv_id":"2012.15562","n_code_links":2,"syntology":null},{"paper":null,"slug":"verb-knowledge-injection-for-multilingual","title":"Verb Knowledge Injection for Multilingual Event Processing","date":"2020-12-31","arxiv_id":"2012.15421","n_code_links":0,"syntology":null},{"paper":null,"slug":"xlm-t-scaling-up-multilingual-machine","title":"XLM-T: Scaling up Multilingual Machine Translation with Pretrained Cross-lingual Transformer Encoders","date":"2020-12-31","arxiv_id":"2012.15547","n_code_links":0,"syntology":null},{"paper":"/paper/deer-a-data-efficient-language-model-for","slug":"deer-a-data-efficient-language-model-for","title":"ECONET: Effective Continual Pretraining of Language Models for Event Temporal Reasoning","date":"2020-12-30","arxiv_id":"2012.15283","n_code_links":2,"syntology":null},{"paper":null,"slug":"deriving-contextualised-semantic-features","title":"Deriving Contextualised Semantic Features from BERT (and Other Transformer Model) Embeddings","date":"2020-12-30","arxiv_id":"2012.15353","n_code_links":0,"syntology":null},{"paper":"/paper/improving-bert-with-syntax-aware-local","slug":"improving-bert-with-syntax-aware-local","title":"Improving BERT with Syntax-aware Local Attention","date":"2020-12-30","arxiv_id":"2012.15150","n_code_links":1,"syntology":null},{"paper":"/paper/mri-brain-tumor-segmentation-and-uncertainty","slug":"mri-brain-tumor-segmentation-and-uncertainty","title":"MRI brain tumor segmentation and uncertainty estimation using 3D-UNet architectures","date":"2020-12-30","arxiv_id":"2012.15294","n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-deeper-transformers-on-small","slug":"optimizing-deeper-transformers-on-small","title":"Optimizing Deeper Transformers on Small Datasets","date":"2020-12-30","arxiv_id":"2012.15355","n_code_links":1,"syntology":null},{"paper":null,"slug":"out-of-order-how-important-is-the-sequential","title":"Out of Order: How Important Is The Sequential Order of Words in a Sentence in Natural Language Understanding Tasks?","date":"2020-12-30","arxiv_id":"2012.15180","n_code_links":0,"syntology":null},{"paper":"/paper/semglove-semantic-co-occurrences-for-glove","slug":"semglove-semantic-co-occurrences-for-glove","title":"SemGloVe: Semantic Co-occurrences for GloVe from BERT","date":"2020-12-30","arxiv_id":"2012.15197","n_code_links":3,"syntology":null},{"paper":null,"slug":"skinet-a-deep-learning-solution-for-skin","title":"SkiNet: A Deep Learning Solution for Skin Lesion Diagnosis with Uncertainty Estimation and Explainability","date":"2020-12-30","arxiv_id":"2012.15049","n_code_links":0,"syntology":null},{"paper":"/paper/towards-unsupervised-deep-image-enhancement","slug":"towards-unsupervised-deep-image-enhancement","title":"Towards Unsupervised Deep Image Enhancement with Generative Adversarial Network","date":"2020-12-30","arxiv_id":"2012.15020","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-for-image-quality-assessment","title":"Transformer for Image Quality Assessment","date":"2020-12-30","arxiv_id":"2101.01097","n_code_links":0,"syntology":null},{"paper":"/paper/unnatural-language-inference","slug":"unnatural-language-inference","title":"UnNatural Language Inference","date":"2020-12-30","arxiv_id":"2101.00010","n_code_links":1,"syntology":null},{"paper":"/paper/a-hierarchical-transformer-with-speaker","slug":"a-hierarchical-transformer-with-speaker","title":"A Hierarchical Transformer with Speaker Modeling for Emotion Recognition in Conversation","date":"2020-12-29","arxiv_id":"2012.14781","n_code_links":1,"syntology":null},{"paper":"/paper/kaleidoscope-an-efficient-learnable-1","slug":"kaleidoscope-an-efficient-learnable-1","title":"Kaleidoscope: An Efficient, Learnable Representation For All Structured Linear Maps","date":"2020-12-29","arxiv_id":"2012.14966","n_code_links":2,"syntology":{"ran":16,"of":23,"n_ran_checked":9,"n_instrument":7,"unverified":7,"pointer_only":0,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","official":{"repos":["HazyResearch/butterfly","HazyResearch/learning-circuits"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","slug":"layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","arxiv_id":"2012.14740","n_code_links":9,"syntology":null},{"paper":"/paper/robust-dialogue-utterance-rewriting-as","slug":"robust-dialogue-utterance-rewriting-as","title":"Robust Dialogue Utterance Rewriting as Sequence Tagging","date":"2020-12-29","arxiv_id":"2012.14535","n_code_links":1,"syntology":null},{"paper":"/paper/sit3-code-summarization-with-structure","slug":"sit3-code-summarization-with-structure","title":"Code Summarization with Structure-induced Transformer","date":"2020-12-29","arxiv_id":"2012.14710","n_code_links":1,"syntology":null},{"paper":"/paper/a-paragraph-level-multi-task-learning-model","slug":"a-paragraph-level-multi-task-learning-model","title":"A Paragraph-level Multi-task Learning Model for Scientific Fact-Verification","date":"2020-12-28","arxiv_id":"2012.14500","n_code_links":1,"syntology":null},{"paper":null,"slug":"burt-bert-inspired-universal-representation-1","title":"BURT: BERT-inspired Universal Representation from Learning Meaningful Segment","date":"2020-12-28","arxiv_id":"2012.14320","n_code_links":0,"syntology":null},{"paper":"/paper/lattice-free-mmi-adaptation-of-self","slug":"lattice-free-mmi-adaptation-of-self","title":"Lattice-Free MMI Adaptation Of Self-Supervised Pretrained Acoustic Models","date":"2020-12-28","arxiv_id":"2012.14252","n_code_links":2,"syntology":null},{"paper":"/paper/red-dragon-ai-at-textgraphs-2020-shared-task","slug":"red-dragon-ai-at-textgraphs-2020-shared-task","title":"Red Dragon AI at TextGraphs 2020 Shared Task: LIT : LSTM-Interleaved Transformer for Multi-Hop Explanation Ranking","date":"2020-12-28","arxiv_id":"2012.14164","n_code_links":1,"syntology":null},{"paper":"/paper/syntax-enhanced-pre-trained-model","slug":"syntax-enhanced-pre-trained-model","title":"Syntax-Enhanced Pre-trained Model","date":"2020-12-28","arxiv_id":"2012.14116","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-a-category-extended-object-detector","title":"Towards a category-extended object detector with limited data","date":"2020-12-28","arxiv_id":"2012.14115","n_code_links":0,"syntology":null},{"paper":"/paper/transpose-towards-explainable-human-pose","slug":"transpose-towards-explainable-human-pose","title":"TransPose: Keypoint Localization via Transformer","date":"2020-12-28","arxiv_id":"2012.14214","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-task-learning-network-using-shared","title":"A multi-task learning network using shared BERT models for aspect-based sentiment analysis","date":"2020-12-27","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"alp-kd-attention-based-layer-projection-for","title":"ALP-KD: Attention-Based Layer Projection for Knowledge Distillation","date":"2020-12-27","arxiv_id":"2012.14022","n_code_links":0,"syntology":null},{"paper":"/paper/an-embarrassingly-simple-model-for-dialogue","slug":"an-embarrassingly-simple-model-for-dialogue","title":"An Embarrassingly Simple Model for Dialogue Relation Extraction","date":"2020-12-27","arxiv_id":"2012.13873","n_code_links":1,"syntology":null},{"paper":"/paper/inserting-information-bottlenecks-for","slug":"inserting-information-bottlenecks-for","title":"Inserting Information Bottlenecks for Attribution in Transformers","date":"2020-12-27","arxiv_id":"2012.13838","n_code_links":1,"syntology":null},{"paper":"/paper/learning-light-weight-translation-models-from","slug":"learning-light-weight-translation-models-from","title":"Learning Light-Weight Translation Models from Deep Transformer","date":"2020-12-27","arxiv_id":"2012.13866","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["libeineu/GPKD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/medal-medical-abbreviation-disambiguation","slug":"medal-medical-abbreviation-disambiguation","title":"MeDAL: Medical Abbreviation Disambiguation Dataset for Natural Language Understanding Pretraining","date":"2020-12-27","arxiv_id":"2012.13978","n_code_links":1,"syntology":null},{"paper":null,"slug":"portfolio-optimization-with-2d-relative","title":"Portfolio Optimization with 2D Relative-Attentional Gated Transformer","date":"2020-12-27","arxiv_id":"2101.03138","n_code_links":0,"syntology":null},{"paper":null,"slug":"sg-net-syntax-guided-transformer-for-language","title":"SG-Net: Syntax Guided Transformer for Language Representation","date":"2020-12-27","arxiv_id":"2012.13915","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-inductive-learner-for-graph","title":"Bayesian Graph Neural Network for Fast identification of critical nodes in Uncertain Complex Networks","date":"2020-12-26","arxiv_id":"2012.15733","n_code_links":0,"syntology":null}],"record_sha256":"0085cbd3ab474282df7facfc4c4022ca2c2e94376af2aeacf5202c0380041735","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}