{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/61","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":61,"pages_in_order":71,"rows_per_page":100,"rows":[6001,6100],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/60","next":"/method/linear-warmup-with-linear-decay/papers/62","papers":[{"paper":null,"slug":"contextual-and-non-contextual-word-embeddings","title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"copybert-a-unified-approach-to-question","title":"CopyBERT: A Unified Approach to Question Generation with Self-Attention","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cross-lingual-disaster-related-multi-label","slug":"cross-lingual-disaster-related-multi-label","title":"Cross-Lingual Disaster-related Multi-label Tweet Classification with Manifold Mixup","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-sarcasm-in-conversation-context","title":"Detecting Sarcasm in Conversation Context Using Transformer-Based Models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-utility-of-model","title":"Evaluating the Utility of Model Configurations and Data Augmentation on Clinical Semantic Textual Similarity","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-limits-of-simple-learners-in","title":"Exploring the Limits of Simple Learners in Knowledge Distillation for Document Classification with DocBERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-projection-for-improved-text","title":"Feature Projection for Improved Text Classification","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/gan-bert-generative-adversarial-learning-for","slug":"gan-bert-generative-adversarial-learning-for","title":"GAN-BERT: Generative Adversarial Learning for Robust Text Classification with a Bunch of Labeled Examples","date":"2020-07-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"getting-the-life-out-of-living-how-adequate","title":"Getting the \\#\\#life out of living: How Adequate Are Word-Pieces for Modelling Complex Morphology?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"go-figure-multi-task-transformer-based","title":"Go Figure! Multi-task transformer-based architecture for metaphor detection using idioms: ETS team in 2020 metaphor shared task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"go-wide-then-narrow-efficient-training-of","title":"Go Wide, Then Narrow: Efficient Training of Deep Thin Networks","date":"2020-07-01","arxiv_id":"2007.00811","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-bert-s-attention-change-when-you","title":"How does BERT's attention change when you fine-tune? An analysis methodology and a case study in negation scope","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"illinimet-illinois-system-for-metaphor","title":"IlliniMet: Illinois System for Metaphor Detection with Contextual and Linguistic Information","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-multimodal-named-entity-recognition","slug":"improving-multimodal-named-entity-recognition","title":"Improving Multimodal Named Entity Recognition via Entity Span Detection with Unified Multimodal Transformer","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"information-retrieval-and-extraction-on-covid","title":"Information Retrieval and Extraction on COVID-19 Clinical Articles Using Graph Community Detection and Bio-BERT Embeddings","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"intermediate-task-transfer-learning-with-1","title":"Intermediate-Task Transfer Learning with Pretrained Language Models: When and Why Does It Work?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-effect-of-auxiliary","title":"Investigating the effect of auxiliary objectives for the automated grading of learner English speech transcriptions","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"item-based-collaborative-filtering-with-bert","title":"Item-based Collaborative Filtering with BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-training-with-semantic-role-labeling","title":"Joint Training with Semantic Role Labeling for Better Generalization in Natural Language Inference","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"k-opsala-transition-based-graph-parsing-via","title":"K\\opsala: Transition-Based Graph Parsing via Efficient Training and Effective Encoding","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"metaphor-detection-using-contextual-word","title":"Metaphor Detection Using Contextual Word Embeddings From Transformers","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/modelling-context-and-syntactical-features","slug":"modelling-context-and-syntactical-features","title":"Modelling Context and Syntactical Features for Aspect-based Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-sarcasm-detection-using-conversation","title":"Neural Sarcasm Detection using Conversation Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-higher-order-dependency-parsers","title":"Revisiting Higher-Order Dependency Parsers","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robertnlp-at-the-iwpt-2020-shared-task","title":"RobertNLP at the IWPT 2020 Shared Task: Surprisingly Simple Enhanced UD Parsing for English","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/roles-and-utilization-of-attention-heads-in","slug":"roles-and-utilization-of-attention-heads-in","title":"Roles and Utilization of Attention Heads in Transformer-based Neural Language Models","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"sarcasm-identification-and-detection-in","title":"Sarcasm Identification and Detection in Conversion Context using BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-context-aware-covid-19","slug":"self-supervised-context-aware-covid-19","title":"Self-supervised context-aware COVID-19 document exploration through atlas grounding","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"sentitel-tabsa-for-twitter-reviews-on-uganda","title":"SentiTel: TABSA for Twitter reviews on Uganda Telecoms","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"should-you-fine-tune-bert-for-automated-essay","title":"Should You Fine-Tune BERT for Automated Essay Scoring?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tbert-topic-models-and-bert-joining-forces","slug":"tbert-topic-models-and-bert-joining-forces","title":"tBERT: Topic Models and BERT Joining Forces for Semantic Similarity Detection","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"the-hw-tsc-video-speech-translation-system-at","title":"The HW-TSC Video Speech Translation System at IWSLT 2020","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformers-on-sarcasm-detection-with","title":"Transformers on Sarcasm Detection with Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-semantic-dependency-parsing-1","slug":"transition-based-semantic-dependency-parsing-1","title":"Transition-based Semantic Dependency Parsing with Pointer Networks","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"turku-enhanced-parser-pipeline-from-raw-text","title":"Turku Enhanced Parser Pipeline: From Raw Text to Enhanced Graphs in the IWPT 2020 Shared Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-advertisements-with-bert","title":"Understanding Advertisements with BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-faq-retrieval-with-question","title":"Unsupervised FAQ Retrieval with Question Generation and BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"why-is-penguin-more-similar-to-polar-bear","title":"Why is penguin more similar to polar bear than to sea gull? Analyzing conceptual knowledge in distributional models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"would-you-rather-a-new-benchmark-for-learning","title":"Would you Rather? A New Benchmark for Learning Machine Alignment with Cultural Values and Social Preferences","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/data-movement-is-all-you-need-a-case-study-of","slug":"data-movement-is-all-you-need-a-case-study-of","title":"Data Movement Is All You Need: A Case Study on Optimizing Transformers","date":"2020-06-30","arxiv_id":"2007.00072","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["spcl/substation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"se3m-a-model-for-software-effort-estimation","title":"SE3M: A Model for Software Effort Estimation Using Pre-trained Embedding Models","date":"2020-06-30","arxiv_id":"2006.16831","n_code_links":0,"syntology":null},{"paper":null,"slug":"segmentation-approach-for-coreference","title":"Segmentation Approach for Coreference Resolution Task","date":"2020-06-30","arxiv_id":"2007.04301","n_code_links":0,"syntology":null},{"paper":"/paper/improving-sequence-tagging-for-vietnamese","slug":"improving-sequence-tagging-for-vietnamese","title":"Improving Sequence Tagging for Vietnamese Text Using Transformer-based Neural Models","date":"2020-06-29","arxiv_id":"2006.15994","n_code_links":2,"syntology":null},{"paper":null,"slug":"interpreting-hierarchical-linguistic","title":"Building Interpretable Interaction Trees for Deep NLP Models","date":"2020-06-29","arxiv_id":"2007.04298","n_code_links":0,"syntology":null},{"paper":null,"slug":"want-to-identify-extract-and-normalize","title":"Want to Identify, Extract and Normalize Adverse Drug Reactions in Tweets? Use RoBERTa","date":"2020-06-29","arxiv_id":"2006.16146","n_code_links":0,"syntology":null},{"paper":"/paper/bond-bert-assisted-open-domain-named-entity","slug":"bond-bert-assisted-open-domain-named-entity","title":"BOND: BERT-Assisted Open-Domain Named Entity Recognition with Distant Supervision","date":"2020-06-28","arxiv_id":"2006.15509","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cliang1453/BOND"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-the-positional-encoding-in","slug":"rethinking-the-positional-encoding-in","title":"Rethinking Positional Encoding in Language Pre-training","date":"2020-06-28","arxiv_id":"2006.15595","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["guolinke/TUPE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/fastspec-scalable-generation-and-detection-of","slug":"fastspec-scalable-generation-and-detection-of","title":"FastSpec: Scalable Generation and Detection of Spectre Gadgets Using Neural Embeddings","date":"2020-06-25","arxiv_id":"2006.14147","n_code_links":1,"syntology":null},{"paper":"/paper/lsbert-a-simple-framework-for-lexical","slug":"lsbert-a-simple-framework-for-lexical","title":"LSBert: A Simple Framework for Lexical Simplification","date":"2020-06-25","arxiv_id":"2006.14939","n_code_links":1,"syntology":null},{"paper":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","n_code_links":0,"syntology":null},{"paper":"/paper/accelerated-large-batch-optimization-of-bert","slug":"accelerated-large-batch-optimization-of-bert","title":"Accelerated Large Batch Optimization of BERT Pretraining in 54 minutes","date":"2020-06-24","arxiv_id":"2006.13484","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-constituency-parsing-by-pointing-1","title":"Efficient Constituency Parsing by Pointing","date":"2020-06-24","arxiv_id":"2006.13557","n_code_links":0,"syntology":null},{"paper":"/paper/reco-a-large-scale-chinese-reading","slug":"reco-a-large-scale-chinese-reading","title":"ReCO: A Large Scale Chinese Reading Comprehension Dataset on Opinion","date":"2020-06-22","arxiv_id":"2006.12146","n_code_links":1,"syntology":null},{"paper":"/paper/students-need-more-attention-bert-based","slug":"students-need-more-attention-bert-based","title":"Students Need More Attention: BERT-based AttentionModel for Small Data with Application to AutomaticPatient Message Triage","date":"2020-06-22","arxiv_id":"2006.11991","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-learning-rates-with-maximum","slug":"adaptive-learning-rates-with-maximum","title":"MaxVA: Fast Adaptation of Step Sizes by Maximizing Observed Variance of Gradients","date":"2020-06-21","arxiv_id":"2006.11918","n_code_links":1,"syntology":null},{"paper":null,"slug":"sarcasm-detection-in-tweets-with-bert-and-1","title":"Sarcasm Detection in Tweets with BERT and GloVe Embeddings","date":"2020-06-20","arxiv_id":"2006.11512","n_code_links":0,"syntology":null},{"paper":"/paper/a-qualitative-evaluation-of-language-models","slug":"a-qualitative-evaluation-of-language-models","title":"A Qualitative Evaluation of Language Models on Automatic Question-Answering for COVID-19","date":"2020-06-19","arxiv_id":"2006.10964","n_code_links":1,"syntology":null},{"paper":"/paper/new-vietnamese-corpus-for-machine","slug":"new-vietnamese-corpus-for-machine","title":"New Vietnamese Corpus for Machine Reading Comprehension of Health News Articles","date":"2020-06-19","arxiv_id":"2006.11138","n_code_links":0,"syntology":null},{"paper":"/paper/squeezebert-what-can-computer-vision-teach","slug":"squeezebert-what-can-computer-vision-teach","title":"SqueezeBERT: What can computer vision teach NLP about efficient neural networks?","date":"2020-06-19","arxiv_id":"2006.11316","n_code_links":6,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huggingface/transformers"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"exploring-the-bert-cross-lingual","title":"Exploring the BERT Cross-Lingual Transferability: a Case Study in Reading Comprehension","date":"2020-06-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tagging-and-parsing-of-multidomain","slug":"tagging-and-parsing-of-multidomain","title":"Tagging and parsing of multidomain collections","date":"2020-06-17","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"end-to-end-code-switching-language-models-for","title":"End-to-End Code Switching Language Models for Automatic Speech Recognition","date":"2020-06-16","arxiv_id":"2006.08870","n_code_links":0,"syntology":null},{"paper":"/paper/improving-accuracy-and-speeding-up-document","slug":"improving-accuracy-and-speeding-up-document","title":"Improving accuracy and speeding up Document Image Classification through parallel systems","date":"2020-06-16","arxiv_id":"2006.09141","n_code_links":1,"syntology":null},{"paper":"/paper/memory-efficient-pipeline-parallel-dnn","slug":"memory-efficient-pipeline-parallel-dnn","title":"Memory-Efficient Pipeline-Parallel DNN Training","date":"2020-06-16","arxiv_id":"2006.09503","n_code_links":1,"syntology":{"ran":2,"of":10,"n_ran_checked":2,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":"/paper/perl-pivot-based-domain-adaptation-for-pre","slug":"perl-pivot-based-domain-adaptation-for-pre","title":"PERL: Pivot-based Domain Adaptation for Pre-trained Deep Contextualized Embedding Models","date":"2020-06-16","arxiv_id":"2006.09075","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalable-cross-lingual-pivots-to-model","title":"Scalable Cross Lingual Pivots to Model Pronoun Gender for Translation","date":"2020-06-16","arxiv_id":"2006.08881","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-sppd-system-for-schema-guided-dialogue","title":"The SPPD System for Schema Guided Dialogue State Tracking Challenge","date":"2020-06-16","arxiv_id":"2006.09035","n_code_links":0,"syntology":null},{"paper":null,"slug":"cooking-is-all-about-people-comment","title":"Cooking Is All About People: Comment Classification On Cookery Channels Using BERT and Classification Models (Malayalam-English Mix-Code)","date":"2020-06-15","arxiv_id":"2007.04249","n_code_links":0,"syntology":null},{"paper":"/paper/document-classification-for-covid-19","slug":"document-classification-for-covid-19","title":"Document Classification for COVID-19 Literature","date":"2020-06-15","arxiv_id":"2006.13816","n_code_links":1,"syntology":null},{"paper":"/paper/finbert-a-pretrained-language-model-for","slug":"finbert-a-pretrained-language-model-for","title":"FinBERT: A Pretrained Language Model for Financial Communications","date":"2020-06-15","arxiv_id":"2006.08097","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yya518/FinBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"finest-bert-and-crosloengual-bert-less-is","title":"FinEst BERT and CroSloEngual BERT: less is more in multilingual models","date":"2020-06-14","arxiv_id":"2006.07890","n_code_links":0,"syntology":null},{"paper":null,"slug":"transferring-monolingual-model-to-low","title":"Transferring Monolingual Model to Low-Resource Language: The Case of Tigrinya","date":"2020-06-13","arxiv_id":"2006.07698","n_code_links":0,"syntology":null},{"paper":"/paper/a-monolingual-approach-to-contextualized-word","slug":"a-monolingual-approach-to-contextualized-word","title":"A Monolingual Approach to Contextualized Word Embeddings for Mid-Resource Languages","date":"2020-06-11","arxiv_id":"2006.06202","n_code_links":0,"syntology":null},{"paper":"/paper/mc-bert-efficient-language-pre-training-via-a","slug":"mc-bert-efficient-language-pre-training-via-a","title":"MC-BERT: Efficient Language Pre-Training via a Meta Controller","date":"2020-06-10","arxiv_id":"2006.05744","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["MC-BERT/MC-BERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/revisiting-few-sample-bert-fine-tuning","slug":"revisiting-few-sample-bert-fine-tuning","title":"Revisiting Few-sample BERT Fine-tuning","date":"2020-06-10","arxiv_id":"2006.05987","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-stability-of-fine-tuning-bert","slug":"on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","arxiv_id":"2006.04884","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":9,"n_instrument":4,"unverified":7,"pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["uds-lsv/bert-stable-fine-tuning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"medical-concept-normalization-in-user","title":"Medical Concept Normalization in User Generated Texts by Learning Target Concept Embeddings","date":"2020-06-07","arxiv_id":"2006.04014","n_code_links":0,"syntology":null},{"paper":"/paper/pre-training-polish-transformer-based","slug":"pre-training-polish-transformer-based","title":"Pre-training Polish Transformer-based Language Models at Scale","date":"2020-06-07","arxiv_id":"2006.04229","n_code_links":1,"syntology":null},{"paper":"/paper/accelerating-natural-language-understanding","slug":"accelerating-natural-language-understanding","title":"Accelerating Natural Language Understanding in Task-Oriented Dialog","date":"2020-06-05","arxiv_id":"2006.03701","n_code_links":1,"syntology":null},{"paper":"/paper/deberta-decoding-enhanced-bert-with","slug":"deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","arxiv_id":"2006.03654","n_code_links":14,"syntology":{"ran":4,"of":13,"n_ran_checked":3,"n_instrument":1,"unverified":9,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["microsoft/DeBERTa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper","unlocated"]}}},{"paper":null,"slug":"udpipe-at-evalatin-2020-contextualized-1","title":"UDPipe at EvaLatin 2020: Contextualized Embeddings and Treebank Embeddings","date":"2020-06-05","arxiv_id":"2006.03687","n_code_links":0,"syntology":null},{"paper":"/paper/the-sofc-exp-corpus-and-neural-approaches-to","slug":"the-sofc-exp-corpus-and-neural-approaches-to","title":"The SOFC-Exp Corpus and Neural Approaches to Information Extraction in the Materials Science Domain","date":"2020-06-04","arxiv_id":"2006.03039","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-text-summarization-of-covid-19","slug":"automatic-text-summarization-of-covid-19","title":"Automatic Text Summarization of COVID-19 Medical Research Articles using BERT and GPT-2","date":"2020-06-03","arxiv_id":"2006.01997","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-pairwise-probe-for-understanding-bert-fine","title":"A Pairwise Probe for Understanding BERT Fine-Tuning on Machine Reading Comprehension","date":"2020-06-02","arxiv_id":"2006.01346","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-multilingual-machine-comprehension","slug":"bert-based-multilingual-machine-comprehension","title":"BERT Based Multilingual Machine Comprehension in English and Hindi","date":"2020-06-02","arxiv_id":"2006.01432","n_code_links":2,"syntology":null},{"paper":"/paper/exploring-cross-sentence-contexts-for-named","slug":"exploring-cross-sentence-contexts-for-named","title":"Exploring Cross-sentence Contexts for Named Entity Recognition with BERT","date":"2020-06-02","arxiv_id":"2006.01563","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jouniluoma/bert-ner-cmv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"position-masking-for-language-models","title":"Position Masking for Language Models","date":"2020-06-02","arxiv_id":"2006.05676","n_code_links":0,"syntology":null},{"paper":"/paper/question-answering-on-scholarly-knowledge","slug":"question-answering-on-scholarly-knowledge","title":"Question Answering on Scholarly Knowledge Graphs","date":"2020-06-02","arxiv_id":"2006.01527","n_code_links":0,"syntology":null},{"paper":null,"slug":"wikibert-models-deep-transfer-learning-for","title":"WikiBERT models: deep transfer learning for many languages","date":"2020-06-02","arxiv_id":"2006.01538","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-ensembles-for-modeling-disclosure","slug":"bert-based-ensembles-for-modeling-disclosure","title":"BERT-based Ensembles for Modeling Disclosure and Support in Conversational Social Media Text","date":"2020-06-01","arxiv_id":"2006.01222","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-machine-comprehension-a","title":"Conversational Machine Comprehension: a Literature Review","date":"2020-06-01","arxiv_id":"2006.00671","n_code_links":0,"syntology":null},{"paper":"/paper/emergence-of-separable-manifolds-in-deep","slug":"emergence-of-separable-manifolds-in-deep","title":"Emergence of Separable Manifolds in Deep Language Representations","date":"2020-06-01","arxiv_id":"2006.01095","n_code_links":1,"syntology":null},{"paper":null,"slug":"etude-des-variations-s-emantiques-a-travers","title":"\\'Etude des variations s\\'emantiques \\`a travers plusieurs dimensions (Studying semantic variations through several dimensions )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"introduction-d-informations-s-emantiques-dans","title":"Introduction d'informations s\\'emantiques dans un syst\\`eme de reconnaissance de la parole (Despite spectacular advances in recent years, the Automatic Speech Recognition (ASR) systems still make mistakes, especially in noisy environments)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"les-mod-eles-de-langue-contextuels-camembert","title":"Les mod\\`eles de langue contextuels Camembert pour le fran\\ccais : impact de la taille et de l'h\\'et\\'erog\\'en\\'eit\\'e des donn\\'ees d'entrainement (C AMEM BERT Contextual Language Models for French: Impact of Training Data Size and Heterogeneity )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"qu-apporte-bert-a-l-analyse-syntaxique-en","title":"Qu'apporte BERT \\`a l'analyse syntaxique en constituants discontinus ? Une suite de tests pour \\'evaluer les pr\\'edictions de structures syntaxiques discontinues en anglais (What does BERT contribute to discontinuous constituency parsing ? A test suite to evaluate discontinuous constituency structure predictions in English)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"r-e-entra-iner-ou-entra-iner-soi-m-eme-strat","title":"R\\'e-entra\\^\\iner ou entra\\^\\iner soi-m\\^eme ? Strat\\'egies de pr\\'e-entra\\^\\inement de BERT en domaine m\\'edical (Re-train or train from scratch ? Pre-training strategies for BERT in the medical domain )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-bert-forgets-how-to-pos-amnesic-probing","title":"Amnesic Probing: Behavioral Explanation with Amnesic Counterfactuals","date":"2020-06-01","arxiv_id":"2006.00995","n_code_links":0,"syntology":null},{"paper":null,"slug":"bpgc-at-semeval-2020-task-11-propaganda","title":"BPGC at SemEval-2020 Task 11: Propaganda Detection in News Articles with Multi-Granularity Knowledge Sharing and Linguistic Features based Ensemble Learning","date":"2020-05-31","arxiv_id":"2006.00593","n_code_links":0,"syntology":null}],"record_sha256":"655cfdfa02a2e52f6c8c57fd77292a42619c2b6c9052814721c3d511919520ae","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}