{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/42","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":42,"pages_in_order":71,"rows_per_page":100,"rows":[4101,4200],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/41","next":"/method/linear-warmup-with-linear-decay/papers/43","papers":[{"paper":null,"slug":"exploring-story-generation-with-multi-task","title":"Exploring Story Generation with Multi-task Objectives in Variational Autoencoders","date":"2021-11-15","arxiv_id":"2111.08133","n_code_links":0,"syntology":null},{"paper":"/paper/iiitt-dravidian-codemix-fire2021","slug":"iiitt-dravidian-codemix-fire2021","title":"IIITT@Dravidian-CodeMix-FIRE2021: Transliterate or translate? Sentiment analysis of code-mixed text in Dravidian languages","date":"2021-11-15","arxiv_id":"2111.07906","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-prosody-for-unseen-texts-in-speech","title":"Improving Prosody for Unseen Texts in Speech Synthesis by Utilizing Linguistic Information and Noisy Data","date":"2021-11-15","arxiv_id":"2111.07549","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-law-for-recommendation-models-towards","title":"Scaling Law for Recommendation Models: Towards General-purpose User Representations","date":"2021-11-15","arxiv_id":"2111.11294","n_code_links":0,"syntology":null},{"paper":null,"slug":"will-you-find-these-shortcuts-a-protocol-for","title":"\"Will You Find These Shortcuts?\" A Protocol for Evaluating the Faithfulness of Input Salience Methods for Text Classification","date":"2021-11-14","arxiv_id":"2111.07367","n_code_links":0,"syntology":null},{"paper":null,"slug":"socialbert-transformers-for-online","title":"SocialBERT -- Transformers for Online SocialNetwork Language Modelling","date":"2021-11-13","arxiv_id":"2111.07148","n_code_links":0,"syntology":null},{"paper":"/paper/ms-latte-a-dataset-of-where-and-when-to-do","slug":"ms-latte-a-dataset-of-where-and-when-to-do","title":"MS-LaTTE: A Dataset of Where and When To-do Tasks are Completed","date":"2021-11-12","arxiv_id":"2111.06902","n_code_links":1,"syntology":null},{"paper":"/paper/character-level-hypernetworks-for-hate-speech","slug":"character-level-hypernetworks-for-hate-speech","title":"Character-level HyperNetworks for Hate Speech Detection","date":"2021-11-11","arxiv_id":"2111.06336","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-large-scale-language-models-and","title":"Improving Large-scale Language Models and Resources for Filipino","date":"2021-11-11","arxiv_id":"2111.06053","n_code_links":0,"syntology":null},{"paper":null,"slug":"amazon-sagemaker-model-parallelism-a-general","title":"Amazon SageMaker Model Parallelism: A General and Flexible Framework for Large Model Training","date":"2021-11-10","arxiv_id":"2111.05972","n_code_links":0,"syntology":null},{"paper":"/paper/bagbert-bert-based-bagging-stacking-for-multi","slug":"bagbert-bert-based-bagging-stacking-for-multi","title":"BagBERT: BERT-based bagging-stacking for multi-topic classification","date":"2021-11-10","arxiv_id":"2111.05808","n_code_links":1,"syntology":null},{"paper":null,"slug":"cehr-bert-incorporating-temporal-information","title":"CEHR-BERT: Incorporating temporal information from structured EHR data to improve prediction tasks","date":"2021-11-10","arxiv_id":"2111.08585","n_code_links":0,"syntology":null},{"paper":"/paper/prune-once-for-all-sparse-pre-trained","slug":"prune-once-for-all-sparse-pre-trained","title":"Prune Once for All: Sparse Pre-Trained Language Models","date":"2021-11-10","arxiv_id":"2111.05754","n_code_links":2,"syntology":null},{"paper":null,"slug":"dsbert-unsupervised-dialogue-structure","title":"DSBERT:Unsupervised Dialogue Structure learning with BERT","date":"2021-11-09","arxiv_id":"2111.04933","n_code_links":0,"syntology":null},{"paper":null,"slug":"fpm-a-collection-of-large-scale-foundation","title":"FPM: A Collection of Large-scale Foundation Pre-trained Language Models","date":"2021-11-09","arxiv_id":"2111.04909","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-in-the-loop-disinformation-detection","title":"Human-in-the-Loop Disinformation Detection: Stance, Sentiment, or Something Else?","date":"2021-11-09","arxiv_id":"2111.05139","n_code_links":0,"syntology":null},{"paper":"/paper/ai-upv-at-iberlef-2021-detoxis-task-toxicity","slug":"ai-upv-at-iberlef-2021-detoxis-task-toxicity","title":"AI-UPV at IberLEF-2021 DETOXIS task: Toxicity Detection in Immigration-Related Web News Comments Using Transformers and Statistical Models","date":"2021-11-08","arxiv_id":"2111.04530","n_code_links":1,"syntology":null},{"paper":"/paper/chemical-detection-and-indexing-in-pubmed","slug":"chemical-detection-and-indexing-in-pubmed","title":"Chemical detection and indexing in PubMed full text articles using deep learning and rule-based methods","date":"2021-11-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-depression-in-thai-blog-posts-a-1","title":"Detecting Depression in Thai Blog Posts: a Dataset and a Baseline","date":"2021-11-08","arxiv_id":"2111.04574","n_code_links":0,"syntology":null},{"paper":"/paper/guiding-multi-step-rearrangement-tasks-with","slug":"guiding-multi-step-rearrangement-tasks-with","title":"Guiding Multi-Step Rearrangement Tasks with Natural Language Instructions","date":"2021-11-08","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/sexism-prediction-in-spanish-and-english","slug":"sexism-prediction-in-spanish-and-english","title":"Sexism Prediction in Spanish and English Tweets Using Monolingual and Multilingual BERT and Ensemble Models","date":"2021-11-08","arxiv_id":"2111.04551","n_code_links":1,"syntology":null},{"paper":"/paper/synthesizing-collective-communication","slug":"synthesizing-collective-communication","title":"TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches","date":"2021-11-08","arxiv_id":"2111.04867","n_code_links":2,"syntology":null},{"paper":"/paper/tacl-improving-bert-pre-training-with-token","slug":"tacl-improving-bert-pre-training-with-token","title":"TaCL: Improving BERT Pre-training with Token-aware Contrastive Learning","date":"2021-11-07","arxiv_id":"2111.04198","n_code_links":2,"syntology":null},{"paper":null,"slug":"profitable-trade-off-between-memory-and","title":"Profitable Trade-Off Between Memory and Performance In Multi-Domain Chatbot Architectures","date":"2021-11-06","arxiv_id":"2111.03963","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-transformer-transducer-for","title":"Context-Aware Transformer Transducer for Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03250","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-speech-recognition-leveraging","title":"Effective Cross-Utterance Language Modeling for Conversational Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03333","n_code_links":0,"syntology":null},{"paper":null,"slug":"ibert-idiom-cloze-style-reading-comprehension","title":"IBERT: Idiom Cloze-style reading comprehension with Attention","date":"2021-11-05","arxiv_id":"2112.02994","n_code_links":0,"syntology":null},{"paper":null,"slug":"sexism-identification-in-tweets-and-gabs","title":"Sexism Identification in Tweets and Gabs using Deep Neural Networks","date":"2021-11-05","arxiv_id":"2111.03612","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-text-autoencoder-from-transformer-for-fast","title":"A text autoencoder from transformer for fast encoding language representation","date":"2021-11-04","arxiv_id":"2111.02844","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-the-effectiveness-of-an","slug":"an-empirical-study-of-the-effectiveness-of-an","title":"An Empirical Study of the Effectiveness of an Ensemble of Stand-alone Sentiment Detection Tools for Software Engineering Datasets","date":"2021-11-04","arxiv_id":"2111.03196","n_code_links":1,"syntology":null},{"paper":"/paper/conformal-prediction-for-text-infilling-and","slug":"conformal-prediction-for-text-infilling-and","title":"Conformal prediction for text infilling and part-of-speech prediction","date":"2021-11-04","arxiv_id":"2111.02592","n_code_links":1,"syntology":null},{"paper":"/paper/an-empirical-study-of-training-end-to-end","slug":"an-empirical-study-of-training-end-to-end","title":"An Empirical Study of Training End-to-End Vision-and-Language Transformers","date":"2021-11-03","arxiv_id":"2111.02387","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zdou0830/meter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"bert-dre-bert-with-deep-recursive-encoder-for","title":"BERT-DRE: BERT with Deep Recursive Encoder for Natural Language Sentence Matching","date":"2021-11-03","arxiv_id":"2111.02188","n_code_links":0,"syntology":null},{"paper":null,"slug":"detection-of-hate-speech-using-bert-and-hate","title":"Detection of Hate Speech using BERT and Hate Speech Word Embedding with Deep Model","date":"2021-11-02","arxiv_id":"2111.01515","n_code_links":0,"syntology":null},{"paper":"/paper/sentence-encoding-for-dialogue-act","slug":"sentence-encoding-for-dialogue-act","title":"Sentence encoding for Dialogue Act classification","date":"2021-11-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/uquad1-0-development-of-an-urdu-question","slug":"uquad1-0-development-of-an-urdu-question","title":"UQuAD1.0: Development of an Urdu Question Answering Training Data for Machine Reading Comprehension","date":"2021-11-02","arxiv_id":"2111.01543","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-study-of-long-document","title":"Comparative Study of Long Document Classification","date":"2021-11-01","arxiv_id":"2111.00702","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-causal-associations-in-tweets","slug":"identifying-causal-associations-in-tweets","title":"Identifying causal relations in tweets using deep learning: Use case on diabetes-related tweets from 2017-2021","date":"2021-11-01","arxiv_id":"2111.01225","n_code_links":1,"syntology":null},{"paper":"/paper/maple-masking-words-to-generate-blackout","slug":"maple-masking-words-to-generate-blackout","title":"MAPLE – MAsking words to generate blackout Poetry using sequence-to-sequence LEarning","date":"2021-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"recent-advances-in-natural-language","title":"Recent Advances in Natural Language Processing via Large Pre-Trained Language Models: A Survey","date":"2021-11-01","arxiv_id":"2111.01243","n_code_links":0,"syntology":null},{"paper":"/paper/fineas-financial-embedding-analysis-of","slug":"fineas-financial-embedding-analysis-of","title":"FinEAS: Financial Embedding Analysis of Sentiment","date":"2021-10-31","arxiv_id":"2111.00526","n_code_links":1,"syntology":null},{"paper":"/paper/backdoor-pre-trained-models-can-transfer-to","slug":"backdoor-pre-trained-models-can-transfer-to","title":"Backdoor Pre-trained Models Can Transfer to All","date":"2021-10-30","arxiv_id":"2111.00197","n_code_links":1,"syntology":null},{"paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","slug":"dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","arxiv_id":"2111.00160","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vita-group/dsee"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"magic-pyramid-accelerating-inference-with","title":"Magic Pyramid: Accelerating Inference with Early Exiting and Token Pruning","date":"2021-10-30","arxiv_id":"2111.00230","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-sequence-tagging-framework-for","title":"ICDM 2020 Knowledge Graph Contest: Consumer Event-Cause Extraction","date":"2021-10-28","arxiv_id":"2110.15722","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sequence-to-sequence-model-for-extracting","title":"A Sequence to Sequence Model for Extracting Multiple Product Name Entities from Dialog","date":"2021-10-28","arxiv_id":"2110.14843","n_code_links":0,"syntology":null},{"paper":"/paper/bridge-the-gap-between-cv-and-nlp-a-gradient","slug":"bridge-the-gap-between-cv-and-nlp-a-gradient","title":"Bridge the Gap Between CV and NLP! A Gradient-based Textual Adversarial Attack Framework","date":"2021-10-28","arxiv_id":"2110.15317","n_code_links":1,"syntology":null},{"paper":"/paper/colossal-ai-a-unified-deep-learning-system","slug":"colossal-ai-a-unified-deep-learning-system","title":"Colossal-AI: A Unified Deep Learning System For Large-Scale Parallel Training","date":"2021-10-28","arxiv_id":"2110.14883","n_code_links":1,"syntology":null},{"paper":null,"slug":"pruning-attention-heads-of-transformer-models","title":"Pruning Attention Heads of Transformer Models Using A* Search: A Novel Approach to Compress Big NLP Architectures","date":"2021-10-28","arxiv_id":"2110.15225","n_code_links":0,"syntology":null},{"paper":"/paper/semi-siamese-bi-encoder-neural-ranking-model","slug":"semi-siamese-bi-encoder-neural-ranking-model","title":"Semi-Siamese Bi-encoder Neural Ranking Model Using Lightweight Fine-Tuning","date":"2021-10-28","arxiv_id":"2110.14943","n_code_links":1,"syntology":null},{"paper":null,"slug":"anomaly-injected-deep-support-vector-data","title":"Anomaly-Injected Deep Support Vector Data Description for Text Outlier Detection","date":"2021-10-27","arxiv_id":"2110.14729","n_code_links":0,"syntology":null},{"paper":null,"slug":"clauserec-a-clause-recommendation-framework","title":"CLAUSEREC: A Clause Recommendation Framework for AI-aided Contract Authoring","date":"2021-10-26","arxiv_id":"2110.15794","n_code_links":0,"syntology":null},{"paper":"/paper/post-processing-for-individual-fairness","slug":"post-processing-for-individual-fairness","title":"Post-processing for Individual Fairness","date":"2021-10-26","arxiv_id":"2110.13796","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["felix-petersen/fairness-post-processing"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/s2s-ft-fine-tuning-pretrained-transformer","slug":"s2s-ft-fine-tuning-pretrained-transformer","title":"s2s-ft: Fine-Tuning Pretrained Transformer Encoders for Sequence-to-Sequence Learning","date":"2021-10-26","arxiv_id":"2110.13640","n_code_links":1,"syntology":null},{"paper":"/paper/tribert-full-body-human-centric-audio-visual","slug":"tribert-full-body-human-centric-audio-visual","title":"TriBERT: Full-body Human-centric Audio-visual Representation Learning for Visual Sound Separation","date":"2021-10-26","arxiv_id":"2110.13412","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ubc-vision/tribert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-tuning-of-pre-trained-transformers-for","slug":"fine-tuning-of-pre-trained-transformers-for","title":"Fine-tuning of Pre-trained Transformers for Hate, Offensive, and Profane Content Detection in English and Marathi","date":"2021-10-25","arxiv_id":"2110.12687","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-cnn-for-highly-inflected-bengali","title":"Paradigm Shift in Language Modeling: Revisiting CNN for Modeling Sanskrit Originated Bengali and Hindi Language","date":"2021-10-25","arxiv_id":"2110.13032","n_code_links":0,"syntology":null},{"paper":null,"slug":"hate-and-offensive-speech-detection-in-hindi","title":"Hate and Offensive Speech Detection in Hindi and Marathi","date":"2021-10-23","arxiv_id":"2110.12200","n_code_links":0,"syntology":null},{"paper":"/paper/double-trouble-how-to-not-explain-a-text","slug":"double-trouble-how-to-not-explain-a-text","title":"Double Trouble: How to not explain a text classifier's decisions using counterfactuals synthesized by masked language models?","date":"2021-10-22","arxiv_id":"2110.11929","n_code_links":1,"syntology":null},{"paper":"/paper/learning-text-image-joint-embedding-for","slug":"learning-text-image-joint-embedding-for","title":"Learning Text-Image Joint Embedding for Efficient Cross-Modal Retrieval with Deep Feature Engineering","date":"2021-10-22","arxiv_id":"2110.11592","n_code_links":1,"syntology":null},{"paper":"/paper/cloob-modern-hopfield-networks-with-infoloob-1","slug":"cloob-modern-hopfield-networks-with-infoloob-1","title":"CLOOB: Modern Hopfield Networks with InfoLOOB Outperform CLIP","date":"2021-10-21","arxiv_id":"2110.11316","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":1,"n_instrument":6,"unverified":4,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ml-jku/cloob"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/fast-model-editing-at-scale-1","slug":"fast-model-editing-at-scale-1","title":"Fast Model Editing at Scale","date":"2021-10-21","arxiv_id":"2110.11309","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["eric-mitchell/mend"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modeling-performance-in-open-domain-dialogue","title":"Modeling Performance in Open-Domain Dialogue with PARADISE","date":"2021-10-21","arxiv_id":"2110.11164","n_code_links":0,"syntology":null},{"paper":"/paper/distributionally-robust-classifiers-in","slug":"distributionally-robust-classifiers-in","title":"Distributionally Robust Classifiers in Sentiment Analysis","date":"2021-10-20","arxiv_id":"2110.10372","n_code_links":1,"syntology":null},{"paper":null,"slug":"slam-a-unified-encoder-for-speech-and","title":"SLAM: A Unified Encoder for Speech and Language Modeling via Speech-Text Joint Pre-Training","date":"2021-10-20","arxiv_id":"2110.10329","n_code_links":0,"syntology":null},{"paper":"/paper/ensemble-albert-on-squad-2-0","slug":"ensemble-albert-on-squad-2-0","title":"Ensemble ALBERT on SQuAD 2.0","date":"2021-10-19","arxiv_id":"2110.09665","n_code_links":1,"syntology":null},{"paper":null,"slug":"risks-of-ai-foundation-models-in-education","title":"Risks of AI Foundation Models in Education","date":"2021-10-19","arxiv_id":"2110.10024","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-data-bootstrapping-recipe-for-low-resource","title":"A Data Bootstrapping Recipe for Low Resource Multilingual Relation Classification","date":"2021-10-18","arxiv_id":"2110.09570","n_code_links":0,"syntology":null},{"paper":null,"slug":"bermo-what-can-bert-learn-from-elmo-1","title":"BERMo: What can BERT learn from ELMo?","date":"2021-10-18","arxiv_id":"2110.15802","n_code_links":0,"syntology":null},{"paper":null,"slug":"ceasing-hate-withmoh-hate-speech-detection-in","title":"Ceasing hate withMoH: Hate Speech Detection in Hindi-English Code-Switched Language","date":"2021-10-18","arxiv_id":"2110.09393","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-hate-speech-detection-in-code","title":"Contextual Hate Speech Detection in Code Mixed Text using Transformer Based Approaches","date":"2021-10-18","arxiv_id":"2110.09338","n_code_links":0,"syntology":null},{"paper":null,"slug":"virapart-a-text-refinement-framework-for-asr","title":"ViraPart: A Text Refinement Framework for Automatic Speech Recognition and Natural Language Processing Tasks in Persian","date":"2021-10-18","arxiv_id":"2110.09086","n_code_links":0,"syntology":null},{"paper":null,"slug":"bitfit-simple-parameter-efficient-fine-tuning-1","title":"BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enct5-fine-tuning-t5-encoder-for-non","slug":"enct5-fine-tuning-t5-encoder-for-non","title":"EncT5: A Framework for Fine-tuning T5 as Non-autoregressive Models","date":"2021-10-16","arxiv_id":"2110.08426","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-transformer-networks-for-long","title":"Hierarchical Transformer Networks for Long-sequence and Multiple Clinical Documents Classification","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"impli-investigating-nli-models-performance-on","title":"IMPLI: Investigating NLI Models' Performance on Figurative Language","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"models-in-a-spelling-bee-language-models-1","title":"Models In a Spelling Bee: Language Models Implicitly Learn the Character Composition of Tokens","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/old-bert-new-tricks-artificial-language-1","slug":"old-bert-new-tricks-artificial-language-1","title":"Old BERT, New Tricks: Artificial Language Learning for Pre-Trained Language Models","date":"2021-10-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-current-state-of-reproducibility-and","title":"On the current state of reproducibility and reporting of uncertainty for Aspect-based Sentiment Analysis","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/on-the-robustness-of-reading-comprehension","slug":"on-the-robustness-of-reading-comprehension","title":"On the Robustness of Reading Comprehension Models to Entity Renaming","date":"2021-10-16","arxiv_id":"2110.08555","n_code_links":1,"syntology":null},{"paper":"/paper/primer-pyramid-based-masked-sentence-pre","slug":"primer-pyramid-based-masked-sentence-pre","title":"PRIMERA: Pyramid-based Masked Sentence Pre-training for Multi-document Summarization","date":"2021-10-16","arxiv_id":"2110.08499","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/primer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"semantic-search-as-extractive-paraphrase-span","title":"Semantic Search as Extractive Paraphrase Span Detection","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"spellm-augmenting-chinese-spell-check-using","title":"SpelLM: Augmenting Chinese Spell Check Using Input Salience","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wechsel-effective-initialization-of-subword","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"what-do-compressed-large-language-models","title":"Robustness Challenges in Model Distillation and Pruning for Natural Language Understanding","date":"2021-10-16","arxiv_id":"2110.08419","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-gender-bias-in-transformer-based","title":"Detecting Gender Bias in Transformer-based Models: A Case Study on BERT","date":"2021-10-15","arxiv_id":"2110.15733","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-faithfulness-of-importance","slug":"evaluating-the-faithfulness-of-importance","title":"Evaluating the Faithfulness of Importance Measures in NLP by Recursively Masking Allegedly Important Tokens and Retraining","date":"2021-10-15","arxiv_id":"2110.08412","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AndreasMadsen/nlp-roar-interpretability"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generating-natural-language-adversarial-1","title":"Generating Natural Language Adversarial Examples through An Improved Beam Search Algorithm","date":"2021-10-15","arxiv_id":"2110.08036","n_code_links":0,"syntology":null},{"paper":null,"slug":"intent-based-product-collections-for-e","title":"Intent-based Product Collections for E-commerce using Pretrained Language Models","date":"2021-10-15","arxiv_id":"2110.08241","n_code_links":0,"syntology":null},{"paper":"/paper/probing-as-quantifying-the-inductive-bias-of","slug":"probing-as-quantifying-the-inductive-bias-of","title":"Probing as Quantifying Inductive Bias","date":"2021-10-15","arxiv_id":"2110.08388","n_code_links":1,"syntology":null},{"paper":"/paper/tracing-origins-coref-aware-machine-reading","slug":"tracing-origins-coref-aware-machine-reading","title":"Tracing Origins: Coreference-aware Machine Reading Comprehension","date":"2021-10-15","arxiv_id":"2110.07961","n_code_links":1,"syntology":null},{"paper":"/paper/a-simple-strong-and-robust-baseline-for","slug":"a-simple-strong-and-robust-baseline-for","title":"PARE: A Simple and Strong Baseline for Monolingual and Multilingual Distantly Supervised Relation Extraction","date":"2021-10-14","arxiv_id":"2110.07415","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert2bert-towards-reusable-pretrained","title":"bert2BERT: Towards Reusable Pretrained Language Models","date":"2021-10-14","arxiv_id":"2110.07143","n_code_links":0,"syntology":null},{"paper":"/paper/bi-rads-bert-using-section-tokenization-to","slug":"bi-rads-bert-using-section-tokenization-to","title":"BI-RADS BERT & Using Section Segmentation to Understand Radiology Reports","date":"2021-10-14","arxiv_id":"2110.07552","n_code_links":1,"syntology":null},{"paper":"/paper/building-chinese-biomedical-language-models","slug":"building-chinese-biomedical-language-models","title":"Building Chinese Biomedical Language Models via Multi-Level Text Discrimination","date":"2021-10-14","arxiv_id":"2110.07244","n_code_links":1,"syntology":null},{"paper":null,"slug":"causally-estimating-the-sensitivity-of-neural-1","title":"Interpreting the Robustness of Neural NLP Models to Textual Perturbations","date":"2021-10-14","arxiv_id":"2110.07159","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-gloss-augmentation-for-improving-word","title":"Context-gloss Augmentation for Improving Word Sense Disambiguation","date":"2021-10-14","arxiv_id":"2110.07174","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-off-the-shelf-machine-listening","title":"Evaluating Off-the-Shelf Machine Listening and Natural Language Models for Automated Audio Captioning","date":"2021-10-14","arxiv_id":"2110.07410","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","n_code_links":1,"syntology":null},{"paper":"/paper/p-adapters-robustly-extracting-factual-1","slug":"p-adapters-robustly-extracting-factual-1","title":"P-Adapters: Robustly Extracting Factual Information from Language Models with Diverse Prompts","date":"2021-10-14","arxiv_id":"2110.07280","n_code_links":1,"syntology":null}],"record_sha256":"04b651058c263b60ba2849594f3d48d1092d1b4a5484f5f8c105e541fa233b97","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}