{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/47","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":47,"pages_in_order":71,"rows_per_page":100,"rows":[4601,4700],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/46","next":"/method/linear-warmup-with-linear-decay/papers/48","papers":[{"paper":"/paper/a-review-of-bangla-natural-language","slug":"a-review-of-bangla-natural-language","title":"A Review of Bangla Natural Language Processing Tasks and the Utility of Transformer Models","date":"2021-07-08","arxiv_id":"2107.03844","n_code_links":2,"syntology":null},{"paper":null,"slug":"bumblebee-a-transformer-for-music","title":"BumbleBee: A Transformer for Music","date":"2021-07-07","arxiv_id":"2107.03443","n_code_links":0,"syntology":null},{"paper":"/paper/can-transformer-models-measure-coherence-in-1","slug":"can-transformer-models-measure-coherence-in-1","title":"Can Transformer Models Measure Coherence In Text? Re-Thinking the Shuffle Test","date":"2021-07-07","arxiv_id":"2107.03448","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-transformer-for-direct-speech","title":"Efficient Transformer for Direct Speech Translation","date":"2021-07-07","arxiv_id":"2107.03069","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-hijacked-reviews","title":"Identifying Hijacked Reviews","date":"2021-07-07","arxiv_id":"2107.05385","n_code_links":0,"syntology":null},{"paper":null,"slug":"languagerefer-spatial-language-model-for-3d","title":"LanguageRefer: Spatial-Language Model for 3D Visual Grounding","date":"2021-07-07","arxiv_id":"2107.03438","n_code_links":0,"syntology":null},{"paper":null,"slug":"contradiction-detection-in-persian-text","title":"Contradiction Detection in Persian Text","date":"2021-07-05","arxiv_id":"2107.01987","n_code_links":0,"syntology":null},{"paper":null,"slug":"experiments-with-adversarial-attacks-on-text","title":"Experiments with adversarial attacks on text genres","date":"2021-07-05","arxiv_id":"2107.02246","n_code_links":0,"syntology":null},{"paper":"/paper/what-helps-transformers-recognize","slug":"what-helps-transformers-recognize","title":"What Helps Transformers Recognize Conversational Structure? Importance of Context, Punctuation, and Labels in Dialog Act Recognition","date":"2021-07-05","arxiv_id":"2107.02294","n_code_links":1,"syntology":null},{"paper":"/paper/kaisa-an-adaptive-second-order-optimizer","slug":"kaisa-an-adaptive-second-order-optimizer","title":"KAISA: An Adaptive Second-Order Optimizer Framework for Deep Neural Networks","date":"2021-07-04","arxiv_id":"2107.01739","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gpauloski/kfac_pytorch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/he-thinks-he-knows-better-than-the-doctors","slug":"he-thinks-he-knows-better-than-the-doctors","title":"He Thinks He Knows Better than the Doctors: BERT for Event Factuality Fails on Pragmatics","date":"2021-07-02","arxiv_id":"2107.00807","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-identification-of-hindi-english","title":"Language Identification of Hindi-English tweets using code-mixed BERT","date":"2021-07-02","arxiv_id":"2107.01202","n_code_links":0,"syntology":null},{"paper":"/paper/ultrasound-video-transformers-for-cardiac","slug":"ultrasound-video-transformers-for-cardiac","title":"Ultrasound Video Transformers for Cardiac Ejection Fraction Estimation","date":"2021-07-02","arxiv_id":"2107.00977","n_code_links":1,"syntology":null},{"paper":null,"slug":"elbert-fast-albert-with-confidence-window","title":"Elbert: Fast Albert with Confidence-Window Based Early Exit","date":"2021-07-01","arxiv_id":"2107.00175","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-domain-agnostic-and-specific","title":"Leveraging Domain Agnostic and Specific Knowledge for Acronym Disambiguation","date":"2021-07-01","arxiv_id":"2107.00316","n_code_links":0,"syntology":null},{"paper":null,"slug":"autolaw-augmented-legal-reasoning-through","title":"AutoLAW: Augmented Legal Reasoning through Legal Precedent Prediction","date":"2021-06-30","arxiv_id":"2106.16034","n_code_links":0,"syntology":null},{"paper":null,"slug":"early-risk-detection-of-pathological-gambling","title":"Early Risk Detection of Pathological Gambling, Self-Harm and Depression Using BERT","date":"2021-06-30","arxiv_id":"2106.16175","n_code_links":0,"syntology":null},{"paper":"/paper/the-multiberts-bert-reproductions-for","slug":"the-multiberts-bert-reproductions-for","title":"The MultiBERTs: BERT Reproductions for Robustness Analysis","date":"2021-06-30","arxiv_id":"2106.16163","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/language"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hate-speech-detection-using-static-bert","slug":"hate-speech-detection-using-static-bert","title":"Hate speech detection using static BERT embeddings","date":"2021-06-29","arxiv_id":"2106.15537","n_code_links":0,"syntology":null},{"paper":null,"slug":"new-arabic-medical-dataset-for-diseases","title":"New Arabic Medical Dataset for Diseases Classification","date":"2021-06-29","arxiv_id":"2106.15236","n_code_links":0,"syntology":null},{"paper":"/paper/packing-towards-2x-nlp-bert-acceleration","slug":"packing-towards-2x-nlp-bert-acceleration","title":"Efficient Sequence Packing without Cross-contamination: Accelerating Large Language Models without Impacting Performance","date":"2021-06-29","arxiv_id":"2107.02027","n_code_links":1,"syntology":null},{"paper":"/paper/a-3d-cnn-network-with-bert-for-automatic","slug":"a-3d-cnn-network-with-bert-for-automatic","title":"A 3D CNN Network with BERT For Automatic COVID-19 Diagnosis From CT-Scan Images","date":"2021-06-28","arxiv_id":"2106.14403","n_code_links":1,"syntology":null},{"paper":null,"slug":"current-landscape-of-the-russian-sentiment","title":"Current Landscape of the Russian Sentiment Corpora","date":"2021-06-28","arxiv_id":"2106.14434","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-the-generalization-for-intent","title":"Enhancing the Generalization for Intent Classification and Out-of-Domain Detection in SLU","date":"2021-06-28","arxiv_id":"2106.14464","n_code_links":0,"syntology":null},{"paper":null,"slug":"traditional-machine-learning-and-deep","title":"Traditional Machine Learning and Deep Learning Models for Argumentation Mining in Russian Texts","date":"2021-06-28","arxiv_id":"2106.14438","n_code_links":0,"syntology":null},{"paper":"/paper/a-closer-look-at-how-fine-tuning-changes-bert","slug":"a-closer-look-at-how-fine-tuning-changes-bert","title":"A Closer Look at How Fine-tuning Changes BERT","date":"2021-06-27","arxiv_id":"2106.14282","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["utahnlp/BERT-fine-tuning-analysis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ai-based-presentation-creator-with-customized","title":"AI based Presentation Creator With Customized Audio Content Delivery","date":"2021-06-27","arxiv_id":"2106.14213","n_code_links":0,"syntology":null},{"paper":null,"slug":"answering-chinese-elementary-school-social","title":"Answering Chinese Elementary School Social Study Multiple Choice Questions","date":"2021-06-26","arxiv_id":"2107.02893","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-differential-privacy-and","slug":"benchmarking-differential-privacy-and","title":"Benchmarking Differential Privacy and Federated Learning for BERT Models","date":"2021-06-26","arxiv_id":"2106.13973","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-precision-training-in-logarithmic-number","title":"LNS-Madam: Low-Precision Training in Logarithmic Number System using Multiplicative Weight Update","date":"2021-06-26","arxiv_id":"2106.13914","n_code_links":0,"syntology":null},{"paper":"/paper/spreadsheetcoder-formula-prediction-from-semi-1","slug":"spreadsheetcoder-formula-prediction-from-semi-1","title":"SpreadsheetCoder: Formula Prediction from Semi-structured Context","date":"2021-06-26","arxiv_id":"2106.15339","n_code_links":1,"syntology":null},{"paper":"/paper/umic-an-unreferenced-metric-for-image","slug":"umic-an-unreferenced-metric-for-image","title":"UMIC: An Unreferenced Metric for Image Captioning via Contrastive Learning","date":"2021-06-26","arxiv_id":"2106.14019","n_code_links":1,"syntology":{"ran":7,"of":14,"n_ran_checked":7,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["hwanheelee1993/UMIC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adapt-and-distill-developing-small-fast-and","title":"Adapt-and-Distill: Developing Small, Fast and Effective Pretrained Language Models for Domains","date":"2021-06-25","arxiv_id":"2106.13474","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-sample-replacements-for-electra","title":"Learning to Sample Replacements for ELECTRA Pre-Training","date":"2021-06-25","arxiv_id":"2106.13715","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-automated-knowledge-mining-and-document","title":"An Automated Knowledge Mining and Document Classification System with Multi-model Transfer Learning","date":"2021-06-24","arxiv_id":"2106.12744","n_code_links":0,"syntology":null},{"paper":null,"slug":"discovering-novel-drug-supplement","title":"Discovering novel drug-supplement interactions using a dietary supplements knowledge graph generated from the biomedical literature","date":"2021-06-24","arxiv_id":"2106.12741","n_code_links":0,"syntology":null},{"paper":"/paper/education-to-skill-mapping-using-hierarchical","slug":"education-to-skill-mapping-using-hierarchical","title":"Education-to-Skill Mapping Using Hierarchical Classification and Transformer Neural Network","date":"2021-06-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"quantization-aware-training-ernie-and","title":"Quantization Aware Training, ERNIE and Kurtosis Regularizer: a short empirical study","date":"2021-06-24","arxiv_id":"2106.13035","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-topic-segmentation-of-meetings","slug":"unsupervised-topic-segmentation-of-meetings","title":"Unsupervised Topic Segmentation of Meetings with BERT Embeddings","date":"2021-06-24","arxiv_id":"2106.12978","n_code_links":2,"syntology":null},{"paper":"/paper/classifying-textual-data-with-pre-trained","slug":"classifying-textual-data-with-pre-trained","title":"Classifying Textual Data with Pre-trained Vision Models through Transfer Learning and Data Transformations","date":"2021-06-23","arxiv_id":"2106.12479","n_code_links":1,"syntology":null},{"paper":"/paper/learnt-sparsity-for-effective-and","slug":"learnt-sparsity-for-effective-and","title":"Extractive Explanations for Interpretable Text Ranking","date":"2021-06-23","arxiv_id":"2106.12460","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-case-study-in-bootstrapping-ontology-graphs","title":"A Case Study in Bootstrapping Ontology Graphs from Textbooks","date":"2021-06-22","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-exploration-of-pre-training","slug":"a-comprehensive-exploration-of-pre-training","title":"A Comprehensive Comparison of Pre-training Language Models","date":"2021-06-22","arxiv_id":"2106.11483","n_code_links":2,"syntology":null},{"paper":"/paper/combining-analogy-with-language-models-for","slug":"combining-analogy-with-language-models-for","title":"Combining Analogy with Language Models for Knowledge Extraction","date":"2021-06-22","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/fine-tune-the-entire-rag-architecture","slug":"fine-tune-the-entire-rag-architecture","title":"Fine-tune the Entire RAG Architecture (including DPR retriever) for Question-Answering","date":"2021-06-22","arxiv_id":"2106.11517","n_code_links":2,"syntology":null},{"paper":"/paper/lv-bert-exploiting-layer-variety-for-bert","slug":"lv-bert-exploiting-layer-variety-for-bert","title":"LV-BERT: Exploiting Layer Variety for BERT","date":"2021-06-22","arxiv_id":"2106.11740","n_code_links":1,"syntology":null},{"paper":"/paper/one-shot-to-weakly-supervised-relation","slug":"one-shot-to-weakly-supervised-relation","title":"One-shot to Weakly-Supervised Relation Classification using Language Models","date":"2021-06-22","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"ad-text-classification-with-transformer-based","title":"Ad Text Classification with Transformer-Based Natural Language Processing Methods","date":"2021-06-21","arxiv_id":"2106.10899","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-models-in-detection-of-dietary","title":"Deep Learning Models in Detection of Dietary Supplement Adverse Event Signals from Twitter","date":"2021-06-21","arxiv_id":"2106.11403","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-network-pruning-with-uncertainty","slug":"iterative-network-pruning-with-uncertainty","title":"Iterative Network Pruning with Uncertainty Regularization for Lifelong Sentiment Classification","date":"2021-06-21","arxiv_id":"2106.11197","n_code_links":1,"syntology":null},{"paper":"/paper/pseudo-relevance-feedback-for-multiple","slug":"pseudo-relevance-feedback-for-multiple","title":"Pseudo-Relevance Feedback for Multiple Representation Dense Retrieval","date":"2021-06-21","arxiv_id":"2106.11251","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["terrierteam/pyterrier_colbert","cmacdonald/pyterrier_colbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/context-aware-legal-citation-recommendation","slug":"context-aware-legal-citation-recommendation","title":"Context-Aware Legal Citation Recommendation using Deep Learning","date":"2021-06-20","arxiv_id":"2106.10776","n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-approach-to-detecting-symptoms-of","title":"Hybrid approach to detecting symptoms of depression in social media entries","date":"2021-06-19","arxiv_id":"2106.10485","n_code_links":0,"syntology":null},{"paper":null,"slug":"vln-bert-a-recurrent-vision-and-language-bert","title":"VLN BERT: A Recurrent Vision-and-Language BERT for Navigation","date":"2021-06-19","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/all-you-can-embed-natural-language-based","slug":"all-you-can-embed-natural-language-based","title":"All You Can Embed: Natural Language based Vehicle Retrieval with Spatio-Temporal Transformers","date":"2021-06-18","arxiv_id":"2106.10153","n_code_links":1,"syntology":null},{"paper":"/paper/bitfit-simple-parameter-efficient-fine-tuning","slug":"bitfit-simple-parameter-efficient-fine-tuning","title":"BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models","date":"2021-06-18","arxiv_id":"2106.10199","n_code_links":6,"syntology":{"ran":4,"of":13,"n_ran_checked":3,"n_instrument":1,"unverified":9,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["benzakenelad/BitFit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"graph-based-joint-pandemic-concern-and","title":"Graph-based Joint Pandemic Concern and Relation Extraction on Twitter","date":"2021-06-18","arxiv_id":"2106.09929","n_code_links":0,"syntology":null},{"paper":"/paper/knowledgeable-or-educated-guess-revisiting","slug":"knowledgeable-or-educated-guess-revisiting","title":"Knowledgeable or Educated Guess? Revisiting Language Models as Knowledge Bases","date":"2021-06-17","arxiv_id":"2106.09231","n_code_links":1,"syntology":null},{"paper":"/paper/large-scale-private-learning-via-low-rank","slug":"large-scale-private-learning-via-low-rank","title":"Large Scale Private Learning via Low-rank Reparametrization","date":"2021-06-17","arxiv_id":"2106.09352","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":1,"n_instrument":3,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dayu11/Differentially-Private-Deep-Learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lnn-el-a-neuro-symbolic-approach-to-short","slug":"lnn-el-a-neuro-symbolic-approach-to-short","title":"LNN-EL: A Neuro-Symbolic Approach to Short-text Entity Linking","date":"2021-06-17","arxiv_id":"2106.09795","n_code_links":1,"syntology":null},{"paper":"/paper/lora-low-rank-adaptation-of-large-language","slug":"lora-low-rank-adaptation-of-large-language","title":"LoRA: Low-Rank Adaptation of Large Language Models","date":"2021-06-17","arxiv_id":"2106.09685","n_code_links":74,"syntology":{"ran":51,"of":84,"n_ran_checked":44,"n_instrument":7,"unverified":33,"pointer_only":30,"phrase":"51 ran (of which 19 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 7 where Syntology's instrument failed) · 33 unverified","official":{"repos":["microsoft/LoRA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"algorithm-to-compilation-codesign-an","title":"Algorithm to Compilation Co-design: An Integrated View of Neural Network Sparsity","date":"2021-06-16","arxiv_id":"2106.08846","n_code_links":0,"syntology":null},{"paper":"/paper/tssubert-tweet-stream-summarization-using","slug":"tssubert-tweet-stream-summarization-using","title":"TSSuBERT: Tweet Stream Summarization Using BERT","date":"2021-06-16","arxiv_id":"2106.08770","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-automated-quality-evaluation-framework-of","title":"An Automated Quality Evaluation Framework of Psychotherapy Conversations with Local Quality Estimates","date":"2021-06-15","arxiv_id":"2106.07922","n_code_links":0,"syntology":null},{"paper":"/paper/beit-bert-pre-training-of-image-transformers","slug":"beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","arxiv_id":"2106.08254","n_code_links":14,"syntology":{"ran":6,"of":11,"n_ran_checked":4,"n_instrument":2,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["microsoft/unilm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/incorporating-word-sense-disambiguation-in","slug":"incorporating-word-sense-disambiguation-in","title":"Incorporating Word Sense Disambiguation in Neural Language Models","date":"2021-06-15","arxiv_id":"2106.07967","n_code_links":2,"syntology":null},{"paper":"/paper/knowledge-rich-bert-embeddings-for","slug":"knowledge-rich-bert-embeddings-for","title":"BERT Embeddings for Automatic Readability Assessment","date":"2021-06-15","arxiv_id":"2106.07935","n_code_links":1,"syntology":null},{"paper":"/paper/medical-code-prediction-from-discharge","slug":"medical-code-prediction-from-discharge","title":"Medical Code Prediction from Discharge Summary: Document to Sequence BERT using Sequence Attention","date":"2021-06-15","arxiv_id":"2106.07932","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-bert-dig-it-named-entity-recognition-for","title":"Can BERT Dig It? -- Named Entity Recognition for Information Retrieval in the Archaeology Domain","date":"2021-06-14","arxiv_id":"2106.07742","n_code_links":0,"syntology":null},{"paper":"/paper/dataset-of-propaganda-techniques-of-the-state","slug":"dataset-of-propaganda-techniques-of-the-state","title":"Dataset of Propaganda Techniques of the State-Sponsored Information Operation of the People's Republic of China","date":"2021-06-14","arxiv_id":"2106.07544","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-sentence-level-representations-for","slug":"exploiting-sentence-level-representations-for","title":"Exploiting Sentence-Level Representations for Passage Ranking","date":"2021-06-14","arxiv_id":"2106.07316","n_code_links":1,"syntology":null},{"paper":"/paper/hubert-self-supervised-speech-representation","slug":"hubert-self-supervised-speech-representation","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","date":"2021-06-14","arxiv_id":"2106.07447","n_code_links":11,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["pytorch/fairseq","huggingface/transformers"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/modeling-profanity-and-hate-speech-in-social","slug":"modeling-profanity-and-hate-speech-in-social","title":"Modeling Profanity and Hate Speech in Social Media with Semantic Subspaces","date":"2021-06-14","arxiv_id":"2106.07505","n_code_links":1,"syntology":null},{"paper":null,"slug":"pre-trained-models-past-present-and-future","title":"Pre-Trained Models: Past, Present and Future","date":"2021-06-14","arxiv_id":"2106.07139","n_code_links":0,"syntology":null},{"paper":"/paper/sas-self-augmented-strategy-for-language","slug":"sas-self-augmented-strategy-for-language","title":"SAS: Self-Augmentation Strategy for Language Model Pre-training","date":"2021-06-14","arxiv_id":"2106.07176","n_code_links":1,"syntology":null},{"paper":null,"slug":"why-can-you-lay-off-heads-investigating-how","title":"Why Can You Lay Off Heads? Investigating How BERT Heads Transfer","date":"2021-06-14","arxiv_id":"2106.07137","n_code_links":0,"syntology":null},{"paper":null,"slug":"sasicm-a-multi-task-benchmark-for-subtext","title":"SASICM A Multi-Task Benchmark For Subtext Recognition","date":"2021-06-13","arxiv_id":"2106.06944","n_code_links":0,"syntology":null},{"paper":null,"slug":"target-model-agnostic-adversarial-attacks","title":"Target Model Agnostic Adversarial Attacks with Query Budgets on Language Understanding Models","date":"2021-06-13","arxiv_id":"2106.07047","n_code_links":0,"syntology":null},{"paper":"/paper/a-sentence-level-hierarchical-bert-model-for","slug":"a-sentence-level-hierarchical-bert-model-for","title":"A Sentence-level Hierarchical BERT Model for Document Classification with Limited Labelled Data","date":"2021-06-12","arxiv_id":"2106.06738","n_code_links":1,"syntology":null},{"paper":null,"slug":"explaining-the-deep-natural-language","title":"Explaining the Deep Natural Language Processing by Mining Textual Interpretable Features","date":"2021-06-12","arxiv_id":"2106.06697","n_code_links":0,"syntology":null},{"paper":"/paper/neural-combinatory-constituency-parsing","slug":"neural-combinatory-constituency-parsing","title":"Neural Combinatory Constituency Parsing","date":"2021-06-12","arxiv_id":"2106.06689","n_code_links":1,"syntology":null},{"paper":"/paper/bioelectra-pretrained-biomedical-text-encoder","slug":"bioelectra-pretrained-biomedical-text-encoder","title":"BioELECTRA:Pretrained Biomedical text Encoder using Discriminators","date":"2021-06-11","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-language-models-for-continuously","title":"Dynamic Language Models for Continuously Evolving Content","date":"2021-06-11","arxiv_id":"2106.06297","n_code_links":0,"syntology":null},{"paper":"/paper/n-best-asr-transformer-enhancing-slu","slug":"n-best-asr-transformer-enhancing-slu","title":"N-Best ASR Transformer: Enhancing SLU Performance using Multiple ASR Hypotheses","date":"2021-06-11","arxiv_id":"2106.06519","n_code_links":1,"syntology":null},{"paper":null,"slug":"refbert-compressing-bert-by-referencing-to","title":"RefBERT: Compressing BERT by Referencing to Pre-computed Representations","date":"2021-06-11","arxiv_id":"2106.08898","n_code_links":0,"syntology":null},{"paper":"/paper/amu-euranova-at-case-2021-task-1-assessing","slug":"amu-euranova-at-case-2021-task-1-assessing","title":"AMU-EURANOVA at CASE 2021 Task 1: Assessing the stability of multilingual BERT","date":"2021-06-10","arxiv_id":"2106.14625","n_code_links":1,"syntology":null},{"paper":"/paper/convolutions-and-self-attention-re","slug":"convolutions-and-self-attention-re","title":"Convolutions and Self-Attention: Re-interpreting Relative Positions in Pre-trained Language Models","date":"2021-06-10","arxiv_id":"2106.05505","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-lingual-emotion-detection","title":"Cross-lingual Emotion Detection","date":"2021-06-10","arxiv_id":"2106.06017","n_code_links":0,"syntology":null},{"paper":null,"slug":"groupbert-enhanced-transformer-architecture","title":"GroupBERT: Enhanced Transformer Architecture with Efficient Grouped Structures","date":"2021-06-10","arxiv_id":"2106.05822","n_code_links":0,"syntology":null},{"paper":"/paper/marginal-utility-diminishes-exploring-the","slug":"marginal-utility-diminishes-exploring-the","title":"Marginal Utility Diminishes: Exploring the Minimum Knowledge for BERT Knowledge Distillation","date":"2021-06-10","arxiv_id":"2106.05691","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["llyx97/Marginal-Utility-Diminishes"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"semantic-aware-binary-code-representation","title":"Semantic-aware Binary Code Representation with BERT","date":"2021-06-10","arxiv_id":"2106.05478","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-sexism-detection-with-multilingual","title":"Automatic Sexism Detection with Multilingual Transformer Models","date":"2021-06-09","arxiv_id":"2106.04908","n_code_links":0,"syntology":null},{"paper":"/paper/phraseformer-multimodal-key-phrase-extraction","slug":"phraseformer-multimodal-key-phrase-extraction","title":"Phraseformer: Multimodal Key-phrase Extraction using Transformer and Graph Embedding","date":"2021-06-09","arxiv_id":"2106.04939","n_code_links":0,"syntology":null},{"paper":"/paper/sentence-embeddings-using-supervised","slug":"sentence-embeddings-using-supervised","title":"Sentence Embeddings using Supervised Contrastive Learning","date":"2021-06-09","arxiv_id":"2106.04791","n_code_links":1,"syntology":null},{"paper":"/paper/cheap-and-good-simple-and-effective-data","slug":"cheap-and-good-simple-and-effective-data","title":"Cheap and Good? Simple and Effective Data Augmentation for Low Resource Machine Reading","date":"2021-06-08","arxiv_id":"2106.04134","n_code_links":1,"syntology":null},{"paper":null,"slug":"speech-bert-embedding-for-improving-prosody","title":"Speech BERT Embedding For Improving Prosody in Neural TTS","date":"2021-06-08","arxiv_id":"2106.04312","n_code_links":0,"syntology":null},{"paper":"/paper/ultra-fine-entity-typing-with-weak","slug":"ultra-fine-entity-typing-with-weak","title":"Ultra-Fine Entity Typing with Weak Supervision from a Masked Language Model","date":"2021-06-08","arxiv_id":"2106.04098","n_code_links":1,"syntology":null},{"paper":"/paper/bertgen-multi-task-generation-through-bert","slug":"bertgen-multi-task-generation-through-bert","title":"BERTGEN: Multi-task Generation through BERT","date":"2021-06-07","arxiv_id":"2106.03484","n_code_links":1,"syntology":null},{"paper":null,"slug":"lawdr-language-agnostic-weighted-document","title":"LAWDR: Language-Agnostic Weighted Document Representations from Pre-trained Models","date":"2021-06-07","arxiv_id":"2106.03379","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-and-improving-bert-s-mathematical","title":"Measuring and Improving BERT's Mathematical Abilities by Predicting the Order of Reasoning","date":"2021-06-07","arxiv_id":"2106.03921","n_code_links":0,"syntology":null}],"record_sha256":"c182b2d87a95b6f7f558796b35dc761878cf9ee6b342429330110305b04c79bb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}