{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/33","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":33,"pages_in_order":71,"rows_per_page":100,"rows":[3201,3300],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/32","next":"/method/linear-warmup-with-linear-decay/papers/34","papers":[{"paper":"/paper/probing-for-targeted-syntactic-knowledge","slug":"probing-for-targeted-syntactic-knowledge","title":"Probing for targeted syntactic knowledge through grammatical error detection","date":"2022-10-28","arxiv_id":"2210.16228","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-flow-vae-a-weakly-supervised-model-for-1","title":"BERT-Flow-VAE: A Weakly-supervised Model for Multi-Label Text Classification","date":"2022-10-27","arxiv_id":"2210.15225","n_code_links":0,"syntology":null},{"paper":"/paper/coco-dr-combating-distribution-shifts-in-zero","slug":"coco-dr-combating-distribution-shifts-in-zero","title":"COCO-DR: Combating Distribution Shifts in Zero-Shot Dense Retrieval with Contrastive and Distributionally Robust Learning","date":"2022-10-27","arxiv_id":"2210.15212","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["openmatch/coco-dr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cost-eff-collaborative-optimization-of","slug":"cost-eff-collaborative-optimization-of","title":"COST-EFF: Collaborative Optimization of Spatial and Temporal Efficiency with Slenderized Multi-exit Language Models","date":"2022-10-27","arxiv_id":"2210.15523","n_code_links":1,"syntology":null},{"paper":"/paper/fast-distilbert-on-cpus","slug":"fast-distilbert-on-cpus","title":"Fast DistilBERT on CPUs","date":"2022-10-27","arxiv_id":"2211.07715","n_code_links":1,"syntology":null},{"paper":"/paper/fctalker-fine-and-coarse-grained-context","slug":"fctalker-fine-and-coarse-grained-context","title":"FCTalker: Fine and Coarse Grained Context Modeling for Expressive Conversational Speech Synthesis","date":"2022-10-27","arxiv_id":"2210.15360","n_code_links":1,"syntology":null},{"paper":"/paper/masked-vision-language-transformer-in-fashion","slug":"masked-vision-language-transformer-in-fashion","title":"Masked Vision-Language Transformer in Fashion","date":"2022-10-27","arxiv_id":"2210.15110","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-boundary-aware-language-model","slug":"unsupervised-boundary-aware-language-model","title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","date":"2022-10-27","arxiv_id":"2210.15231","n_code_links":2,"syntology":null},{"paper":"/paper/automatic-extraction-of-materials-and","slug":"automatic-extraction-of-materials-and","title":"Automatic extraction of materials and properties from superconductors scientific literature","date":"2022-10-26","arxiv_id":"2210.15600","n_code_links":2,"syntology":null},{"paper":null,"slug":"beyond-english-centric-bitexts-for-better","title":"Beyond English-Centric Bitexts for Better Multilingual Language Representation Learning","date":"2022-10-26","arxiv_id":"2210.14867","n_code_links":0,"syntology":null},{"paper":null,"slug":"bi-link-bridging-inductive-link-predictions","title":"Bi-Link: Bridging Inductive Link Predictions from Text via Contrastive Learning of Transformers and Prompts","date":"2022-10-26","arxiv_id":"2210.14463","n_code_links":0,"syntology":null},{"paper":null,"slug":"don-t-prompt-search-mining-based-zero-shot","title":"Don't Prompt, Search! Mining-based Zero-Shot Learning with Language Models","date":"2022-10-26","arxiv_id":"2210.14803","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-robustness-of-prefix-tuning-in","title":"Exploring Robustness of Prefix Tuning in Noisy Data: A Case Study in Financial Sentiment Analysis","date":"2022-10-26","arxiv_id":"2211.05584","n_code_links":0,"syntology":null},{"paper":"/paper/how-long-is-enough-exploring-the-optimal","slug":"how-long-is-enough-exploring-the-optimal","title":"How Long Is Enough? Exploring the Optimal Intervals of Long-Range Clinical Note Language Modeling","date":"2022-10-25","arxiv_id":"2211.07713","n_code_links":1,"syntology":null},{"paper":null,"slug":"ielm-an-open-information-extraction-benchmark","title":"IELM: An Open Information Extraction Benchmark for Pre-Trained Language Models","date":"2022-10-25","arxiv_id":"2210.14128","n_code_links":0,"syntology":null},{"paper":null,"slug":"effective-pre-training-objectives-for","title":"Effective Pre-Training Objectives for Transformer-based Autoencoders","date":"2022-10-24","arxiv_id":"2210.13536","n_code_links":0,"syntology":null},{"paper":null,"slug":"entity-level-sentiment-analysis-in-contact","title":"Entity-level Sentiment Analysis in Contact Center Telephone Conversations","date":"2022-10-24","arxiv_id":"2210.13401","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-translationese-why-are-neural","title":"Explaining Translationese: why are Neural Classifiers Better and what do they Learn?","date":"2022-10-24","arxiv_id":"2210.13391","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-euphemism-detection-in-few-shot-and","slug":"exploring-euphemism-detection-in-few-shot-and","title":"Exploring Euphemism Detection in Few-Shot and Zero-Shot Settings","date":"2022-10-24","arxiv_id":"2210.12926","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-better-your-syntax-the-better-your","title":"The Better Your Syntax, the Better Your Semantics? Probing Pretrained Language Models for the English Comparative Correlative","date":"2022-10-24","arxiv_id":"2210.13181","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bert-based-deep-learning-approach-for","title":"A BERT-based Deep Learning Approach for Reputation Analysis in Social Media","date":"2022-10-23","arxiv_id":"2211.01954","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-essay-scoring-using-transformers","title":"Data Augmentation for Automated Essay Scoring using Transformer Models","date":"2022-10-23","arxiv_id":"2210.12809","n_code_links":0,"syntology":null},{"paper":null,"slug":"discriminative-language-model-as-semantic","title":"Discriminative Language Model as Semantic Consistency Scorer for Prompt-based Few-Shot Text Classification","date":"2022-10-23","arxiv_id":"2210.12763","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-learning-pathologies-from-radiology","title":"Meta-learning Pathologies from Radiology Reports using Variance Aware Prototypical Networks","date":"2022-10-22","arxiv_id":"2210.13979","n_code_links":0,"syntology":null},{"paper":"/paper/amos-an-adam-style-optimizer-with-adaptive","slug":"amos-an-adam-style-optimizer-with-adaptive","title":"Amos: An Adam-style Optimizer with Adaptive Weight Decay towards Model-Oriented Scale","date":"2022-10-21","arxiv_id":"2210.11693","n_code_links":1,"syntology":{"ran":14,"of":18,"n_ran_checked":14,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["google-research/jestimator"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/discovering-differences-in-the-representation","slug":"discovering-differences-in-the-representation","title":"Discovering Differences in the Representation of People using Contextualized Semantic Axes","date":"2022-10-21","arxiv_id":"2210.12170","n_code_links":1,"syntology":null},{"paper":null,"slug":"littlebird-efficient-faster-longer","title":"LittleBird: Efficient Faster & Longer Transformer for Question Answering","date":"2022-10-21","arxiv_id":"2210.11870","n_code_links":0,"syntology":null},{"paper":"/paper/probing-with-noise-unpicking-the-warp-and","slug":"probing-with-noise-unpicking-the-warp-and","title":"Probing with Noise: Unpicking the Warp and Weft of Embeddings","date":"2022-10-21","arxiv_id":"2210.12206","n_code_links":1,"syntology":null},{"paper":null,"slug":"spabert-a-pretrained-language-model-from","title":"SpaBERT: A Pretrained Language Model from Geographic Data for Geo-Entity Representation","date":"2022-10-21","arxiv_id":"2210.12213","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-neural-network-model-for-1","slug":"a-unified-neural-network-model-for-1","title":"A Unified Neural Network Model for Readability Assessment with Feature Projection and Length-Balanced Loss","date":"2022-10-19","arxiv_id":"2210.10305","n_code_links":1,"syntology":null},{"paper":"/paper/biogpt-generative-pre-trained-transformer-for","slug":"biogpt-generative-pre-trained-transformer-for","title":"BioGPT: Generative Pre-trained Transformer for Biomedical Text Generation and Mining","date":"2022-10-19","arxiv_id":"2210.10341","n_code_links":4,"syntology":null},{"paper":"/paper/language-model-decomposition-quantifying-the","slug":"language-model-decomposition-quantifying-the","title":"Language Model Decomposition: Quantifying the Dependency and Correlation of Language Models","date":"2022-10-19","arxiv_id":"2210.10289","n_code_links":1,"syntology":null},{"paper":"/paper/tempo-accelerating-transformer-based-model","slug":"tempo-accelerating-transformer-based-model","title":"Tempo: Accelerating Transformer-Based Model Training through Memory Footprint Reduction","date":"2022-10-19","arxiv_id":"2210.10246","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uoft-ecosystem/tempo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/elastic-numerical-reasoning-with-adaptive","slug":"elastic-numerical-reasoning-with-adaptive","title":"ELASTIC: Numerical Reasoning with Adaptive Symbolic Compiler","date":"2022-10-18","arxiv_id":"2210.10105","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["neurasearch/neurips-2022-submission-3358"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-bert-do-it-controller-area-network","title":"CAN-BERT do it? Controller Area Network Intrusion Detection System based on BERT Language Model","date":"2022-10-17","arxiv_id":"2210.09439","n_code_links":0,"syntology":null},{"paper":"/paper/idna-abf-multi-scale-deep-biological-language","slug":"idna-abf-multi-scale-deep-biological-language","title":"iDNA-ABF: multi-scale deep biological language learning model for the interpretable prediction of DNA methylations","date":"2022-10-17","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"multi-granularity-argument-mining-in-legal","title":"Multi-granularity Argument Mining in Legal Texts","date":"2022-10-17","arxiv_id":"2210.09472","n_code_links":0,"syntology":null},{"paper":"/paper/using-bottleneck-adapters-to-identify-cancer","slug":"using-bottleneck-adapters-to-identify-cancer","title":"Using Bottleneck Adapters to Identify Cancer in Clinical Notes under Low-Resource Constraints","date":"2022-10-17","arxiv_id":"2210.09440","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-ranking-socio-political-texts-with","title":"Zero-Shot Ranking Socio-Political Texts with Transformer Language Models to Reduce Close Reading Time","date":"2022-10-17","arxiv_id":"2210.09179","n_code_links":0,"syntology":null},{"paper":null,"slug":"acoustic-aware-non-autoregressive-spell","title":"Acoustic-aware Non-autoregressive Spell Correction with Mask Sample Decoding","date":"2022-10-16","arxiv_id":"2210.08665","n_code_links":0,"syntology":null},{"paper":null,"slug":"ctcbert-advancing-hidden-unit-bert-with-ctc","title":"CTCBERT: Advancing Hidden-unit BERT with CTC Objectives","date":"2022-10-16","arxiv_id":"2210.08603","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-semantic-matching-through","title":"Improving Semantic Matching through Dependency-Enhanced Pre-trained Model with Adaptive Fusion","date":"2022-10-16","arxiv_id":"2210.08471","n_code_links":0,"syntology":null},{"paper":null,"slug":"aralegal-bert-a-pretrained-language-model-for","title":"AraLegal-BERT: A pretrained language model for Arabic Legal text","date":"2022-10-15","arxiv_id":"2210.08284","n_code_links":0,"syntology":null},{"paper":"/paper/dylora-parameter-efficient-tuning-of-pre","slug":"dylora-parameter-efficient-tuning-of-pre","title":"DyLoRA: Parameter Efficient Tuning of Pre-trained Models using Dynamic Search-Free Low-Rank Adaptation","date":"2022-10-14","arxiv_id":"2210.07558","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["huawei-noah/kd-nlp"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/kernel-whitening-overcome-dataset-bias-with","slug":"kernel-whitening-overcome-dataset-bias-with","title":"Kernel-Whitening: Overcome Dataset Bias with Isotropic Sentence Embedding","date":"2022-10-14","arxiv_id":"2210.07547","n_code_links":2,"syntology":null},{"paper":"/paper/overlooked-video-classification-in-weakly","slug":"overlooked-video-classification-in-weakly","title":"Overlooked Video Classification in Weakly Supervised Video Anomaly Detection","date":"2022-10-13","arxiv_id":"2210.06688","n_code_links":1,"syntology":null},{"paper":null,"slug":"squat-sharpness-and-quantization-aware","title":"SQuAT: Sharpness- and Quantization-Aware Training for BERT","date":"2022-10-13","arxiv_id":"2210.07171","n_code_links":0,"syntology":null},{"paper":null,"slug":"tone-prediction-and-orthographic-conversion","title":"Tone prediction and orthographic conversion for Basaa","date":"2022-10-13","arxiv_id":"2210.06986","n_code_links":0,"syntology":null},{"paper":"/paper/foundation-transformers","slug":"foundation-transformers","title":"Foundation Transformers","date":"2022-10-12","arxiv_id":"2210.06423","n_code_links":4,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"gmp-well-tuned-global-magnitude-pruning-can","title":"GMP*: Well-Tuned Gradual Magnitude Pruning Can Outperform Most BERT-Pruning Methods","date":"2022-10-12","arxiv_id":"2210.06384","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-text-style-transfer-via-style-masked","title":"On Text Style Transfer via Style Masked Language Models","date":"2022-10-12","arxiv_id":"2210.06394","n_code_links":0,"syntology":null},{"paper":"/paper/probing-commonsense-knowledge-in-pre-trained","slug":"probing-commonsense-knowledge-in-pre-trained","title":"Probing Commonsense Knowledge in Pre-trained Language Models with Sense-level Precision and Expanded Vocabulary","date":"2022-10-12","arxiv_id":"2210.06376","n_code_links":1,"syntology":null},{"paper":null,"slug":"rankt5-fine-tuning-t5-for-text-ranking-with","title":"RankT5: Fine-Tuning T5 for Text Ranking with Ranking Losses","date":"2022-10-12","arxiv_id":"2210.10634","n_code_links":0,"syntology":null},{"paper":"/paper/a-win-win-deal-towards-sparse-and-robust-pre","slug":"a-win-win-deal-towards-sparse-and-robust-pre","title":"A Win-win Deal: Towards Sparse and Robust Pre-trained Language Models","date":"2022-10-11","arxiv_id":"2210.05211","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["llyx97/sparse-and-robust-plm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-exploration-of-hierarchical-attention","title":"An Exploration of Hierarchical Attention Transformers for Efficient Long Document Classification","date":"2022-10-11","arxiv_id":"2210.05529","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-also-understands-text-prompting-clip-for","title":"CLIP also Understands Text: Prompting CLIP for Phrase Understanding","date":"2022-10-11","arxiv_id":"2210.05836","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-bert-has-an-accent-evaluating","title":"Multilingual BERT has an accent: Evaluating English influences on fluency in multilingual models","date":"2022-10-11","arxiv_id":"2210.05619","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-interpolation-of-contextualized-term","slug":"on-the-interpolation-of-contextualized-term","title":"On the Interpolation of Contextualized Term-based Ranking with BM25 for Query-by-Example Retrieval","date":"2022-10-11","arxiv_id":"2210.05512","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-use-of-semantically-aligned-speech","title":"On the Use of Semantically-Aligned Speech Representations for Spoken Language Understanding","date":"2022-10-11","arxiv_id":"2210.05291","n_code_links":0,"syntology":null},{"paper":"/paper/vote-n-rank-revision-of-benchmarking-with","slug":"vote-n-rank-revision-of-benchmarking-with","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","date":"2022-10-11","arxiv_id":"2210.05769","n_code_links":1,"syntology":{"ran":6,"of":17,"n_ran_checked":6,"n_instrument":0,"unverified":11,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["pragmaticslab/vote_and_rank"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":"/paper/deptweet-a-typology-for-social-media-texts-to","slug":"deptweet-a-typology-for-social-media-texts-to","title":"DEPTWEET: A Typology for Social Media Texts to Detect Depression Severities","date":"2022-10-10","arxiv_id":"2210.05372","n_code_links":1,"syntology":null},{"paper":"/paper/empowering-the-fact-checkers-automatic","slug":"empowering-the-fact-checkers-automatic","title":"Empowering the Fact-checkers! Automatic Identification of Claim Spans on Twitter","date":"2022-10-10","arxiv_id":"2210.04710","n_code_links":1,"syntology":null},{"paper":"/paper/multi-cls-bert-an-efficient-alternative-to","slug":"multi-cls-bert-an-efficient-alternative-to","title":"Multi-CLS BERT: An Efficient Alternative to Traditional Ensembling","date":"2022-10-10","arxiv_id":"2210.05043","n_code_links":1,"syntology":null},{"paper":"/paper/uncertainty-quantification-with-pre-trained","slug":"uncertainty-quantification-with-pre-trained","title":"Uncertainty Quantification with Pre-trained Language Models: A Large-Scale Empirical Analysis","date":"2022-10-10","arxiv_id":"2210.04714","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xiaoyuxin1002/uq-plm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/fairger-using-nlp-to-measure-support-for","slug":"fairger-using-nlp-to-measure-support-for","title":"Fine-Grained Detection of Solidarity for Women and Migrants in 155 Years of German Parliamentary Debates","date":"2022-10-09","arxiv_id":"2210.04359","n_code_links":2,"syntology":null},{"paper":null,"slug":"improve-transformer-pre-training-with","title":"Better Pre-Training by Reducing Representation Confusion","date":"2022-10-09","arxiv_id":"2210.04246","n_code_links":0,"syntology":null},{"paper":"/paper/spread-love-not-hate-undermining-the","slug":"spread-love-not-hate-undermining-the","title":"Spread Love Not Hate: Undermining the Importance of Hateful Pre-training for Hate Speech Detection","date":"2022-10-09","arxiv_id":"2210.04267","n_code_links":1,"syntology":null},{"paper":null,"slug":"kg-mtt-bert-knowledge-graph-enhanced-bert-for","title":"KG-MTT-BERT: Knowledge Graph Enhanced BERT for Multi-Type Medical Text Classification","date":"2022-10-08","arxiv_id":"2210.03970","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-task-adaptive-pretraining-for-dialogue","title":"On Task-Adaptive Pretraining for Dialogue Response Selection","date":"2022-10-08","arxiv_id":"2210.04073","n_code_links":0,"syntology":null},{"paper":null,"slug":"dabert-dual-attention-enhanced-bert-for","title":"DABERT: Dual Attention Enhanced BERT for Semantic Matching","date":"2022-10-07","arxiv_id":"2210.03454","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-quizzes-to-support-training-on","title":"Generating Quizzes to Support Training on Quality Management and Assurance in Space Science and Engineering","date":"2022-10-07","arxiv_id":"2210.03427","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-injected-prompt-based-fine-tuning","slug":"knowledge-injected-prompt-based-fine-tuning","title":"Knowledge Injected Prompt Based Fine-tuning for Multi-label Few-shot ICD Coding","date":"2022-10-07","arxiv_id":"2210.03304","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whaleloops/KEPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/uu-tax-at-semeval-2022-task-3-improving-the-1","slug":"uu-tax-at-semeval-2022-task-3-improving-the-1","title":"UU-Tax at SemEval-2022 Task 3: Improving the generalizability of language models for taxonomy classification through data augmentation","date":"2022-10-07","arxiv_id":"2210.03378","n_code_links":1,"syntology":null},{"paper":"/paper/bytetransformer-a-high-performance","slug":"bytetransformer-a-high-performance","title":"ByteTransformer: A High-Performance Transformer Boosted for Variable-Length Inputs","date":"2022-10-06","arxiv_id":"2210.03052","n_code_links":1,"syntology":null},{"paper":null,"slug":"explainable-verbal-deception-detection-using","title":"Explainable Verbal Deception Detection using Transformers","date":"2022-10-06","arxiv_id":"2210.03080","n_code_links":0,"syntology":null},{"paper":"/paper/improving-the-domain-adaptation-of-retrieval","slug":"improving-the-domain-adaptation-of-retrieval","title":"Improving the Domain Adaptation of Retrieval Augmented Generation (RAG) Models for Open Domain Question Answering","date":"2022-10-06","arxiv_id":"2210.02627","n_code_links":1,"syntology":null},{"paper":null,"slug":"join-chain-network-a-logical-reasoning-view","title":"Join-Chain Network: A Logical Reasoning View of the Multi-head Attention in Transformer","date":"2022-10-06","arxiv_id":"2210.02729","n_code_links":0,"syntology":null},{"paper":null,"slug":"matching-text-and-audio-embeddings-exploring","title":"Matching Text and Audio Embeddings: Exploring Transfer-learning Strategies for Language-based Audio Retrieval","date":"2022-10-06","arxiv_id":"2210.02833","n_code_links":0,"syntology":null},{"paper":null,"slug":"murag-multimodal-retrieval-augmented","title":"MuRAG: Multimodal Retrieval-Augmented Generator for Open Question Answering over Images and Text","date":"2022-10-06","arxiv_id":"2210.02928","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-preserving-text-classification-on-1","title":"Privacy-Preserving Text Classification on BERT Embeddings with Homomorphic Encryption","date":"2022-10-05","arxiv_id":"2210.02574","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-structured-dropout","title":"Revisiting Structured Dropout","date":"2022-10-05","arxiv_id":"2210.02570","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterization-of-effects-of-transfer","title":"The (In)Effectiveness of Intermediate Task Training For Domain Adaptation and Cross-Lingual Transfer Learning","date":"2022-10-03","arxiv_id":"2210.01091","n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-of-quantitative-values-in-abstractive","title":"Probing of Quantitative Values in Abstractive Summarization Models","date":"2022-10-03","arxiv_id":"2210.00667","n_code_links":0,"syntology":null},{"paper":null,"slug":"construction-and-evaluation-of-a-self","title":"Construction and Evaluation of a Self-Attention Model for Semantic Understanding of Sentence-Final Particles","date":"2022-10-01","arxiv_id":"2210.00282","n_code_links":0,"syntology":null},{"paper":"/paper/promptkg-a-prompt-learning-framework-for","slug":"promptkg-a-prompt-learning-framework-for","title":"LambdaKG: A Library for Pre-trained Language Model-Based Knowledge Graph Embeddings","date":"2022-10-01","arxiv_id":"2210.00305","n_code_links":2,"syntology":null},{"paper":null,"slug":"cefer-a-four-facets-framework-based-on","title":"CEFER: A Four Facets Framework based on Context and Emotion embedded features for Implicit and Explicit Emotion Recognition","date":"2022-09-28","arxiv_id":"2209.13999","n_code_links":0,"syntology":null},{"paper":"/paper/downstream-datasets-make-surprisingly-good","slug":"downstream-datasets-make-surprisingly-good","title":"Downstream Datasets Make Surprisingly Good Pretraining Corpora","date":"2022-09-28","arxiv_id":"2209.14389","n_code_links":1,"syntology":null},{"paper":null,"slug":"supervised-contrastive-learning-as-multi","title":"Supervised Contrastive Learning as Multi-Objective Optimization for Fine-Tuning Large Pre-trained Language Models","date":"2022-09-28","arxiv_id":"2209.14161","n_code_links":0,"syntology":null},{"paper":"/paper/yato-yet-another-deep-learning-based-text","slug":"yato-yet-another-deep-learning-based-text","title":"YATO: Yet Another deep learning based Text analysis Open toolkit","date":"2022-09-28","arxiv_id":"2209.13877","n_code_links":1,"syntology":null},{"paper":null,"slug":"extractive-question-answering-on-queries-in","title":"Extractive Question Answering on Queries in Hindi and Tamil","date":"2022-09-27","arxiv_id":"2210.06356","n_code_links":0,"syntology":null},{"paper":"/paper/outlier-suppression-pushing-the-limit-of-low","slug":"outlier-suppression-pushing-the-limit-of-low","title":"Outlier Suppression: Pushing the Limit of Low-bit Transformer Language Models","date":"2022-09-27","arxiv_id":"2209.13325","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wimh966/outlier_suppression"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/wikides-a-wikipedia-based-dataset-for","slug":"wikides-a-wikipedia-based-dataset-for","title":"WikiDes: A Wikipedia-Based Dataset for Generating Short Descriptions from Paragraphs","date":"2022-09-27","arxiv_id":"2209.13101","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-ever-larger-octopi-still-amplify-reporting","title":"Do ever larger octopi still amplify reporting biases? Evidence from judgments of typical colour","date":"2022-09-26","arxiv_id":"2209.12786","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-simple-and-efficient-task-adaptive","title":"Towards Simple and Efficient Task-Adaptive Pre-training for Text Classification","date":"2022-09-26","arxiv_id":"2209.12943","n_code_links":0,"syntology":null},{"paper":null,"slug":"bigger-faster-two-stage-neural-architecture","title":"SpeedLimit: Neural Architecture Search for Quantized Transformer Models","date":"2022-09-25","arxiv_id":"2209.12127","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentiment-analysis-on-inflation-after-covid","title":"Sentiment Analysis on Inflation after Covid-19","date":"2022-09-25","arxiv_id":"2209.14737","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-transformer-models-effectively-detect","title":"Can Transformer Models Effectively Detect Software Aspects in StackOverflow Discussion?","date":"2022-09-24","arxiv_id":"2209.12065","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-chess-with-language-models-and","title":"Learning Chess With Language Models and Transformers","date":"2022-09-24","arxiv_id":"2209.11902","n_code_links":0,"syntology":null},{"paper":null,"slug":"idea-interactive-double-attentions-from-label","title":"IDEA: Interactive DoublE Attentions from Label Embedding for Text Classification","date":"2022-09-23","arxiv_id":"2209.11407","n_code_links":0,"syntology":null},{"paper":"/paper/adaptation-of-domain-specific-transformer","slug":"adaptation-of-domain-specific-transformer","title":"Adaptation of domain-specific transformer models with text oversampling for sentiment analysis of social media posts on Covid-19 vaccines","date":"2022-09-22","arxiv_id":"2209.10966","n_code_links":1,"syntology":null}],"record_sha256":"2ac2a2c5907aeb0642230ab562b611327d4ff1e997f321749a1b7dab2ed31d15","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}