{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/30","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":30,"pages_in_order":71,"rows_per_page":100,"rows":[2901,3000],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/29","next":"/method/linear-warmup-with-linear-decay/papers/31","papers":[{"paper":"/paper/bertino-an-italian-distilbert-model","slug":"bertino-an-italian-distilbert-model","title":"BERTino: an Italian DistilBERT model","date":"2023-03-31","arxiv_id":"2303.18121","n_code_links":1,"syntology":null},{"paper":null,"slug":"extracting-thyroid-nodules-characteristics","title":"Extracting Thyroid Nodules Characteristics from Ultrasound Reports Using Transformer-based Natural Language Processing Methods","date":"2023-03-31","arxiv_id":"2304.00115","n_code_links":0,"syntology":null},{"paper":null,"slug":"jobham-place-with-smart-recommend-job-options","title":"JobHam-place with smart recommend job options and candidate filtering options","date":"2023-03-31","arxiv_id":"2303.17930","n_code_links":0,"syntology":null},{"paper":null,"slug":"quick-dense-retrievers-consume-kale-post","title":"Quick Dense Retrievers Consume KALE: Post Training Kullback Leibler Alignment of Embeddings for Asymmetrical dual encoders","date":"2023-03-31","arxiv_id":"2304.01016","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-gpt-and-bert-based-models-on","title":"Evaluation of GPT and BERT-based models on identifying protein-protein interactions in biomedical text","date":"2023-03-30","arxiv_id":"2303.17728","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-bert-with-character-level-noise","title":"Fine-Tuning BERT with Character-Level Noise for Zero-Shot Transfer to Dialects and Closely-Related Languages","date":"2023-03-30","arxiv_id":"2303.17683","n_code_links":0,"syntology":null},{"paper":null,"slug":"oberta-improving-sparse-transfer-learning-via","title":"oBERTa: Improving Sparse Transfer Learning via improved initialization, distillation, and pruning regimes","date":"2023-03-30","arxiv_id":"2303.17612","n_code_links":0,"syntology":null},{"paper":"/paper/bert4eth-a-pre-trained-transformer-for","slug":"bert4eth-a-pre-trained-transformer-for","title":"BERT4ETH: A Pre-trained Transformer for Ethereum Fraud Detection","date":"2023-03-29","arxiv_id":"2303.18138","n_code_links":1,"syntology":null},{"paper":"/paper/larger-probes-tell-a-different-story","slug":"larger-probes-tell-a-different-story","title":"Larger Probes Tell a Different Story: Extending Psycholinguistic Datasets Via In-Context Learning","date":"2023-03-29","arxiv_id":"2303.16445","n_code_links":1,"syntology":null},{"paper":null,"slug":"textmi-textualize-multimodal-information-for","title":"TextMI: Textualize Multimodal Information for Integrating Non-verbal Cues in Pre-trained Language Models","date":"2023-03-27","arxiv_id":"2303.15430","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-multimodal-sentiment-analysis-via","title":"Exploring Multimodal Sentiment Analysis via CBAM Attention and Double-layer BiLSTM Architecture","date":"2023-03-26","arxiv_id":"2303.14708","n_code_links":0,"syntology":null},{"paper":"/paper/indonesian-text-to-image-synthesis-with","slug":"indonesian-text-to-image-synthesis-with","title":"Indonesian Text-to-Image Synthesis with Sentence-BERT and FastGAN","date":"2023-03-25","arxiv_id":"2303.14517","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatio-temporal-driven-attention-graph-neural","title":"Spatio-Temporal driven Attention Graph Neural Network with Block Adjacency matrix (STAG-NN-BA)","date":"2023-03-25","arxiv_id":"2303.14322","n_code_links":0,"syntology":null},{"paper":null,"slug":"depression-detection-in-social-media-posts","title":"Depression detection in social media posts using affective and social norm features","date":"2023-03-24","arxiv_id":"2303.14279","n_code_links":0,"syntology":null},{"paper":"/paper/sigmorphon-2023-shared-task-of-interlinear","slug":"sigmorphon-2023-shared-task-of-interlinear","title":"SIGMORPHON 2023 Shared Task of Interlinear Glossing: Baseline Model","date":"2023-03-24","arxiv_id":"2303.14234","n_code_links":1,"syntology":null},{"paper":null,"slug":"toward-open-domain-slot-filling-via-self","title":"Toward Open-domain Slot Filling via Self-supervised Co-training","date":"2023-03-24","arxiv_id":"2303.13801","n_code_links":0,"syntology":null},{"paper":"/paper/where-to-go-next-for-recommender-systems-id","slug":"where-to-go-next-for-recommender-systems-id","title":"Where to Go Next for Recommender Systems? ID- vs. Modality-based Recommender Models Revisited","date":"2023-03-24","arxiv_id":"2303.13835","n_code_links":1,"syntology":null},{"paper":"/paper/a-novel-patent-similarity-measurement","slug":"a-novel-patent-similarity-measurement","title":"A Novel Patent Similarity Measurement Methodology: Semantic Distance and Technological Distance","date":"2023-03-23","arxiv_id":"2303.16767","n_code_links":1,"syntology":null},{"paper":"/paper/retrieval-augmented-classification-with","slug":"retrieval-augmented-classification-with","title":"Retrieval-Augmented Classification with Decoupled Representation","date":"2023-03-23","arxiv_id":"2303.13065","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-the-generalizability-of-deep","title":"Analyzing the Generalizability of Deep Contextualized Language Representations For Text Classification","date":"2023-03-22","arxiv_id":"2303.12936","n_code_links":0,"syntology":null},{"paper":null,"slug":"generate-labeled-training-data-using-prompt","title":"Generate labeled training data using Prompt Programming and GPT-3. An example of Big Five Personality Classification","date":"2023-03-22","arxiv_id":"2303.12279","n_code_links":0,"syntology":null},{"paper":null,"slug":"tron-transformer-neural-network-acceleration","title":"TRON: Transformer Neural Network Acceleration with Non-Coherent Silicon Photonics","date":"2023-03-22","arxiv_id":"2303.12914","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-climatebert-transformer-with","title":"Fine-tuning ClimateBert transformer with ClimaText for the disclosure analysis of climate-related financial risks","date":"2023-03-21","arxiv_id":"2303.13373","n_code_links":0,"syntology":null},{"paper":"/paper/is-bert-blind-exploring-the-effect-of-vision","slug":"is-bert-blind-exploring-the-effect-of-vision","title":"Is BERT Blind? Exploring the Effect of Vision-and-Language Pretraining on Visual Language Understanding","date":"2023-03-21","arxiv_id":"2303.12513","n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-pre-training-framework-for","title":"Multimodal Pre-training Framework for Sequential Recommendation via Contrastive Learning","date":"2023-03-21","arxiv_id":"2303.11879","n_code_links":0,"syntology":null},{"paper":"/paper/character-word-or-both-revisiting-the","slug":"character-word-or-both-revisiting-the","title":"Character, Word, or Both? Revisiting the Segmentation Granularity for Chinese Pre-trained Language Models","date":"2023-03-20","arxiv_id":"2303.10893","n_code_links":1,"syntology":null},{"paper":"/paper/ctran-cnn-transformer-based-network-for","slug":"ctran-cnn-transformer-based-network-for","title":"CTRAN: CNN-Transformer-based Network for Natural Language Understanding","date":"2023-03-19","arxiv_id":"2303.10606","n_code_links":1,"syntology":null},{"paper":null,"slug":"paco-provocation-involving-action-culture-and","title":"PACO: Provocation Involving Action, Culture, and Oppression","date":"2023-03-19","arxiv_id":"2303.12808","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-pre-trained-language","slug":"an-empirical-study-of-pre-trained-language","title":"An Empirical Study of Pre-trained Language Models in Simple Knowledge Graph Question Answering","date":"2023-03-18","arxiv_id":"2303.10368","n_code_links":1,"syntology":null},{"paper":null,"slug":"noisyhate-benchmarking-content-moderation","title":"NoisyHate: Mining Online Human-Written Perturbations for Realistic Robustness Benchmarking of Content Moderation Models","date":"2023-03-18","arxiv_id":"2303.10430","n_code_links":0,"syntology":null},{"paper":"/paper/gadformer-an-attention-based-model-for-group","slug":"gadformer-an-attention-based-model-for-group","title":"GADformer: A Transparent Transformer Model for Group Anomaly Detection on Trajectories","date":"2023-03-17","arxiv_id":"2303.09841","n_code_links":1,"syntology":null},{"paper":"/paper/trained-on-100-million-words-and-still-in","slug":"trained-on-100-million-words-and-still-in","title":"Trained on 100 million words and still in shape: BERT meets British National Corpus","date":"2023-03-17","arxiv_id":"2303.09859","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":{"repos":["ltgoslo/ltg-bert"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"block-wise-bit-compression-of-transformer","title":"Block-wise Bit-Compression of Transformer-based Models","date":"2023-03-16","arxiv_id":"2303.09184","n_code_links":0,"syntology":null},{"paper":"/paper/jump-to-conclusions-short-cutting","slug":"jump-to-conclusions-short-cutting","title":"Jump to Conclusions: Short-Cutting Transformers With Linear Transformations","date":"2023-03-16","arxiv_id":"2303.09435","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sashayd/mat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"measuring-improvement-of-f-1-scores-in","title":"Measuring Improvement of F$_1$-Scores in Detection of Self-Admitted Technical Debt","date":"2023-03-16","arxiv_id":"2303.09617","n_code_links":0,"syntology":null},{"paper":"/paper/smartbert-a-promotion-of-dynamic-early","slug":"smartbert-a-promotion-of-dynamic-early","title":"SmartBERT: A Promotion of Dynamic Early Exiting Mechanism for Accelerating BERT Inference","date":"2023-03-16","arxiv_id":"2303.09266","n_code_links":0,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"efficient-uncertainty-estimation-with","title":"Efficient Uncertainty Estimation with Gaussian Process for Reliable Dialog Response Retrieval","date":"2023-03-15","arxiv_id":"2303.08599","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-transformers-parse-while-predicting-the","title":"Do Transformers Parse while Predicting the Masked Word?","date":"2023-03-14","arxiv_id":"2303.08117","n_code_links":0,"syntology":null},{"paper":null,"slug":"features-matching-using-natural-language","title":"Features matching using natural language processing","date":"2023-03-14","arxiv_id":"2303.12804","n_code_links":0,"syntology":null},{"paper":null,"slug":"finding-the-needle-in-a-haystack-unsupervised","title":"Finding the Needle in a Haystack: Unsupervised Rationale Extraction from Long Text Classifiers","date":"2023-03-14","arxiv_id":"2303.07991","n_code_links":0,"syntology":null},{"paper":null,"slug":"medbert-de-a-comprehensive-german-bert-model","title":"MEDBERT.de: A Comprehensive German BERT Model for the Medical Domain","date":"2023-03-14","arxiv_id":"2303.08179","n_code_links":0,"syntology":null},{"paper":"/paper/neuro-symbolic-commonsense-social-reasoning","slug":"neuro-symbolic-commonsense-social-reasoning","title":"Neuro-symbolic Commonsense Social Reasoning","date":"2023-03-14","arxiv_id":"2303.08264","n_code_links":3,"syntology":null},{"paper":null,"slug":"deep-learning-approach-for-classifying-the","title":"Deep Learning Approach for Classifying the Aggressive Comments on Social Media: Machine Translated Data Vs Real Life Data","date":"2023-03-13","arxiv_id":"2303.07484","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-approaches-to-sentiment","title":"Transformer-based approaches to Sentiment Detection","date":"2023-03-13","arxiv_id":"2303.07292","n_code_links":0,"syntology":null},{"paper":"/paper/luke-graph-a-transformer-based-approach-with","slug":"luke-graph-a-transformer-based-approach-with","title":"LUKE-Graph: A Transformer-based Approach with Gated Relational Graph Attention for Cloze-style Reading Comprehension","date":"2023-03-12","arxiv_id":"2303.06675","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-in-hospital-meta-information-useful-for","title":"Is In-hospital Meta-information Useful for Abstractive Discharge Summary Generation?","date":"2023-03-10","arxiv_id":"2303.06002","n_code_links":0,"syntology":null},{"paper":null,"slug":"research-on-cpi-prediction-based-on-natural","title":"Research on CPI Prediction Based on Natural Language Processing","date":"2023-03-10","arxiv_id":"2303.05666","n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-survey-of-ai-generated","slug":"a-comprehensive-survey-of-ai-generated","title":"A Comprehensive Survey of AI-Generated Content (AIGC): A History of Generative AI from GAN to ChatGPT","date":"2023-03-07","arxiv_id":"2303.04226","n_code_links":1,"syntology":null},{"paper":null,"slug":"adelt-transpilation-between-deep-learning","title":"ADELT: Transpilation Between Deep Learning Frameworks","date":"2023-03-07","arxiv_id":"2303.03593","n_code_links":0,"syntology":null},{"paper":null,"slug":"classifying-text-based-conspiracy-tweets","title":"Classifying Text-Based Conspiracy Tweets related to COVID-19 using Contextualized Word Embeddings","date":"2023-03-07","arxiv_id":"2303.03706","n_code_links":0,"syntology":null},{"paper":null,"slug":"german-bert-model-for-legal-named-entity","title":"German BERT Model for Legal Named Entity Recognition","date":"2023-03-07","arxiv_id":"2303.05388","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-free-structured-pruning-with","title":"Gradient-Free Structured Pruning with Unlabeled Data","date":"2023-03-07","arxiv_id":"2303.04185","n_code_links":0,"syntology":null},{"paper":null,"slug":"video-question-answering-using-clip-guided","title":"Video Question Answering Using CLIP-Guided Visual-Text Attention","date":"2023-03-06","arxiv_id":"2303.03131","n_code_links":0,"syntology":null},{"paper":"/paper/robust-affine-feature-matching-via-quadratic","slug":"robust-affine-feature-matching-via-quadratic","title":"Robust affine point matching via quadratic assignment on Grassmannians","date":"2023-03-05","arxiv_id":"2303.02698","n_code_links":3,"syntology":null},{"paper":null,"slug":"early-warning-signals-of-social-instabilities","title":"Early Warning Signals of Social Instabilities in Twitter Data","date":"2023-03-03","arxiv_id":"2303.05401","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-data-augmentation-methods-on-social","title":"Exploring Data Augmentation Methods on Social Media Corpora","date":"2023-03-03","arxiv_id":"2303.02198","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-label-classification-of-artificial","title":"Multi label classification of Artificial Intelligence related patents using Modified D2SBERT and Sentence Attention mechanism","date":"2023-03-03","arxiv_id":"2303.03165","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-trained-model-representations-and-their","title":"Pre-trained Model Representations and their Robustness against Noise for Speech Emotion Analysis","date":"2023-03-03","arxiv_id":"2303.03177","n_code_links":0,"syntology":null},{"paper":"/paper/trojtext-test-time-invisible-textual-trojan","slug":"trojtext-test-time-invisible-textual-trojan","title":"TrojText: Test-time Invisible Textual Trojan Insertion","date":"2023-03-03","arxiv_id":"2303.02242","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ucf-ml-research/trojtext"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"will-affective-computing-emerge-from","title":"Will Affective Computing Emerge from Foundation Models and General AI? A First Evaluation on ChatGPT","date":"2023-03-03","arxiv_id":"2303.03186","n_code_links":0,"syntology":null},{"paper":null,"slug":"adopting-the-multi-answer-questioning-task","title":"Adopting the Multi-answer Questioning Task with an Auxiliary Metric for Extreme Multi-label Text Classification Utilizing the Label Hierarchy","date":"2023-03-02","arxiv_id":"2303.01064","n_code_links":0,"syntology":null},{"paper":"/paper/can-bert-refrain-from-forgetting-on","slug":"can-bert-refrain-from-forgetting-on","title":"Can BERT Refrain from Forgetting on Sequential Tasks? A Probing Study","date":"2023-03-02","arxiv_id":"2303.01081","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-parameter-efficient-transfer","slug":"evaluating-parameter-efficient-transfer","title":"Evaluating Parameter-Efficient Transfer Learning Approaches on SURE Benchmark for Speech Understanding","date":"2023-03-02","arxiv_id":"2303.03267","n_code_links":1,"syntology":null},{"paper":"/paper/ino-at-factify-2-structure-coherence-based","slug":"ino-at-factify-2-structure-coherence-based","title":"INO at Factify 2: Structure Coherence based Multi-Modal Fact Verification","date":"2023-03-02","arxiv_id":"2303.01510","n_code_links":1,"syntology":null},{"paper":"/paper/sparse-moe-as-the-new-dropout-scaling-dense","slug":"sparse-moe-as-the-new-dropout-scaling-dense","title":"Sparse MoE as the New Dropout: Scaling Dense and Self-Slimmable Transformers","date":"2023-03-02","arxiv_id":"2303.01610","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vita-group/random-moe-as-dropout"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"competence-based-analysis-of-language-models","title":"Competence-Based Analysis of Language Models","date":"2023-03-01","arxiv_id":"2303.00333","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-adapted-large-language-models-for","title":"Domain-adapted large language models for classifying nuclear medicine reports","date":"2023-03-01","arxiv_id":"2303.01258","n_code_links":0,"syntology":null},{"paper":null,"slug":"toxvis-enabling-interpretability-of-implicit","title":"ToxVis: Enabling Interpretability of Implicit vs. Explicit Toxicity Detection Models with Interactive Visualization","date":"2023-03-01","arxiv_id":"2303.09402","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatically-classifying-emotions-based-on","title":"Automatically Classifying Emotions based on Text: A Comparative Exploration of Different Datasets","date":"2023-02-28","arxiv_id":"2302.14727","n_code_links":0,"syntology":null},{"paper":"/paper/text-classification-dataset-and-analysis-for","slug":"text-classification-dataset-and-analysis-for","title":"Text classification dataset and analysis for Uzbek language","date":"2023-02-28","arxiv_id":"2302.14494","n_code_links":1,"syntology":null},{"paper":null,"slug":"weighted-sampling-for-masked-language","title":"Weighted Sampling for Masked Language Modeling","date":"2023-02-28","arxiv_id":"2302.14225","n_code_links":0,"syntology":null},{"paper":null,"slug":"elementwise-language-representation","title":"Elementwise Language Representation","date":"2023-02-27","arxiv_id":"2302.13475","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-auxiliary-tasks-in-multimodal-fusion-of","title":"Using Auxiliary Tasks In Multimodal Fusion Of Wav2vec 2.0 And BERT For Multimodal Emotion Recognition","date":"2023-02-27","arxiv_id":"2302.13661","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-ensemble-architecture-for","slug":"efficient-ensemble-architecture-for","title":"Efficient Ensemble for Multimodal Punctuation Restoration using Time-Delay Neural Network","date":"2023-02-26","arxiv_id":"2302.13376","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-attention-requires-bounded-entries","title":"Fast Attention Requires Bounded Entries","date":"2023-02-26","arxiv_id":"2302.13214","n_code_links":0,"syntology":null},{"paper":"/paper/hulat-at-semeval-2023-task-10-data","slug":"hulat-at-semeval-2023-task-10-data","title":"HULAT at SemEval-2023 Task 10: Data augmentation for pre-trained transformers applied to the detection of sexism in social media","date":"2023-02-24","arxiv_id":"2302.12840","n_code_links":1,"syntology":null},{"paper":"/paper/mux-plms-pre-training-language-models-with","slug":"mux-plms-pre-training-language-models-with","title":"MUX-PLMs: Data Multiplexing for High-throughput Language Models","date":"2023-02-24","arxiv_id":"2302.12441","n_code_links":1,"syntology":null},{"paper":"/paper/window-transformer-for-dialogue-document-a","slug":"window-transformer-for-dialogue-document-a","title":"Window transformer for dialogue document: a joint framework for causal emotion entailment","date":"2023-02-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/teacher-intervention-improving-convergence-of","slug":"teacher-intervention-improving-convergence-of","title":"Teacher Intervention: Improving Convergence of Quantization Aware Training for Ultra-Low Precision Transformers","date":"2023-02-23","arxiv_id":"2302.11812","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marsjacobs/ti-kd-qat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"solution-for-the-epo-codefest-on-green","title":"Solution for the EPO CodeFest on Green Plastics: Hierarchical multi-label classification of patents relating to green plastics using deep learning","date":"2023-02-22","arxiv_id":"2302.13784","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-classification-reliability-of-nlp","title":"Boosting classification reliability of NLP transformer models in the long run","date":"2023-02-20","arxiv_id":"2302.10016","n_code_links":0,"syntology":null},{"paper":"/paper/large-scale-multi-modal-pre-trained-models-a","slug":"large-scale-multi-modal-pre-trained-models-a","title":"Large-scale Multi-Modal Pre-trained Models: A Comprehensive Survey","date":"2023-02-20","arxiv_id":"2302.10035","n_code_links":1,"syntology":null},{"paper":"/paper/can-chatgpt-understand-too-a-comparative","slug":"can-chatgpt-understand-too-a-comparative","title":"Can ChatGPT Understand Too? A Comparative Study on ChatGPT and Fine-tuned BERT","date":"2023-02-19","arxiv_id":"2302.10198","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-effectiveness-of-pre-trained","slug":"evaluating-the-effectiveness-of-pre-trained","title":"Evaluating the Effectiveness of Pre-trained Language Models in Predicting the Helpfulness of Online Product Reviews","date":"2023-02-19","arxiv_id":"2302.10199","n_code_links":1,"syntology":null},{"paper":"/paper/text-classification-in-the-wild-a-large-scale","slug":"text-classification-in-the-wild-a-large-scale","title":"Text Classification in the Wild: a Large-scale Long-tailed Name Normalization Dataset","date":"2023-02-19","arxiv_id":"2302.09509","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-survey-on-pretrained","title":"A Comprehensive Survey on Pretrained Foundation Models: A History from BERT to ChatGPT","date":"2023-02-18","arxiv_id":"2302.09419","n_code_links":0,"syntology":null},{"paper":null,"slug":"hate-speech-and-offensive-language-detection-2","title":"Hate Speech and Offensive Language Detection using an Emotion-aware Shared Encoder","date":"2023-02-17","arxiv_id":"2302.08777","n_code_links":0,"syntology":null},{"paper":null,"slug":"vita-a-vision-transformer-inference","title":"ViTA: A Vision Transformer Inference Accelerator for Edge Applications","date":"2023-02-17","arxiv_id":"2302.09108","n_code_links":0,"syntology":null},{"paper":null,"slug":"foundation-models-for-natural-language","title":"Foundation Models for Natural Language Processing -- Pre-trained Language Models Integrating Media","date":"2023-02-16","arxiv_id":"2302.08575","n_code_links":0,"syntology":null},{"paper":"/paper/marich-a-query-efficient-distributionally","slug":"marich-a-query-efficient-distributionally","title":"Marich: A Query-efficient Distributionally Equivalent Model Extraction Attack using Public Data","date":"2023-02-16","arxiv_id":"2302.08466","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["debabrota-basu/marich"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/retrieval-augmented-image-captioning","slug":"retrieval-augmented-image-captioning","title":"Retrieval-augmented Image Captioning","date":"2023-02-16","arxiv_id":"2302.08268","n_code_links":1,"syntology":null},{"paper":null,"slug":"syntactic-structure-processing-in-the-brain","title":"Syntactic Structure Processing in the Brain while Listening","date":"2023-02-16","arxiv_id":"2302.08589","n_code_links":0,"syntology":null},{"paper":null,"slug":"commonsense-reasoning-for-conversational-ai-a","title":"Commonsense Reasoning for Conversational AI: A Survey of the State of the Art","date":"2023-02-15","arxiv_id":"2302.07926","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-optimal-compression-joint-pruning-and","title":"Towards Optimal Compression: Joint Pruning and Quantization","date":"2023-02-15","arxiv_id":"2302.07612","n_code_links":0,"syntology":null},{"paper":"/paper/a-modern-look-at-the-relationship-between","slug":"a-modern-look-at-the-relationship-between","title":"A Modern Look at the Relationship between Sharpness and Generalization","date":"2023-02-14","arxiv_id":"2302.07011","n_code_links":1,"syntology":null},{"paper":"/paper/a-psycholinguistic-analysis-of-bert-s","slug":"a-psycholinguistic-analysis-of-bert-s","title":"A Psycholinguistic Analysis of BERT's Representations of Compounds","date":"2023-02-14","arxiv_id":"2302.07232","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-category-structure-with-contextual","title":"Exploring Category Structure with Contextual Language Models and Lexical Semantic Networks","date":"2023-02-14","arxiv_id":"2302.06942","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-learning-approaches-for-classifying","title":"Few-shot learning approaches for classifying low resource domain specific software requirements","date":"2023-02-14","arxiv_id":"2302.06951","n_code_links":0,"syntology":null},{"paper":"/paper/reveal-the-unknown-out-of-knowledge-base","slug":"reveal-the-unknown-out-of-knowledge-base","title":"Reveal the Unknown: Out-of-Knowledge-Base Mention Discovery with Entity Linking","date":"2023-02-14","arxiv_id":"2302.07189","n_code_links":3,"syntology":null},{"paper":null,"slug":"linguistic-ambiguity-analysis-in-chatgpt","title":"Linguistic ambiguity analysis in ChatGPT","date":"2023-02-13","arxiv_id":"2302.06426","n_code_links":0,"syntology":null}],"record_sha256":"cb8f7117ee8bd45b6b0b268ba6ebd59d8f7f4ef6f9ce2fbf867fbac8ca2caef5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}