{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/28","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":28,"pages_in_order":71,"rows_per_page":100,"rows":[2701,2800],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/27","next":"/method/linear-warmup-with-linear-decay/papers/29","papers":[{"paper":null,"slug":"stack-over-flowing-with-results-the-case-for","title":"Skill over Scale: The Case for Medium, Domain-Specific Models for SE","date":"2023-06-05","arxiv_id":"2306.03268","n_code_links":0,"syntology":null},{"paper":"/paper/using-sequences-of-life-events-to-predict","slug":"using-sequences-of-life-events-to-predict","title":"Using Sequences of Life-events to Predict Human Lives","date":"2023-06-05","arxiv_id":"2306.03009","n_code_links":2,"syntology":null},{"paper":"/paper/spellmapper-a-non-autoregressive-neural","slug":"spellmapper-a-non-autoregressive-neural","title":"SpellMapper: A non-autoregressive neural spellchecker for ASR customization with candidate retrieval based on n-gram mappings","date":"2023-06-04","arxiv_id":"2306.02317","n_code_links":1,"syntology":null},{"paper":null,"slug":"financial-sentiment-analysis-using-finbert","title":"Financial sentiment analysis using FinBERT with application in predicting stock movement","date":"2023-06-03","arxiv_id":"2306.02136","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilegalpile-a-689gb-multilingual-legal","title":"MultiLegalPile: A 689GB Multilingual Legal Corpus","date":"2023-06-03","arxiv_id":"2306.02069","n_code_links":0,"syntology":null},{"paper":null,"slug":"concurrent-classifier-error-detection-cced-in","title":"Concurrent Classifier Error Detection (CCED) in Large Scale Machine Learning Systems","date":"2023-06-02","arxiv_id":"2306.01820","n_code_links":0,"syntology":null},{"paper":null,"slug":"establishment-of-nlp-based-greenwashing","title":"Establishment of NLP-Based Greenwashing Pattern Detection Service","date":"2023-06-02","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/gateon-an-unsupervised-method-for-large-scale","slug":"gateon-an-unsupervised-method-for-large-scale","title":"Context selectivity with dynamic availability enables lifelong continual learning","date":"2023-06-02","arxiv_id":"2306.01690","n_code_links":1,"syntology":null},{"paper":null,"slug":"word-embeddings-for-banking-industry","title":"Word Embeddings for Banking Industry","date":"2023-06-02","arxiv_id":"2306.01807","n_code_links":0,"syntology":null},{"paper":"/paper/adapting-pre-trained-language-models-to","slug":"adapting-pre-trained-language-models-to","title":"Adapting Pre-trained Language Models to Vision-Language Tasks via Dynamic Visual Prompting","date":"2023-06-01","arxiv_id":"2306.00409","n_code_links":1,"syntology":null},{"paper":null,"slug":"boosting-the-performance-of-transformer","title":"Boosting the Performance of Transformer Architectures for Semantic Textual Similarity","date":"2023-06-01","arxiv_id":"2306.00708","n_code_links":0,"syntology":null},{"paper":"/paper/column-type-annotation-using-chatgpt","slug":"column-type-annotation-using-chatgpt","title":"Column Type Annotation using ChatGPT","date":"2023-06-01","arxiv_id":"2306.00745","n_code_links":1,"syntology":null},{"paper":null,"slug":"feature-engineering-based-detection-of-buffer","title":"Feature Engineering-Based Detection of Buffer Overflow Vulnerability in Source Code Using Neural Networks","date":"2023-06-01","arxiv_id":"2306.07981","n_code_links":0,"syntology":null},{"paper":"/paper/make-pre-trained-model-reversible-from-1","slug":"make-pre-trained-model-reversible-from-1","title":"Make Pre-trained Model Reversible: From Parameter to Memory Efficient Fine-Tuning","date":"2023-06-01","arxiv_id":"2306.00477","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["baohaoliao/mefts"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":"/paper/training-free-neural-architecture-search-for","slug":"training-free-neural-architecture-search-for","title":"Training-free Neural Architecture Search for RNNs and Transformers","date":"2023-06-01","arxiv_id":"2306.00288","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaronserianni/training-free-nas"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ucas-iie-nlp-at-semeval-2023-task-12","slug":"ucas-iie-nlp-at-semeval-2023-task-12","title":"UCAS-IIE-NLP at SemEval-2023 Task 12: Enhancing Generalization of Multilingual BERT for Low-resource Sentiment Analysis","date":"2023-06-01","arxiv_id":"2306.01093","n_code_links":1,"syntology":null},{"paper":"/paper/a-global-context-mechanism-for-sequence","slug":"a-global-context-mechanism-for-sequence","title":"Supplementary Features of BiLSTM for Enhanced Sequence Labeling","date":"2023-05-31","arxiv_id":"2305.19928","n_code_links":1,"syntology":null},{"paper":null,"slug":"building-extractive-question-answering-system","title":"Building Extractive Question Answering System to Support Human-AI Health Coaching Model for Sleep Domain","date":"2023-05-31","arxiv_id":"2305.19707","n_code_links":0,"syntology":null},{"paper":null,"slug":"catalysis-distillation-neural-network-for-the","title":"Catalysis distillation neural network for the few shot open catalyst challenge","date":"2023-05-31","arxiv_id":"2305.19545","n_code_links":0,"syntology":null},{"paper":"/paper/deepmerge-deep-learning-based-region-merging","slug":"deepmerge-deep-learning-based-region-merging","title":"DeepMerge: Deep-Learning-Based Region-Merging for Image Segmentation","date":"2023-05-31","arxiv_id":"2305.19787","n_code_links":1,"syntology":null},{"paper":"/paper/xphonebert-a-pre-trained-multilingual-model","slug":"xphonebert-a-pre-trained-multilingual-model","title":"XPhoneBERT: A Pre-trained Multilingual Model for Phoneme Representations for Text-to-Speech","date":"2023-05-31","arxiv_id":"2305.19709","n_code_links":2,"syntology":{"ran":14,"of":18,"n_ran_checked":14,"n_instrument":0,"unverified":4,"pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["vinairesearch/xphonebert"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"explaining-hate-speech-classification-with","title":"Explaining Hate Speech Classification with Model Agnostic Methods","date":"2023-05-30","arxiv_id":"2306.00021","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-models-in-construction-industry","title":"GPT Models in Construction Industry: Opportunities, Limitations, and a Use Case Validation","date":"2023-05-30","arxiv_id":"2305.18997","n_code_links":0,"syntology":null},{"paper":null,"slug":"multitask-learning-for-recognizing-stress-and","title":"Multitask learning for recognizing stress and depression in social media","date":"2023-05-30","arxiv_id":"2305.18907","n_code_links":0,"syntology":null},{"paper":null,"slug":"prequant-a-task-agnostic-quantization","title":"PreQuant: A Task-agnostic Quantization Approach for Pre-trained Language Models","date":"2023-05-30","arxiv_id":"2306.00014","n_code_links":0,"syntology":null},{"paper":null,"slug":"research-on-multilingual-news-clustering","title":"Research on Multilingual News Clustering Based on Cross-Language Word Embeddings","date":"2023-05-30","arxiv_id":"2305.18880","n_code_links":0,"syntology":null},{"paper":"/paper/scone-benchmarking-negation-reasoning-in","slug":"scone-benchmarking-negation-reasoning-in","title":"ScoNe: Benchmarking Negation Reasoning in Language Models With Fine-Tuning and In-Context Learning","date":"2023-05-30","arxiv_id":"2305.19426","n_code_links":1,"syntology":null},{"paper":null,"slug":"abstractive-summarization-as-augmentation-for","title":"Abstractive Summarization as Augmentation for Document-Level Event Detection","date":"2023-05-29","arxiv_id":"2305.18023","n_code_links":0,"syntology":null},{"paper":"/paper/from-adversarial-arms-race-to-model-centric","slug":"from-adversarial-arms-race-to-model-centric","title":"From Adversarial Arms Race to Model-centric Evaluation: Motivating a Unified Automatic Robustness Evaluation Framework","date":"2023-05-29","arxiv_id":"2305.18503","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/robtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"slimfit-memory-efficient-fine-tuning-of","title":"SlimFit: Memory-Efficient Fine-Tuning of Transformer-based Models Using Training Dynamics","date":"2023-05-29","arxiv_id":"2305.18513","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-language-models-handle-word","title":"Transformer Language Models Handle Word Frequency in Prediction Head","date":"2023-05-29","arxiv_id":"2305.18294","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-masked-language-modeling-for","slug":"rethinking-masked-language-modeling-for","title":"Rethinking Masked Language Modeling for Chinese Spelling Correction","date":"2023-05-28","arxiv_id":"2305.17721","n_code_links":1,"syntology":null},{"paper":null,"slug":"transfer-learning-for-power-outage-detection","title":"Transfer Learning for Power Outage Detection Task with Limited Training Data","date":"2023-05-28","arxiv_id":"2305.17817","n_code_links":0,"syntology":null},{"paper":"/paper/an-investigation-into-the-effects-of-pre","slug":"an-investigation-into-the-effects-of-pre","title":"Diagnosing Transformers: Illuminating Feature Spaces for Clinical Decision-Making","date":"2023-05-27","arxiv_id":"2305.17588","n_code_links":1,"syntology":null},{"paper":null,"slug":"complementary-and-integrative-health-lexicon","title":"Complementary and Integrative Health Lexicon (CIHLex) and Entity Recognition in the Literature","date":"2023-05-27","arxiv_id":"2305.17353","n_code_links":0,"syntology":null},{"paper":"/paper/modeling-adversarial-attack-on-pre-trained","slug":"modeling-adversarial-attack-on-pre-trained","title":"Modeling Adversarial Attack on Pre-trained Language Models as Sequential Decision Making","date":"2023-05-27","arxiv_id":"2305.17440","n_code_links":1,"syntology":null},{"paper":null,"slug":"calibration-of-transformer-based-models-for","title":"Calibration of Transformer-based Models for Identifying Stress and Depression in Social Media","date":"2023-05-26","arxiv_id":"2305.16797","n_code_links":0,"syntology":null},{"paper":"/paper/geovln-learning-geometry-enhanced-visual-1","slug":"geovln-learning-geometry-enhanced-visual-1","title":"GeoVLN: Learning Geometry-Enhanced Visual Representation with Slot Attention for Vision-and-Language Navigation","date":"2023-05-26","arxiv_id":"2305.17102","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-distributions-of-discourse","title":"Incorporating Distributions of Discourse Structure for Long Document Abstractive Summarization","date":"2023-05-26","arxiv_id":"2305.16784","n_code_links":0,"syntology":null},{"paper":null,"slug":"knse-a-knowledge-aware-natural-language","title":"KNSE: A Knowledge-aware Natural Language Inference Framework for Dialogue Symptom Status Recognition","date":"2023-05-26","arxiv_id":"2305.16833","n_code_links":0,"syntology":null},{"paper":null,"slug":"theoretical-and-practical-perspectives-on","title":"Theoretical and Practical Perspectives on what Influence Functions Do","date":"2023-05-26","arxiv_id":"2305.16971","n_code_links":0,"syntology":null},{"paper":"/paper/zero-is-not-hero-yet-benchmarking-zero-shot","slug":"zero-is-not-hero-yet-benchmarking-zero-shot","title":"Zero is Not Hero Yet: Benchmarking Zero-Shot Performance of LLMs for Financial Tasks","date":"2023-05-26","arxiv_id":"2305.16633","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-study-of-pre-trained-bert-models","title":"Comparative Study of Pre-Trained BERT Models for Code-Mixed Hindi-English Data","date":"2023-05-25","arxiv_id":"2305.15722","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-attention-layers-coupled-with","title":"Context-aware attention layers coupled with optimal transport domain adaptation and multimodal fusion methods for recognizing dementia from spontaneous speech","date":"2023-05-25","arxiv_id":"2305.16406","n_code_links":0,"syntology":null},{"paper":null,"slug":"not-wacky-vs-definitely-wacky-a-study-of","title":"Not wacky vs. definitely wacky: A study of scalar adverbs in pretrained language models","date":"2023-05-25","arxiv_id":"2305.16426","n_code_links":0,"syntology":null},{"paper":"/paper/text-to-motion-retrieval-towards-joint","slug":"text-to-motion-retrieval-towards-joint","title":"Text-to-Motion Retrieval: Towards Joint Understanding of Human Motion Data and Natural Language","date":"2023-05-25","arxiv_id":"2305.15842","n_code_links":1,"syntology":null},{"paper":"/paper/a-causal-view-of-entity-bias-in-large","slug":"a-causal-view-of-entity-bias-in-large","title":"A Causal View of Entity Bias in (Large) Language Models","date":"2023-05-24","arxiv_id":"2305.14695","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luka-group/causal-view-of-entity-bias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/complex-mathematical-symbol-definition","slug":"complex-mathematical-symbol-definition","title":"Complex Mathematical Symbol Definition Structures: A Dataset and Model for Coordination Resolution in Definition Extraction","date":"2023-05-24","arxiv_id":"2305.14660","n_code_links":1,"syntology":null},{"paper":"/paper/context-aware-transformer-pre-training-for","slug":"context-aware-transformer-pre-training-for","title":"Context-Aware Transformer Pre-Training for Answer Sentence Selection","date":"2023-05-24","arxiv_id":"2305.15358","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-masking-rate-schedules-for-mlm","title":"Dynamic Masking Rate Schedules for MLM Pretraining","date":"2023-05-24","arxiv_id":"2305.15096","n_code_links":0,"syntology":null},{"paper":null,"slug":"extracting-psychological-indicators-using","title":"Extracting Psychological Indicators Using Question Answering","date":"2023-05-24","arxiv_id":"2305.14891","n_code_links":0,"syntology":null},{"paper":"/paper/ghostbuster-detecting-text-ghostwritten-by","slug":"ghostbuster-detecting-text-ghostwritten-by","title":"Ghostbuster: Detecting Text Ghostwritten by Large Language Models","date":"2023-05-24","arxiv_id":"2305.15047","n_code_links":2,"syntology":null},{"paper":"/paper/how-to-distill-your-bert-an-empirical-study","slug":"how-to-distill-your-bert-an-empirical-study","title":"How to Distill your BERT: An Empirical Study on the Impact of Weight Initialisation and Distillation Objectives","date":"2023-05-24","arxiv_id":"2305.15032","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mainlp/how-to-distill-your-bert"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"neural-summarization-of-electronic-health","title":"Neural Summarization of Electronic Health Records","date":"2023-05-24","arxiv_id":"2305.15222","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-token-dropping-strategy-in","slug":"revisiting-token-dropping-strategy-in","title":"Revisiting Token Dropping Strategy in Efficient BERT Pretraining","date":"2023-05-24","arxiv_id":"2305.15273","n_code_links":1,"syntology":null},{"paper":null,"slug":"2305-14521","title":"Few-shot Adaptation to Distribution Shifts By Mixing Source and Target Embeddings","date":"2023-05-23","arxiv_id":"2305.14521","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-method-for-unsupervised-bilingual","slug":"a-simple-method-for-unsupervised-bilingual","title":"When your Cousin has the Right Connections: Unsupervised Bilingual Lexicon Induction for Related Data-Imbalanced Languages","date":"2023-05-23","arxiv_id":"2305.14012","n_code_links":1,"syntology":null},{"paper":"/paper/all-roads-lead-to-rome-exploring-the","slug":"all-roads-lead-to-rome-exploring-the","title":"All Roads Lead to Rome? Exploring the Invariance of Transformers' Representations","date":"2023-05-23","arxiv_id":"2305.14555","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["twinkle0331/bert-similarity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-linguistic-generalisation-in","title":"Assessing Linguistic Generalisation in Language Models: A Dataset for Brazilian Portuguese","date":"2023-05-23","arxiv_id":"2305.14070","n_code_links":0,"syntology":null},{"paper":"/paper/axomiyaberta-a-phonologically-aware","slug":"axomiyaberta-a-phonologically-aware","title":"AxomiyaBERTa: A Phonologically-aware Transformer Model for Assamese","date":"2023-05-23","arxiv_id":"2305.13641","n_code_links":1,"syntology":null},{"paper":"/paper/connecting-the-dots-what-graph-based-text","slug":"connecting-the-dots-what-graph-based-text","title":"Connecting the Dots: What Graph-Based Text Representations Work Best for Text Classification Using Graph Neural Networks?","date":"2023-05-23","arxiv_id":"2305.14578","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-large-language-models-for-classical","slug":"exploring-large-language-models-for-classical","title":"Exploring Large Language Models for Classical Philology","date":"2023-05-23","arxiv_id":"2305.13698","n_code_links":1,"syntology":null},{"paper":null,"slug":"handling-realistic-label-noise-in-bert-text","title":"Handling Realistic Label Noise in BERT Text Classification","date":"2023-05-23","arxiv_id":"2305.16337","n_code_links":0,"syntology":null},{"paper":"/paper/on-robustness-of-finetuned-transformer-based","slug":"on-robustness-of-finetuned-transformer-based","title":"On Robustness of Finetuned Transformer-based NLP Models","date":"2023-05-23","arxiv_id":"2305.14453","n_code_links":1,"syntology":null},{"paper":"/paper/text-is-all-you-need-learning-language","slug":"text-is-all-you-need-learning-language","title":"Text Is All You Need: Learning Language Representations for Sequential Recommendation","date":"2023-05-23","arxiv_id":"2305.13731","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"training-transitive-and-commutative","title":"Training Transitive and Commutative Multimodal Transformers with LoReTTa","date":"2023-05-23","arxiv_id":"2305.14243","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-energy-based-language-models-with","slug":"exploring-energy-based-language-models-with","title":"Exploring Energy-based Language Models with Different Architectures and Training Methods for Speech Recognition","date":"2023-05-22","arxiv_id":"2305.12676","n_code_links":2,"syntology":null},{"paper":null,"slug":"gatology-for-linguistics-what-syntactic","title":"GATology for Linguistics: What Syntactic Dependencies It Knows","date":"2023-05-22","arxiv_id":"2305.13403","n_code_links":0,"syntology":null},{"paper":null,"slug":"imsimcse-improving-contrastive-learning-for","title":"SimCSE++: Improving Contrastive Learning for Sentence Embeddings from Two Perspectives","date":"2023-05-22","arxiv_id":"2305.13192","n_code_links":0,"syntology":null},{"paper":"/paper/language-agnostic-bias-detection-in-language","slug":"language-agnostic-bias-detection-in-language","title":"Language-Agnostic Bias Detection in Language Models with Bias Probing","date":"2023-05-22","arxiv_id":"2305.13302","n_code_links":1,"syntology":null},{"paper":"/paper/logical-reasoning-for-natural-language","slug":"logical-reasoning-for-natural-language","title":"Atomic Inference for NLI with Generated Facts as Atoms","date":"2023-05-22","arxiv_id":"2305.13214","n_code_links":1,"syntology":null},{"paper":null,"slug":"stock-and-market-index-prediction-using","title":"Stock and market index prediction using Informer network","date":"2023-05-22","arxiv_id":"2305.14382","n_code_links":0,"syntology":null},{"paper":null,"slug":"syntactic-knowledge-via-graph-attention-with","title":"Syntactic Knowledge via Graph Attention with BERT in Machine Translation","date":"2023-05-22","arxiv_id":"2305.13413","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deeper-autoregressive-approach-to-non","title":"A Deeper (Autoregressive) Approach to Non-Convergent Discourse Parsing","date":"2023-05-21","arxiv_id":"2305.12510","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-symbolic-framework-for-systematic","title":"A Symbolic Framework for Evaluating Mathematical Reasoning and Generalisation with Transformers","date":"2023-05-21","arxiv_id":"2305.12563","n_code_links":0,"syntology":null},{"paper":"/paper/bertrlfuzzer-a-bert-and-reinforcement","slug":"bertrlfuzzer-a-bert-and-reinforcement","title":"BertRLFuzzer: A BERT and Reinforcement Learning Based Fuzzer","date":"2023-05-21","arxiv_id":"2305.12534","n_code_links":1,"syntology":null},{"paper":null,"slug":"f-pabee-flexible-patience-based-early-exiting","title":"F-PABEE: Flexible-patience-based Early Exiting for Single-label and Multi-label text Classification Tasks","date":"2023-05-21","arxiv_id":"2305.11916","n_code_links":0,"syntology":null},{"paper":null,"slug":"infor-coef-information-bottleneck-based","title":"Infor-Coef: Information Bottleneck-based Dynamic Token Downsampling for Compact and Efficient language model","date":"2023-05-21","arxiv_id":"2305.12458","n_code_links":0,"syntology":null},{"paper":null,"slug":"ir-models-and-the-covid-19-pandemic-a","title":"IR Models and the COVID-19 Pandemic: A Comparative Study of Performance and Challenges","date":"2023-05-21","arxiv_id":"2305.12528","n_code_links":0,"syntology":null},{"paper":null,"slug":"cdjur-br-a-golden-collection-of-legal","title":"CDJUR-BR -- A Golden Collection of Legal Document from Brazilian Justice with Fine-Grained Named Entities","date":"2023-05-20","arxiv_id":"2305.18315","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentfin-1-0-entity-aware-sentiment-analysis","title":"SEntFiN 1.0: Entity-Aware Sentiment Analysis for Financial News","date":"2023-05-20","arxiv_id":"2305.12257","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sequence-to-sequence-approach-for-arabic","title":"A Sequence-to-Sequence Approach for Arabic Pronoun Resolution","date":"2023-05-19","arxiv_id":"2305.11529","n_code_links":0,"syntology":null},{"paper":null,"slug":"eye-spatialnet-spatial-information-extraction","title":"Eye-SpatialNet: Spatial Information Extraction from Ophthalmology Notes","date":"2023-05-19","arxiv_id":"2305.11948","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-foundation-models-privacy","title":"Federated Foundation Models: Privacy-Preserving and Collaborative Learning for Large Models","date":"2023-05-19","arxiv_id":"2305.11414","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-universal-phonetic-representation-in","title":"Language-Universal Phonetic Representation in Multilingual Speech Pretraining for Low-Resource Speech Recognition","date":"2023-05-19","arxiv_id":"2305.11569","n_code_links":0,"syntology":null},{"paper":null,"slug":"ahead-of-time-p-tuning","title":"Ahead-of-Time P-Tuning","date":"2023-05-18","arxiv_id":"2305.10835","n_code_links":0,"syntology":null},{"paper":"/paper/ditto-a-simple-and-efficient-approach-to","slug":"ditto-a-simple-and-efficient-approach-to","title":"Ditto: A Simple and Efficient Approach to Improve Sentence Embeddings","date":"2023-05-18","arxiv_id":"2305.10786","n_code_links":1,"syntology":null},{"paper":null,"slug":"pdp-parameter-free-differentiable-pruning-is","title":"PDP: Parameter-free Differentiable Pruning is All You Need","date":"2023-05-18","arxiv_id":"2305.11203","n_code_links":0,"syntology":null},{"paper":null,"slug":"trading-syntax-trees-for-wordpieces-target","title":"Trading Syntax Trees for Wordpieces: Target-oriented Opinion Words Extraction with Wordpieces and Aspect Enhancement","date":"2023-05-18","arxiv_id":"2305.11034","n_code_links":0,"syntology":null},{"paper":"/paper/a-quantitative-study-of-nlp-approaches-to","slug":"a-quantitative-study-of-nlp-approaches-to","title":"A quantitative study of NLP approaches to question difficulty estimation","date":"2023-05-17","arxiv_id":"2305.10236","n_code_links":1,"syntology":null},{"paper":"/paper/ad-kd-attribution-driven-knowledge","slug":"ad-kd-attribution-driven-knowledge","title":"AD-KD: Attribution-Driven Knowledge Distillation for Language Model Compression","date":"2023-05-17","arxiv_id":"2305.10010","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["brucewsy/ad-kd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/explaining-black-box-text-modules-in-natural","slug":"explaining-black-box-text-modules-in-natural","title":"Explaining black box text modules in natural language with language models","date":"2023-05-17","arxiv_id":"2305.09863","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["csinva/imodelsX","microsoft/automated-explanations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/solving-cosine-similarity-underestimation","slug":"solving-cosine-similarity-underestimation","title":"Solving Cosine Similarity Underestimation between High Frequency Words by L2 Norm Discounting","date":"2023-05-17","arxiv_id":"2305.10610","n_code_links":1,"syntology":null},{"paper":"/paper/berttm-leveraging-contextualized-word","slug":"berttm-leveraging-contextualized-word","title":"CWTM: Leveraging Contextualized Word Embeddings from BERT for Neural Topic Modeling","date":"2023-05-16","arxiv_id":"2305.09329","n_code_links":1,"syntology":null},{"paper":"/paper/measuring-stereotypes-using-entity-centric","slug":"measuring-stereotypes-using-entity-centric","title":"Measuring Dimensions of Self-Presentation in Twitter Bios and their Links to Misinformation Sharing","date":"2023-05-16","arxiv_id":"2305.09548","n_code_links":1,"syntology":null},{"paper":"/paper/weight-inherited-distillation-for-task","slug":"weight-inherited-distillation-for-task","title":"Weight-Inherited Distillation for Task-Agnostic BERT Compression","date":"2023-05-16","arxiv_id":"2305.09098","n_code_links":1,"syntology":null},{"paper":"/paper/coreference-aware-double-channel-attention","slug":"coreference-aware-double-channel-attention","title":"Coreference-aware Double-channel Attention Network for Multi-party Dialogue Reading Comprehension","date":"2023-05-15","arxiv_id":"2305.08348","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-rumination-for-pre-trained-language","slug":"knowledge-rumination-for-pre-trained-language","title":"Knowledge Rumination for Pre-trained Language Models","date":"2023-05-15","arxiv_id":"2305.08732","n_code_links":1,"syntology":null},{"paper":null,"slug":"private-training-set-inspection-in-mlaas","title":"Private Training Set Inspection in MLaaS","date":"2023-05-15","arxiv_id":"2305.09058","n_code_links":0,"syntology":null},{"paper":null,"slug":"text2gender-a-deep-learning-architecture-for","title":"Text2Gender: A Deep Learning Architecture for Analysis of Blogger's Age and Gender","date":"2023-05-15","arxiv_id":"2305.08633","n_code_links":0,"syntology":null}],"record_sha256":"0670d4267c0b1cd7efa0687371095135f9335733542e3e95a90e771de350b263","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}