{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/120","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":120,"pages_in_order":142,"rows_per_page":100,"rows":[11901,12000],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/119","next":"/task/language-modeling/papers/121","papers":[{"url":null,"slug":"learning-to-selectively-learn-for-weakly","title":"Learning to Selectively Learn for Weakly-supervised Paraphrase Generation","date":"2021-09-25","arxiv_id":"2109.12457","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-diversity-enhanced-and-constraints-relaxed","title":"A Diversity-Enhanced and Constraints-Relaxed Augmentation for Low-Resource Classification","date":"2021-09-24","arxiv_id":"2109.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"identification-of-enzymatic-active-sites-with","title":"Identification of Enzymatic Active Sites with Unsupervised Language Modeling","date":"2021-09-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mlim-vision-and-language-model-pre-training","title":"MLIM: Vision-and-Language Model Pre-training with Masked Language and Image Modeling","date":"2021-09-24","arxiv_id":"2109.12178","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-attention-sparsity-in-transformers","title":"Predicting Attention Sparsity in Transformers","date":"2021-09-24","arxiv_id":"2109.12188","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-language-model-meta-pretraining","title":"Cross-Lingual Language Model Meta-Pretraining","date":"2021-09-23","arxiv_id":"2109.11129","repositories_listed":0,"syntology":null},{"url":null,"slug":"lstm-hyper-parameter-selection-for-malware","title":"LSTM Hyper-Parameter Selection for Malware Detection: Interaction Effects and Hierarchical Selection Approach","date":"2021-09-23","arxiv_id":"2109.11500","repositories_listed":0,"syntology":null},{"url":null,"slug":"bfclass-a-backdoor-free-text-classification","title":"BFClass: A Backdoor-free Text Classification Framework","date":"2021-09-22","arxiv_id":"2109.10855","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"bertweetfr-domain-adaptation-of-pre-trained","title":"BERTweetFR : Domain Adaptation of Pre-Trained Language Models for French Tweets","date":"2021-09-21","arxiv_id":"2109.10234","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-specific-language-models-for","title":"Learning Domain Specific Language Models for Automatic Speech Recognition through Machine Translation","date":"2021-09-21","arxiv_id":"2110.10261","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-trade-offs-of-domain-adaptation-for","title":"The Trade-offs of Domain Adaptation for Neural Language Models","date":"2021-09-21","arxiv_id":"2109.10274","repositories_listed":0,"syntology":null},{"url":null,"slug":"influence-of-asr-and-language-model-on","title":"Influence of ASR and Language Model on Alzheimer's Disease Detection","date":"2021-09-20","arxiv_id":"2110.15704","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-natural-language-generation-from","title":"Learning Natural Language Generation from Scratch","date":"2021-09-20","arxiv_id":"2109.09371","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-training-with-contrastive","title":"Adversarial Training with Contrastive Learning in NLP","date":"2021-09-19","arxiv_id":"2109.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-deacon-multimodal-molecular-domain","title":"Multilingual Molecular Representation Learning via Contrastive Pre-training","date":"2021-09-18","arxiv_id":"2109.08830","repositories_listed":0,"syntology":null},{"url":null,"slug":"bart-light-one-decoder-layer-is-enough","title":"BART-light: One Decoder Layer Is Enough","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-knowledge-augmented-pretrained","title":"Commonsense Knowledge-Augmented Pretrained Language Models for Causal Reasoning Classification","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-multitask-learning-for-low-resource","title":"Exploring Multitask Learning for Low-Resource AbstractiveSummarization","date":"2021-09-17","arxiv_id":"2109.08565","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-range-modeling-of-source-code-files-with","title":"Long-Range Modeling of Source Code Files with eWASH: Extended Window Access by Syntax Hierarchy","date":"2021-09-17","arxiv_id":"2109.08780","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-reading-comprehension-generative-or","title":"Machine Reading Comprehension: Generative or Extractive Reader?","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"relating-neural-text-degeneration-to-exposure","title":"Relating Neural Text Degeneration to Exposure Bias","date":"2021-09-17","arxiv_id":"2109.08705","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiprompt-sentiment-knowledge-enhanced","title":"SentiPrompt: Sentiment Knowledge Enhanced Prompt-Tuning for Aspect-Based Sentiment Analysis","date":"2021-09-17","arxiv_id":"2109.08306","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bag-of-tricks-for-dialogue-summarization","title":"A Bag of Tricks for Dialogue Summarization","date":"2021-09-16","arxiv_id":"2109.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-language-models-know-the-way-to-rome","title":"Do Language Models Know the Way to Rome?","date":"2021-09-16","arxiv_id":"2109.07971","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-the-cat-out-of-the-bag-contrastive","title":"Let the CAT out of the bag: Contrastive Attributed explanations for Text","date":"2021-09-16","arxiv_id":"2109.07983","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-training-of-nearest-neighbor","title":"Regularized Training of Nearest Neighbor Language Models","date":"2021-09-16","arxiv_id":"2109.08249","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-language-model-understood-the-prompt-was","title":"The Language Model Understood the Prompt was Ambiguous: Probing Syntactic Uncertainty Through Generation","date":"2021-09-16","arxiv_id":"2109.07848","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-glass-box-features-uncertainty","title":"Beyond Glass-Box Features: Uncertainty Quantification Enhanced Quality Estimation for Neural Machine Translation","date":"2021-09-15","arxiv_id":"2109.07141","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-text-auto-completion-with-next","title":"Improving Text Auto-Completion with Next Phrase Prediction","date":"2021-09-15","arxiv_id":"2109.07067","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-complementarity-of-data-selection-and","title":"On the Complementarity of Data Selection and Fine Tuning for Domain Adaptation","date":"2021-09-15","arxiv_id":"2109.07591","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranknas-efficient-neural-architecture-search","title":"RankNAS: Efficient Neural Architecture Search by Pairwise Ranking","date":"2021-09-15","arxiv_id":"2109.07383","repositories_listed":0,"syntology":null},{"url":null,"slug":"tied-reduced-rnn-t-decoder","title":"Tied & Reduced RNN-T Decoder","date":"2021-09-15","arxiv_id":"2109.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-read-reconstruction-for-dna-data","title":"Single-Read Reconstruction for DNA Data Storage Using Transformers","date":"2021-09-12","arxiv_id":"2109.05478","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-state-capsule-networks-for-text","title":"Dual-State Capsule Networks for Text Classification","date":"2021-09-10","arxiv_id":"2109.04762","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficientclip-efficient-cross-modal-pre","title":"EfficientCLIP: Efficient Cross-Modal Pre-training by Ensemble Confident Learning and Language Modeling","date":"2021-09-10","arxiv_id":"2109.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaxt-meta-cross-task-transfer-between","title":"MetaXT: Meta Cross-Task Transfer between Disparate Label Spaces","date":"2021-09-09","arxiv_id":"2109.04240","repositories_listed":0,"syntology":null},{"url":"/paper/refinecap-concept-aware-refinement-for-image","slug":"refinecap-concept-aware-refinement-for-image","title":"RefineCap: Concept-Aware Refinement for Image Captioning","date":"2021-09-08","arxiv_id":"2109.03529","repositories_listed":0,"syntology":null},{"url":null,"slug":"sustainable-modular-debiasing-of-language","title":"Sustainable Modular Debiasing of Language Models","date":"2021-09-08","arxiv_id":"2109.03646","repositories_listed":0,"syntology":null},{"url":"/paper/generate-rank-a-multi-task-framework-for-math","slug":"generate-rank-a-multi-task-framework-for-math","title":"Generate & Rank: A Multi-task Framework for Math Word Problems","date":"2021-09-07","arxiv_id":"2109.03034","repositories_listed":0,"syntology":null},{"url":null,"slug":"rare-words-degenerate-all-words","title":"Rare Tokens Degenerate All Tokens: Improving Neural Text Generation via Adaptive Gradient Gating for Rare Token Embeddings","date":"2021-09-07","arxiv_id":"2109.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-should-evaluate-your-language-model-on","title":"You should evaluate your language model on marginal likelihood over tokenisations","date":"2021-09-06","arxiv_id":"2109.02550","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-lexical-translation","title":"Language Modeling, Lexical Translation, Reordering: The Training Process of NMT through the Lens of Classical SMT","date":"2021-09-03","arxiv_id":"2109.01396","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-need-to-know-everything-efficiently","title":"No Need to Know Everything! Efficiently Augmenting Language Models With External Knowledge","date":"2021-09-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-exploration-in-quality-filtering","title":"An Empirical Exploration in Quality Filtering of Text Data","date":"2021-09-02","arxiv_id":"2109.00698","repositories_listed":0,"syntology":null},{"url":null,"slug":"conqx-semantic-expansion-of-spoken-queries","title":"ConQX: Semantic Expansion of Spoken Queries for Intent Detection based on Conditioned Text Generation","date":"2021-09-02","arxiv_id":"2109.00729","repositories_listed":0,"syntology":null},{"url":null,"slug":"legalmfit-efficient-short-legal-text","title":"LegaLMFiT: Efficient Short Legal Text Classification with LSTM Language Model Pre-Training","date":"2021-09-02","arxiv_id":"2109.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-conditionality-for-natural","title":"Multimodal Conditionality for Natural Language Generation","date":"2021-09-02","arxiv_id":"2109.01229","repositories_listed":0,"syntology":null},{"url":null,"slug":"travelbert-pre-training-language-model","title":"Pre-training Language Model Incorporating Domain-specific Heterogeneous Knowledge into A Unified Representation","date":"2021-09-02","arxiv_id":"2109.01048","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-a-clinical-language-model-for","title":"Developing a Clinical Language Model for Swedish: Continued Pretraining of Generic BERT with In-Domain Data","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"does-knowledge-help-general-nlu-an-empirical","title":"Does Knowledge Help General NLU? An Empirical Study","date":"2021-09-01","arxiv_id":"2109.00563","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-specific-japanese-electra-model-using","title":"Domain-Specific Japanese ELECTRA Model Using a Small Corpus","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-character-aware-neural-language","title":"Improving Character-Aware Neural Language Model by Warming up Character Encoder under Skip-gram Architecture","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ircologne-at-germeval-2021-toxicity","title":"IRCologne at GermEval 2021: Toxicity Classification","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-asr-with-an-augmented-language","title":"Low-Resource ASR with an Augmented Language Model","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-adversarial-generation-for-neural","title":"Masked Adversarial Generation for Neural Machine Translation","date":"2021-09-01","arxiv_id":"2109.00417","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-borrowing-detection-with-monolingual","title":"Neural Borrowing Detection with Monolingual Lexical Models","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reducing-repetition-in-abstractive","title":"On Reducing Repetition in Abstractive Summarization","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"split-and-rephrase-in-a-cross-lingual-manner","title":"Split-and-Rephrase in a Cross-Lingual Manner: A Complete Pipeline","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-language-model-for-temporal","title":"Towards a Language Model for Temporal Commonsense Reasoning","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-text-style-transfer-with-content","title":"Unsupervised Text Style Transfer with Content Embeddings","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"watching-a-language-model-learning-chess","title":"Watching a Language Model Learning Chess","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-deep-networks-in-nlp-using","title":"Effectiveness of Deep Networks in NLP using BiDAF as an example architecture","date":"2021-08-31","arxiv_id":"2109.00074","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-adversarial-fine-tuning-benefit-bert","title":"How Does Adversarial Fine-Tuning Benefit BERT?","date":"2021-08-31","arxiv_id":"2108.13602","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effects-of-data-size-on-automated-essay","title":"The effects of data size on Automated Essay Scoring engines","date":"2021-08-30","arxiv_id":"2108.13275","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-memorization-for-fast-learning","title":"Representation Memorization for Fast Learning New Knowledge without Forgetting","date":"2021-08-28","arxiv_id":"2108.12596","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-retraining-free-speech-recognition","title":"Exploring Retraining-Free Speech Recognition for Intra-sentential Code-Switching","date":"2021-08-27","arxiv_id":"2109.00921","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-capacity-of-a-large-scale","title":"Exploring the Capacity of a Large-scale Masked Language Model to Recognize Grammatical Errors","date":"2021-08-27","arxiv_id":"2108.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-invariant-truecasing-with-a-word-and","title":"Position-Invariant Truecasing with a Word-and-Character Hierarchical Recurrent Neural Network","date":"2021-08-26","arxiv_id":"2108.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-of-criminal-texts-for-the-polish","title":"Detection of Criminal Texts for the Polish State Border Guard","date":"2021-08-24","arxiv_id":"2108.10580","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-learning-for-fine-grained-entity","title":"Prompt-Learning for Fine-Grained Entity Typing","date":"2021-08-24","arxiv_id":"2108.10604","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-exposure-bias-in-training-recurrent","title":"Reducing Exposure Bias in Training Recurrent Neural Network Transducers","date":"2021-08-24","arxiv_id":"2108.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-bert-encoding-and-sentence-level","title":"Using BERT Encoding and Sentence-Level Language Model for Sentence Ordering","date":"2021-08-24","arxiv_id":"2108.10986","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbert-pretraining-a-bert-model-for-uzbek","title":"UzBERT: pretraining a BERT model for Uzbek","date":"2021-08-22","arxiv_id":"2108.09814","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-multi-object-relationships-for","title":"Exploiting Multi-Object Relationships for Detecting Adversarial Attacks in Complex Scenes","date":"2021-08-19","arxiv_id":"2108.08421","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-augmented-relevance-score","title":"Language Model Augmented Relevance Score","date":"2021-08-19","arxiv_id":"2108.08485","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-multi-label-prompting-simple-and","title":"Automatic Multi-Label Prompting: Simple and Interpretable Few-Shot Classification","date":"2021-08-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deduplicating-training-data-makes-language-1","title":"Deduplicating Training Data Makes Language Models Better","date":"2021-08-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-deep-learning","title":"Scaling Laws for Deep Learning","date":"2021-08-17","arxiv_id":"2108.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-natural-language-processing-for-linkedin-1","title":"Deep Natural Language Processing for LinkedIn Search","date":"2021-08-16","arxiv_id":"2108.13300","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-structured-dynamic-sparse-pre","title":"Towards Structured Dynamic Sparse Pre-Training of BERT","date":"2021-08-13","arxiv_id":"2108.06277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transformer-based-math-language-model-for","title":"A Transformer-based Math Language Model for Handwritten Math Expression Recognition","date":"2021-08-11","arxiv_id":"2108.05002","repositories_listed":0,"syntology":null},{"url":null,"slug":"mounting-video-metadata-on-transformer-based","title":"Mounting Video Metadata on Transformer-based Language Model for Open-ended Video Question Answering","date":"2021-08-11","arxiv_id":"2108.05158","repositories_listed":0,"syntology":null},{"url":null,"slug":"intent5-search-result-diversification-using","title":"IntenT5: Search Result Diversification using Causal Language Models","date":"2021-08-09","arxiv_id":"2108.04026","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-evaluation-in-open-ended-text","title":"Language Model Evaluation in Open-ended Text Generation","date":"2021-08-08","arxiv_id":"2108.03578","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-commonsense-knowledge-on","title":"Leveraging Commonsense Knowledge on Classifying False News and Determining Checkworthiness of Claims","date":"2021-08-08","arxiv_id":"2108.03731","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-semantic-regression-for-text","title":"Sentence Semantic Regression for Text Generation","date":"2021-08-06","arxiv_id":"2108.02984","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-language-modeling-1","title":"Towards Zero-shot Language Modeling","date":"2021-08-06","arxiv_id":"2108.03334","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmmformer-efficient-and-flexible-transformer","title":"FMMformer: Efficient and Flexible Transformer via Decomposed Near-field and Far-field Attention","date":"2021-08-05","arxiv_id":"2108.02347","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-harm-in-language-models-with","title":"Mitigating harm in language models with conditional-likelihood filtration","date":"2021-08-04","arxiv_id":"2108.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-differentially-private-bert","title":"Large-Scale Differentially Private BERT","date":"2021-08-03","arxiv_id":"2108.01624","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-fairness-may-vary-group-fairness-of","title":"Your fairness may vary: Pretrained language model fairness in toxic text classification","date":"2021-08-03","arxiv_id":"2108.01250","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-my-model-using-the-right-evidence","title":"Is My Model Using The Right Evidence? Systematic Probes for Examining Evidence-Based Tabular Reasoning","date":"2021-08-02","arxiv_id":"2108.00578","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-sentence-weighting-techniques","title":"A Comparison of Sentence-Weighting Techniques for NMT","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-targeted-assessment-of-incremental-1","title":"A Targeted Assessment of Incremental Processing in Neural Language Models and Humans","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"and-does-not-mean-or-using-formal-languages","title":"AND does not mean OR: Using Formal Languages to Study Language Models' Representations","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"astartwice-at-semeval-2021-task-5-toxic-span","title":"AStarTwice at SemEval-2021 Task 5: Toxic Span Detection Using RoBERTa-CRF, Domain Specific Pre-Training and Self-Training","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attending-self-attention-a-case-study-of","title":"Attending Self-Attention: A Case Study of Visually Grounded Supervision in Vision-and-Language Transformers","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"c626489a7a48b41bb3bc4a84a5df1e38e0d719af8b23a10031d32d0d0ac755c2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}