{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/121","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":121,"pages_in_order":142,"rows_per_page":100,"rows":[12001,12100],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/120","next":"/task/language-modeling/papers/122","papers":[{"url":null,"slug":"cambridge-at-semeval-2021-task-2-neural-wic","title":"Cambridge at SemEval-2021 Task 2: Neural WiC-Model with Data Augmentation and Exploration of Representation","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"clac-bp-at-semeval-2021-task-8-scibert-plus","title":"CLaC-BP at SemEval-2021 Task 8: SciBERT Plus Rules for MeasEval","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-fast-and-slow-a-case-study-on","title":"Decoding, Fast and Slow: A Case Study on Balancing Trade-Offs in Incremental, Character-level Pragmatic Reasoning","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deepblueai-at-semeval-2021-task-7-detecting","title":"DeepBlueAI at SemEval-2021 Task 7: Detecting and Rating Humor and Offense with Stacking Diverse Language Model-Based Methods","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"document-grounded-goal-oriented-dialogue","title":"Document-Grounded Goal-Oriented Dialogue Systems on Pre-Trained Language Model with Diverse Input Representation","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-language-generation-with-effective","title":"Enhancing Language Generation with Effective Checkpoints of Pre-trained Language Model","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"entity-and-evidence-guided-document-level","title":"Entity and Evidence Guided Document-Level Relation Extraction","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-morphological-typology-in-zero","title":"Evaluating morphological typology in zero-shot cross-lingual transfer","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ibm-mnlp-ie-at-case-2021-task-1-multigranular","title":"IBM MNLP IE at CASE 2021 Task 1: Multigranular and Multilingual Event Detection on Protest News","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-low-resource-named-entity-1","title":"Improving Low-Resource Named Entity Recognition via Label-Aware Data Augmentation and Curriculum Denoising","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lets-be-explicit-about-that-distant","title":"Let’s be explicit about that: Distant supervision for implicit discourse relation classification via connective prediction","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-improving-bert-s-mathematical-1","title":"Measuring and Improving BERT's Mathematical Abilities by Predicting the Order of Reasoning.","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"medai-at-semeval-2021-task-5-start-to-end","title":"MedAI at SemEval-2021 Task 5: Start-to-end Tagging Framework for Toxic Spans Detection","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-few-shot-named-entity","title":"Meta-Learning for Few-Shot Named Entity Recognition","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mulda-a-multilingual-data-augmentation","title":"MulDA: A Multilingual Data Augmentation Framework for Low-Resource Cross-Lingual NER","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mvp-bert-multi-vocab-pre-training-for-chinese","title":"MVP-BERT: Multi-Vocab Pre-training for Chinese BERT","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nmt5-is-parallel-data-still-relevant-for-pre-1","title":"nmT5 - Is parallel data still relevant for pre-training massively multilingual language models?","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"noobs-at-semeval-2021-task-4-masked-language","title":"Noobs at Semeval-2021 Task 4: Masked Language Modeling for abstract answer prediction","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-response-generation-with-tensor","title":"Personalized Response Generation with Tensor Factorization","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/phmospell-phonological-and-morphological","slug":"phmospell-phonological-and-morphological","title":"PHMOSpell: Phonological and Morphological Knowledge Guided Chinese Spelling Check","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pingan-omini-sinitic-at-semeval-2021-task-4","title":"PINGAN Omini-Sinitic at SemEval-2021 Task 4:Reading Comprehension of Abstract Meaning","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pral-a-tailored-pre-training-model-for-task","title":"PRAL: A Tailored Pre-Training Model for Task-Oriented Dialog Generation","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-multi-modal-machine-translation-with","title":"Probing Multi-modal Machine Translation with Pre-trained Language Model","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large-1","title":"QASR: QCRI Aljazeera Speech Resource A Large Scale Annotated Arabic Speech Corpus","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rakutens-participation-in-wat-2021-examining","title":"Rakuten’s Participation in WAT 2021: Examining the Effectiveness of Pre-trained Models for Multilingual and Multimodal Machine Translation","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"realised-volatility-forecasting-machine","title":"Realised Volatility Forecasting: Machine Learning via Financial Word Embedding","date":"2021-08-01","arxiv_id":"2108.00480","repositories_listed":0,"syntology":null},{"url":null,"slug":"roma-at-semeval-2021-task-7-a-transformer","title":"RoMa at SemEval-2021 Task 7: A Transformer-based Approach for Detecting and Rating Humor and Offense","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"s-nlp-at-semeval-2021-task-5-an-analysis-of","title":"S-NLP at SemEval-2021 Task 5: An Analysis of Dual Networks for Sequence Tagging","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"selecting-informative-contexts-improves-1","title":"Selecting Informative Contexts Improves Language Model Fine-tuning","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"skoltechnlp-at-semeval-2021-task-2-generating","title":"SkoltechNLP at SemEval-2021 Task 2: Generating Cross-Lingual Training Data for the Word-in-Context Task","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stereotyping-norwegian-salmon-an-inventory-of","title":"Stereotyping Norwegian Salmon: An Inventory of Pitfalls in Fairness Benchmark Datasets","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"team-noconflict-at-case-2021-task-1","title":"Team “NoConflict” at CASE 2021 Task 1: Pretraining for Sentence-Level Protest Event Detection","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-university-of-edinburghs-submission-to","title":"The University of Edinburgh’s Submission to the IWSLT21 Simultaneous Translation Task","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unleash-gpt-2-power-for-event-detection","title":"Unleash GPT-2 Power for Event Detection","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-gender-and-polarity-informed-models-to","title":"Using Gender- and Polarity-Informed Models to Investigate Bias","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-continual-entity-learning-in-language","title":"Towards Continual Entity Learning in Language Models for Conversational Agents","date":"2021-07-30","arxiv_id":"2108.00082","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-transferring-of-pre-trained","title":"Cross-lingual Transferring of Pre-trained Contextualized Language Models","date":"2021-07-27","arxiv_id":"2107.12627","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-language-model-for-efficient","title":"Exploiting Language Model for Efficient Linguistic Steganalysis","date":"2021-07-26","arxiv_id":"2107.12168","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-differentiable-language-model-adversarial","title":"A Differentiable Language Model Adversarial Attack on Text Classifiers","date":"2021-07-23","arxiv_id":"2107.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"back-translated-task-adaptive-pretraining","title":"Back-Translated Task Adaptive Pretraining: Improving Accuracy and Robustness on Text Classification","date":"2021-07-22","arxiv_id":"2107.10474","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeptitle-leveraging-bert-to-generate-search","title":"DeepTitle -- Leveraging BERT to generate Search Engine Optimized Headlines","date":"2021-07-22","arxiv_id":"2107.10935","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed-words-based-data-selection-for-language","title":"Seed Words Based Data Selection for Language Model Adaptation","date":"2021-07-20","arxiv_id":"2107.09433","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-language-model-and","title":"Bridging the Gap between Language Model and Reading Comprehension: Unsupervised MRC via Self-Supervision","date":"2021-07-19","arxiv_id":"2107.08582","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-vector-based-approach-to-few-shot-veracity","title":"A Vector-Based Approach to Few-Shot Veracity Classification for Automated Fact-Checking","date":"2021-07-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-language-models-on-low-end-hardware","title":"Using Language Models on Low-end Hardware","date":"2021-07-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"are-multilingual-models-the-best-choice-for","title":"Are Multilingual Models the Best Choice for Moderately Under-resourced Languages? A Comprehensive Assessment for Catalan","date":"2021-07-16","arxiv_id":"2107.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"intersectional-bias-in-causal-language-models","title":"Intersectional Bias in Causal Language Models","date":"2021-07-16","arxiv_id":"2107.07691","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmutants-training-neural-bug-detectors","title":"DeepMutants: Training neural bug detectors with contextual mutations","date":"2021-07-14","arxiv_id":"2107.06657","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-news-classification-using-bert","title":"Large-Scale News Classification using BERT Language Model: Spark NLP Approach","date":"2021-07-14","arxiv_id":"2107.06785","repositories_listed":0,"syntology":null},{"url":null,"slug":"kafisto-a-kalman-filtering-framework-for","title":"KOALA: A Kalman Optimization Algorithm with Loss Adaptivity","date":"2021-07-07","arxiv_id":"2107.03331","repositories_listed":0,"syntology":null},{"url":null,"slug":"languagerefer-spatial-language-model-for-3d","title":"LanguageRefer: Spatial-Language Model for 3D Visual Grounding","date":"2021-07-07","arxiv_id":"2107.03438","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-quite-ask-a-librarian-ai-on-the-nature","title":"Not Quite 'Ask a Librarian': AI on the Nature, Value, and Future of LIS","date":"2021-07-07","arxiv_id":"2107.05383","repositories_listed":0,"syntology":null},{"url":null,"slug":"getting-to-production-with-few-shot-natural","title":"Getting to Production with Few-shot Natural Language Generation Models","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"projection-of-turn-completion-in-incremental","title":"Projection of Turn Completion in Incremental Spoken Dialogue Systems","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"word-free-spoken-language-understanding-for","title":"Word-Free Spoken Language Understanding for Mandarin-Chinese","date":"2021-07-01","arxiv_id":"2107.00186","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-and-efficient-probabilistic-language","title":"A Simple and Efficient Probabilistic Language model for Code-Mixed Text","date":"2021-06-29","arxiv_id":"2106.15102","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-knowledge-grounded-dialog-system-based-on","title":"A Knowledge-Grounded Dialog System Based on Pre-Trained Language Models","date":"2021-06-28","arxiv_id":"2106.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-conceptual-blending-with-large-scale","title":"Visual Conceptual Blending with Large-scale Language and Vision Models","date":"2021-06-27","arxiv_id":"2106.14127","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-are-good-translators","title":"Language Models are Good Translators","date":"2021-06-25","arxiv_id":"2106.13627","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-sample-replacements-for-electra","title":"Learning to Sample Replacements for ELECTRA Pre-Training","date":"2021-06-25","arxiv_id":"2106.13715","repositories_listed":0,"syntology":null},{"url":"/paper/multimodal-few-shot-learning-with-frozen","slug":"multimodal-few-shot-learning-with-frozen","title":"Multimodal Few-Shot Learning with Frozen Language Models","date":"2021-06-25","arxiv_id":"2106.13884","repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large","title":"QASR: QCRI Aljazeera Speech Resource -- A Large Scale Annotated Arabic Speech Corpus","date":"2021-06-24","arxiv_id":"2106.13000","repositories_listed":0,"syntology":null},{"url":null,"slug":"winner-team-mia-at-textvqa-challenge-2021","title":"Winner Team Mia at TextVQA Challenge 2021: Vision-and-Language Representation Learning with Pre-trained Sequence-to-Sequence Model","date":"2021-06-24","arxiv_id":"2106.15332","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterchat-supporting-the-creation-of","title":"CharacterChat: Supporting the Creation of Fictional Characters through Conversation and Progressive Manifestation with a Chatbot","date":"2021-06-23","arxiv_id":"2106.12314","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-case-study-in-bootstrapping-ontology-graphs","title":"A Case Study in Bootstrapping Ontology Graphs from Textbooks","date":"2021-06-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-opcode-sequence-based","title":"Data Augmentation for Opcode Sequence Based Malware Detection","date":"2021-06-22","arxiv_id":"2106.11821","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-discriminative-entity-aware-language-model","title":"A Discriminative Entity-Aware Language Model for Virtual Assistants","date":"2021-06-21","arxiv_id":"2106.11292","repositories_listed":0,"syntology":null},{"url":null,"slug":"ad-text-classification-with-transformer-based","title":"Ad Text Classification with Transformer-Based Natural Language Processing Methods","date":"2021-06-21","arxiv_id":"2106.10899","repositories_listed":0,"syntology":null},{"url":null,"slug":"maxup-lightweight-adversarial-training-with","title":"MaxUp: Lightweight Adversarial Training With Data Augmentation Improves Neural Network Training","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"label-mask-for-multi-label-text","title":"Label prompt for multi-label text classification","date":"2021-06-18","arxiv_id":"2106.10076","repositories_listed":0,"syntology":null},{"url":null,"slug":"process-for-adapting-language-models-to","title":"Process for Adapting Language Models to Society (PALMS) with Values-Targeted Datasets","date":"2021-06-18","arxiv_id":"2106.10328","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-neural-story-generation-with","title":"Augmented Neural Story Generation with Commonsense Inference","date":"2021-06-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"seover-sentence-level-emotion-orientation","title":"SEOVER: Sentence-level Emotion Orientation Vector based Conversation Emotion Recognition Model","date":"2021-06-16","arxiv_id":"2106.08785","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-adaptation-for-e-commerce-chatbots-using","title":"ASR Adaptation for E-commerce Chatbots using Cross-Utterance Context and Multi-Task Language Modeling","date":"2021-06-15","arxiv_id":"2106.09532","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialectal-speech-recognition-and-translation","title":"Dialectal Speech Recognition and Translation of Swiss German Speech to Standard German Text: Microsoft's Submission to SwissText 2021","date":"2021-06-15","arxiv_id":"2106.08126","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairconnect-a-compute-efficient-mlp","title":"PairConnect: A Compute-Efficient MLP Alternative to Attention","date":"2021-06-15","arxiv_id":"2106.08235","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-neural-architecture-search","title":"Differentiable Neural Architecture Search with Morphism-based Transformable Backbone Architectures","date":"2021-06-14","arxiv_id":"2106.07211","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-einstein-more-agreeable-and-less-neurotic","title":"Is Einstein more agreeable and less neurotic than Hitler? A computational exploration of the emotional and personality profiles of historical persons","date":"2021-06-14","arxiv_id":"2106.07237","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-domain-mismatch-in-low-resource","title":"Overcoming Domain Mismatch in Low Resource Sequence-to-Sequence ASR Models using Hybrid Generated Pseudotranscripts","date":"2021-06-14","arxiv_id":"2106.07716","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-the-ordering-of-characters-in","title":"Predicting the Ordering of Characters in Japanese Historical Documents","date":"2021-06-12","arxiv_id":"2106.06786","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-pre-trained-language-model-for","title":"Leveraging Pre-trained Language Model for Speech Sentiment Analysis","date":"2021-06-11","arxiv_id":"2106.06598","repositories_listed":0,"syntology":null},{"url":null,"slug":"mst-masked-self-supervised-transformer-for","title":"MST: Masked Self-Supervised Transformer for Visual Representation","date":"2021-06-10","arxiv_id":"2106.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"dga-net-dynamic-gaussian-attention-network","title":"DGA-Net Dynamic Gaussian Attention Network for Sentence Semantic Matching","date":"2021-06-09","arxiv_id":"2106.04905","repositories_listed":0,"syntology":null},{"url":null,"slug":"hash-layers-for-large-sparse-models","title":"Hash Layers For Large Sparse Models","date":"2021-06-08","arxiv_id":"2106.04426","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-improving-bert-s-mathematical","title":"Measuring and Improving BERT's Mathematical Abilities by Predicting the Order of Reasoning","date":"2021-06-07","arxiv_id":"2106.03921","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-trained-language-model-for-web-scale","title":"Pre-trained Language Model for Web-scale Retrieval in Baidu Search","date":"2021-06-07","arxiv_id":"2106.03373","repositories_listed":0,"syntology":null},{"url":null,"slug":"rosearch-search-for-robust-student","title":"RoSearch: Search for Robust Student Architectures When Distilling Pre-trained Language Models","date":"2021-06-07","arxiv_id":"2106.03613","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-imprint","title":"Video Imprint","date":"2021-06-07","arxiv_id":"2106.03283","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-be-explicit-about-that-distant","title":"Let's be explicit about that: Distant supervision for implicit discourse relation classification via connective prediction","date":"2021-06-06","arxiv_id":"2106.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-adapter-based-tuning","title":"On the Effectiveness of Adapter-based Tuning for Pretrained Language Model Adaptation","date":"2021-06-06","arxiv_id":"2106.03164","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-enhanced-explainable-finetuning-for","title":"Semantic-Enhanced Explainable Finetuning for Open-Domain Dialogues","date":"2021-06-06","arxiv_id":"2106.03065","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-weighted-automata-for-approximate","title":"Extracting Weighted Automata for Approximate Minimization in Language Modelling","date":"2021-06-05","arxiv_id":"2106.02965","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-granularity-contrastive-learning-for-post","title":"Bi-Granularity Contrastive Learning for Post-Training in Few-Shot Scene","date":"2021-06-04","arxiv_id":"2106.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-the-implicit-energy-networks-behind","title":"Exposing the Implicit Energy Networks behind Masked Language Models via Metropolis--Hastings","date":"2021-06-04","arxiv_id":"2106.02736","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-metrics-and-procrustes","title":"Language Model Metrics and Procrustes Analysis for Improved Vector Transformation of NLP Embeddings","date":"2021-06-04","arxiv_id":"2106.02490","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimum-word-error-rate-training-with","title":"Minimum Word Error Rate Training with Language Model Fusion for End-to-End Speech Recognition","date":"2021-06-04","arxiv_id":"2106.02302","repositories_listed":0,"syntology":null},{"url":null,"slug":"nmt5-is-parallel-data-still-relevant-for-pre","title":"nmT5 -- Is parallel data still relevant for pre-training massively multilingual language models?","date":"2021-06-03","arxiv_id":"2106.02171","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-span-extraction-approach-for-information","title":"A Span Extraction Approach for Information Extraction on Visually-Rich Documents","date":"2021-06-02","arxiv_id":"2106.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"belabbert-a-dutch-roberta-based-language","title":"belabBERT: a Dutch RoBERTa-based language model applied to psychiatric classification","date":"2021-06-02","arxiv_id":"2106.01091","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-select-a-fully-attentive-approach","title":"Learning to Select: A Fully Attentive Approach for Novel Object Captioning","date":"2021-06-02","arxiv_id":"2106.01424","repositories_listed":0,"syntology":null}],"record_sha256":"743a46e77ac7c4f6132f01c1785fd2640ce43a6d7b14be7471ee328ef66b52fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}