{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/127","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":127,"pages_in_order":142,"rows_per_page":100,"rows":[12601,12700],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/126","next":"/task/language-modeling/papers/128","papers":[{"url":null,"slug":"learning-to-extract-attribute-value-from","title":"Learning to Extract Attribute Value from Product via Question Answering: A Multi-task Approach","date":"2020-08-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uob-at-semeval-2020-task-12-boosting-bert","title":"UoB at SemEval-2020 Task 12: Boosting BERT with Corpus Level Information","date":"2020-08-19","arxiv_id":"2008.08547","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-language-model-and-parallel-bi","title":"Complementary Language Model and Parallel Bi-LRNN for False Trigger Mitigation","date":"2020-08-18","arxiv_id":"2008.08113","repositories_listed":0,"syntology":null},{"url":null,"slug":"adding-recurrence-to-pretrained-transformers","title":"Adding Recurrence to Pretrained Transformers for Improved Efficiency and Context Size","date":"2020-08-16","arxiv_id":"2008.07027","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-multi-domain-language-model-for","title":"Adaptable Multi-Domain Language Model for Transformer ASR","date":"2020-08-14","arxiv_id":"2008.06208","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-learning-mechanism-for-speech","title":"Prosody Learning Mechanism for Speech Synthesis System Without Text Length Limit","date":"2020-08-13","arxiv_id":"2008.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-with-bidirectional-decoder-for","title":"Transformer with Bidirectional Decoder for Speech Recognition","date":"2020-08-11","arxiv_id":"2008.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-neural-query-auto-completion","title":"Efficient Neural Query Auto Completion","date":"2020-08-06","arxiv_id":"2008.02879","repositories_listed":0,"syntology":null},{"url":null,"slug":"6veclm-language-modeling-in-vector-space-for","title":"6VecLM: Language Modeling in Vector Space for IPv6 Target Generation","date":"2020-08-05","arxiv_id":"2008.02213","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-mdi-adaptation-for-n-gram-language","title":"Efficient MDI Adaptation for n-gram Language Models","date":"2020-08-05","arxiv_id":"2008.02385","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-visual-representations-with-caption","title":"Learning Visual Representations with Caption Annotations","date":"2020-08-04","arxiv_id":"2008.01392","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-effects-of-implicit-and-explicit","title":"A Study on Effects of Implicit and Explicit Language Model Information for DBLSTM-CTC Based Handwriting Recognition","date":"2020-07-31","arxiv_id":"2008.01532","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-vector-enhanced-lstm-language-model","title":"Future Vector Enhanced LSTM Language Model for LVCSR","date":"2020-07-31","arxiv_id":"2008.01832","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-learning-universal-representations-across","title":"On Learning Universal Representations Across Languages","date":"2020-07-31","arxiv_id":"2007.15960","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-federated-learning-1","title":"Communication-Efficient Federated Learning via Optimal Client Sampling","date":"2020-07-30","arxiv_id":"2007.15197","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-therapy-target-discovery-with","title":"COVID-19 therapy target discovery with context-aware literature mining","date":"2020-07-30","arxiv_id":"2007.15681","repositories_listed":0,"syntology":null},{"url":null,"slug":"growing-efficient-deep-networks-by-structured","title":"Growing Efficient Deep Networks by Structured Continuous Sparsification","date":"2020-07-30","arxiv_id":"2007.15353","repositories_listed":0,"syntology":null},{"url":null,"slug":"guir-at-semeval-2020-task-12-domain-tuned","title":"GUIR at SemEval-2020 Task 12: Domain-Tuned Contextualized Models for Offensive Language Detection","date":"2020-07-28","arxiv_id":"2007.14477","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensorcoder-dimension-wise-attention-via","title":"TensorCoder: Dimension-Wise Attention via Tensor Representation for Natural Language Modeling","date":"2020-07-28","arxiv_id":"2008.01547","repositories_listed":0,"syntology":null},{"url":null,"slug":"ids-at-semeval-2020-task-10-does-pre-trained","title":"IDS at SemEval-2020 Task 10: Does Pre-trained Language Model Know What to Emphasize?","date":"2020-07-24","arxiv_id":"2007.12390","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-gpgpu-to-recurrent-neural-network","title":"Applying GPGPU to Recurrent Neural Network Language Model based Fast Network Search in the Real-Time LVCSR","date":"2020-07-23","arxiv_id":"2007.11794","repositories_listed":0,"syntology":null},{"url":null,"slug":"plug-and-play-conversational-models-1","title":"Plug-and-Play Conversational Models","date":"2020-07-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-transformer-based-data-augmentation-with","title":"Deep Transformer based Data Augmentation with Subword Units for Morphologically Rich Online ASR","date":"2020-07-14","arxiv_id":"2007.06949","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-composition-learning-to-generate-from","title":"Neural Composition: Learning to Generate from Multiple Models","date":"2020-07-10","arxiv_id":"2007.16013","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-with-reduced-densities","title":"Language Modeling with Reduced Densities","date":"2020-07-08","arxiv_id":"2007.03834","repositories_listed":0,"syntology":null},{"url":null,"slug":"nlp-service-apis-and-models-for-efficient-1","title":"NLP Service APIs and Models for Efficient Registration of New Clients","date":"2020-07-07","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-go-transformer-natural-language-modeling","title":"The Go Transformer: Natural Language Modeling for Game Play","date":"2020-07-07","arxiv_id":"2007.03500","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmve-at-semeval-2020-task-4-commonsense","title":"LMVE at SemEval-2020 Task 4: Commonsense Validation and Explanation using Pretraining Language Model","date":"2020-07-06","arxiv_id":"2007.02540","repositories_listed":0,"syntology":null},{"url":null,"slug":"birds-of-a-feather-flock-together-satirical","title":"Birds of a Feather Flock Together: Satirical News Detection via Language Model Differentiation","date":"2020-07-04","arxiv_id":"2007.02164","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-data-augmentation-towards-better","title":"Text Data Augmentation: Towards better detection of spear-phishing emails","date":"2020-07-04","arxiv_id":"2007.02033","repositories_listed":0,"syntology":null},{"url":null,"slug":"facts-as-experts-adaptable-and-interpretable","title":"Facts as Experts: Adaptable and Interpretable Neural Memory over Symbolic Knowledge","date":"2020-07-02","arxiv_id":"2007.00849","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-explanations-on-ai-competency","title":"The Impact of Explanations on AI Competency Prediction in VQA","date":"2020-07-02","arxiv_id":"2007.00900","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-h-1-heads-is-better-than-h-heads-1","title":"A Mixture of h - 1 Heads is Better than h Heads","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-and-domain-aware-bert-for-cross","title":"Adversarial and Domain-Aware BERT for Cross-Domain Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-sensing-for-robotics-using-deep-learning","title":"AI Sensing for Robotics using Deep Learning based Visual and Language Modeling","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-subword-segmentation","title":"An Evaluation of Subword Segmentation Strategies for Neural Machine Translation of Morphologically Rich Languages","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"assisting-undergraduate-students-in-writing","title":"Assisting Undergraduate Students in Writing Spanish Methodology Sections","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-machine-translation-evaluation","title":"Automatic Machine Translation Evaluation using Source Language Inputs and Cross-lingual Language Model","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-poetry-generation-from-prosaic-text","title":"Automatic Poetry Generation from Prosaic Text","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-wikipedia-categories-improve-masked","title":"Can Wikipedia Categories Improve Masked Language Model Pretraining?","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-and-non-contextual-word-embeddings","title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"copybert-a-unified-approach-to-question","title":"CopyBERT: A Unified Approach to Question Generation with Self-Attention","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-unsupervised-sentiment","title":"Cross-Lingual Unsupervised Sentiment Classification with Multi-View Transfer Learning","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmet-a-reading-comprehension-paradigm-for","title":"DeepMet: A Reading Comprehension Paradigm for Token-level Metaphor Detection","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-transformer-with-sememe-knowledge","title":"Enhancing Transformer with Sememe Knowledge","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-self-attention-improves-rare-class","title":"How Self-Attention Improves Rare Class Performance in a Question-Answering Dialogue Agent","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effect-of-auxiliary","title":"Investigating the effect of auxiliary objectives for the automated grading of learner English speech transcriptions","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-masked-sequence-to-sequence-model-for","title":"Jointly Masked Sequence-to-Sequence Model for Non-Autoregressive Neural Machine Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"long-tail-predictions-with-continuous-output","title":"Long-Tail Predictions with Continuous-Output Language Models","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-code-switch-languages-using","title":"Modeling Code-Switch Languages Using Bilingual Parallel Corpus","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-contextual-historical-text","title":"Semi-supervised Contextual Historical Text Normalization","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"syntaxgym-an-online-platform-for-targeted","title":"SyntaxGym: An Online Platform for Targeted Evaluation of Language Models","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-afrl-iwslt-2020-systems-work-from-home","title":"The AFRL IWSLT 2020 Systems: Work-From-Home Edition","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tigrinya-automatic-speech-recognition-with","title":"Tigrinya Automatic Speech recognition with Morpheme based recognition units","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"to-pretrain-or-not-to-pretrain-examining-the-1","title":"To Pretrain or Not to Pretrain: Examining the Benefits of Pretrainng on Resource Rich Tasks","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-social-media-for-bitcoin-day-trading","title":"Using Social Media For Bitcoin Day Trading Behavior Prediction","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-does-bert-with-vision-look-at","title":"What Does BERT with Vision Look At?","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-auxiliary-tuning-and-its","title":"Technical Report: Auxiliary Tuning and its Application to Conditional Text Generation","date":"2020-06-30","arxiv_id":"2006.16823","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-aware-language-model-pretraining","title":"Knowledge-Aware Language Model Pretraining","date":"2020-06-29","arxiv_id":"2007.00655","repositories_listed":0,"syntology":null},{"url":null,"slug":"want-to-identify-extract-and-normalize","title":"Want to Identify, Extract and Normalize Adverse Drug Reactions in Tweets? Use RoBERTa","date":"2020-06-29","arxiv_id":"2006.16146","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-facts-knowledge-boosted-coherent","title":"Mind The Facts: Knowledge-Boosted Coherent Abstractive Text Summarization","date":"2020-06-27","arxiv_id":"2006.15435","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-window-for-dynamic-local-1","title":"Differentiable Window for Dynamic Local Attention","date":"2020-06-24","arxiv_id":"2006.13561","repositories_listed":0,"syntology":null},{"url":null,"slug":"clinical-predictive-keyboard-using","title":"Clinical Predictive Keyboard using Statistical and Neural Language Modeling","date":"2020-06-22","arxiv_id":"2006.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-software-naturalness-throughneural","title":"Exploring Software Naturalness through Neural Language Models","date":"2020-06-22","arxiv_id":"2006.12641","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-pretrain-or-not-to-pretrain-examining-the","title":"To Pretrain or Not to Pretrain: Examining the Benefits of Pretraining on Resource Rich Tasks","date":"2020-06-15","arxiv_id":"2006.08671","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-monolingual-model-to-low","title":"Transferring Monolingual Model to Low-Resource Language: The Case of Tigrinya","date":"2020-06-13","arxiv_id":"2006.07698","repositories_listed":0,"syntology":null},{"url":null,"slug":"examination-and-extension-of-strategies-for","title":"Examination and Extension of Strategies for Improving Personalized Language Modeling via Interpolation","date":"2020-06-09","arxiv_id":"2006.05469","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-transfer-learning-for","title":"Improving Cross-Lingual Transfer Learning for End-to-End Speech Recognition with Speech Translation","date":"2020-06-09","arxiv_id":"2006.05474","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-neural-text","title":"On the Effectiveness of Neural Text Generation based Data Augmentation for Recognition of Morphologically Rich Speech","date":"2020-06-09","arxiv_id":"2006.05129","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-for-formal-mathematics","title":"Mathematical Reasoning via Self-supervised Skip-tree Training","date":"2020-06-08","arxiv_id":"2006.04757","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-as-fact-checkers","title":"Language Models as Fact Checkers?","date":"2020-06-07","arxiv_id":"2006.04102","repositories_listed":0,"syntology":null},{"url":"/paper/a-dataset-and-benchmarks-for-multimedia","slug":"a-dataset-and-benchmarks-for-multimedia","title":"A Dataset and Benchmarks for Multimedia Social Analysis","date":"2020-06-05","arxiv_id":"2006.08335","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensorized-transformer-for-dynamical-systems","title":"Tensorized Transformer for Dynamical Systems Modeling","date":"2020-06-05","arxiv_id":"2006.03445","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-rnn-t-for-open-domain-asr","title":"Contextual RNN-T For Open Domain ASR","date":"2020-06-04","arxiv_id":"2006.03411","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-masking-for-language-models","title":"Position Masking for Language Models","date":"2020-06-02","arxiv_id":"2006.05676","repositories_listed":0,"syntology":null},{"url":null,"slug":"segatron-segment-aware-transformer-for","title":"Segatron: Segment-aware Transformer for Language Modeling and Understanding","date":"2020-06-02","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-french-language-models-for","title":"Contextualized French Language Models for Biomedical Named Entity Recognition","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lrg-at-semeval-2020-task-7-assessing-the","title":"LRG at SemEval-2020 Task 7: Assessing the Ability of BERT and Derivative Models to Perform Short-Edits based Humor Grading","date":"2020-05-31","arxiv_id":"2006.00607","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-for-unsupervised-parsing-with","title":"Self-Training for Unsupervised Parsing with PRPN","date":"2020-05-27","arxiv_id":"2005.13455","repositories_listed":0,"syntology":null},{"url":null,"slug":"syntactic-structure-distillation-pretraining","title":"Syntactic Structure Distillation Pretraining For Bidirectional Encoders","date":"2020-05-27","arxiv_id":"2005.13482","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-text-and-image-mutual-translation","title":"TIME: Text and Image Mutual-Translation Adversarial Networks","date":"2020-05-27","arxiv_id":"2005.13192","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-relation-extraction-from-1","title":"Unsupervised Relation Extraction from Language Models using Constrained Cloze Completion","date":"2020-05-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qdkt-question-centric-deep-knowledge-tracing","title":"qDKT: Question-centric Deep Knowledge Tracing","date":"2020-05-25","arxiv_id":"2005.12442","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-does-maml-work-the-best-an-empirical","title":"When does MAML Work the Best? An Empirical Study on Model-Agnostic Meta-Learning in NLP Applications","date":"2020-05-24","arxiv_id":"2005.11700","repositories_listed":0,"syntology":null},{"url":"/paper/asapp-asr-multistream-cnn-and-self-attentive","slug":"asapp-asr-multistream-cnn-and-self-attentive","title":"ASAPP-ASR: Multistream CNN and Self-Attentive SRU for SOTA Speech Recognition","date":"2020-05-21","arxiv_id":"2005.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-text-data-using-hybrid-transformer","title":"Leveraging Text Data Using Hybrid Transformer-LSTM Based End-to-End ASR in Transfer Learning","date":"2020-05-21","arxiv_id":"2005.10407","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-for-debiased-candidate","title":"Contrastive Learning for Debiased Candidate Generation in Large-Scale Recommender Systems","date":"2020-05-20","arxiv_id":"2005.12964","repositories_listed":0,"syntology":null},{"url":null,"slug":"early-stage-lm-integration-using-local-and","title":"Early Stage LM Integration Using Local and Global Log-Linear Combination","date":"2020-05-20","arxiv_id":"2005.10049","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-large-margin-softmax-in","title":"Investigation of Large-Margin Softmax in Neural Language Modeling","date":"2020-05-20","arxiv_id":"2005.10089","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-proper-noun-recognition-in-end-to","title":"Improving Proper Noun Recognition in End-to-End ASR By Customization of the MWER Loss Criterion","date":"2020-05-19","arxiv_id":"2005.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"approaches-to-improving-recognition-of","title":"Approaches to Improving Recognition of Underrepresented Named Entities in Hybrid ASR Systems","date":"2020-05-18","arxiv_id":"2005.08742","repositories_listed":0,"syntology":null},{"url":null,"slug":"semeval-2020-task-5-detecting-counterfactuals","title":"Yseop at SemEval-2020 Task 5: Cascaded BERT Language Model for Counterfactual Statement Analysis","date":"2020-05-18","arxiv_id":"2005.08519","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualizing-asr-lattice-rescoring-with","title":"Contextualizing ASR Lattice Rescoring with Hybrid Pointer Network Language Model","date":"2020-05-15","arxiv_id":"2005.07394","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-communication-meets-natural","title":"Multi-agent Communication meets Natural Language: Synergies between Functional and Structural Language Learning","date":"2020-05-14","arxiv_id":"2005.07064","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-do-not-need-more-data-improving-end-to","title":"You Do Not Need More Data: Improving End-To-End Speech Recognition by Text-To-Speech Data Augmentation","date":"2020-05-14","arxiv_id":"2005.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-h-1-heads-is-better-than-h-heads","title":"A Mixture of $h-1$ Heads is Better than $h$ Heads","date":"2020-05-13","arxiv_id":"2005.06537","repositories_listed":0,"syntology":null},{"url":"/paper/parallel-corpus-filtering-via-pre-trained","slug":"parallel-corpus-filtering-via-pre-trained","title":"Parallel Corpus Filtering via Pre-trained Language Models","date":"2020-05-13","arxiv_id":"2005.06166","repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-evidence-generation-and-injection","title":"Commonsense Evidence Generation and Injection in Reading Comprehension","date":"2020-05-11","arxiv_id":"2005.05240","repositories_listed":0,"syntology":null}],"record_sha256":"b4d9175db61cba1fd6909feb012ebaa7ec5505b1a500c595e8c32a14979bb94b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}