{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modelling/papers/154","list_of":"/task/language-modelling","task":"Language Modelling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":154,"pages_in_order":177,"rows_per_page":100,"rows":[15301,15400],"of":17610,"counts":{"archive_papers_tagged":17610,"with_a_code_link":7012,"where_syntology_ran_a_sample":2428,"not_listed_spam_title":0,"listed":17610,"listed_where_code_ran":2428,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2027,"every_run_a_failure_of_syntologys_instrument":401,"listed_with_a_run_with_no_instrument_failure":2027,"listed_every_run_a_failure_of_syntologys_instrument":401,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modelling","prev":"/task/language-modelling/papers/153","next":"/task/language-modelling/papers/155","papers":[{"url":null,"slug":"improving-tail-performance-of-a-deliberation","title":"Improving Tail Performance of a Deliberation E2E ASR Model Using a Large Text Corpus","date":"2020-08-24","arxiv_id":"2008.10491","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-language-model-with-entanglement","title":"Quantum Language Model with Entanglement Embedding for Question Answering","date":"2020-08-23","arxiv_id":"2008.09943","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-event-extractors-to-medical-data","title":"Adapting Event Extractors to Medical Data: Bridging the Covariate Shift","date":"2020-08-21","arxiv_id":"2008.09266","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-a-knowledge-graph-from","title":"AutoKG: Constructing Virtual Knowledge Graphs from Unstructured Documents for Question Answering","date":"2020-08-20","arxiv_id":"2008.08995","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-useful-sentence-representations","title":"Discovering Useful Sentence Representations from Large Pretrained Language Models","date":"2020-08-20","arxiv_id":"2008.09049","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-extract-attribute-value-from","title":"Learning to Extract Attribute Value from Product via Question Answering: A Multi-task Approach","date":"2020-08-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uob-at-semeval-2020-task-12-boosting-bert","title":"UoB at SemEval-2020 Task 12: Boosting BERT with Corpus Level Information","date":"2020-08-19","arxiv_id":"2008.08547","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-language-model-and-parallel-bi","title":"Complementary Language Model and Parallel Bi-LRNN for False Trigger Mitigation","date":"2020-08-18","arxiv_id":"2008.08113","repositories_listed":0,"syntology":null},{"url":null,"slug":"adding-recurrence-to-pretrained-transformers","title":"Adding Recurrence to Pretrained Transformers for Improved Efficiency and Context Size","date":"2020-08-16","arxiv_id":"2008.07027","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-multi-domain-language-model-for","title":"Adaptable Multi-Domain Language Model for Transformer ASR","date":"2020-08-14","arxiv_id":"2008.06208","repositories_listed":0,"syntology":null},{"url":null,"slug":"hate-speech-detection-and-racial-bias","title":"Hate Speech Detection and Racial Bias Mitigation in Social Media based on BERT model","date":"2020-08-14","arxiv_id":"2008.06460","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-learning-mechanism-for-speech","title":"Prosody Learning Mechanism for Speech Synthesis System Without Text Length Limit","date":"2020-08-13","arxiv_id":"2008.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-with-bidirectional-decoder-for","title":"Transformer with Bidirectional Decoder for Speech Recognition","date":"2020-08-11","arxiv_id":"2008.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-neural-query-auto-completion","title":"Efficient Neural Query Auto Completion","date":"2020-08-06","arxiv_id":"2008.02879","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastlr-non-autoregressive-lipreading-model","title":"FastLR: Non-Autoregressive Lipreading Model with Integrate-and-Fire","date":"2020-08-06","arxiv_id":"2008.02516","repositories_listed":0,"syntology":null},{"url":null,"slug":"6veclm-language-modeling-in-vector-space-for","title":"6VecLM: Language Modeling in Vector Space for IPv6 Target Generation","date":"2020-08-05","arxiv_id":"2008.02213","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-mdi-adaptation-for-n-gram-language","title":"Efficient MDI Adaptation for n-gram Language Models","date":"2020-08-05","arxiv_id":"2008.02385","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-visual-representations-with-caption","title":"Learning Visual Representations with Caption Annotations","date":"2020-08-04","arxiv_id":"2008.01392","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-effects-of-implicit-and-explicit","title":"A Study on Effects of Implicit and Explicit Language Model Information for DBLSTM-CTC Based Handwriting Recognition","date":"2020-07-31","arxiv_id":"2008.01532","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-vector-enhanced-lstm-language-model","title":"Future Vector Enhanced LSTM Language Model for LVCSR","date":"2020-07-31","arxiv_id":"2008.01832","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-learning-universal-representations-across","title":"On Learning Universal Representations Across Languages","date":"2020-07-31","arxiv_id":"2007.15960","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-federated-learning-1","title":"Communication-Efficient Federated Learning via Optimal Client Sampling","date":"2020-07-30","arxiv_id":"2007.15197","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-therapy-target-discovery-with","title":"COVID-19 therapy target discovery with context-aware literature mining","date":"2020-07-30","arxiv_id":"2007.15681","repositories_listed":0,"syntology":null},{"url":null,"slug":"growing-efficient-deep-networks-by-structured","title":"Growing Efficient Deep Networks by Structured Continuous Sparsification","date":"2020-07-30","arxiv_id":"2007.15353","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-deep-neural-networks-via-layer","title":"Compressing Deep Neural Networks via Layer Fusion","date":"2020-07-29","arxiv_id":"2007.14917","repositories_listed":0,"syntology":null},{"url":null,"slug":"guir-at-semeval-2020-task-12-domain-tuned","title":"GUIR at SemEval-2020 Task 12: Domain-Tuned Contextualized Models for Offensive Language Detection","date":"2020-07-28","arxiv_id":"2007.14477","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensorcoder-dimension-wise-attention-via","title":"TensorCoder: Dimension-Wise Attention via Tensor Representation for Natural Language Modeling","date":"2020-07-28","arxiv_id":"2008.01547","repositories_listed":0,"syntology":null},{"url":null,"slug":"ids-at-semeval-2020-task-10-does-pre-trained","title":"IDS at SemEval-2020 Task 10: Does Pre-trained Language Model Know What to Emphasize?","date":"2020-07-24","arxiv_id":"2007.12390","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-gpgpu-to-recurrent-neural-network","title":"Applying GPGPU to Recurrent Neural Network Language Model based Fast Network Search in the Real-Time LVCSR","date":"2020-07-23","arxiv_id":"2007.11794","repositories_listed":0,"syntology":null},{"url":null,"slug":"plug-and-play-conversational-models-1","title":"Plug-and-Play Conversational Models","date":"2020-07-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-perspective-semantic-information","title":"Multi-Perspective Semantic Information Retrieval in the Biomedical Domain","date":"2020-07-17","arxiv_id":"2008.01526","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-speaker-verification-with","title":"Cross-Lingual Speaker Verification with Domain-Balanced Hard Prototype Mining and Language-Dependent Score Normalization","date":"2020-07-15","arxiv_id":"2007.07689","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-transformer-based-data-augmentation-with","title":"Deep Transformer based Data Augmentation with Subword Units for Morphologically Rich Online ASR","date":"2020-07-14","arxiv_id":"2007.06949","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioned-time-dilated-convolutions-for","title":"Conditioned Time-Dilated Convolutions for Sound Event Detection","date":"2020-07-10","arxiv_id":"2007.05183","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-composition-learning-to-generate-from","title":"Neural Composition: Learning to Generate from Multiple Models","date":"2020-07-10","arxiv_id":"2007.16013","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-with-reduced-densities","title":"Language Modeling with Reduced Densities","date":"2020-07-08","arxiv_id":"2007.03834","repositories_listed":0,"syntology":null},{"url":null,"slug":"nlp-service-apis-and-models-for-efficient-1","title":"NLP Service APIs and Models for Efficient Registration of New Clients","date":"2020-07-07","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-go-transformer-natural-language-modeling","title":"The Go Transformer: Natural Language Modeling for Game Play","date":"2020-07-07","arxiv_id":"2007.03500","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-contextual-embeddings-for-address","title":"Deep Contextual Embeddings for Address Classification in E-commerce","date":"2020-07-06","arxiv_id":"2007.03020","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmve-at-semeval-2020-task-4-commonsense","title":"LMVE at SemEval-2020 Task 4: Commonsense Validation and Explanation using Pretraining Language Model","date":"2020-07-06","arxiv_id":"2007.02540","repositories_listed":0,"syntology":null},{"url":null,"slug":"cord19sts-covid-19-semantic-textual","title":"CORD19STS: COVID-19 Semantic Textual Similarity Dataset","date":"2020-07-05","arxiv_id":"2007.02461","repositories_listed":0,"syntology":null},{"url":null,"slug":"birds-of-a-feather-flock-together-satirical","title":"Birds of a Feather Flock Together: Satirical News Detection via Language Model Differentiation","date":"2020-07-04","arxiv_id":"2007.02164","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-data-augmentation-towards-better","title":"Text Data Augmentation: Towards better detection of spear-phishing emails","date":"2020-07-04","arxiv_id":"2007.02033","repositories_listed":0,"syntology":null},{"url":null,"slug":"facts-as-experts-adaptable-and-interpretable","title":"Facts as Experts: Adaptable and Interpretable Neural Memory over Symbolic Knowledge","date":"2020-07-02","arxiv_id":"2007.00849","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-explanations-on-ai-competency","title":"The Impact of Explanations on AI Competency Prediction in VQA","date":"2020-07-02","arxiv_id":"2007.00900","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-h-1-heads-is-better-than-h-heads-1","title":"A Mixture of h - 1 Heads is Better than h Heads","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-and-domain-aware-bert-for-cross","title":"Adversarial and Domain-Aware BERT for Cross-Domain Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-sensing-for-robotics-using-deep-learning","title":"AI Sensing for Robotics using Deep Learning based Visual and Language Modeling","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-subword-segmentation","title":"An Evaluation of Subword Segmentation Strategies for Neural Machine Translation of Morphologically Rich Languages","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"assisting-undergraduate-students-in-writing","title":"Assisting Undergraduate Students in Writing Spanish Methodology Sections","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-machine-translation-evaluation","title":"Automatic Machine Translation Evaluation using Source Language Inputs and Cross-lingual Language Model","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-poetry-generation-from-prosaic-text","title":"Automatic Poetry Generation from Prosaic Text","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-wikipedia-categories-improve-masked","title":"Can Wikipedia Categories Improve Masked Language Model Pretraining?","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"checkpoint-reranking-an-approach-to-select","title":"Checkpoint Reranking: An Approach to Select Better Hypothesis for Neural Machine Translation Systems","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-and-non-contextual-word-embeddings","title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"copybert-a-unified-approach-to-question","title":"CopyBERT: A Unified Approach to Question Generation with Self-Attention","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-unsupervised-sentiment","title":"Cross-Lingual Unsupervised Sentiment Classification with Multi-View Transfer Learning","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmet-a-reading-comprehension-paradigm-for","title":"DeepMet: A Reading Comprehension Paradigm for Token-level Metaphor Detection","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"do-transformers-need-deep-long-range-memory","title":"Do Transformers Need Deep Long-Range Memory?","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-transformer-with-sememe-knowledge","title":"Enhancing Transformer with Sememe Knowledge","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-self-attention-improves-rare-class","title":"How Self-Attention Improves Rare Class Performance in a Question-Answering Dialogue Agent","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effect-of-auxiliary","title":"Investigating the effect of auxiliary objectives for the automated grading of learner English speech transcriptions","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-masked-sequence-to-sequence-model-for","title":"Jointly Masked Sequence-to-Sequence Model for Non-Autoregressive Neural Machine Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"long-tail-predictions-with-continuous-output","title":"Long-Tail Predictions with Continuous-Output Language Models","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"max-margin-incremental-ccg-parsing","title":"Max-Margin Incremental CCG Parsing","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-code-switch-languages-using","title":"Modeling Code-Switch Languages Using Bilingual Parallel Corpus","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"monolingual-corpus-creation-and-evaluation-of","title":"Monolingual corpus creation and evaluation of truly low-resource languages from Peru","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-contextual-historical-text","title":"Semi-supervised Contextual Historical Text Normalization","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"syntaxgym-an-online-platform-for-targeted","title":"SyntaxGym: An Online Platform for Targeted Evaluation of Language Models","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-afrl-iwslt-2020-systems-work-from-home","title":"The AFRL IWSLT 2020 Systems: Work-From-Home Edition","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tigrinya-automatic-speech-recognition-with","title":"Tigrinya Automatic Speech recognition with Morpheme based recognition units","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"to-pretrain-or-not-to-pretrain-examining-the-1","title":"To Pretrain or Not to Pretrain: Examining the Benefits of Pretrainng on Resource Rich Tasks","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-social-media-for-bitcoin-day-trading","title":"Using Social Media For Bitcoin Day Trading Behavior Prediction","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-does-bert-with-vision-look-at","title":"What Does BERT with Vision Look At?","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-auxiliary-tuning-and-its","title":"Technical Report: Auxiliary Tuning and its Application to Conditional Text Generation","date":"2020-06-30","arxiv_id":"2006.16823","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-aware-language-model-pretraining","title":"Knowledge-Aware Language Model Pretraining","date":"2020-06-29","arxiv_id":"2007.00655","repositories_listed":0,"syntology":null},{"url":null,"slug":"want-to-identify-extract-and-normalize","title":"Want to Identify, Extract and Normalize Adverse Drug Reactions in Tweets? Use RoBERTa","date":"2020-06-29","arxiv_id":"2006.16146","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-facts-knowledge-boosted-coherent","title":"Mind The Facts: Knowledge-Boosted Coherent Abstractive Text Summarization","date":"2020-06-27","arxiv_id":"2006.15435","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-window-for-dynamic-local-1","title":"Differentiable Window for Dynamic Local Attention","date":"2020-06-24","arxiv_id":"2006.13561","repositories_listed":0,"syntology":null},{"url":null,"slug":"clinical-predictive-keyboard-using","title":"Clinical Predictive Keyboard using Statistical and Neural Language Modeling","date":"2020-06-22","arxiv_id":"2006.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-software-naturalness-throughneural","title":"Exploring Software Naturalness through Neural Language Models","date":"2020-06-22","arxiv_id":"2006.12641","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooking-is-all-about-people-comment","title":"Cooking Is All About People: Comment Classification On Cookery Channels Using BERT and Classification Models (Malayalam-English Mix-Code)","date":"2020-06-15","arxiv_id":"2007.04249","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-pretrain-or-not-to-pretrain-examining-the","title":"To Pretrain or Not to Pretrain: Examining the Benefits of Pretraining on Resource Rich Tasks","date":"2020-06-15","arxiv_id":"2006.08671","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-monolingual-model-to-low","title":"Transferring Monolingual Model to Low-Resource Language: The Case of Tigrinya","date":"2020-06-13","arxiv_id":"2006.07698","repositories_listed":0,"syntology":null},{"url":null,"slug":"examination-and-extension-of-strategies-for","title":"Examination and Extension of Strategies for Improving Personalized Language Modeling via Interpolation","date":"2020-06-09","arxiv_id":"2006.05469","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-transfer-learning-for","title":"Improving Cross-Lingual Transfer Learning for End-to-End Speech Recognition with Speech Translation","date":"2020-06-09","arxiv_id":"2006.05474","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-neural-text","title":"On the Effectiveness of Neural Text Generation based Data Augmentation for Recognition of Morphologically Rich Speech","date":"2020-06-09","arxiv_id":"2006.05129","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-for-formal-mathematics","title":"Mathematical Reasoning via Self-supervised Skip-tree Training","date":"2020-06-08","arxiv_id":"2006.04757","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lipschitz-constant-of-self-attention","title":"The Lipschitz Constant of Self-Attention","date":"2020-06-08","arxiv_id":"2006.04710","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-as-fact-checkers","title":"Language Models as Fact Checkers?","date":"2020-06-07","arxiv_id":"2006.04102","repositories_listed":0,"syntology":null},{"url":"/paper/a-dataset-and-benchmarks-for-multimedia","slug":"a-dataset-and-benchmarks-for-multimedia","title":"A Dataset and Benchmarks for Multimedia Social Analysis","date":"2020-06-05","arxiv_id":"2006.08335","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensorized-transformer-for-dynamical-systems","title":"Tensorized Transformer for Dynamical Systems Modeling","date":"2020-06-05","arxiv_id":"2006.03445","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-rnn-t-for-open-domain-asr","title":"Contextual RNN-T For Open Domain ASR","date":"2020-06-04","arxiv_id":"2006.03411","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-british-sign-language-1","title":"Transfer Learning for British Sign Language Modelling","date":"2020-06-03","arxiv_id":"2006.02144","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-masking-for-language-models","title":"Position Masking for Language Models","date":"2020-06-02","arxiv_id":"2006.05676","repositories_listed":0,"syntology":null},{"url":null,"slug":"segatron-segment-aware-transformer-for","title":"Segatron: Segment-aware Transformer for Language Modeling and Understanding","date":"2020-06-02","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-french-language-models-for","title":"Contextualized French Language Models for Biomedical Named Entity Recognition","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lrg-at-semeval-2020-task-7-assessing-the","title":"LRG at SemEval-2020 Task 7: Assessing the Ability of BERT and Derivative Models to Perform Short-Edits based Humor Grading","date":"2020-05-31","arxiv_id":"2006.00607","repositories_listed":0,"syntology":null}],"record_sha256":"d0c86ff3c0f337d8a51e02ad9b6adb8fdc761ec8b3cda7577ee595f654f52c36","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}