{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modelling/papers/150","list_of":"/task/language-modelling","task":"Language Modelling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":150,"pages_in_order":177,"rows_per_page":100,"rows":[14901,15000],"of":17610,"counts":{"archive_papers_tagged":17610,"with_a_code_link":7012,"where_syntology_ran_a_sample":2428,"not_listed_spam_title":0,"listed":17610,"listed_where_code_ran":2428,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2027,"every_run_a_failure_of_syntologys_instrument":401,"listed_with_a_run_with_no_instrument_failure":2027,"listed_every_run_a_failure_of_syntologys_instrument":401,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modelling","prev":"/task/language-modelling/papers/149","next":"/task/language-modelling/papers/151","papers":[{"url":null,"slug":"train-your-classifier-first-cascade-neural","title":"Train your classifier first: Cascade Neural Networks Training from upper layers to lower layers","date":"2021-02-09","arxiv_id":"2102.04697","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-fake-cyber-threat-intelligence","title":"Generating Fake Cyber Threat Intelligence Using Transformer-Based Models","date":"2021-02-08","arxiv_id":"2102.04351","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-he-wink-or-does-he-nod-a-challenging","title":"Does He Wink or Does He Nod? A Challenging Benchmark for Evaluating Word Understanding of Language Models","date":"2021-02-06","arxiv_id":"2102.03596","repositories_listed":0,"syntology":null},{"url":null,"slug":"intermediate-loss-regularization-for-ctc","title":"Intermediate Loss Regularization for CTC-based Speech Recognition","date":"2021-02-05","arxiv_id":"2102.03216","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-emails-and-drafting-responses","title":"Understanding Emails and Drafting Responses -- An Approach Using GPT-3","date":"2021-02-05","arxiv_id":"2102.03062","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-semiparametric-language-models","title":"Adaptive Semiparametric Language Models","date":"2021-02-04","arxiv_id":"2102.02557","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-capabilities-limitations","title":"Understanding the Capabilities, Limitations, and Societal Impact of Large Language Models","date":"2021-02-04","arxiv_id":"2102.02503","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-number-of-filters-of-convolutional","title":"Effects of Number of Filters of Convolutional Layers on Speech Recognition Model Accuracy","date":"2021-02-03","arxiv_id":"2102.02326","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-purpose-speech-representation","title":"General-Purpose Speech Representation Learning through a Self-Supervised Multi-Granularity Framework","date":"2021-02-03","arxiv_id":"2102.01930","repositories_listed":0,"syntology":null},{"url":null,"slug":"hebert-hebemo-a-hebrew-bert-model-and-a-tool","title":"HeBERT & HebEMO: a Hebrew BERT Model and a Tool for Polarity Analysis and Emotion Recognition","date":"2021-02-03","arxiv_id":"2102.01909","repositories_listed":0,"syntology":null},{"url":null,"slug":"clickbait-headline-detection-in-indonesian","title":"Clickbait Headline Detection in Indonesian News Sites using Multilingual Bidirectional Encoder Representations from Transformers (M-BERT)","date":"2021-02-02","arxiv_id":"2102.01497","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-training-for-domain","title":"Internal Language Model Training for Domain-Adaptive End-to-End Speech Recognition","date":"2021-02-02","arxiv_id":"2102.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"end2end-acoustic-to-semantic-transduction","title":"End2End Acoustic to Semantic Transduction","date":"2021-02-01","arxiv_id":"2102.01013","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-contrastive-pre-training-for","title":"Adversarial Contrastive Pre-training for Protein Sequences","date":"2021-01-31","arxiv_id":"2102.00466","repositories_listed":0,"syntology":null},{"url":null,"slug":"shuftext-a-simple-black-box-approach-to","title":"ShufText: A Simple Black Box Approach to Evaluate the Fragility of Text Classification Models","date":"2021-01-30","arxiv_id":"2102.00238","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-by-simply-fine-tuning-bert","title":"Speech Recognition by Simply Fine-tuning BERT","date":"2021-01-30","arxiv_id":"2102.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"bcn2brno-asr-system-fusion-for-albayzin-2020","title":"BCN2BRNO: ASR System Fusion for Albayzin 2020 Speech to Text Challenge","date":"2021-01-29","arxiv_id":"2101.12729","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-grams-bayesian-differential-privacy","title":"N-grams Bayesian Differential Privacy","date":"2021-01-29","arxiv_id":"2101.12736","repositories_listed":0,"syntology":null},{"url":null,"slug":"bertau-itau-bert-for-digital-customer-service","title":"BERTaú: Itaú BERT for digital customer service","date":"2021-01-28","arxiv_id":"2101.12015","repositories_listed":0,"syntology":null},{"url":null,"slug":"drag-director-generator-language-modelling","title":"DRAG: Director-Generator Language Modelling Framework for Non-Parallel Author Stylized Rewriting","date":"2021-01-28","arxiv_id":"2101.11836","repositories_listed":0,"syntology":null},{"url":null,"slug":"protoda-efficient-transfer-learning-for-few","title":"ProtoDA: Efficient Transfer Learning for Few-Shot Intent Classification","date":"2021-01-28","arxiv_id":"2101.11753","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-neuro-symbolic-module-1","title":"Weakly Supervised Neuro-Symbolic Module Networks for Numerical Reasoning","date":"2021-01-28","arxiv_id":"2101.11802","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-for-personalised-learning-the-long","title":"Developing for personalised learning: the long road from educational objectives to development and feedback","date":"2021-01-27","arxiv_id":"2101.11333","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modelling-as-a-multi-task-problem","title":"Language Modelling as a Multi-Task Problem","date":"2021-01-27","arxiv_id":"2101.11287","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-zero-shot-cross-lingual-transfer-in","title":"Analyzing Zero-shot Cross-lingual Transfer in Supervised NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10649","repositories_listed":0,"syntology":null},{"url":null,"slug":"climp-a-benchmark-for-chinese-language-model","title":"CLiMP: A Benchmark for Chinese Language Model Evaluation","date":"2021-01-26","arxiv_id":"2101.11131","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-bert-and-albert-sentence","title":"Evaluation of BERT and ALBERT Sentence Embedding Performance on Downstream NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10642","repositories_listed":0,"syntology":null},{"url":null,"slug":"disambiguating-symbolic-expressions-in-1","title":"Disambiguating Symbolic Expressions in Informal Documents","date":"2021-01-25","arxiv_id":"2101.11716","repositories_listed":0,"syntology":null},{"url":null,"slug":"mermaid-metaphor-generation-with-symbolism-1","title":"MERMAID: Metaphor Generation with Symbolism and Discriminative Decoding","date":"2021-01-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"k-neighbor-based-curriculum-sampling-for","title":"$k$-Neighbor Based Curriculum Sampling for Sequence Prediction","date":"2021-01-22","arxiv_id":"2101.09313","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-multiple-parallel-phrase","title":"The Impact of Multiple Parallel Phrase Suggestions on Email Input and Composition Behaviour of Native and Non-Native English Writers","date":"2021-01-22","arxiv_id":"2101.09157","repositories_listed":0,"syntology":null},{"url":null,"slug":"wechat-ai-s-submission-for-dstc9-interactive","title":"WeChat AI & ICT's Submission for DSTC9 Interactive Dialogue Evaluation Track","date":"2021-01-20","arxiv_id":"2101.07947","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-generalization-in-dialog-state","title":"Zero-shot Generalization in Dialog State Tracking through Generative Question Answering","date":"2021-01-20","arxiv_id":"2101.08333","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-wav2vec2-0-and-bert-into-end-to-end","title":"Efficiently Fusing Pretrained Acoustic and Linguistic Encoders for Low-resource Speech Recognition","date":"2021-01-17","arxiv_id":"2101.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"grid-search-hyperparameter-benchmarking-of","title":"Grid Search Hyperparameter Benchmarking of BERT, ALBERT, and LongFormer on DuoRC","date":"2021-01-15","arxiv_id":"2101.06326","repositories_listed":0,"syntology":null},{"url":null,"slug":"kdlsq-bert-a-quantized-bert-combining","title":"KDLSQ-BERT: A Quantized Bert Combining Knowledge Distillation with Learned Step Size Quantization","date":"2021-01-15","arxiv_id":"2101.05938","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-language-model-fine-tuning","title":"Transformer-based Language Model Fine-tuning Methods for COVID-19 Fake News Detection","date":"2021-01-14","arxiv_id":"2101.05509","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-commonsense-causal-reasoning-by","title":"Improving Commonsense Causal Reasoning by Adversarial Training and Data Augmentation","date":"2021-01-13","arxiv_id":"2101.04966","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-deep-learning-approaches-for","title":"Evaluating Deep Learning Approaches for Covid19 Fake News Detection","date":"2021-01-11","arxiv_id":"2101.04012","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-better-sentence-representation-with","title":"Learning Better Sentence Representation with Syntax Information","date":"2021-01-09","arxiv_id":"2101.03343","repositories_listed":0,"syntology":null},{"url":null,"slug":"misspelling-correction-with-pre-trained","title":"Misspelling Correction with Pre-trained Contextual Language Model","date":"2021-01-08","arxiv_id":"2101.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-optimized-n-gram-for-mobile-devices","title":"Real-Time Optimized N-gram For Mobile Devices","date":"2021-01-07","arxiv_id":"2101.03967","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-aware-neural-language-models-for","title":"Domain-aware Neural Language Models for Speech Recognition","date":"2021-01-05","arxiv_id":"2101.03229","repositories_listed":0,"syntology":null},{"url":null,"slug":"edatlas-an-efficient-disambiguation-algorithm","title":"edATLAS: An Efficient Disambiguation Algorithm for Texting in Languages with Abugida Scripts","date":"2021-01-05","arxiv_id":"2101.03916","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-fly-attention-modularization-for","title":"On-the-Fly Attention Modulation for Neural Generation","date":"2021-01-02","arxiv_id":"2101.00371","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-total-variation-regularized-softmax","title":"Graphmax for Text Generation","date":"2021-01-01","arxiv_id":"2101.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"adding-recurrence-to-pretrained-transformers-1","title":"Adding Recurrence to Pretrained Transformers","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-neural-networks-with-variance","title":"Bayesian Neural Networks with Variance Propagation for Uncertainty Evaluation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"block-skim-transformer-for-efficient-question","title":"Block Skim Transformer for Efficient Question Answering","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bros-a-pre-trained-language-model-for","title":"BROS: A Pre-trained Language Model for Understanding Texts in Document","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-temperature-for-language","title":"Context-Aware Temperature for Language Modeling","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-knowledge-distillation-for","title":"Contextual Knowledge Distillation for Transformer Compression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-transfer-learning-for-pre","title":"Cross-lingual Transfer Learning for Pre-trained Contextualized Language Models","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dact-bert-increasing-the-efficiency-and","title":"DACT-BERT: Increasing the efficiency and interpretability of BERT by using adaptive computation time.","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-slot-relationship-modeling-using-a-pre","title":"Domain-slot Relationship Modeling using a Pre-trained Language Encoder for Multi-Domain Dialogue State Tracking","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-chess-blindfolded","title":"Learning Chess Blindfolded","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-representation-in-transformer","title":"Memory Representation in Transformer","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mongoose-a-learnable-lsh-framework-for","title":"MONGOOSE: A Learnable LSH Framework for Efficient Neural Network Training","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-spatio-temporal-reasoning-with-object","title":"Neural spatio-temporal reasoning with object-centric self-supervised learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"non-iterative-parallel-text-generation-via","title":"Non-iterative Parallel Text Generation via Glancing Transformer","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-linguistic-similarities-to","title":"On the use of linguistic similarities to improve Neural Machine Translation for African Languages","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-off-policy-learning-balance-accuracy","title":"Partial Off-Policy Learning: Balance Accuracy and Diversity for Human-Oriented Image Captioning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-text-to-text-transformers-to","title":"Pre-training Text-to-Text Transformers to Write and Reason with Concepts","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrain-knowledge-aware-language-models","title":"Pretrain Knowledge-Aware Language Models","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refine-and-imitate-reducing-repetition-and","title":"Refine and Imitate: Reducing Repetition and Inconsistency in Dialogue Generation via Reinforcement Learning and Human Demonstration","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-and-bias-in-multilingual-nlp","title":"Representation and Bias in Multilingual NLP: Insights from Controlled Experiments on Conditional Language Modeling","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"romul-scale-adaptative-population-based","title":"ROMUL: Scale Adaptative Population Based Training","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/score-pre-training-for-context-representation","slug":"score-pre-training-for-context-representation","title":"SCoRe: Pre-Training for Context Representation in Conversational Semantic Parsing","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-features-how-gru-and-lstm","title":"SEQUENCE-LEVEL FEATURES: HOW GRU AND LSTM CELLS CAPTURE N-GRAMS","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/subformer-a-parameter-reduced-transformer","slug":"subformer-a-parameter-reduced-transformer","title":"Subformer: A Parameter Reduced Transformer","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"syntactic-relevance-xlnet-word-embedding","title":"Syntactic Relevance XLNet Word Embedding Generation in Low-Resource Machine Translation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizer-rethinking-self-attention-for","title":"Synthesizer: Rethinking Self-Attention for Transformer Models","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-agnostic-and-adaptive-size-bert","title":"Task-Agnostic and Adaptive-Size BERT Compression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-document-clustering-wordnet-vs-tf-idf-vs","title":"Text Document Clustering: Wordnet vs. TF-IDF vs. Word Embeddings","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/towards-practical-second-order-optimization","slug":"towards-practical-second-order-optimization","title":"Towards Practical Second Order Optimization for Deep Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-protein-language-models-are","title":"Transformer protein language models are unsupervised structure learners","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-ql-a-step-towards-making","title":"Transformer-QL: A Step Towards Making Transformer Network Quadratically Large","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"translation-memory-guided-neural-machine","title":"Translation Memory Guided Neural Machine Translation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-sentence-representations-learning","title":"Universal Sentence Representations Learning with Conditional Masked Language Model","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refine-and-imitate-reducing-repetition-and-1","title":"Refine and Imitate: Reducing Repetition and Inconsistency in Persuasion Dialogues via Reinforcement Learning and Human Demonstration","date":"2020-12-31","arxiv_id":"2012.15375","repositories_listed":0,"syntology":null},{"url":null,"slug":"studying-strategically-learning-to-mask-for","title":"Studying Strategically: Learning to Mask for Closed-book QA","date":"2020-12-31","arxiv_id":"2012.15856","repositories_listed":0,"syntology":null},{"url":null,"slug":"verb-knowledge-injection-for-multilingual","title":"Verb Knowledge Injection for Multilingual Event Processing","date":"2020-12-31","arxiv_id":"2012.15421","repositories_listed":0,"syntology":null},{"url":null,"slug":"xlm-t-scaling-up-multilingual-machine","title":"XLM-T: Scaling up Multilingual Machine Translation with Pretrained Cross-lingual Transformer Encoders","date":"2020-12-31","arxiv_id":"2012.15547","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-sequence-to-sequence-models-crack","title":"Can Sequence-to-Sequence Models Crack Substitution Ciphers?","date":"2020-12-30","arxiv_id":"2012.15229","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-pre-trained-language-model-with","title":"Enhancing Pre-trained Language Model with Lexical Simplification","date":"2020-12-30","arxiv_id":"2012.15070","repositories_listed":0,"syntology":null},{"url":null,"slug":"reservoir-transformer","title":"Reservoir Transformers","date":"2020-12-30","arxiv_id":"2012.15045","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmv-bert-contrastive-multi-vocab-pretraining","title":"CMV-BERT: Contrastive multi-vocab pretraining of BERT","date":"2020-12-29","arxiv_id":"2012.14763","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-adversarial-examples-in-chinese","title":"Generating Adversarial Examples in Chinese Texts Using Sentence-Pieces","date":"2020-12-29","arxiv_id":"2012.14769","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-handwritten-text-recognition-with-n","title":"Enhancing Handwritten Text Recognition with N-gram sequence decomposition and Multitask Learning","date":"2020-12-28","arxiv_id":"2012.14459","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-sentence-representation-learning","title":"Universal Sentence Representation Learning with Conditional Masked Language Model","date":"2020-12-28","arxiv_id":"2012.14388","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessment-of-the-relative-importance-of","title":"Assessment of the Relative Importance of different hyper-parameters of LSTM for an IDS","date":"2020-12-26","arxiv_id":"2012.14427","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-temperature-for-language-modeling","title":"Contextual Temperature for Language Modeling","date":"2020-12-25","arxiv_id":"2012.13575","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-dependency-parsing-as-domain","title":"Cross-lingual Universal Dependency Parsing Only from One Monolingual Treebank","date":"2020-12-24","arxiv_id":"2012.13163","repositories_listed":0,"syntology":null},{"url":null,"slug":"subicap-towards-subword-informed-image","title":"SubICap: Towards Subword-informed Image Captioning","date":"2020-12-24","arxiv_id":"2012.13122","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-language-model-using","title":"Code Switching Language Model Using Monolingual Training Data","date":"2020-12-23","arxiv_id":"2012.12543","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-a-language-model-without-human","title":"Pre-Training a Language Model Without Human Language","date":"2020-12-22","arxiv_id":"2012.11995","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-and-surprisal-jointly-deliver-the","title":"Uncertainty and Surprisal Jointly Deliver the Punchline: Exploiting Incongruity-Based Features for Humor Recognition","date":"2020-12-22","arxiv_id":"2012.12007","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-document-level-neural-discourse","title":"An End-to-End Document-Level Neural Discourse Parser Exploiting Multi-Granularity Representations","date":"2020-12-21","arxiv_id":"2012.11169","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexically-constrained-text-generation-through","title":"Lexically-constrained Text Generation through Commonsense Knowledge Extraction and Injection","date":"2020-12-19","arxiv_id":"2012.10813","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-the-best-rationalizing","title":"Learning from the Best: Rationalizing Prediction by Adversarial Information Calibration","date":"2020-12-16","arxiv_id":"2012.08884","repositories_listed":0,"syntology":null}],"record_sha256":"a93f08823fca6c07143814e81d9a3507d0099a7e0fcde54f28c2cd9f96615bf5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}