{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modelling/papers/145","list_of":"/task/language-modelling","task":"Language Modelling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":145,"pages_in_order":177,"rows_per_page":100,"rows":[14401,14500],"of":17610,"counts":{"archive_papers_tagged":17610,"with_a_code_link":7012,"where_syntology_ran_a_sample":2428,"not_listed_spam_title":0,"listed":17610,"listed_where_code_ran":2428,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2027,"every_run_a_failure_of_syntologys_instrument":401,"listed_with_a_run_with_no_instrument_failure":2027,"listed_every_run_a_failure_of_syntologys_instrument":401,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modelling","prev":"/task/language-modelling/papers/144","next":"/task/language-modelling/papers/146","papers":[{"url":null,"slug":"attention-augmented-convolutional-transformer","title":"Attention Augmented Convolutional Transformer for Tabular Time-series","date":"2021-10-05","arxiv_id":"2110.01825","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-contextual-adaptation-with-neural","title":"Fast Contextual Adaptation with Neural Associative Memory for On-Device Personalized Speech Recognition","date":"2021-10-05","arxiv_id":"2110.02220","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-using-lmus-10x-better-data","title":"Language Modeling using LMUs: 10x Better Data Efficiency or Improved Scaling Compared to Transformers","date":"2021-10-05","arxiv_id":"2110.02402","repositories_listed":0,"syntology":null},{"url":null,"slug":"teach-me-what-to-say-and-i-will-learn-what-to","title":"Teach Me What to Say and I Will Learn What to Pick: Unsupervised Knowledge Selection Through Response Generation with Pretrained Generative Models","date":"2021-10-05","arxiv_id":"2110.02067","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-chains-transparent-and-controllable-human","title":"AI Chains: Transparent and Controllable Human-AI Interaction by Chaining Large Language Model Prompts","date":"2021-10-04","arxiv_id":"2110.01691","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-information-bottleneck-for","title":"Leveraging Information Bottleneck for Scientific Document Summarization","date":"2021-10-04","arxiv_id":"2110.01280","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-anderson-mixing-for-nonconvex","title":"Stochastic Anderson Mixing for Nonconvex Stochastic Optimization","date":"2021-10-04","arxiv_id":"2110.01543","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-contextualized-language-modeling","title":"A Study on Contextualized Language Modeling for Machine Reading Comprehension","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-low-resource-code-switching-data","title":"Exploiting Low-Resource Code-Switching Data to Mandarin-English Speech Recognition Systems","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-networks-based-on","title":"Generative Adversarial Networks based on Mixed-Attentions for Citation Intent Classification in Scientific Publications","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","repositories_listed":0,"syntology":null},{"url":null,"slug":"span-labeling-approach-for-vietnamese-and","title":"Span Labeling Approach for Vietnamese and Chinese Word Segmentation","date":"2021-10-01","arxiv_id":"2110.00156","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English with Transfer Learning","date":"2021-10-01","arxiv_id":"2110.00678","repositories_listed":0,"syntology":null},{"url":null,"slug":"unpacking-the-interdependent-systems-of","title":"Unpacking the Interdependent Systems of Discrimination: Ableist Bias in NLP Systems through an Intersectional Lens","date":"2021-10-01","arxiv_id":"2110.00521","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-compression-via-concurrent","title":"Deep Neural Compression Via Concurrent Pruning and Self-Distillation","date":"2021-09-30","arxiv_id":"2109.15014","repositories_listed":0,"syntology":null},{"url":null,"slug":"focused-contrastive-training-for-test-based","title":"Focused Contrastive Training for Test-based Constituency Analysis","date":"2021-09-30","arxiv_id":"2109.15159","repositories_listed":0,"syntology":null},{"url":"/paper/a-dot-product-attention-free-transformer","slug":"a-dot-product-attention-free-transformer","title":"A Dot Product Attention Free Transformer","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-step-wise-weighting-approach-for","title":"A Step-Wise Weighting Approach for Controllable Text Generation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-the-effect-of-masking-length","title":"Analysing the Effect of Masking Length Distribution of MLM: An Evaluation Framework and Case Study on Chinese MRC Datasets","date":"2021-09-29","arxiv_id":"2110.15712","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-implicit-position-encoding","title":"Analyzing the Implicit Position Encoding Ability of Transformer Decoder","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"autocog-a-unified-data-modal-co-search","title":"AutoCoG: A Unified Data-Modal Co-Search Framework for Graph Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"banana-a-benchmark-for-the-assessment-of","title":"BANANA: a Benchmark for the Assessment of Neural Architectures for Nucleic Acids","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beliefbank-adding-memory-to-a-pre-trained","title":"BeliefBank: Adding Memory to a Pre-Trained Language Model for a Systematic Notion of Belief","date":"2021-09-29","arxiv_id":"2109.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-storytelling-with-human-actors","title":"Collaborative Storytelling with Human Actors and AI Narrators","date":"2021-09-29","arxiv_id":"2109.14728","repositories_listed":0,"syntology":null},{"url":null,"slug":"dictformer-tiny-transformer-with-shared","title":"DictFormer: Tiny Transformer with Shared Dictionary","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ensembles-and-cocktails-robust-finetuning-for","title":"Ensembles and Cocktails: Robust Finetuning for Natural Language Generation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-representation-for-multilingual","title":"Fairness in Representation for Multilingual NLP: Insights from Controlled Experiments on Conditional Language Modeling","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generate-annotate-and-learn-generative-models-1","title":"Generate, Annotate, and Learn: Generative Models Advance Self-Training and Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gental-generative-denoising-skip-gram","title":"GenTAL: Generative Denoising Skip-gram Transformer for Unsupervised Binary Code Similarity Detection","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-character-tagger-for-short-text","title":"Hierarchical Character Tagger for Short Text Spelling Error Correction","date":"2021-09-29","arxiv_id":"2109.14259","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-adapt-your-large-scale-vision-and","title":"How to Adapt Your Large-Scale Vision-and-Language Model","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"image-bert-pre-training-with-online-tokenizer","title":"Image BERT Pre-training with Online Tokenizer","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-non-autoregressive-translation","title":"Improving Non-Autoregressive Translation Models Without Distillation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-pre-training-improves","title":"Language Model Pre-training Improves Generalization in Policy Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"not-so-fine-tuning-measures-of-common-sense","title":"Not-so fine-tuning: Measures of Common Sense for Language Models","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-large","title":"Offline Reinforcement Learning for Large Scale Language Action Spaces","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reward-maximization-and-distribution","title":"On Reward Maximization and Distribution Matching for Fine-Tuning Language Models","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrained-language-model-in-continual","title":"Pretrained Language Model in Continual Learning: A Comparative Study","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-client-reweighting-for-selfish","title":"Rethinking Client Reweighting for Selfish Federated Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-transformer-a-unified-architecture-for","title":"Scene Transformer: A unified architecture for predicting future trajectories of multiple agents","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-token-generation-for-few-shot","title":"Selective Token Generation for Few-shot Language Modeling","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-distilled-pruning-of-neural-networks","title":"Self-Distilled Pruning Of Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sgornn-combining-scalar-gates-and-orthogonal","title":"SGORNN: Combining Scalar Gates and Orthogonal Constraints in Recurrent Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"short-term-memory-in-neural-language-models","title":"Short-term memory in neural language models","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-attention-with-learning-to-hash","title":"Sparse Attention with Learning to Hash","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-aware-neural-language-model-domain","title":"Topic Aware Neural Language Model: Domain Adaptation of Unconditional Text Generation Models","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transliteration-a-simple-technique-for","title":"Transliteration: A Simple Technique For Improving Multilingual Language Modeling","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transtcn-an-attention-based-tcn-framework-for","title":"TransTCN: An Attention-based TCN Framework for Sequential Modeling","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"private-language-model-adaptation-for-speech","title":"Private Language Model Adaptation for Speech Recognition","date":"2021-09-28","arxiv_id":"2110.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"fquad2-0-french-question-answering-and","title":"FQuAD2.0: French Question Answering and knowing that you know nothing","date":"2021-09-27","arxiv_id":"2109.13209","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-priming-for-cross-lingual","title":"Language Model Priming for Cross-Lingual Event Extraction","date":"2021-09-25","arxiv_id":"2109.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-selectively-learn-for-weakly","title":"Learning to Selectively Learn for Weakly-supervised Paraphrase Generation","date":"2021-09-25","arxiv_id":"2109.12457","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-diversity-enhanced-and-constraints-relaxed","title":"A Diversity-Enhanced and Constraints-Relaxed Augmentation for Low-Resource Classification","date":"2021-09-24","arxiv_id":"2109.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-proposal-of-automatic-error-correction-in","title":"A Proposal of Automatic Error Correction in Text","date":"2021-09-24","arxiv_id":"2112.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"identification-of-enzymatic-active-sites-with","title":"Identification of Enzymatic Active Sites with Unsupervised Language Modeling","date":"2021-09-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mlim-vision-and-language-model-pre-training","title":"MLIM: Vision-and-Language Model Pre-training with Masked Language and Image Modeling","date":"2021-09-24","arxiv_id":"2109.12178","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-attention-sparsity-in-transformers","title":"Predicting Attention Sparsity in Transformers","date":"2021-09-24","arxiv_id":"2109.12188","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-language-model-meta-pretraining","title":"Cross-Lingual Language Model Meta-Pretraining","date":"2021-09-23","arxiv_id":"2109.11129","repositories_listed":0,"syntology":null},{"url":null,"slug":"lstm-hyper-parameter-selection-for-malware","title":"LSTM Hyper-Parameter Selection for Malware Detection: Interaction Effects and Hierarchical Selection Approach","date":"2021-09-23","arxiv_id":"2109.11500","repositories_listed":0,"syntology":null},{"url":null,"slug":"bfclass-a-backdoor-free-text-classification","title":"BFClass: A Backdoor-free Text Classification Framework","date":"2021-09-22","arxiv_id":"2109.10855","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialoguebert-a-self-supervised-learning-based","title":"DialogueBERT: A Self-Supervised Learning based Dialogue Pre-training Encoder","date":"2021-09-22","arxiv_id":"2109.10480","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"bertweetfr-domain-adaptation-of-pre-trained","title":"BERTweetFR : Domain Adaptation of Pre-Trained Language Models for French Tweets","date":"2021-09-21","arxiv_id":"2109.10234","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-specific-language-models-for","title":"Learning Domain Specific Language Models for Automatic Speech Recognition through Machine Translation","date":"2021-09-21","arxiv_id":"2110.10261","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-trade-offs-of-domain-adaptation-for","title":"The Trade-offs of Domain Adaptation for Neural Language Models","date":"2021-09-21","arxiv_id":"2109.10274","repositories_listed":0,"syntology":null},{"url":null,"slug":"influence-of-asr-and-language-model-on","title":"Influence of ASR and Language Model on Alzheimer's Disease Detection","date":"2021-09-20","arxiv_id":"2110.15704","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-natural-language-generation-from","title":"Learning Natural Language Generation from Scratch","date":"2021-09-20","arxiv_id":"2109.09371","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-training-with-contrastive","title":"Adversarial Training with Contrastive Learning in NLP","date":"2021-09-19","arxiv_id":"2109.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-deacon-multimodal-molecular-domain","title":"Multilingual Molecular Representation Learning via Contrastive Pre-training","date":"2021-09-18","arxiv_id":"2109.08830","repositories_listed":0,"syntology":null},{"url":null,"slug":"bart-light-one-decoder-layer-is-enough","title":"BART-light: One Decoder Layer Is Enough","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-knowledge-augmented-pretrained","title":"Commonsense Knowledge-Augmented Pretrained Language Models for Causal Reasoning Classification","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-multitask-learning-for-low-resource","title":"Exploring Multitask Learning for Low-Resource AbstractiveSummarization","date":"2021-09-17","arxiv_id":"2109.08565","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-as-a-knowledge-source-for","title":"Language Models as a Knowledge Source for Cognitive Agents","date":"2021-09-17","arxiv_id":"2109.08270","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-range-modeling-of-source-code-files-with","title":"Long-Range Modeling of Source Code Files with eWASH: Extended Window Access by Syntax Hierarchy","date":"2021-09-17","arxiv_id":"2109.08780","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-reading-comprehension-generative-or","title":"Machine Reading Comprehension: Generative or Extractive Reader?","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"relating-neural-text-degeneration-to-exposure","title":"Relating Neural Text Degeneration to Exposure Bias","date":"2021-09-17","arxiv_id":"2109.08705","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiprompt-sentiment-knowledge-enhanced","title":"SentiPrompt: Sentiment Knowledge Enhanced Prompt-Tuning for Aspect-Based Sentiment Analysis","date":"2021-09-17","arxiv_id":"2109.08306","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bag-of-tricks-for-dialogue-summarization","title":"A Bag of Tricks for Dialogue Summarization","date":"2021-09-16","arxiv_id":"2109.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-algorithmic-question-answering-towards-a","title":"Deep Algorithmic Question Answering: Towards a Compositionally Hybrid AI for Algorithmic Reasoning","date":"2021-09-16","arxiv_id":"2109.08006","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-language-models-know-the-way-to-rome","title":"Do Language Models Know the Way to Rome?","date":"2021-09-16","arxiv_id":"2109.07971","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-the-cat-out-of-the-bag-contrastive","title":"Let the CAT out of the bag: Contrastive Attributed explanations for Text","date":"2021-09-16","arxiv_id":"2109.07983","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-training-of-nearest-neighbor","title":"Regularized Training of Nearest Neighbor Language Models","date":"2021-09-16","arxiv_id":"2109.08249","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-language-model-understood-the-prompt-was","title":"The Language Model Understood the Prompt was Ambiguous: Probing Syntactic Uncertainty Through Generation","date":"2021-09-16","arxiv_id":"2109.07848","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-glass-box-features-uncertainty","title":"Beyond Glass-Box Features: Uncertainty Quantification Enhanced Quality Estimation for Neural Machine Translation","date":"2021-09-15","arxiv_id":"2109.07141","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-domain-adaptation-of-language","title":"Efficient Domain Adaptation of Language Models via Adaptive Tokenization","date":"2021-09-15","arxiv_id":"2109.07460","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-text-auto-completion-with-next","title":"Improving Text Auto-Completion with Next Phrase Prediction","date":"2021-09-15","arxiv_id":"2109.07067","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-complementarity-of-data-selection-and","title":"On the Complementarity of Data Selection and Fine Tuning for Domain Adaptation","date":"2021-09-15","arxiv_id":"2109.07591","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranknas-efficient-neural-architecture-search","title":"RankNAS: Efficient Neural Architecture Search by Pairwise Ranking","date":"2021-09-15","arxiv_id":"2109.07383","repositories_listed":0,"syntology":null},{"url":null,"slug":"tied-reduced-rnn-t-decoder","title":"Tied & Reduced RNN-T Decoder","date":"2021-09-15","arxiv_id":"2109.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-crawler-architecture-for-harvesting-the","title":"A Crawler Architecture for Harvesting the Clear, Social, and Dark Web for IoT-Related Cyber-Threat Intelligence","date":"2021-09-14","arxiv_id":"2109.06932","repositories_listed":0,"syntology":null},{"url":null,"slug":"different-strokes-for-different-folks","title":"Different Strokes for Different Folks: Investigating Appropriate Further Pre-training Approaches for Diverse Dialogue Tasks","date":"2021-09-14","arxiv_id":"2109.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-read-reconstruction-for-dna-data","title":"Single-Read Reconstruction for DNA Data Storage Using Transformers","date":"2021-09-12","arxiv_id":"2109.05478","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-state-capsule-networks-for-text","title":"Dual-State Capsule Networks for Text Classification","date":"2021-09-10","arxiv_id":"2109.04762","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficientclip-efficient-cross-modal-pre","title":"EfficientCLIP: Efficient Cross-Modal Pre-training by Ensemble Confident Learning and Language Modeling","date":"2021-09-10","arxiv_id":"2109.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-self-disclosure-in-neural-dialog","title":"Enhancing Self-Disclosure In Neural Dialog Models By Candidate Re-ranking","date":"2021-09-10","arxiv_id":"2109.05090","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaxt-meta-cross-task-transfer-between","title":"MetaXT: Meta Cross-Task Transfer between Disparate Label Spaces","date":"2021-09-09","arxiv_id":"2109.04240","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-end-to-end-speech","title":"Non-autoregressive End-to-end Speech Translation with Parallel Autoregressive Rescoring","date":"2021-09-09","arxiv_id":"2109.04411","repositories_listed":0,"syntology":null},{"url":"/paper/refinecap-concept-aware-refinement-for-image","slug":"refinecap-concept-aware-refinement-for-image","title":"RefineCap: Concept-Aware Refinement for Image Captioning","date":"2021-09-08","arxiv_id":"2109.03529","repositories_listed":0,"syntology":null}],"record_sha256":"ede14c4c05d47e40438543382092c124b5acee8654153b01809eac4bf6a40991","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}