{"url":"/task/language-modeling","name":"Language Modeling","slug":"language-modeling","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":14182,"papers_with_code":5620,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":1,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[{"url":"/task/dream-generation","name":"Dream Generation"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":5620,"tagged_in_all":14182,"items":[{"url":"/paper/semi-supervised-sequence-learning","title":"Semi-supervised Sequence Learning","date":"2015-11-04","arxiv_id":"1511.01432","repositories_listed":161,"syntology":null},{"url":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","repositories_listed":67,"syntology":{"n":65,"n_ran":15,"n_unverified":50,"n_pointer_only":4}},{"url":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","repositories_listed":67,"syntology":{"n":48,"n_ran":22,"n_unverified":26,"n_pointer_only":23}},{"url":"/paper/universal-language-model-fine-tuning-for-text","title":"Universal Language Model Fine-tuning for Text Classification","date":"2018-01-18","arxiv_id":"1801.06146","repositories_listed":66,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":3}},{"url":"/paper/darts-differentiable-architecture-search","title":"DARTS: Differentiable Architecture Search","date":"2018-06-24","arxiv_id":"1806.09055","repositories_listed":59,"syntology":{"n":156,"n_ran":66,"n_unverified":90,"n_pointer_only":48}},{"url":"/paper/deep-contextualized-word-representations","title":"Deep contextualized word representations","date":"2018-02-15","arxiv_id":"1802.05365","repositories_listed":46,"syntology":{"n":58,"n_ran":23,"n_unverified":35,"n_pointer_only":25}},{"url":"/paper/regularizing-and-optimizing-lstm-language","title":"Regularizing and Optimizing LSTM Language Models","date":"2017-08-07","arxiv_id":"1708.02182","repositories_listed":45,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":7}},{"url":"/paper/end-to-end-memory-networks","title":"End-To-End Memory Networks","date":"2015-03-31","arxiv_id":"1503.08895","repositories_listed":44,"syntology":{"n":15,"n_ran":2,"n_unverified":13,"n_pointer_only":5}},{"url":"/paper/listen-attend-and-spell","title":"Listen, Attend and Spell","date":"2015-08-05","arxiv_id":"1508.01211","repositories_listed":40,"syntology":{"n":51,"n_ran":12,"n_unverified":39,"n_pointer_only":9}},{"url":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","arxiv_id":"1910.01108","repositories_listed":37,"syntology":{"n":27,"n_ran":19,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","arxiv_id":"1901.02860","repositories_listed":37,"syntology":{"n":143,"n_ran":63,"n_unverified":80,"n_pointer_only":43}},{"url":"/paper/mamba-linear-time-sequence-modeling-with","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","date":"2023-12-01","arxiv_id":"2312.00752","repositories_listed":35,"syntology":{"n":62,"n_ran":18,"n_unverified":44,"n_pointer_only":28}},{"url":"/paper/unsupervised-cross-lingual-representation-1","title":"Unsupervised Cross-lingual Representation Learning at Scale","date":"2019-11-05","arxiv_id":"1911.02116","repositories_listed":35,"syntology":{"n":59,"n_ran":25,"n_unverified":34,"n_pointer_only":52}},{"url":"/paper/specaugment-a-simple-data-augmentation-method","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","date":"2019-04-18","arxiv_id":"1904.08779","repositories_listed":30,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/direct-preference-optimization-your-language","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","date":"2023-05-29","arxiv_id":"2305.18290","repositories_listed":29,"syntology":{"n":31,"n_ran":6,"n_unverified":25,"n_pointer_only":2}},{"url":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","arxiv_id":"1906.08237","repositories_listed":27,"syntology":{"n":24,"n_ran":10,"n_unverified":14,"n_pointer_only":3}},{"url":"/paper/matching-networks-for-one-shot-learning","title":"Matching Networks for One Shot Learning","date":"2016-06-13","arxiv_id":"1606.04080","repositories_listed":26,"syntology":{"n":16,"n_ran":6,"n_unverified":10,"n_pointer_only":6}},{"url":"/paper/conformer-convolution-augmented-transformer","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","date":"2020-05-16","arxiv_id":"2005.08100","repositories_listed":25,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":2}},{"url":"/paper/the-pile-an-800gb-dataset-of-diverse-text-for","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","date":"2020-12-31","arxiv_id":"2101.00027","repositories_listed":22,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/longformer-the-long-document-transformer","title":"Longformer: The Long-Document Transformer","date":"2020-04-10","arxiv_id":"2004.05150","repositories_listed":22,"syntology":{"n":35,"n_ran":15,"n_unverified":20,"n_pointer_only":5}},{"url":"/paper/on-the-variance-of-the-adaptive-learning-rate","title":"On the Variance of the Adaptive Learning Rate and Beyond","date":"2019-08-08","arxiv_id":"1908.03265","repositories_listed":21,"syntology":{"n":13,"n_ran":4,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","arxiv_id":null,"repositories_listed":21,"syntology":null},{"url":"/paper/recurrent-neural-network-regularization","title":"Recurrent Neural Network Regularization","date":"2014-09-08","arxiv_id":"1409.2329","repositories_listed":21,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":6}},{"url":"/paper/chain-of-thought-prompting-elicits-reasoning","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","date":"2022-01-28","arxiv_id":"2201.11903","repositories_listed":19,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/electra-pre-training-text-encoders-as-1","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","date":"2020-03-23","arxiv_id":"2003.10555","repositories_listed":19,"syntology":{"n":40,"n_ran":26,"n_unverified":14,"n_pointer_only":10}},{"url":"/paper/variational-autoencoders-for-collaborative","title":"Variational Autoencoders for Collaborative Filtering","date":"2018-02-16","arxiv_id":"1802.05814","repositories_listed":18,"syntology":{"n":21,"n_ran":2,"n_unverified":19,"n_pointer_only":1}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/the-curious-case-of-neural-text-degeneration","title":"The Curious Case of Neural Text Degeneration","date":"2019-04-22","arxiv_id":"1904.09751","repositories_listed":17,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":11}},{"url":"/paper/cross-lingual-language-model-pretraining","title":"Cross-lingual Language Model Pretraining","date":"2019-01-22","arxiv_id":"1901.07291","repositories_listed":17,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":1}}],"syntology_records":28,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}