{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/cosine-annealing/papers/35","list_of":"/method/cosine-annealing","method":"Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":35,"pages_in_order":40,"rows_per_page":100,"rows":[3401,3500],"of":3965,"counts":{"archive_papers_tagged":3965,"with_a_code_link":1734,"where_syntology_ran_a_sample":627,"not_listed_spam_title":0,"listed":3965,"listed_where_code_ran":627,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":513,"every_run_a_failure_of_syntologys_instrument":114,"listed_with_a_run_with_no_instrument_failure":513,"listed_every_run_a_failure_of_syntologys_instrument":114,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/cosine-annealing","prev":"/method/cosine-annealing/papers/34","next":"/method/cosine-annealing/papers/36","papers":[{"paper":null,"slug":"elle-efficient-lifelong-pre-training-for","title":"ELLE: Efficient Lifelong Pre-training for Emerging Data","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-task-oriented-dialog-policy","title":"End-to-end Task-oriented Dialog Policy Learning based on Pre-trained Language Model","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-sparse-learning-hierarchical-efficient","title":"ERNIE-SPARSE: Learning Hierarchical Efficient Transformer Through Regularized Self-Attention","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-and-adapting-chinese-gpt-to-pinyin","title":"Exploring and Adapting Chinese GPT to Pinyin Input Method","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-pre-trained-transformer-for-design","title":"Generative Pre-Trained Transformer for Design Concept Generation: An Exploration","date":"2021-11-16","arxiv_id":"2111.08489","n_code_links":0,"syntology":null},{"paper":null,"slug":"glm-general-language-model-pretraining-with","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-tokenization-on-language-models-an","title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-gpt-3-after-deployment-with-a","title":"Improving GPT-3 after deployment with a dynamic memory of feedback","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-is-in-rescue-task-oriented","title":"Knowledge Graph is in Rescue: Task Oriented Dialogue System for Response Generation without NLU and DM","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"life-after-bert-what-do-other-muppets","title":"Life after BERT: What do Other Muppets Understand about Language?","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"moving-the-eiffel-tower-to-rome-tracing-and","title":"Moving the Eiffel Tower to ROME: Tracing and Editing Facts in GPT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-multilingual-capabilities-of-very-1","title":"On the Multilingual Capabilities of Very Large-Scale English Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-of-ambiguity-in-pre-trained","title":"Representation of Ambiguity in Pre-Trained Sentence Embeddings","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-bottleneck-makes-language-models","title":"Softmax Bottleneck Makes Language Models Unable to Represent Multi-mode Word Distributions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tell-me-who-you-are-and-i-ll-tell-you-what-to","title":"Tell me who you are and i'll tell you what to do: A Persona Grounded Task Oriented Dialogue Generation System","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource-1","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-coding-social-science-datasets-with","title":"Towards Coding Social Science Datasets with Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-classifying-grammatical-role-bert-doesn","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-story-generation-with-multi-task","title":"Exploring Story Generation with Multi-task Objectives in Variational Autoencoders","date":"2021-11-15","arxiv_id":"2111.08133","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-law-for-recommendation-models-towards","title":"Scaling Law for Recommendation Models: Towards General-purpose User Representations","date":"2021-11-15","arxiv_id":"2111.11294","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-corpus-of-discourse-structure-in","slug":"a-novel-corpus-of-discourse-structure-in","title":"A Novel Corpus of Discourse Structure in Humans and Computers","date":"2021-11-10","arxiv_id":"2111.05940","n_code_links":1,"syntology":null},{"paper":null,"slug":"amazon-sagemaker-model-parallelism-a-general","title":"Amazon SageMaker Model Parallelism: A General and Flexible Framework for Large Model Training","date":"2021-11-10","arxiv_id":"2111.05972","n_code_links":0,"syntology":null},{"paper":null,"slug":"distir-an-intermediate-representation-and","title":"DistIR: An Intermediate Representation and Simulator for Efficient Neural Network Distribution","date":"2021-11-09","arxiv_id":"2111.05426","n_code_links":0,"syntology":null},{"paper":null,"slug":"fpm-a-collection-of-large-scale-foundation","title":"FPM: A Collection of Large-scale Foundation Pre-trained Language Models","date":"2021-11-09","arxiv_id":"2111.04909","n_code_links":0,"syntology":null},{"paper":"/paper/synthesizing-collective-communication","slug":"synthesizing-collective-communication","title":"TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches","date":"2021-11-08","arxiv_id":"2111.04867","n_code_links":2,"syntology":null},{"paper":"/paper/an-explanation-of-in-context-learning-as-1","slug":"an-explanation-of-in-context-learning-as-1","title":"An Explanation of In-context Learning as Implicit Bayesian Inference","date":"2021-11-03","arxiv_id":"2111.02080","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["p-lambda/incontext-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"theeyecorpus-experiments-in-reducing-nlp-bias","title":"TheEyeCorpus: Experiments in Reducing NLP Bias and Identifiability for Large LMs","date":"2021-11-03","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","slug":"dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","arxiv_id":"2111.00160","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vita-group/dsee"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/amendable-generation-for-dialogue-state","slug":"amendable-generation-for-dialogue-state","title":"Amendable Generation for Dialogue State Tracking","date":"2021-10-29","arxiv_id":"2110.15659","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-sequence-to-sequence-model-for-extracting","title":"A Sequence to Sequence Model for Extracting Multiple Product Name Entities from Dialog","date":"2021-10-28","arxiv_id":"2110.14843","n_code_links":0,"syntology":null},{"paper":"/paper/colossal-ai-a-unified-deep-learning-system","slug":"colossal-ai-a-unified-deep-learning-system","title":"Colossal-AI: A Unified Deep Learning System For Large-Scale Parallel Training","date":"2021-10-28","arxiv_id":"2110.14883","n_code_links":1,"syntology":null},{"paper":null,"slug":"new-sar-target-recognition-based-on-yolo-and","title":"New SAR target recognition based on YOLO and very deep multi-canonical correlation analysis","date":"2021-10-28","arxiv_id":"2110.15383","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-transformers-are-more-efficient","slug":"hierarchical-transformers-are-more-efficient","title":"Hierarchical Transformers Are More Efficient Language Models","date":"2021-10-26","arxiv_id":"2110.13711","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google/trax"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"generating-artificial-texts-as-substitution","title":"Generating artificial texts as substitution or complement of training data","date":"2021-10-25","arxiv_id":"2110.13016","n_code_links":0,"syntology":null},{"paper":"/paper/fast-model-editing-at-scale-1","slug":"fast-model-editing-at-scale-1","title":"Fast Model Editing at Scale","date":"2021-10-21","arxiv_id":"2110.11309","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["eric-mitchell/mend"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"risks-of-ai-foundation-models-in-education","title":"Risks of AI Foundation Models in Education","date":"2021-10-19","arxiv_id":"2110.10024","n_code_links":0,"syntology":null},{"paper":"/paper/illiterate-dall-cdot-e-learns-to-compose-1","slug":"illiterate-dall-cdot-e-learns-to-compose-1","title":"Illiterate DALL-E Learns to Compose","date":"2021-10-17","arxiv_id":"2110.11405","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["singhgautam/slate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reminding-the-incremental-language-model-via","title":"Reminding the Incremental Language Model via Data-Free Self-Distillation","date":"2021-10-17","arxiv_id":"2110.08745","n_code_links":0,"syntology":null},{"paper":"/paper/taming-visually-guided-sound-generation","slug":"taming-visually-guided-sound-generation","title":"Taming Visually Guided Sound Generation","date":"2021-10-17","arxiv_id":"2110.08791","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":6,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["v-iashin/SpecVQGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"a-short-study-on-compressing-decoder-based","title":"A Short Study on Compressing Decoder-Based Language Models","date":"2021-10-16","arxiv_id":"2110.08460","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-transfer-learning-for-polish","title":"Evaluation of Transfer Learning for Polish with a text-to-text model","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hydra-a-system-for-large-multi-model-deep","slug":"hydra-a-system-for-large-multi-model-deep","title":"Hydra: A System for Large Multi-Model Deep Learning","date":"2021-10-16","arxiv_id":"2110.08633","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-inheritance-for-pre-trained-1","title":"Knowledge Inheritance for Pre-trained Language Models","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pagnol-an-extra-large-french-generative-model","title":"PAGnol: An Extra-Large French Generative Model","date":"2021-10-16","arxiv_id":"2110.08554","n_code_links":0,"syntology":null},{"paper":null,"slug":"sharpness-aware-minimization-improves","title":"Sharpness-Aware Minimization Improves Language Model Generalization","date":"2021-10-16","arxiv_id":"2110.08529","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-10-16","arxiv_id":"2110.08525","n_code_links":0,"syntology":null},{"paper":"/paper/vehicle-speed-estimation-using-computer","slug":"vehicle-speed-estimation-using-computer","title":"Vehicle Speed Estimation Using Computer Vision And Evolutionary Camera Calibration","date":"2021-10-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"wechsel-effective-initialization-of-subword","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kronecker-decomposition-for-gpt-compression","title":"Kronecker Decomposition for GPT Compression","date":"2021-10-15","arxiv_id":"2110.08152","n_code_links":0,"syntology":null},{"paper":"/paper/building-chinese-biomedical-language-models","slug":"building-chinese-biomedical-language-models","title":"Building Chinese Biomedical Language Models via Multi-Level Text Discrimination","date":"2021-10-14","arxiv_id":"2110.07244","n_code_links":1,"syntology":null},{"paper":"/paper/delphi-towards-machine-ethics-and-norms","slug":"delphi-towards-machine-ethics-and-norms","title":"Can Machines Learn Morality? The Delphi Experiment","date":"2021-10-14","arxiv_id":"2110.07574","n_code_links":1,"syntology":null},{"paper":"/paper/symbolic-knowledge-distillation-from-general","slug":"symbolic-knowledge-distillation-from-general","title":"Symbolic Knowledge Distillation: from General Language Models to Commonsense Models","date":"2021-10-14","arxiv_id":"2110.07178","n_code_links":1,"syntology":null},{"paper":null,"slug":"covert-message-passing-over-public-internet","title":"Leveraging Generative Models for Covert Messaging: Challenges and Tradeoffs for \"Dead-Drop\" Deployments","date":"2021-10-13","arxiv_id":"2110.07009","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-modelling-via-learning-to-rank","title":"Language Modelling via Learning to Rank","date":"2021-10-13","arxiv_id":"2110.06961","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-laws-for-the-few-shot-adaptation-of-1","title":"Scaling Laws for the Few-Shot Adaptation of Pre-trained Image Classifiers","date":"2021-10-13","arxiv_id":"2110.06990","n_code_links":0,"syntology":null},{"paper":"/paper/yformer-u-net-inspired-transformer-1","slug":"yformer-u-net-inspired-transformer-1","title":"Yformer: U-Net Inspired Transformer Architecture for Far Horizon Time Series Forecasting","date":"2021-10-13","arxiv_id":"2110.08255","n_code_links":1,"syntology":null},{"paper":"/paper/lightseq-accelerated-training-for-transformer","slug":"lightseq-accelerated-training-for-transformer","title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","date":"2021-10-12","arxiv_id":"2110.05722","n_code_links":1,"syntology":null},{"paper":"/paper/list-lite-self-training-makes-efficient-few-1","slug":"list-lite-self-training-makes-efficient-few-1","title":"LiST: Lite Prompted Self-training Makes Parameter-Efficient Few-shot Learners","date":"2021-10-12","arxiv_id":"2110.06274","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":7,"n_instrument":2,"unverified":3,"pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/list"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"emds-7-environmental-microorganism-image","title":"EMDS-7: Environmental Microorganism Image Dataset Seventh Version for Multiple Object Detection Evaluation","date":"2021-10-11","arxiv_id":"2110.07723","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-learning-for-situated-multi-domain","title":"Multi-Task Learning for Situated Multi-Domain End-to-End Dialogue Systems","date":"2021-10-11","arxiv_id":"2110.05221","n_code_links":0,"syntology":null},{"paper":null,"slug":"dct-dynamic-compressive-transformer-for","title":"DCT: Dynamic Compressive Transformer for Modeling Unbounded Sequence","date":"2021-10-10","arxiv_id":"2110.04821","n_code_links":0,"syntology":null},{"paper":"/paper/yuan-1-0-large-scale-pre-trained-language","slug":"yuan-1-0-large-scale-pre-trained-language","title":"Yuan 1.0: Large-Scale Pre-trained Language Model in Zero-Shot and Few-Shot Learning","date":"2021-10-10","arxiv_id":"2110.04725","n_code_links":1,"syntology":null},{"paper":"/paper/vector-quantized-image-modeling-with-improved-1","slug":"vector-quantized-image-modeling-with-improved-1","title":"Vector-quantized Image Modeling with Improved VQGAN","date":"2021-10-09","arxiv_id":"2110.04627","n_code_links":5,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"m6-10t-a-sharing-delinking-paradigm-for","title":"M6-10T: A Sharing-Delinking Paradigm for Efficient Multi-Trillion Parameter Pretraining","date":"2021-10-08","arxiv_id":"2110.03888","n_code_links":0,"syntology":null},{"paper":"/paper/layer-wise-pruning-of-transformer-attention","slug":"layer-wise-pruning-of-transformer-attention","title":"Layer-wise Pruning of Transformer Attention Heads for Efficient Language Modeling","date":"2021-10-07","arxiv_id":"2110.03252","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-the-inductive-bias-of-large","title":"Leveraging the Inductive Bias of Large Language Models for Abstract Textual Reasoning","date":"2021-10-05","arxiv_id":"2110.02370","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-examples-generation-for-reducing","title":"Adversarial Examples Generation for Reducing Implicit Gender Bias in Pre-trained Models","date":"2021-10-03","arxiv_id":"2110.01094","n_code_links":0,"syntology":null},{"paper":"/paper/survtrace-transformers-for-survival-analysis","slug":"survtrace-transformers-for-survival-analysis","title":"SurvTRACE: Transformers for Survival Analysis with Competing Events","date":"2021-10-02","arxiv_id":"2110.00855","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["RyanWangZf/SurvTRACE"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"collaborative-storytelling-with-human-actors","title":"Collaborative Storytelling with Human Actors and AI Narrators","date":"2021-09-29","arxiv_id":"2109.14728","n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-sparse-robust-efficient-transformer","title":"ERNIE-SPARSE: Robust Efficient Transformer Through Hierarchically Unifying Isolated Information","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"illiterate-dall-cdot-e-learns-to-compose","title":"Illiterate DALL$\\cdot$E Learns to Compose","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"language-model-pre-training-improves","title":"Language Model Pre-training Improves Generalization in Policy Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mapping-language-models-to-grounded","title":"Mapping Language Models to Grounded Conceptual Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-for-large","title":"Offline Reinforcement Learning for Large Scale Language Action Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"seqpate-differentially-private-text","title":"SeqPATE: Differentially Private Text Generation via Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/raft-a-real-world-few-shot-text","slug":"raft-a-real-world-few-shot-text","title":"RAFT: A Real-World Few-Shot Text Classification Benchmark","date":"2021-09-28","arxiv_id":"2109.14076","n_code_links":1,"syntology":null},{"paper":"/paper/turingbench-a-benchmark-environment-for","slug":"turingbench-a-benchmark-environment-for","title":"TURINGBENCH: A Benchmark Environment for Turing Test in the Age of Neural Text Generation","date":"2021-09-27","arxiv_id":"2109.13296","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/a-real-time-and-high-precision-method-for","slug":"a-real-time-and-high-precision-method-for","title":"A real-time and high-precision method for small traffic-signs recognition","date":"2021-09-25","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"training-dataset-generation-for-bridge-game","title":"Training dataset generation for bridge game registration","date":"2021-09-24","arxiv_id":"2109.11861","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-as-recommender-systems","title":"Language Models as Recommender Systems: Evaluations and Limitations","date":"2021-09-22","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"recursively-summarizing-books-with-human","title":"Recursively Summarizing Books with Human Feedback","date":"2021-09-22","arxiv_id":"2109.10862","n_code_links":0,"syntology":null},{"paper":"/paper/a-plug-and-play-method-for-controlled-text","slug":"a-plug-and-play-method-for-controlled-text","title":"A Plug-and-Play Method for Controlled Text Generation","date":"2021-09-20","arxiv_id":"2109.09707","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-bias-in-nlp-application-to-hate-speech","title":"Model Bias in NLP -- Application to Hate Speech Classification using transfer learning techniques","date":"2021-09-20","arxiv_id":"2109.09725","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-zero-label-language-learning","title":"Towards Zero-Label Language Learning","date":"2021-09-19","arxiv_id":"2109.09193","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-low-frequency-patterns-with-a-pre","title":"Learning Low-frequency Patterns with A Pre-trained Document-Grounded Conversation Model","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/primer-searching-for-efficient-transformers","slug":"primer-searching-for-efficient-transformers","title":"Primer: Searching for Efficient Transformers for Language Modeling","date":"2021-09-17","arxiv_id":"2109.08668","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"relating-neural-text-degeneration-to-exposure","title":"Relating Neural Text Degeneration to Exposure Bias","date":"2021-09-17","arxiv_id":"2109.08705","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-are-few-shot-multilingual","slug":"language-models-are-few-shot-multilingual","title":"Language Models are Few-shot Multilingual Learners","date":"2021-09-16","arxiv_id":"2109.07684","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gentaiscool/few-shot-lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-text-auto-completion-with-next","title":"Improving Text Auto-Completion with Next Phrase Prediction","date":"2021-09-15","arxiv_id":"2109.07067","n_code_links":0,"syntology":null},{"paper":"/paper/a-temporal-variational-model-for-story","slug":"a-temporal-variational-model-for-story","title":"A Temporal Variational Model for Story Generation","date":"2021-09-14","arxiv_id":"2109.06807","n_code_links":3,"syntology":null},{"paper":"/paper/multilingual-translation-via-grafting-pre","slug":"multilingual-translation-via-grafting-pre","title":"Multilingual Translation via Grafting Pre-trained Language Models","date":"2021-09-11","arxiv_id":"2109.05256","n_code_links":1,"syntology":null},{"paper":null,"slug":"topicrefine-joint-topic-prediction-and","title":"TopicRefine: Joint Topic Prediction and Dialogue Response Generation for Multi-turn End-to-End Dialogue System","date":"2021-09-11","arxiv_id":"2109.05187","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-self-disclosure-in-neural-dialog","title":"Enhancing Self-Disclosure In Neural Dialog Models By Candidate Re-ranking","date":"2021-09-10","arxiv_id":"2109.05090","n_code_links":0,"syntology":null},{"paper":"/paper/what-changes-can-large-scale-language-models","slug":"what-changes-can-large-scale-language-models","title":"What Changes Can Large-scale Language Models Bring? Intensive Study on HyperCLOVA: Billions-scale Korean Generative Pretrained Transformers","date":"2021-09-10","arxiv_id":"2109.04650","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/all-bark-and-no-bite-rogue-dimensions-in","slug":"all-bark-and-no-bite-rogue-dimensions-in","title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","date":"2021-09-09","arxiv_id":"2109.04404","n_code_links":1,"syntology":null},{"paper":null,"slug":"medically-aware-gpt-3-as-a-data-generator-for","title":"Medically Aware GPT-3 as a Data Generator for Medical Dialogue Summarization","date":"2021-09-09","arxiv_id":"2110.07356","n_code_links":0,"syntology":null},{"paper":"/paper/variational-latent-state-gpt-for-semi","slug":"variational-latent-state-gpt-for-semi","title":"Variational Latent-State GPT for Semi-Supervised Task-Oriented Dialog Systems","date":"2021-09-09","arxiv_id":"2109.04314","n_code_links":2,"syntology":null}],"record_sha256":"0bd8da034b6244a117e3bdfa2ccbecf60f8f924d5535552603357a9f0606e698","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}