{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/35","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":35,"pages_in_order":38,"rows_per_page":100,"rows":[3401,3500],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/34","next":"/method/linear-warmup-with-cosine-annealing/papers/36","papers":[{"paper":null,"slug":"enhancing-self-disclosure-in-neural-dialog","title":"Enhancing Self-Disclosure In Neural Dialog Models By Candidate Re-ranking","date":"2021-09-10","arxiv_id":"2109.05090","n_code_links":0,"syntology":null},{"paper":"/paper/what-changes-can-large-scale-language-models","slug":"what-changes-can-large-scale-language-models","title":"What Changes Can Large-scale Language Models Bring? Intensive Study on HyperCLOVA: Billions-scale Korean Generative Pretrained Transformers","date":"2021-09-10","arxiv_id":"2109.04650","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/all-bark-and-no-bite-rogue-dimensions-in","slug":"all-bark-and-no-bite-rogue-dimensions-in","title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","date":"2021-09-09","arxiv_id":"2109.04404","n_code_links":1,"syntology":null},{"paper":null,"slug":"medically-aware-gpt-3-as-a-data-generator-for","title":"Medically Aware GPT-3 as a Data Generator for Medical Dialogue Summarization","date":"2021-09-09","arxiv_id":"2110.07356","n_code_links":0,"syntology":null},{"paper":"/paper/variational-latent-state-gpt-for-semi","slug":"variational-latent-state-gpt-for-semi","title":"Variational Latent-State GPT for Semi-Supervised Task-Oriented Dialog Systems","date":"2021-09-09","arxiv_id":"2109.04314","n_code_links":2,"syntology":null},{"paper":"/paper/truthfulqa-measuring-how-models-mimic-human","slug":"truthfulqa-measuring-how-models-mimic-human","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","date":"2021-09-08","arxiv_id":"2109.07958","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sylinrl/truthfulqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"empathetic-dialogue-generation-with-pre","title":"Empathetic Dialogue Generation with Pre-trained RoBERTa-GPT2 and External Knowledge","date":"2021-09-07","arxiv_id":"2109.03004","n_code_links":0,"syntology":null},{"paper":null,"slug":"numgpt-improving-numeracy-ability-of","title":"NumGPT: Improving Numeracy Ability of Generative Pre-trained Models","date":"2021-09-07","arxiv_id":"2109.03137","n_code_links":0,"syntology":null},{"paper":"/paper/text-free-prosody-aware-generative-spoken","slug":"text-free-prosody-aware-generative-spoken","title":"Text-Free Prosody-Aware Generative Spoken Language Modeling","date":"2021-09-07","arxiv_id":"2109.03264","n_code_links":1,"syntology":null},{"paper":"/paper/general-purpose-question-answering-with-macaw","slug":"general-purpose-question-answering-with-macaw","title":"General-Purpose Question-Answering with Macaw","date":"2021-09-06","arxiv_id":"2109.02593","n_code_links":2,"syntology":null},{"paper":"/paper/gpt-3-models-are-poor-few-shot-learners-in","slug":"gpt-3-models-are-poor-few-shot-learners-in","title":"GPT-3 Models are Poor Few-Shot Learners in the Biomedical Domain","date":"2021-09-06","arxiv_id":"2109.02555","n_code_links":1,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","slug":"finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","arxiv_id":"2109.01652","n_code_links":8,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["google-research/flan"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/codet5-identifier-aware-unified-pre-trained","slug":"codet5-identifier-aware-unified-pre-trained","title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","date":"2021-09-02","arxiv_id":"2109.00859","n_code_links":5,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["salesforce/codet5"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"conqx-semantic-expansion-of-spoken-queries","title":"ConQX: Semantic Expansion of Spoken Queries for Intent Detection based on Conditioned Text Generation","date":"2021-09-02","arxiv_id":"2109.00729","n_code_links":0,"syntology":null},{"paper":null,"slug":"so-cloze-yet-so-far-n400-amplitude-is-better","title":"So Cloze yet so Far: N400 Amplitude is Better Predicted by Distributional Information than Human Predictability Judgements","date":"2021-09-02","arxiv_id":"2109.01226","n_code_links":0,"syntology":null},{"paper":null,"slug":"fight-fire-with-fire-fine-tuning-hate","title":"Fight Fire with Fire: Fine-tuning Hate Detectors using Large Samples of Generated Hate Speech","date":"2021-09-01","arxiv_id":"2109.00591","n_code_links":0,"syntology":null},{"paper":"/paper/optagan-entropy-based-finetuning-on-text-vae","slug":"optagan-entropy-based-finetuning-on-text-vae","title":"OptAGAN: Entropy-based finetuning on text VAE-GAN","date":"2021-09-01","arxiv_id":"2109.00239","n_code_links":1,"syntology":null},{"paper":"/paper/minif2f-a-cross-system-benchmark-for-formal","slug":"minif2f-a-cross-system-benchmark-for-formal","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","date":"2021-08-31","arxiv_id":"2109.00110","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openai/minif2f"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/task-oriented-dialogue-system-as-natural","slug":"task-oriented-dialogue-system-as-natural","title":"Task-Oriented Dialogue System as Natural Language Generation","date":"2021-08-31","arxiv_id":"2108.13679","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["victorwz/tod_as_nlg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-multilingual-capabilities-of-very","slug":"on-the-multilingual-capabilities-of-very","title":"On the Multilingual Capabilities of Very Large-Scale English Language Models","date":"2021-08-30","arxiv_id":"2108.13349","n_code_links":2,"syntology":null},{"paper":"/paper/want-to-reduce-labeling-cost-gpt-3-can-help","slug":"want-to-reduce-labeling-cost-gpt-3-can-help","title":"Want To Reduce Labeling Cost? GPT-3 Can Help","date":"2021-08-30","arxiv_id":"2108.13487","n_code_links":1,"syntology":null},{"paper":"/paper/headlinecause-a-dataset-of-news-headlines-for","slug":"headlinecause-a-dataset-of-news-headlines-for","title":"HeadlineCause: A Dataset of News Headlines for Detecting Causalities","date":"2021-08-28","arxiv_id":"2108.12626","n_code_links":1,"syntology":null},{"paper":null,"slug":"cgems-a-metric-model-for-automatic-code","title":"CGEMs: A Metric Model for Automatic Code Generation using GPT-3","date":"2021-08-23","arxiv_id":"2108.10168","n_code_links":0,"syntology":null},{"paper":"/paper/table-caption-generation-in-scholarly","slug":"table-caption-generation-in-scholarly","title":"Table Caption Generation in Scholarly Documents Leveraging Pre-trained Language Models","date":"2021-08-18","arxiv_id":"2108.08111","n_code_links":1,"syntology":null},{"paper":"/paper/curriculum-learning-a-regularization-method","slug":"curriculum-learning-a-regularization-method","title":"The Stability-Efficiency Dilemma: Investigating Sequence Length Warmup for Training GPT Models","date":"2021-08-13","arxiv_id":"2108.06084","n_code_links":1,"syntology":null},{"paper":"/paper/ammus-a-survey-of-transformer-based","slug":"ammus-a-survey-of-transformer-based","title":"AMMUS : A Survey of Transformer-based Pretrained Models in Natural Language Processing","date":"2021-08-12","arxiv_id":"2108.05542","n_code_links":1,"syntology":null},{"paper":"/paper/patrickstar-parallel-training-of-pre-trained","slug":"patrickstar-parallel-training-of-pre-trained","title":"PatrickStar: Parallel Training of Pre-trained Models via Chunk-based Memory Management","date":"2021-08-12","arxiv_id":"2108.05818","n_code_links":1,"syntology":null},{"paper":null,"slug":"offensive-language-and-hate-speech-detection-1","title":"Offensive Language and Hate Speech Detection with Deep Learning and Transfer Learning","date":"2021-08-06","arxiv_id":"2108.03305","n_code_links":0,"syntology":null},{"paper":null,"slug":"rockgpt-reconstructing-three-dimensional","title":"RockGPT: Reconstructing three-dimensional digital rocks from single two-dimensional slice from the perspective of video generation","date":"2021-08-05","arxiv_id":"2108.03132","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-offset-block-embedding-array-robe-for","title":"Random Offset Block Embedding Array (ROBE) for CriteoTB Benchmark MLPerf DLRM Model : 1000$\\times$ Compression and 3.1$\\times$ Faster Inference","date":"2021-08-04","arxiv_id":"2108.02191","n_code_links":0,"syntology":null},{"paper":"/paper/q-pain-a-question-answering-dataset-to","slug":"q-pain-a-question-answering-dataset-to","title":"Q-Pain: A Question Answering Dataset to Measure Social Bias in Pain Management","date":"2021-08-03","arxiv_id":"2108.01764","n_code_links":0,"syntology":null},{"paper":"/paper/bertac-enhancing-transformer-based-language","slug":"bertac-enhancing-transformer-based-language","title":"BERTAC: Enhancing Transformer-based Language Models with Adversarially Pretrained Convolutional Neural Networks","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"best-of-both-worlds-making-high-accuracy-non","title":"Best of Both Worlds: Making High Accuracy Non-incremental Transformer-based Disfluency Detection Incremental","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/employing-argumentation-knowledge-graphs-for","slug":"employing-argumentation-knowledge-graphs-for","title":"Employing Argumentation Knowledge Graphs for Neural Argument Generation","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/explanations-for-commonsenseqa-new-dataset","slug":"explanations-for-commonsenseqa-new-dataset","title":"Explanations for CommonsenseQA: New Dataset and Models","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kuileixi-a-chinese-open-ended-text-adventure","title":"KuiLeiXi: a Chinese Open-Ended Text Adventure Game","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pral-a-tailored-pre-training-model-for-task","title":"PRAL: A Tailored Pre-Training Model for Task-Oriented Dialog Generation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tgea-an-error-annotated-dataset-and-benchmark","title":"TGEA: An Error-Annotated Dataset and Benchmark Tasks for TextGeneration from Pretrained Language Models","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unleash-gpt-2-power-for-event-detection","title":"Unleash GPT-2 Power for Event Detection","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-gpt-gpt-2-and-bert-language-models","title":"Adapting GPT, GPT-2 and BERT Language Models for Speech Recognition","date":"2021-07-29","arxiv_id":"2108.07789","n_code_links":0,"syntology":null},{"paper":"/paper/fnetar-mixing-tokens-with-autoregressive","slug":"fnetar-mixing-tokens-with-autoregressive","title":"FNetAR: Mixing Tokens with Autoregressive Fourier Transforms","date":"2021-07-22","arxiv_id":"2107.10932","n_code_links":1,"syntology":null},{"paper":"/paper/chimera-efficiently-training-large-scale","slug":"chimera-efficiently-training-large-scale","title":"Chimera: Efficiently Training Large-Scale Neural Networks with Bidirectional Pipelines","date":"2021-07-14","arxiv_id":"2107.06925","n_code_links":1,"syntology":null},{"paper":"/paper/scalable-memory-protection-in-the-penglai","slug":"scalable-memory-protection-in-the-penglai","title":"Scalable Memory Protection in the PENGLAI Enclave","date":"2021-07-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/transformers-with-multi-modal-features-and","slug":"transformers-with-multi-modal-features-and","title":"Transformers with multi-modal features and post-fusion context for e-commerce session-based recommendation","date":"2021-07-11","arxiv_id":"2107.05124","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-trained-on","slug":"evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","arxiv_id":"2107.03374","n_code_links":13,"syntology":{"ran":26,"of":39,"n_ran_checked":24,"n_instrument":2,"unverified":13,"pointer_only":4,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 24 with no instrument failure: 1 honoured, 0 violated, 23 with no contract checked; 2 where Syntology's instrument failed) · 13 unverified","official":{"repos":["openai/human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["found_in_text","listed","official"]}}},{"paper":null,"slug":"not-quite-ask-a-librarian-ai-on-the-nature","title":"Not Quite 'Ask a Librarian': AI on the Nature, Value, and Future of LIS","date":"2021-07-07","arxiv_id":"2107.05383","n_code_links":0,"syntology":null},{"paper":"/paper/ernie-3-0-large-scale-knowledge-enhanced-pre","slug":"ernie-3-0-large-scale-knowledge-enhanced-pre","title":"ERNIE 3.0: Large-scale Knowledge Enhanced Pre-training for Language Understanding and Generation","date":"2021-07-05","arxiv_id":"2107.02137","n_code_links":2,"syntology":null},{"paper":null,"slug":"scarecrow-a-framework-for-scrutinizing","title":"Is GPT-3 Text Indistinguishable from Human Text? Scarecrow: A Framework for Scrutinizing Machine Text","date":"2021-07-02","arxiv_id":"2107.01294","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-s-in-a-measurement-using-gpt-3-on","title":"What's in a Measurement? Using GPT-3 on SemEval 2021 Task 8 -- MeasEval","date":"2021-06-28","arxiv_id":"2106.14720","n_code_links":0,"syntology":null},{"paper":"/paper/symbolicgpt-a-generative-transformer-model","slug":"symbolicgpt-a-generative-transformer-model","title":"SymbolicGPT: A Generative Transformer Model for Symbolic Regression","date":"2021-06-27","arxiv_id":"2106.14131","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["git.uwaterloo.ca/data-analytics-lab/symbolicgpt2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"toward-less-hidden-cost-of-code-completion","title":"Toward Less Hidden Cost of Code Completion with Acceptance and Ranking Models","date":"2021-06-26","arxiv_id":"2106.13928","n_code_links":0,"syntology":null},{"paper":null,"slug":"process-for-adapting-language-models-to","title":"Process for Adapting Language Models to Society (PALMS) with Values-Targeted Datasets","date":"2021-06-18","arxiv_id":"2106.10328","n_code_links":0,"syntology":null},{"paper":"/paper/lora-low-rank-adaptation-of-large-language","slug":"lora-low-rank-adaptation-of-large-language","title":"LoRA: Low-Rank Adaptation of Large Language Models","date":"2021-06-17","arxiv_id":"2106.09685","n_code_links":74,"syntology":{"ran":51,"of":84,"n_ran_checked":44,"n_instrument":7,"unverified":33,"pointer_only":30,"phrase":"51 ran (of which 19 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 7 where Syntology's instrument failed) · 33 unverified","official":{"repos":["microsoft/LoRA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"asr-adaptation-for-e-commerce-chatbots-using","title":"ASR Adaptation for E-commerce Chatbots using Cross-Utterance Context and Multi-Task Language Modeling","date":"2021-06-15","arxiv_id":"2106.09532","n_code_links":0,"syntology":null},{"paper":null,"slug":"textual-data-distributions-kullback-leibler","title":"Textual Data Distributions: Kullback Leibler Textual Distributions Contrasts on GPT-2 Generated Texts, with Supervised, Unsupervised Learning on Vaccine & Market Topics & Sentiment","date":"2021-06-15","arxiv_id":"2107.02025","n_code_links":0,"syntology":null},{"paper":"/paper/gpt3-to-plan-extracting-plans-from-text-using","slug":"gpt3-to-plan-extracting-plans-from-text-using","title":"GPT3-to-plan: Extracting plans from text using GPT-3","date":"2021-06-14","arxiv_id":"2106.07131","n_code_links":1,"syntology":null},{"paper":null,"slug":"pre-trained-models-past-present-and-future","title":"Pre-Trained Models: Past, Present and Future","date":"2021-06-14","arxiv_id":"2106.07139","n_code_links":0,"syntology":null},{"paper":"/paper/generate-annotate-and-learn-generative-models","slug":"generate-annotate-and-learn-generative-models","title":"Generate, Annotate, and Learn: NLP with Synthetic Text","date":"2021-06-11","arxiv_id":"2106.06168","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xlhex/gal_syntex"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/programming-puzzles","slug":"programming-puzzles","title":"Programming Puzzles","date":"2021-06-10","arxiv_id":"2106.05784","n_code_links":3,"syntology":null},{"paper":null,"slug":"engines-of-power-electricity-ai-and-general","title":"Engines of Power: Electricity, AI, and General-Purpose Military Transformations","date":"2021-06-08","arxiv_id":"2106.04338","n_code_links":0,"syntology":null},{"paper":"/paper/timedial-temporal-commonsense-reasoning-in","slug":"timedial-temporal-commonsense-reasoning-in","title":"TIMEDIAL: Temporal Commonsense Reasoning in Dialog","date":"2021-06-08","arxiv_id":"2106.04571","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-syntactic-probes-probe-syntax-experiments","title":"Do Syntactic Probes Probe Syntax? Experiments with Jabberwocky Probing","date":"2021-06-04","arxiv_id":"2106.02559","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-tagging-of-short-conversational","title":"Auto-tagging of Short Conversational Sentences using Transformer Methods","date":"2021-06-03","arxiv_id":"2106.01735","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-perdetry-test-generating-new-meanings-for","title":"GPT Perdetry Test: Generating new meanings for new words","date":"2021-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-comprehensive-understanding-and","title":"Towards a Comprehensive Understanding and Accurate Evaluation of Societal Biases in Pre-Trained Transformers","date":"2021-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-inheritance-for-pre-trained","slug":"knowledge-inheritance-for-pre-trained","title":"Knowledge Inheritance for Pre-trained Language Models","date":"2021-05-28","arxiv_id":"2105.13880","n_code_links":2,"syntology":null},{"paper":null,"slug":"generative-adversarial-imitation-learning-for","title":"Generative Adversarial Imitation Learning for Empathy-based AI","date":"2021-05-27","arxiv_id":"2105.13328","n_code_links":0,"syntology":null},{"paper":"/paper/data-curation-and-quality-assurance-for","slug":"data-curation-and-quality-assurance-for","title":"Data Curation and Quality Assurance for Machine Learning-based Cyber Intrusion Detection","date":"2021-05-20","arxiv_id":"2105.10041","n_code_links":1,"syntology":null},{"paper":"/paper/methods-for-detoxification-of-texts-for-the","slug":"methods-for-detoxification-of-texts-for-the","title":"Methods for Detoxification of Texts for the Russian Language","date":"2021-05-19","arxiv_id":"2105.09052","n_code_links":3,"syntology":null},{"paper":null,"slug":"neural-predictive-text-for-grammatical-error","title":"Neural Predictive Text for Grammatical Error Prevention","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slgpt-using-transfer-learning-to-directly","slug":"slgpt-using-transfer-learning-to-directly","title":"SLGPT: Using Transfer Learning to Directly Generate Simulink Model Files and Find Bugs in the Simulink Toolchain","date":"2021-05-16","arxiv_id":"2105.07465","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-busters-outlier-layernorm-dimensions","title":"BERT Busters: Outlier Dimensions that Disrupt Transformers","date":"2021-05-14","arxiv_id":"2105.06990","n_code_links":0,"syntology":null},{"paper":"/paper/joint-retrieval-and-generation-training-for","slug":"joint-retrieval-and-generation-training-for","title":"RetGen: A Joint framework for Retrieval and Grounded Text Generation Modeling","date":"2021-05-14","arxiv_id":"2105.06597","n_code_links":1,"syntology":null},{"paper":"/paper/bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","slug":"bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","title":"BERT is to NLP what AlexNet is to CV: Can Pre-Trained Language Models Identify Analogies?","date":"2021-05-11","arxiv_id":"2105.04949","n_code_links":1,"syntology":null},{"paper":"/paper/el-attention-memory-efficient-lossless","slug":"el-attention-memory-efficient-lossless","title":"EL-Attention: Memory Efficient Lossless Attention for Generation","date":"2021-05-11","arxiv_id":"2105.04779","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/fastseq"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/e-vil-a-dataset-and-benchmark-for-natural","slug":"e-vil-a-dataset-and-benchmark-for-natural","title":"e-ViL: A Dataset and Benchmark for Natural Language Explanations in Vision-Language Tasks","date":"2021-05-08","arxiv_id":"2105.03761","n_code_links":2,"syntology":{"ran":7,"of":15,"n_ran_checked":6,"n_instrument":1,"unverified":8,"pointer_only":15,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["maximek3/e-ViL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/on-the-fly-controlled-text-generation-with","slug":"on-the-fly-controlled-text-generation-with","title":"DExperts: Decoding-Time Controlled Text Generation with Experts and Anti-Experts","date":"2021-05-07","arxiv_id":"2105.03023","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alisawuffles/DExperts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"one-model-to-rule-them-all-towards-zero-shot","title":"One Model to Rule them All: Towards Zero-Shot Learning for Databases","date":"2021-05-03","arxiv_id":"2105.00642","n_code_links":0,"syntology":null},{"paper":"/paper/unreasonable-effectiveness-of-rule-based","slug":"unreasonable-effectiveness-of-rule-based","title":"Unreasonable Effectiveness of Rule-Based Heuristics in Solving Russian SuperGLUE Tasks","date":"2021-05-03","arxiv_id":"2105.01192","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-political-bias-in-language-models","title":"Mitigating Political Bias in Language Models Through Reinforced Calibration","date":"2021-04-30","arxiv_id":"2104.14795","n_code_links":0,"syntology":null},{"paper":"/paper/entailment-as-few-shot-learner","slug":"entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","arxiv_id":"2104.14690","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"extractive-and-abstractive-explanations-for","title":"Extractive and Abstractive Explanations for Fact-Checking and Evaluation of News","date":"2021-04-27","arxiv_id":"2104.12918","n_code_links":0,"syntology":null},{"paper":null,"slug":"uot-uwf-partai-at-semeval-2021-task-5-self","title":"UoT-UWF-PartAI at SemEval-2021 Task 5: Self Attention Based Bi-GRU with Multi-Embedding Representation for Toxicity Highlighter","date":"2021-04-27","arxiv_id":"2104.13164","n_code_links":0,"syntology":null},{"paper":null,"slug":"accounting-for-agreement-phenomena-in","title":"Accounting for Agreement Phenomena in Sentence Comprehension with Transformer Language Models: Effects of Similarity-based Interference on Surprisal and Attention","date":"2021-04-26","arxiv_id":"2104.12874","n_code_links":0,"syntology":null},{"paper":"/paper/easy-and-efficient-transformer-scalable","slug":"easy-and-efficient-transformer-scalable","title":"Easy and Efficient Transformer : Scalable Inference Solution For large NLP model","date":"2021-04-26","arxiv_id":"2104.12470","n_code_links":1,"syntology":null},{"paper":"/paper/pangu-a-large-scale-autoregressive-pretrained","slug":"pangu-a-large-scale-autoregressive-pretrained","title":"PanGu-$α$: Large-scale Autoregressive Pretrained Chinese Language Models with Auto-parallel Computation","date":"2021-04-26","arxiv_id":"2104.12369","n_code_links":5,"syntology":null},{"paper":null,"slug":"what-s-the-context-long-context-nlm","title":"Adapting Long Context NLM for ASR Rescoring in Conversational Agents","date":"2021-04-21","arxiv_id":"2104.11070","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-covid-19-tweets-with-transformer","title":"Analyzing COVID-19 Tweets with Transformer-based Language Models","date":"2021-04-20","arxiv_id":"2104.10259","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-pre-training-objectives-for","title":"Efficient pre-training objectives for Transformers","date":"2021-04-20","arxiv_id":"2104.09694","n_code_links":0,"syntology":null},{"paper":"/paper/a-token-level-reference-free-hallucination","slug":"a-token-level-reference-free-hallucination","title":"A Token-level Reference-free Hallucination Detection Benchmark for Free-form Text Generation","date":"2021-04-18","arxiv_id":"2104.08704","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/HaDes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fantastically-ordered-prompts-and-where-to","slug":"fantastically-ordered-prompts-and-where-to","title":"Fantastically Ordered Prompts and Where to Find Them: Overcoming Few-Shot Prompt Order Sensitivity","date":"2021-04-18","arxiv_id":"2104.08786","n_code_links":2,"syntology":{"ran":3,"of":10,"n_ran_checked":0,"n_instrument":3,"unverified":7,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":null}},{"paper":"/paper/gpt3mix-leveraging-large-scale-language","slug":"gpt3mix-leveraging-large-scale-language","title":"GPT3Mix: Leveraging Large-scale Language Models for Text Augmentation","date":"2021-04-18","arxiv_id":"2104.08826","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["naver-ai/hypermix"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/natural-instructions-benchmarking","slug":"natural-instructions-benchmarking","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","date":"2021-04-18","arxiv_id":"2104.08773","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/natural-instructions","allenai/natural-instructions-v1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/the-power-of-scale-for-parameter-efficient","slug":"the-power-of-scale-for-parameter-efficient","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","date":"2021-04-18","arxiv_id":"2104.08691","n_code_links":12,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google-research/prompt-tuning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/an-adversarially-learned-turing-test-for","slug":"an-adversarially-learned-turing-test-for","title":"An Adversarially-Learned Turing Test for Dialog Generation Models","date":"2021-04-16","arxiv_id":"2104.08231","n_code_links":1,"syntology":null},{"paper":"/paper/surface-form-competition-why-the-highest","slug":"surface-form-competition-why-the-highest","title":"Surface Form Competition: Why the Highest Probability Answer Isn't Always Right","date":"2021-04-16","arxiv_id":"2104.08315","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["peterwestuw/surface-form-competition"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/text2app-a-framework-for-creating-android","slug":"text2app-a-framework-for-creating-android","title":"Text2App: A Framework for Creating Android Apps from Text Descriptions","date":"2021-04-16","arxiv_id":"2104.08301","n_code_links":2,"syntology":null},{"paper":"/paper/nareor-the-narrative-reordering-problem","slug":"nareor-the-narrative-reordering-problem","title":"NAREOR: The Narrative Reordering Problem","date":"2021-04-14","arxiv_id":"2104.06669","n_code_links":1,"syntology":null},{"paper":"/paper/understanding-transformers-for-bot-detection","slug":"understanding-transformers-for-bot-detection","title":"Understanding Transformers for Bot Detection in Twitter","date":"2021-04-13","arxiv_id":"2104.06182","n_code_links":1,"syntology":null},{"paper":"/paper/meta-tuning-language-models-to-answer-prompts","slug":"meta-tuning-language-models-to-answer-prompts","title":"Adapting Language Models for Zero-shot Learning by Meta-tuning on Dataset and Prompt Collections","date":"2021-04-10","arxiv_id":"2104.04670","n_code_links":1,"syntology":null}],"record_sha256":"71100b90b63c34f2e4133cb2bdc07465060e45201a7659b4b9e380b40fd72ff8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}