{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-2/papers/6","list_of":"/method/gpt-2","method":"GPT-2","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":8,"rows_per_page":100,"rows":[501,600],"of":768,"counts":{"archive_papers_tagged":768,"with_a_code_link":339,"where_syntology_ran_a_sample":125,"not_listed_spam_title":0,"listed":768,"listed_where_code_ran":125,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-2","prev":"/method/gpt-2/papers/5","next":"/method/gpt-2/papers/7","papers":[{"paper":null,"slug":"efficient-hierarchical-domain-adaptation-for-1","title":"Efficient Hierarchical Domain Adaptation for Pretrained Language Models","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"elastic-weight-consolidation-for-reduction-of","title":"Elastic Weight Consolidation for Reduction of Catastrophic Forgetting in GPT-2","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"jointly-reinforced-user-simulator-and-task","title":"Jointly Reinforced User Simulator and Task-oriented Dialog System with Simplified Generative Architecture","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-conversational-1","title":"Representation Learning for Conversational Data using Discourse Mutual Information Maximization","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-a-sentence-does-not-introduce-a","title":"When a sentence does not introduce a discourse entity, Transformer-based models still often refer to it","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"why-does-surprisal-from-smaller-gpt-2-models","title":"Why Does Surprisal From Smaller GPT-2 Models Provide Better Fit to Human Reading Times?","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/assemble-foundation-models-for-automatic-code","slug":"assemble-foundation-models-for-automatic-code","title":"Assemble Foundation Models for Automatic Code Summarization","date":"2022-01-13","arxiv_id":"2201.05222","n_code_links":1,"syntology":null},{"paper":null,"slug":"submix-practical-private-prediction-for-large-1","title":"Submix: Practical Private Prediction for Large-Scale Language Models","date":"2022-01-04","arxiv_id":"2201.00971","n_code_links":0,"syntology":null},{"paper":"/paper/call-for-customized-conversation-customized","slug":"call-for-customized-conversation-customized","title":"Call for Customized Conversation: Customized Conversation Grounding Persona and Knowledge","date":"2021-12-16","arxiv_id":"2112.08619","n_code_links":3,"syntology":null},{"paper":"/paper/efficient-hierarchical-domain-adaptation-for","slug":"efficient-hierarchical-domain-adaptation-for","title":"Efficient Hierarchical Domain Adaptation for Pretrained Language Models","date":"2021-12-16","arxiv_id":"2112.08786","n_code_links":1,"syntology":null},{"paper":null,"slug":"reconsidering-the-past-optimizing-hidden-1","title":"Reconsidering the Past: Optimizing Hidden States in Language Models","date":"2021-12-16","arxiv_id":"2112.08653","n_code_links":0,"syntology":null},{"paper":"/paper/wechsel-effective-initialization-of-subword-1","slug":"wechsel-effective-initialization-of-subword-1","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","date":"2021-12-13","arxiv_id":"2112.06598","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cpjku/wechsel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-logical-level-natural-language","title":"Improving Logical-Level Natural Language Generation with Topic-Conditioned Data Augmentation and Logical Form Generation","date":"2021-12-12","arxiv_id":"2112.06240","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-conversational","title":"Representation Learning for Conversational Data using Discourse Mutual Information Maximization","date":"2021-12-04","arxiv_id":"2112.05787","n_code_links":0,"syntology":null},{"paper":"/paper/pixelated-butterfly-simple-and-efficient-1","slug":"pixelated-butterfly-simple-and-efficient-1","title":"Pixelated Butterfly: Simple and Efficient Sparse training for Neural Network Models","date":"2021-11-30","arxiv_id":"2112.00029","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["HazyResearch/pixelfly"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"context-matters-in-semantically-controlled","title":"Context Matters in Semantically Controlled Language Generation for Task-oriented Dialogue Systems","date":"2021-11-28","arxiv_id":"2111.14119","n_code_links":0,"syntology":null},{"paper":"/paper/clipcap-clip-prefix-for-image-captioning","slug":"clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","arxiv_id":"2111.09734","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rmokady/clip_prefix_caption"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"guiding-generative-language-models-for-data","title":"Guiding Generative Language Models for Data Augmentation in Few-Shot Text Classification","date":"2021-11-17","arxiv_id":"2111.09064","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-information-theoretic-measurement-of","title":"An Information Theoretic Measurement of Topical Relevance in Learner Essays","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-task-oriented-dialog-policy","title":"End-to-end Task-oriented Dialog Policy Learning based on Pre-trained Language Model","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-pre-trained-transformer-for-design","title":"Generative Pre-Trained Transformer for Design Concept Generation: An Exploration","date":"2021-11-16","arxiv_id":"2111.08489","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-is-in-rescue-task-oriented","title":"Knowledge Graph is in Rescue: Task Oriented Dialogue System for Response Generation without NLU and DM","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"moving-the-eiffel-tower-to-rome-tracing-and","title":"Moving the Eiffel Tower to ROME: Tracing and Editing Facts in GPT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-of-ambiguity-in-pre-trained","title":"Representation of Ambiguity in Pre-Trained Sentence Embeddings","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-bottleneck-makes-language-models","title":"Softmax Bottleneck Makes Language Models Unable to Represent Multi-mode Word Distributions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tell-me-who-you-are-and-i-ll-tell-you-what-to","title":"Tell me who you are and i'll tell you what to do: A Persona Grounded Task Oriented Dialogue Generation System","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-classifying-grammatical-role-bert-doesn","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-story-generation-with-multi-task","title":"Exploring Story Generation with Multi-task Objectives in Variational Autoencoders","date":"2021-11-15","arxiv_id":"2111.08133","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-corpus-of-discourse-structure-in","slug":"a-novel-corpus-of-discourse-structure-in","title":"A Novel Corpus of Discourse Structure in Humans and Computers","date":"2021-11-10","arxiv_id":"2111.05940","n_code_links":1,"syntology":null},{"paper":null,"slug":"distir-an-intermediate-representation-and","title":"DistIR: An Intermediate Representation and Simulator for Efficient Neural Network Distribution","date":"2021-11-09","arxiv_id":"2111.05426","n_code_links":0,"syntology":null},{"paper":null,"slug":"fpm-a-collection-of-large-scale-foundation","title":"FPM: A Collection of Large-scale Foundation Pre-trained Language Models","date":"2021-11-09","arxiv_id":"2111.04909","n_code_links":0,"syntology":null},{"paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","slug":"dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","arxiv_id":"2111.00160","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vita-group/dsee"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/amendable-generation-for-dialogue-state","slug":"amendable-generation-for-dialogue-state","title":"Amendable Generation for Dialogue State Tracking","date":"2021-10-29","arxiv_id":"2110.15659","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-sequence-to-sequence-model-for-extracting","title":"A Sequence to Sequence Model for Extracting Multiple Product Name Entities from Dialog","date":"2021-10-28","arxiv_id":"2110.14843","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-artificial-texts-as-substitution","title":"Generating artificial texts as substitution or complement of training data","date":"2021-10-25","arxiv_id":"2110.13016","n_code_links":0,"syntology":null},{"paper":null,"slug":"reminding-the-incremental-language-model-via","title":"Reminding the Incremental Language Model via Data-Free Self-Distillation","date":"2021-10-17","arxiv_id":"2110.08745","n_code_links":0,"syntology":null},{"paper":"/paper/taming-visually-guided-sound-generation","slug":"taming-visually-guided-sound-generation","title":"Taming Visually Guided Sound Generation","date":"2021-10-17","arxiv_id":"2110.08791","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":6,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["v-iashin/SpecVQGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"a-short-study-on-compressing-decoder-based","title":"A Short Study on Compressing Decoder-Based Language Models","date":"2021-10-16","arxiv_id":"2110.08460","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-transfer-learning-for-polish","title":"Evaluation of Transfer Learning for Polish with a text-to-text model","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hydra-a-system-for-large-multi-model-deep","slug":"hydra-a-system-for-large-multi-model-deep","title":"Hydra: A System for Large Multi-Model Deep Learning","date":"2021-10-16","arxiv_id":"2110.08633","n_code_links":1,"syntology":null},{"paper":null,"slug":"wechsel-effective-initialization-of-subword","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kronecker-decomposition-for-gpt-compression","title":"Kronecker Decomposition for GPT Compression","date":"2021-10-15","arxiv_id":"2110.08152","n_code_links":0,"syntology":null},{"paper":null,"slug":"covert-message-passing-over-public-internet","title":"Leveraging Generative Models for Covert Messaging: Challenges and Tradeoffs for \"Dead-Drop\" Deployments","date":"2021-10-13","arxiv_id":"2110.07009","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-modelling-via-learning-to-rank","title":"Language Modelling via Learning to Rank","date":"2021-10-13","arxiv_id":"2110.06961","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-learning-for-situated-multi-domain","title":"Multi-Task Learning for Situated Multi-Domain End-to-End Dialogue Systems","date":"2021-10-11","arxiv_id":"2110.05221","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-model-pre-training-improves","title":"Language Model Pre-training Improves Generalization in Policy Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mapping-language-models-to-grounded","title":"Mapping Language Models to Grounded Conceptual Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-for-large","title":"Offline Reinforcement Learning for Large Scale Language Action Spaces","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-plug-and-play-method-for-controlled-text","slug":"a-plug-and-play-method-for-controlled-text","title":"A Plug-and-Play Method for Controlled Text Generation","date":"2021-09-20","arxiv_id":"2109.09707","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-bias-in-nlp-application-to-hate-speech","title":"Model Bias in NLP -- Application to Hate Speech Classification using transfer learning techniques","date":"2021-09-20","arxiv_id":"2109.09725","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-low-frequency-patterns-with-a-pre","title":"Learning Low-frequency Patterns with A Pre-trained Document-Grounded Conversation Model","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"relating-neural-text-degeneration-to-exposure","title":"Relating Neural Text Degeneration to Exposure Bias","date":"2021-09-17","arxiv_id":"2109.08705","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-text-auto-completion-with-next","title":"Improving Text Auto-Completion with Next Phrase Prediction","date":"2021-09-15","arxiv_id":"2109.07067","n_code_links":0,"syntology":null},{"paper":"/paper/a-temporal-variational-model-for-story","slug":"a-temporal-variational-model-for-story","title":"A Temporal Variational Model for Story Generation","date":"2021-09-14","arxiv_id":"2109.06807","n_code_links":3,"syntology":null},{"paper":null,"slug":"enhancing-self-disclosure-in-neural-dialog","title":"Enhancing Self-Disclosure In Neural Dialog Models By Candidate Re-ranking","date":"2021-09-10","arxiv_id":"2109.05090","n_code_links":0,"syntology":null},{"paper":"/paper/all-bark-and-no-bite-rogue-dimensions-in","slug":"all-bark-and-no-bite-rogue-dimensions-in","title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","date":"2021-09-09","arxiv_id":"2109.04404","n_code_links":1,"syntology":null},{"paper":"/paper/variational-latent-state-gpt-for-semi","slug":"variational-latent-state-gpt-for-semi","title":"Variational Latent-State GPT for Semi-Supervised Task-Oriented Dialog Systems","date":"2021-09-09","arxiv_id":"2109.04314","n_code_links":2,"syntology":null},{"paper":"/paper/truthfulqa-measuring-how-models-mimic-human","slug":"truthfulqa-measuring-how-models-mimic-human","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","date":"2021-09-08","arxiv_id":"2109.07958","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sylinrl/truthfulqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"empathetic-dialogue-generation-with-pre","title":"Empathetic Dialogue Generation with Pre-trained RoBERTa-GPT2 and External Knowledge","date":"2021-09-07","arxiv_id":"2109.03004","n_code_links":0,"syntology":null},{"paper":"/paper/text-free-prosody-aware-generative-spoken","slug":"text-free-prosody-aware-generative-spoken","title":"Text-Free Prosody-Aware Generative Spoken Language Modeling","date":"2021-09-07","arxiv_id":"2109.03264","n_code_links":1,"syntology":null},{"paper":null,"slug":"conqx-semantic-expansion-of-spoken-queries","title":"ConQX: Semantic Expansion of Spoken Queries for Intent Detection based on Conditioned Text Generation","date":"2021-09-02","arxiv_id":"2109.00729","n_code_links":0,"syntology":null},{"paper":"/paper/optagan-entropy-based-finetuning-on-text-vae","slug":"optagan-entropy-based-finetuning-on-text-vae","title":"OptAGAN: Entropy-based finetuning on text VAE-GAN","date":"2021-09-01","arxiv_id":"2109.00239","n_code_links":1,"syntology":null},{"paper":"/paper/task-oriented-dialogue-system-as-natural","slug":"task-oriented-dialogue-system-as-natural","title":"Task-Oriented Dialogue System as Natural Language Generation","date":"2021-08-31","arxiv_id":"2108.13679","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["victorwz/tod_as_nlg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/headlinecause-a-dataset-of-news-headlines-for","slug":"headlinecause-a-dataset-of-news-headlines-for","title":"HeadlineCause: A Dataset of News Headlines for Detecting Causalities","date":"2021-08-28","arxiv_id":"2108.12626","n_code_links":1,"syntology":null},{"paper":"/paper/table-caption-generation-in-scholarly","slug":"table-caption-generation-in-scholarly","title":"Table Caption Generation in Scholarly Documents Leveraging Pre-trained Language Models","date":"2021-08-18","arxiv_id":"2108.08111","n_code_links":1,"syntology":null},{"paper":"/paper/curriculum-learning-a-regularization-method","slug":"curriculum-learning-a-regularization-method","title":"The Stability-Efficiency Dilemma: Investigating Sequence Length Warmup for Training GPT Models","date":"2021-08-13","arxiv_id":"2108.06084","n_code_links":1,"syntology":null},{"paper":null,"slug":"offensive-language-and-hate-speech-detection-1","title":"Offensive Language and Hate Speech Detection with Deep Learning and Transfer Learning","date":"2021-08-06","arxiv_id":"2108.03305","n_code_links":0,"syntology":null},{"paper":"/paper/q-pain-a-question-answering-dataset-to","slug":"q-pain-a-question-answering-dataset-to","title":"Q-Pain: A Question Answering Dataset to Measure Social Bias in Pain Management","date":"2021-08-03","arxiv_id":"2108.01764","n_code_links":0,"syntology":null},{"paper":null,"slug":"best-of-both-worlds-making-high-accuracy-non","title":"Best of Both Worlds: Making High Accuracy Non-incremental Transformer-based Disfluency Detection Incremental","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/employing-argumentation-knowledge-graphs-for","slug":"employing-argumentation-knowledge-graphs-for","title":"Employing Argumentation Knowledge Graphs for Neural Argument Generation","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/explanations-for-commonsenseqa-new-dataset","slug":"explanations-for-commonsenseqa-new-dataset","title":"Explanations for CommonsenseQA: New Dataset and Models","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kuileixi-a-chinese-open-ended-text-adventure","title":"KuiLeiXi: a Chinese Open-Ended Text Adventure Game","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pral-a-tailored-pre-training-model-for-task","title":"PRAL: A Tailored Pre-Training Model for Task-Oriented Dialog Generation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tgea-an-error-annotated-dataset-and-benchmark","title":"TGEA: An Error-Annotated Dataset and Benchmark Tasks for TextGeneration from Pretrained Language Models","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unleash-gpt-2-power-for-event-detection","title":"Unleash GPT-2 Power for Event Detection","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-gpt-gpt-2-and-bert-language-models","title":"Adapting GPT, GPT-2 and BERT Language Models for Speech Recognition","date":"2021-07-29","arxiv_id":"2108.07789","n_code_links":0,"syntology":null},{"paper":"/paper/chimera-efficiently-training-large-scale","slug":"chimera-efficiently-training-large-scale","title":"Chimera: Efficiently Training Large-Scale Neural Networks with Bidirectional Pipelines","date":"2021-07-14","arxiv_id":"2107.06925","n_code_links":1,"syntology":null},{"paper":null,"slug":"scarecrow-a-framework-for-scrutinizing","title":"Is GPT-3 Text Indistinguishable from Human Text? Scarecrow: A Framework for Scrutinizing Machine Text","date":"2021-07-02","arxiv_id":"2107.01294","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-less-hidden-cost-of-code-completion","title":"Toward Less Hidden Cost of Code Completion with Acceptance and Ranking Models","date":"2021-06-26","arxiv_id":"2106.13928","n_code_links":0,"syntology":null},{"paper":"/paper/lora-low-rank-adaptation-of-large-language","slug":"lora-low-rank-adaptation-of-large-language","title":"LoRA: Low-Rank Adaptation of Large Language Models","date":"2021-06-17","arxiv_id":"2106.09685","n_code_links":74,"syntology":{"ran":51,"of":84,"n_ran_checked":44,"n_instrument":7,"unverified":33,"pointer_only":30,"phrase":"51 ran (of which 19 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 7 where Syntology's instrument failed) · 33 unverified","official":{"repos":["microsoft/LoRA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"textual-data-distributions-kullback-leibler","title":"Textual Data Distributions: Kullback Leibler Textual Distributions Contrasts on GPT-2 Generated Texts, with Supervised, Unsupervised Learning on Vaccine & Market Topics & Sentiment","date":"2021-06-15","arxiv_id":"2107.02025","n_code_links":0,"syntology":null},{"paper":"/paper/generate-annotate-and-learn-generative-models","slug":"generate-annotate-and-learn-generative-models","title":"Generate, Annotate, and Learn: NLP with Synthetic Text","date":"2021-06-11","arxiv_id":"2106.06168","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xlhex/gal_syntex"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"auto-tagging-of-short-conversational","title":"Auto-tagging of Short Conversational Sentences using Transformer Methods","date":"2021-06-03","arxiv_id":"2106.01735","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-comprehensive-understanding-and","title":"Towards a Comprehensive Understanding and Accurate Evaluation of Societal Biases in Pre-Trained Transformers","date":"2021-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-adversarial-imitation-learning-for","title":"Generative Adversarial Imitation Learning for Empathy-based AI","date":"2021-05-27","arxiv_id":"2105.13328","n_code_links":0,"syntology":null},{"paper":"/paper/methods-for-detoxification-of-texts-for-the","slug":"methods-for-detoxification-of-texts-for-the","title":"Methods for Detoxification of Texts for the Russian Language","date":"2021-05-19","arxiv_id":"2105.09052","n_code_links":3,"syntology":null},{"paper":null,"slug":"neural-predictive-text-for-grammatical-error","title":"Neural Predictive Text for Grammatical Error Prevention","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slgpt-using-transfer-learning-to-directly","slug":"slgpt-using-transfer-learning-to-directly","title":"SLGPT: Using Transfer Learning to Directly Generate Simulink Model Files and Find Bugs in the Simulink Toolchain","date":"2021-05-16","arxiv_id":"2105.07465","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-busters-outlier-layernorm-dimensions","title":"BERT Busters: Outlier Dimensions that Disrupt Transformers","date":"2021-05-14","arxiv_id":"2105.06990","n_code_links":0,"syntology":null},{"paper":"/paper/bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","slug":"bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","title":"BERT is to NLP what AlexNet is to CV: Can Pre-Trained Language Models Identify Analogies?","date":"2021-05-11","arxiv_id":"2105.04949","n_code_links":1,"syntology":null},{"paper":"/paper/el-attention-memory-efficient-lossless","slug":"el-attention-memory-efficient-lossless","title":"EL-Attention: Memory Efficient Lossless Attention for Generation","date":"2021-05-11","arxiv_id":"2105.04779","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/fastseq"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/e-vil-a-dataset-and-benchmark-for-natural","slug":"e-vil-a-dataset-and-benchmark-for-natural","title":"e-ViL: A Dataset and Benchmark for Natural Language Explanations in Vision-Language Tasks","date":"2021-05-08","arxiv_id":"2105.03761","n_code_links":2,"syntology":{"ran":7,"of":15,"n_ran_checked":6,"n_instrument":1,"unverified":8,"pointer_only":15,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["maximek3/e-ViL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"mitigating-political-bias-in-language-models","title":"Mitigating Political Bias in Language Models Through Reinforced Calibration","date":"2021-04-30","arxiv_id":"2104.14795","n_code_links":0,"syntology":null},{"paper":null,"slug":"extractive-and-abstractive-explanations-for","title":"Extractive and Abstractive Explanations for Fact-Checking and Evaluation of News","date":"2021-04-27","arxiv_id":"2104.12918","n_code_links":0,"syntology":null},{"paper":null,"slug":"uot-uwf-partai-at-semeval-2021-task-5-self","title":"UoT-UWF-PartAI at SemEval-2021 Task 5: Self Attention Based Bi-GRU with Multi-Embedding Representation for Toxicity Highlighter","date":"2021-04-27","arxiv_id":"2104.13164","n_code_links":0,"syntology":null},{"paper":null,"slug":"accounting-for-agreement-phenomena-in","title":"Accounting for Agreement Phenomena in Sentence Comprehension with Transformer Language Models: Effects of Similarity-based Interference on Surprisal and Attention","date":"2021-04-26","arxiv_id":"2104.12874","n_code_links":0,"syntology":null},{"paper":"/paper/easy-and-efficient-transformer-scalable","slug":"easy-and-efficient-transformer-scalable","title":"Easy and Efficient Transformer : Scalable Inference Solution For large NLP model","date":"2021-04-26","arxiv_id":"2104.12470","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-pre-training-objectives-for","title":"Efficient pre-training objectives for Transformers","date":"2021-04-20","arxiv_id":"2104.09694","n_code_links":0,"syntology":null}],"record_sha256":"cad7deb440243418cd4c74ba7080089e445469205e4c74fe568ceed5b111af3c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}